diff --git a/.github/workflows/copilot-cli-safeoutputs.yml b/.github/workflows/copilot-cli-safeoutputs.yml index eff862f20..71967b4b3 100644 --- a/.github/workflows/copilot-cli-safeoutputs.yml +++ b/.github/workflows/copilot-cli-safeoutputs.yml @@ -4,6 +4,7 @@ on: pull_request: paths: - "src/**" + - "scripts/ado-script/**" - "tests/**" - "Cargo.toml" - "Cargo.lock" @@ -62,6 +63,12 @@ jobs: mcpg:MCPG_VERSION:src/compile/common.rs VERSIONS + - name: Build Copilot invoker bundle + run: | + set -euo pipefail + npm --prefix scripts/ado-script ci + npm --prefix scripts/ado-script run build:copilot-invoker + - name: Install compiler-pinned GitHub Copilot CLI run: | set -euo pipefail @@ -111,6 +118,7 @@ jobs: ADO_AW_BIN: ${{ github.workspace }}/target/debug/ado-aw AWF_BIN: ${{ runner.temp }}/bin/awf COPILOT_BIN: ${{ runner.temp }}/bin/copilot + COPILOT_INVOKER_BUNDLE: ${{ github.workspace }}/scripts/ado-script/copilot-invoker.js AWF_VERSION: ${{ steps.versions.outputs.awf }} MCPG_VERSION: ${{ steps.versions.outputs.mcpg }} run: bash tests/awf-copilot-safeoutputs/run.sh diff --git a/.github/workflows/pr-sous-chef.lock.yml b/.github/workflows/pr-sous-chef.lock.yml index f1ce12c41..6f325cf18 100644 --- a/.github/workflows/pr-sous-chef.lock.yml +++ b/.github/workflows/pr-sous-chef.lock.yml @@ -1,4 +1,4 @@ -# gh-aw-metadata: {"schema_version":"v4","frontmatter_hash":"d4e6e641206e227ad9c8c71f06a4f8d75b46814949bf0464d0bc3fb84cfc2706","body_hash":"0b760609f27236f11ded4123610e522e8a73558533e14609a69bf0d176abdd4f","compiler_version":"v0.86.2","strict":true,"agent_id":"copilot","engine_versions":{"copilot":"1.0.79"}} +# gh-aw-metadata: {"schema_version":"v4","frontmatter_hash":"d4e6e641206e227ad9c8c71f06a4f8d75b46814949bf0464d0bc3fb84cfc2706","body_hash":"c58344b90694d21c8e1521cfba92d39375d8b8877e0827d4c9dff5a150bc2613","compiler_version":"v0.86.2","strict":true,"agent_id":"copilot","engine_versions":{"copilot":"1.0.79"}} # gh-aw-manifest: {"version":1,"secrets":["GH_AW_CI_TRIGGER_TOKEN","GH_AW_GITHUB_MCP_SERVER_TOKEN","GH_AW_GITHUB_TOKEN","GITHUB_TOKEN"],"actions":[{"repo":"actions/checkout","sha":"3d3c42e5aac5ba805825da76410c181273ba90b1","version":"v7.0.1"},{"repo":"actions/download-artifact","sha":"3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c","version":"v8.0.1"},{"repo":"actions/github-script","sha":"3a2844b7e9c422d3c10d287c895573f7108da1b3","version":"v9.0.0"},{"repo":"actions/setup-node","sha":"820762786026740c76f36085b0efc47a31fe5020","version":"v7.0.0"},{"repo":"actions/upload-artifact","sha":"043fb46d1a93c77aae656e7c1c64a875d1fc6a0a","version":"v7.0.1"},{"repo":"github/gh-aw-actions/setup","sha":"6aab9e5b5c91c615506061f09bedd81a23babe3c","version":"v0.86.2"}],"containers":[{"image":"ghcr.io/github/gh-aw-firewall/agent:0.27.44","digest":"sha256:0d727725c737b58c7bdf51f640cffb928385ec46517e0917c7f1a02f1bada8b4","pinned_image":"ghcr.io/github/gh-aw-firewall/agent:0.27.44@sha256:0d727725c737b58c7bdf51f640cffb928385ec46517e0917c7f1a02f1bada8b4"},{"image":"ghcr.io/github/gh-aw-firewall/api-proxy:0.27.44","digest":"sha256:b50fbadba138f6e9aba94aca09711335c489bb3b15861220cb66f6092e042dc7","pinned_image":"ghcr.io/github/gh-aw-firewall/api-proxy:0.27.44@sha256:b50fbadba138f6e9aba94aca09711335c489bb3b15861220cb66f6092e042dc7"},{"image":"ghcr.io/github/gh-aw-firewall/squid:0.27.44","digest":"sha256:83e48bbe12c634be8c228a576832fe45f66c529ac3659db92bddbcf2eeb6d627","pinned_image":"ghcr.io/github/gh-aw-firewall/squid:0.27.44@sha256:83e48bbe12c634be8c228a576832fe45f66c529ac3659db92bddbcf2eeb6d627"},{"image":"ghcr.io/github/gh-aw-mcpg:v0.4.9","digest":"sha256:e5a1569aeaf41820fa7bdee3e94468cae448133cdbf00119ad24f5b74db1ab9f","pinned_image":"ghcr.io/github/gh-aw-mcpg:v0.4.9@sha256:e5a1569aeaf41820fa7bdee3e94468cae448133cdbf00119ad24f5b74db1ab9f"},{"image":"ghcr.io/github/gh-aw-node","digest":"sha256:0d9f1fb5fd6610c0ac1f5194a38e45a8a1e81f8a390d5142d8e4e6f26a4b3196","pinned_image":"ghcr.io/github/gh-aw-node@sha256:0d9f1fb5fd6610c0ac1f5194a38e45a8a1e81f8a390d5142d8e4e6f26a4b3196"},{"image":"ghcr.io/github/github-mcp-server:v1.9.0","digest":"sha256:881b53d6f75f69bdbc1b5b10fc2f1361717c19054143b3a8529fb5c32061a50e","pinned_image":"ghcr.io/github/github-mcp-server:v1.9.0@sha256:881b53d6f75f69bdbc1b5b10fc2f1361717c19054143b3a8529fb5c32061a50e"}]} # This file was automatically generated by gh-aw (v0.86.2). DO NOT EDIT. To debug this workflow, load the skill at https://github.com/github/gh-aw/blob/main/debug.md # diff --git a/.github/workflows/pr-sous-chef.md b/.github/workflows/pr-sous-chef.md index 333c6595e..4a39ccd61 100644 --- a/.github/workflows/pr-sous-chef.md +++ b/.github/workflows/pr-sous-chef.md @@ -368,7 +368,6 @@ recommendations visible; wrap verbose detail in ## agent: `pr-processor` --- description: Decides skip/nudge actions for a single pull request using a minimal number of API calls -model: small --- You are given one PR number and its compact metadata. Decide what should happen to it, using as few tool calls as possible. diff --git a/.github/workflows/review-rust.lock.yml b/.github/workflows/review-rust.lock.yml index 1feaf1ebd..c507177fc 100644 --- a/.github/workflows/review-rust.lock.yml +++ b/.github/workflows/review-rust.lock.yml @@ -1,4 +1,4 @@ -# gh-aw-metadata: {"schema_version":"v4","frontmatter_hash":"ea67889e811c8c0a7ac2e9a1b512ef32f072734d71026f82eae0a8489198f59f","body_hash":"c6aacd86f655324be0dfa10d467221cdb5bf4e14431f715b874744d2ee0304c5","compiler_version":"v0.86.2","strict":true,"agent_id":"copilot","engine_versions":{"copilot":"1.0.79"}} +# gh-aw-metadata: {"schema_version":"v4","frontmatter_hash":"ea67889e811c8c0a7ac2e9a1b512ef32f072734d71026f82eae0a8489198f59f","body_hash":"5b9b49749ef63834afcced2cd715df56c5717bc69179daca0c867cc6941e0f84","compiler_version":"v0.86.2","strict":true,"agent_id":"copilot","engine_versions":{"copilot":"1.0.79"}} # gh-aw-manifest: {"version":1,"secrets":["GH_AW_GITHUB_MCP_SERVER_TOKEN","GH_AW_GITHUB_TOKEN","GITHUB_TOKEN"],"actions":[{"repo":"actions/cache","sha":"55cc8345863c7cc4c66a329aec7e433d2d1c52a9","version":"v6.1.0"},{"repo":"actions/cache/restore","sha":"55cc8345863c7cc4c66a329aec7e433d2d1c52a9","version":"v6.1.0"},{"repo":"actions/cache/save","sha":"55cc8345863c7cc4c66a329aec7e433d2d1c52a9","version":"v6.1.0"},{"repo":"actions/checkout","sha":"3d3c42e5aac5ba805825da76410c181273ba90b1","version":"v7.0.1"},{"repo":"actions/download-artifact","sha":"3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c","version":"v8.0.1"},{"repo":"actions/github-script","sha":"3a2844b7e9c422d3c10d287c895573f7108da1b3","version":"v9.0.0"},{"repo":"actions/setup-node","sha":"820762786026740c76f36085b0efc47a31fe5020","version":"v7.0.0"},{"repo":"actions/upload-artifact","sha":"043fb46d1a93c77aae656e7c1c64a875d1fc6a0a","version":"v7.0.1"},{"repo":"github/gh-aw-actions/setup","sha":"6aab9e5b5c91c615506061f09bedd81a23babe3c","version":"v0.86.2"}],"containers":[{"image":"ghcr.io/github/gh-aw-firewall/agent:0.27.44","digest":"sha256:0d727725c737b58c7bdf51f640cffb928385ec46517e0917c7f1a02f1bada8b4","pinned_image":"ghcr.io/github/gh-aw-firewall/agent:0.27.44@sha256:0d727725c737b58c7bdf51f640cffb928385ec46517e0917c7f1a02f1bada8b4"},{"image":"ghcr.io/github/gh-aw-firewall/api-proxy:0.27.44","digest":"sha256:b50fbadba138f6e9aba94aca09711335c489bb3b15861220cb66f6092e042dc7","pinned_image":"ghcr.io/github/gh-aw-firewall/api-proxy:0.27.44@sha256:b50fbadba138f6e9aba94aca09711335c489bb3b15861220cb66f6092e042dc7"},{"image":"ghcr.io/github/gh-aw-firewall/squid:0.27.44","digest":"sha256:83e48bbe12c634be8c228a576832fe45f66c529ac3659db92bddbcf2eeb6d627","pinned_image":"ghcr.io/github/gh-aw-firewall/squid:0.27.44@sha256:83e48bbe12c634be8c228a576832fe45f66c529ac3659db92bddbcf2eeb6d627"},{"image":"ghcr.io/github/gh-aw-mcpg:v0.4.9","digest":"sha256:e5a1569aeaf41820fa7bdee3e94468cae448133cdbf00119ad24f5b74db1ab9f","pinned_image":"ghcr.io/github/gh-aw-mcpg:v0.4.9@sha256:e5a1569aeaf41820fa7bdee3e94468cae448133cdbf00119ad24f5b74db1ab9f"},{"image":"ghcr.io/github/gh-aw-node","digest":"sha256:0d9f1fb5fd6610c0ac1f5194a38e45a8a1e81f8a390d5142d8e4e6f26a4b3196","pinned_image":"ghcr.io/github/gh-aw-node@sha256:0d9f1fb5fd6610c0ac1f5194a38e45a8a1e81f8a390d5142d8e4e6f26a4b3196"},{"image":"ghcr.io/github/github-mcp-server:v1.9.0","digest":"sha256:881b53d6f75f69bdbc1b5b10fc2f1361717c19054143b3a8529fb5c32061a50e","pinned_image":"ghcr.io/github/github-mcp-server:v1.9.0@sha256:881b53d6f75f69bdbc1b5b10fc2f1361717c19054143b3a8529fb5c32061a50e"}],"has_pull_request":true} # This file was automatically generated by gh-aw (v0.86.2). DO NOT EDIT. To debug this workflow, load the skill at https://github.com/github/gh-aw/blob/main/debug.md # diff --git a/.github/workflows/review-rust.md b/.github/workflows/review-rust.md index 136182b1b..2da67b265 100644 --- a/.github/workflows/review-rust.md +++ b/.github/workflows/review-rust.md @@ -169,7 +169,6 @@ and the themes in a `
` block. ## agent: `rust-critic` --- description: Hostile first-pass Rust reviewer that mines merge-blocking defects from changed lines -model: small --- You are a hostile senior Rust reviewer performing a first-pass audit. diff --git a/.github/workflows/review-typescript.lock.yml b/.github/workflows/review-typescript.lock.yml index 5af4a1983..24752ad10 100644 --- a/.github/workflows/review-typescript.lock.yml +++ b/.github/workflows/review-typescript.lock.yml @@ -1,4 +1,4 @@ -# gh-aw-metadata: {"schema_version":"v4","frontmatter_hash":"7259abb8515a4bb6dd1d124c1db92aa00b5e5357d4b61b78ae045ba045d425ac","body_hash":"684f375762c2d2dfb48458e8efe8760e29325adddb9edfdd9b22990ac7f3b8a7","compiler_version":"v0.86.2","strict":true,"agent_id":"copilot","engine_versions":{"copilot":"1.0.79"}} +# gh-aw-metadata: {"schema_version":"v4","frontmatter_hash":"7259abb8515a4bb6dd1d124c1db92aa00b5e5357d4b61b78ae045ba045d425ac","body_hash":"e370274e6f63c5c3accfa7545ad59e1499267f58360fe90346f2a5f5c58c12bd","compiler_version":"v0.86.2","strict":true,"agent_id":"copilot","engine_versions":{"copilot":"1.0.79"}} # gh-aw-manifest: {"version":1,"secrets":["GH_AW_GITHUB_MCP_SERVER_TOKEN","GH_AW_GITHUB_TOKEN","GITHUB_TOKEN"],"actions":[{"repo":"actions/cache","sha":"55cc8345863c7cc4c66a329aec7e433d2d1c52a9","version":"v6.1.0"},{"repo":"actions/cache/restore","sha":"55cc8345863c7cc4c66a329aec7e433d2d1c52a9","version":"v6.1.0"},{"repo":"actions/cache/save","sha":"55cc8345863c7cc4c66a329aec7e433d2d1c52a9","version":"v6.1.0"},{"repo":"actions/checkout","sha":"3d3c42e5aac5ba805825da76410c181273ba90b1","version":"v7.0.1"},{"repo":"actions/download-artifact","sha":"3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c","version":"v8.0.1"},{"repo":"actions/github-script","sha":"3a2844b7e9c422d3c10d287c895573f7108da1b3","version":"v9.0.0"},{"repo":"actions/setup-node","sha":"820762786026740c76f36085b0efc47a31fe5020","version":"v7.0.0"},{"repo":"actions/upload-artifact","sha":"043fb46d1a93c77aae656e7c1c64a875d1fc6a0a","version":"v7.0.1"},{"repo":"github/gh-aw-actions/setup","sha":"6aab9e5b5c91c615506061f09bedd81a23babe3c","version":"v0.86.2"}],"containers":[{"image":"ghcr.io/github/gh-aw-firewall/agent:0.27.44","digest":"sha256:0d727725c737b58c7bdf51f640cffb928385ec46517e0917c7f1a02f1bada8b4","pinned_image":"ghcr.io/github/gh-aw-firewall/agent:0.27.44@sha256:0d727725c737b58c7bdf51f640cffb928385ec46517e0917c7f1a02f1bada8b4"},{"image":"ghcr.io/github/gh-aw-firewall/api-proxy:0.27.44","digest":"sha256:b50fbadba138f6e9aba94aca09711335c489bb3b15861220cb66f6092e042dc7","pinned_image":"ghcr.io/github/gh-aw-firewall/api-proxy:0.27.44@sha256:b50fbadba138f6e9aba94aca09711335c489bb3b15861220cb66f6092e042dc7"},{"image":"ghcr.io/github/gh-aw-firewall/squid:0.27.44","digest":"sha256:83e48bbe12c634be8c228a576832fe45f66c529ac3659db92bddbcf2eeb6d627","pinned_image":"ghcr.io/github/gh-aw-firewall/squid:0.27.44@sha256:83e48bbe12c634be8c228a576832fe45f66c529ac3659db92bddbcf2eeb6d627"},{"image":"ghcr.io/github/gh-aw-mcpg:v0.4.9","digest":"sha256:e5a1569aeaf41820fa7bdee3e94468cae448133cdbf00119ad24f5b74db1ab9f","pinned_image":"ghcr.io/github/gh-aw-mcpg:v0.4.9@sha256:e5a1569aeaf41820fa7bdee3e94468cae448133cdbf00119ad24f5b74db1ab9f"},{"image":"ghcr.io/github/gh-aw-node","digest":"sha256:0d9f1fb5fd6610c0ac1f5194a38e45a8a1e81f8a390d5142d8e4e6f26a4b3196","pinned_image":"ghcr.io/github/gh-aw-node@sha256:0d9f1fb5fd6610c0ac1f5194a38e45a8a1e81f8a390d5142d8e4e6f26a4b3196"},{"image":"ghcr.io/github/github-mcp-server:v1.9.0","digest":"sha256:881b53d6f75f69bdbc1b5b10fc2f1361717c19054143b3a8529fb5c32061a50e","pinned_image":"ghcr.io/github/github-mcp-server:v1.9.0@sha256:881b53d6f75f69bdbc1b5b10fc2f1361717c19054143b3a8529fb5c32061a50e"}],"has_pull_request":true} # This file was automatically generated by gh-aw (v0.86.2). DO NOT EDIT. To debug this workflow, load the skill at https://github.com/github/gh-aw/blob/main/debug.md # diff --git a/.github/workflows/review-typescript.md b/.github/workflows/review-typescript.md index 4b99d2d16..a19db8f12 100644 --- a/.github/workflows/review-typescript.md +++ b/.github/workflows/review-typescript.md @@ -171,7 +171,6 @@ wrong output; otherwise `COMMENT`. ## agent: `ts-critic` --- description: Hostile first-pass TypeScript reviewer for bundled Azure DevOps runtime helpers -model: small --- You are a hostile senior TypeScript reviewer performing a first-pass audit of code that is bundled and executed on Azure DevOps build agents. diff --git a/AGENTS.md b/AGENTS.md index 2cd957f00..ed63bfe48 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -319,6 +319,7 @@ fail-closed and only pauses when the agent actually proposed a reviewed output. │ ├── conclusion/ # Conclusion-job reporter source (bundled to conclusion.js) │ ├── approval-summary/ # Safe-outputs summary renderer (bundled to approval-summary.js; end-of-Agent-job summary tab) │ ├── github-app-token/ # GitHub App token minter (bundled to github-app-token.js; mints installation token in Agent + Detection when engine.github-app-token is set) +│ ├── copilot-invoker/ # Sandboxed Copilot process harness (bundled to copilot-invoker.js): strict versioned invocation/result documents, runtime model resolution, typed argv, signal forwarding, exact exit propagation │ ├── executor-e2e/ # Stage 3 safe-output E2E test harness (not a bundle; runs deterministic scenarios against a real ADO project and files a GitHub issue on failure) │ ├── compiler-smoke-e2e/ # Smoke E2E orchestrator (not a bundle): stages each case in `tests/smoke/cases.json` to the fixed `.smoke/pipeline.yml` path on its own per-case `ado-aw-mirror` ref, queues it against its credential *lane* definition, and asserts they go green. Two modes via `SMOKE_COMPILER_SOURCE`: `candidate` (compiler built from this commit, pinned pipeline-artifact) and `released` (latest release asset, release URLs required). Built to `test-bin/` by `build:compiler-smoke-e2e`, listed in `NON_BUNDLE_DIRS`. │ ├── prepare-pr-base/ # create-pull-request preparer (bundled to prepare-pr-base.js): Agent mode uses ADO diff metadata + bounded fallback; SafeOutputs fetches the target tip; cross-org targets use isolated credentials + exact remote matching @@ -463,7 +464,7 @@ index to jump to the right page. (`gate.js`, `import.js`, the execution-context `exec-context-*.js` bundles, `conclusion.js`, `approval-summary.js`, `github-app-token.js`, `prepare-pr-base.js`, and - `azure-wif-refresh.js`), schemars-driven + `azure-wif-refresh.js`, `copilot-invoker.js`), schemars-driven type codegen, the A2 design decision, the bundle env contract modelled in `src/compile/ado_bundle.rs`, and the `trigger-e2e/` gate-spec drift guard (kept in sync via `export-fact-catalog`). diff --git a/docs/ado-script.md b/docs/ado-script.md index fa1c726e5..9c932e97a 100644 --- a/docs/ado-script.md +++ b/docs/ado-script.md @@ -98,6 +98,13 @@ pipeline** as runtime helpers. Today it produces the following shipped bundles: never enter the agent, MCP environment, Docker arguments, logs, status documents, or artifacts. See [`mcp.md`](mcp.md#renewable-azure-workload-identity). +- `copilot-invoker.js` — the sandboxed Copilot process harness used by every + Agent job and every enabled Detection job. It validates a versioned, + compiler-emitted invocation document, resolves role-specific runtime model + variables, constructs Copilot argv without a shell, forwards termination + signals, publishes the requested-model result, and preserves Copilot's exit + status. It does not own AWF topology, credentials, retries, or OTel + interpretation. > **Internal-only.** `ado-script` is not a user-facing front-matter > feature. Authors never write an `ado-script:` block in their agent @@ -666,6 +673,9 @@ scripts/ado-script/ │ ├── azure-wif-refresh/ # azure-wif-refresh.js entry point + renewable WIF assertion sidecar │ │ ├── index.ts # main(): rotate a private token file for user-defined stdio MCP servers │ │ └── __tests__/ # unit tests for rotation and isolation behaviour +│ ├── copilot-invoker/ # copilot-invoker.js entry point + versioned Copilot process harness +│ │ ├── index.ts # document/result validation, model resolution, argv, signals, exact exit +│ │ └── index.test.ts # schema, precedence, argv, environment, result, and process tests │ ├── trigger-e2e/ # test-only: FACT_META gate-spec table + trigger-evaluation E2E scenarios (not a bundle) │ │ ├── gate-spec.ts # FACT_META mirror of Rust Fact::ALL; drift-guarded by export-fact-catalog + fact-catalog.gen.json │ │ ├── fact-catalog.gen.json # generated by `cargo run -- export-fact-catalog`; deep-compared by gate-spec.test.ts @@ -689,7 +699,8 @@ scripts/ado-script/ ├── github-app-token.js # ncc bundle output (gitignored) ├── prepare-pr-base.js # ncc bundle output (gitignored) ├── ado-proxy.js # ncc bundle output (gitignored) -└── azure-wif-refresh.js # ncc bundle output (gitignored) +├── azure-wif-refresh.js # ncc bundle output (gitignored) +└── copilot-invoker.js # ncc bundle output (gitignored) ``` The release workflow (`.github/workflows/release.yml`) runs @@ -700,8 +711,8 @@ captures every bundle, including `gate.js`, `import.js`, `exec-context-ci-push.js`, `exec-context-workitem.js`, `exec-context-schedule.js`, `exec-context-pr-checks.js`, `exec-context-repo.js`, `conclusion.js`, `approval-summary.js`, -`github-app-token.js`, `prepare-pr-base.js`, `ado-proxy.js`, and -`azure-wif-refresh.js` — into the +`github-app-token.js`, `prepare-pr-base.js`, `ado-proxy.js`, +`azure-wif-refresh.js`, and `copilot-invoker.js` — into the `ado-script.zip` release asset. Pipelines download that asset at runtime by URL pinned to the compiler's `CARGO_PKG_VERSION`, verify its SHA-256 against the `checksums.txt` asset, then extract. @@ -742,9 +753,8 @@ cargo run -- export-gate-schema --output schema/gate-spec.schema.json `AdoScriptExtension` (`src/compile/extensions/ado_script.rs`) is the always-on single -extension that owns all `ado-script` wiring. It has two independent -features, each emitted **into the job that actually consumes the -bundle**: +extension that owns shared `ado-script` delivery. Each download is emitted +**into the job that actually consumes the bundle**: ### Setup job (gate evaluator) @@ -762,12 +772,11 @@ returns three typed `Declarations::setup_steps` entries for the Setup job: runs the gate with `GATE_SPEC` and the env-var contract documented above. -### Agent job (runtime-import resolver + PR-context precompute) +### Agent job (Copilot invoker + optional helpers) -When `inlined-imports: false` (the default) OR the execution-context -PR contributor activates (`on.pr` configured and not disabled), -`AdoScriptExtension::declarations()` returns the install + download pair in -`Declarations::agent_prepare_steps` for the Agent job: +Every Agent job receives the install + download pair in +`Declarations::agent_prepare_steps`, because every Copilot run uses +`copilot-invoker.js`: 1. **`UseNode@1`** — same shape as above. 2. **`curl` download + verify + extract** — same artefact, same @@ -777,6 +786,11 @@ PR contributor activates (`on.pr` configured and not disabled), `/tmp/awf-tools/agent-prompt.md` in place. See [`runtime-imports.md`](runtime-imports.md) for marker syntax. **Only emitted when `inlined-imports: false`.** +4. The AWF run executes the fixed command + `node /tmp/ado-aw-scripts/ado-script/copilot-invoker.js run + /tmp/awf-tools/copilot-invocation.json`. Author values, prompt contents, and + Azure DevOps model macros are data in the invocation document or typed + environment mappings, never shell fragments. The PR-context precompute step (`node exec-context-pr.js`) is owned by `ExecContextExtension` (not `AdoScriptExtension`) and emitted through @@ -785,27 +799,53 @@ its own Tool-phase `Declarations::agent_prepare_steps`. Phase ordering guarantees the bundle is installed and on disk before the exec-context invocation runs. +### Detection job + +Every enabled Detection job installs Node and stages `ado-script.zip`, then +uses the same fixed invoker command with a Detection-role document. Detection +does not receive the Agent MCP configuration. Workflows that explicitly disable +threat detection emit no Detection download, document, model mappings, or +invoker command. + +### Copilot invocation and result protocol + +The compiler writes schema version 1 JSON under +`/tmp/awf-tools/copilot-invocation.json`. The document contains only +compiler-validated, non-secret data: role, command, prompt path, optional MCP +config path, argv entries, optional explicit model, and result path. The +invoker rejects unknown fields and unsupported versions, reads the prompt as +UTF-8, spawns Copilot without a shell, and sets or removes `COPILOT_MODEL` only +in the child environment. + +Before spawning Copilot, the invoker atomically writes +`/tmp/awf-tools/copilot-invocation-result.json` with the role and requested +model. Trusted host code validates that document with the invoker's +`read-result` mode and merges it into `aw_info.json`. The invoker never edits +trusted metadata and never receives Stage 3 credentials. + ### Per-job download (NOT a duplication bug) ADO jobs use **isolated VMs** — `/tmp` is not shared between jobs. The `ado-script.zip` bundle therefore has to be downloaded once per -job that consumes it. When both Setup and Agent need it, install + -download steps appear in **both**. That's correct architecture given -ADO's topology, not waste. +job that consumes it. Agent and enabled Detection therefore always stage the +bundle; Setup stages another copy when a gate or synthetic-PR resolver needs +one. That's correct architecture given ADO's topology, not waste. ### What gets emitted, by case The rows below assume the synthetic-PR resolver is **not** active (`pr_trigger_for_synth = None`): -| Setup consumer | Agent consumer | Setup-job steps | Agent-job extra steps | +| Setup consumer | Agent optional consumers | Setup-job steps | Agent-job steps | |---|---|---|---| -| no gate | none | (none) | (none) | -| no gate | `inlined-imports: false` only | (no Setup job) | install + download + resolver | -| no gate | execution-context contributor(s) only | (no Setup job) | install + download + exec-context bundle(s) | -| no gate | resolver + execution-context | (no Setup job) | install + download + resolver + exec-context bundle(s) | -| gate | none | install + download + gate | (none) | -| gate | any combination of resolver / exec-context | install + download + gate | install + download + (resolver?) + (exec-context bundle(s)?) | +| no gate | none | (none) | install + download + invoker | +| no gate | runtime imports | (none) | install + download + resolver + invoker | +| no gate | execution-context contributor(s) | (none) | install + download + exec-context bundle(s) + invoker | +| gate | none | install + download + gate | install + download + invoker | +| gate | resolver / exec-context | install + download + gate | install + download + optional helpers + invoker | + +Every row also has an enabled Detection job with its own install + download + +invoker unless threat detection is explicitly disabled. When the synthetic-PR resolver **is** active (`pr_trigger_for_synth = Some(_)`, i.e. `synthetic_pr_active()` is @@ -813,18 +853,16 @@ true) the Setup job gains the `synthPr` step (`node exec-context-pr-synth.js`) before any gate step — and the Setup job is emitted even with no gate: -| Setup consumer | Setup-job steps | Agent-job extra steps | +| Setup consumer | Setup-job steps | Agent-job steps | |---|---|---| -| synth-PR (no gate) | install + download + synth-PR | (per Agent consumer above) | -| gate (no synth-PR) | install + download + gate | (per Agent consumer above) | -| synth-PR + gate | install + download + synth-PR + gate | (per Agent consumer above) | +| synth-PR (no gate) | install + download + synth-PR | install + download + optional helpers + invoker | +| gate (no synth-PR) | install + download + gate | install + download + optional helpers + invoker | +| synth-PR + gate | install + download + synth-PR + gate | install + download + optional helpers + invoker | The "Setup consumer" column is gated on `filters:` lowering to non-empty -checks **or** `synthetic_pr_active()` being true. The "Agent consumer" -columns are gated on `inlined-imports: false` (resolver) and the PR -contributor's activation predicate (exec-context-pr; see -`pr_contributor_will_activate` in -`src/compile/extensions/exec_context/mod.rs`). +checks **or** `synthetic_pr_active()` being true. The Agent download is +unconditional; optional resolver and execution-context steps retain their own +feature predicates. The IR-to-bash codegen that produces the gate step is `compile_gate_step_external` in `src/compile/filter_ir.rs`. diff --git a/docs/audit.md b/docs/audit.md index e2fdf6980..b2bd46de1 100644 --- a/docs/audit.md +++ b/docs/audit.md @@ -61,13 +61,25 @@ URL-encoded project segments are decoded before the ADO context is resolved. `t= │ ├── mcpg/ # MCP Gateway logs (includes the SafeOutputs stdio child's stdout/stderr) │ └── agent-output.txt # Filtered agent stdout ├── analyzed_outputs[_]/ # Downloaded artifact (Detection stage) +│ ├── aw_info.json # Agent metadata + Detection runtime model │ ├── threat-analysis.json # Aggregate verdict + reasons │ └── threat-analysis-output.txt └── safe_outputs[_]/ # Downloaded artifact (SafeOutputs stage) └── safe-outputs-executed.ndjson # Per-item execution log ``` -`aw_info.json`, `otel.jsonl`, and `safe_outputs.ndjson` are searched in `staging/` first and then at the artifact top level so older layouts still audit cleanly. +Agent `aw_info.json`, `otel.jsonl`, and `safe_outputs.ndjson` are searched in +`staging/` first and then at the artifact top level so older layouts still +audit cleanly. When present, the Detection-enriched `aw_info.json` from +`analyzed_outputs` overlays only Detection-owned runtime fields in the report. + +`overview.aw_info.model` is the model requested for the Agent's Copilot +session. Copilot custom-agent definitions may pin a different model. When OTel +contains `gen_ai.request.model`, `engine_config.model` is the model observed +during execution; otherwise it falls back to the requested model. Console +output shows `requested_model` and `observed_model` separately when they differ. +`overview.aw_info.detection_model` is the requested Detection session model; +Detection OTel is not currently included in the analyzed artifact. ## Report shape (`AuditData`) @@ -75,7 +87,7 @@ Current top-level keys include the following. Optional sections are omitted from | Key | Source | | --- | --- | -| `overview` | ADO build metadata + `aw_info.json` (engine, model, optional threat-detection enabled/engine/model, agent name, source, target). | +| `overview` | ADO build metadata + `aw_info.json` (engine, requested session model, optional threat-detection enabled/engine/requested model, agent name, source, target). | | `task_domain` | Audit heuristics over the run's prompts and outputs. | | `behavior_fingerprint` | Higher-level audit heuristics over the run's behavior. | | `agentic_assessments` | Higher-level audit assessments emitted by the analyzers. | @@ -83,7 +95,7 @@ Current top-level keys include the following. Optional sections are omitted from | `key_findings` | Heuristic rules + analyzer-emitted findings (for example aggregate-gate rejection). | | `recommendations` | Follow-up actions derived from findings. | | `performance_metrics` | Derived from `metrics`, runtime duration, tool usage, and firewall counts. | -| `engine_config` | Runtime engine configuration derived from `aw_info.json`. | +| `engine_config` | Runtime engine configuration; the Agent model prefers the OTel-observed model and falls back to the requested `aw_info.json` model. | | `safe_output_summary` | Counts of proposed / executed / rejected / not processed items. | | `safe_output_execution` | Per-item trace joining proposal + detection + execution. | | `rejected_safe_outputs` | Rollup of rejections by reason / threat flag. | diff --git a/docs/engine.md b/docs/engine.md index ce7724e07..0c39757bb 100644 --- a/docs/engine.md +++ b/docs/engine.md @@ -22,18 +22,70 @@ engine: | Field | Type | Default | Description | |-------|------|---------|-------------| | `id` | string | `copilot` | Engine identifier. Currently only `copilot` (GitHub Copilot CLI) is supported. | -| `model` | string | *(none)* | AI model to use (e.g., `gpt-5-mini`). When omitted, the compiler does not emit `--model` and the Copilot CLI chooses its own default. When set, the compiler passes the value directly to the Copilot CLI `--model` flag — any model identifier the Copilot CLI accepts is valid. | +| `model` | string | *(none)* | AI model to use (e.g., `gpt-5-mini`). When set, the compiler records it in the versioned Copilot invocation document and the invoker sets Copilot CLI's native `COPILOT_MODEL` environment variable. When omitted, runtime model controls can select a model; if no runtime control is set, `COPILOT_MODEL` remains unset and the Copilot CLI chooses its own default. | | `timeout-minutes` | integer | *(none)* | Maximum time in minutes the agent job is allowed to run. Sets `timeoutInMinutes` on the `Agent` job in the generated pipeline. | | `version` | string | *(none)* | Engine CLI version to install (e.g., `"1.0.70"`, `"latest"`). Overrides the pinned `COPILOT_CLI_VERSION`. Set to `"latest"` to use the newest available version. | | `agent` | string | *(none)* | Custom agent file identifier (Copilot only). Adds `--agent ` to the CLI invocation, selecting a custom agent from `.github/agents/`. | | `api-target` | string | *(none)* | Custom API endpoint hostname for GHES/GHEC (e.g., `"api.acme.ghe.com"`). Adds `--api-target ` to the CLI invocation and adds the hostname to the AWF network allowlist. | -| `args` | list | `[]` | Custom CLI arguments appended after compiler-generated args. Subject to shell-safety validation and blocked from overriding compiler-controlled flags (`--prompt`, `--additional-mcp-config`, `--allow-tool`, `--allow-all-tools`, `--allow-all-paths`, `--disable-builtin-mcps`, `--no-ask-user`, `--ask-user`). | -| `env` | map | *(none)* | Engine-specific environment variables merged into the sandbox step's `env:` block. Keys must be valid env var names. Values are literal-only and must not contain ADO expressions (`$(`, `${{`, `$[`) or pipeline command injection (`##vso[`), **except** the Copilot provider keys (`COPILOT_PROVIDER_BASE_URL`, `COPILOT_PROVIDER_API_KEY`, `COPILOT_PROVIDER_BEARER_TOKEN`, `COPILOT_PROVIDER_WIRE_API`), which may carry an ADO macro (`$(...)`) expression. Prefer the typed [`provider`](#copilot-model-provider-byok-configuration) block over raw provider env keys. Compiler-controlled keys (`GITHUB_TOKEN`, `PATH`, `BASH_ENV`, etc.) are blocked. | +| `args` | list | `[]` | Custom CLI arguments appended after compiler-generated args. Subject to shell-safety validation and blocked from overriding compiler-controlled flags (`--prompt`, `--model`, `--additional-mcp-config`, `--allow-tool`, `--allow-all-tools`, `--allow-all-paths`, `--disable-builtin-mcps`, `--no-ask-user`, `--ask-user`). Use `engine.model` or the runtime variables below instead of a raw `--model` argument. | +| `env` | map | *(none)* | Engine-specific environment variables merged into the sandbox step's `env:` block. Keys must be valid env var names. Values are literal-only and must not contain ADO expressions (`$(`, `${{`, `$[`) or pipeline command injection (`##vso[`), **except** the Copilot provider keys (`COPILOT_PROVIDER_BASE_URL`, `COPILOT_PROVIDER_API_KEY`, `COPILOT_PROVIDER_BEARER_TOKEN`, `COPILOT_PROVIDER_WIRE_API`), which may carry an ADO macro (`$(...)`) expression. Prefer the typed [`provider`](#copilot-model-provider-byok-configuration) block over raw provider env keys. Compiler-controlled keys (`GITHUB_TOKEN`, `COPILOT_MODEL`, `PATH`, `BASH_ENV`, etc.) are blocked. | | `provider` | map | *(none)* | Copilot external model-provider (BYOK) configuration: `base-url`, `type`, `wire-api`, `token` (compiler-minted bearer via a service connection), `api-key`. Maps to the `COPILOT_PROVIDER_*` env vars. See [Copilot model provider (BYOK) configuration](#copilot-model-provider-byok-configuration). | | `command` | string | *(none)* | Custom engine executable path (skips the default engine binary installation — NuGet for `target: 1es`, GitHub Releases for all other targets). The path must be accessible inside the AWF container (e.g., `/tmp/...` or workspace-mounted paths). | | `github-app-token` | map | *(none)* | GitHub App-backed Copilot engine authentication. When set, the compiler mints (and, by default, revokes) a GitHub App installation token in the Agent and Detection jobs and sources `GITHUB_TOKEN` from it (for Copilot only). See [GitHub App-backed Copilot engine auth](#github-app-backed-copilot-engine-auth). | +### Runtime model controls + +For the Copilot engine, operators can switch models at Azure DevOps pipeline +runtime without editing workflow markdown or recompiling lock files. Configure +these as pipeline variables or variable-group entries: + +| Variable | Applies to | +|----------|------------| +| `ADO_AW_MODEL_AGENT_COPILOT` | Agent job only | +| `ADO_AW_MODEL_DETECTION_COPILOT` | Detection job only | +| `ADO_AW_DEFAULT_MODEL_COPILOT` | Fallback for both jobs | + +Precedence is: + +1. Explicit `engine.model` for the effective engine config. +2. Role-specific runtime variable (`ADO_AW_MODEL_AGENT_COPILOT` or + `ADO_AW_MODEL_DETECTION_COPILOT`). +3. Shared runtime variable (`ADO_AW_DEFAULT_MODEL_COPILOT`). +4. Existing default behavior (`COPILOT_MODEL` is unset; the Copilot CLI + chooses). + +Detection uses its effective engine config after applying +`safe-outputs.threat-detection.engine`, so an inherited or nested explicit model +still wins over runtime variables. This mirrors gh-aw and means the runtime +variables are an operational escape hatch only for workflows that leave +`engine.model` unset; changing a pinned frontmatter model still requires +recompilation. + +Runtime values are passed through typed step environment mappings, so Azure +DevOps YAML variables, UI variables, variable groups, and variables set by an +earlier trusted `##vso[task.setvariable]` step all resolve at task start. Inside +AWF, the compiler-owned `copilot-invoker.js` validates the selected value and +sets Copilot CLI's native `COPILOT_MODEL` only in the child process environment. +When no value resolves, the invoker removes `COPILOT_MODEL` rather than +supplying a compiler default. Raw `engine.args --model` and +`engine.env.COPILOT_MODEL` are rejected so they cannot bypass this precedence. + +The invoker writes a strict result document before starting Copilot. The trusted +Agent host task reads that result after AWF returns and records the requested +session model in `aw_info.json`; Detection uses the same result contract and +later enriches the copied metadata in `analyzed_outputs_` from its own +job scope. A prior trusted step can therefore set a variable and both execution +and metadata see the same task-start value. `ado-aw audit` merges those +job-owned fields. + +When `engine.agent` selects a custom agent whose definition declares `model` or +`models`, Copilot CLI may use that agent-pinned model instead of the requested +session model. `aw_info.json` records the requested session model; when Copilot +OTel is present, `ado-aw audit` reports the observed Agent model separately. +Detection metadata is requested-model-only because the analyzed artifact does +not currently include Detection OTel. + ### `timeout-minutes` The `timeout-minutes` field sets a wall-clock limit (in minutes) for the entire agent job. It maps to the Azure DevOps `timeoutInMinutes` job property on `Agent`. This is useful for: @@ -246,7 +298,8 @@ runtime (raw `engine.env` cross-job macros like `$(Setup.FOUNDRY_TOKEN)` do | `token` | optional | `COPILOT_PROVIDER_API_KEY` | Compiler-minted credential via Azure CLI (see below). Mutually exclusive with `api-key`. | | `api-key` | optional | `COPILOT_PROVIDER_API_KEY` | Static API key, typically a `$(VAR)` secret pipeline variable. Mutually exclusive with `token`. | -The model itself is set via `engine.model` (or a `COPILOT_MODEL` env var). +The model itself is set via `engine.model` or the +[`ADO_AW_MODEL_*`](#runtime-model-controls) runtime controls. #### Compiler-owned token acquisition (`provider.token`) diff --git a/scripts/ado-script/.gitignore b/scripts/ado-script/.gitignore index 019aca5c2..196e2b199 100644 --- a/scripts/ado-script/.gitignore +++ b/scripts/ado-script/.gitignore @@ -17,6 +17,7 @@ github-app-token.js prepare-pr-base.js ado-proxy.js azure-wif-refresh.js +copilot-invoker.js schema *.tsbuildinfo test-bin diff --git a/scripts/ado-script/package.json b/scripts/ado-script/package.json index 4e6c5e1f7..049ae9d8a 100644 --- a/scripts/ado-script/package.json +++ b/scripts/ado-script/package.json @@ -7,8 +7,8 @@ "node": ">=20.0.0" }, "scripts": { - "build": "npm run codegen && npm run clean && npm run build:gate && npm run build:import && npm run build:exec-context-pr && npm run build:exec-context-pr-synth && npm run build:exec-context-manual && npm run build:exec-context-pipeline && npm run build:exec-context-ci-push && npm run build:exec-context-workitem && npm run build:exec-context-schedule && npm run build:exec-context-pr-checks && npm run build:exec-context-repo && npm run build:conclusion && npm run build:approval-summary && npm run build:github-app-token && npm run build:prepare-pr-base && npm run build:ado-proxy && npm run build:azure-wif-refresh", - "clean": "node -e \"const fs=require('node:fs'); fs.rmSync('.ado-build',{recursive:true,force:true}); for (const n of ['gate','import','exec-context-pr','exec-context-pr-synth','exec-context-manual','exec-context-pipeline','exec-context-ci-push','exec-context-workitem','exec-context-schedule','exec-context-pr-checks','exec-context-repo','conclusion','approval-summary','github-app-token','prepare-pr-base','ado-proxy','azure-wif-refresh']) fs.rmSync(n+'.js',{force:true});\"", + "build": "npm run codegen && npm run clean && npm run build:gate && npm run build:import && npm run build:exec-context-pr && npm run build:exec-context-pr-synth && npm run build:exec-context-manual && npm run build:exec-context-pipeline && npm run build:exec-context-ci-push && npm run build:exec-context-workitem && npm run build:exec-context-schedule && npm run build:exec-context-pr-checks && npm run build:exec-context-repo && npm run build:conclusion && npm run build:approval-summary && npm run build:github-app-token && npm run build:prepare-pr-base && npm run build:ado-proxy && npm run build:azure-wif-refresh && npm run build:copilot-invoker", + "clean": "node -e \"const fs=require('node:fs'); fs.rmSync('.ado-build',{recursive:true,force:true}); for (const n of ['gate','import','exec-context-pr','exec-context-pr-synth','exec-context-manual','exec-context-pipeline','exec-context-ci-push','exec-context-workitem','exec-context-schedule','exec-context-pr-checks','exec-context-repo','conclusion','approval-summary','github-app-token','prepare-pr-base','ado-proxy','azure-wif-refresh','copilot-invoker']) fs.rmSync(n+'.js',{force:true});\"", "build:gate": "ncc build src/gate/index.ts -o .ado-build/gate -m -t && node -e \"const fs=require('node:fs'); fs.copyFileSync('.ado-build/gate/index.js','gate.js'); fs.rmSync('.ado-build/gate',{recursive:true,force:true});\"", "build:import": "ncc build src/import/index.ts -o .ado-build/import -m -t && node -e \"const fs=require('node:fs'); fs.copyFileSync('.ado-build/import/index.js','import.js'); fs.rmSync('.ado-build/import',{recursive:true,force:true});\"", "build:exec-context-pr": "ncc build src/exec-context-pr/index.ts -o .ado-build/exec-context-pr -m -t && node -e \"const fs=require('node:fs'); fs.copyFileSync('.ado-build/exec-context-pr/index.js','exec-context-pr.js'); fs.rmSync('.ado-build/exec-context-pr',{recursive:true,force:true});\"", @@ -26,13 +26,14 @@ "build:prepare-pr-base": "ncc build src/prepare-pr-base/index.ts -o .ado-build/prepare-pr-base -m -t && node -e \"const fs=require('node:fs'); fs.copyFileSync('.ado-build/prepare-pr-base/index.js','prepare-pr-base.js'); fs.rmSync('.ado-build/prepare-pr-base',{recursive:true,force:true});\"", "build:ado-proxy": "ncc build src/ado-proxy/index.ts -o .ado-build/ado-proxy -m -t && node -e \"const fs=require('node:fs'); fs.copyFileSync('.ado-build/ado-proxy/index.js','ado-proxy.js'); fs.rmSync('.ado-build/ado-proxy',{recursive:true,force:true});\"", "build:azure-wif-refresh": "ncc build src/azure-wif-refresh/index.ts -o .ado-build/azure-wif-refresh -m -t && node -e \"const fs=require('node:fs'); fs.copyFileSync('.ado-build/azure-wif-refresh/index.js','azure-wif-refresh.js'); fs.rmSync('.ado-build/azure-wif-refresh',{recursive:true,force:true});\"", + "build:copilot-invoker": "ncc build src/copilot-invoker/index.ts -o .ado-build/copilot-invoker -m -t && node -e \"const fs=require('node:fs'); fs.copyFileSync('.ado-build/copilot-invoker/index.js','copilot-invoker.js'); fs.rmSync('.ado-build/copilot-invoker',{recursive:true,force:true});\"", "build:executor-e2e": "ncc build src/executor-e2e/index.ts -o .ado-build/executor-e2e -m -t && node -e \"const fs=require('node:fs'); fs.mkdirSync('test-bin',{recursive:true}); fs.copyFileSync('.ado-build/executor-e2e/index.js','test-bin/executor-e2e.js'); fs.rmSync('.ado-build/executor-e2e',{recursive:true,force:true});\"", "build:trigger-e2e": "ncc build src/trigger-e2e/index.ts -o .ado-build/trigger-e2e -m -t && node -e \"const fs=require('node:fs'); fs.mkdirSync('test-bin',{recursive:true}); fs.copyFileSync('.ado-build/trigger-e2e/index.js','test-bin/trigger-e2e.js'); fs.rmSync('.ado-build/trigger-e2e',{recursive:true,force:true});\"", "build:compiler-smoke-e2e": "ncc build src/compiler-smoke-e2e/index.ts -o .ado-build/compiler-smoke-e2e -m -t && node -e \"const fs=require('node:fs'); fs.mkdirSync('test-bin',{recursive:true}); fs.copyFileSync('.ado-build/compiler-smoke-e2e/index.js','test-bin/compiler-smoke-e2e.js'); fs.rmSync('.ado-build/compiler-smoke-e2e',{recursive:true,force:true});\"", "build:check": "ls -lh gate.js && wc -c gate.js", "codegen": "node -e \"require('node:fs').mkdirSync('schema', { recursive: true })\" && cargo run --quiet --manifest-path ../../Cargo.toml -- export-gate-schema --output schema/gate-spec.schema.json && npx json2ts schema/gate-spec.schema.json -o src/shared/types.gen.ts --bannerComment \"// AUTO-GENERATED from Rust IR via cargo run -- export-gate-schema. Do not edit; run npm run codegen.\" && cargo run --quiet --manifest-path ../../Cargo.toml -- export-fact-catalog --output src/trigger-e2e/fact-catalog.gen.json && cargo run --quiet --manifest-path ../../Cargo.toml -- export-ado-proxy-catalog-schema --output schema/ado-proxy-catalog.schema.json && npx json2ts schema/ado-proxy-catalog.schema.json -o src/shared/ado-proxy-catalog.types.gen.ts --bannerComment \"// AUTO-GENERATED from Rust via cargo run -- export-ado-proxy-catalog-schema. Do not edit; run npm run codegen.\" && cargo run --quiet --manifest-path ../../Cargo.toml -- export-ado-proxy-catalog --output src/ado-proxy/catalog.gen.json", "test": "vitest run", - "test:smoke": "npm run build:gate && npm run build:import && npm run build:exec-context-pr && npm run build:exec-context-pr-synth && npm run build:exec-context-manual && npm run build:exec-context-pipeline && npm run build:exec-context-ci-push && npm run build:exec-context-workitem && npm run build:exec-context-schedule && npm run build:exec-context-pr-checks && npm run build:exec-context-repo && npm run build:conclusion && npm run build:approval-summary && npm run build:github-app-token && npm run build:prepare-pr-base && npm run build:ado-proxy && npm run build:azure-wif-refresh && vitest run -c vitest.config.smoke.ts", + "test:smoke": "npm run build:gate && npm run build:import && npm run build:exec-context-pr && npm run build:exec-context-pr-synth && npm run build:exec-context-manual && npm run build:exec-context-pipeline && npm run build:exec-context-ci-push && npm run build:exec-context-workitem && npm run build:exec-context-schedule && npm run build:exec-context-pr-checks && npm run build:exec-context-repo && npm run build:conclusion && npm run build:approval-summary && npm run build:github-app-token && npm run build:prepare-pr-base && npm run build:ado-proxy && npm run build:azure-wif-refresh && npm run build:copilot-invoker && vitest run -c vitest.config.smoke.ts", "lint": "echo TODO", "typecheck": "tsc --noEmit" }, diff --git a/scripts/ado-script/src/copilot-invoker/index.test.ts b/scripts/ado-script/src/copilot-invoker/index.test.ts new file mode 100644 index 000000000..d04688f19 --- /dev/null +++ b/scripts/ado-script/src/copilot-invoker/index.test.ts @@ -0,0 +1,359 @@ +import { EventEmitter } from "node:events"; +import { mkdtempSync, readFileSync, writeFileSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { describe, expect, it, vi } from "vitest"; + +import { + buildChildEnvironment, + buildCopilotArgs, + main, + parseInvocationDocument, + parseInvocationResult, + resolveRequestedModel, + runInvocation, + type InvocationDocument, + type InvocationResult, +} from "./index.js"; + +function document(overrides: Partial = {}): InvocationDocument { + return { + schema_version: 1, + role: "agent", + command: "/tmp/awf-tools/copilot", + prompt_path: "/tmp/awf-tools/agent-prompt.md", + mcp_config_path: "/tmp/awf-tools/mcp-config.json", + args: ["--disable-builtin-mcps", "--allow-tool", "shell(cat *)"], + explicit_model: null, + result_path: "/tmp/awf-tools/copilot-invocation-result.json", + ...overrides, + }; +} + +describe("copilot invoker document", () => { + it("parses a strict versioned document", () => { + expect(parseInvocationDocument(JSON.stringify(document()))).toEqual(document()); + }); + + describe("copilot invoker result", () => { + it("parses a strict result document", () => { + expect( + parseInvocationResult( + JSON.stringify({ + schema_version: 1, + role: "detection", + requested_model: "detector", + }), + ), + ).toEqual({ + schema_version: 1, + role: "detection", + requested_model: "detector", + }); + }); + + it.each([ + [{ schema_version: 2, role: "agent", requested_model: null }, "unsupported"], + [ + { schema_version: 1, role: "agent", requested_model: null, extra: true }, + "unknown field", + ], + [{ schema_version: 1, role: "other", requested_model: null }, "must be"], + [{ schema_version: 1, role: "agent", requested_model: "bad model" }, "invalid"], + ])("rejects malformed results %#", (value, message) => { + expect(() => parseInvocationResult(JSON.stringify(value))).toThrow(message); + }); + }); + + it.each([ + [{ ...document(), schema_version: 2 }, "unsupported invocation schema"], + [{ ...document(), unexpected: true }, "unknown field 'unexpected'"], + [{ ...document(), role: "other" }, "must be 'agent' or 'detection'"], + [{ ...document(), command: "copilot;sh" }, "field 'command' is invalid"], + [{ ...document(), command: ".." }, "field 'command' is invalid"], + [{ ...document(), command: "bin/copilot" }, "field 'command' is invalid"], + [{ ...document(), command: "/tmp/../copilot" }, "field 'command' is invalid"], + [{ ...document(), command: "/tmp//copilot" }, "field 'command' is invalid"], + [{ ...document(), command: "/tmp/copilot/" }, "field 'command' is invalid"], + [{ ...document(), prompt_path: "relative.md" }, "field 'prompt_path' is invalid"], + [{ ...document(), prompt_path: "/tmp/../prompt.md" }, "field 'prompt_path' is invalid"], + [{ ...document(), result_path: "/" }, "field 'result_path' is invalid"], + [{ ...document(), args: [1] }, "field 'args' must be an array of strings"], + [{ ...document(), explicit_model: "bad model" }, "field 'explicit_model' is invalid"], + ])("rejects malformed input %#", (value, message) => { + expect(() => parseInvocationDocument(JSON.stringify(value))).toThrow(message); + }); +}); + +describe("model resolution", () => { + it("keeps an explicit model authoritative", () => { + expect( + resolveRequestedModel(document({ explicit_model: "frontmatter-model" }), { + ADO_AW_MODEL_AGENT_COPILOT: "role-model", + ADO_AW_DEFAULT_MODEL_COPILOT: "default-model", + }), + ).toBe("frontmatter-model"); + }); + + it("uses role-specific then shared values", () => { + expect( + resolveRequestedModel(document(), { + ADO_AW_MODEL_AGENT_COPILOT: "role-model", + ADO_AW_DEFAULT_MODEL_COPILOT: "default-model", + }), + ).toBe("role-model"); + expect( + resolveRequestedModel(document({ role: "detection" }), { + ADO_AW_MODEL_DETECTION_COPILOT: "", + ADO_AW_DEFAULT_MODEL_COPILOT: "default-model", + }), + ).toBe("default-model"); + }); + + it("treats unresolved ADO macros as absent", () => { + expect( + resolveRequestedModel(document(), { + ADO_AW_MODEL_AGENT_COPILOT: "$(ADO_AW_MODEL_AGENT_COPILOT)", + ADO_AW_DEFAULT_MODEL_COPILOT: "$(ADO_AW_DEFAULT_MODEL_COPILOT)", + }), + ).toBeNull(); + }); + + it("rejects invalid runtime values without printing them", () => { + let message = ""; + try { + resolveRequestedModel(document(), { + ADO_AW_MODEL_AGENT_COPILOT: "secret value", + }); + } catch (error) { + message = error instanceof Error ? error.message : String(error); + } + expect(message).toContain("contains invalid characters"); + expect(message).not.toContain("secret value"); + }); +}); + +describe("argv and child environment", () => { + it("passes prompt, MCP config, and authored values as distinct argv elements", () => { + expect(buildCopilotArgs(document(), "line one\nline two")).toEqual([ + "--prompt=line one\nline two", + "--additional-mcp-config", + "@/tmp/awf-tools/mcp-config.json", + "--disable-builtin-mcps", + "--allow-tool", + "shell(cat *)", + ]); + }); + + it("rejects NUL bytes in prompt content", () => { + expect(() => buildCopilotArgs(document(), "before\0after")).toThrow( + "prompt contains an invalid NUL byte", + ); + }); + + it("sets or removes only the child COPILOT_MODEL", () => { + const original = { KEEP: "yes", COPILOT_MODEL: "old" }; + expect(buildChildEnvironment(original, "selected")).toEqual({ + KEEP: "yes", + COPILOT_MODEL: "selected", + }); + expect(buildChildEnvironment(original, null)).toEqual({ KEEP: "yes" }); + expect(original.COPILOT_MODEL).toBe("old"); + }); +}); + +describe("read-result command", () => { + it("writes the requested model after validating the result", async () => { + const directory = mkdtempSync(join(tmpdir(), "copilot-invoker-result-")); + const resultPath = join(directory, "result.json"); + writeFileSync( + resultPath, + JSON.stringify({ + schema_version: 1, + role: "agent", + requested_model: "gpt-test", + }), + ); + const write = vi.spyOn(process.stdout, "write").mockImplementation( + ((_chunk: unknown, callback?: (error?: Error | null) => void) => { + callback?.(); + return true; + }) as typeof process.stdout.write, + ); + + await expect(main(["read-result", resultPath, "agent"])).resolves.toBe(0); + expect(write).toHaveBeenCalledWith("gpt-test", expect.any(Function)); + write.mockRestore(); + }); + + it("rejects a result for the wrong role", async () => { + const directory = mkdtempSync(join(tmpdir(), "copilot-invoker-result-")); + const resultPath = join(directory, "result.json"); + writeFileSync( + resultPath, + JSON.stringify({ + schema_version: 1, + role: "agent", + requested_model: "gpt-test", + }), + ); + const error = vi.spyOn(console, "error").mockImplementation(() => undefined); + + await expect(main(["read-result", resultPath, "detection"])).resolves.toBe(1); + expect(error).toHaveBeenCalledWith( + "copilot-invoker: invocation result role 'agent' does not match expected role 'detection'", + ); + error.mockRestore(); + }); + + it("reports a missing or malformed result", async () => { + const directory = mkdtempSync(join(tmpdir(), "copilot-invoker-result-")); + const resultPath = join(directory, "missing.json"); + const error = vi.spyOn(console, "error").mockImplementation(() => undefined); + + await expect(main(["read-result", resultPath, "agent"])).resolves.toBe(1); + expect(error).toHaveBeenCalledWith(expect.stringContaining("copilot-invoker:")); + error.mockRestore(); + }); + + it("reports stdout write failures", async () => { + const directory = mkdtempSync(join(tmpdir(), "copilot-invoker-result-")); + const resultPath = join(directory, "result.json"); + writeFileSync( + resultPath, + JSON.stringify({ + schema_version: 1, + role: "agent", + requested_model: "gpt-test", + }), + ); + const write = vi.spyOn(process.stdout, "write").mockImplementation( + ((_chunk: unknown, callback?: (error?: Error | null) => void) => { + callback?.(new Error("EPIPE")); + return false; + }) as typeof process.stdout.write, + ); + const error = vi.spyOn(console, "error").mockImplementation(() => undefined); + + await expect(main(["read-result", resultPath, "agent"])).resolves.toBe(1); + expect(error).toHaveBeenCalledWith("copilot-invoker: EPIPE"); + write.mockRestore(); + error.mockRestore(); + }); +}); + +describe("process lifecycle", () => { + it("publishes the result before spawning and preserves the child exit code", async () => { + const events: string[] = []; + const child = new EventEmitter() as EventEmitter & { + killed: boolean; + kill: ReturnType; + }; + child.killed = false; + child.kill = vi.fn(); + const results: InvocationResult[] = []; + let spawnedArgs: readonly string[] | undefined; + const spawn = vi.fn((_command: string, args: readonly string[]) => { + spawnedArgs = args; + events.push("spawn"); + queueMicrotask(() => child.emit("close", 23, null)); + return child; + }); + + const exit = await runInvocation( + document(), + { ADO_AW_MODEL_AGENT_COPILOT: "runtime-model" }, + { + readFile: () => "prompt", + writeResult: (_path, result) => { + events.push("result"); + results.push(result); + }, + spawn: spawn as never, + }, + ); + + expect(events).toEqual(["result", "spawn"]); + expect(results).toEqual([ + { schema_version: 1, role: "agent", requested_model: "runtime-model" }, + ]); + expect(exit).toBe(23); + expect(spawnedArgs).toEqual([ + "--prompt=prompt", + "--additional-mcp-config", + "@/tmp/awf-tools/mcp-config.json", + "--disable-builtin-mcps", + "--allow-tool", + "shell(cat *)", + ]); + }); + + it("writes results atomically with private permissions", async () => { + const directory = mkdtempSync(join(tmpdir(), "copilot-invoker-")); + const resultPath = join(directory, "result.json").replaceAll("\\", "/"); + const child = new EventEmitter() as EventEmitter & { + killed: boolean; + kill: ReturnType; + }; + child.killed = false; + child.kill = vi.fn(); + const invocation = document({ result_path: resultPath }); + const promise = runInvocation(invocation, {}, { + readFile: () => "prompt", + writeResult: (await import("./index.js")).writeResultAtomic, + spawn: (() => { + queueMicrotask(() => child.emit("close", 0, null)); + return child; + }) as never, + }); + await expect(promise).resolves.toBe(0); + expect(JSON.parse(readFileSync(resultPath, "utf8"))).toEqual({ + schema_version: 1, + role: "agent", + requested_model: null, + }); + }); + + it("returns a deterministic failure when Copilot cannot start", async () => { + const child = new EventEmitter() as EventEmitter & { + killed: boolean; + kill: ReturnType; + }; + child.killed = false; + child.kill = vi.fn(); + const error = vi.spyOn(console, "error").mockImplementation(() => undefined); + const promise = runInvocation(document(), {}, { + readFile: () => "prompt", + writeResult: () => undefined, + spawn: (() => { + queueMicrotask(() => child.emit("error", new Error("ENOENT"))); + return child; + }) as never, + }); + await expect(promise).resolves.toBe(1); + expect(error).toHaveBeenCalledWith( + "copilot-invoker: failed to start Copilot: ENOENT", + ); + error.mockRestore(); + }); + + it("forwards termination signals and maps a signaled exit", async () => { + const child = new EventEmitter() as EventEmitter & { + killed: boolean; + kill: ReturnType; + }; + child.killed = false; + child.kill = vi.fn((signal: NodeJS.Signals) => { + queueMicrotask(() => child.emit("close", null, signal)); + return true; + }); + const promise = runInvocation(document(), {}, { + readFile: () => "prompt", + writeResult: () => undefined, + spawn: (() => child) as never, + }); + process.emit("SIGTERM", "SIGTERM"); + await expect(promise).resolves.toBe(143); + expect(child.kill).toHaveBeenCalledWith("SIGTERM"); + }); +}); diff --git a/scripts/ado-script/src/copilot-invoker/index.ts b/scripts/ado-script/src/copilot-invoker/index.ts new file mode 100644 index 000000000..e6a6d4450 --- /dev/null +++ b/scripts/ado-script/src/copilot-invoker/index.ts @@ -0,0 +1,386 @@ +import { spawn, type ChildProcess } from "node:child_process"; +import { readFileSync, renameSync, writeFileSync } from "node:fs"; +import { constants } from "node:os"; +import { resolve } from "node:path"; +import { fileURLToPath } from "node:url"; + +const SCHEMA_VERSION = 1; +const MODEL_PATTERN = /^[A-Za-z0-9._:-]+$/; +const COMMAND_PATTERN = /^[A-Za-z0-9._/-]+$/; +const DOCUMENT_KEYS = new Set([ + "schema_version", + "role", + "command", + "prompt_path", + "mcp_config_path", + "args", + "explicit_model", + "result_path", +]); +const RESULT_KEYS = new Set(["schema_version", "role", "requested_model"]); + +export type InvocationRole = "agent" | "detection"; + +export interface InvocationDocument { + schema_version: 1; + role: InvocationRole; + command: string; + prompt_path: string; + mcp_config_path: string | null; + args: string[]; + explicit_model: string | null; + result_path: string; +} + +export interface InvocationResult { + schema_version: 1; + role: InvocationRole; + requested_model: string | null; +} + +interface SpawnLike { + ( + command: string, + args: readonly string[], + options: { + env: NodeJS.ProcessEnv; + stdio: "inherit"; + }, + ): ChildProcess; +} + +export interface RunDependencies { + spawn: SpawnLike; + readFile(path: string): string; + writeResult(path: string, result: InvocationResult): void; +} + +const DEFAULT_DEPENDENCIES: RunDependencies = { + spawn, + readFile: (path) => readFileSync(path, "utf8"), + writeResult: writeResultAtomic, +}; + +function isRecord(value: unknown): value is Record { + return typeof value === "object" && value !== null && !Array.isArray(value); +} + +function requiredString( + value: Record, + key: string, + validator?: (input: string) => boolean, +): string { + const candidate = value[key]; + if (typeof candidate !== "string" || candidate.length === 0) { + throw new Error(`invocation document field '${key}' must be a non-empty string`); + } + if (candidate.includes("\0") || (validator && !validator(candidate))) { + throw new Error(`invocation document field '${key}' is invalid`); + } + return candidate; +} + +function nullableString( + value: Record, + key: string, + validator?: (input: string) => boolean, +): string | null { + const candidate = value[key]; + if (candidate === null) return null; + if (typeof candidate !== "string" || candidate.includes("\0")) { + throw new Error(`invocation document field '${key}' must be a string or null`); + } + if (validator && !validator(candidate)) { + throw new Error(`invocation document field '${key}' is invalid`); + } + return candidate; +} + +function hasSafePathSegments(value: string, allowBareCommand: boolean): boolean { + if ( + value.includes("\n") || + value.includes("\r") || + value.includes(":") || + value.endsWith("/") + ) { + return false; + } + if (allowBareCommand && !value.includes("/")) { + return value !== "." && value !== ".."; + } + if (!value.startsWith("/")) return false; + const segments = value.slice(1).split("/"); + return ( + segments.length > 0 && + segments.every((segment) => segment.length > 0 && segment !== "." && segment !== "..") + ); +} + +function isAbsoluteContainerPath(value: string): boolean { + return hasSafePathSegments(value, false); +} + +function isSafeCommand(value: string): boolean { + return COMMAND_PATTERN.test(value) && hasSafePathSegments(value, true); +} + +export function parseInvocationDocument(raw: string): InvocationDocument { + let parsed: unknown; + try { + parsed = JSON.parse(raw); + } catch { + throw new Error("invocation document is not valid JSON"); + } + + if (!isRecord(parsed)) { + throw new Error("invocation document must be a JSON object"); + } + const unknown = Object.keys(parsed).filter((key) => !DOCUMENT_KEYS.has(key)); + if (unknown.length > 0) { + throw new Error(`invocation document contains unknown field '${unknown.sort()[0]}'`); + } + if (parsed.schema_version !== SCHEMA_VERSION) { + throw new Error( + `unsupported invocation schema version '${String(parsed.schema_version)}'`, + ); + } + if (parsed.role !== "agent" && parsed.role !== "detection") { + throw new Error("invocation document field 'role' must be 'agent' or 'detection'"); + } + if (!Array.isArray(parsed.args) || !parsed.args.every((arg) => typeof arg === "string")) { + throw new Error("invocation document field 'args' must be an array of strings"); + } + if (parsed.args.some((arg) => arg.includes("\0"))) { + throw new Error("invocation document field 'args' contains an invalid NUL byte"); + } + + return { + schema_version: SCHEMA_VERSION, + role: parsed.role, + command: requiredString(parsed, "command", isSafeCommand), + prompt_path: requiredString(parsed, "prompt_path", isAbsoluteContainerPath), + mcp_config_path: nullableString(parsed, "mcp_config_path", isAbsoluteContainerPath), + args: [...parsed.args], + explicit_model: nullableString(parsed, "explicit_model", (value) => + MODEL_PATTERN.test(value), + ), + result_path: requiredString(parsed, "result_path", isAbsoluteContainerPath), + }; +} + +export function parseInvocationResult(raw: string): InvocationResult { + let parsed: unknown; + try { + parsed = JSON.parse(raw); + } catch { + throw new Error("invocation result is not valid JSON"); + } + if (!isRecord(parsed)) { + throw new Error("invocation result must be a JSON object"); + } + const unknown = Object.keys(parsed).filter((key) => !RESULT_KEYS.has(key)); + if (unknown.length > 0) { + throw new Error(`invocation result contains unknown field '${unknown.sort()[0]}'`); + } + if (parsed.schema_version !== SCHEMA_VERSION) { + throw new Error( + `unsupported invocation result schema version '${String(parsed.schema_version)}'`, + ); + } + if (parsed.role !== "agent" && parsed.role !== "detection") { + throw new Error("invocation result field 'role' must be 'agent' or 'detection'"); + } + const requestedModel = nullableString(parsed, "requested_model", (value) => + MODEL_PATTERN.test(value), + ); + return { + schema_version: SCHEMA_VERSION, + role: parsed.role, + requested_model: requestedModel, + }; +} + +function runtimeModelVariable(role: InvocationRole): string { + return role === "agent" + ? "ADO_AW_MODEL_AGENT_COPILOT" + : "ADO_AW_MODEL_DETECTION_COPILOT"; +} + +function isUnresolvedAdoMacro(value: string, variable: string): boolean { + return value === `$(${variable})`; +} + +export function resolveRequestedModel( + document: InvocationDocument, + env: NodeJS.ProcessEnv, +): string | null { + if (document.explicit_model !== null) { + return document.explicit_model; + } + const specific = runtimeModelVariable(document.role); + const candidates: Array<[string, string | undefined]> = [ + [specific, env[specific]], + ["ADO_AW_DEFAULT_MODEL_COPILOT", env.ADO_AW_DEFAULT_MODEL_COPILOT], + ]; + for (const [variable, candidate] of candidates) { + if ( + candidate === undefined || + candidate.length === 0 || + isUnresolvedAdoMacro(candidate, variable) + ) { + continue; + } + if (!MODEL_PATTERN.test(candidate)) { + throw new Error( + `runtime Copilot model from ${specific}/ADO_AW_DEFAULT_MODEL_COPILOT contains invalid characters`, + ); + } + return candidate; + } + return null; +} + +export function buildCopilotArgs( + document: InvocationDocument, + prompt: string, +): string[] { + if (prompt.includes("\0")) { + throw new Error("prompt contains an invalid NUL byte"); + } + const args = [`--prompt=${prompt}`]; + if (document.mcp_config_path !== null) { + args.push("--additional-mcp-config", `@${document.mcp_config_path}`); + } + args.push(...document.args); + return args; +} + +export function buildChildEnvironment( + env: NodeJS.ProcessEnv, + requestedModel: string | null, +): NodeJS.ProcessEnv { + // AWF filters host-only credentials before launching the invoker. Preserve + // the remaining environment because Copilot providers and MCPs consume it. + const childEnv = { ...env }; + if (requestedModel === null) { + delete childEnv.COPILOT_MODEL; + } else { + childEnv.COPILOT_MODEL = requestedModel; + } + return childEnv; +} + +export function writeResultAtomic(path: string, result: InvocationResult): void { + const temporary = `${path}.tmp-${process.pid}`; + writeFileSync(temporary, `${JSON.stringify(result)}\n`, { + encoding: "utf8", + mode: 0o600, + }); + renameSync(temporary, path); +} + +function signalExitCode(signal: NodeJS.Signals): number { + return 128 + (constants.signals[signal] ?? 0); +} + +export async function runInvocation( + document: InvocationDocument, + env: NodeJS.ProcessEnv = process.env, + dependencies: RunDependencies = DEFAULT_DEPENDENCIES, +): Promise { + const requestedModel = resolveRequestedModel(document, env); + dependencies.writeResult(document.result_path, { + schema_version: SCHEMA_VERSION, + role: document.role, + requested_model: requestedModel, + }); + + const prompt = dependencies.readFile(document.prompt_path); + const args = buildCopilotArgs(document, prompt); + const child = dependencies.spawn(document.command, args, { + env: buildChildEnvironment(env, requestedModel), + stdio: "inherit", + }); + + return await new Promise((resolveExit) => { + let settled = false; + const settle = (code: number) => { + if (settled) return; + settled = true; + for (const signal of forwardedSignals) { + process.off(signal, handlers[signal]); + } + resolveExit(code); + }; + const forwardedSignals: NodeJS.Signals[] = ["SIGINT", "SIGTERM", "SIGHUP"]; + const handlers = Object.fromEntries( + forwardedSignals.map((signal) => [ + signal, + () => { + if (!child.killed) child.kill(signal); + }, + ]), + ) as Record void>; + for (const signal of forwardedSignals) { + process.on(signal, handlers[signal]); + } + child.once("error", (error) => { + console.error(`copilot-invoker: failed to start Copilot: ${error.message}`); + settle(1); + }); + child.once("close", (code, signal) => { + settle(code ?? (signal ? signalExitCode(signal) : 1)); + }); + }); +} + +export async function main(argv: string[]): Promise { + if (argv[0] === "read-result" && argv.length === 3) { + try { + const result = parseInvocationResult(readFileSync(argv[1]!, "utf8")); + if (result.role !== argv[2]) { + throw new Error( + `invocation result role '${result.role}' does not match expected role '${argv[2]}'`, + ); + } + await new Promise((resolveWrite, rejectWrite) => { + process.stdout.write(result.requested_model ?? "", (error) => { + if (error) rejectWrite(error); + else resolveWrite(); + }); + }); + return 0; + } catch (error) { + const message = error instanceof Error ? error.message : "unknown error"; + console.error(`copilot-invoker: ${message}`); + return 1; + } + } + if (argv[0] !== "run" || argv.length !== 2) { + console.error( + "usage: copilot-invoker run | read-result ", + ); + return 2; + } + try { + const document = parseInvocationDocument(readFileSync(argv[1]!, "utf8")); + return await runInvocation(document); + } catch (error) { + const message = error instanceof Error ? error.message : "unknown error"; + console.error(`copilot-invoker: ${message}`); + return 1; + } +} + +const invokedPath = process.argv[1] ? resolve(process.argv[1]) : ""; +if (invokedPath === fileURLToPath(import.meta.url)) { + void main(process.argv.slice(2)) + .then((code) => { + process.exitCode = code; + }) + .catch((error: unknown) => { + const message = error instanceof Error ? error.message : "unknown error"; + console.error(`copilot-invoker: unexpected failure: ${message}`); + process.exitCode = 1; + }); +} diff --git a/scripts/ado-script/test/azure-wif-isolation.test.ts b/scripts/ado-script/test/azure-wif-isolation.test.ts index 607a23443..d8ae17c2b 100644 --- a/scripts/ado-script/test/azure-wif-isolation.test.ts +++ b/scripts/ado-script/test/azure-wif-isolation.test.ts @@ -209,10 +209,18 @@ describe.skipIf(!awfEnabled)("Azure WIF real AWF boundary", () => { const workspace = join(directory, "workspace"); const temp = join(directory, "runner-temp"); const tools = join(directory, "tools"); + const awfTools = join(tools, "awf-tools"); + const adoScripts = join(tools, "ado-aw-scripts"); const home = join(directory, "home"); const auth = join(temp, "ado-aw-azure-auth", "fixture"); mkdirSync(workspace, { recursive: true }); mkdirSync(home); + mkdirSync(awfTools, { recursive: true }); + mkdirSync(join(adoScripts, "ado-script"), { recursive: true }); + copyFileSync( + resolve(testDir, "../copilot-invoker.js"), + join(adoScripts, "ado-script/copilot-invoker.js"), + ); mkdirSync(join(auth, "token.d"), { recursive: true }); chmodSync(join(temp, "ado-aw-azure-auth"), 0o700); chmodSync(auth, 0o700); @@ -224,15 +232,22 @@ describe.skipIf(!awfEnabled)("Azure WIF real AWF boundary", () => { const runStep = pipeline.jobs.find((job) => job.job === "Agent")?.steps .find((step) => step.bash?.includes("AWF_ARGS+=(--skip-pull --env-all)")); if (!runStep?.bash) throw new Error("compiled AWF invocation is missing"); + expect(runStep.bash).toContain( + "copilot-invoker.js run /tmp/awf-tools/copilot-invocation.json", + ); const capture = join(directory, "awf-args"); + const invocationResult = join(awfTools, "copilot-invocation-result.json"); writeFileSync(join(tools, "awf/awf"), `#!/bin/sh if [ "$1" = logs ]; then exit 0; fi printf '%s\\0' "$@" > '${capture}' +printf '%s\\n' '{"schema_version":1,"role":"agent","requested_model":null}' > '${invocationResult}' `, { mode: 0o755 }); const script = runStep.bash .replaceAll("$(Agent.TempDirectory)", temp) .replaceAll("$(Pipeline.Workspace)", tools) - .replaceAll("$(Build.SourcesDirectory)", workspace); + .replaceAll("$(Build.SourcesDirectory)", workspace) + .replaceAll("/tmp/awf-tools", awfTools) + .replaceAll("/tmp/ado-aw-scripts", adoScripts); const env: NodeJS.ProcessEnv = { PATH: process.env.PATH, HOME: home, @@ -260,6 +275,9 @@ printf '%s\\0' "$@" > '${capture}' const commandIndex = captured.indexOf("--"); expect(commandIndex).toBeGreaterThan(0); + expect(captured[commandIndex + 1]).toContain( + `copilot-invoker.js run ${join(awfTools, "copilot-invocation.json")}`, + ); const args: string[] = []; for (let i = 0; i < commandIndex; i++) { // This probe exercises filesystem/env isolation, not MCP networking: diff --git a/scripts/ado-script/test/smoke.test.ts b/scripts/ado-script/test/smoke.test.ts index 630ae1906..f03b5c188 100644 --- a/scripts/ado-script/test/smoke.test.ts +++ b/scripts/ado-script/test/smoke.test.ts @@ -10,6 +10,7 @@ import { spawnSync } from "node:child_process"; import { randomUUID } from "node:crypto"; import { + chmodSync, copyFileSync, existsSync, mkdirSync, @@ -27,6 +28,7 @@ const gateBundlePath = resolve(__dirname, "../gate.js"); const importBundlePath = resolve(__dirname, "../import.js"); const execContextPrBundlePath = resolve(__dirname, "../exec-context-pr.js"); const preparePrBaseBundlePath = resolve(__dirname, "../prepare-pr-base.js"); +const copilotInvokerBundlePath = resolve(__dirname, "../copilot-invoker.js"); const gateFixturePath = resolve( __dirname, "fixtures/gate-spec-pr-title-match.json", @@ -131,6 +133,61 @@ describe("import.js smoke", () => { }, 20000); }); +describe.skipIf(process.platform === "win32")("copilot-invoker.js smoke", () => { + it("runs a fake Copilot child and publishes the requested model", () => { + withSmokeScratchDir("copilot-invoker", (dir) => { + const command = resolve(dir, "fake-copilot"); + const promptPath = resolve(dir, "prompt.md"); + const resultPath = resolve(dir, "result.json"); + const capturePath = resolve(dir, "capture.json"); + const documentPath = resolve(dir, "invocation.json"); + writeFileSync( + command, + `#!/bin/sh +node -e 'require("node:fs").writeFileSync(process.env.CAPTURE_PATH, JSON.stringify({ argv: process.argv.slice(1), model: process.env.COPILOT_MODEL }))' -- "$@" +exit 7 +`, + ); + chmodSync(command, 0o755); + writeFileSync(promptPath, "smoke prompt\n"); + writeFileSync( + documentPath, + JSON.stringify({ + schema_version: 1, + role: "agent", + command, + prompt_path: promptPath, + mcp_config_path: null, + args: ["--no-ask-user"], + explicit_model: "gpt-smoke", + result_path: resultPath, + }), + ); + + const run = spawnSync( + process.execPath, + [copilotInvokerBundlePath, "run", documentPath], + { + env: { ...process.env, CAPTURE_PATH: capturePath }, + encoding: "utf8", + }, + ); + expect(run.status).toBe(7); + expect(run.stdout).toBe(""); + expect(run.stderr).toBe(""); + expect(JSON.parse(readFileSync(capturePath, "utf8"))).toEqual({ + argv: ["--prompt=smoke prompt\n", "--no-ask-user"], + model: "gpt-smoke", + }); + expect(JSON.parse(readFileSync(resultPath, "utf8"))).toEqual({ + schema_version: 1, + role: "agent", + requested_model: "gpt-smoke", + }); + }); + }); +}); + function runGitInRepo(repoDir: string, args: string[]): void { const result = spawnSync("git", args, { cwd: repoDir, diff --git a/src/audit/analyzers/detection.rs b/src/audit/analyzers/detection.rs index 49432ee02..23592cb1e 100644 --- a/src/audit/analyzers/detection.rs +++ b/src/audit/analyzers/detection.rs @@ -1,10 +1,10 @@ -use anyhow::Result; +use anyhow::{Context, Result}; use log::{debug, warn}; use serde_json::Value; use std::io::ErrorKind; use std::path::{Path, PathBuf}; -use crate::audit::model::{DetectionAnalysis, DetectionThreats}; +use crate::audit::model::{AwInfo, DetectionAnalysis, DetectionThreats}; /// Read the detection verdict from `analyzed_outputs_/threat-analysis.json`. /// @@ -65,7 +65,36 @@ pub async fn analyze_detection(download_root: &Path) -> Result` artifact. +pub async fn load_aw_info(download_root: &Path) -> Result> { + let Some(directory) = find_analyzed_outputs_dir(download_root).await else { + return Ok(None); + }; + let path = [ + directory.join("aw_info.json"), + directory.join("staging").join("aw_info.json"), + ] + .into_iter() + .find(|path| path.is_file()); + let Some(path) = path else { + return Ok(None); + }; + let contents = tokio::fs::read_to_string(&path) + .await + .with_context(|| format!("Failed to read Detection aw_info file {}", path.display()))?; + serde_json::from_str(&contents) + .with_context(|| format!("Failed to parse Detection aw_info file {}", path.display())) + .map(Some) +} + async fn find_verdict_path(download_root: &Path) -> Option { + find_analyzed_outputs_dir(download_root) + .await + .map(|directory| directory.join("threat-analysis.json")) +} + +async fn find_analyzed_outputs_dir(download_root: &Path) -> Option { let mut entries = match tokio::fs::read_dir(download_root).await { Ok(entries) => entries, Err(err) if err.kind() == ErrorKind::NotFound => return None, @@ -122,7 +151,7 @@ async fn find_verdict_path(download_root: &Path) -> Option { } } - latest_dir.map(|(_, dir)| dir.join("threat-analysis.json")) + latest_dir.map(|(_, dir)| dir) } fn extract_bool(v: &Value, key: &str) -> bool { @@ -163,7 +192,7 @@ fn extract_reasons(v: &Value, verdict_path: &Path) -> Vec { #[cfg(test)] mod tests { - use super::analyze_detection; + use super::{analyze_detection, load_aw_info}; use crate::audit::model::DetectionThreats; use tempfile::TempDir; @@ -180,6 +209,14 @@ mod tests { .unwrap(); } + async fn write_aw_info(temp_dir: &TempDir, dir_name: &str, contents: &str) { + let dir = temp_dir.path().join(dir_name); + tokio::fs::create_dir_all(&dir).await.unwrap(); + tokio::fs::write(dir.join("aw_info.json"), contents) + .await + .unwrap(); + } + async fn write_verdict(temp_dir: &TempDir, dir_name: &str, contents: &str) { let dir = temp_dir.path().join(dir_name); tokio::fs::create_dir_all(&dir).await.unwrap(); @@ -346,4 +383,43 @@ mod tests { Some(expected_verdict_path("analyzed_outputs_10")) ); } + + #[tokio::test] + async fn loads_detection_enriched_aw_info_from_latest_artifact() { + let temp_dir = TempDir::new().unwrap(); + write_aw_info( + &temp_dir, + "analyzed_outputs_9", + r#"{"detection_model":"older"}"#, + ) + .await; + write_aw_info( + &temp_dir, + "analyzed_outputs_10", + r#"{"detection_model":"detector-model"}"#, + ) + .await; + + let aw_info = load_aw_info(temp_dir.path()).await.unwrap().unwrap(); + + assert_eq!(aw_info.detection_model.as_deref(), Some("detector-model")); + } + + #[tokio::test] + async fn detection_aw_info_is_optional_for_older_artifacts() { + let temp_dir = TempDir::new().unwrap(); + create_analyzed_outputs_dir(&temp_dir, "analyzed_outputs_42").await; + + assert!(load_aw_info(temp_dir.path()).await.unwrap().is_none()); + } + + #[tokio::test] + async fn malformed_detection_aw_info_is_an_error() { + let temp_dir = TempDir::new().unwrap(); + write_aw_info(&temp_dir, "analyzed_outputs_42", "{not valid json").await; + + let error = load_aw_info(temp_dir.path()).await.unwrap_err().to_string(); + + assert!(error.contains("Failed to parse Detection aw_info"), "{error}"); + } } diff --git a/src/audit/analyzers/otel.rs b/src/audit/analyzers/otel.rs index ae24740bc..a06de22f6 100644 --- a/src/audit/analyzers/otel.rs +++ b/src/audit/analyzers/otel.rs @@ -12,6 +12,7 @@ pub struct OtelAnalysis { pub engine_config: Option, pub performance: Option, pub aw_info: Option, + pub observed_model: Option, pub warnings: Vec, } @@ -36,6 +37,7 @@ pub async fn analyze_otel(agent_outputs_dir: &std::path::Path) -> anyhow::Result let stats = AgentStats::from_otel_file(&otel_path, "audit") .await .with_context(|| format!("Failed to analyze OTel file: {}", otel_path.display()))?; + analysis.observed_model = stats.model.clone(); let total_tokens = stats.input_tokens + stats.output_tokens; analysis.metrics = MetricsData { @@ -81,7 +83,10 @@ pub async fn analyze_otel(agent_outputs_dir: &std::path::Path) -> anyhow::Result Ok(aw_info) => { analysis.engine_config = Some(AuditEngineConfig { engine: aw_info.engine.clone().unwrap_or_default(), - model: aw_info.model.clone(), + model: analysis + .observed_model + .clone() + .or_else(|| aw_info.model.clone()), version: aw_info.compiler_version.clone(), timeout_minutes: None, }); @@ -176,6 +181,10 @@ mod tests { assert_eq!(analysis.metrics.token_usage, 33185); assert_eq!(analysis.metrics.turns, 2); + assert_eq!( + analysis.observed_model.as_deref(), + Some("claude-sonnet-4.5") + ); assert!(analysis.engine_config.is_none()); assert!(analysis.aw_info.is_none()); } @@ -198,6 +207,10 @@ mod tests { .and_then(|config| config.model.as_deref()), Some("claude-sonnet-4.5") ); + assert_eq!( + analysis.observed_model.as_deref(), + Some("claude-sonnet-4.5") + ); assert_eq!( analysis .aw_info @@ -214,6 +227,35 @@ mod tests { ); } + #[tokio::test] + async fn observed_otel_model_overrides_requested_model_in_engine_config() { + let temp_dir = TempDir::new().unwrap(); + let staging_dir = temp_dir.path().join("staging"); + write_file(&staging_dir.join("otel.jsonl"), COPILOT_OTEL_FIXTURE).await; + write_file( + &staging_dir.join("aw_info.json"), + &AW_INFO_JSON.replace("claude-sonnet-4.5", "requested-session-model"), + ) + .await; + + let analysis = analyze_otel(temp_dir.path()).await.unwrap(); + + assert_eq!( + analysis + .aw_info + .as_ref() + .and_then(|info| info.model.as_deref()), + Some("requested-session-model") + ); + assert_eq!( + analysis + .engine_config + .as_ref() + .and_then(|config| config.model.as_deref()), + Some("claude-sonnet-4.5") + ); + } + #[tokio::test] async fn malformed_aw_info_preserves_otel_metrics_and_emits_shared_warning() { let temp_dir = TempDir::new().unwrap(); diff --git a/src/audit/cli.rs b/src/audit/cli.rs index 978273654..99eb73b60 100644 --- a/src/audit/cli.rs +++ b/src/audit/cli.rs @@ -15,7 +15,7 @@ use crate::audit::analyzers::{ use crate::audit::cache::{RunSummary, load_run_summary, save_run_summary}; use crate::audit::find_artifact_dir; use crate::audit::findings; -use crate::audit::model::{AuditData, ErrorInfo, FileInfo, OverviewData}; +use crate::audit::model::{AuditData, AwInfo, ErrorInfo, FileInfo, OverviewData}; use crate::audit::pipeline_graph; use crate::audit::render; use crate::audit::url::{ParsedBuildRef, parse_build_ref}; @@ -458,6 +458,19 @@ async fn run_analyzers( detection::analyze_detection(run_dir).await, |a, result| a.detection_analysis = result, ); + match detection::load_aw_info(run_dir).await { + Ok(Some(detection_aw_info)) => { + merge_detection_aw_info(&mut audit.overview.aw_info, detection_aw_info) + } + Ok(None) => {} + Err(error) => { + log::warn!("{error:#}"); + crate::audit::push_warning_once( + audit, + crate::audit::malformed_detection_aw_info_warning(), + ); + } + } } run_analyzer( audit, @@ -468,6 +481,16 @@ async fn run_analyzers( ); } +fn merge_detection_aw_info(base: &mut Option, detection: AwInfo) { + let Some(base) = base.as_mut() else { + *base = Some(detection); + return; + }; + if detection.detection_model.is_some() { + base.detection_model = detection.detection_model; + } +} + fn artifact_family_selected(filters: Option<&[String]>, family: &str) -> bool { filters.is_none_or(|filters| filters.iter().any(|filter| filter == family)) } @@ -1039,7 +1062,67 @@ fn render_audit(audit: &AuditData, json: bool) -> Result<()> { #[cfg(test)] mod tests { use super::*; - use crate::audit::model::{CustomSafeOutputJobAudit, Finding, JobData, Recommendation}; + use crate::audit::model::{ + AwInfo, CustomSafeOutputJobAudit, Finding, JobData, Recommendation, + }; + + #[test] + fn detection_aw_info_overlays_only_detection_owned_fields() { + let mut base = Some(AwInfo { + engine: Some(String::from("copilot")), + model: Some(String::from("agent-model")), + detection_model: Some(String::from("old-detector")), + source: Some(String::from("agents/test.md")), + ..Default::default() + }); + let detection = AwInfo { + engine: Some(String::from("must-not-replace")), + model: Some(String::from("must-not-replace")), + detection_model: Some(String::from("detector-model")), + source: Some(String::from("must-not-replace")), + ..Default::default() + }; + + merge_detection_aw_info(&mut base, detection); + + let merged = base.unwrap(); + assert_eq!(merged.engine.as_deref(), Some("copilot")); + assert_eq!(merged.model.as_deref(), Some("agent-model")); + assert_eq!(merged.source.as_deref(), Some("agents/test.md")); + assert_eq!( + merged.detection_model.as_deref(), + Some("detector-model") + ); + } + + #[test] + fn detection_only_aw_info_populates_overview_metadata() { + let mut base = None; + let detection = AwInfo { + engine: Some(String::from("copilot")), + detection_model: Some(String::from("detector-model")), + ..Default::default() + }; + + merge_detection_aw_info(&mut base, detection.clone()); + + assert_eq!(base, Some(detection)); + } + + #[test] + fn older_detection_aw_info_does_not_erase_static_model() { + let mut base = Some(AwInfo { + detection_model: Some(String::from("static-detector")), + ..Default::default() + }); + + merge_detection_aw_info(&mut base, AwInfo::default()); + + assert_eq!( + base.unwrap().detection_model.as_deref(), + Some("static-detector") + ); + } #[tokio::test] async fn agent_output_analyzers_load_proxy_logs_from_canonical_path() { diff --git a/src/audit/mod.rs b/src/audit/mod.rs index bae5429cf..646036747 100644 --- a/src/audit/mod.rs +++ b/src/audit/mod.rs @@ -27,12 +27,38 @@ pub(crate) fn malformed_aw_info_warning() -> model::ErrorInfo { } } +pub(crate) fn malformed_detection_aw_info_warning() -> model::ErrorInfo { + model::ErrorInfo { + source: String::from("audit::detection_aw_info"), + message: String::from( + "Detection's analyzed aw_info.json could not be read or parsed; Detection runtime model enrichment is unavailable", + ), + timestamp: None, + } +} + pub(crate) fn push_warning_once(audit: &mut model::AuditData, warning: model::ErrorInfo) { if !audit.warnings.contains(&warning) { audit.warnings.push(warning); } } +#[cfg(test)] +mod tests { + #[test] + fn malformed_detection_metadata_warning_is_scoped_to_detection_enrichment() { + let warning = super::malformed_detection_aw_info_warning(); + + assert_eq!(warning.source, "audit::detection_aw_info"); + assert!( + warning + .message + .contains("Detection runtime model enrichment") + ); + assert!(!warning.message.contains("pipeline graph")); + } +} + /// Compare two `_` directory names by their trailing /// integer suffix, falling back to a full lexicographic comparison /// when the suffix isn't a u64. diff --git a/src/audit/model.rs b/src/audit/model.rs index 2b9e6365a..0ab1466f3 100644 --- a/src/audit/model.rs +++ b/src/audit/model.rs @@ -153,21 +153,26 @@ pub struct OverviewData { /// Local path where build logs or downloaded artifacts were stored. #[serde(skip_serializing_if = "Option::is_none")] pub logs_path: Option, - /// Runtime-emitted AW metadata from `staging/aw_info.json`. + /// Runtime-emitted AW metadata merged from Agent and Detection artifacts. #[serde(skip_serializing_if = "Option::is_none")] pub aw_info: Option, } /// Runtime-emitted agentic workflow metadata. /// -/// This is read from `staging/aw_info.json`, which mirrors the compiled marker metadata plus runtime context. +/// Agent metadata is read from `staging/aw_info.json`; Detection may enrich the +/// copied `aw_info.json` in its analyzed-output artifact with Detection-owned +/// runtime fields. #[derive(Debug, Clone, Default, PartialEq, Serialize, Deserialize)] #[serde(default)] pub struct AwInfo { /// Configured engine name for the run. #[serde(skip_serializing_if = "Option::is_none")] pub engine: Option, - /// Model identifier used by the agent runtime. + /// Model identifier requested for the Agent's Copilot session. + /// + /// A selected custom agent may pin a different model. When Copilot OTel is + /// available, `AuditData.engine_config.model` reports the observed model. #[serde(skip_serializing_if = "Option::is_none")] pub model: Option, /// Whether AI threat detection was enabled for this workflow. @@ -176,7 +181,7 @@ pub struct AwInfo { /// Engine identifier used by the Detection job when explicitly configured. #[serde(skip_serializing_if = "Option::is_none")] pub detection_engine: Option, - /// Model identifier used by the Detection job when explicitly configured. + /// Model requested for the Detection Copilot session, when available. #[serde(skip_serializing_if = "Option::is_none")] pub detection_model: Option, /// Agent name emitted by the compiled workflow metadata. diff --git a/src/audit/render/console.rs b/src/audit/render/console.rs index ecdb8661d..210d03cfe 100644 --- a/src/audit/render/console.rs +++ b/src/audit/render/console.rs @@ -87,14 +87,23 @@ fn render_overview_section( .filter(|config| !config.engine.is_empty()) .map(|config| config.engine.clone()) }); - let model = aw_info + let requested_model = aw_info .and_then(|info| info.model.as_deref()) .filter(|value| !value.is_empty()) - .map(str::to_string) - .or_else(|| engine_config.and_then(|config| config.model.clone())); + .map(str::to_string); + let observed_model = engine_config + .and_then(|config| config.model.as_deref()) + .filter(|value| !value.is_empty()) + .map(str::to_string); push_opt_owned_row(&mut rows, "engine", engine); - push_opt_owned_row(&mut rows, "model", model); + match (&requested_model, &observed_model) { + (Some(requested), Some(observed)) if requested != observed => { + push_opt_owned_row(&mut rows, "requested_model", requested_model); + push_opt_owned_row(&mut rows, "observed_model", observed_model); + } + _ => push_opt_owned_row(&mut rows, "model", observed_model.or(requested_model)), + } if let Some(enabled) = aw_info.and_then(|info| info.threat_detection_enabled) { rows.push(("threat_detection_enabled".to_string(), enabled.to_string())); } @@ -1235,6 +1244,35 @@ mod tests { assert_eq!(headings, vec!["## Overview", "## Metrics"]); } + #[test] + fn overview_distinguishes_requested_and_observed_models() { + let audit = AuditData { + overview: crate::audit::model::OverviewData { + aw_info: Some(AwInfo { + engine: Some("copilot".to_string()), + model: Some("requested-model".to_string()), + ..Default::default() + }), + ..Default::default() + }, + engine_config: Some(AuditEngineConfig { + engine: "copilot".to_string(), + model: Some("observed-model".to_string()), + ..Default::default() + }), + ..Default::default() + }; + + let out = render_console(&audit); + + assert!(out.contains("- requested_model: requested-model"), "{out}"); + assert!(out.contains("- observed_model: observed-model"), "{out}"); + assert!( + !out.contains("- model: requested-model"), + "{out}" + ); + } + #[test] fn ado_proxy_analysis_renders_lifecycle_rollups_and_recent_events() { let audit = AuditData { diff --git a/src/compile/ado_bundle.rs b/src/compile/ado_bundle.rs index 4bb45757f..f795ea469 100644 --- a/src/compile/ado_bundle.rs +++ b/src/compile/ado_bundle.rs @@ -53,6 +53,10 @@ pub enum Bundle { ExecContextRepo, ApprovalSummary, Conclusion, + /// Copilot process harness executed inside AWF. It receives only a + /// compiler-generated invocation document and the already-filtered Agent + /// environment, so it requires no ADO bearer. + CopilotInvoker, /// GitHub App installation-token minter/revoker (issue #1316). Runs before /// the Copilot invocation in the Agent and Detection jobs. It authenticates /// to the **GitHub** API (not ADO REST), so it needs no ADO bearer. @@ -155,6 +159,7 @@ impl Bundle { Bundle::ExecContextRepo, Bundle::ApprovalSummary, Bundle::Conclusion, + Bundle::CopilotInvoker, Bundle::GithubAppToken, Bundle::PreparePrBase, Bundle::AzureWifRefresh, @@ -183,6 +188,7 @@ impl Bundle { Bundle::ExecContextRepo => paths::EXEC_CONTEXT_REPO_PATH, Bundle::ApprovalSummary => paths::APPROVAL_SUMMARY_PATH, Bundle::Conclusion => paths::CONCLUSION_PATH, + Bundle::CopilotInvoker => paths::COPILOT_INVOKER_PATH, Bundle::GithubAppToken => paths::GITHUB_APP_TOKEN_PATH, Bundle::PreparePrBase => paths::PREPARE_PR_BASE_PATH, Bundle::AzureWifRefresh => paths::AZURE_WIF_REFRESH_PATH, @@ -212,6 +218,7 @@ impl Bundle { | Bundle::ExecContextManual | Bundle::ExecContextRepo | Bundle::ApprovalSummary + | Bundle::CopilotInvoker // Authenticates to the GitHub API with its own App JWT / minted // token, not the ADO bearer. | Bundle::GithubAppToken diff --git a/src/compile/agentic_pipeline.rs b/src/compile/agentic_pipeline.rs index 176fb2e0b..e5645a7a5 100644 --- a/src/compile/agentic_pipeline.rs +++ b/src/compile/agentic_pipeline.rs @@ -269,8 +269,8 @@ fn fanout_extension_declarations( /// function's cognitive complexity manageable — behaviour is unchanged. struct EngineSetup { compiler_version: String, - engine_run: String, - engine_run_detection: String, + agent_invocation_json: String, + detection_invocation_json: String, engine_install_steps_yaml: String, detection_engine_install_steps_yaml: String, engine_log_dir: String, @@ -293,19 +293,31 @@ fn build_engine_setup( let compiler_version = env!("CARGO_PKG_VERSION").to_string(); let detection_engine = crate::engine::get_engine(detection_engine_config.engine_id())?; - let engine_run = ctx.engine.invocation( + let agent_invocation = ctx.engine.invocation_document( ctx.front_matter, extension_declarations, - "/tmp/awf-tools/agent-prompt.md", - Some("/tmp/awf-tools/mcp-config.json"), + crate::engine::CopilotInvocationContext::new( + crate::engine::RuntimeModelRole::Agent, + "/tmp/awf-tools/agent-prompt.md", + Some("/tmp/awf-tools/mcp-config.json"), + "/tmp/awf-tools/copilot-invocation-result.json", + ), )?; - let engine_run_detection = detection_engine.invocation_with_config( + let detection_invocation = detection_engine.invocation_document_with_config( detection_engine_config, ctx.front_matter, extension_declarations, - "/tmp/awf-tools/threat-analysis-prompt.md", - None, + crate::engine::CopilotInvocationContext::new( + crate::engine::RuntimeModelRole::Detection, + "/tmp/awf-tools/threat-analysis-prompt.md", + None, + "/tmp/awf-tools/copilot-invocation-result.json", + ), )?; + let agent_invocation_json = serde_json::to_string(&agent_invocation) + .context("failed to serialize Agent Copilot invocation document")?; + let detection_invocation_json = serde_json::to_string(&detection_invocation) + .context("failed to serialize Detection Copilot invocation document")?; let engine_install_steps_yaml = ctx.engine .install_steps(&front_matter.engine, &front_matter.target, ctx.ado_org())?; @@ -348,8 +360,8 @@ fn build_engine_setup( Ok(EngineSetup { compiler_version, - engine_run, - engine_run_detection, + agent_invocation_json, + detection_invocation_json, engine_install_steps_yaml, detection_engine_install_steps_yaml, engine_log_dir, @@ -440,8 +452,8 @@ pub(crate) fn build_pipeline_context( )?; let EngineSetup { compiler_version, - engine_run, - engine_run_detection, + agent_invocation_json, + detection_invocation_json, engine_install_steps_yaml, detection_engine_install_steps_yaml, engine_log_dir, @@ -573,8 +585,8 @@ pub(crate) fn build_pipeline_context( compiler_version: compiler_version.clone(), engine_install_steps_yaml, detection_engine_install_steps_yaml, - engine_run, - engine_run_detection, + agent_invocation_json, + detection_invocation_json, detection_engine_config, threat_detection, engine_env, @@ -828,8 +840,8 @@ pub(crate) struct StandaloneCtx { /// to typed steps. pub(crate) engine_install_steps_yaml: String, pub(crate) detection_engine_install_steps_yaml: String, - pub(crate) engine_run: String, - pub(crate) engine_run_detection: String, + pub(crate) agent_invocation_json: String, + pub(crate) detection_invocation_json: String, pub(crate) detection_engine_config: EngineConfig, pub(crate) threat_detection: ThreatDetectionConfig, /// Composed engine env block — `KEY: VALUE` lines, one per line. @@ -1328,8 +1340,7 @@ fn build_agent_job( // `checkout: self` (step 1) so the clone exists, and before the Copilot // run so the refs are present when the agent proposes a PR. The // `prepare-pr-base.js` bundle is staged by the ado-script extension's - // agent-prepare steps (`prepare_pr_base_active` is OR'd into that - // extension's Agent-job download predicate), so it is guaranteed present. + // always-on Agent preparation. if front_matter.create_pr_config().is_some() { // The prepare step deepens every checkout dir the SafeOutputs MCP server // may generate a patch from — see `create_pr_prepare_repos`. The @@ -1364,11 +1375,7 @@ fn build_agent_job( // step sets. Never runs for SafeOutputs/user steps. // // The ado-script bundle is staged by the ado-script extension's - // agent-prepare steps: `github_app_token_active` is OR'd into that - // extension's Agent-job download predicate (mirroring - // `safe_outputs_summary_active`), so the bundle is guaranteed present by - // the time we reach this step — no need to inspect emitted steps or - // re-download here. + // always-on Agent preparation, so no feature-specific download is needed. if let Some(app_token) = front_matter.engine.github_app_token() { steps.push(super::extensions::ado_script::github_app_token_step_typed( app_token, @@ -1389,7 +1396,7 @@ fn build_agent_job( &cfg.allowed_domains, &cfg.awf_mounts, &cfg.working_directory, - &cfg.engine_run, + &cfg.agent_invocation_json, &cfg.engine_env, &cfg.byom_exclude_keys, front_matter.supply_chain(), @@ -1412,9 +1419,7 @@ fn build_agent_job( // emitted when any safe-output tool is enabled (transparency for every // run); when manual review is configured the reviewed proposals are listed // first. The ado-script bundle was delivered earlier in this job by the - // ado-script extension, gated on the SAME predicate - // (`has_any_safe_output_tool` → `safe_outputs_summary_active`), so the - // bundle is downloaded iff this step is emitted. + // ado-script extension's always-on Agent preparation. if front_matter.has_any_safe_output_tool() { let (_, reviewed_summary_tools) = front_matter.partition_safe_outputs_by_approval(); steps.push(Step::Bash(safe_outputs_summary_step( @@ -1546,17 +1551,6 @@ fn agent_job_variables_hoist( /// (see `AdoScriptExtension::build_agent_conditions` for today's /// only contributor — synth-PR-skip, PR-filter gate, pipeline-filter /// gate, and user `expression:` escape hatches). -/// Whether the Detection job must stage the `ado-script` bundle. The Detection -/// job has no extension-prepare phase (unlike the Agent job, whose bundle -/// download is contributed by `AdoScriptExtension`), so it stages the bundle -/// itself — but gated on this single predicate so exactly one download is -/// emitted. Today only the GitHub App token step needs it; future -/// detection-only bundle consumers should `||` their own condition in here -/// rather than adding a second `install_and_download_steps_typed` call. -fn detection_job_needs_ado_script_bundle(engine_config: &EngineConfig) -> bool { - engine_config.github_app_token().is_some() -} - fn build_detection_job( front_matter: &FrontMatter, cfg: &StandaloneCtx, @@ -1612,15 +1606,14 @@ fn build_detection_job( )?)); steps.push(Step::Bash(setup_compiler_step())); - // Stage auth support before custom pre-steps, but mint credentials only - // after them so trusted setup code receives the least privilege needed. - if detection_job_needs_ado_script_bundle(&cfg.detection_engine_config) { - steps.extend( - super::extensions::ado_script::install_and_download_steps_typed( - front_matter.supply_chain(), - ), - ); - } + // Detection always executes the Copilot invoker bundle. Stage it before + // custom pre-steps; mint optional credentials only after those steps so + // trusted setup code receives the least privilege needed. + steps.extend( + super::extensions::ado_script::install_and_download_steps_typed( + front_matter.supply_chain(), + ), + ); for user_step in &cfg.threat_detection.steps { steps.push(Step::RawYaml(step_to_raw_yaml_string(user_step)?)); } @@ -1640,7 +1633,7 @@ fn build_detection_job( steps.push(Step::Bash(run_threat_analysis_step( &cfg.detection_allowed_domains, &cfg.working_directory, - &cfg.engine_run_detection, + &cfg.detection_invocation_json, &cfg.detection_byom_exclude_keys, &cfg.detection_engine_env, crate::engine::github_token_source_var(&cfg.detection_engine_config), @@ -1657,6 +1650,9 @@ fn build_detection_job( steps.push(Step::RawYaml(step_to_raw_yaml_string(user_step)?)); } steps.push(Step::Bash(prepare_analyzed_outputs_step())); + if cfg.detection_engine_config.model().is_none() { + steps.push(Step::Bash(record_detection_runtime_model_step())); + } steps.push(Step::Bash(evaluate_threat_analysis_step())); } else { steps.push(Step::Bash(prepare_analyzed_outputs_passthrough_step())); @@ -3623,7 +3619,7 @@ shell_script! { externals: [], fragments: [tail], body: r###" -set -eo pipefail +set -o pipefail mkdir -p "$DEST" locate_one() { @@ -4456,22 +4452,41 @@ shell_script! { /// - `image_flags` — `--image-tag` plus optional `--image-registry` /// - `exclude_env` — provider credentials and internal MCP identity keys /// - `awf_mounts` — the compiler-supplied chain of `--mount "…"` args - /// - `routed_engine_run` — the single-quoted `NO_PROXY` prefix + engine - /// command that AWF invokes inside the sandbox + /// - `routed_invoker` — the fixed single-quoted `NO_PROXY` prefix + + /// compiler-owned invoker command that AWF runs inside the sandbox RUN_AGENT { interpreter: Bash, - bindings: [AGENT_TEMP, PIPELINE_WORKSPACE, ALLOWED_DOMAINS], + bindings: [ + AGENT_TEMP, + PIPELINE_WORKSPACE, + ALLOWED_DOMAINS, + INVOCATION_DOCUMENT, + INVOCATION_DOCUMENT_PATH, + INVOCATION_RESULT_PATH, + COPILOT_INVOKER_PATH + ], externals: [WORKING_DIRECTORY], - fragments: [topology_attach, image_flags, exclude_env, awf_mounts, routed_engine_run], + fragments: [ + append_aw_info_field, + topology_attach, + image_flags, + exclude_env, + awf_mounts, + routed_invoker + ], + phases: [append_aw_info_field = super::extensions::APPEND_AW_INFO_FIELD], body: r###" -set -o pipefail +set -eo pipefail AGENT_OUTPUT_FILE="$AGENT_TEMP/staging/logs/agent-output.txt" mkdir -p "$AGENT_TEMP/staging/logs" AGENT_EXIT_CODE=0 +printf '%s\n' "$INVOCATION_DOCUMENT" > "$INVOCATION_DOCUMENT_PATH" +rm -f "$INVOCATION_RESULT_PATH" echo "=== Running AI agent with AWF network isolation ===" echo "Allowed domains: $ALLOWED_DOMAINS" +set +e # AWF provides L7 domain whitelisting via a rootless Docker topology. # The named MCPG container is attached to AWF's internal network as a @@ -4495,7 +4510,7 @@ AWF_ARGS+=( --log-level info --proxy-logs-dir "$AGENT_TEMP/staging/logs/firewall" ) -# ado-aw:fragment routed_engine_run +# ado-aw:fragment routed_invoker # Stream agent output in real-time while filtering VSO commands. # sed -u = unbuffered (line-by-line) so output appears immediately. @@ -4507,6 +4522,27 @@ AWF_ARGS+=( | tee "$AGENT_OUTPUT_FILE" \ || AGENT_EXIT_CODE=$? +MODEL_RESULT_STATUS=0 +REQUESTED_MODEL=$(node "$COPILOT_INVOKER_PATH" read-result "$INVOCATION_RESULT_PATH" agent) \ + || MODEL_RESULT_STATUS=$? +if [ "$MODEL_RESULT_STATUS" -ne 0 ]; then + echo "ERROR: Agent Copilot invocation result is missing or malformed" >&2 + if [ "$AGENT_EXIT_CODE" -eq 0 ]; then + AGENT_EXIT_CODE="$MODEL_RESULT_STATUS" + fi +else + ADO_AW_INFO_JSON="$AGENT_TEMP/staging/aw_info.json" + if [ -f "$ADO_AW_INFO_JSON" ]; then + # ado-aw:fragment append_aw_info_field + ado_aw_append_info_field \ + "model" \ + "$REQUESTED_MODEL" \ + "$ADO_AW_INFO_JSON" + else + echo "Warning: Agent metadata file not found at $ADO_AW_INFO_JSON; model metadata was not recorded" >&2 + fi +fi + # Print firewall summary if available if [ -x "$PIPELINE_WORKSPACE/awf/awf" ]; then echo "=== Firewall Summary ===" @@ -4523,7 +4559,7 @@ fn run_agent_step( allowed_domains: &str, awf_mounts: &str, working_directory: &str, - engine_run: &str, + invocation_document: &str, engine_env: &str, byom_exclude_keys: &[String], supply_chain: Option<&SupplyChainConfig>, @@ -4565,7 +4601,6 @@ fn run_agent_step( }; let image_flags_block = awf_image_flags(supply_chain); let exclude_env_block = awf_exclude_env_flags(byom_exclude_keys); - // AWF attaches externally-launched trusted containers to its internal // network by name. The flag is repeatable, which is what lets the policy // engine join alongside MCPG. Attaching also gives the agent an @@ -4620,9 +4655,10 @@ fn run_agent_step( } else { MCPG_CONTAINER_NAME.to_string() }; - let routed_engine_run = format!( + let routed_invoker = format!( "AWF_ARGS+=(-- 'export NO_PROXY=\"${{NO_PROXY:+$NO_PROXY,}}{no_proxy_peers}\"; \ - export no_proxy=\"$NO_PROXY\"; {engine_run}')" + export no_proxy=\"$NO_PROXY\"; exec node {} run /tmp/awf-tools/copilot-invocation.json')", + super::extensions::ado_script::COPILOT_INVOKER_PATH ); let mut step = ShellScript::new(&RUN_AGENT) @@ -4632,11 +4668,31 @@ fn run_agent_step( Binding::ado_macro("Pipeline.Workspace"), ) .bind_text("ALLOWED_DOMAINS", allowed_domains) + .bind( + "INVOCATION_DOCUMENT", + Binding::document(invocation_document), + ) + .bind_text( + "INVOCATION_DOCUMENT_PATH", + "/tmp/awf-tools/copilot-invocation.json", + ) + .bind_text( + "INVOCATION_RESULT_PATH", + "/tmp/awf-tools/copilot-invocation-result.json", + ) + .bind_text( + "COPILOT_INVOKER_PATH", + super::extensions::ado_script::COPILOT_INVOKER_PATH, + ) + .fragment( + "append_aw_info_field", + phase_body(&super::extensions::APPEND_AW_INFO_FIELD), + ) .fragment("topology_attach", topology_attach_block) .fragment("image_flags", image_flags_line) .fragment("exclude_env", exclude_env_line) .fragment("awf_mounts", awf_mounts_block) - .fragment("routed_engine_run", routed_engine_run) + .fragment("routed_invoker", routed_invoker) .into_step("Run copilot (AWF network isolated)"); step.working_directory = Some(working_directory.to_string()); // Engine env comes as a multi-line YAML env block — `KEY: VALUE` lines @@ -4797,8 +4853,7 @@ node "$APPROVAL_SUMMARY_PATH" || echo "##vso[task.logissue type=warning]approval /// Emitted at the **end of the Agent job** (after `collect_safe_outputs_step` /// has staged `safe_outputs.ndjson`), never in the Detection/threat-analysis /// job. The ado-script bundle is delivered earlier in the same job by the -/// ado-script extension's agent-prepare steps (gated on -/// `safe_outputs_summary_active`). +/// ado-script extension's always-on Agent preparation. /// /// `reviewed` is the compiler-resolved set of approval-gated tool names; when /// non-empty the bundle lists those proposals first under a "Pending approval" @@ -6205,15 +6260,25 @@ shell_script! { /// verbatim. RUN_THREAT_ANALYSIS { interpreter: Bash, - bindings: [AGENT_TEMP, PIPELINE_WORKSPACE, ALLOWED_DOMAINS], + bindings: [ + AGENT_TEMP, + PIPELINE_WORKSPACE, + ALLOWED_DOMAINS, + INVOCATION_DOCUMENT, + INVOCATION_DOCUMENT_PATH, + INVOCATION_RESULT_PATH, + COPILOT_INVOKER_PATH + ], externals: [WORKING_DIRECTORY], - fragments: [image_flags, exclude_env, engine_run_detection], + fragments: [image_flags, exclude_env, run_invoker], body: r###" set -o pipefail # Run threat analysis with AWF network isolation THREAT_OUTPUT_FILE="$AGENT_TEMP/threat-analysis-output.txt" AGENT_EXIT_CODE=0 +printf '%s\n' "$INVOCATION_DOCUMENT" > "$INVOCATION_DOCUMENT_PATH" +rm -f "$INVOCATION_RESULT_PATH" # The argument list is assembled into an array so runtime-supplied # fragments splice in as ordinary shell statements (`AWF_ARGS+=(...)`) @@ -6230,7 +6295,7 @@ AWF_ARGS+=( --log-level info --proxy-logs-dir "$AGENT_TEMP/threat-analysis-logs/firewall" ) -# ado-aw:fragment engine_run_detection +# ado-aw:fragment run_invoker # Stream threat analysis output in real-time with VSO command filtering # shellcheck disable=SC2016 # The single-quoted engine command inside AWF_ARGS is intentionally expanded by AWF inside the sandbox @@ -6239,6 +6304,18 @@ AWF_ARGS+=( | tee "$THREAT_OUTPUT_FILE" \ || AGENT_EXIT_CODE=$? +MODEL_RESULT_STATUS=0 +REQUESTED_MODEL=$(node "$COPILOT_INVOKER_PATH" read-result "$INVOCATION_RESULT_PATH" detection) \ + || MODEL_RESULT_STATUS=$? +if [ "$MODEL_RESULT_STATUS" -ne 0 ]; then + echo "ERROR: Detection Copilot invocation result is missing or malformed" >&2 + if [ "$AGENT_EXIT_CODE" -eq 0 ]; then + AGENT_EXIT_CODE="$MODEL_RESULT_STATUS" + fi +else + printf '%s' "$REQUESTED_MODEL" > "$AGENT_TEMP/detection-runtime-model" +fi + exit "$AGENT_EXIT_CODE" "###, } @@ -6247,7 +6324,7 @@ exit "$AGENT_EXIT_CODE" fn run_threat_analysis_step( allowed_domains: &str, working_directory: &str, - engine_run_detection: &str, + invocation_document: &str, byom_exclude_keys: &[String], detection_engine_env: &[(String, String)], github_token_var: &str, @@ -6283,7 +6360,10 @@ fn run_threat_analysis_step( format!("AWF_ARGS+=({})", parts.join(" ")) } }; - let engine_run_detection_line = format!("AWF_ARGS+=(-- '{engine_run_detection}')"); + let run_invoker = format!( + "AWF_ARGS+=(-- 'exec node {} run /tmp/awf-tools/copilot-invocation.json')", + super::extensions::ado_script::COPILOT_INVOKER_PATH + ); let mut step = ShellScript::new(&RUN_THREAT_ANALYSIS) .bind("AGENT_TEMP", Binding::ado_macro("Agent.TempDirectory")) @@ -6292,9 +6372,25 @@ fn run_threat_analysis_step( Binding::ado_macro("Pipeline.Workspace"), ) .bind_text("ALLOWED_DOMAINS", allowed_domains) + .bind( + "INVOCATION_DOCUMENT", + Binding::document(invocation_document), + ) + .bind_text( + "INVOCATION_DOCUMENT_PATH", + "/tmp/awf-tools/copilot-invocation.json", + ) + .bind_text( + "INVOCATION_RESULT_PATH", + "/tmp/awf-tools/copilot-invocation-result.json", + ) + .bind_text( + "COPILOT_INVOKER_PATH", + super::extensions::ado_script::COPILOT_INVOKER_PATH, + ) .fragment("image_flags", image_flags_line) .fragment("exclude_env", exclude_env_line) - .fragment("engine_run_detection", engine_run_detection_line) + .fragment("run_invoker", run_invoker) .into_step("Run threat analysis (AWF network isolated)"); step.working_directory = Some(working_directory.to_string()); // env block: GITHUB_TOKEN + GITHUB_READ_ONLY — emit the latter as @@ -6374,6 +6470,48 @@ fn prepare_analyzed_outputs_step() -> BashStep { .with_condition(Condition::Always) } +shell_script! { + /// Detection job: resolve the effective runtime model in Detection's own + /// variable scope and enrich the copied Agent metadata. + RECORD_DETECTION_RUNTIME_MODEL { + interpreter: Bash, + bindings: [AGENT_TEMP], + externals: [], + fragments: [append_aw_info_field], + phases: [append_aw_info_field = super::extensions::APPEND_AW_INFO_FIELD], + body: r###" +set -eo pipefail + +ADO_AW_INFO_JSON="$AGENT_TEMP/analyzed_outputs/aw_info.json" +ADO_AW_MODEL_FILE="$AGENT_TEMP/detection-runtime-model" +if [ ! -f "$ADO_AW_INFO_JSON" ]; then + echo "ERROR: Detection could not find copied aw_info.json at $ADO_AW_INFO_JSON" >&2 + exit 1 +fi +if [ ! -f "$ADO_AW_MODEL_FILE" ]; then + exit 0 +fi + +# ado-aw:fragment append_aw_info_field +ado_aw_append_info_field \ + "detection_model" \ + "$(cat "$ADO_AW_MODEL_FILE")" \ + "$ADO_AW_INFO_JSON" +"###, + } +} + +fn record_detection_runtime_model_step() -> BashStep { + ShellScript::new(&RECORD_DETECTION_RUNTIME_MODEL) + .bind("AGENT_TEMP", Binding::ado_macro("Agent.TempDirectory")) + .fragment( + "append_aw_info_field", + phase_body(&super::extensions::APPEND_AW_INFO_FIELD), + ) + .into_step("Record Detection runtime model") + .with_condition(Condition::Always) +} + shell_script! { /// Detection job (AI threat detection disabled): copy Agent proposals to /// `analyzed_outputs/` unchanged. The Detection stage still runs as a @@ -6964,6 +7102,8 @@ const _SUBMODULES_OPT_BIND: Option = None; mod tests { use super::*; use crate::compile::mcpg::McpgLaunchEnvironment; + #[cfg(unix)] + use std::process::{Command, Output}; fn test_front_matter(yaml: &str) -> FrontMatter { serde_yaml::from_str(yaml).expect("front matter should parse") @@ -6989,8 +7129,8 @@ mod tests { compiler_version: "0.0.0-test".to_string(), engine_install_steps_yaml: String::new(), detection_engine_install_steps_yaml: String::new(), - engine_run: "echo agent".to_string(), - engine_run_detection: "echo detection".to_string(), + agent_invocation_json: "{}".to_string(), + detection_invocation_json: "{}".to_string(), detection_engine_config: EngineConfig::default(), threat_detection: ThreatDetectionConfig::default(), engine_env: "GITHUB_READ_ONLY: 1".to_string(), @@ -7661,7 +7801,7 @@ safe-outputs: "example.com", "\\", "/work", - "copilot -p prompt", + r#"{"schema_version":1,"role":"agent"}"#, "FOO: bar", &[], None, @@ -7671,6 +7811,20 @@ safe-outputs: .script } + fn runtime_agent_step_for_test() -> BashStep { + run_agent_step( + "example.com", + "\\", + "/work", + r#"{"schema_version":1,"role":"agent"}"#, + "ADO_AW_MODEL_AGENT_COPILOT: $(ADO_AW_MODEL_AGENT_COPILOT)\nADO_AW_DEFAULT_MODEL_COPILOT: $(ADO_AW_DEFAULT_MODEL_COPILOT)", + &[], + None, + false, + ) + .expect("run_agent_step should build") + } + #[test] fn agent_attaches_only_mcpg_when_the_policy_engine_is_disabled() { let script = agent_step_for_test(false); @@ -7706,6 +7860,81 @@ safe-outputs: ))); } + #[test] + fn agent_runtime_model_is_delegated_to_invoker_and_recorded_after_awf() { + let step = runtime_agent_step_for_test(); + assert!(matches!( + step.env + .get(crate::engine::ADO_AW_MODEL_AGENT_COPILOT), + Some(EnvValue::PipelineVar(name)) + if name == crate::engine::ADO_AW_MODEL_AGENT_COPILOT + )); + assert!(matches!( + step.env + .get(crate::engine::ADO_AW_DEFAULT_MODEL_COPILOT), + Some(EnvValue::PipelineVar(name)) + if name == crate::engine::ADO_AW_DEFAULT_MODEL_COPILOT + )); + assert!(step.script.contains("copilot-invoker.js run")); + assert!(step + .script + .contains("node \"$COPILOT_INVOKER_PATH\" read-result")); + assert!(step.script.contains("\"model\"")); + let metadata_index = step + .script + .rfind("ado_aw_append_info_field") + .expect("metadata append"); + let awf_index = step + .script + .find("\"$PIPELINE_WORKSPACE/awf/awf\"") + .expect("AWF invocation"); + assert!(awf_index < metadata_index); + assert!(!step.script.contains("$(ADO_AW_MODEL_AGENT_COPILOT)")); + assert!(!step.script.contains("$(ADO_AW_DEFAULT_MODEL_COPILOT)")); + assert!(!step.script.contains("ADO_AW_EFFECTIVE_MODEL")); + assert!(!step.script.contains("--model")); + } + + #[test] + fn missing_agent_metadata_does_not_block_agent_execution() { + let step = runtime_agent_step_for_test(); + assert!(step + .script + .contains("if [ -f \"$ADO_AW_INFO_JSON\" ]; then")); + assert!(step.script.contains( + "Warning: Agent metadata file not found at $ADO_AW_INFO_JSON; model metadata was not recorded" + )); + assert!(!step + .script + .contains("ERROR: Agent could not find aw_info.json")); + + let warning_index = step.script.find("model metadata was not recorded").unwrap(); + let awf_index = step + .script + .find("\"$PIPELINE_WORKSPACE/awf/awf\"") + .expect("AWF invocation"); + assert!(awf_index < warning_index); + } + + #[test] + fn prefixed_user_env_key_does_not_change_fixed_invoker_command() { + let step = run_agent_step( + "example.com", + "\\", + "/work", + r#"{"schema_version":1,"role":"agent","explicit_model":"static-model"}"#, + "ADO_AW_MODEL_AGENT_COPILOT_X: harmless", + &[], + None, + false, + ) + .expect("run_agent_step should build"); + + assert!(!step.script.contains("ADO_AW_EFFECTIVE_MODEL")); + assert!(!step.script.contains("unset COPILOT_MODEL")); + assert!(step.script.contains("copilot-invoker.js run")); + } + #[test] fn enabling_the_policy_engine_changes_only_attachment_and_no_proxy() { // Guards against the continuation-indent damage that a hand-built @@ -8217,6 +8446,8 @@ safe-outputs: let fm = parse_and_resolve(source); let threat_detection = fm.threat_detection_config().unwrap(); let detection_engine_config = fm.effective_detection_engine(&threat_detection); + let detection_engine_env = + crate::engine::copilot_detection_env(&detection_engine_config).unwrap(); let ctx = super::super::extensions::CompileContext::for_test(&fm); let extensions = super::super::extensions::collect_extensions(&fm); let decls: Vec<_> = extensions @@ -8260,8 +8491,8 @@ safe-outputs: compiler_version: "0.0.0-test".to_string(), engine_install_steps_yaml: String::new(), detection_engine_install_steps_yaml: String::new(), - engine_run: String::new(), - engine_run_detection: String::new(), + agent_invocation_json: "{}".to_string(), + detection_invocation_json: "{}".to_string(), detection_engine_config, threat_detection, engine_env: "env:\n GITHUB_TOKEN: $(GITHUB_TOKEN)\n".to_string(), @@ -8284,7 +8515,7 @@ safe-outputs: debug_pipeline: false, byom_exclude_keys: vec![], detection_byom_exclude_keys: vec![], - detection_engine_env: vec![], + detection_engine_env, }; build_canonical_jobs( &fm, @@ -8302,6 +8533,68 @@ safe-outputs: jobs.iter().find(|j| j.id.as_ref() == id).map(|j| &j.pool) } + fn detection_runtime_model_step(jobs: &[Job]) -> Option<&BashStep> { + jobs.iter() + .find(|job| job.id.as_ref() == "Detection") + .and_then(|job| { + job.steps.iter().find_map(|step| match step { + Step::Bash(step) if step.display_name == "Record Detection runtime model" => { + Some(step) + } + _ => None, + }) + }) + } + + fn detection_run_step(jobs: &[Job]) -> Option<&BashStep> { + jobs.iter() + .find(|job| job.id.as_ref() == "Detection") + .and_then(|job| { + job.steps.iter().find_map(|step| match step { + Step::Bash(step) + if step.display_name == "Run threat analysis (AWF network isolated)" => + { + Some(step) + } + _ => None, + }) + }) + } + + #[cfg(unix)] + fn run_detection_runtime_model_script(model: Option<&str>) -> (Output, tempfile::TempDir) { + let temp = tempfile::tempdir().expect("temp dir"); + let analyzed_outputs = temp.path().join("analyzed_outputs"); + std::fs::create_dir_all(&analyzed_outputs).expect("create analyzed outputs"); + std::fs::write( + analyzed_outputs.join("aw_info.json"), + r#"{"schema":"ado-aw/aw_info/1"}"#, + ) + .expect("write aw_info.json"); + if let Some(model) = model { + std::fs::write(temp.path().join("detection-runtime-model"), model) + .expect("write runtime model"); + } + let script = ShellScript::new(&RECORD_DETECTION_RUNTIME_MODEL) + .bind_text("AGENT_TEMP", temp.path().display().to_string()) + .fragment( + "append_aw_info_field", + phase_body(&super::super::extensions::APPEND_AW_INFO_FIELD), + ) + .render(); + let mut command = Command::new("bash"); + command.arg("-c").arg(script).env_clear(); + (command.output().expect("bash should run"), temp) + } + + #[cfg(unix)] + fn read_detection_aw_info(temp: &tempfile::TempDir) -> serde_json::Value { + let contents = + std::fs::read_to_string(temp.path().join("analyzed_outputs/aw_info.json")) + .expect("read aw_info.json"); + serde_json::from_str(&contents).expect("parse aw_info.json") + } + #[test] fn threat_detection_enabled_and_disabled_match_expected_ir_graph() { use std::collections::{BTreeMap, BTreeSet}; @@ -8381,6 +8674,7 @@ safe-outputs: assert_eq!(location.job, job("Detection")); assert_eq!(&location.outputs, outputs); } + } let disabled_detection = disabled_jobs @@ -8418,6 +8712,117 @@ safe-outputs: assert!(reviewed_index < copy_logs_index); } + #[test] + fn detection_runtime_model_metadata_uses_detection_job_scope_only_when_enabled() { + let runtime = build_jobs( + "---\nname: test\ndescription: test\nsafe-outputs:\n threat-detection: true\n---\nbody\n", + ); + let disabled = build_jobs( + "---\nname: test\ndescription: test\nsafe-outputs:\n threat-detection: false\n---\nbody\n", + ); + let static_model = build_jobs( + "---\nname: test\ndescription: test\nengine:\n model: static-model\nsafe-outputs:\n threat-detection: true\n---\nbody\n", + ); + + let step = + detection_runtime_model_step(&runtime).expect("runtime Detection emits metadata step"); + assert!(matches!(step.condition, Some(Condition::Always))); + assert!(step.env.is_empty()); + assert!(!step.script.contains("ADO_AW_MODEL_DETECTION_COPILOT")); + assert!(!step.script.contains("ADO_AW_DEFAULT_MODEL_COPILOT")); + + let run_step = detection_run_step(&runtime).expect("enabled Detection runs analysis"); + assert!( + run_step + .env + .contains_key(crate::engine::ADO_AW_MODEL_DETECTION_COPILOT) + ); + assert!( + run_step + .env + .contains_key(crate::engine::ADO_AW_DEFAULT_MODEL_COPILOT) + ); + assert!(matches!( + run_step + .env + .get(crate::engine::ADO_AW_MODEL_DETECTION_COPILOT), + Some(EnvValue::PipelineVar(name)) + if name == crate::engine::ADO_AW_MODEL_DETECTION_COPILOT + )); + assert!(matches!( + run_step + .env + .get(crate::engine::ADO_AW_DEFAULT_MODEL_COPILOT), + Some(EnvValue::PipelineVar(name)) + if name == crate::engine::ADO_AW_DEFAULT_MODEL_COPILOT + )); + assert!( + run_step + .script + .contains("$AGENT_TEMP/detection-runtime-model") + ); + assert!(run_step.script.contains("copilot-invoker.js run")); + assert!(run_step + .script + .contains("node \"$COPILOT_INVOKER_PATH\" read-result")); + assert!(!run_step.script.contains("ADO_AW_EFFECTIVE_MODEL")); + assert!(!run_step.script.contains("--model")); + assert!( + run_step + .script + .find("\"$PIPELINE_WORKSPACE/awf/awf\"") + .unwrap() + < run_step + .script + .find("$AGENT_TEMP/detection-runtime-model") + .unwrap(), + "the trusted host must consume the invoker result after Detection runs" + ); + assert!( + !run_step + .script + .contains("$(ADO_AW_MODEL_DETECTION_COPILOT)") + ); + assert!( + !run_step + .script + .contains("$(ADO_AW_DEFAULT_MODEL_COPILOT)") + ); + + assert!( + detection_runtime_model_step(&disabled).is_none(), + "disabled Detection must not resolve or validate model variables" + ); + assert!( + detection_runtime_model_step(&static_model).is_none(), + "static Detection models are already present in compile-time metadata" + ); + } + + #[test] + #[cfg(unix)] + fn detection_runtime_metadata_records_captured_model() { + let (output, temp) = run_detection_runtime_model_script(Some("detector-model")); + + assert!(output.status.success(), "{output:?}"); + assert_eq!( + read_detection_aw_info(&temp)["detection_model"], + "detector-model" + ); + } + + #[test] + #[cfg(unix)] + fn detection_runtime_metadata_omits_missing_model() { + let (output, temp) = run_detection_runtime_model_script(None); + assert!(output.status.success(), "{output:?}"); + assert!( + read_detection_aw_info(&temp) + .get("detection_model") + .is_none() + ); + } + #[test] fn pool_overrides_detection_only_flows_to_compiled_job() { let source = concat!( diff --git a/src/compile/extensions/ado_aw_marker.rs b/src/compile/extensions/ado_aw_marker.rs index 47c249ad7..af5da1e0f 100644 --- a/src/compile/extensions/ado_aw_marker.rs +++ b/src/compile/extensions/ado_aw_marker.rs @@ -47,6 +47,47 @@ shell_script! { } } +shell_script! { + /// Append one validated string field to a single-line JSON object. + /// + /// This phase is shared by the Agent metadata writer and the Detection + /// metadata enrichment step so their file-update behavior cannot drift. + APPEND_AW_INFO_FIELD { + interpreter: Bash, + bindings: [], + externals: [], + fragments: [], + body: r#" +ado_aw_append_info_field() { + local field="$1" + local value="$2" + local file="$3" + if [ -z "$value" ] || grep -Eq "\"${field}\"[[:space:]]*:" "$file"; then + return 0 + fi + local json + local tmp + local separator="," + json="$(cat "$file")" + if [ "$json" = "{}" ]; then + separator="" + else + case "$json" in + \{*\}) ;; + *) + echo "ERROR: aw_info.json is not a single-line JSON object" >&2 + exit 1 + ;; + esac + fi + tmp="$(mktemp)" + printf '%s%s"%s":"%s"}' "${json%?}" "$separator" "$field" "$value" > "$tmp" + mv "$tmp" "$file" +} +"#, + } +} + shell_script! { /// Write `aw_info.json` to Agent.TempDirectory/staging. /// @@ -242,23 +283,20 @@ impl CompileMetadata { .front_matter .safe_outputs .contains_key(crate::compile::types::THREAT_DETECTION_KEY); - let (threat_detection_enabled, detection_engine, detection_model) = - if explicit_threat_detection { - let config = ctx.front_matter.threat_detection_config()?; - let (engine, model) = if config.engine.is_some() { - let effective = ctx.front_matter.effective_detection_engine(&config); - let engine = crate::engine::get_engine(effective.engine_id())?; - let model = match engine { - crate::engine::Engine::Copilot => effective.model().map(str::to_string), - }; - (Some(effective.engine_id().to_string()), model) - } else { - (None, None) - }; - (Some(config.is_enabled()), engine, model) - } else { - (None, None, None) - }; + let config = ctx.front_matter.threat_detection_config()?; + let effective = ctx.front_matter.effective_detection_engine(&config); + let engine = crate::engine::get_engine(effective.engine_id())?; + let detection_model = if config.is_enabled() { + match engine { + crate::engine::Engine::Copilot => effective.model().map(str::to_string), + } + } else { + None + }; + let threat_detection_enabled = + explicit_threat_detection.then_some(config.is_enabled()); + let detection_engine = (explicit_threat_detection && config.engine.is_some()) + .then_some(effective.engine_id().to_string()); Ok(Some(Self { source: super::super::common::normalize_source_path(input_path), org: ctx @@ -330,7 +368,10 @@ impl CompileMetadata { ); } if let Some(model) = &self.model { - object.insert("model".to_string(), serde_json::Value::String(model.clone())); + object.insert( + "model".to_string(), + serde_json::Value::String(model.clone()), + ); } if let Some(model) = &self.detection_model { object.insert( @@ -459,6 +500,8 @@ mod tests { use crate::compile::extensions::CompileContext; use crate::compile::types::FrontMatter; use std::path::Path; + #[cfg(unix)] + use std::process::{Command, Output}; fn parse_fm(yaml: &str) -> FrontMatter { serde_yaml::from_str(yaml).expect("front matter parses") @@ -478,6 +521,37 @@ mod tests { } } + #[cfg(unix)] + fn run_append_aw_info_field( + field: &str, + value: &str, + aw_info_json: &str, + ) -> (Output, tempfile::TempDir) { + let temp = tempfile::tempdir().expect("temp dir"); + let path = temp.path().join("aw_info.json"); + std::fs::write(&path, aw_info_json).expect("write aw_info.json"); + let script = format!( + "{}\nado_aw_append_info_field \"$FIELD\" \"$VALUE\" \"$FILE\"", + APPEND_AW_INFO_FIELD.body.trim() + ); + let mut command = Command::new("bash"); + command + .arg("-c") + .arg(script) + .env_clear() + .env("FIELD", field) + .env("VALUE", value) + .env("FILE", &path); + (command.output().expect("bash should run"), temp) + } + + #[cfg(unix)] + fn read_aw_info_json(temp: &tempfile::TempDir) -> serde_json::Value { + let path = temp.path().join("aw_info.json"); + let contents = std::fs::read_to_string(path).expect("aw_info.json should be written"); + serde_json::from_str(&contents).expect("aw_info.json should parse") + } + #[test] fn returns_no_step_when_input_path_absent() { let fm = parse_fm("name: t\ndescription: x\n"); @@ -591,7 +665,7 @@ mod tests { step.script ); assert!( - !step.script.contains("\"model\""), + !step.script.contains("\"model\":\""), "step should omit model when no model is configured:\n{}", step.script ); @@ -623,14 +697,81 @@ mod tests { "step missing build_definition_id macro:\n{}", step.script ); - assert!(!step.script.contains("detection_model")); - assert!(!step.script.contains("threat_detection_enabled")); + assert!(!step.script.contains("\"threat_detection_enabled\"")); + assert!( + !step + .env + .contains_key(crate::engine::ADO_AW_MODEL_DETECTION_COPILOT) + ); + assert!( + !step + .env + .contains_key(crate::engine::ADO_AW_MODEL_AGENT_COPILOT) + ); + assert!( + !step + .env + .contains_key(crate::engine::ADO_AW_DEFAULT_MODEL_COPILOT) + ); + assert!( + !step.script.contains("$(ADO_AW_MODEL_DETECTION_COPILOT)"), + "runtime model macros must only appear in env mappings:\n{}", + step.script + ); + assert!( + !step.script.contains("$(ADO_AW_DEFAULT_MODEL_COPILOT)"), + "runtime model macros must only appear in env mappings:\n{}", + step.script + ); + } + + #[test] + #[cfg(unix)] + fn append_aw_info_field_does_not_confuse_value_with_model_key() { + let (output, temp) = run_append_aw_info_field( + "model", + "gpt-5", + r#"{"agent_name":"model","schema":"ado-aw/aw_info/1"}"#, + ); + + assert!(output.status.success(), "{output:?}"); + let value = read_aw_info_json(&temp); + assert_eq!(value["agent_name"], "model"); + assert_eq!(value["model"], "gpt-5"); + } + + #[test] + #[cfg(unix)] + fn append_aw_info_field_does_not_confuse_value_with_detection_model_key() { + let (output, temp) = run_append_aw_info_field( + "detection_model", + "gpt-5-mini", + r#"{"agent_name":"detection_model","schema":"ado-aw/aw_info/1"}"#, + ); + + assert!(output.status.success(), "{output:?}"); + let value = read_aw_info_json(&temp); + assert_eq!(value["agent_name"], "detection_model"); + assert_eq!(value["detection_model"], "gpt-5-mini"); + } + + #[test] + #[cfg(unix)] + fn append_aw_info_field_preserves_existing_key_with_whitespace() { + let (output, temp) = run_append_aw_info_field( + "model", + "replacement", + r#"{"model" : "original","schema":"ado-aw/aw_info/1"}"#, + ); + + assert!(output.status.success(), "{output:?}"); + let value = read_aw_info_json(&temp); + assert_eq!(value["model"], "original"); } #[test] fn explicit_model_emits_aw_info_model_metadata() { - let fm = - parse_fm("name: t\ndescription: x\nengine:\n id: copilot\n model: some-model\n"); + let fm = parse_fm("name: t\ndescription: x\nengine:\n id: copilot\n model: some-model\n"); let input_path = Path::new("agents/foo.md"); let ctx = CompileContext { agent_name: &fm.name, @@ -651,7 +792,7 @@ mod tests { } #[test] - fn explicit_threat_detection_emits_detector_metadata() { + fn disabled_threat_detection_omits_detector_model() { let fm = parse_fm( "name: t\ndescription: x\nengine:\n id: copilot\n model: agent-model\n\ safe-outputs:\n threat-detection:\n enabled: false\n engine:\n \ @@ -670,8 +811,7 @@ mod tests { let steps = agent_prepare_steps(&ctx); let step = bash_step(&steps[1]); assert!( - step.script - .contains("\"threat_detection_enabled\":false"), + step.script.contains("\"threat_detection_enabled\":false"), "{}", step.script ); @@ -680,18 +820,86 @@ mod tests { "{}", step.script ); + assert!(!step.script.contains("\"detection_model\""), "{}", step.script); + } + + #[test] + fn explicit_default_threat_detection_emits_enabled_state_only() { + let fm = parse_fm("name: t\ndescription: x\nsafe-outputs:\n threat-detection: true\n"); + let input_path = Path::new("agents/foo.md"); + let ctx = CompileContext { + agent_name: &fm.name, + front_matter: &fm, + ado_context: None, + engine: crate::engine::Engine::Copilot, + compile_dir: None, + input_path: Some(input_path), + imported_prompt_body: String::new(), + }; + let steps = agent_prepare_steps(&ctx); + let step = bash_step(&steps[1]); + assert!( + step.script.contains("\"threat_detection_enabled\":true"), + "{}", + step.script + ); + assert!(!step.script.contains("\"detection_engine\"")); + assert!(!step.script.contains("ADO_AW_MODEL_DETECTION_COPILOT")); + } + + #[test] + fn inherited_detection_model_is_emitted_as_static_metadata() { + let fm = parse_fm( + "name: t\ndescription: x\nengine:\n id: copilot\n model: agent-model\n\ + safe-outputs:\n threat-detection: true\n", + ); + let input_path = Path::new("agents/foo.md"); + let ctx = CompileContext { + agent_name: &fm.name, + front_matter: &fm, + ado_context: None, + engine: crate::engine::Engine::Copilot, + compile_dir: None, + input_path: Some(input_path), + imported_prompt_body: String::new(), + }; + let steps = agent_prepare_steps(&ctx); + let step = bash_step(&steps[1]); assert!( + step.script.contains("\"detection_model\":\"agent-model\""), + "{}", step.script - .contains("\"detection_model\":\"detector-model\""), + ); + } + + #[test] + fn implicit_detection_inherits_static_agent_model_metadata() { + let fm = + parse_fm("name: t\ndescription: x\nengine:\n id: copilot\n model: agent-model\n"); + let input_path = Path::new("agents/foo.md"); + let ctx = CompileContext { + agent_name: &fm.name, + front_matter: &fm, + ado_context: None, + engine: crate::engine::Engine::Copilot, + compile_dir: None, + input_path: Some(input_path), + imported_prompt_body: String::new(), + }; + let steps = agent_prepare_steps(&ctx); + let step = bash_step(&steps[1]); + assert!( + step.script + .contains("\"detection_model\":\"agent-model\""), "{}", step.script ); } #[test] - fn explicit_default_threat_detection_emits_enabled_state_only() { + fn enabled_detection_specific_static_model_is_emitted() { let fm = parse_fm( - "name: t\ndescription: x\nsafe-outputs:\n threat-detection: true\n", + "name: t\ndescription: x\nsafe-outputs:\n threat-detection:\n enabled: true\n engine:\n model: detector-model\n", ); let input_path = Path::new("agents/foo.md"); let ctx = CompileContext { @@ -706,12 +914,11 @@ mod tests { let steps = agent_prepare_steps(&ctx); let step = bash_step(&steps[1]); assert!( - step.script.contains("\"threat_detection_enabled\":true"), + step.script + .contains("\"detection_model\":\"detector-model\""), "{}", step.script ); - assert!(!step.script.contains("\"detection_engine\"")); - assert!(!step.script.contains("\"detection_model\"")); } #[test] diff --git a/src/compile/extensions/ado_script.rs b/src/compile/extensions/ado_script.rs index 26bd43c07..9cc39371a 100644 --- a/src/compile/extensions/ado_script.rs +++ b/src/compile/extensions/ado_script.rs @@ -220,6 +220,8 @@ node "$BUNDLE" pub(crate) const GATE_EVAL_PATH: &str = "/tmp/ado-aw-scripts/ado-script/gate.js"; pub(crate) const IMPORT_EVAL_PATH: &str = "/tmp/ado-aw-scripts/ado-script/import.js"; +pub(crate) const COPILOT_INVOKER_PATH: &str = + "/tmp/ado-aw-scripts/ado-script/copilot-invoker.js"; /// Path to the ado-proxy bundle inside the unpacked `ado-script.zip`. /// /// Unlike every other bundle this one is not executed by a pipeline step. It @@ -308,83 +310,6 @@ pub struct AdoScriptExtension { pub pr_filters: Option, pub pipeline_filters: Option, pub inlined_imports: bool, - /// Whether the PR-context contributor will activate. When true, - /// the Agent-job install/download must fire even if - /// `runtime_imports_active()` is false (i.e. the user has - /// `inlined-imports: true` but a PR trigger configured), so that - /// `exec-context-pr.js` is present for the `pr.rs` invocation. - /// - /// Populated at construction by `collect_extensions` using the - /// shared `exec_context_pr_active` predicate so this stays in - /// lock-step with `ExecContextExtension`'s own activation gate. - pub exec_context_pr_active: bool, - /// Whether the Manual-context contributor (Stage 1 of the - /// exec-context contributor build-out — see plan.md) will - /// activate. When true, the Agent-job install/download must - /// fire so that `exec-context-manual.js` is present. - /// - /// Populated at construction by `collect_extensions` using the - /// shared `manual_contributor_will_activate` predicate so this - /// stays in lock-step with the contributor's `should_activate`. - pub exec_context_manual_active: bool, - /// Whether the Pipeline-context contributor (Stage 2 of the - /// exec-context contributor build-out — see plan.md) will - /// activate. When true, the Agent-job install/download must - /// fire so that `exec-context-pipeline.js` is present. - /// - /// Populated at construction by `collect_extensions` using the - /// shared `pipeline_contributor_will_activate` predicate so this - /// stays in lock-step with the contributor's `should_activate`. - pub exec_context_pipeline_active: bool, - /// Whether the CI-push-context contributor (Stage 3 of the - /// exec-context contributor build-out — see plan.md) will - /// activate. Default-off opt-in feature; when true the - /// install/download must fire so that - /// `exec-context-ci-push.js` is present. - pub exec_context_ci_push_active: bool, - /// Whether the Workitem-context contributor (Stage 4 of the - /// exec-context contributor build-out — see plan.md) will - /// activate. Activates whenever the PR contributor activates - /// unless explicitly disabled. **Crosses an untrusted-prose - /// boundary** — see workitem.rs. - pub exec_context_workitem_active: bool, - /// Whether the Schedule-context contributor (Stage 5 of the - /// exec-context contributor build-out — see plan.md) will - /// activate. Opt-in (default OFF). - pub exec_context_schedule_active: bool, - /// Whether the PR-checks extension (Stage 6 of the build-out — - /// see plan.md) will activate. Opt-in (default OFF) AND - /// requires the PR contributor to activate. - pub exec_context_pr_checks_active: bool, - /// Whether the Repo-context contributor (Stage 7 of the - /// build-out — see plan.md) will activate. Always-on capability, - /// default OFF (opt-in). - pub exec_context_repo_active: bool, - /// Whether the safe-outputs approval-summary step will run at the - /// end of the Agent job. True whenever the workflow enables any - /// safe-output tool. When true the Agent-job install/download must - /// fire so that `approval-summary.js` is present for the - /// end-of-job render step (emitted by `build_agent_job`). - pub safe_outputs_summary_active: bool, - /// Whether GitHub App-backed Copilot auth is configured - /// (`engine.github-app-token`, issue #1316). When true the Agent-job - /// install/download must fire so that `github-app-token.js` is present for - /// the mint (and revoke) steps that `build_agent_job` emits immediately - /// around the Copilot run. Mirrors `safe_outputs_summary_active`: the - /// consuming steps are emitted by `build_agent_job`, not this extension, so - /// the flag drives the shared bundle download — the builder never has to - /// inspect emitted steps to decide whether to download. - pub github_app_token_active: bool, - /// Whether `create-pull-request` is configured (issue #1413). When true the - /// Agent-job install/download must fire so that `prepare-pr-base.js` is - /// present for the base-ref prepare step that `build_agent_job` emits before - /// the Copilot run. Mirrors `github_app_token_active`: the consuming step is - /// emitted by `build_agent_job`, not this extension, so the flag drives the - /// shared bundle download. - pub prepare_pr_base_active: bool, - /// Whether any user-defined stdio MCP server configures `azure-auth`. - /// Drives Agent-job bundle delivery for `azure-wif-refresh.js`. - pub azure_mcp_auth_active: bool, /// PR trigger config required to build `PR_SYNTH_SPEC`. `Some(_)` /// is the single source of truth for "synthetic-from-ci path is /// active for this agent" — `is_some()` replaces what used to be a @@ -1156,25 +1081,9 @@ impl CompilerExtension for AdoScriptExtension { // ─── Agent job ───────────────────────────────────────── let mut agent_prepare_steps: Vec = Vec::new(); let import_active = self.runtime_imports_active(); - if import_active - || self.exec_context_pr_active - || self.exec_context_manual_active - || self.exec_context_pipeline_active - || self.exec_context_ci_push_active - || self.exec_context_workitem_active - || self.exec_context_schedule_active - || self.exec_context_pr_checks_active - || self.exec_context_repo_active - || self.safe_outputs_summary_active - || self.github_app_token_active - || self.prepare_pr_base_active - || self.azure_mcp_auth_active - { - agent_prepare_steps - .extend(install_and_download_steps_typed(self.supply_chain.as_ref())); - if import_active { - agent_prepare_steps.push(resolver_step_typed()); - } + agent_prepare_steps.extend(install_and_download_steps_typed(self.supply_chain.as_ref())); + if import_active { + agent_prepare_steps.push(resolver_step_typed()); } // ─── Agent-job condition contribution ────────────────── @@ -1365,18 +1274,6 @@ mod tests { pr_filters: pr, pipeline_filters: pipeline, inlined_imports: inlined, - exec_context_pr_active: false, - exec_context_manual_active: false, - exec_context_pipeline_active: false, - exec_context_ci_push_active: false, - exec_context_workitem_active: false, - exec_context_schedule_active: false, - exec_context_pr_checks_active: false, - exec_context_repo_active: false, - safe_outputs_summary_active: false, - github_app_token_active: false, - prepare_pr_base_active: false, - azure_mcp_auth_active: false, pr_trigger_for_synth: None, supply_chain: None, } @@ -1445,18 +1342,6 @@ mod tests { pr_filters: None, pipeline_filters: None, inlined_imports: true, - exec_context_pr_active: false, - exec_context_manual_active: false, - exec_context_pipeline_active: false, - exec_context_ci_push_active: false, - exec_context_workitem_active: false, - exec_context_schedule_active: false, - exec_context_pr_checks_active: false, - exec_context_repo_active: false, - safe_outputs_summary_active: false, - github_app_token_active: false, - prepare_pr_base_active: false, - azure_mcp_auth_active: false, pr_trigger_for_synth: Some(PrTriggerConfig { branches: Some(BranchFilter { include: vec!["main".into()], @@ -1505,18 +1390,6 @@ mod tests { pr_filters: Some(filters), pipeline_filters: None, inlined_imports: true, - exec_context_pr_active: false, - exec_context_manual_active: false, - exec_context_pipeline_active: false, - exec_context_ci_push_active: false, - exec_context_workitem_active: false, - exec_context_schedule_active: false, - exec_context_pr_checks_active: false, - exec_context_repo_active: false, - safe_outputs_summary_active: false, - github_app_token_active: false, - prepare_pr_base_active: false, - azure_mcp_auth_active: false, pr_trigger_for_synth: Some(PrTriggerConfig { branches: Some(BranchFilter { include: vec!["main".into()], @@ -2090,14 +1963,13 @@ mod tests { } #[test] - fn declarations_agent_prepare_download_fires_when_only_prepare_pr_base_active() { - let mut ext = ext_with(None, None, true); - ext.prepare_pr_base_active = true; + fn declarations_agent_prepare_always_stages_bundle() { + let ext = ext_with(None, None, true); let fm: FrontMatter = serde_yaml::from_str("name: t\ndescription: t").unwrap(); let ctx = CompileContext::for_test(&fm); let steps = ext.declarations(&ctx).unwrap().agent_prepare_steps; - // Install + download fire (so prepare-pr-base.js is staged), but no - // runtime-import resolver (inlined_imports: true). + // The invoker is required by every Agent job, so install + download + // fire even when no other ado-script consumer is active. assert_eq!(steps.len(), 2, "install + download only"); assert!(matches!(&steps[0], Step::Task(t) if t.task == "UseNode@1")); assert!( @@ -2266,18 +2138,6 @@ mod tests { pr_filters: pr, pipeline_filters: pipeline, inlined_imports: true, - exec_context_pr_active: false, - exec_context_manual_active: false, - exec_context_pipeline_active: false, - exec_context_ci_push_active: false, - exec_context_workitem_active: false, - exec_context_schedule_active: false, - exec_context_pr_checks_active: false, - exec_context_repo_active: false, - safe_outputs_summary_active: false, - github_app_token_active: false, - prepare_pr_base_active: false, - azure_mcp_auth_active: false, pr_trigger_for_synth: Some(PrTriggerConfig { branches: Some(BranchFilter { include: vec!["main".into()], @@ -2740,30 +2600,19 @@ mod tests { // ── Typed-IR declarations (port-ado-script) ───────────────────── - /// `declarations()` returns empty step lists when neither - /// runtime-import nor exec-context-pr nor any gate / synth path - /// is active. + /// Setup remains empty when no gate / synth path is active, while Agent + /// preparation always stages the Copilot invoker bundle. #[test] - fn declarations_empty_when_nothing_active() { + fn declarations_stages_agent_bundle_when_nothing_else_active() { let ext = ext_with(None, None, true); let fm: FrontMatter = serde_yaml::from_str("name: t\ndescription: t").unwrap(); let ctx = CompileContext::for_test(&fm); let decl = ext.declarations(&ctx).unwrap(); assert!(decl.setup_steps.is_empty()); - assert!(decl.agent_prepare_steps.is_empty()); - } - - #[test] - fn declarations_agent_prepare_download_fires_for_azure_mcp_auth() { - let mut ext = ext_with(None, None, true); - ext.azure_mcp_auth_active = true; - let fm: FrontMatter = serde_yaml::from_str("name: t\ndescription: t").unwrap(); - let ctx = CompileContext::for_test(&fm); - let steps = ext.declarations(&ctx).unwrap().agent_prepare_steps; - assert_eq!(steps.len(), 2, "install + download only"); - assert!(matches!(&steps[0], Step::Task(t) if t.task == "UseNode@1")); + assert_eq!(decl.agent_prepare_steps.len(), 2, "install + download"); + assert!(matches!(&decl.agent_prepare_steps[0], Step::Task(t) if t.task == "UseNode@1")); assert!( - matches!(&steps[1], Step::Bash(b) if b.display_name.contains("Download ado-aw scripts")) + matches!(&decl.agent_prepare_steps[1], Step::Bash(b) if b.display_name.contains("Download ado-aw scripts")) ); } @@ -2816,18 +2665,6 @@ mod tests { pr_filters: None, pipeline_filters: None, inlined_imports: true, - exec_context_pr_active: false, - exec_context_manual_active: false, - exec_context_pipeline_active: false, - exec_context_ci_push_active: false, - exec_context_workitem_active: false, - exec_context_schedule_active: false, - exec_context_pr_checks_active: false, - exec_context_repo_active: false, - safe_outputs_summary_active: false, - github_app_token_active: false, - prepare_pr_base_active: false, - azure_mcp_auth_active: false, pr_trigger_for_synth: Some(PrTriggerConfig { branches: Some(BranchFilter { include: vec!["main".into()], diff --git a/src/compile/extensions/exec_context/mod.rs b/src/compile/extensions/exec_context/mod.rs index 82d949013..4d1e5bf93 100644 --- a/src/compile/extensions/exec_context/mod.rs +++ b/src/compile/extensions/exec_context/mod.rs @@ -55,118 +55,6 @@ use repo::RepoContextContributor; use schedule::ScheduleContextContributor; use workitem::WorkitemContextContributor; -/// Returns `true` iff the PR-context contributor will activate for the -/// given front matter. Shared between `ExecContextExtension::new` (for -/// its own `any_contributor_active` precomputation) and -/// `collect_extensions` (which passes it to `AdoScriptExtension` so -/// the Agent-job install/download fires whenever the bundle is needed). -/// -/// MAINTENANCE: this MUST match `PrContextContributor::should_activate` -/// (in `pr.rs`). The duplication is intentional — `should_activate` -/// takes a `CompileContext` that includes both front matter and target, -/// while this helper only needs the front matter (because `target` is -/// not relevant to PR activation today). -pub fn pr_contributor_will_activate(front_matter: &FrontMatter) -> bool { - // Borrow the embedded config when present; fall back to a stack- - // local default. Avoids the per-call clone — this helper is called - // on every `collect_extensions` invocation, which is hot during - // compile. - let default_cfg = ExecutionContextConfig::default(); - let cfg = front_matter - .execution_context - .as_ref() - .unwrap_or(&default_cfg); - pr_contributor_will_activate_with_cfg(cfg, front_matter) -} - -/// Returns `true` iff the Manual-context contributor will activate -/// for the given front matter. Shared between `ExecContextExtension::new` -/// (for its own `any_contributor_active` aggregate) and -/// `collect_extensions` (which passes it to `AdoScriptExtension` so -/// the Agent-job install/download fires whenever the bundle is needed). -/// -/// MAINTENANCE: this MUST match -/// `ManualContextContributor::should_activate` (in `manual.rs`). -/// Tests in `tests::manual` exercise both paths. -pub fn manual_contributor_will_activate(front_matter: &FrontMatter) -> bool { - let default_cfg = ExecutionContextConfig::default(); - let cfg = front_matter - .execution_context - .as_ref() - .unwrap_or(&default_cfg); - manual_contributor_will_activate_with_cfg(cfg, front_matter) -} - -/// Returns `true` iff the Pipeline-context contributor will activate -/// for the given front matter. Same pattern as the helpers above. -/// -/// MAINTENANCE: this MUST match -/// `PipelineContextContributor::should_activate` (in `pipeline.rs`). -pub fn pipeline_contributor_will_activate(front_matter: &FrontMatter) -> bool { - let default_cfg = ExecutionContextConfig::default(); - let cfg = front_matter - .execution_context - .as_ref() - .unwrap_or(&default_cfg); - pipeline_contributor_will_activate_with_cfg(cfg, front_matter) -} - -/// Returns `true` iff the CI-push-context contributor will activate -/// for the given front matter. Purely config-driven (opt-in, -/// default OFF). -pub fn ci_push_contributor_will_activate(front_matter: &FrontMatter) -> bool { - let default_cfg = ExecutionContextConfig::default(); - let cfg = front_matter - .execution_context - .as_ref() - .unwrap_or(&default_cfg); - ci_push_contributor_will_activate_with_cfg(cfg, front_matter) -} - -/// Returns `true` iff the Workitem contributor will activate. -/// PR-linked mode only — depends on the PR trigger being configured. -pub fn workitem_contributor_will_activate(front_matter: &FrontMatter) -> bool { - let default_cfg = ExecutionContextConfig::default(); - let cfg = front_matter - .execution_context - .as_ref() - .unwrap_or(&default_cfg); - workitem_contributor_will_activate_with_cfg(cfg, front_matter) -} - -/// Returns `true` iff the Schedule contributor will activate. Opt-in -/// (default OFF) AND requires `on.schedule` to be declared. -pub fn schedule_contributor_will_activate(front_matter: &FrontMatter) -> bool { - let default_cfg = ExecutionContextConfig::default(); - let cfg = front_matter - .execution_context - .as_ref() - .unwrap_or(&default_cfg); - schedule_contributor_will_activate_with_cfg(cfg, front_matter) -} - -/// Returns `true` iff the PR-checks extension will activate. Opt-in -/// (default OFF) AND requires the PR contributor to activate. -pub fn pr_checks_contributor_will_activate(front_matter: &FrontMatter) -> bool { - let default_cfg = ExecutionContextConfig::default(); - let cfg = front_matter - .execution_context - .as_ref() - .unwrap_or(&default_cfg); - pr_checks_contributor_will_activate_with_cfg(cfg, front_matter) -} - -/// Returns `true` iff the Repo contributor will activate. Pure -/// config-driven (opt-in, default OFF). -pub fn repo_contributor_will_activate(front_matter: &FrontMatter) -> bool { - let default_cfg = ExecutionContextConfig::default(); - let cfg = front_matter - .execution_context - .as_ref() - .unwrap_or(&default_cfg); - repo_contributor_will_activate_with_cfg(cfg, front_matter) -} - /// Variant that takes the resolved `ExecutionContextConfig` explicitly. /// Used by [`ExecContextExtension::new`] so its internal /// `any_contributor_active` precomputation tracks the config it was diff --git a/src/compile/extensions/exec_context/pr.rs b/src/compile/extensions/exec_context/pr.rs index 21d50c44e..bff38c33d 100644 --- a/src/compile/extensions/exec_context/pr.rs +++ b/src/compile/extensions/exec_context/pr.rs @@ -138,13 +138,6 @@ impl ContextContributor for PrContextContributor { } fn should_activate(&self, ctx: &CompileContext) -> bool { - // MAINTENANCE: this MUST stay in lock-step with - // `super::pr_contributor_will_activate` (the shared helper used - // by `collect_extensions` to populate - // `AdoScriptExtension::exec_context_pr_active`). The divergence- - // trap tests in `super::tests` exercise the helper path; this - // method is the runtime-context-aware version used by the - // declarations path. if ctx.front_matter.pr_trigger().is_none() { return false; } diff --git a/src/compile/extensions/mod.rs b/src/compile/extensions/mod.rs index 26a7f6736..2970f2c1a 100644 --- a/src/compile/extensions/mod.rs +++ b/src/compile/extensions/mod.rs @@ -686,14 +686,10 @@ pub use crate::runtimes::python::PythonExtension; pub use crate::tools::azure_devops::AzureDevOpsExtension; pub use crate::tools::cache_memory::CacheMemoryExtension; pub use ado_aw_marker::AdoAwMarkerExtension; +pub(crate) use ado_aw_marker::APPEND_AW_INFO_FIELD; pub use ado_script::AdoScriptExtension; pub use azure_cli::AzureCliExtension; -pub use exec_context::{ - ExecContextExtension, ci_push_contributor_will_activate, manual_contributor_will_activate, - pipeline_contributor_will_activate, pr_checks_contributor_will_activate, - pr_contributor_will_activate, repo_contributor_will_activate, - schedule_contributor_will_activate, workitem_contributor_will_activate, -}; +pub use exec_context::ExecContextExtension; pub use github::GitHubExtension; pub use safe_outputs::SafeOutputsExtension; @@ -773,59 +769,6 @@ pub fn collect_extensions(front_matter: &FrontMatter) -> Vec { pr_filters: front_matter.pr_filters().cloned(), pipeline_filters: front_matter.pipeline_filters().cloned(), inlined_imports: front_matter.inlined_imports, - // Tell the ado-script extension whether the PR-context - // contributor will activate so it can fire the Agent-job - // install/download even when `inlined-imports: true` (no - // import.js needed). The two extensions stay loosely - // coupled: ExecContextExtension owns invoking the bundle; - // AdoScriptExtension owns installing it. Shared helper - // keeps the activation predicate in lock-step. - exec_context_pr_active: pr_contributor_will_activate(front_matter), - // Same loose-coupling pattern for the Manual contributor - // (Stage 1 of the exec-context contributor build-out — - // see plan.md). Activates whenever any `parameters:` - // block is declared and the contributor isn't explicitly - // disabled. - exec_context_manual_active: manual_contributor_will_activate(front_matter), - // Same loose-coupling pattern for the Pipeline contributor - // (Stage 2 of the exec-context contributor build-out — - // see plan.md). Activates whenever `on.pipeline` is - // configured and the contributor isn't explicitly - // disabled. - exec_context_pipeline_active: pipeline_contributor_will_activate(front_matter), - // CI-push contributor (Stage 3 — opt-in, default OFF). - exec_context_ci_push_active: ci_push_contributor_will_activate(front_matter), - // Workitem contributor (Stage 4 — PR-linked mode only). - // Activates whenever the PR contributor activates and - // workitem isn't explicitly disabled. - exec_context_workitem_active: workitem_contributor_will_activate(front_matter), - // Schedule contributor (Stage 5 — opt-in, default OFF). - exec_context_schedule_active: schedule_contributor_will_activate(front_matter), - // PR-checks extension (Stage 6 — opt-in, default OFF). - exec_context_pr_checks_active: pr_checks_contributor_will_activate(front_matter), - // Repo contributor (Stage 7 — opt-in, default OFF, no - // bearer / no REST, pure git). - exec_context_repo_active: repo_contributor_will_activate(front_matter), - // True whenever any safe-output tool is enabled — drives the - // Agent-job bundle install/download so `approval-summary.js` - // is present for the end-of-job render step that - // `build_agent_job` emits. MUST use the same predicate as that - // step (see `FrontMatter::has_any_safe_output_tool`). - safe_outputs_summary_active: front_matter.has_any_safe_output_tool(), - // True when `engine.github-app-token` is configured — drives the - // Agent-job bundle install/download so `github-app-token.js` is - // present for the mint/revoke steps that `build_agent_job` emits - // around the Copilot run. Same loose-coupling pattern as - // `safe_outputs_summary_active`: the consuming steps live in - // `build_agent_job`, not this extension. - github_app_token_active: front_matter.engine.github_app_token().is_some(), - // True when `create-pull-request` is configured (issue #1413) — - // drives the Agent-job bundle download so `prepare-pr-base.js` - // is present for the base-ref prepare step `build_agent_job` - // emits before the Copilot run. Same loose-coupling pattern as - // `github_app_token_active`. - prepare_pr_base_active: front_matter.create_pr_config().is_some(), - azure_mcp_auth_active: front_matter.has_azure_authenticated_mcp_servers(), pr_trigger_for_synth, supply_chain: front_matter.supply_chain().cloned(), } diff --git a/src/compile/types.rs b/src/compile/types.rs index 08eb3afe4..219d8d4d4 100644 --- a/src/compile/types.rs +++ b/src/compile/types.rs @@ -1626,15 +1626,6 @@ impl FrontMatter { servers } - pub fn has_azure_authenticated_mcp_servers(&self) -> bool { - self.mcp_servers.values().any(|config| { - matches!( - config, - McpConfig::WithOptions(options) - if options.enabled.unwrap_or(true) && options.azure_auth.is_some() - ) - }) - } } /// Compile-time source for a remote reusable import. @@ -2115,14 +2106,7 @@ impl FrontMatter { /// Whether the workflow enables **any** safe-output tool. /// - /// Single source of truth for the safe-outputs-summary feature gate: it - /// drives BOTH the ado-script bundle download - /// (`AdoScriptExtension::safe_outputs_summary_active`, set in - /// `collect_extensions`) and the end-of-Agent-job render step emission - /// (`build_agent_job`). Both call sites MUST go through this so the bundle - /// is downloaded iff the step that runs it is emitted — a drift between two - /// independent copies of this predicate would make the step invoke a bundle - /// that was never downloaded. + /// Single source of truth for the end-of-Agent-job summary step. pub fn has_any_safe_output_tool(&self) -> bool { self.safe_output_tool_names().next().is_some() || !self.custom_safe_output_tool_names().is_empty() diff --git a/src/engine.rs b/src/engine.rs index b01cbfe43..92a9155d2 100644 --- a/src/engine.rs +++ b/src/engine.rs @@ -1,6 +1,7 @@ use std::collections::HashMap; use anyhow::Result; +use serde::Serialize; use crate::compile::extensions::Declarations; use crate::compile::shell::{Binding, ShellScript}; @@ -24,6 +25,9 @@ const BLOCKED_ARG_PREFIXES: &[&str] = &[ "--ask-user", ]; +/// Native Copilot CLI environment variable used for model selection. +pub const COPILOT_MODEL: &str = "COPILOT_MODEL"; + /// Environment variable keys that the compiler controls — users must not override these. pub const BLOCKED_ENV_KEYS: &[&str] = &[ "GITHUB_TOKEN", @@ -31,6 +35,10 @@ pub const BLOCKED_ENV_KEYS: &[&str] = &[ "COPILOT_OTEL_ENABLED", "COPILOT_OTEL_EXPORTER_TYPE", "COPILOT_OTEL_FILE_EXPORTER_PATH", + COPILOT_MODEL, + "ADO_AW_MODEL_AGENT_COPILOT", + "ADO_AW_MODEL_DETECTION_COPILOT", + "ADO_AW_DEFAULT_MODEL_COPILOT", // Shell/system vars that could affect AWF or pipeline behavior "PATH", "HOME", @@ -74,6 +82,65 @@ pub const COPILOT_PROVIDER_EXPR_ENV_KEYS: &[&str] = &[ /// select the provider subset for the Detection step. const COPILOT_PROVIDER_PREFIX: &str = "COPILOT_PROVIDER_"; +/// Runtime pipeline-variable override for the main Copilot agent model. +pub const ADO_AW_MODEL_AGENT_COPILOT: &str = "ADO_AW_MODEL_AGENT_COPILOT"; +/// Runtime pipeline-variable override for the Detection Copilot model. +pub const ADO_AW_MODEL_DETECTION_COPILOT: &str = "ADO_AW_MODEL_DETECTION_COPILOT"; +/// Shared runtime pipeline-variable fallback for Copilot models. +pub const ADO_AW_DEFAULT_MODEL_COPILOT: &str = "ADO_AW_DEFAULT_MODEL_COPILOT"; + +#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize)] +#[serde(rename_all = "lowercase")] +pub(crate) enum RuntimeModelRole { + Agent, + Detection, +} + +impl RuntimeModelRole { + fn specific_var(self) -> &'static str { + match self { + RuntimeModelRole::Agent => ADO_AW_MODEL_AGENT_COPILOT, + RuntimeModelRole::Detection => ADO_AW_MODEL_DETECTION_COPILOT, + } + } +} + +#[derive(Debug, Clone, Serialize)] +pub(crate) struct CopilotInvocationDocument { + pub(crate) schema_version: u32, + pub(crate) role: RuntimeModelRole, + pub(crate) command: String, + pub(crate) prompt_path: String, + pub(crate) mcp_config_path: Option, + pub(crate) args: Vec, + pub(crate) explicit_model: Option, + pub(crate) result_path: String, +} + +#[derive(Debug, Clone, Copy)] +pub(crate) struct CopilotInvocationContext<'a> { + role: RuntimeModelRole, + prompt_path: &'a str, + mcp_config_path: Option<&'a str>, + result_path: &'a str, +} + +impl<'a> CopilotInvocationContext<'a> { + pub(crate) const fn new( + role: RuntimeModelRole, + prompt_path: &'a str, + mcp_config_path: Option<&'a str>, + result_path: &'a str, + ) -> Self { + Self { + role, + prompt_path, + mcp_config_path, + result_path, + } + } +} + /// Returns true when `key` is an allowlisted BYOM/BYOK provider env-var key that /// may carry ADO macro/runtime expressions in `engine.env`. Case-sensitive: see /// [`COPILOT_PROVIDER_EXPR_ENV_KEYS`] for why exact case is required. @@ -281,30 +348,28 @@ pub fn get_engine(engine_id: &str) -> Result { } impl Engine { - /// The default engine binary name (e.g., "copilot"). - /// - /// Currently scaffolding — the pipeline templates hard-code the binary path - /// (`/tmp/awf-tools/copilot`). This will be wired into template substitution - /// when additional engines are added. Can be overridden per-agent via - /// `engine.command` in front matter. - #[allow(dead_code)] + /// Test-only legacy display of the default engine binary name. + #[cfg(test)] pub fn command(&self) -> &str { match self { Engine::Copilot => "copilot", } } - /// Generate CLI arguments for the engine invocation. + /// Test-only legacy display form of the Copilot CLI arguments. + #[cfg(test)] pub fn args( &self, front_matter: &FrontMatter, extension_declarations: &[Declarations], ) -> Result { - self.args_with_config( - &front_matter.engine, - front_matter, - extension_declarations, - ) + Ok(self + .args_with_config( + &front_matter.engine, + front_matter, + extension_declarations, + )? + .join(" ")) } /// Generate CLI arguments using an explicit engine configuration while @@ -314,7 +379,7 @@ impl Engine { engine_config: &EngineConfig, front_matter: &FrontMatter, extension_declarations: &[Declarations], - ) -> Result { + ) -> Result> { match self { Engine::Copilot => copilot_args(engine_config, front_matter, extension_declarations), } @@ -408,60 +473,37 @@ impl Engine { } } - /// Generate the full AWF `--` command string for running the engine. - /// - /// Returns the content for the AWF `-- ''` argument, including the - /// binary path, prompt delivery flag, MCP config flag, and all CLI arguments. - /// The engine controls how the prompt is provided (e.g., `--prompt="$(cat ...)"` - /// for Copilot) and how MCP config is referenced. - /// - /// `prompt_path` is the path to the prompt file inside the AWF container. - /// `mcp_config_path` is optionally the path to the MCP config file - /// (Some for Agent job, None for Detection job which has no MCP). - pub fn invocation( + pub(crate) fn invocation_document( &self, front_matter: &FrontMatter, extension_declarations: &[Declarations], - prompt_path: &str, - mcp_config_path: Option<&str>, - ) -> Result { - let args = self.args(front_matter, extension_declarations)?; - self.invocation_with_args( + invocation: CopilotInvocationContext<'_>, + ) -> Result { + self.invocation_document_with_config( &front_matter.engine, - prompt_path, - mcp_config_path, - &args, + front_matter, + extension_declarations, + invocation, ) } - /// Generate an invocation using an explicit engine configuration. - pub fn invocation_with_config( + pub(crate) fn invocation_document_with_config( &self, engine_config: &EngineConfig, front_matter: &FrontMatter, extension_declarations: &[Declarations], - prompt_path: &str, - mcp_config_path: Option<&str>, - ) -> Result { + invocation: CopilotInvocationContext<'_>, + ) -> Result { let args = self.args_with_config(engine_config, front_matter, extension_declarations)?; - self.invocation_with_args(engine_config, prompt_path, mcp_config_path, &args) - } - - fn invocation_with_args( - &self, - engine_config: &EngineConfig, - prompt_path: &str, - mcp_config_path: Option<&str>, - args: &str, - ) -> Result { match self { Engine::Copilot => { let command_path = match engine_config.command() { Some(cmd) => { if !is_valid_command_path(cmd) { anyhow::bail!( - "engine.command '{}' contains invalid characters. \ - Only ASCII alphanumerics, '.', '_', '/', and '-' are allowed.", + "engine.command '{}' is invalid. Use a bare executable name or an \ + absolute container path with ASCII alphanumerics, '.', '_', '/', \ + and '-' and no dot, empty, or traversal segments.", cmd ); } @@ -469,12 +511,19 @@ impl Engine { } None => "/tmp/awf-tools/copilot".to_string(), }; - Ok(copilot_invocation( - &command_path, - prompt_path, - mcp_config_path, + if let Some(model) = engine_config.model() { + validate_model_name(model)?; + } + Ok(CopilotInvocationDocument { + schema_version: 1, + role: invocation.role, + command: command_path, + prompt_path: invocation.prompt_path.to_string(), + mcp_config_path: invocation.mcp_config_path.map(str::to_string), args, - )) + explicit_model: engine_config.model().map(str::to_string), + result_path: invocation.result_path.to_string(), + }) } } } @@ -604,6 +653,13 @@ fn validate_user_arg(arg: &str) -> Result<()> { arg ); } + if arg == "--model" || arg.starts_with("--model=") { + anyhow::bail!( + "engine.args entry '{}' conflicts with compiler-controlled model selection. \ + Use engine.model or the ADO_AW_MODEL_*_COPILOT pipeline variables instead.", + arg + ); + } // Reject args that attempt to override compiler-controlled flags for blocked in BLOCKED_ARG_PREFIXES { if arg.starts_with(blocked) { @@ -622,7 +678,7 @@ fn copilot_args( engine_config: &EngineConfig, front_matter: &FrontMatter, extension_declarations: &[Declarations], -) -> Result { +) -> Result> { // Check if bash triggers --allow-all-tools. This happens when: // 1. Bash has an explicit wildcard entry (":*" or "*"), OR // 2. Bash is not specified at all (None) — ado-aw agents always run in AWF sandbox, @@ -657,21 +713,8 @@ fn copilot_args( let mut params = Vec::new(); - // Validate model name to prevent shell injection — copilot_params are embedded - // inside a single-quoted bash string in the AWF command. if let Some(model) = engine_config.model() { - if model.is_empty() - || !model - .chars() - .all(|c| c.is_ascii_alphanumeric() || matches!(c, '.' | '_' | ':' | '-')) - { - anyhow::bail!( - "Model name '{}' contains invalid characters. \ - Only ASCII alphanumerics, '.', '_', ':', and '-' are allowed.", - model - ); - } - params.push(format!("--model {}", model)); + validate_model_name(model)?; } if let Some(0) = engine_config.timeout_minutes() { eprintln!( @@ -691,7 +734,8 @@ fn copilot_args( agent ); } - params.push(format!("--agent {}", agent)); + params.push("--agent".to_string()); + params.push(agent.to_string()); } // Wire engine.api-target — sets the GHES/GHEC API endpoint hostname @@ -703,7 +747,8 @@ fn copilot_args( api_target ); } - params.push(format!("--api-target {}", api_target)); + params.push("--api-target".to_string()); + params.push(api_target.to_string()); } params.push("--disable-builtin-mcps".to_string()); @@ -714,13 +759,8 @@ fn copilot_args( } for tool in allowed_tools { - if tool.contains('(') || tool.contains(')') || tool.contains(' ') { - // Use double quotes - the copilot_params are embedded inside a single-quoted - // bash string in the AWF command, so single quotes would break quoting. - params.push(format!("--allow-tool \"{}\"", tool)); - } else { - params.push(format!("--allow-tool {}", tool)); - } + params.push("--allow-tool".to_string()); + params.push(tool); } // --allow-all-paths when edit is enabled — lets the agent write to any file path. @@ -730,14 +770,31 @@ fn copilot_args( } // Wire engine.args — append user-provided CLI arguments after compiler-generated args. - // User args are additive; they cannot remove compiler security flags but may override - // non-security defaults via last-wins semantics (e.g., --model). + // User args are additive and cannot override compiler-controlled flags or model selection. for arg in engine_config.args() { validate_user_arg(arg)?; params.push(arg.to_string()); } - Ok(params.join(" ")) + Ok(params) +} + +fn validate_model_name(model: &str) -> Result<()> { + // Validate model names before they become invocation-document values or + // runtime-selected COPILOT_MODEL values. Keep this character set in sync + // with the Copilot invoker. + if model.is_empty() + || !model + .chars() + .all(|c| c.is_ascii_alphanumeric() || matches!(c, '.' | '_' | ':' | '-')) + { + anyhow::bail!( + "Model name '{}' contains invalid characters. \ + Only ASCII alphanumerics, '.', '_', ':', and '-' are allowed.", + model + ); + } + Ok(()) } /// The masked, same-job pipeline variable the `github-app-token` ado-script @@ -835,6 +892,11 @@ fn copilot_env(engine_config: &EngineConfig) -> Result { "COPILOT_OTEL_EXPORTER_TYPE: \"file\"".to_string(), "COPILOT_OTEL_FILE_EXPORTER_PATH: \"/tmp/awf-tools/staging/otel.jsonl\"".to_string(), ]; + if let Some(model) = engine_config.model() { + validate_model_name(model)?; + } else { + add_runtime_model_env_lines(&mut lines, RuntimeModelRole::Agent); + } // Wire engine.env — merge user-provided environment variables plus any // `COPILOT_PROVIDER_*` vars derived from an `engine.provider` block. @@ -851,6 +913,23 @@ fn copilot_env(engine_config: &EngineConfig) -> Result { Ok(lines.join("\n")) } +fn add_runtime_model_env_lines(lines: &mut Vec, role: RuntimeModelRole) { + let specific = role.specific_var(); + lines.push(format!("{specific}: $({specific})")); + lines.push(format!( + "{ADO_AW_DEFAULT_MODEL_COPILOT}: $({ADO_AW_DEFAULT_MODEL_COPILOT})" + )); +} + +fn add_runtime_model_env_pairs(pairs: &mut Vec<(String, String)>, role: RuntimeModelRole) { + let specific = role.specific_var(); + pairs.push((specific.to_string(), format!("$({specific})"))); + pairs.push(( + ADO_AW_DEFAULT_MODEL_COPILOT.to_string(), + format!("$({ADO_AW_DEFAULT_MODEL_COPILOT})"), + )); +} + /// Return the `COPILOT_PROVIDER_*` entries of `engine.env` as validated, /// **raw** `(key, value)` pairs (value un-rendered; empty vec when none present), /// sorted by key. @@ -858,9 +937,9 @@ fn copilot_env(engine_config: &EngineConfig) -> Result { /// Used by the Detection (threat-analysis) step so the detection Copilot run /// inherits the same BYOM/BYOK provider routing (and credential isolation) as /// the main agent. Mirrors gh-aw, whose detection engine config inherits the -/// main engine's `Env` (`threat_detection_inline_engine.go`). The main model is -/// already threaded via the `--model` flag on the detection invocation, so only -/// the provider routing/credential keys are needed here. +/// main engine's `Env` (`threat_detection_inline_engine.go`). Model delivery is +/// handled separately through compiler-owned `COPILOT_MODEL`, so only provider +/// routing/credential keys are selected here. /// /// Returning raw pairs (rather than a rendered YAML string) lets the call site /// build typed `EnvValue`s directly — no render-to-YAML-then-reparse round-trip, @@ -928,6 +1007,11 @@ pub fn copilot_detection_env(engine_config: &EngineConfig) -> Result, - args: &str, -) -> String { - let mut parts = vec![ - command_path.to_string(), - format!("--prompt=\"$(cat {prompt_path})\""), - ]; - - if let Some(mcp_path) = mcp_config_path { - parts.push(format!("--additional-mcp-config @{mcp_path}")); - } - - if !args.is_empty() { - parts.push(args.to_string()); - } - - parts.join(" ") -} - #[cfg(test)] mod tests { use super::{ - Engine, GITHUB_APP_TOKEN_VAR, copilot_byom_active, copilot_byom_credential_keys, + ADO_AW_DEFAULT_MODEL_COPILOT, ADO_AW_MODEL_AGENT_COPILOT, ADO_AW_MODEL_DETECTION_COPILOT, + COPILOT_MODEL, CopilotInvocationContext, Engine, GITHUB_APP_TOKEN_VAR, RuntimeModelRole, + copilot_byom_active, copilot_byom_credential_keys, copilot_detection_env, copilot_provider_env, get_engine, github_app_token_secrecy_advisory, github_token_source_var, normalize_version_tag, validate_engine_feature_support, }; @@ -1391,32 +1452,143 @@ mod tests { } #[test] - fn copilot_engine_command() { - assert_eq!(Engine::Copilot.command(), "copilot"); + fn copilot_engine_with_explicit_model() { + let (front_matter, _) = parse_markdown( + "---\nname: test\ndescription: test\nengine:\n id: copilot\n model: gpt-5\n---\n", + ) + .unwrap(); + let params = Engine::Copilot + .args(&front_matter, &declarations_for(&front_matter)) + .unwrap(); + assert!(!params.contains("--model")); + let env = Engine::Copilot.env(&front_matter.engine).unwrap(); + assert!(!env.contains(COPILOT_MODEL), "{env}"); + let invocation = Engine::Copilot + .invocation_document( + &front_matter, + &declarations_for(&front_matter), + CopilotInvocationContext::new( + RuntimeModelRole::Agent, + "/tmp/prompt.md", + None, + "/tmp/result.json", + ), + ) + .unwrap(); + assert_eq!(invocation.explicit_model.as_deref(), Some("gpt-5")); } #[test] - fn copilot_engine_args() { + fn copilot_invocation_document_defers_runtime_model_resolution() { let (front_matter, _) = parse_markdown("---\nname: test\ndescription: test\n---\n").unwrap(); - let params = Engine::Copilot - .args(&front_matter, &declarations_for(&front_matter)) + let invocation = Engine::Copilot + .invocation_document( + &front_matter, + &declarations_for(&front_matter), + CopilotInvocationContext::new( + RuntimeModelRole::Agent, + "/tmp/prompt.md", + Some("/tmp/mcp.json"), + "/tmp/result.json", + ), + ) .unwrap(); - // Default engine (copilot) lets the Copilot CLI choose its default model. - assert!(!params.contains("--model ")); - assert!(params.contains("--disable-builtin-mcps")); + + assert_eq!(invocation.role, RuntimeModelRole::Agent); + assert_eq!(invocation.explicit_model, None); + assert_eq!( + invocation.mcp_config_path.as_deref(), + Some("/tmp/mcp.json") + ); + assert!(!invocation.args.iter().any(|arg| arg.starts_with("--model"))); } #[test] - fn copilot_engine_with_explicit_model() { + fn copilot_invocation_document_keeps_explicit_model_static() { let (front_matter, _) = parse_markdown( "---\nname: test\ndescription: test\nengine:\n id: copilot\n model: gpt-5\n---\n", ) .unwrap(); - let params = Engine::Copilot - .args(&front_matter, &declarations_for(&front_matter)) + let invocation = Engine::Copilot + .invocation_document( + &front_matter, + &declarations_for(&front_matter), + CopilotInvocationContext::new( + RuntimeModelRole::Agent, + "/tmp/prompt.md", + None, + "/tmp/result.json", + ), + ) .unwrap(); - assert!(params.contains("--model gpt-5")); + + assert_eq!(invocation.explicit_model.as_deref(), Some("gpt-5")); + assert!(!invocation.args.iter().any(|arg| arg.starts_with("--model"))); + } + + #[test] + fn copilot_detection_invocation_document_uses_independent_runtime_model() { + let (front_matter, _) = + parse_markdown("---\nname: test\ndescription: test\n---\n").unwrap(); + let invocation = Engine::Copilot + .invocation_document_with_config( + &front_matter.engine, + &front_matter, + &declarations_for(&front_matter), + CopilotInvocationContext::new( + RuntimeModelRole::Detection, + "/tmp/threat.md", + None, + "/tmp/result.json", + ), + ) + .unwrap(); + + assert_eq!(invocation.role, RuntimeModelRole::Detection); + assert_eq!(invocation.explicit_model, None); + assert!(!invocation.args.iter().any(|arg| arg.starts_with("--model"))); + let env = copilot_detection_env(&front_matter.engine).unwrap(); + assert!( + env.iter() + .any(|(key, _)| key == ADO_AW_MODEL_DETECTION_COPILOT) + ); + assert!( + env.iter() + .any(|(key, _)| key == ADO_AW_DEFAULT_MODEL_COPILOT) + ); + } + + #[test] + fn engine_args_reject_model_flag() { + for args in ["[--model, gpt-5]", "[--model=gpt-5]"] { + let source = format!( + "---\nname: test\ndescription: test\nengine:\n id: copilot\n args: {args}\n---\n" + ); + let (front_matter, _) = parse_markdown(&source).unwrap(); + let error = Engine::Copilot + .args(&front_matter, &declarations_for(&front_matter)) + .unwrap_err() + .to_string(); + assert!( + error.contains("compiler-controlled model selection"), + "{error}" + ); + } + } + + #[test] + fn engine_env_rejects_raw_copilot_model() { + let (front_matter, _) = parse_markdown( + "---\nname: test\ndescription: test\nengine:\n id: copilot\n env:\n COPILOT_MODEL: gpt-5\n---\n", + ) + .unwrap(); + let error = Engine::Copilot + .env(&front_matter.engine) + .unwrap_err() + .to_string(); + assert!(error.contains(COPILOT_MODEL), "{error}"); + assert!(error.contains("compiler-controlled"), "{error}"); } #[test] @@ -1445,6 +1617,78 @@ mod tests { assert!(!env.contains("AZURE_DEVOPS_EXT_PAT")); } + #[test] + fn copilot_engine_env_maps_runtime_agent_model_vars_when_no_explicit_model() { + let (front_matter, _) = + parse_markdown("---\nname: test\ndescription: test\n---\n").unwrap(); + let env = Engine::Copilot.env(&front_matter.engine).unwrap(); + + assert!(env.contains(&format!( + "{ADO_AW_MODEL_AGENT_COPILOT}: $({ADO_AW_MODEL_AGENT_COPILOT})" + ))); + assert!(env.contains(&format!( + "{ADO_AW_DEFAULT_MODEL_COPILOT}: $({ADO_AW_DEFAULT_MODEL_COPILOT})" + ))); + } + + #[test] + fn copilot_engine_env_omits_runtime_agent_model_vars_for_explicit_model() { + let (front_matter, _) = parse_markdown( + "---\nname: test\ndescription: test\nengine:\n id: copilot\n model: gpt-5\n---\n", + ) + .unwrap(); + let env = Engine::Copilot.env(&front_matter.engine).unwrap(); + + assert!(!env.contains(ADO_AW_MODEL_AGENT_COPILOT)); + assert!(!env.contains(ADO_AW_DEFAULT_MODEL_COPILOT)); + } + + #[test] + fn copilot_detection_env_maps_independent_runtime_model_vars() { + let (front_matter, _) = + parse_markdown("---\nname: test\ndescription: test\n---\n").unwrap(); + let env = copilot_detection_env(&front_matter.engine).unwrap(); + + assert!(env.contains(&( + ADO_AW_MODEL_DETECTION_COPILOT.to_string(), + format!("$({ADO_AW_MODEL_DETECTION_COPILOT})") + ))); + assert!(env.contains(&( + ADO_AW_DEFAULT_MODEL_COPILOT.to_string(), + format!("$({ADO_AW_DEFAULT_MODEL_COPILOT})") + ))); + assert!(!env.iter().any(|(key, _)| key == ADO_AW_MODEL_AGENT_COPILOT)); + } + + #[test] + fn copilot_detection_env_omits_runtime_model_vars_for_explicit_model() { + let (front_matter, _) = parse_markdown( + "---\nname: test\ndescription: test\nsafe-outputs:\n threat-detection:\n engine:\n model: detector-model\n---\n", + ) + .unwrap(); + let threat_detection = front_matter.threat_detection_config().unwrap(); + let detection_engine = front_matter.effective_detection_engine(&threat_detection); + let env = copilot_detection_env(&detection_engine).unwrap(); + + assert!(!env.iter().any(|(key, _)| key == ADO_AW_MODEL_DETECTION_COPILOT)); + assert!(!env.iter().any(|(key, _)| key == ADO_AW_DEFAULT_MODEL_COPILOT)); + } + + #[test] + fn copilot_engine_env_rejects_user_runtime_model_var_override() { + let (front_matter, _) = parse_markdown( + "---\nname: test\ndescription: test\nengine:\n id: copilot\n env:\n ADO_AW_MODEL_AGENT_COPILOT: gpt-5\n---\n", + ) + .unwrap(); + let err = Engine::Copilot + .env(&front_matter.engine) + .unwrap_err() + .to_string(); + + assert!(err.contains("compiler-controlled environment variable")); + assert!(err.contains(ADO_AW_MODEL_AGENT_COPILOT)); + } + #[test] fn copilot_engine_env_sources_github_token_from_app_token_var_when_configured() { let src = "---\nname: test\ndescription: test\nengine:\n id: copilot\n \ @@ -1542,29 +1786,36 @@ mod tests { "---\nname: test\ndescription: test\nengine:\n id: copilot\n command: /usr/local/bin/my-copilot\n---\n", ).unwrap(); let result = Engine::Copilot - .invocation( + .invocation_document( &fm, &declarations_for(&fm), - "/tmp/prompt.md", - Some("/tmp/mcp.json"), + CopilotInvocationContext::new( + RuntimeModelRole::Agent, + "/tmp/prompt.md", + Some("/tmp/mcp.json"), + "/tmp/result.json", + ), ) .unwrap(); - assert!(result.starts_with("/usr/local/bin/my-copilot ")); - assert!(!result.contains("/tmp/awf-tools/copilot")); + assert_eq!(result.command, "/usr/local/bin/my-copilot"); } #[test] fn engine_command_default_uses_awf_path() { let (fm, _) = parse_markdown("---\nname: test\ndescription: test\n---\n").unwrap(); let result = Engine::Copilot - .invocation( + .invocation_document( &fm, &declarations_for(&fm), - "/tmp/prompt.md", - Some("/tmp/mcp.json"), + CopilotInvocationContext::new( + RuntimeModelRole::Agent, + "/tmp/prompt.md", + Some("/tmp/mcp.json"), + "/tmp/result.json", + ), ) .unwrap(); - assert!(result.starts_with("/tmp/awf-tools/copilot ")); + assert_eq!(result.command, "/tmp/awf-tools/copilot"); } #[test] @@ -1572,14 +1823,22 @@ mod tests { let (fm, _) = parse_markdown( "---\nname: test\ndescription: test\nengine:\n id: copilot\n command: \"/tmp/copilot; rm -rf /\"\n---\n", ).unwrap(); - let result = - Engine::Copilot.invocation(&fm, &declarations_for(&fm), "/tmp/prompt.md", None); + let result = Engine::Copilot.invocation_document( + &fm, + &declarations_for(&fm), + CopilotInvocationContext::new( + RuntimeModelRole::Agent, + "/tmp/prompt.md", + None, + "/tmp/result.json", + ), + ); assert!(result.is_err()); assert!( result .unwrap_err() .to_string() - .contains("invalid characters") + .contains("is invalid") ); } @@ -1588,8 +1847,16 @@ mod tests { let (fm, _) = parse_markdown( "---\nname: test\ndescription: test\nengine:\n id: copilot\n command: \"/tmp/co'pilot\"\n---\n", ).unwrap(); - let result = - Engine::Copilot.invocation(&fm, &declarations_for(&fm), "/tmp/prompt.md", None); + let result = Engine::Copilot.invocation_document( + &fm, + &declarations_for(&fm), + CopilotInvocationContext::new( + RuntimeModelRole::Agent, + "/tmp/prompt.md", + None, + "/tmp/result.json", + ), + ); assert!(result.is_err()); } diff --git a/src/validate.rs b/src/validate.rs index a12fb5c7e..a3059478e 100644 --- a/src/validate.rs +++ b/src/validate.rs @@ -42,12 +42,24 @@ pub fn is_safe_path_segment(s: &str) -> bool { .all(|c| c.is_ascii_alphanumeric() || matches!(c, '-' | '_' | '.')) } -/// Characters allowed in engine.command paths (absolute path chars only). -/// Prevents shell injection when the path is embedded in AWF single-quoted commands. +/// Validate an engine command as either a bare executable name or an absolute +/// container path with no empty, dot, or traversal segments. pub fn is_valid_command_path(s: &str) -> bool { - !s.is_empty() - && s.chars() + if s.is_empty() + || !s + .chars() .all(|c| c.is_ascii_alphanumeric() || matches!(c, '.' | '_' | '/' | '-')) + { + return false; + } + if !s.contains('/') { + return s != "." && s != ".."; + } + s.starts_with('/') + && !s.ends_with('/') + && s[1..] + .split('/') + .all(|segment| !segment.is_empty() && segment != "." && segment != "..") } /// Characters allowed in engine.agent and engine.model identifiers. @@ -1003,6 +1015,12 @@ mod tests { assert!(is_valid_command_path("/tmp/awf-tools/copilot")); assert!(is_valid_command_path("copilot")); assert!(is_valid_command_path("/usr/local/bin/my-tool_v2")); + assert!(!is_valid_command_path("bin/copilot")); + assert!(!is_valid_command_path(".")); + assert!(!is_valid_command_path("..")); + assert!(!is_valid_command_path("/tmp/../copilot")); + assert!(!is_valid_command_path("/tmp//copilot")); + assert!(!is_valid_command_path("/tmp/copilot/")); assert!(!is_valid_command_path("")); assert!(!is_valid_command_path("/tmp/copilot; rm -rf /")); assert!(!is_valid_command_path("/tmp/copilot'")); diff --git a/tests/awf-copilot-safeoutputs/run.sh b/tests/awf-copilot-safeoutputs/run.sh index b25c15a08..5710a6937 100644 --- a/tests/awf-copilot-safeoutputs/run.sh +++ b/tests/awf-copilot-safeoutputs/run.sh @@ -6,6 +6,7 @@ umask 077 : "${ADO_AW_BIN:?ADO_AW_BIN is required}" : "${AWF_BIN:?AWF_BIN is required}" : "${COPILOT_BIN:?COPILOT_BIN is required}" +: "${COPILOT_INVOKER_BUNDLE:?COPILOT_INVOKER_BUNDLE is required}" : "${AWF_VERSION:?AWF_VERSION is required}" : "${MCPG_VERSION:?MCPG_VERSION is required}" : "${ADO_AW_COPILOT_CLI_ARTIFACT_DIR:?ADO_AW_COPILOT_CLI_ARTIFACT_DIR is required}" @@ -71,6 +72,10 @@ for binary in "${ADO_AW_BIN}" "${AWF_BIN}" "${COPILOT_BIN}"; do exit 1 } done +[[ -f "${COPILOT_INVOKER_BUNDLE}" ]] || { + echo "Copilot invoker bundle is missing: ${COPILOT_INVOKER_BUNDLE}" >&2 + exit 1 +} MCP_GATEWAY_API_KEY="$(openssl rand -base64 45 | tr -d '/+=')" install -m 0755 "${ADO_AW_BIN}" "${TOOLS_DIR}/ado-aw" @@ -223,15 +228,41 @@ jq \ chmod 600 "${TOOLS_DIR}/mcp-config.json" install -m 0755 "${COPILOT_BIN}" "${TOOLS_DIR}/copilot" +mkdir -p /tmp/ado-aw-scripts/ado-script +install -m 0644 "${COPILOT_INVOKER_BUNDLE}" \ + /tmp/ado-aw-scripts/ado-script/copilot-invoker.js cat >"${TOOLS_DIR}/agent-prompt.md" <"${TOOLS_DIR}/copilot-invocation.json" +chmod 600 "${TOOLS_DIR}/copilot-invocation.json" readonly ALLOWED_DOMAINS="api.business.githubcopilot.com,api.enterprise.githubcopilot.com,api.github.com,api.githubcopilot.com,api.individual.githubcopilot.com,config.edge.skype.com,copilot-proxy.githubusercontent.com,github.com,telemetry.enterprise.githubcopilot.com,*.copilot.github.com,*.githubcopilot.com" # shellcheck disable=SC2016 # AWF expands the engine command inside the sandbox. -readonly ENGINE_RUN='export NO_PROXY="${NO_PROXY:+$NO_PROXY,}awmg-mcpg"; export no_proxy="$NO_PROXY"; /tmp/awf-tools/copilot --prompt="$(cat /tmp/awf-tools/agent-prompt.md)" --additional-mcp-config @/tmp/awf-tools/mcp-config.json --model gpt-5-mini --disable-builtin-mcps --no-ask-user --allow-all-tools --allow-tool safeoutputs --allow-all-paths' +readonly ENGINE_RUN='export NO_PROXY="${NO_PROXY:+$NO_PROXY,}awmg-mcpg"; export no_proxy="$NO_PROXY"; exec node /tmp/ado-aw-scripts/ado-script/copilot-invoker.js run /tmp/awf-tools/copilot-invocation.json' set +e "${AWF_BIN}" \ @@ -254,6 +285,15 @@ if [[ "${AWF_STATUS}" -ne 0 ]]; then exit "${AWF_STATUS}" fi +REQUESTED_MODEL="$( + node /tmp/ado-aw-scripts/ado-script/copilot-invoker.js \ + read-result "${TOOLS_DIR}/copilot-invocation-result.json" agent +)" +[[ "${REQUESTED_MODEL}" == "gpt-5-mini" ]] || { + echo "Unexpected requested model from invoker: ${REQUESTED_MODEL}" >&2 + exit 1 +} + NDJSON_PATH="${SAFE_OUTPUTS_DIR}/safe_outputs.ndjson" for _ in $(seq 1 30); do [[ -s "${NDJSON_PATH}" ]] && break diff --git a/tests/compiler_tests.rs b/tests/compiler_tests.rs index 179337dd7..b5080cf73 100644 --- a/tests/compiler_tests.rs +++ b/tests/compiler_tests.rs @@ -2049,33 +2049,31 @@ Call the noop tool exactly once. let detection = extract_job_block(&compiled, "Detection").expect("Detection job should exist"); assert!( - agent.contains( - "/tmp/awf-tools/copilot --prompt=\"$(cat /tmp/awf-tools/agent-prompt.md)\" \ - --additional-mcp-config @/tmp/awf-tools/mcp-config.json" - ), - "agent job should pass compiler-emitted MCP config to Copilot CLI: {agent}" + agent.contains(r#""prompt_path":"/tmp/awf-tools/agent-prompt.md""#) + && agent.contains(r#""mcp_config_path":"/tmp/awf-tools/mcp-config.json""#), + "agent invocation document should reference the prompt and MCP config: {agent}" ); assert!( - !agent.contains("--prompt \"$(cat "), - "agent job should not pass prompt as a separate option value: {agent}" + agent.contains("copilot-invoker.js run") + && !agent.contains("/tmp/awf-tools/copilot --prompt"), + "agent job should execute only the fixed invoker command: {agent}" ); assert!( - detection.contains( - "/tmp/awf-tools/copilot --prompt=\"$(cat \ - /tmp/awf-tools/threat-analysis-prompt.md)\"" - ), - "detection job should pass prompt using attached form: {detection}" + detection.contains(r#""prompt_path":"/tmp/awf-tools/threat-analysis-prompt.md""#), + "detection invocation document should reference the threat prompt: {detection}" ); assert!( - !detection.contains("--prompt \"$(cat "), - "detection job should not pass prompt as a separate option value: {detection}" + detection.contains("copilot-invoker.js run") + && !detection.contains("/tmp/awf-tools/copilot --prompt"), + "detection job should execute only the fixed invoker command: {detection}" ); assert!( agent.contains("--allow-all-tools"), "default unrestricted tools path should emit --allow-all-tools: {agent}" ); assert!( - !detection.contains("--additional-mcp-config"), + detection.contains(r#""mcp_config_path":null"#) + && !detection.contains(r#""mcp_config_path":"/tmp/awf-tools/mcp-config.json""#), "detection job should not receive the SafeOutputs MCP config: {detection}" ); assert!( @@ -2119,19 +2117,17 @@ fn test_runtime_import_frontmatter_prompt_uses_attached_copilot_prompt_flag() { "agent prompt should runtime-import the full markdown fixture: {compiled}" ); assert!( - agent.contains("/tmp/awf-tools/copilot --prompt=\"$(cat /tmp/awf-tools/agent-prompt.md)\""), - "agent job should pass prompt using attached form: {agent}" + agent.contains(r#""prompt_path":"/tmp/awf-tools/agent-prompt.md""#), + "agent invocation document should reference the resolved prompt file: {agent}" ); assert!( - detection.contains( - "/tmp/awf-tools/copilot --prompt=\"$(cat \ - /tmp/awf-tools/threat-analysis-prompt.md)\"" - ), - "detection job should pass prompt using attached form: {detection}" + detection.contains(r#""prompt_path":"/tmp/awf-tools/threat-analysis-prompt.md""#), + "detection invocation document should reference the threat prompt: {detection}" ); assert!( - !compiled.contains("--prompt \"$(cat "), - "compiled pipeline should not pass prompt as a separate option value: {compiled}" + compiled.contains("copilot-invoker.js run") + && !compiled.contains("/tmp/awf-tools/copilot --prompt"), + "compiled pipeline should use the fixed invoker command: {compiled}" ); exercise_attached_prompt_with_pinned_copilot_cli(&fixture); @@ -2171,19 +2167,19 @@ Call the noop tool exactly once. "restricted bash path should not emit --allow-all-tools: {agent}" ); assert!( - agent.contains("--allow-tool safeoutputs"), + agent.contains(r#""--allow-tool","safeoutputs""#), "restricted bash path must explicitly allow the SafeOutputs MCP server: {agent}" ); assert!( - agent.contains("--allow-tool \"shell(echo)\""), + agent.contains(r#""--allow-tool","shell(echo)""#), "restricted bash path must emit the configured bash allowlist: {agent}" ); assert!( - agent.contains("--agent my-custom-agent"), + agent.contains(r#""--agent","my-custom-agent""#), "engine.agent should flow through the compiled Copilot CLI invocation: {agent}" ); assert!( - agent.contains("--api-target api.example.com"), + agent.contains(r#""--api-target","api.example.com""#), "engine.api-target should flow through the compiled Copilot CLI invocation: {agent}" ); assert!( @@ -2191,7 +2187,7 @@ Call the noop tool exactly once. "engine.args should append additive Copilot CLI arguments: {agent}" ); assert!( - agent.contains("--additional-mcp-config @/tmp/awf-tools/mcp-config.json"), + agent.contains(r#""mcp_config_path":"/tmp/awf-tools/mcp-config.json""#), "restricted tools path should still use the compiler-emitted MCP config: {agent}" ); } @@ -2279,7 +2275,7 @@ fn permissions_read_enables_proxy_and_wrapped_az_without_mcp() { "displayName: Install az wrapper (ado-proxy)", "displayName: Detect Azure CLI on host (for AWF mount)", "--topology-attach \"awmg-ado-proxy\"", - "--allow-tool \"shell(az)\"", + r#""--allow-tool","shell(az)""#, ] { assert!( compiled.contains(required), @@ -2490,7 +2486,7 @@ fn test_fixture_azure_devops_mcp_compiled_output() { "MCPG config should have entrypointArgs field" ); assert!( - !compiled.contains("\"command\""), + !compiled.contains("\"command\": "), "MCPG config should NOT use command field" ); @@ -2604,7 +2600,7 @@ fn test_mcpg_config_container_based_mcp() { assert!(compiled.contains("/host/data:/app/data:ro")); assert!(compiled.contains("\"API_KEY\": \"test-key\"")); assert!(compiled.contains("\"tool_a\"")); - assert!(!compiled.contains("\"command\"")); + assert!(!compiled.contains("\"command\": ")); let _ = fs::remove_dir_all(&temp_dir); } @@ -2728,7 +2724,7 @@ fn test_mcpg_config_http_based_mcp() { assert!(compiled.contains("\"url\": \"https://mcp.dev.azure.com/myorg\"")); assert!(compiled.contains("\"X-MCP-Toolsets\": \"repos,wit\"")); assert!(compiled.contains("\"wit_get_work_item\"")); - assert!(!compiled.contains("\"command\"")); + assert!(!compiled.contains("\"command\": ")); let _ = fs::remove_dir_all(&temp_dir); } @@ -5310,8 +5306,8 @@ fn test_1es_compiled_output_is_valid_yaml() { "1ES output should contain SafeOutputs references" ); assert!( - compiled.contains("copilot --prompt="), - "1ES output should contain copilot invocation (engine_run substituted)" + compiled.contains("copilot-invoker.js run"), + "1ES output should contain the fixed Copilot invoker command" ); assert!( compiled.contains("threat-analysis"), @@ -5849,11 +5845,10 @@ fn extract_job_block<'a>(yaml: &'a str, name: &str) -> Option<&'a str> { Some(&yaml[start..end]) } -/// Per-job download placement: gate-only pipeline must put the download in -/// Setup and NOT in Agent. ADO jobs run on isolated VMs, so the gate's -/// install/download has to land in the same job as the gate step. +/// Gate-only pipelines stage the bundle in Setup for the gate and in both +/// Copilot jobs for the invoker. #[test] -fn test_gate_only_pipeline_downloads_bundle_in_setup_job_not_agent() { +fn test_gate_only_pipeline_downloads_bundle_in_all_consuming_jobs() { let yaml = compile_fixture("dedupe_gate_only.md"); let setup = extract_job_block(&yaml, "Setup").expect("Setup job should exist"); let agent = extract_job_block(&yaml, "Agent").expect("Agent job should exist"); @@ -5862,9 +5857,8 @@ fn test_gate_only_pipeline_downloads_bundle_in_setup_job_not_agent() { "Setup job is missing the script bundle download (gate consumer lives here)" ); assert!( - !agent.contains("Download ado-aw scripts"), - "Agent job should NOT have the script bundle download (gate-only, no runtime imports). \ - Agent block contents: {}", + agent.contains("Download ado-aw scripts"), + "Agent job must stage the Copilot invoker bundle. Agent block contents: {}", agent ); } @@ -5890,9 +5884,8 @@ fn test_imports_only_pipeline_downloads_bundle_in_agent_job_not_setup() { } } -/// Per-job download placement: when both gate and runtime imports are active, -/// the bundle is downloaded twice — once per consuming job. ADO's VM -/// isolation makes this correct architecture, not duplication waste. +/// When both gate and runtime imports are active, each isolated consuming job +/// stages the bundle: Setup, Agent, and Detection. #[test] fn test_both_features_active_downloads_bundle_in_both_jobs() { let yaml = compile_fixture("dedupe_both.md"); @@ -5908,23 +5901,19 @@ fn test_both_features_active_downloads_bundle_in_both_jobs() { ); assert_eq!( yaml.matches("Download ado-aw scripts").count(), - 2, - "Expected exactly two downloads — one per consuming job (Setup + Agent)" + 3, + "Expected exactly three downloads — Setup, Agent, and Detection" ); } -/// Per-job download placement: with neither gate nor runtime imports active, -/// no Node install or script-bundle download should appear anywhere. +/// Even with no gate or runtime imports, Agent and Detection stage the invoker. #[test] -fn test_neither_feature_active_emits_no_node_or_download_anywhere() { +fn test_neither_feature_active_stages_invoker_in_copilot_jobs() { let yaml = compile_fixture("dedupe_neither.md"); - assert!( - !yaml.contains("UseNode@1"), - "No UseNode@1 expected when neither gate nor runtime imports are active" - ); - assert!( - !yaml.contains("Download ado-aw scripts"), - "No script bundle download expected when neither gate nor runtime imports are active" + assert_eq!( + yaml.matches("Download ado-aw scripts").count(), + 2, + "Agent and Detection must each stage the invoker bundle" ); } @@ -5993,12 +5982,11 @@ fn test_node_runtime_install_orders_after_ado_script_so_user_version_wins() { ado-script idx = {ado_script_install_idx}, user idx = {user_runtime_install_idx}" ); - // Both downloads of ado-script.zip remain unaffected (still exactly one - // in the Agent job in this fixture — no filters, so no Setup-side download). + // Agent and Detection each stage ado-script.zip; no Setup download exists. assert_eq!( yaml.matches("Download ado-aw scripts").count(), - 1, - "Expected exactly one ado-script.zip download (Agent job only; no gate active)" + 2, + "Expected exactly two ado-script.zip downloads (Agent + Detection)" ); } @@ -8175,12 +8163,23 @@ safe-outputs: let agent = job_block(&compiled, "Agent"); let detection = job_block(&compiled, "Detection"); - assert!(agent.contains("--model agent-model"), "{agent}"); + assert!( + agent.contains(r#""explicit_model":"agent-model""#), + "{agent}" + ); assert!(agent.contains("--reasoning-effort=high"), "{agent}"); - assert!(!agent.contains("--model detection-model"), "{agent}"); + assert!(!agent.contains("--model"), "{agent}"); + assert!( + !agent.contains(r#""explicit_model":"detection-model""#), + "{agent}" + ); assert!(!agent.contains("DETECTION_ENV"), "{agent}"); - assert!(detection.contains("--model detection-model"), "{detection}"); + assert!( + detection.contains(r#""explicit_model":"detection-model""#), + "{detection}" + ); + assert!(!detection.contains("--model"), "{detection}"); assert!(detection.contains("--reasoning-effort=low"), "{detection}"); assert!( !detection.contains("--reasoning-effort=high"), @@ -8204,6 +8203,86 @@ safe-outputs: assert!(detection.contains("2.0.2"), "{detection}"); } +#[test] +fn runtime_model_controls_compile_across_all_targets() { + for target in ["standalone", "1es", "job", "stage"] { + let target_field = if target == "standalone" { + String::new() + } else { + format!("target: {target}\n") + }; + let source = format!( + "---\nname: Runtime Model {target}\ndescription: Runtime model target coverage\n\ + {target_field}safe-outputs:\n noop: {{}}\n threat-detection: true\n---\n\n## Agent\n" + ); + let (ok, compiled, stderr) = + compile_inline_source(&format!("runtime-model-{target}"), &source); + assert!(ok, "{target} should compile: {stderr}"); + assert!( + compiled.contains( + "ADO_AW_MODEL_AGENT_COPILOT: $(ADO_AW_MODEL_AGENT_COPILOT)" + ), + "{target}: missing Agent runtime model env mapping" + ); + assert!( + compiled.contains( + "ADO_AW_MODEL_DETECTION_COPILOT: $(ADO_AW_MODEL_DETECTION_COPILOT)" + ), + "{target}: missing Detection runtime model env mapping" + ); + assert!( + compiled.contains("copilot-invoker.js run"), + "{target}: fixed Copilot invoker command must be emitted" + ); + assert!( + compiled.contains(r#""role":"agent""#) + && compiled.contains(r#""role":"detection""#), + "{target}: Agent and Detection invocation documents must be emitted" + ); + assert!( + !compiled.contains("ADO_AW_EFFECTIVE_MODEL"), + "{target}: runtime model shell resolver must be absent" + ); + assert!( + !compiled.contains("--model"), + "{target}: compiler-generated model flags must be absent" + ); + } +} + +#[test] +fn runtime_model_control_resolves_after_prior_agent_step() { + let source = r###"--- +name: Runtime Model Set Variable +description: Runtime model task-scope resolution +steps: + - bash: | + echo "##vso[task.setvariable variable=ADO_AW_MODEL_AGENT_COPILOT]gpt-runtime" +safe-outputs: + threat-detection: false +--- + +## Agent +"###; + let (ok, compiled, stderr) = compile_inline_source("runtime-model-set-variable", source); + assert!(ok, "pipeline should compile: {stderr}"); + let agent = job_block(&compiled, "Agent"); + let producer = agent + .find("task.setvariable variable=ADO_AW_MODEL_AGENT_COPILOT") + .expect("model variable producer should be emitted"); + let consumer = agent + .find("Run copilot (AWF network isolated)") + .expect("Copilot invoker step should be emitted"); + assert!( + producer < consumer, + "trusted variable producer must run before the invoker task: {agent}" + ); + assert!( + agent.contains("ADO_AW_MODEL_AGENT_COPILOT: $(ADO_AW_MODEL_AGENT_COPILOT)"), + "invoker task must resolve the model through its typed env mapping: {agent}" + ); +} + #[test] fn threat_detection_disabled_preserves_outputs_artifacts_and_manual_review() { let source = r#"--- @@ -10124,10 +10203,8 @@ fn test_github_app_token_hyphenated_private_key_variable() { /// When another ado-script bundle feature is active in the Agent job (here a /// safe-output activates the approval-summary bundle download), the mint step -/// must NOT trigger a second bundle download in that job — it reuses the -/// already-staged bundle. Proven by a delta: adding `github-app-token` to an -/// otherwise-identical workflow adds exactly ONE bundle download (the -/// Detection job, which has no extension-prepare phase), never two. +/// must NOT trigger another bundle download in either Copilot job because both +/// already stage the invoker bundle. #[test] fn test_github_app_token_reuses_staged_bundle_in_agent() { fn count_downloads(compiled: &str) -> usize { @@ -10150,14 +10227,11 @@ fn test_github_app_token_reuses_staged_bundle_in_agent() { ); assert_github_app_token_wiring(&with); - // Adding github-app-token stages the bundle only in Detection (Agent - // reuses its already-staged copy), so the download count grows by exactly 1. + // Adding github-app-token reuses the always-staged Agent and Detection bundles. assert_eq!( count_downloads(&with), - count_downloads(&without) + 1, - "github-app-token must add exactly one bundle download (Detection), \ - proving the Agent job reuses its staged bundle rather than \ - double-downloading. without={}, with={}", + count_downloads(&without), + "github-app-token must not add bundle downloads. without={}, with={}", count_downloads(&without), count_downloads(&with), );