diff --git a/.github/workflows/copilot-cli-safeoutputs.yml b/.github/workflows/copilot-cli-safeoutputs.yml index eff862f20..13eb68503 100644 --- a/.github/workflows/copilot-cli-safeoutputs.yml +++ b/.github/workflows/copilot-cli-safeoutputs.yml @@ -4,6 +4,7 @@ on: pull_request: paths: - "src/**" + - "scripts/ado-script/**" - "tests/**" - "Cargo.toml" - "Cargo.lock" @@ -62,6 +63,13 @@ jobs: mcpg:MCPG_VERSION:src/compile/common.rs VERSIONS + - name: Build Copilot controller and runner bundles + run: | + set -euo pipefail + npm --prefix scripts/ado-script ci + npm --prefix scripts/ado-script run build:copilot-controller + npm --prefix scripts/ado-script run build:copilot-runner + - name: Install compiler-pinned GitHub Copilot CLI run: | set -euo pipefail @@ -108,9 +116,12 @@ jobs: - name: Run handwritten AWF + Copilot + SafeOutputs contract env: ADO_AW_COPILOT_CLI_ARTIFACT_DIR: ${{ runner.temp }}/copilot-cli-safeoutputs + ADO_AW_COPILOT_CLI_CONTROL_DIR: ${{ runner.temp }}/copilot-cli-control ADO_AW_BIN: ${{ github.workspace }}/target/debug/ado-aw AWF_BIN: ${{ runner.temp }}/bin/awf COPILOT_BIN: ${{ runner.temp }}/bin/copilot + COPILOT_CONTROLLER_BUNDLE: ${{ github.workspace }}/scripts/ado-script/copilot-controller.js + COPILOT_RUNNER_BUNDLE: ${{ github.workspace }}/scripts/ado-script/copilot-runner.js AWF_VERSION: ${{ steps.versions.outputs.awf }} MCPG_VERSION: ${{ steps.versions.outputs.mcpg }} run: bash tests/awf-copilot-safeoutputs/run.sh diff --git a/.github/workflows/pr-sous-chef.lock.yml b/.github/workflows/pr-sous-chef.lock.yml index f1ce12c41..6f325cf18 100644 --- a/.github/workflows/pr-sous-chef.lock.yml +++ b/.github/workflows/pr-sous-chef.lock.yml @@ -1,4 +1,4 @@ -# gh-aw-metadata: {"schema_version":"v4","frontmatter_hash":"d4e6e641206e227ad9c8c71f06a4f8d75b46814949bf0464d0bc3fb84cfc2706","body_hash":"0b760609f27236f11ded4123610e522e8a73558533e14609a69bf0d176abdd4f","compiler_version":"v0.86.2","strict":true,"agent_id":"copilot","engine_versions":{"copilot":"1.0.79"}} +# gh-aw-metadata: {"schema_version":"v4","frontmatter_hash":"d4e6e641206e227ad9c8c71f06a4f8d75b46814949bf0464d0bc3fb84cfc2706","body_hash":"c58344b90694d21c8e1521cfba92d39375d8b8877e0827d4c9dff5a150bc2613","compiler_version":"v0.86.2","strict":true,"agent_id":"copilot","engine_versions":{"copilot":"1.0.79"}} # gh-aw-manifest: {"version":1,"secrets":["GH_AW_CI_TRIGGER_TOKEN","GH_AW_GITHUB_MCP_SERVER_TOKEN","GH_AW_GITHUB_TOKEN","GITHUB_TOKEN"],"actions":[{"repo":"actions/checkout","sha":"3d3c42e5aac5ba805825da76410c181273ba90b1","version":"v7.0.1"},{"repo":"actions/download-artifact","sha":"3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c","version":"v8.0.1"},{"repo":"actions/github-script","sha":"3a2844b7e9c422d3c10d287c895573f7108da1b3","version":"v9.0.0"},{"repo":"actions/setup-node","sha":"820762786026740c76f36085b0efc47a31fe5020","version":"v7.0.0"},{"repo":"actions/upload-artifact","sha":"043fb46d1a93c77aae656e7c1c64a875d1fc6a0a","version":"v7.0.1"},{"repo":"github/gh-aw-actions/setup","sha":"6aab9e5b5c91c615506061f09bedd81a23babe3c","version":"v0.86.2"}],"containers":[{"image":"ghcr.io/github/gh-aw-firewall/agent:0.27.44","digest":"sha256:0d727725c737b58c7bdf51f640cffb928385ec46517e0917c7f1a02f1bada8b4","pinned_image":"ghcr.io/github/gh-aw-firewall/agent:0.27.44@sha256:0d727725c737b58c7bdf51f640cffb928385ec46517e0917c7f1a02f1bada8b4"},{"image":"ghcr.io/github/gh-aw-firewall/api-proxy:0.27.44","digest":"sha256:b50fbadba138f6e9aba94aca09711335c489bb3b15861220cb66f6092e042dc7","pinned_image":"ghcr.io/github/gh-aw-firewall/api-proxy:0.27.44@sha256:b50fbadba138f6e9aba94aca09711335c489bb3b15861220cb66f6092e042dc7"},{"image":"ghcr.io/github/gh-aw-firewall/squid:0.27.44","digest":"sha256:83e48bbe12c634be8c228a576832fe45f66c529ac3659db92bddbcf2eeb6d627","pinned_image":"ghcr.io/github/gh-aw-firewall/squid:0.27.44@sha256:83e48bbe12c634be8c228a576832fe45f66c529ac3659db92bddbcf2eeb6d627"},{"image":"ghcr.io/github/gh-aw-mcpg:v0.4.9","digest":"sha256:e5a1569aeaf41820fa7bdee3e94468cae448133cdbf00119ad24f5b74db1ab9f","pinned_image":"ghcr.io/github/gh-aw-mcpg:v0.4.9@sha256:e5a1569aeaf41820fa7bdee3e94468cae448133cdbf00119ad24f5b74db1ab9f"},{"image":"ghcr.io/github/gh-aw-node","digest":"sha256:0d9f1fb5fd6610c0ac1f5194a38e45a8a1e81f8a390d5142d8e4e6f26a4b3196","pinned_image":"ghcr.io/github/gh-aw-node@sha256:0d9f1fb5fd6610c0ac1f5194a38e45a8a1e81f8a390d5142d8e4e6f26a4b3196"},{"image":"ghcr.io/github/github-mcp-server:v1.9.0","digest":"sha256:881b53d6f75f69bdbc1b5b10fc2f1361717c19054143b3a8529fb5c32061a50e","pinned_image":"ghcr.io/github/github-mcp-server:v1.9.0@sha256:881b53d6f75f69bdbc1b5b10fc2f1361717c19054143b3a8529fb5c32061a50e"}]} # This file was automatically generated by gh-aw (v0.86.2). DO NOT EDIT. To debug this workflow, load the skill at https://github.com/github/gh-aw/blob/main/debug.md # diff --git a/.github/workflows/pr-sous-chef.md b/.github/workflows/pr-sous-chef.md index 333c6595e..4a39ccd61 100644 --- a/.github/workflows/pr-sous-chef.md +++ b/.github/workflows/pr-sous-chef.md @@ -368,7 +368,6 @@ recommendations visible; wrap verbose detail in ## agent: `pr-processor` --- description: Decides skip/nudge actions for a single pull request using a minimal number of API calls -model: small --- You are given one PR number and its compact metadata. Decide what should happen to it, using as few tool calls as possible. diff --git a/.github/workflows/review-rust.lock.yml b/.github/workflows/review-rust.lock.yml index 1feaf1ebd..c507177fc 100644 --- a/.github/workflows/review-rust.lock.yml +++ b/.github/workflows/review-rust.lock.yml @@ -1,4 +1,4 @@ -# gh-aw-metadata: {"schema_version":"v4","frontmatter_hash":"ea67889e811c8c0a7ac2e9a1b512ef32f072734d71026f82eae0a8489198f59f","body_hash":"c6aacd86f655324be0dfa10d467221cdb5bf4e14431f715b874744d2ee0304c5","compiler_version":"v0.86.2","strict":true,"agent_id":"copilot","engine_versions":{"copilot":"1.0.79"}} +# gh-aw-metadata: {"schema_version":"v4","frontmatter_hash":"ea67889e811c8c0a7ac2e9a1b512ef32f072734d71026f82eae0a8489198f59f","body_hash":"5b9b49749ef63834afcced2cd715df56c5717bc69179daca0c867cc6941e0f84","compiler_version":"v0.86.2","strict":true,"agent_id":"copilot","engine_versions":{"copilot":"1.0.79"}} # gh-aw-manifest: {"version":1,"secrets":["GH_AW_GITHUB_MCP_SERVER_TOKEN","GH_AW_GITHUB_TOKEN","GITHUB_TOKEN"],"actions":[{"repo":"actions/cache","sha":"55cc8345863c7cc4c66a329aec7e433d2d1c52a9","version":"v6.1.0"},{"repo":"actions/cache/restore","sha":"55cc8345863c7cc4c66a329aec7e433d2d1c52a9","version":"v6.1.0"},{"repo":"actions/cache/save","sha":"55cc8345863c7cc4c66a329aec7e433d2d1c52a9","version":"v6.1.0"},{"repo":"actions/checkout","sha":"3d3c42e5aac5ba805825da76410c181273ba90b1","version":"v7.0.1"},{"repo":"actions/download-artifact","sha":"3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c","version":"v8.0.1"},{"repo":"actions/github-script","sha":"3a2844b7e9c422d3c10d287c895573f7108da1b3","version":"v9.0.0"},{"repo":"actions/setup-node","sha":"820762786026740c76f36085b0efc47a31fe5020","version":"v7.0.0"},{"repo":"actions/upload-artifact","sha":"043fb46d1a93c77aae656e7c1c64a875d1fc6a0a","version":"v7.0.1"},{"repo":"github/gh-aw-actions/setup","sha":"6aab9e5b5c91c615506061f09bedd81a23babe3c","version":"v0.86.2"}],"containers":[{"image":"ghcr.io/github/gh-aw-firewall/agent:0.27.44","digest":"sha256:0d727725c737b58c7bdf51f640cffb928385ec46517e0917c7f1a02f1bada8b4","pinned_image":"ghcr.io/github/gh-aw-firewall/agent:0.27.44@sha256:0d727725c737b58c7bdf51f640cffb928385ec46517e0917c7f1a02f1bada8b4"},{"image":"ghcr.io/github/gh-aw-firewall/api-proxy:0.27.44","digest":"sha256:b50fbadba138f6e9aba94aca09711335c489bb3b15861220cb66f6092e042dc7","pinned_image":"ghcr.io/github/gh-aw-firewall/api-proxy:0.27.44@sha256:b50fbadba138f6e9aba94aca09711335c489bb3b15861220cb66f6092e042dc7"},{"image":"ghcr.io/github/gh-aw-firewall/squid:0.27.44","digest":"sha256:83e48bbe12c634be8c228a576832fe45f66c529ac3659db92bddbcf2eeb6d627","pinned_image":"ghcr.io/github/gh-aw-firewall/squid:0.27.44@sha256:83e48bbe12c634be8c228a576832fe45f66c529ac3659db92bddbcf2eeb6d627"},{"image":"ghcr.io/github/gh-aw-mcpg:v0.4.9","digest":"sha256:e5a1569aeaf41820fa7bdee3e94468cae448133cdbf00119ad24f5b74db1ab9f","pinned_image":"ghcr.io/github/gh-aw-mcpg:v0.4.9@sha256:e5a1569aeaf41820fa7bdee3e94468cae448133cdbf00119ad24f5b74db1ab9f"},{"image":"ghcr.io/github/gh-aw-node","digest":"sha256:0d9f1fb5fd6610c0ac1f5194a38e45a8a1e81f8a390d5142d8e4e6f26a4b3196","pinned_image":"ghcr.io/github/gh-aw-node@sha256:0d9f1fb5fd6610c0ac1f5194a38e45a8a1e81f8a390d5142d8e4e6f26a4b3196"},{"image":"ghcr.io/github/github-mcp-server:v1.9.0","digest":"sha256:881b53d6f75f69bdbc1b5b10fc2f1361717c19054143b3a8529fb5c32061a50e","pinned_image":"ghcr.io/github/github-mcp-server:v1.9.0@sha256:881b53d6f75f69bdbc1b5b10fc2f1361717c19054143b3a8529fb5c32061a50e"}],"has_pull_request":true} # This file was automatically generated by gh-aw (v0.86.2). DO NOT EDIT. To debug this workflow, load the skill at https://github.com/github/gh-aw/blob/main/debug.md # diff --git a/.github/workflows/review-rust.md b/.github/workflows/review-rust.md index 136182b1b..2da67b265 100644 --- a/.github/workflows/review-rust.md +++ b/.github/workflows/review-rust.md @@ -169,7 +169,6 @@ and the themes in a `
` block. ## agent: `rust-critic` --- description: Hostile first-pass Rust reviewer that mines merge-blocking defects from changed lines -model: small --- You are a hostile senior Rust reviewer performing a first-pass audit. diff --git a/.github/workflows/review-typescript.lock.yml b/.github/workflows/review-typescript.lock.yml index 5af4a1983..24752ad10 100644 --- a/.github/workflows/review-typescript.lock.yml +++ b/.github/workflows/review-typescript.lock.yml @@ -1,4 +1,4 @@ -# gh-aw-metadata: {"schema_version":"v4","frontmatter_hash":"7259abb8515a4bb6dd1d124c1db92aa00b5e5357d4b61b78ae045ba045d425ac","body_hash":"684f375762c2d2dfb48458e8efe8760e29325adddb9edfdd9b22990ac7f3b8a7","compiler_version":"v0.86.2","strict":true,"agent_id":"copilot","engine_versions":{"copilot":"1.0.79"}} +# gh-aw-metadata: {"schema_version":"v4","frontmatter_hash":"7259abb8515a4bb6dd1d124c1db92aa00b5e5357d4b61b78ae045ba045d425ac","body_hash":"e370274e6f63c5c3accfa7545ad59e1499267f58360fe90346f2a5f5c58c12bd","compiler_version":"v0.86.2","strict":true,"agent_id":"copilot","engine_versions":{"copilot":"1.0.79"}} # gh-aw-manifest: {"version":1,"secrets":["GH_AW_GITHUB_MCP_SERVER_TOKEN","GH_AW_GITHUB_TOKEN","GITHUB_TOKEN"],"actions":[{"repo":"actions/cache","sha":"55cc8345863c7cc4c66a329aec7e433d2d1c52a9","version":"v6.1.0"},{"repo":"actions/cache/restore","sha":"55cc8345863c7cc4c66a329aec7e433d2d1c52a9","version":"v6.1.0"},{"repo":"actions/cache/save","sha":"55cc8345863c7cc4c66a329aec7e433d2d1c52a9","version":"v6.1.0"},{"repo":"actions/checkout","sha":"3d3c42e5aac5ba805825da76410c181273ba90b1","version":"v7.0.1"},{"repo":"actions/download-artifact","sha":"3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c","version":"v8.0.1"},{"repo":"actions/github-script","sha":"3a2844b7e9c422d3c10d287c895573f7108da1b3","version":"v9.0.0"},{"repo":"actions/setup-node","sha":"820762786026740c76f36085b0efc47a31fe5020","version":"v7.0.0"},{"repo":"actions/upload-artifact","sha":"043fb46d1a93c77aae656e7c1c64a875d1fc6a0a","version":"v7.0.1"},{"repo":"github/gh-aw-actions/setup","sha":"6aab9e5b5c91c615506061f09bedd81a23babe3c","version":"v0.86.2"}],"containers":[{"image":"ghcr.io/github/gh-aw-firewall/agent:0.27.44","digest":"sha256:0d727725c737b58c7bdf51f640cffb928385ec46517e0917c7f1a02f1bada8b4","pinned_image":"ghcr.io/github/gh-aw-firewall/agent:0.27.44@sha256:0d727725c737b58c7bdf51f640cffb928385ec46517e0917c7f1a02f1bada8b4"},{"image":"ghcr.io/github/gh-aw-firewall/api-proxy:0.27.44","digest":"sha256:b50fbadba138f6e9aba94aca09711335c489bb3b15861220cb66f6092e042dc7","pinned_image":"ghcr.io/github/gh-aw-firewall/api-proxy:0.27.44@sha256:b50fbadba138f6e9aba94aca09711335c489bb3b15861220cb66f6092e042dc7"},{"image":"ghcr.io/github/gh-aw-firewall/squid:0.27.44","digest":"sha256:83e48bbe12c634be8c228a576832fe45f66c529ac3659db92bddbcf2eeb6d627","pinned_image":"ghcr.io/github/gh-aw-firewall/squid:0.27.44@sha256:83e48bbe12c634be8c228a576832fe45f66c529ac3659db92bddbcf2eeb6d627"},{"image":"ghcr.io/github/gh-aw-mcpg:v0.4.9","digest":"sha256:e5a1569aeaf41820fa7bdee3e94468cae448133cdbf00119ad24f5b74db1ab9f","pinned_image":"ghcr.io/github/gh-aw-mcpg:v0.4.9@sha256:e5a1569aeaf41820fa7bdee3e94468cae448133cdbf00119ad24f5b74db1ab9f"},{"image":"ghcr.io/github/gh-aw-node","digest":"sha256:0d9f1fb5fd6610c0ac1f5194a38e45a8a1e81f8a390d5142d8e4e6f26a4b3196","pinned_image":"ghcr.io/github/gh-aw-node@sha256:0d9f1fb5fd6610c0ac1f5194a38e45a8a1e81f8a390d5142d8e4e6f26a4b3196"},{"image":"ghcr.io/github/github-mcp-server:v1.9.0","digest":"sha256:881b53d6f75f69bdbc1b5b10fc2f1361717c19054143b3a8529fb5c32061a50e","pinned_image":"ghcr.io/github/github-mcp-server:v1.9.0@sha256:881b53d6f75f69bdbc1b5b10fc2f1361717c19054143b3a8529fb5c32061a50e"}],"has_pull_request":true} # This file was automatically generated by gh-aw (v0.86.2). DO NOT EDIT. To debug this workflow, load the skill at https://github.com/github/gh-aw/blob/main/debug.md # diff --git a/.github/workflows/review-typescript.md b/.github/workflows/review-typescript.md index 4b99d2d16..a19db8f12 100644 --- a/.github/workflows/review-typescript.md +++ b/.github/workflows/review-typescript.md @@ -171,7 +171,6 @@ wrong output; otherwise `COMMENT`. ## agent: `ts-critic` --- description: Hostile first-pass TypeScript reviewer for bundled Azure DevOps runtime helpers -model: small --- You are a hostile senior TypeScript reviewer performing a first-pass audit of code that is bundled and executed on Azure DevOps build agents. diff --git a/AGENTS.md b/AGENTS.md index 2cd957f00..ed8368a24 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -319,6 +319,9 @@ fail-closed and only pauses when the agent actually proposed a reviewed output. │ ├── conclusion/ # Conclusion-job reporter source (bundled to conclusion.js) │ ├── approval-summary/ # Safe-outputs summary renderer (bundled to approval-summary.js; end-of-Agent-job summary tab) │ ├── github-app-token/ # GitHub App token minter (bundled to github-app-token.js; mints installation token in Agent + Detection when engine.github-app-token is set) +│ ├── copilot-shared/ # Strict schema-v2 request/prepared/result protocol, model resolution, typed argv/env, atomic writes shared by the controller and runner +│ ├── copilot-controller/ # Trusted host control plane (bundled to copilot-controller.js): prepare + read-result only; never mounted into AWF +│ ├── copilot-runner/ # Sandbox-only process harness (bundled to copilot-runner.js): run only, self-removal, typed argv, signal forwarding, exact exit propagation │ ├── executor-e2e/ # Stage 3 safe-output E2E test harness (not a bundle; runs deterministic scenarios against a real ADO project and files a GitHub issue on failure) │ ├── compiler-smoke-e2e/ # Smoke E2E orchestrator (not a bundle): stages each case in `tests/smoke/cases.json` to the fixed `.smoke/pipeline.yml` path on its own per-case `ado-aw-mirror` ref, queues it against its credential *lane* definition, and asserts they go green. Two modes via `SMOKE_COMPILER_SOURCE`: `candidate` (compiler built from this commit, pinned pipeline-artifact) and `released` (latest release asset, release URLs required). Built to `test-bin/` by `build:compiler-smoke-e2e`, listed in `NON_BUNDLE_DIRS`. │ ├── prepare-pr-base/ # create-pull-request preparer (bundled to prepare-pr-base.js): Agent mode uses ADO diff metadata + bounded fallback; SafeOutputs fetches the target tip; cross-org targets use isolated credentials + exact remote matching @@ -463,7 +466,8 @@ index to jump to the right page. (`gate.js`, `import.js`, the execution-context `exec-context-*.js` bundles, `conclusion.js`, `approval-summary.js`, `github-app-token.js`, `prepare-pr-base.js`, and - `azure-wif-refresh.js`), schemars-driven + `azure-wif-refresh.js`, `copilot-controller.js`, `copilot-runner.js`), + schemars-driven type codegen, the A2 design decision, the bundle env contract modelled in `src/compile/ado_bundle.rs`, and the `trigger-e2e/` gate-spec drift guard (kept in sync via `export-fact-catalog`). @@ -533,7 +537,11 @@ Following the gh-aw security model: assume deletion will make the exchange safe. This trap has caused repeated incorrect designs in credential-bearing work. Stream private material over stdin or use a container-private volume; publish only intentionally public - files (for example the interception CA certificate) under `/tmp`. + files (for example the interception CA certificate) under `/tmp`. The same + boundary applies to integrity, not only secrecy: after AWF starts, never + execute host-side code from `/tmp` or treat `/tmp` metadata as authoritative. + Copy trusted executables and results beneath `$(Agent.TempDirectory)` before + AWF and consume only those private copies afterward. 3. **Tool Allow-listing**: Agents have access to a limited, controlled set of tools — see [`docs/tools.md`](docs/tools.md) and [`docs/mcp.md`](docs/mcp.md). diff --git a/docs/ado-script.md b/docs/ado-script.md index fa1c726e5..77db181ab 100644 --- a/docs/ado-script.md +++ b/docs/ado-script.md @@ -98,6 +98,17 @@ pipeline** as runtime helpers. Today it produces the following shipped bundles: never enter the agent, MCP environment, Docker arguments, logs, status documents, or artifacts. See [`mcp.md`](mcp.md#renewable-azure-workload-identity). +- `copilot-controller.js` — the trusted host control plane used by every Agent + job and every enabled Detection job. Before AWF starts, it validates the + compiler request, resolves role-specific runtime model variables once, and + writes both the prepared sandbox invocation and an authoritative + host-private result. After AWF, its private copy validates that private + result. +- `copilot-runner.js` — the sandbox-only Copilot process harness. It validates + a prepared invocation, removes that document and its own bundle before + starting Copilot, constructs argv without a shell, forwards termination + signals, and preserves Copilot's exit status. It has no controller modes and + writes no authoritative metadata. > **Internal-only.** `ado-script` is not a user-facing front-matter > feature. Authors never write an `ado-script:` block in their agent @@ -666,6 +677,15 @@ scripts/ado-script/ │ ├── azure-wif-refresh/ # azure-wif-refresh.js entry point + renewable WIF assertion sidecar │ │ ├── index.ts # main(): rotate a private token file for user-defined stdio MCP servers │ │ └── __tests__/ # unit tests for rotation and isolation behaviour +│ ├── copilot-shared/ # strict schema-v2 request/prepared/result protocol shared by both bundles +│ │ ├── protocol.ts # validation, model resolution, argv/environment construction, atomic JSON writes +│ │ └── protocol.test.ts # schema, precedence, argv, environment, and result tests +│ ├── copilot-controller/ # copilot-controller.js trusted-host entry point +│ │ ├── index.ts # prepare + read-result modes only +│ │ └── index.test.ts # controller CLI and trusted-result tests +│ ├── copilot-runner/ # copilot-runner.js sandbox entry point +│ │ ├── index.ts # run mode, self-removal, signals, exact exit +│ │ └── index.test.ts # process, lifecycle, and mode-isolation tests │ ├── trigger-e2e/ # test-only: FACT_META gate-spec table + trigger-evaluation E2E scenarios (not a bundle) │ │ ├── gate-spec.ts # FACT_META mirror of Rust Fact::ALL; drift-guarded by export-fact-catalog + fact-catalog.gen.json │ │ ├── fact-catalog.gen.json # generated by `cargo run -- export-fact-catalog`; deep-compared by gate-spec.test.ts @@ -689,7 +709,9 @@ scripts/ado-script/ ├── github-app-token.js # ncc bundle output (gitignored) ├── prepare-pr-base.js # ncc bundle output (gitignored) ├── ado-proxy.js # ncc bundle output (gitignored) -└── azure-wif-refresh.js # ncc bundle output (gitignored) +├── azure-wif-refresh.js # ncc bundle output (gitignored) +├── copilot-controller.js # ncc bundle output (gitignored) +└── copilot-runner.js # ncc bundle output (gitignored) ``` The release workflow (`.github/workflows/release.yml`) runs @@ -700,8 +722,8 @@ captures every bundle, including `gate.js`, `import.js`, `exec-context-ci-push.js`, `exec-context-workitem.js`, `exec-context-schedule.js`, `exec-context-pr-checks.js`, `exec-context-repo.js`, `conclusion.js`, `approval-summary.js`, -`github-app-token.js`, `prepare-pr-base.js`, `ado-proxy.js`, and -`azure-wif-refresh.js` — into the +`github-app-token.js`, `prepare-pr-base.js`, `ado-proxy.js`, +`azure-wif-refresh.js`, `copilot-controller.js`, and `copilot-runner.js` — into the `ado-script.zip` release asset. Pipelines download that asset at runtime by URL pinned to the compiler's `CARGO_PKG_VERSION`, verify its SHA-256 against the `checksums.txt` asset, then extract. @@ -742,9 +764,8 @@ cargo run -- export-gate-schema --output schema/gate-spec.schema.json `AdoScriptExtension` (`src/compile/extensions/ado_script.rs`) is the always-on single -extension that owns all `ado-script` wiring. It has two independent -features, each emitted **into the job that actually consumes the -bundle**: +extension that owns shared `ado-script` delivery. Each download is emitted +**into the job that actually consumes the bundle**: ### Setup job (gate evaluator) @@ -762,12 +783,11 @@ returns three typed `Declarations::setup_steps` entries for the Setup job: runs the gate with `GATE_SPEC` and the env-var contract documented above. -### Agent job (runtime-import resolver + PR-context precompute) +### Agent job (Copilot controller/runner + optional helpers) -When `inlined-imports: false` (the default) OR the execution-context -PR contributor activates (`on.pr` configured and not disabled), -`AdoScriptExtension::declarations()` returns the install + download pair in -`Declarations::agent_prepare_steps` for the Agent job: +Every Agent job receives the install + download pair in +`Declarations::agent_prepare_steps`, because every Copilot run uses +the controller and runner: 1. **`UseNode@1`** — same shape as above. 2. **`curl` download + verify + extract** — same artefact, same @@ -777,6 +797,19 @@ PR contributor activates (`on.pr` configured and not disabled), `/tmp/awf-tools/agent-prompt.md` in place. See [`runtime-imports.md`](runtime-imports.md) for marker syntax. **Only emitted when `inlined-imports: false`.** +4. The Agent task copies the verified controller into a mode-0700 directory + beneath `$(Agent.TempDirectory)`, writes the schema-v2 request there, and + runs `prepare`. Preparation atomically writes the host-private result and + the non-authoritative prepared invocation under `/tmp/awf-tools`. +5. The AWF run executes the fixed command + `node /tmp/ado-aw-scripts/ado-script/copilot-runner.js run + /tmp/awf-tools/copilot-invocation.json`. Author values and prompt contents + are data in the prepared document, never shell fragments. The private + controller copy and its request/result paths are not mounted or passed into + AWF. +6. After AWF, only the private controller copy reads the private result. The + controller source staged under `/tmp` is removed before AWF, and no + sandbox-writable code or metadata is trusted on the host afterward. The PR-context precompute step (`node exec-context-pr.js`) is owned by `ExecContextExtension` (not `AdoScriptExtension`) and emitted through @@ -785,27 +818,73 @@ its own Tool-phase `Declarations::agent_prepare_steps`. Phase ordering guarantees the bundle is installed and on disk before the exec-context invocation runs. +### Detection job + +Every enabled Detection job installs Node and stages `ado-script.zip`, then +uses the same trusted controller plus fixed runner command with a +Detection-role request. Detection does not receive the Agent MCP configuration. +Workflows that explicitly disable threat detection emit no Detection download, +documents, model mappings, controller, or runner command. + +### Copilot invocation and result protocol + +Schema version 2 has three strict document kinds: + +- **request** — compiler-emitted under the private + `$(Agent.TempDirectory)` controller directory. It contains role, command, + prompt path, optional MCP config path, argv entries, and optional explicit + model. +- **prepared** — controller-emitted under + `/tmp/awf-tools/copilot-invocation.json`. It contains the same execution data + plus the resolved `requested_model`; it has no result path and performs no + ADO-variable lookup inside AWF. +- **result** — controller-emitted beside the private request. It contains the + role and the same `requested_model` written into the prepared document. + +The controller supports only `prepare ` and +`read-result `. The runner supports only +`run `. Both reject unknown fields, wrong document kinds, unsupported +versions, unsafe paths/commands, invalid models, and NUL bytes. + +Preparation happens after Azure DevOps resolves the task's typed environment +mappings, so YAML variables, UI variables, variable groups, and earlier +same-job `task.setvariable` values retain their task-start semantics. The model +is resolved once and atomically committed to both execution and audit metadata. +The runner ignores runtime selector variables, uses only the prepared model, +reads the prompt as UTF-8, spawns Copilot without a shell, and sets or removes +`COPILOT_MODEL` only in the child environment. + +AWF exposes host `/tmp` read-write to the sandbox. Therefore every file beneath +`/tmp`, including the runner and prepared document, is untrusted after AWF +starts. The runner removes itself and the prepared document before giving +Copilot control, but that is defense in depth rather than a single-process +security boundary. The authoritative controller and result remain outside +`/tmp`; host code never executes or trusts sandbox-writable code or metadata +afterward. + ### Per-job download (NOT a duplication bug) ADO jobs use **isolated VMs** — `/tmp` is not shared between jobs. The `ado-script.zip` bundle therefore has to be downloaded once per -job that consumes it. When both Setup and Agent need it, install + -download steps appear in **both**. That's correct architecture given -ADO's topology, not waste. +job that consumes it. Agent and enabled Detection therefore always stage the +bundle; Setup stages another copy when a gate or synthetic-PR resolver needs +one. That's correct architecture given ADO's topology, not waste. ### What gets emitted, by case The rows below assume the synthetic-PR resolver is **not** active (`pr_trigger_for_synth = None`): -| Setup consumer | Agent consumer | Setup-job steps | Agent-job extra steps | +| Setup consumer | Agent optional consumers | Setup-job steps | Agent-job steps | |---|---|---|---| -| no gate | none | (none) | (none) | -| no gate | `inlined-imports: false` only | (no Setup job) | install + download + resolver | -| no gate | execution-context contributor(s) only | (no Setup job) | install + download + exec-context bundle(s) | -| no gate | resolver + execution-context | (no Setup job) | install + download + resolver + exec-context bundle(s) | -| gate | none | install + download + gate | (none) | -| gate | any combination of resolver / exec-context | install + download + gate | install + download + (resolver?) + (exec-context bundle(s)?) | +| no gate | none | (none) | install + download + controller/runner | +| no gate | runtime imports | (none) | install + download + resolver + controller/runner | +| no gate | execution-context contributor(s) | (none) | install + download + exec-context bundle(s) + controller/runner | +| gate | none | install + download + gate | install + download + controller/runner | +| gate | resolver / exec-context | install + download + gate | install + download + optional helpers + controller/runner | + +Every row also has an enabled Detection job with its own install + download + +controller/runner unless threat detection is explicitly disabled. When the synthetic-PR resolver **is** active (`pr_trigger_for_synth = Some(_)`, i.e. `synthetic_pr_active()` is @@ -813,18 +892,16 @@ true) the Setup job gains the `synthPr` step (`node exec-context-pr-synth.js`) before any gate step — and the Setup job is emitted even with no gate: -| Setup consumer | Setup-job steps | Agent-job extra steps | +| Setup consumer | Setup-job steps | Agent-job steps | |---|---|---| -| synth-PR (no gate) | install + download + synth-PR | (per Agent consumer above) | -| gate (no synth-PR) | install + download + gate | (per Agent consumer above) | -| synth-PR + gate | install + download + synth-PR + gate | (per Agent consumer above) | +| synth-PR (no gate) | install + download + synth-PR | install + download + optional helpers + controller/runner | +| gate (no synth-PR) | install + download + gate | install + download + optional helpers + controller/runner | +| synth-PR + gate | install + download + synth-PR + gate | install + download + optional helpers + controller/runner | The "Setup consumer" column is gated on `filters:` lowering to non-empty -checks **or** `synthetic_pr_active()` being true. The "Agent consumer" -columns are gated on `inlined-imports: false` (resolver) and the PR -contributor's activation predicate (exec-context-pr; see -`pr_contributor_will_activate` in -`src/compile/extensions/exec_context/mod.rs`). +checks **or** `synthetic_pr_active()` being true. The Agent download is +unconditional; optional resolver and execution-context steps retain their own +feature predicates. The IR-to-bash codegen that produces the gate step is `compile_gate_step_external` in `src/compile/filter_ir.rs`. diff --git a/docs/audit.md b/docs/audit.md index e2fdf6980..b2bd46de1 100644 --- a/docs/audit.md +++ b/docs/audit.md @@ -61,13 +61,25 @@ URL-encoded project segments are decoded before the ADO context is resolved. `t= │ ├── mcpg/ # MCP Gateway logs (includes the SafeOutputs stdio child's stdout/stderr) │ └── agent-output.txt # Filtered agent stdout ├── analyzed_outputs[_]/ # Downloaded artifact (Detection stage) +│ ├── aw_info.json # Agent metadata + Detection runtime model │ ├── threat-analysis.json # Aggregate verdict + reasons │ └── threat-analysis-output.txt └── safe_outputs[_]/ # Downloaded artifact (SafeOutputs stage) └── safe-outputs-executed.ndjson # Per-item execution log ``` -`aw_info.json`, `otel.jsonl`, and `safe_outputs.ndjson` are searched in `staging/` first and then at the artifact top level so older layouts still audit cleanly. +Agent `aw_info.json`, `otel.jsonl`, and `safe_outputs.ndjson` are searched in +`staging/` first and then at the artifact top level so older layouts still +audit cleanly. When present, the Detection-enriched `aw_info.json` from +`analyzed_outputs` overlays only Detection-owned runtime fields in the report. + +`overview.aw_info.model` is the model requested for the Agent's Copilot +session. Copilot custom-agent definitions may pin a different model. When OTel +contains `gen_ai.request.model`, `engine_config.model` is the model observed +during execution; otherwise it falls back to the requested model. Console +output shows `requested_model` and `observed_model` separately when they differ. +`overview.aw_info.detection_model` is the requested Detection session model; +Detection OTel is not currently included in the analyzed artifact. ## Report shape (`AuditData`) @@ -75,7 +87,7 @@ Current top-level keys include the following. Optional sections are omitted from | Key | Source | | --- | --- | -| `overview` | ADO build metadata + `aw_info.json` (engine, model, optional threat-detection enabled/engine/model, agent name, source, target). | +| `overview` | ADO build metadata + `aw_info.json` (engine, requested session model, optional threat-detection enabled/engine/requested model, agent name, source, target). | | `task_domain` | Audit heuristics over the run's prompts and outputs. | | `behavior_fingerprint` | Higher-level audit heuristics over the run's behavior. | | `agentic_assessments` | Higher-level audit assessments emitted by the analyzers. | @@ -83,7 +95,7 @@ Current top-level keys include the following. Optional sections are omitted from | `key_findings` | Heuristic rules + analyzer-emitted findings (for example aggregate-gate rejection). | | `recommendations` | Follow-up actions derived from findings. | | `performance_metrics` | Derived from `metrics`, runtime duration, tool usage, and firewall counts. | -| `engine_config` | Runtime engine configuration derived from `aw_info.json`. | +| `engine_config` | Runtime engine configuration; the Agent model prefers the OTel-observed model and falls back to the requested `aw_info.json` model. | | `safe_output_summary` | Counts of proposed / executed / rejected / not processed items. | | `safe_output_execution` | Per-item trace joining proposal + detection + execution. | | `rejected_safe_outputs` | Rollup of rejections by reason / threat flag. | diff --git a/docs/engine.md b/docs/engine.md index ce7724e07..4562a5c43 100644 --- a/docs/engine.md +++ b/docs/engine.md @@ -22,18 +22,76 @@ engine: | Field | Type | Default | Description | |-------|------|---------|-------------| | `id` | string | `copilot` | Engine identifier. Currently only `copilot` (GitHub Copilot CLI) is supported. | -| `model` | string | *(none)* | AI model to use (e.g., `gpt-5-mini`). When omitted, the compiler does not emit `--model` and the Copilot CLI chooses its own default. When set, the compiler passes the value directly to the Copilot CLI `--model` flag — any model identifier the Copilot CLI accepts is valid. | +| `model` | string | *(none)* | AI model to use (e.g., `gpt-5-mini`). When set, the compiler records it in the versioned Copilot request; the trusted controller preserves it in the prepared invocation, and the sandbox runner sets Copilot CLI's native `COPILOT_MODEL` environment variable. When omitted, runtime model controls can select a model; if no runtime control is set, `COPILOT_MODEL` remains unset and the Copilot CLI chooses its own default. | | `timeout-minutes` | integer | *(none)* | Maximum time in minutes the agent job is allowed to run. Sets `timeoutInMinutes` on the `Agent` job in the generated pipeline. | | `version` | string | *(none)* | Engine CLI version to install (e.g., `"1.0.70"`, `"latest"`). Overrides the pinned `COPILOT_CLI_VERSION`. Set to `"latest"` to use the newest available version. | | `agent` | string | *(none)* | Custom agent file identifier (Copilot only). Adds `--agent ` to the CLI invocation, selecting a custom agent from `.github/agents/`. | | `api-target` | string | *(none)* | Custom API endpoint hostname for GHES/GHEC (e.g., `"api.acme.ghe.com"`). Adds `--api-target ` to the CLI invocation and adds the hostname to the AWF network allowlist. | -| `args` | list | `[]` | Custom CLI arguments appended after compiler-generated args. Subject to shell-safety validation and blocked from overriding compiler-controlled flags (`--prompt`, `--additional-mcp-config`, `--allow-tool`, `--allow-all-tools`, `--allow-all-paths`, `--disable-builtin-mcps`, `--no-ask-user`, `--ask-user`). | -| `env` | map | *(none)* | Engine-specific environment variables merged into the sandbox step's `env:` block. Keys must be valid env var names. Values are literal-only and must not contain ADO expressions (`$(`, `${{`, `$[`) or pipeline command injection (`##vso[`), **except** the Copilot provider keys (`COPILOT_PROVIDER_BASE_URL`, `COPILOT_PROVIDER_API_KEY`, `COPILOT_PROVIDER_BEARER_TOKEN`, `COPILOT_PROVIDER_WIRE_API`), which may carry an ADO macro (`$(...)`) expression. Prefer the typed [`provider`](#copilot-model-provider-byok-configuration) block over raw provider env keys. Compiler-controlled keys (`GITHUB_TOKEN`, `PATH`, `BASH_ENV`, etc.) are blocked. | +| `args` | list | `[]` | Custom CLI arguments appended after compiler-generated args. Subject to shell-safety validation and blocked from overriding compiler-controlled flags (`--prompt`, `--model`, `--additional-mcp-config`, `--allow-tool`, `--allow-all-tools`, `--allow-all-paths`, `--disable-builtin-mcps`, `--no-ask-user`, `--ask-user`). Use `engine.model` or the runtime variables below instead of a raw `--model` argument. | +| `env` | map | *(none)* | Engine-specific environment variables merged into the sandbox step's `env:` block. Keys must be valid env var names. Values are literal-only and must not contain ADO expressions (`$(`, `${{`, `$[`) or pipeline command injection (`##vso[`), **except** the Copilot provider keys (`COPILOT_PROVIDER_BASE_URL`, `COPILOT_PROVIDER_API_KEY`, `COPILOT_PROVIDER_BEARER_TOKEN`, `COPILOT_PROVIDER_WIRE_API`), which may carry an ADO macro (`$(...)`) expression. Prefer the typed [`provider`](#copilot-model-provider-byok-configuration) block over raw provider env keys. Compiler-controlled keys (`GITHUB_TOKEN`, `COPILOT_MODEL`, `PATH`, `BASH_ENV`, etc.) are blocked. | | `provider` | map | *(none)* | Copilot external model-provider (BYOK) configuration: `base-url`, `type`, `wire-api`, `token` (compiler-minted bearer via a service connection), `api-key`. Maps to the `COPILOT_PROVIDER_*` env vars. See [Copilot model provider (BYOK) configuration](#copilot-model-provider-byok-configuration). | | `command` | string | *(none)* | Custom engine executable path (skips the default engine binary installation — NuGet for `target: 1es`, GitHub Releases for all other targets). The path must be accessible inside the AWF container (e.g., `/tmp/...` or workspace-mounted paths). | | `github-app-token` | map | *(none)* | GitHub App-backed Copilot engine authentication. When set, the compiler mints (and, by default, revokes) a GitHub App installation token in the Agent and Detection jobs and sources `GITHUB_TOKEN` from it (for Copilot only). See [GitHub App-backed Copilot engine auth](#github-app-backed-copilot-engine-auth). | +### Runtime model controls + +For the Copilot engine, operators can switch models at Azure DevOps pipeline +runtime without editing workflow markdown or recompiling lock files. Configure +these as pipeline variables or variable-group entries: + +| Variable | Applies to | +|----------|------------| +| `ADO_AW_MODEL_AGENT_COPILOT` | Agent job only | +| `ADO_AW_MODEL_DETECTION_COPILOT` | Detection job only | +| `ADO_AW_DEFAULT_MODEL_COPILOT` | Fallback for both jobs | + +Precedence is: + +1. Explicit `engine.model` for the effective engine config. +2. Role-specific runtime variable (`ADO_AW_MODEL_AGENT_COPILOT` or + `ADO_AW_MODEL_DETECTION_COPILOT`). +3. Shared runtime variable (`ADO_AW_DEFAULT_MODEL_COPILOT`). +4. Existing default behavior (`COPILOT_MODEL` is unset; the Copilot CLI + chooses). + +Detection uses its effective engine config after applying +`safe-outputs.threat-detection.engine`, so an inherited or nested explicit model +still wins over runtime variables. This mirrors gh-aw and means the runtime +variables are an operational escape hatch only for workflows that leave +`engine.model` unset; changing a pinned frontmatter model still requires +recompilation. + +Runtime values are passed through typed step environment mappings, so Azure +DevOps YAML variables, UI variables, variable groups, and variables set by an +earlier trusted `##vso[task.setvariable]` step all resolve at task start. Inside +the trusted host task, `copilot-controller.js` validates and resolves the +selected value before AWF starts. It writes both a host-private result and a +prepared sandbox invocation containing that same model decision. Inside AWF, +the run-only `copilot-runner.js` sets Copilot CLI's native `COPILOT_MODEL` only +in the child process environment. When no value resolves, the runner removes +`COPILOT_MODEL` rather than supplying a compiler default. Raw +`engine.args --model` and +`engine.env.COPILOT_MODEL` are rejected so they cannot bypass this precedence. + +The authoritative result and controller copy live beneath +`$(Agent.TempDirectory)`, outside AWF's automatic host `/tmp` mount. After AWF +returns, the host validates that private result with the private controller and +records the requested session model in `aw_info.json`; Detection uses the same +contract and later enriches the copied metadata in +`analyzed_outputs_` from its own job scope. No JavaScript or result +file exposed through sandbox-writable `/tmp` is executed or trusted afterward. +A prior trusted step can therefore set a variable and both execution and +metadata see the same task-start value. `ado-aw audit` merges those job-owned +fields. + +When `engine.agent` selects a custom agent whose definition declares `model` or +`models`, Copilot CLI may use that agent-pinned model instead of the requested +session model. `aw_info.json` records the requested session model; when Copilot +OTel is present, `ado-aw audit` reports the observed Agent model separately. +Detection metadata is requested-model-only because the analyzed artifact does +not currently include Detection OTel. + ### `timeout-minutes` The `timeout-minutes` field sets a wall-clock limit (in minutes) for the entire agent job. It maps to the Azure DevOps `timeoutInMinutes` job property on `Agent`. This is useful for: @@ -246,7 +304,8 @@ runtime (raw `engine.env` cross-job macros like `$(Setup.FOUNDRY_TOKEN)` do | `token` | optional | `COPILOT_PROVIDER_API_KEY` | Compiler-minted credential via Azure CLI (see below). Mutually exclusive with `api-key`. | | `api-key` | optional | `COPILOT_PROVIDER_API_KEY` | Static API key, typically a `$(VAR)` secret pipeline variable. Mutually exclusive with `token`. | -The model itself is set via `engine.model` (or a `COPILOT_MODEL` env var). +The model itself is set via `engine.model` or the +[`ADO_AW_MODEL_*`](#runtime-model-controls) runtime controls. #### Compiler-owned token acquisition (`provider.token`) diff --git a/scripts/ado-script/.gitignore b/scripts/ado-script/.gitignore index 019aca5c2..db750d1f3 100644 --- a/scripts/ado-script/.gitignore +++ b/scripts/ado-script/.gitignore @@ -17,6 +17,8 @@ github-app-token.js prepare-pr-base.js ado-proxy.js azure-wif-refresh.js +copilot-controller.js +copilot-runner.js schema *.tsbuildinfo test-bin diff --git a/scripts/ado-script/package.json b/scripts/ado-script/package.json index 4e6c5e1f7..0e611488a 100644 --- a/scripts/ado-script/package.json +++ b/scripts/ado-script/package.json @@ -7,8 +7,8 @@ "node": ">=20.0.0" }, "scripts": { - "build": "npm run codegen && npm run clean && npm run build:gate && npm run build:import && npm run build:exec-context-pr && npm run build:exec-context-pr-synth && npm run build:exec-context-manual && npm run build:exec-context-pipeline && npm run build:exec-context-ci-push && npm run build:exec-context-workitem && npm run build:exec-context-schedule && npm run build:exec-context-pr-checks && npm run build:exec-context-repo && npm run build:conclusion && npm run build:approval-summary && npm run build:github-app-token && npm run build:prepare-pr-base && npm run build:ado-proxy && npm run build:azure-wif-refresh", - "clean": "node -e \"const fs=require('node:fs'); fs.rmSync('.ado-build',{recursive:true,force:true}); for (const n of ['gate','import','exec-context-pr','exec-context-pr-synth','exec-context-manual','exec-context-pipeline','exec-context-ci-push','exec-context-workitem','exec-context-schedule','exec-context-pr-checks','exec-context-repo','conclusion','approval-summary','github-app-token','prepare-pr-base','ado-proxy','azure-wif-refresh']) fs.rmSync(n+'.js',{force:true});\"", + "build": "npm run codegen && npm run clean && npm run build:gate && npm run build:import && npm run build:exec-context-pr && npm run build:exec-context-pr-synth && npm run build:exec-context-manual && npm run build:exec-context-pipeline && npm run build:exec-context-ci-push && npm run build:exec-context-workitem && npm run build:exec-context-schedule && npm run build:exec-context-pr-checks && npm run build:exec-context-repo && npm run build:conclusion && npm run build:approval-summary && npm run build:github-app-token && npm run build:prepare-pr-base && npm run build:ado-proxy && npm run build:azure-wif-refresh && npm run build:copilot-controller && npm run build:copilot-runner", + "clean": "node -e \"const fs=require('node:fs'); fs.rmSync('.ado-build',{recursive:true,force:true}); for (const n of ['gate','import','exec-context-pr','exec-context-pr-synth','exec-context-manual','exec-context-pipeline','exec-context-ci-push','exec-context-workitem','exec-context-schedule','exec-context-pr-checks','exec-context-repo','conclusion','approval-summary','github-app-token','prepare-pr-base','ado-proxy','azure-wif-refresh','copilot-invoker','copilot-controller','copilot-runner']) fs.rmSync(n+'.js',{force:true});\"", "build:gate": "ncc build src/gate/index.ts -o .ado-build/gate -m -t && node -e \"const fs=require('node:fs'); fs.copyFileSync('.ado-build/gate/index.js','gate.js'); fs.rmSync('.ado-build/gate',{recursive:true,force:true});\"", "build:import": "ncc build src/import/index.ts -o .ado-build/import -m -t && node -e \"const fs=require('node:fs'); fs.copyFileSync('.ado-build/import/index.js','import.js'); fs.rmSync('.ado-build/import',{recursive:true,force:true});\"", "build:exec-context-pr": "ncc build src/exec-context-pr/index.ts -o .ado-build/exec-context-pr -m -t && node -e \"const fs=require('node:fs'); fs.copyFileSync('.ado-build/exec-context-pr/index.js','exec-context-pr.js'); fs.rmSync('.ado-build/exec-context-pr',{recursive:true,force:true});\"", @@ -26,13 +26,15 @@ "build:prepare-pr-base": "ncc build src/prepare-pr-base/index.ts -o .ado-build/prepare-pr-base -m -t && node -e \"const fs=require('node:fs'); fs.copyFileSync('.ado-build/prepare-pr-base/index.js','prepare-pr-base.js'); fs.rmSync('.ado-build/prepare-pr-base',{recursive:true,force:true});\"", "build:ado-proxy": "ncc build src/ado-proxy/index.ts -o .ado-build/ado-proxy -m -t && node -e \"const fs=require('node:fs'); fs.copyFileSync('.ado-build/ado-proxy/index.js','ado-proxy.js'); fs.rmSync('.ado-build/ado-proxy',{recursive:true,force:true});\"", "build:azure-wif-refresh": "ncc build src/azure-wif-refresh/index.ts -o .ado-build/azure-wif-refresh -m -t && node -e \"const fs=require('node:fs'); fs.copyFileSync('.ado-build/azure-wif-refresh/index.js','azure-wif-refresh.js'); fs.rmSync('.ado-build/azure-wif-refresh',{recursive:true,force:true});\"", + "build:copilot-controller": "ncc build src/copilot-controller/index.ts -o .ado-build/copilot-controller -m -t && node -e \"const fs=require('node:fs'); fs.copyFileSync('.ado-build/copilot-controller/index.js','copilot-controller.js'); fs.rmSync('.ado-build/copilot-controller',{recursive:true,force:true});\"", + "build:copilot-runner": "ncc build src/copilot-runner/index.ts -o .ado-build/copilot-runner -m -t && node -e \"const fs=require('node:fs'); fs.copyFileSync('.ado-build/copilot-runner/index.js','copilot-runner.js'); fs.rmSync('.ado-build/copilot-runner',{recursive:true,force:true});\"", "build:executor-e2e": "ncc build src/executor-e2e/index.ts -o .ado-build/executor-e2e -m -t && node -e \"const fs=require('node:fs'); fs.mkdirSync('test-bin',{recursive:true}); fs.copyFileSync('.ado-build/executor-e2e/index.js','test-bin/executor-e2e.js'); fs.rmSync('.ado-build/executor-e2e',{recursive:true,force:true});\"", "build:trigger-e2e": "ncc build src/trigger-e2e/index.ts -o .ado-build/trigger-e2e -m -t && node -e \"const fs=require('node:fs'); fs.mkdirSync('test-bin',{recursive:true}); fs.copyFileSync('.ado-build/trigger-e2e/index.js','test-bin/trigger-e2e.js'); fs.rmSync('.ado-build/trigger-e2e',{recursive:true,force:true});\"", "build:compiler-smoke-e2e": "ncc build src/compiler-smoke-e2e/index.ts -o .ado-build/compiler-smoke-e2e -m -t && node -e \"const fs=require('node:fs'); fs.mkdirSync('test-bin',{recursive:true}); fs.copyFileSync('.ado-build/compiler-smoke-e2e/index.js','test-bin/compiler-smoke-e2e.js'); fs.rmSync('.ado-build/compiler-smoke-e2e',{recursive:true,force:true});\"", "build:check": "ls -lh gate.js && wc -c gate.js", "codegen": "node -e \"require('node:fs').mkdirSync('schema', { recursive: true })\" && cargo run --quiet --manifest-path ../../Cargo.toml -- export-gate-schema --output schema/gate-spec.schema.json && npx json2ts schema/gate-spec.schema.json -o src/shared/types.gen.ts --bannerComment \"// AUTO-GENERATED from Rust IR via cargo run -- export-gate-schema. Do not edit; run npm run codegen.\" && cargo run --quiet --manifest-path ../../Cargo.toml -- export-fact-catalog --output src/trigger-e2e/fact-catalog.gen.json && cargo run --quiet --manifest-path ../../Cargo.toml -- export-ado-proxy-catalog-schema --output schema/ado-proxy-catalog.schema.json && npx json2ts schema/ado-proxy-catalog.schema.json -o src/shared/ado-proxy-catalog.types.gen.ts --bannerComment \"// AUTO-GENERATED from Rust via cargo run -- export-ado-proxy-catalog-schema. Do not edit; run npm run codegen.\" && cargo run --quiet --manifest-path ../../Cargo.toml -- export-ado-proxy-catalog --output src/ado-proxy/catalog.gen.json", "test": "vitest run", - "test:smoke": "npm run build:gate && npm run build:import && npm run build:exec-context-pr && npm run build:exec-context-pr-synth && npm run build:exec-context-manual && npm run build:exec-context-pipeline && npm run build:exec-context-ci-push && npm run build:exec-context-workitem && npm run build:exec-context-schedule && npm run build:exec-context-pr-checks && npm run build:exec-context-repo && npm run build:conclusion && npm run build:approval-summary && npm run build:github-app-token && npm run build:prepare-pr-base && npm run build:ado-proxy && npm run build:azure-wif-refresh && vitest run -c vitest.config.smoke.ts", + "test:smoke": "npm run build:gate && npm run build:import && npm run build:exec-context-pr && npm run build:exec-context-pr-synth && npm run build:exec-context-manual && npm run build:exec-context-pipeline && npm run build:exec-context-ci-push && npm run build:exec-context-workitem && npm run build:exec-context-schedule && npm run build:exec-context-pr-checks && npm run build:exec-context-repo && npm run build:conclusion && npm run build:approval-summary && npm run build:github-app-token && npm run build:prepare-pr-base && npm run build:ado-proxy && npm run build:azure-wif-refresh && npm run build:copilot-controller && npm run build:copilot-runner && vitest run -c vitest.config.smoke.ts", "lint": "echo TODO", "typecheck": "tsc --noEmit" }, diff --git a/scripts/ado-script/src/__tests__/bundle-coverage.test.ts b/scripts/ado-script/src/__tests__/bundle-coverage.test.ts index 053207945..4c9861590 100644 --- a/scripts/ado-script/src/__tests__/bundle-coverage.test.ts +++ b/scripts/ado-script/src/__tests__/bundle-coverage.test.ts @@ -44,6 +44,7 @@ const packageJsonPath = join(here, "..", "..", "package.json"); */ const NON_BUNDLE_DIRS = new Set([ "shared", + "copilot-shared", "__tests__", "executor-e2e", "trigger-e2e", diff --git a/scripts/ado-script/src/compiler-smoke-e2e/__tests__/ado-rest.test.ts b/scripts/ado-script/src/compiler-smoke-e2e/__tests__/ado-rest.test.ts index bce3a951c..e6b66be7b 100644 --- a/scripts/ado-script/src/compiler-smoke-e2e/__tests__/ado-rest.test.ts +++ b/scripts/ado-script/src/compiler-smoke-e2e/__tests__/ado-rest.test.ts @@ -99,6 +99,26 @@ describe("AdoRest.queueBuild", () => { }); }); + it("serializes queue variables through the Build Queue parameters string", async () => { + let sentBody: unknown; + const fetchImpl = vi.fn( + async (_input: RequestInfo | URL, init?: RequestInit) => { + sentBody = JSON.parse(String(init?.body)); + return jsonResponse(200, { id: 556 }); + }, + ); + const rest = makeRest(fetchImpl as unknown as typeof fetch); + await rest.queueBuild(2560, { + sourceBranch: "refs/heads/x", + sourceVersion: "deadbeef", + variables: { ADO_AW_MODEL_AGENT_COPILOT: "auto" }, + }); + expect(sentBody).toMatchObject({ + parameters: JSON.stringify({ ADO_AW_MODEL_AGENT_COPILOT: "auto" }), + }); + expect(sentBody).not.toHaveProperty("variables"); + }); + it("throws with a descriptive error on a non-2xx response", async () => { const fetchImpl = vi.fn(async () => new Response("nope", { status: 500 })); const rest = makeRest(fetchImpl as unknown as typeof fetch); diff --git a/scripts/ado-script/src/compiler-smoke-e2e/__tests__/cases.test.ts b/scripts/ado-script/src/compiler-smoke-e2e/__tests__/cases.test.ts index b0c6e64d5..04098633d 100644 --- a/scripts/ado-script/src/compiler-smoke-e2e/__tests__/cases.test.ts +++ b/scripts/ado-script/src/compiler-smoke-e2e/__tests__/cases.test.ts @@ -313,7 +313,7 @@ describe("parseManifest", () => { }, ]), ), - ).toThrow(/must declare agentCommand, pipelineText and\/or requiredBuildTags/); + ).toThrow(/must declare agentCommand, pipelineText, requiredBuildTags and\/or requestedModels/); }); it("rejects an agentCommand with no snippets", () => { @@ -356,6 +356,67 @@ describe("parseManifest", () => { forbidden: ["$SC_READ_TOKEN"], }); }); + + it("parses and validates requested model assertions", () => { + const parsed = parseManifest( + cases([ + { + id: "x", + lane: "agentic", + kind: "compiled", + modes: ["candidate", "released"], + source: "a.md", + assertions: { requestedModels: { agent: "auto", detection: "gpt-5.4" } }, + }, + ]), + ); + expect(parsed.cases[0]?.assertions?.requestedModels).toEqual({ + agent: "auto", + detection: "gpt-5.4", + }); + + expect(() => + parseManifest( + cases([ + { + id: "x", + lane: "agentic", + kind: "compiled", + modes: ["candidate", "released"], + source: "a.md", + assertions: { requestedModels: {} }, + }, + ]), + ), + ).toThrow(/must declare agent and\/or detection/); + }); + }); + + describe("queue variable validation", () => { + it("parses valid variables and rejects empty or malformed values", () => { + const entry = { + id: "x", + lane: "agentic", + kind: "compiled", + modes: ["candidate", "released"], + source: "a.md", + }; + const parsed = parseManifest( + cases([{ ...entry, queueVariables: { ADO_AW_MODEL_AGENT_COPILOT: "auto" } }]), + ); + expect(parsed.cases[0]?.queueVariables).toEqual({ + ADO_AW_MODEL_AGENT_COPILOT: "auto", + }); + expect(() => parseManifest(cases([{ ...entry, queueVariables: {} }]))).toThrow( + /must not be empty/, + ); + expect(() => + parseManifest(cases([{ ...entry, queueVariables: { "not-valid": "auto" } }])), + ).toThrow(/must match/); + expect(() => + parseManifest(cases([{ ...entry, queueVariables: { ADO_AW_MODEL_AGENT_COPILOT: "" } }])), + ).toThrow(/must be a non-empty string/); + }); }); }); diff --git a/scripts/ado-script/src/compiler-smoke-e2e/__tests__/index.test.ts b/scripts/ado-script/src/compiler-smoke-e2e/__tests__/index.test.ts index 2606cf65a..549ac290f 100644 --- a/scripts/ado-script/src/compiler-smoke-e2e/__tests__/index.test.ts +++ b/scripts/ado-script/src/compiler-smoke-e2e/__tests__/index.test.ts @@ -8,7 +8,13 @@ const mockCalls: string[] = []; const compiledCasePaths: string[] = []; const stagedWrites: { to: string; contents: string }[] = []; let queuedCaseIds: string[] = []; -let queuedRequests: { caseId: string; lane: string; definitionId: number; sourceBranch: string }[] = []; +let queuedRequests: { + caseId: string; + lane: string; + definitionId: number; + sourceBranch: string; + variables?: Readonly>; +}[] = []; let deletedRefs: string[] = []; const HERE = dirname(fileURLToPath(import.meta.url)); @@ -171,7 +177,7 @@ vi.mock("../signals.js", async (importOriginal) => { const actual = await importOriginal(); return { ...actual, - verifyCandidateAudit: vi.fn(async (results: readonly FixtureBuildResult[]) => ({ + verifyCandidateAudit: vi.fn(async (_cases: unknown, results: readonly FixtureBuildResult[]) => ({ ok: true, results: results.map((result) => ({ ...result })), })), @@ -185,7 +191,13 @@ vi.mock("../runner.js", async (importOriginal) => { runFixtures: vi.fn( async ( _client: unknown, - requests: { caseId: string; lane: string; definitionId: number; sourceBranch: string }[], + requests: { + caseId: string; + lane: string; + definitionId: number; + sourceBranch: string; + variables?: Readonly>; + }[], ) => { mockCalls.push("runFixtures"); queuedCaseIds = requests.map((request) => request.caseId); @@ -271,6 +283,8 @@ describe("smoke-e2e index.main (happy path, candidate mode)", () => { "noop-target", "custom-safe-output", "multi-repo", + "runtime-model-queue", + "runtime-model-set-variable", ]); expect(queuedCaseIds).not.toContain("janitor"); expect(compiledCasePaths).toEqual([ @@ -279,7 +293,12 @@ describe("smoke-e2e index.main (happy path, candidate mode)", () => { "tests/safe-outputs/noop-target.md", "tests/smoke/custom-safe-output.md", "tests/smoke/multi-repo.md", + "tests/smoke/runtime-model-queue.md", + "tests/smoke/runtime-model-set-variable.md", ]); + expect( + queuedRequests.find((request) => request.caseId === "runtime-model-queue")?.variables, + ).toEqual({ ADO_AW_MODEL_AGENT_COPILOT: "gpt-6-luna" }); // Cleanup ordering: remote refs deleted BEFORE the local worktree is removed. expect(mockCalls.indexOf("deleteRemoteRefs")).toBeGreaterThanOrEqual(0); @@ -297,9 +316,11 @@ describe("smoke-e2e index.main (happy path, candidate mode)", () => { "refs/heads/ado-aw-smoke-candidate/630001/noop-target", "refs/heads/ado-aw-smoke-candidate/630001/custom-safe-output", "refs/heads/ado-aw-smoke-candidate/630001/multi-repo", + "refs/heads/ado-aw-smoke-candidate/630001/runtime-model-queue", + "refs/heads/ado-aw-smoke-candidate/630001/runtime-model-set-variable", ]); // Every case is staged to the SAME path — the ref is what distinguishes them. - expect(stagedWrites.length).toBe(5); + expect(stagedWrites.length).toBe(7); for (const write of stagedWrites) { expect(write.to).toBe(join(WORKTREE, "candidate", ".smoke", "pipeline.yml")); // The compiler emits no trigger keys once `on:` is stripped, and a @@ -329,7 +350,7 @@ describe("smoke-e2e index.main (happy path, candidate mode)", () => { const gitModule = await import("../git.js"); const resets = vi.mocked(gitModule.resetWorktree).mock.calls; - expect(resets.length).toBe(5); + expect(resets.length).toBe(7); for (const call of resets) { expect(call[0]).toMatchObject({ commitish: "basecommit" }); } @@ -456,6 +477,8 @@ describe("smoke-e2e index.main (per-case ref retention)", () => { "refs/heads/ado-aw-smoke-candidate/630001/noop-target", "refs/heads/ado-aw-smoke-candidate/630001/custom-safe-output", "refs/heads/ado-aw-smoke-candidate/630001/multi-repo", + "refs/heads/ado-aw-smoke-candidate/630001/runtime-model-queue", + "refs/heads/ado-aw-smoke-candidate/630001/runtime-model-set-variable", ]); expect(deletedRefs).not.toContain("refs/heads/ado-aw-smoke-candidate/630001/ado-proxy"); }); diff --git a/scripts/ado-script/src/compiler-smoke-e2e/__tests__/runner.test.ts b/scripts/ado-script/src/compiler-smoke-e2e/__tests__/runner.test.ts index 575119e05..886ce9b04 100644 --- a/scripts/ado-script/src/compiler-smoke-e2e/__tests__/runner.test.ts +++ b/scripts/ado-script/src/compiler-smoke-e2e/__tests__/runner.test.ts @@ -101,6 +101,49 @@ describe("runFixtures", () => { expect(outcome.results.every((r) => r.terminalProven)).toBe(true); }); + it("forwards queue variables unchanged", async () => { + let queuedOptions: Parameters[1] | undefined; + const client: FixtureBuildClient = { + async queueBuild(_definitionId, options) { + queuedOptions = options; + return { id: 103 }; + }, + async getBuild() { + return { + status: "completed", + result: "succeeded", + definition: { id: 1 }, + sourceBranch: "refs/heads/x", + sourceVersion: "sha", + }; + }, + async cancelBuild() {}, + async addBuildTags() {}, + buildUrl(buildId) { + return `https://example/${buildId}`; + }, + }; + const request = { + ...req("runtime-model-queue", 1), + variables: { ADO_AW_MODEL_AGENT_COPILOT: "auto" }, + }; + + const outcome = await runFixtures(client, [request], { + concurrency: 1, + timeoutMs: 10_000, + pollMs: 1, + log: () => {}, + sleepImpl: noopSleep, + }); + + expect(outcome.ok).toBe(true); + expect(queuedOptions).toEqual({ + sourceBranch: "refs/heads/x", + sourceVersion: "sha", + variables: { ADO_AW_MODEL_AGENT_COPILOT: "auto" }, + }); + }); + it("preserves declaration order in results regardless of completion order", async () => { const { client } = makeFakeClient({ queueResults: { 1: { ok: true, id: 201 }, 2: { ok: true, id: 202 } }, diff --git a/scripts/ado-script/src/compiler-smoke-e2e/__tests__/signals.test.ts b/scripts/ado-script/src/compiler-smoke-e2e/__tests__/signals.test.ts index 22e3134be..d67b7cf3d 100644 --- a/scripts/ado-script/src/compiler-smoke-e2e/__tests__/signals.test.ts +++ b/scripts/ado-script/src/compiler-smoke-e2e/__tests__/signals.test.ts @@ -36,6 +36,15 @@ const CASES: ResolvedCase[] = [ source: "tests/safe-outputs/canary.md", definitionId: 3006, }, + { + id: "runtime-model-queue", + lane: "agentic", + kind: "compiled", + modes: ["candidate"], + source: "tests/smoke/runtime-model-queue.md", + assertions: { requestedModels: { agent: "gpt-6-luna" } }, + definitionId: 3006, + }, ]; function result(overrides: Partial = {}): FixtureBuildResult { @@ -177,7 +186,7 @@ describe("verifyCandidateAudit", () => { }); const canary = result({ caseId: "canary" }); - const outcome = await verifyCandidateAudit([canary], options); + const outcome = await verifyCandidateAudit(CASES, [canary], options); expect(outcome).toEqual({ ok: true, results: [canary] }); expect(safeSpawnMock).toHaveBeenCalledWith( @@ -200,7 +209,11 @@ describe("verifyCandidateAudit", () => { stderrTruncated: false, }); - const outcome = await verifyCandidateAudit([result({ caseId: "canary" })], options); + const outcome = await verifyCandidateAudit( + CASES, + [result({ caseId: "canary" })], + options, + ); expect(outcome.ok).toBe(false); expect(outcome.results[0]?.message).toContain("***"); @@ -209,6 +222,7 @@ describe("verifyCandidateAudit", () => { it("fails closed without spawning when no successful canary build exists", async () => { const outcome = await verifyCandidateAudit( + CASES, [result({ caseId: "canary", status: "failed", result: "failed" })], options, ); @@ -216,4 +230,45 @@ describe("verifyCandidateAudit", () => { expect(outcome.ok).toBe(false); expect(safeSpawnMock).not.toHaveBeenCalled(); }); + + it("verifies the requested Agent model from audit metadata", async () => { + safeSpawnMock + .mockResolvedValueOnce({ + status: 0, + stdout: JSON.stringify({ + overview: { build_id: 42 }, + downloaded_files: [ + { path: "agent_outputs_42/agent-output.json" }, + { path: "analyzed_outputs_42/verdict.json" }, + { path: "safe_outputs/executed-safe-outputs.ndjson" }, + ], + }), + stderr: "", + timedOut: false, + stdoutTruncated: false, + stderrTruncated: false, + }) + .mockResolvedValueOnce({ + status: 0, + stdout: JSON.stringify({ + overview: { build_id: 43, aw_info: { model: "auto" } }, + }), + stderr: "", + timedOut: false, + stdoutTruncated: false, + stderrTruncated: false, + }); + + const outcome = await verifyCandidateAudit( + CASES, + [ + result({ caseId: "canary" }), + result({ caseId: "runtime-model-queue", buildId: 43 }), + ], + options, + ); + + expect(outcome.ok).toBe(false); + expect(outcome.results[1]?.message).toMatch(/expected "gpt-6-luna"/); + }); }); diff --git a/scripts/ado-script/src/compiler-smoke-e2e/ado-rest.ts b/scripts/ado-script/src/compiler-smoke-e2e/ado-rest.ts index 3e953b320..688cc54bb 100644 --- a/scripts/ado-script/src/compiler-smoke-e2e/ado-rest.ts +++ b/scripts/ado-script/src/compiler-smoke-e2e/ado-rest.ts @@ -249,14 +249,25 @@ export class AdoRest { */ async queueBuild( definitionId: number, - opts: { sourceBranch: string; sourceVersion: string }, + opts: { + sourceBranch: string; + sourceVersion: string; + variables?: Readonly>; + }, ): Promise<{ id: number }> { const path = this.projPath(`_apis/build/builds?api-version=7.1`); - const body = { + const body: Record = { definition: { id: definitionId }, sourceBranch: opts.sourceBranch, sourceVersion: opts.sourceVersion, }; + if (opts.variables && Object.keys(opts.variables).length > 0) { + // Build Queue's legacy `parameters` string is how `az pipelines run + // --variables` sends queue-time variables. A top-level `variables` + // object belongs to the Pipelines Runs API and is silently ignored by + // this endpoint. + body.parameters = JSON.stringify(opts.variables); + } const res = await this.request<{ id: number }>(path, { method: "POST", body }); if (!res) throw new Error(`queueBuild(${definitionId}) returned no body`); return res; diff --git a/scripts/ado-script/src/compiler-smoke-e2e/cases.ts b/scripts/ado-script/src/compiler-smoke-e2e/cases.ts index 0dc4cfa3e..7c1761247 100644 --- a/scripts/ado-script/src/compiler-smoke-e2e/cases.ts +++ b/scripts/ado-script/src/compiler-smoke-e2e/cases.ts @@ -62,6 +62,11 @@ export interface CaseAssertions { readonly pipelineText?: AgentCommandAssertion; /** Build tags the child run must carry, with `{buildId}` expanded to the child build id. */ readonly requiredBuildTags?: readonly string[]; + /** Requested runtime models recorded in the child's audit metadata. */ + readonly requestedModels?: { + readonly agent?: string; + readonly detection?: string; + }; } export interface SmokeLane { @@ -77,6 +82,8 @@ export interface SmokeCase { readonly modes: readonly CompilerSource[]; /** Repo-relative source path (`.md` for compiled, `.yml`/`.yaml` for raw). */ readonly source: string; + /** Non-secret variables supplied through the ADO queue-build API. */ + readonly queueVariables?: Readonly>; readonly assertions?: CaseAssertions; } @@ -162,6 +169,27 @@ function validateKindMatchesExtension(kind: CaseKind, source: string, caseId: st } } +function parseQueueVariables( + raw: unknown, + caseId: string, +): Readonly> | undefined { + if (raw === undefined) return undefined; + const obj = asRecord(raw, `case '${caseId}' queueVariables`); + const entries = Object.entries(obj); + if (entries.length === 0) { + fail(`case '${caseId}' queueVariables must not be empty`); + } + + const variables: Record = {}; + for (const [name, rawValue] of entries) { + if (!ENV_NAME_RE.test(name)) { + fail(`case '${caseId}' queue variable '${name}' must match ${ENV_NAME_RE}`); + } + variables[name] = asString(rawValue, `case '${caseId}' queueVariables.${name}`); + } + return variables; +} + function parseAssertions(raw: unknown, caseId: string): CaseAssertions | undefined { if (raw === undefined) return undefined; const obj = asRecord(raw, `case '${caseId}' assertions`); @@ -217,16 +245,41 @@ function parseAssertions(raw: unknown, caseId: string): CaseAssertions | undefin } } + let requestedModels: CaseAssertions["requestedModels"]; + if (obj.requestedModels !== undefined) { + const models = asRecord( + obj.requestedModels, + `case '${caseId}' assertions.requestedModels`, + ); + requestedModels = { + agent: + models.agent === undefined + ? undefined + : asString(models.agent, `case '${caseId}' assertions.requestedModels.agent`), + detection: + models.detection === undefined + ? undefined + : asString( + models.detection, + `case '${caseId}' assertions.requestedModels.detection`, + ), + }; + if (requestedModels.agent === undefined && requestedModels.detection === undefined) { + fail(`case '${caseId}' assertions.requestedModels must declare agent and/or detection`); + } + } + if ( agentCommand === undefined && pipelineText === undefined && - requiredBuildTags === undefined + requiredBuildTags === undefined && + requestedModels === undefined ) { fail( - `case '${caseId}' assertions must declare agentCommand, pipelineText and/or requiredBuildTags`, + `case '${caseId}' assertions must declare agentCommand, pipelineText, requiredBuildTags and/or requestedModels`, ); } - return { agentCommand, pipelineText, requiredBuildTags }; + return { agentCommand, pipelineText, requiredBuildTags, requestedModels }; } /** Expand `{buildId}` in a declared build tag. */ @@ -316,7 +369,15 @@ export function parseManifest(text: string): SmokeManifest { const source = validateSourcePath(entry.source, id); validateKindMatchesExtension(kind, source, id); - cases.push({ id, lane, kind, modes, source, assertions: parseAssertions(entry.assertions, id) }); + cases.push({ + id, + lane, + kind, + modes, + source, + queueVariables: parseQueueVariables(entry.queueVariables, id), + assertions: parseAssertions(entry.assertions, id), + }); } if (cases.length === 0) fail("cases must declare at least one case"); diff --git a/scripts/ado-script/src/compiler-smoke-e2e/index.ts b/scripts/ado-script/src/compiler-smoke-e2e/index.ts index 110499ba9..5afa09ba6 100644 --- a/scripts/ado-script/src/compiler-smoke-e2e/index.ts +++ b/scripts/ado-script/src/compiler-smoke-e2e/index.ts @@ -368,6 +368,7 @@ export async function main(): Promise { definitionId: entry.definitionId, sourceBranch: staged.get(entry.id)!.ref, sourceVersion: staged.get(entry.id)!.sha, + variables: entry.queueVariables, tags: [`smoke-case:${entry.id}`, `smoke-candidate:${config.buildId}`], })); @@ -386,7 +387,7 @@ export async function main(): Promise { const signalOutcome = await verifyCaseSignals(rest, resolved.cases, outcome.results); const auditOutcome = config.compilerSource === "candidate" - ? await verifyCandidateAudit(signalOutcome.results, { + ? await verifyCandidateAudit(resolved.cases, signalOutcome.results, { adoAwBin: config.adoAwBin, cwd: config.sourcesDirectory, orgUrl: config.orgUrl, diff --git a/scripts/ado-script/src/compiler-smoke-e2e/runner.ts b/scripts/ado-script/src/compiler-smoke-e2e/runner.ts index 4e0370921..512e0f3b0 100644 --- a/scripts/ado-script/src/compiler-smoke-e2e/runner.ts +++ b/scripts/ado-script/src/compiler-smoke-e2e/runner.ts @@ -50,7 +50,14 @@ export interface PolledBuild { /** The minimal ADO Build surface this state machine needs. */ export interface FixtureBuildClient { - queueBuild(definitionId: number, opts: { sourceBranch: string; sourceVersion: string }): Promise<{ id: number }>; + queueBuild( + definitionId: number, + opts: { + sourceBranch: string; + sourceVersion: string; + variables?: Readonly>; + }, + ): Promise<{ id: number }>; getBuild(buildId: number): Promise; cancelBuild(buildId: number): Promise; buildUrl(buildId: number): string; @@ -70,6 +77,8 @@ export interface FixtureBuildRequest { definitionId: number; sourceBranch: string; sourceVersion: string; + /** Non-secret ADO variables supplied when the child build is queued. */ + variables?: Readonly>; /** Tags applied to the queued run so it is identifiable in a shared lane's history. */ tags?: readonly string[]; } @@ -312,6 +321,7 @@ export async function runFixtures( const build = await client.queueBuild(req.definitionId, { sourceBranch: req.sourceBranch, sourceVersion: req.sourceVersion, + variables: req.variables, }); queued.push({ index: i, buildId: build.id, start }); results[i] = { diff --git a/scripts/ado-script/src/compiler-smoke-e2e/signals.ts b/scripts/ado-script/src/compiler-smoke-e2e/signals.ts index 80b270094..17d25e8d5 100644 --- a/scripts/ado-script/src/compiler-smoke-e2e/signals.ts +++ b/scripts/ado-script/src/compiler-smoke-e2e/signals.ts @@ -77,8 +77,20 @@ export async function verifyCaseSignals( }; } -/** Audit one completed candidate child through the released CLI contract. */ +interface CandidateAuditReport { + overview?: { + build_id?: number; + aw_info?: { + model?: string | null; + detection_model?: string | null; + }; + }; + downloaded_files?: { path?: string }[]; +} + +/** Audit candidate children through the released CLI contract. */ export async function verifyCandidateAudit( + cases: readonly ResolvedCase[], results: readonly FixtureBuildResult[], options: { adoAwBin: string; @@ -89,78 +101,106 @@ export async function verifyCandidateAudit( timeoutMs: number; }, ): Promise { - const target = results.find( + const canary = results.find( (result) => result.caseId === "canary" && result.status === "succeeded" && result.buildId !== undefined, ); - if (!target?.buildId) return { ok: false, results: results.map((result) => ({ ...result })) }; + if (!canary?.buildId) return { ok: false, results: results.map((result) => ({ ...result })) }; + + const casesById = new Map(cases.map((entry) => [entry.id, entry])); + const targets = results.filter((result) => { + if (result.status !== "succeeded" || result.buildId === undefined) return false; + return ( + result.caseId === "canary" || + casesById.get(result.caseId)?.assertions?.requestedModels !== undefined + ); + }); + const verified = results.map((result) => ({ ...result })); - const outputDir = await mkdtemp(join(tmpdir(), "ado-aw-smoke-audit-")); - try { - const outcome = await safeSpawn({ - cmd: options.adoAwBin, - args: [ - "audit", - String(target.buildId), - "--json", - "--no-cache", - "--output", - outputDir, - "--org", - options.orgUrl, - "--project", - options.project, - ], - cwd: options.cwd, - env: { AZURE_DEVOPS_EXT_PAT: options.token }, - timeoutMs: options.timeoutMs, - }); + for (const target of targets) { + const outputDir = await mkdtemp(join(tmpdir(), "ado-aw-smoke-audit-")); let error: string | undefined; - if (outcome.timedOut || outcome.status !== 0) { - error = `exit=${outcome.status ?? "signal"} timedOut=${outcome.timedOut}; stderr=${redact(outcome.stderr, [options.token])}`; - } else { - try { - const audit = JSON.parse(outcome.stdout) as { - overview?: { build_id?: number }; - downloaded_files?: { path?: string }[]; - }; - const paths = - audit.downloaded_files - ?.flatMap((file) => file.path ?? []) - .map((path) => path.replaceAll("\\", "/")) ?? []; - const expectedRoots = [ - `agent_outputs_${target.buildId}/`, - `analyzed_outputs_${target.buildId}/`, - "safe_outputs/", - ]; - const missingRoots = expectedRoots.filter( - (root) => !paths.some((path) => path.startsWith(root)), - ); - if (audit.overview?.build_id !== target.buildId || missingRoots.length > 0) { - error = - `JSON report did not contain the child build id and every published artifact family; ` + - `missing roots: ${missingRoots.join(", ") || ""}`; + try { + const outcome = await safeSpawn({ + cmd: options.adoAwBin, + args: [ + "audit", + String(target.buildId), + "--json", + "--no-cache", + "--output", + outputDir, + "--org", + options.orgUrl, + "--project", + options.project, + ], + cwd: options.cwd, + env: { AZURE_DEVOPS_EXT_PAT: options.token }, + timeoutMs: options.timeoutMs, + }); + if (outcome.timedOut || outcome.status !== 0) { + error = `exit=${outcome.status ?? "signal"} timedOut=${outcome.timedOut}; stderr=${redact(outcome.stderr, [options.token])}`; + } else { + try { + const audit = JSON.parse(outcome.stdout) as CandidateAuditReport; + if (audit.overview?.build_id !== target.buildId) { + error = `JSON report build id was ${audit.overview?.build_id ?? ""}; expected ${target.buildId}`; + } + + if (!error && target.caseId === "canary") { + const paths = + audit.downloaded_files + ?.flatMap((file) => file.path ?? []) + .map((path) => path.replaceAll("\\", "/")) ?? []; + const expectedRoots = [ + `agent_outputs_${target.buildId}/`, + `analyzed_outputs_${target.buildId}/`, + "safe_outputs/", + ]; + const missingRoots = expectedRoots.filter( + (root) => !paths.some((path) => path.startsWith(root)), + ); + if (missingRoots.length > 0) { + error = + `JSON report did not contain every published artifact family; ` + + `missing roots: ${missingRoots.join(", ")}`; + } + } + + const requested = casesById.get(target.caseId)?.assertions?.requestedModels; + if (!error && requested?.agent !== undefined) { + const actual = audit.overview?.aw_info?.model; + if (actual !== requested.agent) { + error = `requested Agent model was ${JSON.stringify(actual ?? null)}; expected ${JSON.stringify(requested.agent)}`; + } + } + if (!error && requested?.detection !== undefined) { + const actual = audit.overview?.aw_info?.detection_model; + if (actual !== requested.detection) { + error = `requested Detection model was ${JSON.stringify(actual ?? null)}; expected ${JSON.stringify(requested.detection)}`; + } + } + } catch (parseError) { + error = `invalid JSON report: ${parseError instanceof Error ? parseError.message : String(parseError)}`; } - } catch (parseError) { - error = `invalid JSON report: ${parseError instanceof Error ? parseError.message : String(parseError)}`; } + } finally { + await rm(outputDir, { recursive: true, force: true }); } - if (!error) return { ok: true, results: results.map((result) => ({ ...result })) }; - return { - ok: false, - results: results.map((result) => - result.caseId === target.caseId - ? { - ...result, - status: "failed", - message: - `candidate audit contract failed for build #${target.buildId} (${target.url ?? "URL unavailable"}); ` + - `expected artifacts agent_outputs_${target.buildId}, analyzed_outputs_${target.buildId}, and safe_outputs: ${error}`, - } - : { ...result }, - ), - }; - } finally { - await rm(outputDir, { recursive: true, force: true }); + if (error) { + const index = verified.findIndex((result) => result.caseId === target.caseId); + verified[index] = { + ...verified[index]!, + status: "failed", + message: + `candidate audit verification failed for build #${target.buildId} (${target.url ?? "URL unavailable"}): ${error}`, + }; + } } + + return { + ok: verified.every((result) => result.status === "succeeded"), + results: verified, + }; } diff --git a/scripts/ado-script/src/copilot-controller/index.test.ts b/scripts/ado-script/src/copilot-controller/index.test.ts new file mode 100644 index 000000000..c124bbb46 --- /dev/null +++ b/scripts/ado-script/src/copilot-controller/index.test.ts @@ -0,0 +1,131 @@ +import { mkdtempSync, readFileSync, writeFileSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { afterEach, describe, expect, it, vi } from "vitest"; + +import { main } from "./index.js"; + +function writeRequest(directory: string): string { + const path = join(directory, "request.json"); + writeFileSync( + path, + JSON.stringify({ + schema_version: 2, + document_kind: "request", + role: "agent", + command: "/tmp/awf-tools/copilot", + prompt_path: "/tmp/awf-tools/agent-prompt.md", + mcp_config_path: null, + args: ["--no-ask-user"], + explicit_model: null, + }), + ); + return path; +} + +afterEach(() => { + vi.unstubAllEnvs(); + vi.restoreAllMocks(); +}); + +describe("copilot controller", () => { + it("prepares sandbox execution and host result from the same model decision", async () => { + const directory = mkdtempSync(join(tmpdir(), "copilot-controller-")); + const requestPath = writeRequest(directory); + const preparedPath = join(directory, "prepared.json"); + const resultPath = join(directory, "result.json"); + vi.stubEnv("ADO_AW_MODEL_AGENT_COPILOT", "runtime-model"); + + await expect( + main(["prepare", requestPath, preparedPath, resultPath]), + ).resolves.toBe(0); + const prepared = JSON.parse(readFileSync(preparedPath, "utf8")); + const result = JSON.parse(readFileSync(resultPath, "utf8")); + expect(prepared).toMatchObject({ + schema_version: 2, + document_kind: "prepared", + requested_model: "runtime-model", + }); + expect(result).toEqual({ + schema_version: 2, + document_kind: "result", + role: "agent", + requested_model: prepared.requested_model, + }); + }); + + it("commits the trusted result before exposing a prepared sandbox document", async () => { + const directory = mkdtempSync(join(tmpdir(), "copilot-controller-")); + const requestPath = writeRequest(directory); + const preparedPath = join(directory, "missing", "prepared.json"); + const resultPath = join(directory, "result.json"); + const error = vi.spyOn(console, "error").mockImplementation(() => undefined); + + await expect( + main(["prepare", requestPath, preparedPath, resultPath]), + ).resolves.toBe(1); + expect(JSON.parse(readFileSync(resultPath, "utf8"))).toMatchObject({ + document_kind: "result", + role: "agent", + }); + expect(error).toHaveBeenCalledWith( + expect.stringContaining("copilot-controller:"), + ); + }); + + it("reads only a matching strict result", async () => { + const directory = mkdtempSync(join(tmpdir(), "copilot-controller-")); + const resultPath = join(directory, "result.json"); + writeFileSync( + resultPath, + JSON.stringify({ + schema_version: 2, + document_kind: "result", + role: "agent", + requested_model: "gpt-test", + }), + ); + const write = vi.spyOn(process.stdout, "write").mockImplementation( + ((_chunk: unknown, callback?: (error?: Error | null) => void) => { + callback?.(); + return true; + }) as typeof process.stdout.write, + ); + await expect(main(["read-result", resultPath, "agent"])).resolves.toBe(0); + expect(write).toHaveBeenCalledWith("gpt-test", expect.any(Function)); + }); + + it("rejects role mismatch, missing results, and stdout failures", async () => { + const directory = mkdtempSync(join(tmpdir(), "copilot-controller-")); + const resultPath = join(directory, "result.json"); + writeFileSync( + resultPath, + JSON.stringify({ + schema_version: 2, + document_kind: "result", + role: "agent", + requested_model: "gpt-test", + }), + ); + const error = vi.spyOn(console, "error").mockImplementation(() => undefined); + await expect(main(["read-result", resultPath, "detection"])).resolves.toBe(1); + await expect( + main(["read-result", join(directory, "missing.json"), "agent"]), + ).resolves.toBe(1); + + vi.spyOn(process.stdout, "write").mockImplementation( + ((_chunk: unknown, callback?: (error?: Error | null) => void) => { + callback?.(new Error("EPIPE")); + return false; + }) as typeof process.stdout.write, + ); + await expect(main(["read-result", resultPath, "agent"])).resolves.toBe(1); + expect(error).toHaveBeenCalledWith("copilot-controller: EPIPE"); + }); + + it("does not expose sandbox run mode", async () => { + const error = vi.spyOn(console, "error").mockImplementation(() => undefined); + await expect(main(["run", "/tmp/prepared.json"])).resolves.toBe(2); + expect(error).toHaveBeenCalledWith(expect.stringContaining("usage:")); + }); +}); diff --git a/scripts/ado-script/src/copilot-controller/index.ts b/scripts/ado-script/src/copilot-controller/index.ts new file mode 100644 index 000000000..f731867eb --- /dev/null +++ b/scripts/ado-script/src/copilot-controller/index.ts @@ -0,0 +1,70 @@ +import { readFileSync } from "node:fs"; +import { resolve } from "node:path"; +import { fileURLToPath } from "node:url"; + +import { + parseInvocationRequest, + parseInvocationResult, + prepareInvocation, + writeJsonAtomic, +} from "../copilot-shared/protocol.js"; + +async function writeStdout(value: string): Promise { + await new Promise((resolveWrite, rejectWrite) => { + process.stdout.write(value, (error) => { + if (error) rejectWrite(error); + else resolveWrite(); + }); + }); +} + +export async function main(argv: string[]): Promise { + if (argv[0] === "prepare" && argv.length === 4) { + try { + const request = parseInvocationRequest(readFileSync(argv[1]!, "utf8")); + const { prepared, result } = prepareInvocation(request, process.env); + // The sandbox document is the commit point: trusted preflight removes + // any prior result and cannot start AWF unless both writes succeed. + writeJsonAtomic(argv[3]!, result); + writeJsonAtomic(argv[2]!, prepared); + return 0; + } catch (error) { + const message = error instanceof Error ? error.message : "unknown error"; + console.error(`copilot-controller: ${message}`); + return 1; + } + } + if (argv[0] === "read-result" && argv.length === 3) { + try { + const result = parseInvocationResult(readFileSync(argv[1]!, "utf8")); + if (result.role !== argv[2]) { + throw new Error( + `invocation result role '${result.role}' does not match expected role '${argv[2]}'`, + ); + } + await writeStdout(result.requested_model ?? ""); + return 0; + } catch (error) { + const message = error instanceof Error ? error.message : "unknown error"; + console.error(`copilot-controller: ${message}`); + return 1; + } + } + console.error( + "usage: copilot-controller prepare | read-result ", + ); + return 2; +} + +const invokedPath = process.argv[1] ? resolve(process.argv[1]) : ""; +if (invokedPath === fileURLToPath(import.meta.url)) { + void main(process.argv.slice(2)) + .then((code) => { + process.exitCode = code; + }) + .catch((error: unknown) => { + const message = error instanceof Error ? error.message : "unknown error"; + console.error(`copilot-controller: unexpected failure: ${message}`); + process.exitCode = 1; + }); +} diff --git a/scripts/ado-script/src/copilot-runner/index.test.ts b/scripts/ado-script/src/copilot-runner/index.test.ts new file mode 100644 index 000000000..7a34e32fc --- /dev/null +++ b/scripts/ado-script/src/copilot-runner/index.test.ts @@ -0,0 +1,173 @@ +import { EventEmitter } from "node:events"; +import { mkdtempSync, writeFileSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { describe, expect, it, vi } from "vitest"; + +import { + main, + runInvocation, + type RunDependencies, +} from "./index.js"; +import type { PreparedInvocation } from "../copilot-shared/protocol.js"; + +function prepared( + overrides: Partial = {}, +): PreparedInvocation { + return { + schema_version: 2, + document_kind: "prepared", + role: "agent", + command: "/tmp/awf-tools/copilot", + prompt_path: "/tmp/awf-tools/agent-prompt.md", + mcp_config_path: null, + args: ["--no-ask-user"], + requested_model: "prepared-model", + ...overrides, + }; +} + +function childProcess() { + const child = new EventEmitter() as EventEmitter & { + killed: boolean; + kill: ReturnType; + }; + child.killed = false; + child.kill = vi.fn(); + return child; +} + +describe("copilot runner", () => { + it("removes runner and prepared document before spawning", async () => { + const events: string[] = []; + const child = childProcess(); + let spawnedEnv: NodeJS.ProcessEnv | undefined; + const dependencies: RunDependencies = { + readFile: () => "prompt", + removeFile: (path) => events.push(`remove:${path}`), + spawn: (( + _command: string, + _args: readonly string[], + options: { env: NodeJS.ProcessEnv; stdio: "inherit" }, + ) => { + events.push("spawn"); + spawnedEnv = options.env; + queueMicrotask(() => child.emit("close", 23, null)); + return child; + }) as never, + }; + + await expect( + runInvocation( + prepared(), + "/tmp/prepared.json", + "/tmp/copilot-runner.js", + { + ADO_AW_MODEL_AGENT_COPILOT: "conflicting-model", + ADO_AW_DEFAULT_MODEL_COPILOT: "conflicting-default", + }, + dependencies, + ), + ).resolves.toBe(23); + expect(events).toEqual([ + "remove:/tmp/prepared.json", + "remove:/tmp/copilot-runner.js", + "spawn", + ]); + expect(spawnedEnv).toEqual({ COPILOT_MODEL: "prepared-model" }); + }); + + it("fails closed when self-removal fails", async () => { + const spawn = vi.fn(); + await expect( + runInvocation( + prepared(), + "/tmp/prepared.json", + "/tmp/copilot-runner.js", + {}, + { + readFile: () => "prompt", + removeFile: () => { + throw new Error("EPERM"); + }, + spawn: spawn as never, + }, + ), + ).rejects.toThrow("EPERM"); + expect(spawn).not.toHaveBeenCalled(); + }); + + it("returns deterministic spawn failure and forwards termination signals", async () => { + const child = childProcess(); + const error = vi.spyOn(console, "error").mockImplementation(() => undefined); + const failure = runInvocation( + prepared(), + "/tmp/prepared.json", + "/tmp/copilot-runner.js", + {}, + { + readFile: () => "prompt", + removeFile: () => undefined, + spawn: (() => { + queueMicrotask(() => child.emit("error", new Error("ENOENT"))); + return child; + }) as never, + }, + ); + await expect(failure).resolves.toBe(1); + expect(error).toHaveBeenCalledWith( + "copilot-runner: failed to start Copilot: ENOENT", + ); + error.mockRestore(); + + const signaledChild = childProcess(); + const listenersBefore = process.listenerCount("SIGTERM"); + signaledChild.kill = vi.fn((signal: NodeJS.Signals) => { + queueMicrotask(() => signaledChild.emit("close", null, signal)); + return true; + }); + const signaled = runInvocation( + prepared(), + "/tmp/prepared.json", + "/tmp/copilot-runner.js", + {}, + { + readFile: () => "prompt", + removeFile: () => undefined, + spawn: (() => signaledChild) as never, + }, + ); + process.emit("SIGTERM", "SIGTERM"); + await expect(signaled).resolves.toBe(143); + expect(signaledChild.kill).toHaveBeenCalledWith("SIGTERM"); + expect(process.listenerCount("SIGTERM")).toBe(listenersBefore); + }); + + it("does not expose controller modes", async () => { + const error = vi.spyOn(console, "error").mockImplementation(() => undefined); + await expect(main(["prepare", "a", "b", "c"], "/tmp/runner.js")).resolves.toBe(2); + await expect(main(["read-result", "a", "agent"], "/tmp/runner.js")).resolves.toBe(2); + expect(error).toHaveBeenCalledWith( + "usage: copilot-runner run ", + ); + error.mockRestore(); + }); + + it("reports unreadable and malformed prepared invocations", async () => { + const directory = mkdtempSync(join(tmpdir(), "copilot-runner-")); + const missingPath = join(directory, "missing.json"); + const malformedPath = join(directory, "malformed.json"); + writeFileSync(malformedPath, "not-json"); + const error = vi.spyOn(console, "error").mockImplementation(() => undefined); + + await expect(main(["run", missingPath], "/tmp/runner.js")).resolves.toBe(1); + await expect(main(["run", malformedPath], "/tmp/runner.js")).resolves.toBe(1); + expect(error).toHaveBeenCalledWith( + expect.stringContaining("copilot-runner:"), + ); + expect(error).toHaveBeenCalledWith( + "copilot-runner: prepared invocation is not valid JSON", + ); + error.mockRestore(); + }); +}); diff --git a/scripts/ado-script/src/copilot-runner/index.ts b/scripts/ado-script/src/copilot-runner/index.ts new file mode 100644 index 000000000..29cdc3cc8 --- /dev/null +++ b/scripts/ado-script/src/copilot-runner/index.ts @@ -0,0 +1,119 @@ +import { spawn, type ChildProcess } from "node:child_process"; +import { readFileSync, unlinkSync } from "node:fs"; +import { constants } from "node:os"; +import { resolve } from "node:path"; +import { fileURLToPath } from "node:url"; + +import { + buildChildEnvironment, + buildCopilotArgs, + parsePreparedInvocation, + type PreparedInvocation, +} from "../copilot-shared/protocol.js"; + +interface SpawnLike { + ( + command: string, + args: readonly string[], + options: { + env: NodeJS.ProcessEnv; + stdio: "inherit"; + }, + ): ChildProcess; +} + +export interface RunDependencies { + spawn: SpawnLike; + readFile(path: string): string; + removeFile(path: string): void; +} + +const DEFAULT_DEPENDENCIES: RunDependencies = { + spawn, + readFile: (path) => readFileSync(path, "utf8"), + removeFile: unlinkSync, +}; + +function signalExitCode(signal: NodeJS.Signals): number { + return 128 + (constants.signals[signal] ?? 0); +} + +export async function runInvocation( + document: PreparedInvocation, + preparedPath: string, + runnerPath: string, + env: NodeJS.ProcessEnv = process.env, + dependencies: RunDependencies = DEFAULT_DEPENDENCIES, +): Promise { + const prompt = dependencies.readFile(document.prompt_path); + const args = buildCopilotArgs(document, prompt); + dependencies.removeFile(preparedPath); + dependencies.removeFile(runnerPath); + const child = dependencies.spawn(document.command, args, { + env: buildChildEnvironment(env, document.requested_model), + stdio: "inherit", + }); + + return await new Promise((resolveExit) => { + let settled = false; + const settle = (code: number) => { + if (settled) return; + settled = true; + for (const signal of forwardedSignals) { + process.off(signal, handlers[signal]); + } + resolveExit(code); + }; + const forwardedSignals: NodeJS.Signals[] = ["SIGINT", "SIGTERM", "SIGHUP"]; + const handlers = Object.fromEntries( + forwardedSignals.map((signal) => [ + signal, + () => { + if (!child.killed) child.kill(signal); + }, + ]), + ) as Record void>; + for (const signal of forwardedSignals) { + process.on(signal, handlers[signal]); + } + child.once("error", (error) => { + const message = error instanceof Error ? error.message : "unknown error"; + console.error(`copilot-runner: failed to start Copilot: ${message}`); + settle(1); + }); + child.once("close", (code, signal) => { + settle(code ?? (signal ? signalExitCode(signal) : 1)); + }); + }); +} + +export async function main( + argv: string[], + runnerPath = fileURLToPath(import.meta.url), +): Promise { + if (argv[0] !== "run" || argv.length !== 2) { + console.error("usage: copilot-runner run "); + return 2; + } + try { + const document = parsePreparedInvocation(readFileSync(argv[1]!, "utf8")); + return await runInvocation(document, argv[1]!, runnerPath); + } catch (error) { + const message = error instanceof Error ? error.message : "unknown error"; + console.error(`copilot-runner: ${message}`); + return 1; + } +} + +const invokedPath = process.argv[1] ? resolve(process.argv[1]) : ""; +if (invokedPath === fileURLToPath(import.meta.url)) { + void main(process.argv.slice(2), invokedPath) + .then((code) => { + process.exitCode = code; + }) + .catch((error: unknown) => { + const message = error instanceof Error ? error.message : "unknown error"; + console.error(`copilot-runner: unexpected failure: ${message}`); + process.exitCode = 1; + }); +} diff --git a/scripts/ado-script/src/copilot-shared/protocol.test.ts b/scripts/ado-script/src/copilot-shared/protocol.test.ts new file mode 100644 index 000000000..429b5b24c --- /dev/null +++ b/scripts/ado-script/src/copilot-shared/protocol.test.ts @@ -0,0 +1,308 @@ +import { mkdtempSync, readdirSync, readFileSync, statSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { describe, expect, it } from "vitest"; + +import { + buildChildEnvironment, + buildCopilotArgs, + parseInvocationRequest, + parseInvocationResult, + parsePreparedInvocation, + prepareInvocation, + resolveRequestedModel, + writeJsonAtomic, + type InvocationRequest, + type PreparedInvocation, +} from "./protocol.js"; + +function request( + overrides: Partial = {}, +): InvocationRequest { + return { + schema_version: 2, + document_kind: "request", + role: "agent", + command: "/tmp/awf-tools/copilot", + prompt_path: "/tmp/awf-tools/agent-prompt.md", + mcp_config_path: "/tmp/awf-tools/mcp-config.json", + args: ["--disable-builtin-mcps", "--allow-tool", "shell(cat *)"], + explicit_model: null, + ...overrides, + }; +} + +function prepared( + overrides: Partial = {}, +): PreparedInvocation { + return { + schema_version: 2, + document_kind: "prepared", + role: "agent", + command: "/tmp/awf-tools/copilot", + prompt_path: "/tmp/awf-tools/agent-prompt.md", + mcp_config_path: "/tmp/awf-tools/mcp-config.json", + args: ["--disable-builtin-mcps", "--allow-tool", "shell(cat *)"], + requested_model: null, + ...overrides, + }; +} + +describe("Copilot invocation protocol", () => { + it("parses strict request, prepared, and result documents", () => { + expect(parseInvocationRequest(JSON.stringify(request()))).toEqual(request()); + expect(parsePreparedInvocation(JSON.stringify(prepared()))).toEqual(prepared()); + expect( + parseInvocationResult( + JSON.stringify({ + schema_version: 2, + document_kind: "result", + role: "detection", + requested_model: "detector", + }), + ), + ).toEqual({ + schema_version: 2, + document_kind: "result", + role: "detection", + requested_model: "detector", + }); + }); + + it.each([ + [{ ...request(), schema_version: 1 }, "unsupported invocation request schema"], + [{ ...request(), document_kind: "prepared" }, "must be 'request'"], + [{ ...request(), unexpected: true }, "unknown field 'unexpected'"], + [{ ...request(), role: "other" }, "must be 'agent' or 'detection'"], + [{ ...request(), command: "copilot;sh" }, "field 'command' is invalid"], + [{ ...request(), command: "bin/copilot" }, "field 'command' is invalid"], + [{ ...request(), command: "." }, "field 'command' is invalid"], + [{ ...request(), command: ".." }, "field 'command' is invalid"], + [{ ...request(), command: "/tmp/../copilot" }, "field 'command' is invalid"], + [{ ...request(), command: "/tmp//copilot" }, "field 'command' is invalid"], + [{ ...request(), command: "/tmp/copilot/" }, "field 'command' is invalid"], + [{ ...request(), prompt_path: "relative.md" }, "field 'prompt_path' is invalid"], + [ + { ...request(), mcp_config_path: "/tmp/../mcp.json" }, + "field 'mcp_config_path' is invalid", + ], + [{ ...request(), args: [1] }, "field 'args' must be an array of strings"], + [{ ...request(), explicit_model: "bad model" }, "field 'explicit_model' is invalid"], + ])("rejects malformed requests %#", (value, message) => { + expect(() => parseInvocationRequest(JSON.stringify(value))).toThrow(message); + }); + + it("rejects request/prepared document-kind confusion", () => { + expect(() => parsePreparedInvocation(JSON.stringify(request()))).toThrow( + "unknown field 'explicit_model'", + ); + expect(() => parseInvocationRequest(JSON.stringify(prepared()))).toThrow( + "unknown field 'requested_model'", + ); + }); + + it.each([ + [{ ...prepared(), schema_version: 1 }, "unsupported prepared invocation schema"], + [{ ...prepared(), document_kind: "request" }, "must be 'prepared'"], + [{ ...prepared(), unexpected: true }, "unknown field 'unexpected'"], + [{ ...prepared(), requested_model: "bad model" }, "field 'requested_model' is invalid"], + ])("rejects malformed prepared invocations %#", (value, message) => { + expect(() => parsePreparedInvocation(JSON.stringify(value))).toThrow(message); + }); + + it.each([ + [ + { + schema_version: 1, + document_kind: "result", + role: "agent", + requested_model: null, + }, + "unsupported invocation result schema", + ], + [ + { + schema_version: 2, + document_kind: "prepared", + role: "agent", + requested_model: null, + }, + "must be 'result'", + ], + [ + { + schema_version: 2, + document_kind: "result", + role: "agent", + requested_model: null, + unexpected: true, + }, + "unknown field 'unexpected'", + ], + ])("rejects malformed invocation results %#", (value, message) => { + expect(() => parseInvocationResult(JSON.stringify(value))).toThrow(message); + }); + + it("writes JSON atomically without leaving a temporary sibling", () => { + const directory = mkdtempSync(join(tmpdir(), "copilot-protocol-")); + const path = join(directory, "result.json"); + writeJsonAtomic(path, { + schema_version: 2, + document_kind: "result", + role: "agent", + requested_model: null, + }); + expect(JSON.parse(readFileSync(path, "utf8"))).toMatchObject({ + document_kind: "result", + role: "agent", + }); + expect(readdirSync(directory)).toEqual(["result.json"]); + if (process.platform !== "win32") { + expect(statSync(path).mode & 0o777).toBe(0o600); + } + }); +}); + +describe("model preparation", () => { + it("keeps explicit model authoritative", () => { + expect( + resolveRequestedModel(request({ explicit_model: "frontmatter-model" }), { + ADO_AW_MODEL_AGENT_COPILOT: "role-model", + ADO_AW_DEFAULT_MODEL_COPILOT: "default-model", + }), + ).toBe("frontmatter-model"); + }); + + it("uses role-specific then shared values and ignores unresolved macros", () => { + expect( + resolveRequestedModel(request(), { + ADO_AW_MODEL_AGENT_COPILOT: "role-model", + ADO_AW_DEFAULT_MODEL_COPILOT: "default-model", + }), + ).toBe("role-model"); + expect( + resolveRequestedModel(request({ role: "detection" }), { + ADO_AW_MODEL_DETECTION_COPILOT: "", + ADO_AW_DEFAULT_MODEL_COPILOT: "default-model", + }), + ).toBe("default-model"); + expect( + resolveRequestedModel(request(), { + ADO_AW_MODEL_AGENT_COPILOT: "$(ADO_AW_MODEL_AGENT_COPILOT)", + ADO_AW_DEFAULT_MODEL_COPILOT: "default-model", + }), + ).toBe("default-model"); + expect( + resolveRequestedModel(request(), { + ADO_AW_MODEL_AGENT_COPILOT: "$(ADO_AW_MODEL_AGENT_COPILOT)", + ADO_AW_DEFAULT_MODEL_COPILOT: "$(ADO_AW_DEFAULT_MODEL_COPILOT)", + }), + ).toBeNull(); + expect( + resolveRequestedModel(request(), { + ADO_AW_DEFAULT_MODEL_COPILOT: "", + }), + ).toBeNull(); + }); + + it("rejects invalid runtime values without printing them", () => { + expect(() => + resolveRequestedModel(request(), { + ADO_AW_MODEL_AGENT_COPILOT: "secret value", + }), + ).toThrow("contains invalid characters"); + try { + resolveRequestedModel(request(), { + ADO_AW_MODEL_AGENT_COPILOT: "secret value", + }); + } catch (error) { + expect(String(error)).not.toContain("secret value"); + } + }); + + it("emits identical requested models in prepared and result documents", () => { + const { prepared: execution, result } = prepareInvocation(request(), { + ADO_AW_MODEL_AGENT_COPILOT: "runtime-model", + }); + expect(execution.requested_model).toBe("runtime-model"); + expect(result.requested_model).toBe(execution.requested_model); + expect(execution).not.toHaveProperty("explicit_model"); + expect(execution).not.toHaveProperty("result_path"); + }); +}); + +describe("argv and child environment", () => { + it("preserves prompt, MCP config, and authored values as argv elements", () => { + expect(buildCopilotArgs(prepared(), "line one\nline two")).toEqual([ + "--prompt=line one\nline two", + "--additional-mcp-config", + "@/tmp/awf-tools/mcp-config.json", + "--disable-builtin-mcps", + "--allow-tool", + "shell(cat *)", + ]); + }); + + it("keeps option-looking prompt text inside the attached prompt argument", () => { + expect(buildCopilotArgs(prepared(), "--model=attacker")).toEqual([ + "--prompt=--model=attacker", + "--additional-mcp-config", + "@/tmp/awf-tools/mcp-config.json", + "--disable-builtin-mcps", + "--allow-tool", + "shell(cat *)", + ]); + }); + + it("omits MCP arguments when no MCP config is prepared", () => { + expect( + buildCopilotArgs(prepared({ mcp_config_path: null }), "prompt"), + ).toEqual([ + "--prompt=prompt", + "--disable-builtin-mcps", + "--allow-tool", + "shell(cat *)", + ]); + }); + + it("rejects NUL bytes in prompt content", () => { + expect(() => buildCopilotArgs(prepared(), "before\0after")).toThrow( + "prompt contains an invalid NUL byte", + ); + }); + + it("uses only the prepared model and removes runtime selector variables", () => { + const original = { + KEEP: "yes", + COPILOT_MODEL: "old", + ADO_AW_MODEL_AGENT_COPILOT: "conflicting", + ADO_AW_DEFAULT_MODEL_COPILOT: "conflicting-default", + }; + expect(buildChildEnvironment(original, "selected")).toEqual({ + KEEP: "yes", + COPILOT_MODEL: "selected", + }); + expect(original.COPILOT_MODEL).toBe("old"); + }); + + it("removes a stale Copilot model when no model was prepared", () => { + expect( + buildChildEnvironment( + { + KEEP: "yes", + COPILOT_MODEL: "stale", + ADO_AW_MODEL_DETECTION_COPILOT: "conflicting", + }, + null, + ), + ).toEqual({ KEEP: "yes" }); + }); + + it("accepts a validated bare executable command", () => { + expect( + parsePreparedInvocation( + JSON.stringify(prepared({ command: "copilot" })), + ).command, + ).toBe("copilot"); + }); +}); diff --git a/scripts/ado-script/src/copilot-shared/protocol.ts b/scripts/ado-script/src/copilot-shared/protocol.ts new file mode 100644 index 000000000..b39b41180 --- /dev/null +++ b/scripts/ado-script/src/copilot-shared/protocol.ts @@ -0,0 +1,385 @@ +import { renameSync, writeFileSync } from "node:fs"; + +export const SCHEMA_VERSION = 2; + +const MODEL_PATTERN = /^[A-Za-z0-9._:-]+$/; +const COMMAND_PATTERN = /^[A-Za-z0-9._/-]+$/; +const REQUEST_KEYS = new Set([ + "schema_version", + "document_kind", + "role", + "command", + "prompt_path", + "mcp_config_path", + "args", + "explicit_model", +]); +const PREPARED_KEYS = new Set([ + "schema_version", + "document_kind", + "role", + "command", + "prompt_path", + "mcp_config_path", + "args", + "requested_model", +]); +const RESULT_KEYS = new Set([ + "schema_version", + "document_kind", + "role", + "requested_model", +]); + +export type InvocationRole = "agent" | "detection"; + +export interface InvocationRequest { + schema_version: 2; + document_kind: "request"; + role: InvocationRole; + command: string; + prompt_path: string; + mcp_config_path: string | null; + args: string[]; + explicit_model: string | null; +} + +export interface PreparedInvocation { + schema_version: 2; + document_kind: "prepared"; + role: InvocationRole; + command: string; + prompt_path: string; + mcp_config_path: string | null; + args: string[]; + requested_model: string | null; +} + +export interface InvocationResult { + schema_version: 2; + document_kind: "result"; + role: InvocationRole; + requested_model: string | null; +} + +function isRecord(value: unknown): value is Record { + return typeof value === "object" && value !== null && !Array.isArray(value); +} + +function parseObject(raw: string, label: string): Record { + let parsed: unknown; + try { + parsed = JSON.parse(raw); + } catch { + throw new Error(`${label} is not valid JSON`); + } + if (!isRecord(parsed)) { + throw new Error(`${label} must be a JSON object`); + } + return parsed; +} + +function rejectUnknown( + parsed: Record, + allowed: ReadonlySet, + label: string, +): void { + const unknown = Object.keys(parsed).filter((key) => !allowed.has(key)); + if (unknown.length > 0) { + throw new Error(`${label} contains unknown field '${unknown.sort()[0]}'`); + } +} + +function requireVersionAndKind( + parsed: Record, + kind: "request" | "prepared" | "result", + label: string, +): void { + if (parsed.schema_version !== SCHEMA_VERSION) { + throw new Error( + `unsupported ${label} schema version '${String(parsed.schema_version)}'`, + ); + } + if (parsed.document_kind !== kind) { + throw new Error(`${label} field 'document_kind' must be '${kind}'`); + } +} + +function parseRole( + parsed: Record, + label: string, +): InvocationRole { + if (parsed.role !== "agent" && parsed.role !== "detection") { + throw new Error(`${label} field 'role' must be 'agent' or 'detection'`); + } + return parsed.role; +} + +function requiredString( + value: Record, + key: string, + label: string, + validator?: (input: string) => boolean, +): string { + const candidate = value[key]; + if (typeof candidate !== "string" || candidate.length === 0) { + throw new Error(`${label} field '${key}' must be a non-empty string`); + } + if (candidate.includes("\0") || (validator && !validator(candidate))) { + throw new Error(`${label} field '${key}' is invalid`); + } + return candidate; +} + +function nullableString( + value: Record, + key: string, + label: string, + validator?: (input: string) => boolean, +): string | null { + const candidate = value[key]; + if (candidate === null) return null; + if (typeof candidate !== "string" || candidate.includes("\0")) { + throw new Error(`${label} field '${key}' must be a string or null`); + } + if (validator && !validator(candidate)) { + throw new Error(`${label} field '${key}' is invalid`); + } + return candidate; +} + +function parseArgs( + parsed: Record, + label: string, +): string[] { + if ( + !Array.isArray(parsed.args) || + !parsed.args.every((arg) => typeof arg === "string") + ) { + throw new Error(`${label} field 'args' must be an array of strings`); + } + if (parsed.args.some((arg) => arg.includes("\0"))) { + throw new Error(`${label} field 'args' contains an invalid NUL byte`); + } + return [...parsed.args]; +} + +function hasSafePathSegments(value: string, allowBareCommand: boolean): boolean { + if ( + value.includes("\n") || + value.includes("\r") || + value.includes(":") || + value.endsWith("/") + ) { + return false; + } + if (allowBareCommand && !value.includes("/")) { + return value !== "." && value !== ".."; + } + if (!value.startsWith("/")) return false; + const segments = value.slice(1).split("/"); + return ( + segments.length > 0 && + segments.every( + (segment) => segment.length > 0 && segment !== "." && segment !== "..", + ) + ); +} + +function isAbsoluteContainerPath(value: string): boolean { + return hasSafePathSegments(value, false); +} + +function isSafeCommand(value: string): boolean { + return COMMAND_PATTERN.test(value) && hasSafePathSegments(value, true); +} + +export function parseInvocationRequest(raw: string): InvocationRequest { + const label = "invocation request"; + const parsed = parseObject(raw, label); + rejectUnknown(parsed, REQUEST_KEYS, label); + requireVersionAndKind(parsed, "request", label); + return { + schema_version: SCHEMA_VERSION, + document_kind: "request", + role: parseRole(parsed, label), + command: requiredString(parsed, "command", label, isSafeCommand), + prompt_path: requiredString( + parsed, + "prompt_path", + label, + isAbsoluteContainerPath, + ), + mcp_config_path: nullableString( + parsed, + "mcp_config_path", + label, + isAbsoluteContainerPath, + ), + args: parseArgs(parsed, label), + explicit_model: nullableString( + parsed, + "explicit_model", + label, + (value) => MODEL_PATTERN.test(value), + ), + }; +} + +export function parsePreparedInvocation(raw: string): PreparedInvocation { + const label = "prepared invocation"; + const parsed = parseObject(raw, label); + rejectUnknown(parsed, PREPARED_KEYS, label); + requireVersionAndKind(parsed, "prepared", label); + return { + schema_version: SCHEMA_VERSION, + document_kind: "prepared", + role: parseRole(parsed, label), + command: requiredString(parsed, "command", label, isSafeCommand), + prompt_path: requiredString( + parsed, + "prompt_path", + label, + isAbsoluteContainerPath, + ), + mcp_config_path: nullableString( + parsed, + "mcp_config_path", + label, + isAbsoluteContainerPath, + ), + args: parseArgs(parsed, label), + requested_model: nullableString( + parsed, + "requested_model", + label, + (value) => MODEL_PATTERN.test(value), + ), + }; +} + +export function parseInvocationResult(raw: string): InvocationResult { + const label = "invocation result"; + const parsed = parseObject(raw, label); + rejectUnknown(parsed, RESULT_KEYS, label); + requireVersionAndKind(parsed, "result", label); + return { + schema_version: SCHEMA_VERSION, + document_kind: "result", + role: parseRole(parsed, label), + requested_model: nullableString( + parsed, + "requested_model", + label, + (value) => MODEL_PATTERN.test(value), + ), + }; +} + +function runtimeModelVariable(role: InvocationRole): string { + return role === "agent" + ? "ADO_AW_MODEL_AGENT_COPILOT" + : "ADO_AW_MODEL_DETECTION_COPILOT"; +} + +function isUnresolvedAdoMacro(value: string, variable: string): boolean { + return value === `$(${variable})`; +} + +export function resolveRequestedModel( + request: InvocationRequest, + env: NodeJS.ProcessEnv, +): string | null { + if (request.explicit_model !== null) { + return request.explicit_model; + } + const specific = runtimeModelVariable(request.role); + const candidates: Array<[string, string | undefined]> = [ + [specific, env[specific]], + ["ADO_AW_DEFAULT_MODEL_COPILOT", env.ADO_AW_DEFAULT_MODEL_COPILOT], + ]; + for (const [variable, candidate] of candidates) { + if ( + candidate === undefined || + candidate.length === 0 || + isUnresolvedAdoMacro(candidate, variable) + ) { + continue; + } + if (!MODEL_PATTERN.test(candidate)) { + throw new Error( + `runtime Copilot model from ${specific}/ADO_AW_DEFAULT_MODEL_COPILOT contains invalid characters`, + ); + } + return candidate; + } + return null; +} + +export function prepareInvocation( + request: InvocationRequest, + env: NodeJS.ProcessEnv, +): { prepared: PreparedInvocation; result: InvocationResult } { + const requestedModel = resolveRequestedModel(request, env); + return { + prepared: { + schema_version: SCHEMA_VERSION, + document_kind: "prepared", + role: request.role, + command: request.command, + prompt_path: request.prompt_path, + mcp_config_path: request.mcp_config_path, + args: [...request.args], + requested_model: requestedModel, + }, + result: { + schema_version: SCHEMA_VERSION, + document_kind: "result", + role: request.role, + requested_model: requestedModel, + }, + }; +} + +export function buildCopilotArgs( + document: PreparedInvocation, + prompt: string, +): string[] { + if (prompt.includes("\0")) { + throw new Error("prompt contains an invalid NUL byte"); + } + const args = [`--prompt=${prompt}`]; + if (document.mcp_config_path !== null) { + args.push("--additional-mcp-config", `@${document.mcp_config_path}`); + } + args.push(...document.args); + return args; +} + +export function buildChildEnvironment( + env: NodeJS.ProcessEnv, + requestedModel: string | null, +): NodeJS.ProcessEnv { + const childEnv = { ...env }; + delete childEnv.ADO_AW_MODEL_AGENT_COPILOT; + delete childEnv.ADO_AW_MODEL_DETECTION_COPILOT; + delete childEnv.ADO_AW_DEFAULT_MODEL_COPILOT; + if (requestedModel === null) { + delete childEnv.COPILOT_MODEL; + } else { + childEnv.COPILOT_MODEL = requestedModel; + } + return childEnv; +} + +export function writeJsonAtomic( + path: string, + value: PreparedInvocation | InvocationResult, +): void { + const temporary = `${path}.tmp-${process.pid}`; + writeFileSync(temporary, `${JSON.stringify(value)}\n`, { + encoding: "utf8", + mode: 0o600, + }); + renameSync(temporary, path); +} diff --git a/scripts/ado-script/test/azure-wif-isolation.test.ts b/scripts/ado-script/test/azure-wif-isolation.test.ts index 607a23443..8bf3b2194 100644 --- a/scripts/ado-script/test/azure-wif-isolation.test.ts +++ b/scripts/ado-script/test/azure-wif-isolation.test.ts @@ -209,10 +209,22 @@ describe.skipIf(!awfEnabled)("Azure WIF real AWF boundary", () => { const workspace = join(directory, "workspace"); const temp = join(directory, "runner-temp"); const tools = join(directory, "tools"); + const awfTools = join(tools, "awf-tools"); + const adoScripts = join(tools, "ado-aw-scripts"); const home = join(directory, "home"); const auth = join(temp, "ado-aw-azure-auth", "fixture"); mkdirSync(workspace, { recursive: true }); mkdirSync(home); + mkdirSync(awfTools, { recursive: true }); + mkdirSync(join(adoScripts, "ado-script"), { recursive: true }); + copyFileSync( + resolve(testDir, "../copilot-controller.js"), + join(adoScripts, "ado-script/copilot-controller.js"), + ); + copyFileSync( + resolve(testDir, "../copilot-runner.js"), + join(adoScripts, "ado-script/copilot-runner.js"), + ); mkdirSync(join(auth, "token.d"), { recursive: true }); chmodSync(join(temp, "ado-aw-azure-auth"), 0o700); chmodSync(auth, 0o700); @@ -224,15 +236,24 @@ describe.skipIf(!awfEnabled)("Azure WIF real AWF boundary", () => { const runStep = pipeline.jobs.find((job) => job.job === "Agent")?.steps .find((step) => step.bash?.includes("AWF_ARGS+=(--skip-pull --env-all)")); if (!runStep?.bash) throw new Error("compiled AWF invocation is missing"); + expect(runStep.bash).toContain( + "copilot-runner.js run /tmp/awf-tools/copilot-invocation.json", + ); const capture = join(directory, "awf-args"); + const controllerSource = join(adoScripts, "ado-script/copilot-controller.js"); + const forgedResult = join(awfTools, "copilot-invocation-result.json"); writeFileSync(join(tools, "awf/awf"), `#!/bin/sh if [ "$1" = logs ]; then exit 0; fi printf '%s\\0' "$@" > '${capture}' +printf '%s\\n' 'throw new Error("sandbox controller executed on host")' > '${controllerSource}' +printf '%s\\n' '{"schema_version":2,"document_kind":"result","role":"agent","requested_model":"forged"}' > '${forgedResult}' `, { mode: 0o755 }); const script = runStep.bash .replaceAll("$(Agent.TempDirectory)", temp) .replaceAll("$(Pipeline.Workspace)", tools) - .replaceAll("$(Build.SourcesDirectory)", workspace); + .replaceAll("$(Build.SourcesDirectory)", workspace) + .replaceAll("/tmp/awf-tools", awfTools) + .replaceAll("/tmp/ado-aw-scripts", adoScripts); const env: NodeJS.ProcessEnv = { PATH: process.env.PATH, HOME: home, @@ -242,6 +263,22 @@ printf '%s\\0' "$@" > '${capture}' }; for (const name of identities) env[name] = "synthetic-identity"; run("bash", ["-c", script], env); + expect(readFileSync(controllerSource, "utf8")).toContain( + "sandbox controller executed on host", + ); + expect(JSON.parse(readFileSync(forgedResult, "utf8"))).toMatchObject({ + document_kind: "result", + requested_model: "forged", + }); + expect(JSON.parse(readFileSync( + join(temp, "ado-aw-copilot-controller/invocation-result.json"), + "utf8", + ))).toEqual({ + schema_version: 2, + document_kind: "result", + role: "agent", + requested_model: null, + }); const captured = readFileSync(capture, "utf8").split("\0").filter(Boolean); const version = captured[captured.indexOf("--image-tag") + 1]; expect(version).toMatch(/^\d+\.\d+\.\d+$/); @@ -260,6 +297,9 @@ printf '%s\\0' "$@" > '${capture}' const commandIndex = captured.indexOf("--"); expect(commandIndex).toBeGreaterThan(0); + expect(captured[commandIndex + 1]).toContain( + `copilot-runner.js run ${join(awfTools, "copilot-invocation.json")}`, + ); const args: string[] = []; for (let i = 0; i < commandIndex; i++) { // This probe exercises filesystem/env isolation, not MCP networking: diff --git a/scripts/ado-script/test/smoke.test.ts b/scripts/ado-script/test/smoke.test.ts index 630ae1906..0b539d21d 100644 --- a/scripts/ado-script/test/smoke.test.ts +++ b/scripts/ado-script/test/smoke.test.ts @@ -10,6 +10,7 @@ import { spawnSync } from "node:child_process"; import { randomUUID } from "node:crypto"; import { + chmodSync, copyFileSync, existsSync, mkdirSync, @@ -27,6 +28,8 @@ const gateBundlePath = resolve(__dirname, "../gate.js"); const importBundlePath = resolve(__dirname, "../import.js"); const execContextPrBundlePath = resolve(__dirname, "../exec-context-pr.js"); const preparePrBaseBundlePath = resolve(__dirname, "../prepare-pr-base.js"); +const copilotControllerBundlePath = resolve(__dirname, "../copilot-controller.js"); +const copilotRunnerBundlePath = resolve(__dirname, "../copilot-runner.js"); const gateFixturePath = resolve( __dirname, "fixtures/gate-spec-pr-title-match.json", @@ -131,6 +134,79 @@ describe("import.js smoke", () => { }, 20000); }); +describe.skipIf(process.platform === "win32")("Copilot controller/runner smoke", () => { + it("prepares, runs, self-removes, and preserves trusted model metadata", () => { + withSmokeScratchDir("copilot-runner", (dir) => { + const command = resolve(dir, "fake-copilot"); + const promptPath = resolve(dir, "prompt.md"); + const resultPath = resolve(dir, "result.json"); + const capturePath = resolve(dir, "capture.json"); + const requestPath = resolve(dir, "request.json"); + const preparedPath = resolve(dir, "prepared.json"); + const runnerPath = resolve(dir, "copilot-runner.js"); + copyFileSync(copilotRunnerBundlePath, runnerPath); + writeFileSync( + command, + `#!/bin/sh +node -e 'require("node:fs").writeFileSync(process.env.CAPTURE_PATH, JSON.stringify({ argv: process.argv.slice(1), model: process.env.COPILOT_MODEL }))' -- "$@" +exit 7 +`, + ); + chmodSync(command, 0o755); + writeFileSync(promptPath, "smoke prompt\n"); + writeFileSync( + requestPath, + JSON.stringify({ + schema_version: 2, + document_kind: "request", + role: "agent", + command, + prompt_path: promptPath, + mcp_config_path: null, + args: ["--no-ask-user"], + explicit_model: "gpt-smoke", + }), + ); + + const prepare = spawnSync( + process.execPath, + [ + copilotControllerBundlePath, + "prepare", + requestPath, + preparedPath, + resultPath, + ], + { env: { ...process.env }, encoding: "utf8" }, + ); + expect(prepare.status).toBe(0); + const run = spawnSync( + process.execPath, + [runnerPath, "run", preparedPath], + { + env: { ...process.env, CAPTURE_PATH: capturePath }, + encoding: "utf8", + }, + ); + expect(run.status).toBe(7); + expect(run.stdout).toBe(""); + expect(run.stderr).toBe(""); + expect(existsSync(runnerPath)).toBe(false); + expect(existsSync(preparedPath)).toBe(false); + expect(JSON.parse(readFileSync(capturePath, "utf8"))).toEqual({ + argv: ["--prompt=smoke prompt\n", "--no-ask-user"], + model: "gpt-smoke", + }); + expect(JSON.parse(readFileSync(resultPath, "utf8"))).toEqual({ + schema_version: 2, + document_kind: "result", + role: "agent", + requested_model: "gpt-smoke", + }); + }); + }); +}); + function runGitInRepo(repoDir: string, args: string[]): void { const result = spawnSync("git", args, { cwd: repoDir, diff --git a/src/audit/analyzers/detection.rs b/src/audit/analyzers/detection.rs index 49432ee02..23592cb1e 100644 --- a/src/audit/analyzers/detection.rs +++ b/src/audit/analyzers/detection.rs @@ -1,10 +1,10 @@ -use anyhow::Result; +use anyhow::{Context, Result}; use log::{debug, warn}; use serde_json::Value; use std::io::ErrorKind; use std::path::{Path, PathBuf}; -use crate::audit::model::{DetectionAnalysis, DetectionThreats}; +use crate::audit::model::{AwInfo, DetectionAnalysis, DetectionThreats}; /// Read the detection verdict from `analyzed_outputs_/threat-analysis.json`. /// @@ -65,7 +65,36 @@ pub async fn analyze_detection(download_root: &Path) -> Result` artifact. +pub async fn load_aw_info(download_root: &Path) -> Result> { + let Some(directory) = find_analyzed_outputs_dir(download_root).await else { + return Ok(None); + }; + let path = [ + directory.join("aw_info.json"), + directory.join("staging").join("aw_info.json"), + ] + .into_iter() + .find(|path| path.is_file()); + let Some(path) = path else { + return Ok(None); + }; + let contents = tokio::fs::read_to_string(&path) + .await + .with_context(|| format!("Failed to read Detection aw_info file {}", path.display()))?; + serde_json::from_str(&contents) + .with_context(|| format!("Failed to parse Detection aw_info file {}", path.display())) + .map(Some) +} + async fn find_verdict_path(download_root: &Path) -> Option { + find_analyzed_outputs_dir(download_root) + .await + .map(|directory| directory.join("threat-analysis.json")) +} + +async fn find_analyzed_outputs_dir(download_root: &Path) -> Option { let mut entries = match tokio::fs::read_dir(download_root).await { Ok(entries) => entries, Err(err) if err.kind() == ErrorKind::NotFound => return None, @@ -122,7 +151,7 @@ async fn find_verdict_path(download_root: &Path) -> Option { } } - latest_dir.map(|(_, dir)| dir.join("threat-analysis.json")) + latest_dir.map(|(_, dir)| dir) } fn extract_bool(v: &Value, key: &str) -> bool { @@ -163,7 +192,7 @@ fn extract_reasons(v: &Value, verdict_path: &Path) -> Vec { #[cfg(test)] mod tests { - use super::analyze_detection; + use super::{analyze_detection, load_aw_info}; use crate::audit::model::DetectionThreats; use tempfile::TempDir; @@ -180,6 +209,14 @@ mod tests { .unwrap(); } + async fn write_aw_info(temp_dir: &TempDir, dir_name: &str, contents: &str) { + let dir = temp_dir.path().join(dir_name); + tokio::fs::create_dir_all(&dir).await.unwrap(); + tokio::fs::write(dir.join("aw_info.json"), contents) + .await + .unwrap(); + } + async fn write_verdict(temp_dir: &TempDir, dir_name: &str, contents: &str) { let dir = temp_dir.path().join(dir_name); tokio::fs::create_dir_all(&dir).await.unwrap(); @@ -346,4 +383,43 @@ mod tests { Some(expected_verdict_path("analyzed_outputs_10")) ); } + + #[tokio::test] + async fn loads_detection_enriched_aw_info_from_latest_artifact() { + let temp_dir = TempDir::new().unwrap(); + write_aw_info( + &temp_dir, + "analyzed_outputs_9", + r#"{"detection_model":"older"}"#, + ) + .await; + write_aw_info( + &temp_dir, + "analyzed_outputs_10", + r#"{"detection_model":"detector-model"}"#, + ) + .await; + + let aw_info = load_aw_info(temp_dir.path()).await.unwrap().unwrap(); + + assert_eq!(aw_info.detection_model.as_deref(), Some("detector-model")); + } + + #[tokio::test] + async fn detection_aw_info_is_optional_for_older_artifacts() { + let temp_dir = TempDir::new().unwrap(); + create_analyzed_outputs_dir(&temp_dir, "analyzed_outputs_42").await; + + assert!(load_aw_info(temp_dir.path()).await.unwrap().is_none()); + } + + #[tokio::test] + async fn malformed_detection_aw_info_is_an_error() { + let temp_dir = TempDir::new().unwrap(); + write_aw_info(&temp_dir, "analyzed_outputs_42", "{not valid json").await; + + let error = load_aw_info(temp_dir.path()).await.unwrap_err().to_string(); + + assert!(error.contains("Failed to parse Detection aw_info"), "{error}"); + } } diff --git a/src/audit/analyzers/otel.rs b/src/audit/analyzers/otel.rs index ae24740bc..a06de22f6 100644 --- a/src/audit/analyzers/otel.rs +++ b/src/audit/analyzers/otel.rs @@ -12,6 +12,7 @@ pub struct OtelAnalysis { pub engine_config: Option, pub performance: Option, pub aw_info: Option, + pub observed_model: Option, pub warnings: Vec, } @@ -36,6 +37,7 @@ pub async fn analyze_otel(agent_outputs_dir: &std::path::Path) -> anyhow::Result let stats = AgentStats::from_otel_file(&otel_path, "audit") .await .with_context(|| format!("Failed to analyze OTel file: {}", otel_path.display()))?; + analysis.observed_model = stats.model.clone(); let total_tokens = stats.input_tokens + stats.output_tokens; analysis.metrics = MetricsData { @@ -81,7 +83,10 @@ pub async fn analyze_otel(agent_outputs_dir: &std::path::Path) -> anyhow::Result Ok(aw_info) => { analysis.engine_config = Some(AuditEngineConfig { engine: aw_info.engine.clone().unwrap_or_default(), - model: aw_info.model.clone(), + model: analysis + .observed_model + .clone() + .or_else(|| aw_info.model.clone()), version: aw_info.compiler_version.clone(), timeout_minutes: None, }); @@ -176,6 +181,10 @@ mod tests { assert_eq!(analysis.metrics.token_usage, 33185); assert_eq!(analysis.metrics.turns, 2); + assert_eq!( + analysis.observed_model.as_deref(), + Some("claude-sonnet-4.5") + ); assert!(analysis.engine_config.is_none()); assert!(analysis.aw_info.is_none()); } @@ -198,6 +207,10 @@ mod tests { .and_then(|config| config.model.as_deref()), Some("claude-sonnet-4.5") ); + assert_eq!( + analysis.observed_model.as_deref(), + Some("claude-sonnet-4.5") + ); assert_eq!( analysis .aw_info @@ -214,6 +227,35 @@ mod tests { ); } + #[tokio::test] + async fn observed_otel_model_overrides_requested_model_in_engine_config() { + let temp_dir = TempDir::new().unwrap(); + let staging_dir = temp_dir.path().join("staging"); + write_file(&staging_dir.join("otel.jsonl"), COPILOT_OTEL_FIXTURE).await; + write_file( + &staging_dir.join("aw_info.json"), + &AW_INFO_JSON.replace("claude-sonnet-4.5", "requested-session-model"), + ) + .await; + + let analysis = analyze_otel(temp_dir.path()).await.unwrap(); + + assert_eq!( + analysis + .aw_info + .as_ref() + .and_then(|info| info.model.as_deref()), + Some("requested-session-model") + ); + assert_eq!( + analysis + .engine_config + .as_ref() + .and_then(|config| config.model.as_deref()), + Some("claude-sonnet-4.5") + ); + } + #[tokio::test] async fn malformed_aw_info_preserves_otel_metrics_and_emits_shared_warning() { let temp_dir = TempDir::new().unwrap(); diff --git a/src/audit/cli.rs b/src/audit/cli.rs index 978273654..99eb73b60 100644 --- a/src/audit/cli.rs +++ b/src/audit/cli.rs @@ -15,7 +15,7 @@ use crate::audit::analyzers::{ use crate::audit::cache::{RunSummary, load_run_summary, save_run_summary}; use crate::audit::find_artifact_dir; use crate::audit::findings; -use crate::audit::model::{AuditData, ErrorInfo, FileInfo, OverviewData}; +use crate::audit::model::{AuditData, AwInfo, ErrorInfo, FileInfo, OverviewData}; use crate::audit::pipeline_graph; use crate::audit::render; use crate::audit::url::{ParsedBuildRef, parse_build_ref}; @@ -458,6 +458,19 @@ async fn run_analyzers( detection::analyze_detection(run_dir).await, |a, result| a.detection_analysis = result, ); + match detection::load_aw_info(run_dir).await { + Ok(Some(detection_aw_info)) => { + merge_detection_aw_info(&mut audit.overview.aw_info, detection_aw_info) + } + Ok(None) => {} + Err(error) => { + log::warn!("{error:#}"); + crate::audit::push_warning_once( + audit, + crate::audit::malformed_detection_aw_info_warning(), + ); + } + } } run_analyzer( audit, @@ -468,6 +481,16 @@ async fn run_analyzers( ); } +fn merge_detection_aw_info(base: &mut Option, detection: AwInfo) { + let Some(base) = base.as_mut() else { + *base = Some(detection); + return; + }; + if detection.detection_model.is_some() { + base.detection_model = detection.detection_model; + } +} + fn artifact_family_selected(filters: Option<&[String]>, family: &str) -> bool { filters.is_none_or(|filters| filters.iter().any(|filter| filter == family)) } @@ -1039,7 +1062,67 @@ fn render_audit(audit: &AuditData, json: bool) -> Result<()> { #[cfg(test)] mod tests { use super::*; - use crate::audit::model::{CustomSafeOutputJobAudit, Finding, JobData, Recommendation}; + use crate::audit::model::{ + AwInfo, CustomSafeOutputJobAudit, Finding, JobData, Recommendation, + }; + + #[test] + fn detection_aw_info_overlays_only_detection_owned_fields() { + let mut base = Some(AwInfo { + engine: Some(String::from("copilot")), + model: Some(String::from("agent-model")), + detection_model: Some(String::from("old-detector")), + source: Some(String::from("agents/test.md")), + ..Default::default() + }); + let detection = AwInfo { + engine: Some(String::from("must-not-replace")), + model: Some(String::from("must-not-replace")), + detection_model: Some(String::from("detector-model")), + source: Some(String::from("must-not-replace")), + ..Default::default() + }; + + merge_detection_aw_info(&mut base, detection); + + let merged = base.unwrap(); + assert_eq!(merged.engine.as_deref(), Some("copilot")); + assert_eq!(merged.model.as_deref(), Some("agent-model")); + assert_eq!(merged.source.as_deref(), Some("agents/test.md")); + assert_eq!( + merged.detection_model.as_deref(), + Some("detector-model") + ); + } + + #[test] + fn detection_only_aw_info_populates_overview_metadata() { + let mut base = None; + let detection = AwInfo { + engine: Some(String::from("copilot")), + detection_model: Some(String::from("detector-model")), + ..Default::default() + }; + + merge_detection_aw_info(&mut base, detection.clone()); + + assert_eq!(base, Some(detection)); + } + + #[test] + fn older_detection_aw_info_does_not_erase_static_model() { + let mut base = Some(AwInfo { + detection_model: Some(String::from("static-detector")), + ..Default::default() + }); + + merge_detection_aw_info(&mut base, AwInfo::default()); + + assert_eq!( + base.unwrap().detection_model.as_deref(), + Some("static-detector") + ); + } #[tokio::test] async fn agent_output_analyzers_load_proxy_logs_from_canonical_path() { diff --git a/src/audit/mod.rs b/src/audit/mod.rs index bae5429cf..646036747 100644 --- a/src/audit/mod.rs +++ b/src/audit/mod.rs @@ -27,12 +27,38 @@ pub(crate) fn malformed_aw_info_warning() -> model::ErrorInfo { } } +pub(crate) fn malformed_detection_aw_info_warning() -> model::ErrorInfo { + model::ErrorInfo { + source: String::from("audit::detection_aw_info"), + message: String::from( + "Detection's analyzed aw_info.json could not be read or parsed; Detection runtime model enrichment is unavailable", + ), + timestamp: None, + } +} + pub(crate) fn push_warning_once(audit: &mut model::AuditData, warning: model::ErrorInfo) { if !audit.warnings.contains(&warning) { audit.warnings.push(warning); } } +#[cfg(test)] +mod tests { + #[test] + fn malformed_detection_metadata_warning_is_scoped_to_detection_enrichment() { + let warning = super::malformed_detection_aw_info_warning(); + + assert_eq!(warning.source, "audit::detection_aw_info"); + assert!( + warning + .message + .contains("Detection runtime model enrichment") + ); + assert!(!warning.message.contains("pipeline graph")); + } +} + /// Compare two `_` directory names by their trailing /// integer suffix, falling back to a full lexicographic comparison /// when the suffix isn't a u64. diff --git a/src/audit/model.rs b/src/audit/model.rs index 2b9e6365a..0ab1466f3 100644 --- a/src/audit/model.rs +++ b/src/audit/model.rs @@ -153,21 +153,26 @@ pub struct OverviewData { /// Local path where build logs or downloaded artifacts were stored. #[serde(skip_serializing_if = "Option::is_none")] pub logs_path: Option, - /// Runtime-emitted AW metadata from `staging/aw_info.json`. + /// Runtime-emitted AW metadata merged from Agent and Detection artifacts. #[serde(skip_serializing_if = "Option::is_none")] pub aw_info: Option, } /// Runtime-emitted agentic workflow metadata. /// -/// This is read from `staging/aw_info.json`, which mirrors the compiled marker metadata plus runtime context. +/// Agent metadata is read from `staging/aw_info.json`; Detection may enrich the +/// copied `aw_info.json` in its analyzed-output artifact with Detection-owned +/// runtime fields. #[derive(Debug, Clone, Default, PartialEq, Serialize, Deserialize)] #[serde(default)] pub struct AwInfo { /// Configured engine name for the run. #[serde(skip_serializing_if = "Option::is_none")] pub engine: Option, - /// Model identifier used by the agent runtime. + /// Model identifier requested for the Agent's Copilot session. + /// + /// A selected custom agent may pin a different model. When Copilot OTel is + /// available, `AuditData.engine_config.model` reports the observed model. #[serde(skip_serializing_if = "Option::is_none")] pub model: Option, /// Whether AI threat detection was enabled for this workflow. @@ -176,7 +181,7 @@ pub struct AwInfo { /// Engine identifier used by the Detection job when explicitly configured. #[serde(skip_serializing_if = "Option::is_none")] pub detection_engine: Option, - /// Model identifier used by the Detection job when explicitly configured. + /// Model requested for the Detection Copilot session, when available. #[serde(skip_serializing_if = "Option::is_none")] pub detection_model: Option, /// Agent name emitted by the compiled workflow metadata. diff --git a/src/audit/render/console.rs b/src/audit/render/console.rs index ecdb8661d..210d03cfe 100644 --- a/src/audit/render/console.rs +++ b/src/audit/render/console.rs @@ -87,14 +87,23 @@ fn render_overview_section( .filter(|config| !config.engine.is_empty()) .map(|config| config.engine.clone()) }); - let model = aw_info + let requested_model = aw_info .and_then(|info| info.model.as_deref()) .filter(|value| !value.is_empty()) - .map(str::to_string) - .or_else(|| engine_config.and_then(|config| config.model.clone())); + .map(str::to_string); + let observed_model = engine_config + .and_then(|config| config.model.as_deref()) + .filter(|value| !value.is_empty()) + .map(str::to_string); push_opt_owned_row(&mut rows, "engine", engine); - push_opt_owned_row(&mut rows, "model", model); + match (&requested_model, &observed_model) { + (Some(requested), Some(observed)) if requested != observed => { + push_opt_owned_row(&mut rows, "requested_model", requested_model); + push_opt_owned_row(&mut rows, "observed_model", observed_model); + } + _ => push_opt_owned_row(&mut rows, "model", observed_model.or(requested_model)), + } if let Some(enabled) = aw_info.and_then(|info| info.threat_detection_enabled) { rows.push(("threat_detection_enabled".to_string(), enabled.to_string())); } @@ -1235,6 +1244,35 @@ mod tests { assert_eq!(headings, vec!["## Overview", "## Metrics"]); } + #[test] + fn overview_distinguishes_requested_and_observed_models() { + let audit = AuditData { + overview: crate::audit::model::OverviewData { + aw_info: Some(AwInfo { + engine: Some("copilot".to_string()), + model: Some("requested-model".to_string()), + ..Default::default() + }), + ..Default::default() + }, + engine_config: Some(AuditEngineConfig { + engine: "copilot".to_string(), + model: Some("observed-model".to_string()), + ..Default::default() + }), + ..Default::default() + }; + + let out = render_console(&audit); + + assert!(out.contains("- requested_model: requested-model"), "{out}"); + assert!(out.contains("- observed_model: observed-model"), "{out}"); + assert!( + !out.contains("- model: requested-model"), + "{out}" + ); + } + #[test] fn ado_proxy_analysis_renders_lifecycle_rollups_and_recent_events() { let audit = AuditData { diff --git a/src/compile/ado_bundle.rs b/src/compile/ado_bundle.rs index 4bb45757f..69684e984 100644 --- a/src/compile/ado_bundle.rs +++ b/src/compile/ado_bundle.rs @@ -53,6 +53,11 @@ pub enum Bundle { ExecContextRepo, ApprovalSummary, Conclusion, + /// Trusted Copilot invocation controller. Runs only on the pipeline host + /// for preflight preparation and postflight result validation. + CopilotController, + /// Minimal Copilot process runner executed only inside AWF. + CopilotRunner, /// GitHub App installation-token minter/revoker (issue #1316). Runs before /// the Copilot invocation in the Agent and Detection jobs. It authenticates /// to the **GitHub** API (not ADO REST), so it needs no ADO bearer. @@ -155,6 +160,8 @@ impl Bundle { Bundle::ExecContextRepo, Bundle::ApprovalSummary, Bundle::Conclusion, + Bundle::CopilotController, + Bundle::CopilotRunner, Bundle::GithubAppToken, Bundle::PreparePrBase, Bundle::AzureWifRefresh, @@ -183,6 +190,8 @@ impl Bundle { Bundle::ExecContextRepo => paths::EXEC_CONTEXT_REPO_PATH, Bundle::ApprovalSummary => paths::APPROVAL_SUMMARY_PATH, Bundle::Conclusion => paths::CONCLUSION_PATH, + Bundle::CopilotController => paths::COPILOT_CONTROLLER_PATH, + Bundle::CopilotRunner => paths::COPILOT_RUNNER_PATH, Bundle::GithubAppToken => paths::GITHUB_APP_TOKEN_PATH, Bundle::PreparePrBase => paths::PREPARE_PR_BASE_PATH, Bundle::AzureWifRefresh => paths::AZURE_WIF_REFRESH_PATH, @@ -212,6 +221,8 @@ impl Bundle { | Bundle::ExecContextManual | Bundle::ExecContextRepo | Bundle::ApprovalSummary + | Bundle::CopilotController + | Bundle::CopilotRunner // Authenticates to the GitHub API with its own App JWT / minted // token, not the ADO bearer. | Bundle::GithubAppToken diff --git a/src/compile/agentic_pipeline.rs b/src/compile/agentic_pipeline.rs index 176fb2e0b..568aa2974 100644 --- a/src/compile/agentic_pipeline.rs +++ b/src/compile/agentic_pipeline.rs @@ -269,8 +269,8 @@ fn fanout_extension_declarations( /// function's cognitive complexity manageable — behaviour is unchanged. struct EngineSetup { compiler_version: String, - engine_run: String, - engine_run_detection: String, + agent_invocation_json: String, + detection_invocation_json: String, engine_install_steps_yaml: String, detection_engine_install_steps_yaml: String, engine_log_dir: String, @@ -293,19 +293,29 @@ fn build_engine_setup( let compiler_version = env!("CARGO_PKG_VERSION").to_string(); let detection_engine = crate::engine::get_engine(detection_engine_config.engine_id())?; - let engine_run = ctx.engine.invocation( + let agent_invocation = ctx.engine.invocation_request( ctx.front_matter, extension_declarations, - "/tmp/awf-tools/agent-prompt.md", - Some("/tmp/awf-tools/mcp-config.json"), + crate::engine::CopilotInvocationContext::new( + crate::engine::RuntimeModelRole::Agent, + "/tmp/awf-tools/agent-prompt.md", + Some("/tmp/awf-tools/mcp-config.json"), + ), )?; - let engine_run_detection = detection_engine.invocation_with_config( + let detection_invocation = detection_engine.invocation_request_with_config( detection_engine_config, ctx.front_matter, extension_declarations, - "/tmp/awf-tools/threat-analysis-prompt.md", - None, + crate::engine::CopilotInvocationContext::new( + crate::engine::RuntimeModelRole::Detection, + "/tmp/awf-tools/threat-analysis-prompt.md", + None, + ), )?; + let agent_invocation_json = serde_json::to_string(&agent_invocation) + .context("failed to serialize Agent Copilot invocation request")?; + let detection_invocation_json = serde_json::to_string(&detection_invocation) + .context("failed to serialize Detection Copilot invocation request")?; let engine_install_steps_yaml = ctx.engine .install_steps(&front_matter.engine, &front_matter.target, ctx.ado_org())?; @@ -348,8 +358,8 @@ fn build_engine_setup( Ok(EngineSetup { compiler_version, - engine_run, - engine_run_detection, + agent_invocation_json, + detection_invocation_json, engine_install_steps_yaml, detection_engine_install_steps_yaml, engine_log_dir, @@ -440,8 +450,8 @@ pub(crate) fn build_pipeline_context( )?; let EngineSetup { compiler_version, - engine_run, - engine_run_detection, + agent_invocation_json, + detection_invocation_json, engine_install_steps_yaml, detection_engine_install_steps_yaml, engine_log_dir, @@ -573,8 +583,8 @@ pub(crate) fn build_pipeline_context( compiler_version: compiler_version.clone(), engine_install_steps_yaml, detection_engine_install_steps_yaml, - engine_run, - engine_run_detection, + agent_invocation_json, + detection_invocation_json, detection_engine_config, threat_detection, engine_env, @@ -828,8 +838,8 @@ pub(crate) struct StandaloneCtx { /// to typed steps. pub(crate) engine_install_steps_yaml: String, pub(crate) detection_engine_install_steps_yaml: String, - pub(crate) engine_run: String, - pub(crate) engine_run_detection: String, + pub(crate) agent_invocation_json: String, + pub(crate) detection_invocation_json: String, pub(crate) detection_engine_config: EngineConfig, pub(crate) threat_detection: ThreatDetectionConfig, /// Composed engine env block — `KEY: VALUE` lines, one per line. @@ -1328,8 +1338,7 @@ fn build_agent_job( // `checkout: self` (step 1) so the clone exists, and before the Copilot // run so the refs are present when the agent proposes a PR. The // `prepare-pr-base.js` bundle is staged by the ado-script extension's - // agent-prepare steps (`prepare_pr_base_active` is OR'd into that - // extension's Agent-job download predicate), so it is guaranteed present. + // always-on Agent preparation. if front_matter.create_pr_config().is_some() { // The prepare step deepens every checkout dir the SafeOutputs MCP server // may generate a patch from — see `create_pr_prepare_repos`. The @@ -1364,11 +1373,7 @@ fn build_agent_job( // step sets. Never runs for SafeOutputs/user steps. // // The ado-script bundle is staged by the ado-script extension's - // agent-prepare steps: `github_app_token_active` is OR'd into that - // extension's Agent-job download predicate (mirroring - // `safe_outputs_summary_active`), so the bundle is guaranteed present by - // the time we reach this step — no need to inspect emitted steps or - // re-download here. + // always-on Agent preparation, so no feature-specific download is needed. if let Some(app_token) = front_matter.engine.github_app_token() { steps.push(super::extensions::ado_script::github_app_token_step_typed( app_token, @@ -1389,7 +1394,7 @@ fn build_agent_job( &cfg.allowed_domains, &cfg.awf_mounts, &cfg.working_directory, - &cfg.engine_run, + &cfg.agent_invocation_json, &cfg.engine_env, &cfg.byom_exclude_keys, front_matter.supply_chain(), @@ -1412,9 +1417,7 @@ fn build_agent_job( // emitted when any safe-output tool is enabled (transparency for every // run); when manual review is configured the reviewed proposals are listed // first. The ado-script bundle was delivered earlier in this job by the - // ado-script extension, gated on the SAME predicate - // (`has_any_safe_output_tool` → `safe_outputs_summary_active`), so the - // bundle is downloaded iff this step is emitted. + // ado-script extension's always-on Agent preparation. if front_matter.has_any_safe_output_tool() { let (_, reviewed_summary_tools) = front_matter.partition_safe_outputs_by_approval(); steps.push(Step::Bash(safe_outputs_summary_step( @@ -1546,17 +1549,6 @@ fn agent_job_variables_hoist( /// (see `AdoScriptExtension::build_agent_conditions` for today's /// only contributor — synth-PR-skip, PR-filter gate, pipeline-filter /// gate, and user `expression:` escape hatches). -/// Whether the Detection job must stage the `ado-script` bundle. The Detection -/// job has no extension-prepare phase (unlike the Agent job, whose bundle -/// download is contributed by `AdoScriptExtension`), so it stages the bundle -/// itself — but gated on this single predicate so exactly one download is -/// emitted. Today only the GitHub App token step needs it; future -/// detection-only bundle consumers should `||` their own condition in here -/// rather than adding a second `install_and_download_steps_typed` call. -fn detection_job_needs_ado_script_bundle(engine_config: &EngineConfig) -> bool { - engine_config.github_app_token().is_some() -} - fn build_detection_job( front_matter: &FrontMatter, cfg: &StandaloneCtx, @@ -1612,15 +1604,15 @@ fn build_detection_job( )?)); steps.push(Step::Bash(setup_compiler_step())); - // Stage auth support before custom pre-steps, but mint credentials only - // after them so trusted setup code receives the least privilege needed. - if detection_job_needs_ado_script_bundle(&cfg.detection_engine_config) { - steps.extend( - super::extensions::ado_script::install_and_download_steps_typed( - front_matter.supply_chain(), - ), - ); - } + // Detection always executes the Copilot controller/runner bundles. + // Stage them before custom pre-steps; mint optional credentials only + // after those steps so trusted setup code receives the least privilege + // needed. + steps.extend( + super::extensions::ado_script::install_and_download_steps_typed( + front_matter.supply_chain(), + ), + ); for user_step in &cfg.threat_detection.steps { steps.push(Step::RawYaml(step_to_raw_yaml_string(user_step)?)); } @@ -1640,7 +1632,7 @@ fn build_detection_job( steps.push(Step::Bash(run_threat_analysis_step( &cfg.detection_allowed_domains, &cfg.working_directory, - &cfg.engine_run_detection, + &cfg.detection_invocation_json, &cfg.detection_byom_exclude_keys, &cfg.detection_engine_env, crate::engine::github_token_source_var(&cfg.detection_engine_config), @@ -1657,6 +1649,9 @@ fn build_detection_job( steps.push(Step::RawYaml(step_to_raw_yaml_string(user_step)?)); } steps.push(Step::Bash(prepare_analyzed_outputs_step())); + if cfg.detection_engine_config.model().is_none() { + steps.push(Step::Bash(record_detection_runtime_model_step())); + } steps.push(Step::Bash(evaluate_threat_analysis_step())); } else { steps.push(Step::Bash(prepare_analyzed_outputs_passthrough_step())); @@ -4444,6 +4439,21 @@ fn awf_exclude_env_flags(exclude_keys: &[String]) -> String { block } +const COPILOT_TRUSTED_PREFLIGHT: &str = r#"TRUSTED_CONTROLLER_DIR="$AGENT_TEMP/ado-aw-copilot-controller" +TRUSTED_CONTROLLER_PATH="$TRUSTED_CONTROLLER_DIR/copilot-controller.js" +TRUSTED_REQUEST_PATH="$TRUSTED_CONTROLLER_DIR/invocation-request.json" +TRUSTED_RESULT_PATH="$TRUSTED_CONTROLLER_DIR/invocation-result.json" +install -d -m 0700 "$TRUSTED_CONTROLLER_DIR" +install -m 0500 "$COPILOT_CONTROLLER_SOURCE_PATH" "$TRUSTED_CONTROLLER_PATH" +printf '%s\n' "$INVOCATION_REQUEST" > "$TRUSTED_REQUEST_PATH" +chmod 0600 "$TRUSTED_REQUEST_PATH" +rm -f "$SANDBOX_INVOCATION_PATH" "$TRUSTED_RESULT_PATH" +node "$TRUSTED_CONTROLLER_PATH" prepare \ + "$TRUSTED_REQUEST_PATH" \ + "$SANDBOX_INVOCATION_PATH" \ + "$TRUSTED_RESULT_PATH" +rm -f "$COPILOT_CONTROLLER_SOURCE_PATH""#; + shell_script! { /// Invoke the AI agent inside AWF's network-isolated Docker topology. /// @@ -4456,22 +4466,51 @@ shell_script! { /// - `image_flags` — `--image-tag` plus optional `--image-registry` /// - `exclude_env` — provider credentials and internal MCP identity keys /// - `awf_mounts` — the compiler-supplied chain of `--mount "…"` args - /// - `routed_engine_run` — the single-quoted `NO_PROXY` prefix + engine - /// command that AWF invokes inside the sandbox + /// - `routed_runner` — the fixed single-quoted `NO_PROXY` prefix + + /// compiler-owned runner command that AWF runs inside the sandbox RUN_AGENT { interpreter: Bash, - bindings: [AGENT_TEMP, PIPELINE_WORKSPACE, ALLOWED_DOMAINS], - externals: [WORKING_DIRECTORY], - fragments: [topology_attach, image_flags, exclude_env, awf_mounts, routed_engine_run], + bindings: [ + AGENT_TEMP, + PIPELINE_WORKSPACE, + ALLOWED_DOMAINS, + INVOCATION_REQUEST, + SANDBOX_INVOCATION_PATH, + COPILOT_CONTROLLER_SOURCE_PATH + ], + externals: [ + WORKING_DIRECTORY, + TRUSTED_CONTROLLER_PATH, + TRUSTED_RESULT_PATH + ], + fragments: [ + trusted_preflight, + append_aw_info_field, + topology_attach, + image_flags, + exclude_env, + awf_mounts, + routed_runner + ], + phases: [append_aw_info_field = super::extensions::APPEND_AW_INFO_FIELD], + fragment_uses: [ + trusted_preflight => [ + INVOCATION_REQUEST, + SANDBOX_INVOCATION_PATH, + COPILOT_CONTROLLER_SOURCE_PATH + ], + ], body: r###" -set -o pipefail +set -eo pipefail AGENT_OUTPUT_FILE="$AGENT_TEMP/staging/logs/agent-output.txt" mkdir -p "$AGENT_TEMP/staging/logs" AGENT_EXIT_CODE=0 +# ado-aw:fragment trusted_preflight echo "=== Running AI agent with AWF network isolation ===" echo "Allowed domains: $ALLOWED_DOMAINS" +set +e # AWF provides L7 domain whitelisting via a rootless Docker topology. # The named MCPG container is attached to AWF's internal network as a @@ -4495,7 +4534,7 @@ AWF_ARGS+=( --log-level info --proxy-logs-dir "$AGENT_TEMP/staging/logs/firewall" ) -# ado-aw:fragment routed_engine_run +# ado-aw:fragment routed_runner # Stream agent output in real-time while filtering VSO commands. # sed -u = unbuffered (line-by-line) so output appears immediately. @@ -4507,13 +4546,35 @@ AWF_ARGS+=( | tee "$AGENT_OUTPUT_FILE" \ || AGENT_EXIT_CODE=$? +MODEL_RESULT_STATUS=0 +REQUESTED_MODEL=$(node "$TRUSTED_CONTROLLER_PATH" read-result "$TRUSTED_RESULT_PATH" agent) \ + || MODEL_RESULT_STATUS=$? +if [ "$MODEL_RESULT_STATUS" -ne 0 ]; then + echo "ERROR: Agent Copilot invocation result is missing or malformed" >&2 +else + ADO_AW_INFO_JSON="$AGENT_TEMP/staging/aw_info.json" + if [ -f "$ADO_AW_INFO_JSON" ]; then + # ado-aw:fragment append_aw_info_field + ado_aw_append_info_field \ + "model" \ + "$REQUESTED_MODEL" \ + "$ADO_AW_INFO_JSON" \ + || MODEL_RESULT_STATUS=$? + else + echo "Warning: Agent metadata file not found at $ADO_AW_INFO_JSON; model metadata was not recorded" >&2 + fi +fi + # Print firewall summary if available if [ -x "$PIPELINE_WORKSPACE/awf/awf" ]; then echo "=== Firewall Summary ===" "$PIPELINE_WORKSPACE/awf/awf" logs summary --source "$AGENT_TEMP/staging/logs/firewall" 2>/dev/null || true fi -exit "$AGENT_EXIT_CODE" +if [ "$AGENT_EXIT_CODE" -ne 0 ]; then + exit "$AGENT_EXIT_CODE" +fi +exit "$MODEL_RESULT_STATUS" "###, } } @@ -4523,7 +4584,7 @@ fn run_agent_step( allowed_domains: &str, awf_mounts: &str, working_directory: &str, - engine_run: &str, + invocation_request: &str, engine_env: &str, byom_exclude_keys: &[String], supply_chain: Option<&SupplyChainConfig>, @@ -4565,7 +4626,6 @@ fn run_agent_step( }; let image_flags_block = awf_image_flags(supply_chain); let exclude_env_block = awf_exclude_env_flags(byom_exclude_keys); - // AWF attaches externally-launched trusted containers to its internal // network by name. The flag is repeatable, which is what lets the policy // engine join alongside MCPG. Attaching also gives the agent an @@ -4620,9 +4680,10 @@ fn run_agent_step( } else { MCPG_CONTAINER_NAME.to_string() }; - let routed_engine_run = format!( + let routed_runner = format!( "AWF_ARGS+=(-- 'export NO_PROXY=\"${{NO_PROXY:+$NO_PROXY,}}{no_proxy_peers}\"; \ - export no_proxy=\"$NO_PROXY\"; {engine_run}')" + export no_proxy=\"$NO_PROXY\"; exec node {} run /tmp/awf-tools/copilot-invocation.json')", + super::extensions::ado_script::COPILOT_RUNNER_PATH ); let mut step = ShellScript::new(&RUN_AGENT) @@ -4632,11 +4693,28 @@ fn run_agent_step( Binding::ado_macro("Pipeline.Workspace"), ) .bind_text("ALLOWED_DOMAINS", allowed_domains) + .bind( + "INVOCATION_REQUEST", + Binding::document(invocation_request), + ) + .bind_text( + "SANDBOX_INVOCATION_PATH", + "/tmp/awf-tools/copilot-invocation.json", + ) + .bind_text( + "COPILOT_CONTROLLER_SOURCE_PATH", + super::extensions::ado_script::COPILOT_CONTROLLER_PATH, + ) + .fragment("trusted_preflight", COPILOT_TRUSTED_PREFLIGHT) + .fragment( + "append_aw_info_field", + phase_body(&super::extensions::APPEND_AW_INFO_FIELD), + ) .fragment("topology_attach", topology_attach_block) .fragment("image_flags", image_flags_line) .fragment("exclude_env", exclude_env_line) .fragment("awf_mounts", awf_mounts_block) - .fragment("routed_engine_run", routed_engine_run) + .fragment("routed_runner", routed_runner) .into_step("Run copilot (AWF network isolated)"); step.working_directory = Some(working_directory.to_string()); // Engine env comes as a multi-line YAML env block — `KEY: VALUE` lines @@ -4797,8 +4875,7 @@ node "$APPROVAL_SUMMARY_PATH" || echo "##vso[task.logissue type=warning]approval /// Emitted at the **end of the Agent job** (after `collect_safe_outputs_step` /// has staged `safe_outputs.ndjson`), never in the Detection/threat-analysis /// job. The ado-script bundle is delivered earlier in the same job by the -/// ado-script extension's agent-prepare steps (gated on -/// `safe_outputs_summary_active`). +/// ado-script extension's always-on Agent preparation. /// /// `reviewed` is the compiler-resolved set of approval-gated tool names; when /// non-empty the bundle lists those proposals first under a "Pending approval" @@ -6205,15 +6282,34 @@ shell_script! { /// verbatim. RUN_THREAT_ANALYSIS { interpreter: Bash, - bindings: [AGENT_TEMP, PIPELINE_WORKSPACE, ALLOWED_DOMAINS], - externals: [WORKING_DIRECTORY], - fragments: [image_flags, exclude_env, engine_run_detection], + bindings: [ + AGENT_TEMP, + PIPELINE_WORKSPACE, + ALLOWED_DOMAINS, + INVOCATION_REQUEST, + SANDBOX_INVOCATION_PATH, + COPILOT_CONTROLLER_SOURCE_PATH + ], + externals: [ + WORKING_DIRECTORY, + TRUSTED_CONTROLLER_PATH, + TRUSTED_RESULT_PATH + ], + fragments: [trusted_preflight, image_flags, exclude_env, run_runner], + fragment_uses: [ + trusted_preflight => [ + INVOCATION_REQUEST, + SANDBOX_INVOCATION_PATH, + COPILOT_CONTROLLER_SOURCE_PATH + ], + ], body: r###" -set -o pipefail +set -eo pipefail # Run threat analysis with AWF network isolation THREAT_OUTPUT_FILE="$AGENT_TEMP/threat-analysis-output.txt" AGENT_EXIT_CODE=0 +# ado-aw:fragment trusted_preflight # The argument list is assembled into an array so runtime-supplied # fragments splice in as ordinary shell statements (`AWF_ARGS+=(...)`) @@ -6230,16 +6326,29 @@ AWF_ARGS+=( --log-level info --proxy-logs-dir "$AGENT_TEMP/threat-analysis-logs/firewall" ) -# ado-aw:fragment engine_run_detection +# ado-aw:fragment run_runner # Stream threat analysis output in real-time with VSO command filtering # shellcheck disable=SC2016 # The single-quoted engine command inside AWF_ARGS is intentionally expanded by AWF inside the sandbox +set +e "$PIPELINE_WORKSPACE/awf/awf" "${AWF_ARGS[@]}" 2>&1 \ | sed -u 's/##vso\[/[VSO-FILTERED] vso[/g; s/##\[/[VSO-FILTERED] [/g' \ | tee "$THREAT_OUTPUT_FILE" \ || AGENT_EXIT_CODE=$? -exit "$AGENT_EXIT_CODE" +MODEL_RESULT_STATUS=0 +REQUESTED_MODEL=$(node "$TRUSTED_CONTROLLER_PATH" read-result "$TRUSTED_RESULT_PATH" detection) \ + || MODEL_RESULT_STATUS=$? +if [ "$MODEL_RESULT_STATUS" -ne 0 ]; then + echo "ERROR: Detection Copilot invocation result is missing or malformed" >&2 +else + printf '%s' "$REQUESTED_MODEL" > "$AGENT_TEMP/detection-runtime-model" +fi + +if [ "$AGENT_EXIT_CODE" -ne 0 ]; then + exit "$AGENT_EXIT_CODE" +fi +exit "$MODEL_RESULT_STATUS" "###, } } @@ -6247,7 +6356,7 @@ exit "$AGENT_EXIT_CODE" fn run_threat_analysis_step( allowed_domains: &str, working_directory: &str, - engine_run_detection: &str, + invocation_request: &str, byom_exclude_keys: &[String], detection_engine_env: &[(String, String)], github_token_var: &str, @@ -6283,7 +6392,10 @@ fn run_threat_analysis_step( format!("AWF_ARGS+=({})", parts.join(" ")) } }; - let engine_run_detection_line = format!("AWF_ARGS+=(-- '{engine_run_detection}')"); + let run_runner = format!( + "AWF_ARGS+=(-- 'exec node {} run /tmp/awf-tools/copilot-invocation.json')", + super::extensions::ado_script::COPILOT_RUNNER_PATH + ); let mut step = ShellScript::new(&RUN_THREAT_ANALYSIS) .bind("AGENT_TEMP", Binding::ado_macro("Agent.TempDirectory")) @@ -6292,9 +6404,22 @@ fn run_threat_analysis_step( Binding::ado_macro("Pipeline.Workspace"), ) .bind_text("ALLOWED_DOMAINS", allowed_domains) + .bind( + "INVOCATION_REQUEST", + Binding::document(invocation_request), + ) + .bind_text( + "SANDBOX_INVOCATION_PATH", + "/tmp/awf-tools/copilot-invocation.json", + ) + .bind_text( + "COPILOT_CONTROLLER_SOURCE_PATH", + super::extensions::ado_script::COPILOT_CONTROLLER_PATH, + ) + .fragment("trusted_preflight", COPILOT_TRUSTED_PREFLIGHT) .fragment("image_flags", image_flags_line) .fragment("exclude_env", exclude_env_line) - .fragment("engine_run_detection", engine_run_detection_line) + .fragment("run_runner", run_runner) .into_step("Run threat analysis (AWF network isolated)"); step.working_directory = Some(working_directory.to_string()); // env block: GITHUB_TOKEN + GITHUB_READ_ONLY — emit the latter as @@ -6374,6 +6499,48 @@ fn prepare_analyzed_outputs_step() -> BashStep { .with_condition(Condition::Always) } +shell_script! { + /// Detection job: resolve the effective runtime model in Detection's own + /// variable scope and enrich the copied Agent metadata. + RECORD_DETECTION_RUNTIME_MODEL { + interpreter: Bash, + bindings: [AGENT_TEMP], + externals: [], + fragments: [append_aw_info_field], + phases: [append_aw_info_field = super::extensions::APPEND_AW_INFO_FIELD], + body: r###" +set -eo pipefail + +ADO_AW_INFO_JSON="$AGENT_TEMP/analyzed_outputs/aw_info.json" +ADO_AW_MODEL_FILE="$AGENT_TEMP/detection-runtime-model" +if [ ! -f "$ADO_AW_INFO_JSON" ]; then + echo "Warning: Detection metadata file not found at $ADO_AW_INFO_JSON; model metadata was not recorded" >&2 + exit 0 +fi +if [ ! -f "$ADO_AW_MODEL_FILE" ]; then + exit 0 +fi + +# ado-aw:fragment append_aw_info_field +ado_aw_append_info_field \ + "detection_model" \ + "$(cat "$ADO_AW_MODEL_FILE")" \ + "$ADO_AW_INFO_JSON" +"###, + } +} + +fn record_detection_runtime_model_step() -> BashStep { + ShellScript::new(&RECORD_DETECTION_RUNTIME_MODEL) + .bind("AGENT_TEMP", Binding::ado_macro("Agent.TempDirectory")) + .fragment( + "append_aw_info_field", + phase_body(&super::extensions::APPEND_AW_INFO_FIELD), + ) + .into_step("Record Detection runtime model") + .with_condition(Condition::Always) +} + shell_script! { /// Detection job (AI threat detection disabled): copy Agent proposals to /// `analyzed_outputs/` unchanged. The Detection stage still runs as a @@ -6964,11 +7131,24 @@ const _SUBMODULES_OPT_BIND: Option = None; mod tests { use super::*; use crate::compile::mcpg::McpgLaunchEnvironment; + #[cfg(unix)] + use std::process::{Command, Output}; fn test_front_matter(yaml: &str) -> FrontMatter { serde_yaml::from_str(yaml).expect("front matter should parse") } + #[test] + fn candidate_artifact_staging_keeps_fail_fast_verification() { + assert!( + STAGE_CANDIDATE_ARTIFACT_PAYLOAD + .body + .trim_start() + .starts_with("set -eo pipefail"), + "candidate checksum and provenance verification must abort on the first failure" + ); + } + fn test_ctx() -> StandaloneCtx { let test_pool = Pool::VmImage("ubuntu-latest".to_string()); StandaloneCtx { @@ -6989,8 +7169,8 @@ mod tests { compiler_version: "0.0.0-test".to_string(), engine_install_steps_yaml: String::new(), detection_engine_install_steps_yaml: String::new(), - engine_run: "echo agent".to_string(), - engine_run_detection: "echo detection".to_string(), + agent_invocation_json: "{}".to_string(), + detection_invocation_json: "{}".to_string(), detection_engine_config: EngineConfig::default(), threat_detection: ThreatDetectionConfig::default(), engine_env: "GITHUB_READ_ONLY: 1".to_string(), @@ -7661,7 +7841,7 @@ safe-outputs: "example.com", "\\", "/work", - "copilot -p prompt", + r#"{"schema_version":2,"document_kind":"request","role":"agent"}"#, "FOO: bar", &[], None, @@ -7671,6 +7851,20 @@ safe-outputs: .script } + fn runtime_agent_step_for_test() -> BashStep { + run_agent_step( + "example.com", + "\\", + "/work", + r#"{"schema_version":2,"document_kind":"request","role":"agent"}"#, + "ADO_AW_MODEL_AGENT_COPILOT: $(ADO_AW_MODEL_AGENT_COPILOT)\nADO_AW_DEFAULT_MODEL_COPILOT: $(ADO_AW_DEFAULT_MODEL_COPILOT)", + &[], + None, + false, + ) + .expect("run_agent_step should build") + } + #[test] fn agent_attaches_only_mcpg_when_the_policy_engine_is_disabled() { let script = agent_step_for_test(false); @@ -7706,6 +7900,93 @@ safe-outputs: ))); } + #[test] + fn agent_runtime_model_is_prepared_by_controller_and_recorded_after_awf() { + let step = runtime_agent_step_for_test(); + assert!(matches!( + step.env + .get(crate::engine::ADO_AW_MODEL_AGENT_COPILOT), + Some(EnvValue::PipelineVar(name)) + if name == crate::engine::ADO_AW_MODEL_AGENT_COPILOT + )); + assert!(matches!( + step.env + .get(crate::engine::ADO_AW_DEFAULT_MODEL_COPILOT), + Some(EnvValue::PipelineVar(name)) + if name == crate::engine::ADO_AW_DEFAULT_MODEL_COPILOT + )); + assert!(step.script.contains("copilot-runner.js run")); + assert!(step + .script + .contains("node \"$TRUSTED_CONTROLLER_PATH\" prepare")); + assert!(step + .script + .contains("node \"$TRUSTED_CONTROLLER_PATH\" read-result")); + assert!(!step + .script + .contains("AGENT_EXIT_CODE=\"$MODEL_RESULT_STATUS\"")); + assert!(step.script.contains( + "if [ \"$AGENT_EXIT_CODE\" -ne 0 ]; then\n exit \"$AGENT_EXIT_CODE\"\nfi\nexit \"$MODEL_RESULT_STATUS\"" + )); + assert!(!step + .script + .contains("node \"$COPILOT_CONTROLLER_SOURCE_PATH\" read-result")); + assert!(step.script.contains("\"model\"")); + let metadata_index = step + .script + .rfind("ado_aw_append_info_field") + .expect("metadata append"); + let awf_index = step + .script + .find("\"$PIPELINE_WORKSPACE/awf/awf\"") + .expect("AWF invocation"); + assert!(awf_index < metadata_index); + assert!(!step.script.contains("$(ADO_AW_MODEL_AGENT_COPILOT)")); + assert!(!step.script.contains("$(ADO_AW_DEFAULT_MODEL_COPILOT)")); + assert!(!step.script.contains("ADO_AW_EFFECTIVE_MODEL")); + assert!(!step.script.contains("--model")); + } + + #[test] + fn missing_agent_metadata_does_not_block_agent_execution() { + let step = runtime_agent_step_for_test(); + assert!(step + .script + .contains("if [ -f \"$ADO_AW_INFO_JSON\" ]; then")); + assert!(step.script.contains( + "Warning: Agent metadata file not found at $ADO_AW_INFO_JSON; model metadata was not recorded" + )); + assert!(!step + .script + .contains("ERROR: Agent could not find aw_info.json")); + + let warning_index = step.script.find("model metadata was not recorded").unwrap(); + let awf_index = step + .script + .find("\"$PIPELINE_WORKSPACE/awf/awf\"") + .expect("AWF invocation"); + assert!(awf_index < warning_index); + } + + #[test] + fn prefixed_user_env_key_does_not_change_fixed_runner_command() { + let step = run_agent_step( + "example.com", + "\\", + "/work", + r#"{"schema_version":2,"document_kind":"request","role":"agent","explicit_model":"static-model"}"#, + "ADO_AW_MODEL_AGENT_COPILOT_X: harmless", + &[], + None, + false, + ) + .expect("run_agent_step should build"); + + assert!(!step.script.contains("ADO_AW_EFFECTIVE_MODEL")); + assert!(!step.script.contains("unset COPILOT_MODEL")); + assert!(step.script.contains("copilot-runner.js run")); + } + #[test] fn enabling_the_policy_engine_changes_only_attachment_and_no_proxy() { // Guards against the continuation-indent damage that a hand-built @@ -8217,6 +8498,8 @@ safe-outputs: let fm = parse_and_resolve(source); let threat_detection = fm.threat_detection_config().unwrap(); let detection_engine_config = fm.effective_detection_engine(&threat_detection); + let detection_engine_env = + crate::engine::copilot_detection_env(&detection_engine_config).unwrap(); let ctx = super::super::extensions::CompileContext::for_test(&fm); let extensions = super::super::extensions::collect_extensions(&fm); let decls: Vec<_> = extensions @@ -8260,8 +8543,8 @@ safe-outputs: compiler_version: "0.0.0-test".to_string(), engine_install_steps_yaml: String::new(), detection_engine_install_steps_yaml: String::new(), - engine_run: String::new(), - engine_run_detection: String::new(), + agent_invocation_json: "{}".to_string(), + detection_invocation_json: "{}".to_string(), detection_engine_config, threat_detection, engine_env: "env:\n GITHUB_TOKEN: $(GITHUB_TOKEN)\n".to_string(), @@ -8284,7 +8567,7 @@ safe-outputs: debug_pipeline: false, byom_exclude_keys: vec![], detection_byom_exclude_keys: vec![], - detection_engine_env: vec![], + detection_engine_env, }; build_canonical_jobs( &fm, @@ -8302,6 +8585,78 @@ safe-outputs: jobs.iter().find(|j| j.id.as_ref() == id).map(|j| &j.pool) } + fn detection_runtime_model_step(jobs: &[Job]) -> Option<&BashStep> { + jobs.iter() + .find(|job| job.id.as_ref() == "Detection") + .and_then(|job| { + job.steps.iter().find_map(|step| match step { + Step::Bash(step) if step.display_name == "Record Detection runtime model" => { + Some(step) + } + _ => None, + }) + }) + } + + fn detection_run_step(jobs: &[Job]) -> Option<&BashStep> { + jobs.iter() + .find(|job| job.id.as_ref() == "Detection") + .and_then(|job| { + job.steps.iter().find_map(|step| match step { + Step::Bash(step) + if step.display_name == "Run threat analysis (AWF network isolated)" => + { + Some(step) + } + _ => None, + }) + }) + } + + #[cfg(unix)] + fn run_detection_runtime_model_script_with_aw_info( + model: Option<&str>, + aw_info: Option<&str>, + ) -> (Output, tempfile::TempDir) { + let temp = tempfile::tempdir().expect("temp dir"); + let analyzed_outputs = temp.path().join("analyzed_outputs"); + std::fs::create_dir_all(&analyzed_outputs).expect("create analyzed outputs"); + if let Some(aw_info) = aw_info { + std::fs::write(analyzed_outputs.join("aw_info.json"), aw_info) + .expect("write aw_info.json"); + } + if let Some(model) = model { + std::fs::write(temp.path().join("detection-runtime-model"), model) + .expect("write runtime model"); + } + let script = ShellScript::new(&RECORD_DETECTION_RUNTIME_MODEL) + .bind_text("AGENT_TEMP", temp.path().display().to_string()) + .fragment( + "append_aw_info_field", + phase_body(&super::super::extensions::APPEND_AW_INFO_FIELD), + ) + .render(); + let mut command = Command::new("bash"); + command.arg("-c").arg(script).env_clear(); + (command.output().expect("bash should run"), temp) + } + + #[cfg(unix)] + fn run_detection_runtime_model_script(model: Option<&str>) -> (Output, tempfile::TempDir) { + run_detection_runtime_model_script_with_aw_info( + model, + Some(r#"{"schema":"ado-aw/aw_info/1"}"#), + ) + } + + #[cfg(unix)] + fn read_detection_aw_info(temp: &tempfile::TempDir) -> serde_json::Value { + let contents = + std::fs::read_to_string(temp.path().join("analyzed_outputs/aw_info.json")) + .expect("read aw_info.json"); + serde_json::from_str(&contents).expect("parse aw_info.json") + } + #[test] fn threat_detection_enabled_and_disabled_match_expected_ir_graph() { use std::collections::{BTreeMap, BTreeSet}; @@ -8381,6 +8736,7 @@ safe-outputs: assert_eq!(location.job, job("Detection")); assert_eq!(&location.outputs, outputs); } + } let disabled_detection = disabled_jobs @@ -8418,6 +8774,165 @@ safe-outputs: assert!(reviewed_index < copy_logs_index); } + #[test] + fn detection_runtime_model_metadata_uses_detection_job_scope_only_when_enabled() { + let runtime = build_jobs( + "---\nname: test\ndescription: test\nsafe-outputs:\n threat-detection: true\n---\nbody\n", + ); + let disabled = build_jobs( + "---\nname: test\ndescription: test\nsafe-outputs:\n threat-detection: false\n---\nbody\n", + ); + let static_model = build_jobs( + "---\nname: test\ndescription: test\nengine:\n model: static-model\nsafe-outputs:\n threat-detection: true\n---\nbody\n", + ); + + let step = + detection_runtime_model_step(&runtime).expect("runtime Detection emits metadata step"); + assert!(matches!(step.condition, Some(Condition::Always))); + assert!(step.env.is_empty()); + assert!(!step.script.contains("ADO_AW_MODEL_DETECTION_COPILOT")); + assert!(!step.script.contains("ADO_AW_DEFAULT_MODEL_COPILOT")); + + let run_step = detection_run_step(&runtime).expect("enabled Detection runs analysis"); + assert!( + run_step + .env + .contains_key(crate::engine::ADO_AW_MODEL_DETECTION_COPILOT) + ); + assert!( + run_step + .env + .contains_key(crate::engine::ADO_AW_DEFAULT_MODEL_COPILOT) + ); + assert!(matches!( + run_step + .env + .get(crate::engine::ADO_AW_MODEL_DETECTION_COPILOT), + Some(EnvValue::PipelineVar(name)) + if name == crate::engine::ADO_AW_MODEL_DETECTION_COPILOT + )); + assert!(matches!( + run_step + .env + .get(crate::engine::ADO_AW_DEFAULT_MODEL_COPILOT), + Some(EnvValue::PipelineVar(name)) + if name == crate::engine::ADO_AW_DEFAULT_MODEL_COPILOT + )); + assert!( + run_step + .script + .contains("$AGENT_TEMP/detection-runtime-model") + ); + assert!(run_step.script.contains("copilot-runner.js run")); + assert!(run_step + .script + .contains("node \"$TRUSTED_CONTROLLER_PATH\" prepare")); + assert!(run_step + .script + .contains("node \"$TRUSTED_CONTROLLER_PATH\" read-result")); + assert!(!run_step + .script + .contains("AGENT_EXIT_CODE=\"$MODEL_RESULT_STATUS\"")); + assert!(run_step.script.contains( + "if [ \"$AGENT_EXIT_CODE\" -ne 0 ]; then\n exit \"$AGENT_EXIT_CODE\"\nfi\nexit \"$MODEL_RESULT_STATUS\"" + )); + let prepare = run_step + .script + .find("node \"$TRUSTED_CONTROLLER_PATH\" prepare") + .unwrap(); + let disable_errexit = run_step.script.find("set +e").unwrap(); + let awf = run_step + .script + .find("\"$PIPELINE_WORKSPACE/awf/awf\"") + .unwrap(); + assert!(run_step.script.contains("set -eo pipefail")); + assert!( + prepare < disable_errexit && disable_errexit < awf, + "Detection preflight must fail closed before AWF exit capture begins" + ); + assert!(!run_step.script.contains("ADO_AW_EFFECTIVE_MODEL")); + assert!(!run_step.script.contains("--model")); + assert!( + run_step + .script + .find("\"$PIPELINE_WORKSPACE/awf/awf\"") + .unwrap() + < run_step + .script + .find("$AGENT_TEMP/detection-runtime-model") + .unwrap(), + "the trusted host must consume the controller result after Detection runs" + ); + assert!( + !run_step + .script + .contains("$(ADO_AW_MODEL_DETECTION_COPILOT)") + ); + assert!( + !run_step + .script + .contains("$(ADO_AW_DEFAULT_MODEL_COPILOT)") + ); + + assert!( + detection_runtime_model_step(&disabled).is_none(), + "disabled Detection must not resolve or validate model variables" + ); + assert!( + detection_runtime_model_step(&static_model).is_none(), + "static Detection models are already present in compile-time metadata" + ); + } + + #[test] + #[cfg(unix)] + fn detection_runtime_metadata_records_captured_model() { + let (output, temp) = run_detection_runtime_model_script(Some("detector-model")); + + assert!(output.status.success(), "{output:?}"); + assert_eq!( + read_detection_aw_info(&temp)["detection_model"], + "detector-model" + ); + } + + #[test] + #[cfg(unix)] + fn detection_runtime_metadata_omits_missing_model() { + let (output, temp) = run_detection_runtime_model_script(None); + assert!(output.status.success(), "{output:?}"); + assert!( + read_detection_aw_info(&temp) + .get("detection_model") + .is_none() + ); + } + + #[test] + #[cfg(unix)] + fn missing_detection_metadata_does_not_mask_the_detection_result() { + let (output, _) = + run_detection_runtime_model_script_with_aw_info(Some("detector-model"), None); + + assert!(output.status.success(), "{output:?}"); + assert!(String::from_utf8_lossy(&output.stderr).contains( + "Warning: Detection metadata file not found" + )); + } + + #[test] + #[cfg(unix)] + fn malformed_detection_metadata_fails_enrichment() { + let (output, _) = run_detection_runtime_model_script_with_aw_info( + Some("detector-model"), + Some("not-json"), + ); + + assert!(!output.status.success(), "{output:?}"); + assert!(String::from_utf8_lossy(&output.stderr) + .contains("aw_info.json is not a single-line JSON object")); + } + #[test] fn pool_overrides_detection_only_flows_to_compiled_job() { let source = concat!( diff --git a/src/compile/extensions/ado_aw_marker.rs b/src/compile/extensions/ado_aw_marker.rs index 47c249ad7..977ceefd6 100644 --- a/src/compile/extensions/ado_aw_marker.rs +++ b/src/compile/extensions/ado_aw_marker.rs @@ -47,6 +47,52 @@ shell_script! { } } +shell_script! { + /// Append one validated string field to a single-line JSON object. + /// + /// This phase is shared by the Agent metadata writer and the Detection + /// metadata enrichment step so their file-update behavior cannot drift. + APPEND_AW_INFO_FIELD { + interpreter: Bash, + bindings: [], + externals: [], + fragments: [], + body: r#" +ado_aw_append_info_field() { + local field="$1" + local value="$2" + local file="$3" + if [ -z "$value" ]; then + return 0 + fi + local json + local compact_json + local tmp + local separator="," + json="$(cat "$file")" + if [ "$json" = "{}" ]; then + separator="" + else + case "$json" in + \{*\}) ;; + *) + echo "ERROR: aw_info.json is not a single-line JSON object" >&2 + return 1 + ;; + esac + fi + compact_json="$(printf '%s' "$json" | tr -d '[:space:]')" + if printf '%s' "$compact_json" | grep -Fq "\"${field}\":"; then + return 0 + fi + tmp="$(mktemp)" + printf '%s%s"%s":"%s"}' "${json%?}" "$separator" "$field" "$value" > "$tmp" + mv "$tmp" "$file" +} +"#, + } +} + shell_script! { /// Write `aw_info.json` to Agent.TempDirectory/staging. /// @@ -242,23 +288,20 @@ impl CompileMetadata { .front_matter .safe_outputs .contains_key(crate::compile::types::THREAT_DETECTION_KEY); - let (threat_detection_enabled, detection_engine, detection_model) = - if explicit_threat_detection { - let config = ctx.front_matter.threat_detection_config()?; - let (engine, model) = if config.engine.is_some() { - let effective = ctx.front_matter.effective_detection_engine(&config); - let engine = crate::engine::get_engine(effective.engine_id())?; - let model = match engine { - crate::engine::Engine::Copilot => effective.model().map(str::to_string), - }; - (Some(effective.engine_id().to_string()), model) - } else { - (None, None) - }; - (Some(config.is_enabled()), engine, model) - } else { - (None, None, None) - }; + let config = ctx.front_matter.threat_detection_config()?; + let effective = ctx.front_matter.effective_detection_engine(&config); + let engine = crate::engine::get_engine(effective.engine_id())?; + let detection_model = if config.is_enabled() { + match engine { + crate::engine::Engine::Copilot => effective.model().map(str::to_string), + } + } else { + None + }; + let threat_detection_enabled = + explicit_threat_detection.then_some(config.is_enabled()); + let detection_engine = (explicit_threat_detection && config.engine.is_some()) + .then_some(effective.engine_id().to_string()); Ok(Some(Self { source: super::super::common::normalize_source_path(input_path), org: ctx @@ -330,7 +373,10 @@ impl CompileMetadata { ); } if let Some(model) = &self.model { - object.insert("model".to_string(), serde_json::Value::String(model.clone())); + object.insert( + "model".to_string(), + serde_json::Value::String(model.clone()), + ); } if let Some(model) = &self.detection_model { object.insert( @@ -459,6 +505,8 @@ mod tests { use crate::compile::extensions::CompileContext; use crate::compile::types::FrontMatter; use std::path::Path; + #[cfg(unix)] + use std::process::{Command, Output}; fn parse_fm(yaml: &str) -> FrontMatter { serde_yaml::from_str(yaml).expect("front matter parses") @@ -478,6 +526,37 @@ mod tests { } } + #[cfg(unix)] + fn run_append_aw_info_field( + field: &str, + value: &str, + aw_info_json: &str, + ) -> (Output, tempfile::TempDir) { + let temp = tempfile::tempdir().expect("temp dir"); + let path = temp.path().join("aw_info.json"); + std::fs::write(&path, aw_info_json).expect("write aw_info.json"); + let script = format!( + "{}\nado_aw_append_info_field \"$FIELD\" \"$VALUE\" \"$FILE\"", + APPEND_AW_INFO_FIELD.body.trim() + ); + let mut command = Command::new("bash"); + command + .arg("-c") + .arg(script) + .env_clear() + .env("FIELD", field) + .env("VALUE", value) + .env("FILE", &path); + (command.output().expect("bash should run"), temp) + } + + #[cfg(unix)] + fn read_aw_info_json(temp: &tempfile::TempDir) -> serde_json::Value { + let path = temp.path().join("aw_info.json"); + let contents = std::fs::read_to_string(path).expect("aw_info.json should be written"); + serde_json::from_str(&contents).expect("aw_info.json should parse") + } + #[test] fn returns_no_step_when_input_path_absent() { let fm = parse_fm("name: t\ndescription: x\n"); @@ -591,7 +670,7 @@ mod tests { step.script ); assert!( - !step.script.contains("\"model\""), + !step.script.contains("\"model\":\""), "step should omit model when no model is configured:\n{}", step.script ); @@ -623,14 +702,105 @@ mod tests { "step missing build_definition_id macro:\n{}", step.script ); - assert!(!step.script.contains("detection_model")); - assert!(!step.script.contains("threat_detection_enabled")); + assert!(!step.script.contains("\"threat_detection_enabled\"")); + assert!( + !step + .env + .contains_key(crate::engine::ADO_AW_MODEL_DETECTION_COPILOT) + ); + assert!( + !step + .env + .contains_key(crate::engine::ADO_AW_MODEL_AGENT_COPILOT) + ); + assert!( + !step + .env + .contains_key(crate::engine::ADO_AW_DEFAULT_MODEL_COPILOT) + ); + assert!( + !step.script.contains("$(ADO_AW_MODEL_DETECTION_COPILOT)"), + "runtime model macros must only appear in env mappings:\n{}", + step.script + ); + assert!( + !step.script.contains("$(ADO_AW_DEFAULT_MODEL_COPILOT)"), + "runtime model macros must only appear in env mappings:\n{}", + step.script + ); + } + + #[test] + #[cfg(unix)] + fn append_aw_info_field_does_not_confuse_value_with_model_key() { + let (output, temp) = run_append_aw_info_field( + "model", + "gpt-5", + r#"{"agent_name":"model","schema":"ado-aw/aw_info/1"}"#, + ); + + assert!(output.status.success(), "{output:?}"); + let value = read_aw_info_json(&temp); + assert_eq!(value["agent_name"], "model"); + assert_eq!(value["model"], "gpt-5"); + } + + #[test] + #[cfg(unix)] + fn append_aw_info_field_does_not_confuse_value_with_detection_model_key() { + let (output, temp) = run_append_aw_info_field( + "detection_model", + "gpt-5-mini", + r#"{"agent_name":"detection_model","schema":"ado-aw/aw_info/1"}"#, + ); + + assert!(output.status.success(), "{output:?}"); + let value = read_aw_info_json(&temp); + assert_eq!(value["agent_name"], "detection_model"); + assert_eq!(value["detection_model"], "gpt-5-mini"); + } + + #[test] + #[cfg(unix)] + fn append_aw_info_field_preserves_existing_key_with_whitespace() { + let (output, temp) = run_append_aw_info_field( + "model", + "replacement", + r#"{"model" : "original","schema":"ado-aw/aw_info/1"}"#, + ); + + assert!(output.status.success(), "{output:?}"); + let value = read_aw_info_json(&temp); + assert_eq!(value["model"], "original"); + } + + #[test] + #[cfg(unix)] + fn append_aw_info_field_returns_failure_without_exiting_its_caller() { + let temp = tempfile::tempdir().expect("temp dir"); + let path = temp.path().join("aw_info.json"); + std::fs::write(&path, "not-json").expect("write aw_info.json"); + let script = format!( + "{}\nstatus=0\nado_aw_append_info_field model gpt-5 \"$FILE\" || status=$?\nprintf 'after:%s' \"$status\"", + APPEND_AW_INFO_FIELD.body.trim() + ); + let output = Command::new("bash") + .arg("-c") + .arg(script) + .env_clear() + .env("FILE", path) + .output() + .expect("bash should run"); + + assert!(output.status.success(), "{output:?}"); + assert_eq!(String::from_utf8_lossy(&output.stdout), "after:1"); + assert!(String::from_utf8_lossy(&output.stderr) + .contains("aw_info.json is not a single-line JSON object")); } #[test] fn explicit_model_emits_aw_info_model_metadata() { - let fm = - parse_fm("name: t\ndescription: x\nengine:\n id: copilot\n model: some-model\n"); + let fm = parse_fm("name: t\ndescription: x\nengine:\n id: copilot\n model: some-model\n"); let input_path = Path::new("agents/foo.md"); let ctx = CompileContext { agent_name: &fm.name, @@ -651,7 +821,7 @@ mod tests { } #[test] - fn explicit_threat_detection_emits_detector_metadata() { + fn disabled_threat_detection_omits_detector_model() { let fm = parse_fm( "name: t\ndescription: x\nengine:\n id: copilot\n model: agent-model\n\ safe-outputs:\n threat-detection:\n enabled: false\n engine:\n \ @@ -670,8 +840,7 @@ mod tests { let steps = agent_prepare_steps(&ctx); let step = bash_step(&steps[1]); assert!( - step.script - .contains("\"threat_detection_enabled\":false"), + step.script.contains("\"threat_detection_enabled\":false"), "{}", step.script ); @@ -680,18 +849,86 @@ mod tests { "{}", step.script ); + assert!(!step.script.contains("\"detection_model\""), "{}", step.script); + } + + #[test] + fn explicit_default_threat_detection_emits_enabled_state_only() { + let fm = parse_fm("name: t\ndescription: x\nsafe-outputs:\n threat-detection: true\n"); + let input_path = Path::new("agents/foo.md"); + let ctx = CompileContext { + agent_name: &fm.name, + front_matter: &fm, + ado_context: None, + engine: crate::engine::Engine::Copilot, + compile_dir: None, + input_path: Some(input_path), + imported_prompt_body: String::new(), + }; + let steps = agent_prepare_steps(&ctx); + let step = bash_step(&steps[1]); assert!( + step.script.contains("\"threat_detection_enabled\":true"), + "{}", step.script - .contains("\"detection_model\":\"detector-model\""), + ); + assert!(!step.script.contains("\"detection_engine\"")); + assert!(!step.script.contains("ADO_AW_MODEL_DETECTION_COPILOT")); + } + + #[test] + fn inherited_detection_model_is_emitted_as_static_metadata() { + let fm = parse_fm( + "name: t\ndescription: x\nengine:\n id: copilot\n model: agent-model\n\ + safe-outputs:\n threat-detection: true\n", + ); + let input_path = Path::new("agents/foo.md"); + let ctx = CompileContext { + agent_name: &fm.name, + front_matter: &fm, + ado_context: None, + engine: crate::engine::Engine::Copilot, + compile_dir: None, + input_path: Some(input_path), + imported_prompt_body: String::new(), + }; + let steps = agent_prepare_steps(&ctx); + let step = bash_step(&steps[1]); + assert!( + step.script.contains("\"detection_model\":\"agent-model\""), "{}", step.script ); } #[test] - fn explicit_default_threat_detection_emits_enabled_state_only() { + fn implicit_detection_inherits_static_agent_model_metadata() { + let fm = + parse_fm("name: t\ndescription: x\nengine:\n id: copilot\n model: agent-model\n"); + let input_path = Path::new("agents/foo.md"); + let ctx = CompileContext { + agent_name: &fm.name, + front_matter: &fm, + ado_context: None, + engine: crate::engine::Engine::Copilot, + compile_dir: None, + input_path: Some(input_path), + imported_prompt_body: String::new(), + }; + let steps = agent_prepare_steps(&ctx); + let step = bash_step(&steps[1]); + assert!( + step.script + .contains("\"detection_model\":\"agent-model\""), + "{}", + step.script + ); + } + + #[test] + fn enabled_detection_specific_static_model_is_emitted() { let fm = parse_fm( - "name: t\ndescription: x\nsafe-outputs:\n threat-detection: true\n", + "name: t\ndescription: x\nsafe-outputs:\n threat-detection:\n enabled: true\n engine:\n model: detector-model\n", ); let input_path = Path::new("agents/foo.md"); let ctx = CompileContext { @@ -706,12 +943,11 @@ mod tests { let steps = agent_prepare_steps(&ctx); let step = bash_step(&steps[1]); assert!( - step.script.contains("\"threat_detection_enabled\":true"), + step.script + .contains("\"detection_model\":\"detector-model\""), "{}", step.script ); - assert!(!step.script.contains("\"detection_engine\"")); - assert!(!step.script.contains("\"detection_model\"")); } #[test] diff --git a/src/compile/extensions/ado_script.rs b/src/compile/extensions/ado_script.rs index 26bd43c07..cf3f2ebf6 100644 --- a/src/compile/extensions/ado_script.rs +++ b/src/compile/extensions/ado_script.rs @@ -220,6 +220,13 @@ node "$BUNDLE" pub(crate) const GATE_EVAL_PATH: &str = "/tmp/ado-aw-scripts/ado-script/gate.js"; pub(crate) const IMPORT_EVAL_PATH: &str = "/tmp/ado-aw-scripts/ado-script/import.js"; +/// Sandbox-visible controller source. The Agent/Detection step copies this to +/// `$(Agent.TempDirectory)` and removes this copy before AWF starts. +pub(crate) const COPILOT_CONTROLLER_PATH: &str = + "/tmp/ado-aw-scripts/ado-script/copilot-controller.js"; +/// Untrusted run-only bundle executed as AWF's initial Copilot command. +pub(crate) const COPILOT_RUNNER_PATH: &str = + "/tmp/ado-aw-scripts/ado-script/copilot-runner.js"; /// Path to the ado-proxy bundle inside the unpacked `ado-script.zip`. /// /// Unlike every other bundle this one is not executed by a pipeline step. It @@ -308,83 +315,6 @@ pub struct AdoScriptExtension { pub pr_filters: Option, pub pipeline_filters: Option, pub inlined_imports: bool, - /// Whether the PR-context contributor will activate. When true, - /// the Agent-job install/download must fire even if - /// `runtime_imports_active()` is false (i.e. the user has - /// `inlined-imports: true` but a PR trigger configured), so that - /// `exec-context-pr.js` is present for the `pr.rs` invocation. - /// - /// Populated at construction by `collect_extensions` using the - /// shared `exec_context_pr_active` predicate so this stays in - /// lock-step with `ExecContextExtension`'s own activation gate. - pub exec_context_pr_active: bool, - /// Whether the Manual-context contributor (Stage 1 of the - /// exec-context contributor build-out — see plan.md) will - /// activate. When true, the Agent-job install/download must - /// fire so that `exec-context-manual.js` is present. - /// - /// Populated at construction by `collect_extensions` using the - /// shared `manual_contributor_will_activate` predicate so this - /// stays in lock-step with the contributor's `should_activate`. - pub exec_context_manual_active: bool, - /// Whether the Pipeline-context contributor (Stage 2 of the - /// exec-context contributor build-out — see plan.md) will - /// activate. When true, the Agent-job install/download must - /// fire so that `exec-context-pipeline.js` is present. - /// - /// Populated at construction by `collect_extensions` using the - /// shared `pipeline_contributor_will_activate` predicate so this - /// stays in lock-step with the contributor's `should_activate`. - pub exec_context_pipeline_active: bool, - /// Whether the CI-push-context contributor (Stage 3 of the - /// exec-context contributor build-out — see plan.md) will - /// activate. Default-off opt-in feature; when true the - /// install/download must fire so that - /// `exec-context-ci-push.js` is present. - pub exec_context_ci_push_active: bool, - /// Whether the Workitem-context contributor (Stage 4 of the - /// exec-context contributor build-out — see plan.md) will - /// activate. Activates whenever the PR contributor activates - /// unless explicitly disabled. **Crosses an untrusted-prose - /// boundary** — see workitem.rs. - pub exec_context_workitem_active: bool, - /// Whether the Schedule-context contributor (Stage 5 of the - /// exec-context contributor build-out — see plan.md) will - /// activate. Opt-in (default OFF). - pub exec_context_schedule_active: bool, - /// Whether the PR-checks extension (Stage 6 of the build-out — - /// see plan.md) will activate. Opt-in (default OFF) AND - /// requires the PR contributor to activate. - pub exec_context_pr_checks_active: bool, - /// Whether the Repo-context contributor (Stage 7 of the - /// build-out — see plan.md) will activate. Always-on capability, - /// default OFF (opt-in). - pub exec_context_repo_active: bool, - /// Whether the safe-outputs approval-summary step will run at the - /// end of the Agent job. True whenever the workflow enables any - /// safe-output tool. When true the Agent-job install/download must - /// fire so that `approval-summary.js` is present for the - /// end-of-job render step (emitted by `build_agent_job`). - pub safe_outputs_summary_active: bool, - /// Whether GitHub App-backed Copilot auth is configured - /// (`engine.github-app-token`, issue #1316). When true the Agent-job - /// install/download must fire so that `github-app-token.js` is present for - /// the mint (and revoke) steps that `build_agent_job` emits immediately - /// around the Copilot run. Mirrors `safe_outputs_summary_active`: the - /// consuming steps are emitted by `build_agent_job`, not this extension, so - /// the flag drives the shared bundle download — the builder never has to - /// inspect emitted steps to decide whether to download. - pub github_app_token_active: bool, - /// Whether `create-pull-request` is configured (issue #1413). When true the - /// Agent-job install/download must fire so that `prepare-pr-base.js` is - /// present for the base-ref prepare step that `build_agent_job` emits before - /// the Copilot run. Mirrors `github_app_token_active`: the consuming step is - /// emitted by `build_agent_job`, not this extension, so the flag drives the - /// shared bundle download. - pub prepare_pr_base_active: bool, - /// Whether any user-defined stdio MCP server configures `azure-auth`. - /// Drives Agent-job bundle delivery for `azure-wif-refresh.js`. - pub azure_mcp_auth_active: bool, /// PR trigger config required to build `PR_SYNTH_SPEC`. `Some(_)` /// is the single source of truth for "synthetic-from-ci path is /// active for this agent" — `is_some()` replaces what used to be a @@ -1156,25 +1086,9 @@ impl CompilerExtension for AdoScriptExtension { // ─── Agent job ───────────────────────────────────────── let mut agent_prepare_steps: Vec = Vec::new(); let import_active = self.runtime_imports_active(); - if import_active - || self.exec_context_pr_active - || self.exec_context_manual_active - || self.exec_context_pipeline_active - || self.exec_context_ci_push_active - || self.exec_context_workitem_active - || self.exec_context_schedule_active - || self.exec_context_pr_checks_active - || self.exec_context_repo_active - || self.safe_outputs_summary_active - || self.github_app_token_active - || self.prepare_pr_base_active - || self.azure_mcp_auth_active - { - agent_prepare_steps - .extend(install_and_download_steps_typed(self.supply_chain.as_ref())); - if import_active { - agent_prepare_steps.push(resolver_step_typed()); - } + agent_prepare_steps.extend(install_and_download_steps_typed(self.supply_chain.as_ref())); + if import_active { + agent_prepare_steps.push(resolver_step_typed()); } // ─── Agent-job condition contribution ────────────────── @@ -1365,18 +1279,6 @@ mod tests { pr_filters: pr, pipeline_filters: pipeline, inlined_imports: inlined, - exec_context_pr_active: false, - exec_context_manual_active: false, - exec_context_pipeline_active: false, - exec_context_ci_push_active: false, - exec_context_workitem_active: false, - exec_context_schedule_active: false, - exec_context_pr_checks_active: false, - exec_context_repo_active: false, - safe_outputs_summary_active: false, - github_app_token_active: false, - prepare_pr_base_active: false, - azure_mcp_auth_active: false, pr_trigger_for_synth: None, supply_chain: None, } @@ -1445,18 +1347,6 @@ mod tests { pr_filters: None, pipeline_filters: None, inlined_imports: true, - exec_context_pr_active: false, - exec_context_manual_active: false, - exec_context_pipeline_active: false, - exec_context_ci_push_active: false, - exec_context_workitem_active: false, - exec_context_schedule_active: false, - exec_context_pr_checks_active: false, - exec_context_repo_active: false, - safe_outputs_summary_active: false, - github_app_token_active: false, - prepare_pr_base_active: false, - azure_mcp_auth_active: false, pr_trigger_for_synth: Some(PrTriggerConfig { branches: Some(BranchFilter { include: vec!["main".into()], @@ -1505,18 +1395,6 @@ mod tests { pr_filters: Some(filters), pipeline_filters: None, inlined_imports: true, - exec_context_pr_active: false, - exec_context_manual_active: false, - exec_context_pipeline_active: false, - exec_context_ci_push_active: false, - exec_context_workitem_active: false, - exec_context_schedule_active: false, - exec_context_pr_checks_active: false, - exec_context_repo_active: false, - safe_outputs_summary_active: false, - github_app_token_active: false, - prepare_pr_base_active: false, - azure_mcp_auth_active: false, pr_trigger_for_synth: Some(PrTriggerConfig { branches: Some(BranchFilter { include: vec!["main".into()], @@ -2090,14 +1968,14 @@ mod tests { } #[test] - fn declarations_agent_prepare_download_fires_when_only_prepare_pr_base_active() { - let mut ext = ext_with(None, None, true); - ext.prepare_pr_base_active = true; + fn declarations_agent_prepare_always_stages_bundle() { + let ext = ext_with(None, None, true); let fm: FrontMatter = serde_yaml::from_str("name: t\ndescription: t").unwrap(); let ctx = CompileContext::for_test(&fm); let steps = ext.declarations(&ctx).unwrap().agent_prepare_steps; - // Install + download fire (so prepare-pr-base.js is staged), but no - // runtime-import resolver (inlined_imports: true). + // The controller and runner are required by every Agent job, so + // install + download fire even when no other ado-script consumer is + // active. assert_eq!(steps.len(), 2, "install + download only"); assert!(matches!(&steps[0], Step::Task(t) if t.task == "UseNode@1")); assert!( @@ -2266,18 +2144,6 @@ mod tests { pr_filters: pr, pipeline_filters: pipeline, inlined_imports: true, - exec_context_pr_active: false, - exec_context_manual_active: false, - exec_context_pipeline_active: false, - exec_context_ci_push_active: false, - exec_context_workitem_active: false, - exec_context_schedule_active: false, - exec_context_pr_checks_active: false, - exec_context_repo_active: false, - safe_outputs_summary_active: false, - github_app_token_active: false, - prepare_pr_base_active: false, - azure_mcp_auth_active: false, pr_trigger_for_synth: Some(PrTriggerConfig { branches: Some(BranchFilter { include: vec!["main".into()], @@ -2740,30 +2606,19 @@ mod tests { // ── Typed-IR declarations (port-ado-script) ───────────────────── - /// `declarations()` returns empty step lists when neither - /// runtime-import nor exec-context-pr nor any gate / synth path - /// is active. + /// Setup remains empty when no gate / synth path is active, while Agent + /// preparation always stages the Copilot controller/runner bundles. #[test] - fn declarations_empty_when_nothing_active() { + fn declarations_stages_agent_bundle_when_nothing_else_active() { let ext = ext_with(None, None, true); let fm: FrontMatter = serde_yaml::from_str("name: t\ndescription: t").unwrap(); let ctx = CompileContext::for_test(&fm); let decl = ext.declarations(&ctx).unwrap(); assert!(decl.setup_steps.is_empty()); - assert!(decl.agent_prepare_steps.is_empty()); - } - - #[test] - fn declarations_agent_prepare_download_fires_for_azure_mcp_auth() { - let mut ext = ext_with(None, None, true); - ext.azure_mcp_auth_active = true; - let fm: FrontMatter = serde_yaml::from_str("name: t\ndescription: t").unwrap(); - let ctx = CompileContext::for_test(&fm); - let steps = ext.declarations(&ctx).unwrap().agent_prepare_steps; - assert_eq!(steps.len(), 2, "install + download only"); - assert!(matches!(&steps[0], Step::Task(t) if t.task == "UseNode@1")); + assert_eq!(decl.agent_prepare_steps.len(), 2, "install + download"); + assert!(matches!(&decl.agent_prepare_steps[0], Step::Task(t) if t.task == "UseNode@1")); assert!( - matches!(&steps[1], Step::Bash(b) if b.display_name.contains("Download ado-aw scripts")) + matches!(&decl.agent_prepare_steps[1], Step::Bash(b) if b.display_name.contains("Download ado-aw scripts")) ); } @@ -2816,18 +2671,6 @@ mod tests { pr_filters: None, pipeline_filters: None, inlined_imports: true, - exec_context_pr_active: false, - exec_context_manual_active: false, - exec_context_pipeline_active: false, - exec_context_ci_push_active: false, - exec_context_workitem_active: false, - exec_context_schedule_active: false, - exec_context_pr_checks_active: false, - exec_context_repo_active: false, - safe_outputs_summary_active: false, - github_app_token_active: false, - prepare_pr_base_active: false, - azure_mcp_auth_active: false, pr_trigger_for_synth: Some(PrTriggerConfig { branches: Some(BranchFilter { include: vec!["main".into()], diff --git a/src/compile/extensions/exec_context/mod.rs b/src/compile/extensions/exec_context/mod.rs index 82d949013..4d1e5bf93 100644 --- a/src/compile/extensions/exec_context/mod.rs +++ b/src/compile/extensions/exec_context/mod.rs @@ -55,118 +55,6 @@ use repo::RepoContextContributor; use schedule::ScheduleContextContributor; use workitem::WorkitemContextContributor; -/// Returns `true` iff the PR-context contributor will activate for the -/// given front matter. Shared between `ExecContextExtension::new` (for -/// its own `any_contributor_active` precomputation) and -/// `collect_extensions` (which passes it to `AdoScriptExtension` so -/// the Agent-job install/download fires whenever the bundle is needed). -/// -/// MAINTENANCE: this MUST match `PrContextContributor::should_activate` -/// (in `pr.rs`). The duplication is intentional — `should_activate` -/// takes a `CompileContext` that includes both front matter and target, -/// while this helper only needs the front matter (because `target` is -/// not relevant to PR activation today). -pub fn pr_contributor_will_activate(front_matter: &FrontMatter) -> bool { - // Borrow the embedded config when present; fall back to a stack- - // local default. Avoids the per-call clone — this helper is called - // on every `collect_extensions` invocation, which is hot during - // compile. - let default_cfg = ExecutionContextConfig::default(); - let cfg = front_matter - .execution_context - .as_ref() - .unwrap_or(&default_cfg); - pr_contributor_will_activate_with_cfg(cfg, front_matter) -} - -/// Returns `true` iff the Manual-context contributor will activate -/// for the given front matter. Shared between `ExecContextExtension::new` -/// (for its own `any_contributor_active` aggregate) and -/// `collect_extensions` (which passes it to `AdoScriptExtension` so -/// the Agent-job install/download fires whenever the bundle is needed). -/// -/// MAINTENANCE: this MUST match -/// `ManualContextContributor::should_activate` (in `manual.rs`). -/// Tests in `tests::manual` exercise both paths. -pub fn manual_contributor_will_activate(front_matter: &FrontMatter) -> bool { - let default_cfg = ExecutionContextConfig::default(); - let cfg = front_matter - .execution_context - .as_ref() - .unwrap_or(&default_cfg); - manual_contributor_will_activate_with_cfg(cfg, front_matter) -} - -/// Returns `true` iff the Pipeline-context contributor will activate -/// for the given front matter. Same pattern as the helpers above. -/// -/// MAINTENANCE: this MUST match -/// `PipelineContextContributor::should_activate` (in `pipeline.rs`). -pub fn pipeline_contributor_will_activate(front_matter: &FrontMatter) -> bool { - let default_cfg = ExecutionContextConfig::default(); - let cfg = front_matter - .execution_context - .as_ref() - .unwrap_or(&default_cfg); - pipeline_contributor_will_activate_with_cfg(cfg, front_matter) -} - -/// Returns `true` iff the CI-push-context contributor will activate -/// for the given front matter. Purely config-driven (opt-in, -/// default OFF). -pub fn ci_push_contributor_will_activate(front_matter: &FrontMatter) -> bool { - let default_cfg = ExecutionContextConfig::default(); - let cfg = front_matter - .execution_context - .as_ref() - .unwrap_or(&default_cfg); - ci_push_contributor_will_activate_with_cfg(cfg, front_matter) -} - -/// Returns `true` iff the Workitem contributor will activate. -/// PR-linked mode only — depends on the PR trigger being configured. -pub fn workitem_contributor_will_activate(front_matter: &FrontMatter) -> bool { - let default_cfg = ExecutionContextConfig::default(); - let cfg = front_matter - .execution_context - .as_ref() - .unwrap_or(&default_cfg); - workitem_contributor_will_activate_with_cfg(cfg, front_matter) -} - -/// Returns `true` iff the Schedule contributor will activate. Opt-in -/// (default OFF) AND requires `on.schedule` to be declared. -pub fn schedule_contributor_will_activate(front_matter: &FrontMatter) -> bool { - let default_cfg = ExecutionContextConfig::default(); - let cfg = front_matter - .execution_context - .as_ref() - .unwrap_or(&default_cfg); - schedule_contributor_will_activate_with_cfg(cfg, front_matter) -} - -/// Returns `true` iff the PR-checks extension will activate. Opt-in -/// (default OFF) AND requires the PR contributor to activate. -pub fn pr_checks_contributor_will_activate(front_matter: &FrontMatter) -> bool { - let default_cfg = ExecutionContextConfig::default(); - let cfg = front_matter - .execution_context - .as_ref() - .unwrap_or(&default_cfg); - pr_checks_contributor_will_activate_with_cfg(cfg, front_matter) -} - -/// Returns `true` iff the Repo contributor will activate. Pure -/// config-driven (opt-in, default OFF). -pub fn repo_contributor_will_activate(front_matter: &FrontMatter) -> bool { - let default_cfg = ExecutionContextConfig::default(); - let cfg = front_matter - .execution_context - .as_ref() - .unwrap_or(&default_cfg); - repo_contributor_will_activate_with_cfg(cfg, front_matter) -} - /// Variant that takes the resolved `ExecutionContextConfig` explicitly. /// Used by [`ExecContextExtension::new`] so its internal /// `any_contributor_active` precomputation tracks the config it was diff --git a/src/compile/extensions/exec_context/pr.rs b/src/compile/extensions/exec_context/pr.rs index 21d50c44e..bff38c33d 100644 --- a/src/compile/extensions/exec_context/pr.rs +++ b/src/compile/extensions/exec_context/pr.rs @@ -138,13 +138,6 @@ impl ContextContributor for PrContextContributor { } fn should_activate(&self, ctx: &CompileContext) -> bool { - // MAINTENANCE: this MUST stay in lock-step with - // `super::pr_contributor_will_activate` (the shared helper used - // by `collect_extensions` to populate - // `AdoScriptExtension::exec_context_pr_active`). The divergence- - // trap tests in `super::tests` exercise the helper path; this - // method is the runtime-context-aware version used by the - // declarations path. if ctx.front_matter.pr_trigger().is_none() { return false; } diff --git a/src/compile/extensions/mod.rs b/src/compile/extensions/mod.rs index 26a7f6736..2970f2c1a 100644 --- a/src/compile/extensions/mod.rs +++ b/src/compile/extensions/mod.rs @@ -686,14 +686,10 @@ pub use crate::runtimes::python::PythonExtension; pub use crate::tools::azure_devops::AzureDevOpsExtension; pub use crate::tools::cache_memory::CacheMemoryExtension; pub use ado_aw_marker::AdoAwMarkerExtension; +pub(crate) use ado_aw_marker::APPEND_AW_INFO_FIELD; pub use ado_script::AdoScriptExtension; pub use azure_cli::AzureCliExtension; -pub use exec_context::{ - ExecContextExtension, ci_push_contributor_will_activate, manual_contributor_will_activate, - pipeline_contributor_will_activate, pr_checks_contributor_will_activate, - pr_contributor_will_activate, repo_contributor_will_activate, - schedule_contributor_will_activate, workitem_contributor_will_activate, -}; +pub use exec_context::ExecContextExtension; pub use github::GitHubExtension; pub use safe_outputs::SafeOutputsExtension; @@ -773,59 +769,6 @@ pub fn collect_extensions(front_matter: &FrontMatter) -> Vec { pr_filters: front_matter.pr_filters().cloned(), pipeline_filters: front_matter.pipeline_filters().cloned(), inlined_imports: front_matter.inlined_imports, - // Tell the ado-script extension whether the PR-context - // contributor will activate so it can fire the Agent-job - // install/download even when `inlined-imports: true` (no - // import.js needed). The two extensions stay loosely - // coupled: ExecContextExtension owns invoking the bundle; - // AdoScriptExtension owns installing it. Shared helper - // keeps the activation predicate in lock-step. - exec_context_pr_active: pr_contributor_will_activate(front_matter), - // Same loose-coupling pattern for the Manual contributor - // (Stage 1 of the exec-context contributor build-out — - // see plan.md). Activates whenever any `parameters:` - // block is declared and the contributor isn't explicitly - // disabled. - exec_context_manual_active: manual_contributor_will_activate(front_matter), - // Same loose-coupling pattern for the Pipeline contributor - // (Stage 2 of the exec-context contributor build-out — - // see plan.md). Activates whenever `on.pipeline` is - // configured and the contributor isn't explicitly - // disabled. - exec_context_pipeline_active: pipeline_contributor_will_activate(front_matter), - // CI-push contributor (Stage 3 — opt-in, default OFF). - exec_context_ci_push_active: ci_push_contributor_will_activate(front_matter), - // Workitem contributor (Stage 4 — PR-linked mode only). - // Activates whenever the PR contributor activates and - // workitem isn't explicitly disabled. - exec_context_workitem_active: workitem_contributor_will_activate(front_matter), - // Schedule contributor (Stage 5 — opt-in, default OFF). - exec_context_schedule_active: schedule_contributor_will_activate(front_matter), - // PR-checks extension (Stage 6 — opt-in, default OFF). - exec_context_pr_checks_active: pr_checks_contributor_will_activate(front_matter), - // Repo contributor (Stage 7 — opt-in, default OFF, no - // bearer / no REST, pure git). - exec_context_repo_active: repo_contributor_will_activate(front_matter), - // True whenever any safe-output tool is enabled — drives the - // Agent-job bundle install/download so `approval-summary.js` - // is present for the end-of-job render step that - // `build_agent_job` emits. MUST use the same predicate as that - // step (see `FrontMatter::has_any_safe_output_tool`). - safe_outputs_summary_active: front_matter.has_any_safe_output_tool(), - // True when `engine.github-app-token` is configured — drives the - // Agent-job bundle install/download so `github-app-token.js` is - // present for the mint/revoke steps that `build_agent_job` emits - // around the Copilot run. Same loose-coupling pattern as - // `safe_outputs_summary_active`: the consuming steps live in - // `build_agent_job`, not this extension. - github_app_token_active: front_matter.engine.github_app_token().is_some(), - // True when `create-pull-request` is configured (issue #1413) — - // drives the Agent-job bundle download so `prepare-pr-base.js` - // is present for the base-ref prepare step `build_agent_job` - // emits before the Copilot run. Same loose-coupling pattern as - // `github_app_token_active`. - prepare_pr_base_active: front_matter.create_pr_config().is_some(), - azure_mcp_auth_active: front_matter.has_azure_authenticated_mcp_servers(), pr_trigger_for_synth, supply_chain: front_matter.supply_chain().cloned(), } diff --git a/src/compile/types.rs b/src/compile/types.rs index 08eb3afe4..219d8d4d4 100644 --- a/src/compile/types.rs +++ b/src/compile/types.rs @@ -1626,15 +1626,6 @@ impl FrontMatter { servers } - pub fn has_azure_authenticated_mcp_servers(&self) -> bool { - self.mcp_servers.values().any(|config| { - matches!( - config, - McpConfig::WithOptions(options) - if options.enabled.unwrap_or(true) && options.azure_auth.is_some() - ) - }) - } } /// Compile-time source for a remote reusable import. @@ -2115,14 +2106,7 @@ impl FrontMatter { /// Whether the workflow enables **any** safe-output tool. /// - /// Single source of truth for the safe-outputs-summary feature gate: it - /// drives BOTH the ado-script bundle download - /// (`AdoScriptExtension::safe_outputs_summary_active`, set in - /// `collect_extensions`) and the end-of-Agent-job render step emission - /// (`build_agent_job`). Both call sites MUST go through this so the bundle - /// is downloaded iff the step that runs it is emitted — a drift between two - /// independent copies of this predicate would make the step invoke a bundle - /// that was never downloaded. + /// Single source of truth for the end-of-Agent-job summary step. pub fn has_any_safe_output_tool(&self) -> bool { self.safe_output_tool_names().next().is_some() || !self.custom_safe_output_tool_names().is_empty() diff --git a/src/engine.rs b/src/engine.rs index b01cbfe43..b12fa9c31 100644 --- a/src/engine.rs +++ b/src/engine.rs @@ -1,6 +1,7 @@ use std::collections::HashMap; use anyhow::Result; +use serde::Serialize; use crate::compile::extensions::Declarations; use crate::compile::shell::{Binding, ShellScript}; @@ -24,6 +25,9 @@ const BLOCKED_ARG_PREFIXES: &[&str] = &[ "--ask-user", ]; +/// Native Copilot CLI environment variable used for model selection. +pub const COPILOT_MODEL: &str = "COPILOT_MODEL"; + /// Environment variable keys that the compiler controls — users must not override these. pub const BLOCKED_ENV_KEYS: &[&str] = &[ "GITHUB_TOKEN", @@ -31,6 +35,10 @@ pub const BLOCKED_ENV_KEYS: &[&str] = &[ "COPILOT_OTEL_ENABLED", "COPILOT_OTEL_EXPORTER_TYPE", "COPILOT_OTEL_FILE_EXPORTER_PATH", + COPILOT_MODEL, + "ADO_AW_MODEL_AGENT_COPILOT", + "ADO_AW_MODEL_DETECTION_COPILOT", + "ADO_AW_DEFAULT_MODEL_COPILOT", // Shell/system vars that could affect AWF or pipeline behavior "PATH", "HOME", @@ -74,6 +82,62 @@ pub const COPILOT_PROVIDER_EXPR_ENV_KEYS: &[&str] = &[ /// select the provider subset for the Detection step. const COPILOT_PROVIDER_PREFIX: &str = "COPILOT_PROVIDER_"; +/// Runtime pipeline-variable override for the main Copilot agent model. +pub const ADO_AW_MODEL_AGENT_COPILOT: &str = "ADO_AW_MODEL_AGENT_COPILOT"; +/// Runtime pipeline-variable override for the Detection Copilot model. +pub const ADO_AW_MODEL_DETECTION_COPILOT: &str = "ADO_AW_MODEL_DETECTION_COPILOT"; +/// Shared runtime pipeline-variable fallback for Copilot models. +pub const ADO_AW_DEFAULT_MODEL_COPILOT: &str = "ADO_AW_DEFAULT_MODEL_COPILOT"; + +#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize)] +#[serde(rename_all = "lowercase")] +pub(crate) enum RuntimeModelRole { + Agent, + Detection, +} + +impl RuntimeModelRole { + fn specific_var(self) -> &'static str { + match self { + RuntimeModelRole::Agent => ADO_AW_MODEL_AGENT_COPILOT, + RuntimeModelRole::Detection => ADO_AW_MODEL_DETECTION_COPILOT, + } + } +} + +#[derive(Debug, Clone, Serialize)] +pub(crate) struct CopilotInvocationRequest { + pub(crate) schema_version: u32, + pub(crate) document_kind: &'static str, + pub(crate) role: RuntimeModelRole, + pub(crate) command: String, + pub(crate) prompt_path: String, + pub(crate) mcp_config_path: Option, + pub(crate) args: Vec, + pub(crate) explicit_model: Option, +} + +#[derive(Debug, Clone, Copy)] +pub(crate) struct CopilotInvocationContext<'a> { + role: RuntimeModelRole, + prompt_path: &'a str, + mcp_config_path: Option<&'a str>, +} + +impl<'a> CopilotInvocationContext<'a> { + pub(crate) const fn new( + role: RuntimeModelRole, + prompt_path: &'a str, + mcp_config_path: Option<&'a str>, + ) -> Self { + Self { + role, + prompt_path, + mcp_config_path, + } + } +} + /// Returns true when `key` is an allowlisted BYOM/BYOK provider env-var key that /// may carry ADO macro/runtime expressions in `engine.env`. Case-sensitive: see /// [`COPILOT_PROVIDER_EXPR_ENV_KEYS`] for why exact case is required. @@ -281,30 +345,28 @@ pub fn get_engine(engine_id: &str) -> Result { } impl Engine { - /// The default engine binary name (e.g., "copilot"). - /// - /// Currently scaffolding — the pipeline templates hard-code the binary path - /// (`/tmp/awf-tools/copilot`). This will be wired into template substitution - /// when additional engines are added. Can be overridden per-agent via - /// `engine.command` in front matter. - #[allow(dead_code)] + /// Test-only legacy display of the default engine binary name. + #[cfg(test)] pub fn command(&self) -> &str { match self { Engine::Copilot => "copilot", } } - /// Generate CLI arguments for the engine invocation. + /// Test-only legacy display form of the Copilot CLI arguments. + #[cfg(test)] pub fn args( &self, front_matter: &FrontMatter, extension_declarations: &[Declarations], ) -> Result { - self.args_with_config( - &front_matter.engine, - front_matter, - extension_declarations, - ) + Ok(self + .args_with_config( + &front_matter.engine, + front_matter, + extension_declarations, + )? + .join(" ")) } /// Generate CLI arguments using an explicit engine configuration while @@ -314,7 +376,7 @@ impl Engine { engine_config: &EngineConfig, front_matter: &FrontMatter, extension_declarations: &[Declarations], - ) -> Result { + ) -> Result> { match self { Engine::Copilot => copilot_args(engine_config, front_matter, extension_declarations), } @@ -408,60 +470,37 @@ impl Engine { } } - /// Generate the full AWF `--` command string for running the engine. - /// - /// Returns the content for the AWF `-- ''` argument, including the - /// binary path, prompt delivery flag, MCP config flag, and all CLI arguments. - /// The engine controls how the prompt is provided (e.g., `--prompt="$(cat ...)"` - /// for Copilot) and how MCP config is referenced. - /// - /// `prompt_path` is the path to the prompt file inside the AWF container. - /// `mcp_config_path` is optionally the path to the MCP config file - /// (Some for Agent job, None for Detection job which has no MCP). - pub fn invocation( + pub(crate) fn invocation_request( &self, front_matter: &FrontMatter, extension_declarations: &[Declarations], - prompt_path: &str, - mcp_config_path: Option<&str>, - ) -> Result { - let args = self.args(front_matter, extension_declarations)?; - self.invocation_with_args( + invocation: CopilotInvocationContext<'_>, + ) -> Result { + self.invocation_request_with_config( &front_matter.engine, - prompt_path, - mcp_config_path, - &args, + front_matter, + extension_declarations, + invocation, ) } - /// Generate an invocation using an explicit engine configuration. - pub fn invocation_with_config( + pub(crate) fn invocation_request_with_config( &self, engine_config: &EngineConfig, front_matter: &FrontMatter, extension_declarations: &[Declarations], - prompt_path: &str, - mcp_config_path: Option<&str>, - ) -> Result { + invocation: CopilotInvocationContext<'_>, + ) -> Result { let args = self.args_with_config(engine_config, front_matter, extension_declarations)?; - self.invocation_with_args(engine_config, prompt_path, mcp_config_path, &args) - } - - fn invocation_with_args( - &self, - engine_config: &EngineConfig, - prompt_path: &str, - mcp_config_path: Option<&str>, - args: &str, - ) -> Result { match self { Engine::Copilot => { let command_path = match engine_config.command() { Some(cmd) => { if !is_valid_command_path(cmd) { anyhow::bail!( - "engine.command '{}' contains invalid characters. \ - Only ASCII alphanumerics, '.', '_', '/', and '-' are allowed.", + "engine.command '{}' is invalid. Use a bare executable name or an \ + absolute container path with ASCII alphanumerics, '.', '_', '/', \ + and '-' and no dot, empty, or traversal segments.", cmd ); } @@ -469,12 +508,19 @@ impl Engine { } None => "/tmp/awf-tools/copilot".to_string(), }; - Ok(copilot_invocation( - &command_path, - prompt_path, - mcp_config_path, + if let Some(model) = engine_config.model() { + validate_model_name(model)?; + } + Ok(CopilotInvocationRequest { + schema_version: 2, + document_kind: "request", + role: invocation.role, + command: command_path, + prompt_path: invocation.prompt_path.to_string(), + mcp_config_path: invocation.mcp_config_path.map(str::to_string), args, - )) + explicit_model: engine_config.model().map(str::to_string), + }) } } } @@ -604,6 +650,13 @@ fn validate_user_arg(arg: &str) -> Result<()> { arg ); } + if arg == "--model" || arg.starts_with("--model=") { + anyhow::bail!( + "engine.args entry '{}' conflicts with compiler-controlled model selection. \ + Use engine.model or the ADO_AW_MODEL_*_COPILOT pipeline variables instead.", + arg + ); + } // Reject args that attempt to override compiler-controlled flags for blocked in BLOCKED_ARG_PREFIXES { if arg.starts_with(blocked) { @@ -622,7 +675,7 @@ fn copilot_args( engine_config: &EngineConfig, front_matter: &FrontMatter, extension_declarations: &[Declarations], -) -> Result { +) -> Result> { // Check if bash triggers --allow-all-tools. This happens when: // 1. Bash has an explicit wildcard entry (":*" or "*"), OR // 2. Bash is not specified at all (None) — ado-aw agents always run in AWF sandbox, @@ -657,21 +710,8 @@ fn copilot_args( let mut params = Vec::new(); - // Validate model name to prevent shell injection — copilot_params are embedded - // inside a single-quoted bash string in the AWF command. if let Some(model) = engine_config.model() { - if model.is_empty() - || !model - .chars() - .all(|c| c.is_ascii_alphanumeric() || matches!(c, '.' | '_' | ':' | '-')) - { - anyhow::bail!( - "Model name '{}' contains invalid characters. \ - Only ASCII alphanumerics, '.', '_', ':', and '-' are allowed.", - model - ); - } - params.push(format!("--model {}", model)); + validate_model_name(model)?; } if let Some(0) = engine_config.timeout_minutes() { eprintln!( @@ -691,7 +731,8 @@ fn copilot_args( agent ); } - params.push(format!("--agent {}", agent)); + params.push("--agent".to_string()); + params.push(agent.to_string()); } // Wire engine.api-target — sets the GHES/GHEC API endpoint hostname @@ -703,7 +744,8 @@ fn copilot_args( api_target ); } - params.push(format!("--api-target {}", api_target)); + params.push("--api-target".to_string()); + params.push(api_target.to_string()); } params.push("--disable-builtin-mcps".to_string()); @@ -714,13 +756,8 @@ fn copilot_args( } for tool in allowed_tools { - if tool.contains('(') || tool.contains(')') || tool.contains(' ') { - // Use double quotes - the copilot_params are embedded inside a single-quoted - // bash string in the AWF command, so single quotes would break quoting. - params.push(format!("--allow-tool \"{}\"", tool)); - } else { - params.push(format!("--allow-tool {}", tool)); - } + params.push("--allow-tool".to_string()); + params.push(tool); } // --allow-all-paths when edit is enabled — lets the agent write to any file path. @@ -730,14 +767,31 @@ fn copilot_args( } // Wire engine.args — append user-provided CLI arguments after compiler-generated args. - // User args are additive; they cannot remove compiler security flags but may override - // non-security defaults via last-wins semantics (e.g., --model). + // User args are additive and cannot override compiler-controlled flags or model selection. for arg in engine_config.args() { validate_user_arg(arg)?; params.push(arg.to_string()); } - Ok(params.join(" ")) + Ok(params) +} + +fn validate_model_name(model: &str) -> Result<()> { + // Validate model names before they become invocation-document values or + // runtime-selected COPILOT_MODEL values. Keep this character set in sync + // with the Copilot controller/runner protocol. + if model.is_empty() + || !model + .chars() + .all(|c| c.is_ascii_alphanumeric() || matches!(c, '.' | '_' | ':' | '-')) + { + anyhow::bail!( + "Model name '{}' contains invalid characters. \ + Only ASCII alphanumerics, '.', '_', ':', and '-' are allowed.", + model + ); + } + Ok(()) } /// The masked, same-job pipeline variable the `github-app-token` ado-script @@ -835,6 +889,11 @@ fn copilot_env(engine_config: &EngineConfig) -> Result { "COPILOT_OTEL_EXPORTER_TYPE: \"file\"".to_string(), "COPILOT_OTEL_FILE_EXPORTER_PATH: \"/tmp/awf-tools/staging/otel.jsonl\"".to_string(), ]; + if let Some(model) = engine_config.model() { + validate_model_name(model)?; + } else { + add_runtime_model_env_lines(&mut lines, RuntimeModelRole::Agent); + } // Wire engine.env — merge user-provided environment variables plus any // `COPILOT_PROVIDER_*` vars derived from an `engine.provider` block. @@ -851,6 +910,23 @@ fn copilot_env(engine_config: &EngineConfig) -> Result { Ok(lines.join("\n")) } +fn add_runtime_model_env_lines(lines: &mut Vec, role: RuntimeModelRole) { + let specific = role.specific_var(); + lines.push(format!("{specific}: $({specific})")); + lines.push(format!( + "{ADO_AW_DEFAULT_MODEL_COPILOT}: $({ADO_AW_DEFAULT_MODEL_COPILOT})" + )); +} + +fn add_runtime_model_env_pairs(pairs: &mut Vec<(String, String)>, role: RuntimeModelRole) { + let specific = role.specific_var(); + pairs.push((specific.to_string(), format!("$({specific})"))); + pairs.push(( + ADO_AW_DEFAULT_MODEL_COPILOT.to_string(), + format!("$({ADO_AW_DEFAULT_MODEL_COPILOT})"), + )); +} + /// Return the `COPILOT_PROVIDER_*` entries of `engine.env` as validated, /// **raw** `(key, value)` pairs (value un-rendered; empty vec when none present), /// sorted by key. @@ -858,9 +934,9 @@ fn copilot_env(engine_config: &EngineConfig) -> Result { /// Used by the Detection (threat-analysis) step so the detection Copilot run /// inherits the same BYOM/BYOK provider routing (and credential isolation) as /// the main agent. Mirrors gh-aw, whose detection engine config inherits the -/// main engine's `Env` (`threat_detection_inline_engine.go`). The main model is -/// already threaded via the `--model` flag on the detection invocation, so only -/// the provider routing/credential keys are needed here. +/// main engine's `Env` (`threat_detection_inline_engine.go`). Model delivery is +/// handled separately through compiler-owned `COPILOT_MODEL`, so only provider +/// routing/credential keys are selected here. /// /// Returning raw pairs (rather than a rendered YAML string) lets the call site /// build typed `EnvValue`s directly — no render-to-YAML-then-reparse round-trip, @@ -928,6 +1004,11 @@ pub fn copilot_detection_env(engine_config: &EngineConfig) -> Result, - args: &str, -) -> String { - let mut parts = vec![ - command_path.to_string(), - format!("--prompt=\"$(cat {prompt_path})\""), - ]; - - if let Some(mcp_path) = mcp_config_path { - parts.push(format!("--additional-mcp-config @{mcp_path}")); - } - - if !args.is_empty() { - parts.push(args.to_string()); - } - - parts.join(" ") -} - #[cfg(test)] mod tests { use super::{ - Engine, GITHUB_APP_TOKEN_VAR, copilot_byom_active, copilot_byom_credential_keys, + ADO_AW_DEFAULT_MODEL_COPILOT, ADO_AW_MODEL_AGENT_COPILOT, ADO_AW_MODEL_DETECTION_COPILOT, + COPILOT_MODEL, CopilotInvocationContext, Engine, GITHUB_APP_TOKEN_VAR, RuntimeModelRole, + copilot_byom_active, copilot_byom_credential_keys, copilot_detection_env, copilot_provider_env, get_engine, github_app_token_secrecy_advisory, github_token_source_var, normalize_version_tag, validate_engine_feature_support, }; @@ -1391,32 +1449,141 @@ mod tests { } #[test] - fn copilot_engine_command() { - assert_eq!(Engine::Copilot.command(), "copilot"); + fn copilot_engine_with_explicit_model() { + let (front_matter, _) = parse_markdown( + "---\nname: test\ndescription: test\nengine:\n id: copilot\n model: gpt-5\n---\n", + ) + .unwrap(); + let params = Engine::Copilot + .args(&front_matter, &declarations_for(&front_matter)) + .unwrap(); + assert!(!params.contains("--model")); + let env = Engine::Copilot.env(&front_matter.engine).unwrap(); + assert!(!env.contains(COPILOT_MODEL), "{env}"); + let invocation = Engine::Copilot + .invocation_request( + &front_matter, + &declarations_for(&front_matter), + CopilotInvocationContext::new( + RuntimeModelRole::Agent, + "/tmp/prompt.md", + None, + ), + ) + .unwrap(); + assert_eq!(invocation.explicit_model.as_deref(), Some("gpt-5")); } #[test] - fn copilot_engine_args() { + fn copilot_invocation_request_defers_runtime_model_resolution() { let (front_matter, _) = parse_markdown("---\nname: test\ndescription: test\n---\n").unwrap(); - let params = Engine::Copilot - .args(&front_matter, &declarations_for(&front_matter)) + let invocation = Engine::Copilot + .invocation_request( + &front_matter, + &declarations_for(&front_matter), + CopilotInvocationContext::new( + RuntimeModelRole::Agent, + "/tmp/prompt.md", + Some("/tmp/mcp.json"), + ), + ) .unwrap(); - // Default engine (copilot) lets the Copilot CLI choose its default model. - assert!(!params.contains("--model ")); - assert!(params.contains("--disable-builtin-mcps")); + + assert_eq!(invocation.role, RuntimeModelRole::Agent); + assert_eq!(invocation.schema_version, 2); + assert_eq!(invocation.document_kind, "request"); + assert_eq!(invocation.explicit_model, None); + assert_eq!( + invocation.mcp_config_path.as_deref(), + Some("/tmp/mcp.json") + ); + assert!(!invocation.args.iter().any(|arg| arg.starts_with("--model"))); } #[test] - fn copilot_engine_with_explicit_model() { + fn copilot_invocation_request_keeps_explicit_model_static() { let (front_matter, _) = parse_markdown( "---\nname: test\ndescription: test\nengine:\n id: copilot\n model: gpt-5\n---\n", ) .unwrap(); - let params = Engine::Copilot - .args(&front_matter, &declarations_for(&front_matter)) + let invocation = Engine::Copilot + .invocation_request( + &front_matter, + &declarations_for(&front_matter), + CopilotInvocationContext::new( + RuntimeModelRole::Agent, + "/tmp/prompt.md", + None, + ), + ) .unwrap(); - assert!(params.contains("--model gpt-5")); + + assert_eq!(invocation.explicit_model.as_deref(), Some("gpt-5")); + assert!(!invocation.args.iter().any(|arg| arg.starts_with("--model"))); + } + + #[test] + fn copilot_detection_invocation_request_uses_independent_runtime_model() { + let (front_matter, _) = + parse_markdown("---\nname: test\ndescription: test\n---\n").unwrap(); + let invocation = Engine::Copilot + .invocation_request_with_config( + &front_matter.engine, + &front_matter, + &declarations_for(&front_matter), + CopilotInvocationContext::new( + RuntimeModelRole::Detection, + "/tmp/threat.md", + None, + ), + ) + .unwrap(); + + assert_eq!(invocation.role, RuntimeModelRole::Detection); + assert_eq!(invocation.explicit_model, None); + assert!(!invocation.args.iter().any(|arg| arg.starts_with("--model"))); + let env = copilot_detection_env(&front_matter.engine).unwrap(); + assert!( + env.iter() + .any(|(key, _)| key == ADO_AW_MODEL_DETECTION_COPILOT) + ); + assert!( + env.iter() + .any(|(key, _)| key == ADO_AW_DEFAULT_MODEL_COPILOT) + ); + } + + #[test] + fn engine_args_reject_model_flag() { + for args in ["[--model, gpt-5]", "[--model=gpt-5]"] { + let source = format!( + "---\nname: test\ndescription: test\nengine:\n id: copilot\n args: {args}\n---\n" + ); + let (front_matter, _) = parse_markdown(&source).unwrap(); + let error = Engine::Copilot + .args(&front_matter, &declarations_for(&front_matter)) + .unwrap_err() + .to_string(); + assert!( + error.contains("compiler-controlled model selection"), + "{error}" + ); + } + } + + #[test] + fn engine_env_rejects_raw_copilot_model() { + let (front_matter, _) = parse_markdown( + "---\nname: test\ndescription: test\nengine:\n id: copilot\n env:\n COPILOT_MODEL: gpt-5\n---\n", + ) + .unwrap(); + let error = Engine::Copilot + .env(&front_matter.engine) + .unwrap_err() + .to_string(); + assert!(error.contains(COPILOT_MODEL), "{error}"); + assert!(error.contains("compiler-controlled"), "{error}"); } #[test] @@ -1445,6 +1612,78 @@ mod tests { assert!(!env.contains("AZURE_DEVOPS_EXT_PAT")); } + #[test] + fn copilot_engine_env_maps_runtime_agent_model_vars_when_no_explicit_model() { + let (front_matter, _) = + parse_markdown("---\nname: test\ndescription: test\n---\n").unwrap(); + let env = Engine::Copilot.env(&front_matter.engine).unwrap(); + + assert!(env.contains(&format!( + "{ADO_AW_MODEL_AGENT_COPILOT}: $({ADO_AW_MODEL_AGENT_COPILOT})" + ))); + assert!(env.contains(&format!( + "{ADO_AW_DEFAULT_MODEL_COPILOT}: $({ADO_AW_DEFAULT_MODEL_COPILOT})" + ))); + } + + #[test] + fn copilot_engine_env_omits_runtime_agent_model_vars_for_explicit_model() { + let (front_matter, _) = parse_markdown( + "---\nname: test\ndescription: test\nengine:\n id: copilot\n model: gpt-5\n---\n", + ) + .unwrap(); + let env = Engine::Copilot.env(&front_matter.engine).unwrap(); + + assert!(!env.contains(ADO_AW_MODEL_AGENT_COPILOT)); + assert!(!env.contains(ADO_AW_DEFAULT_MODEL_COPILOT)); + } + + #[test] + fn copilot_detection_env_maps_independent_runtime_model_vars() { + let (front_matter, _) = + parse_markdown("---\nname: test\ndescription: test\n---\n").unwrap(); + let env = copilot_detection_env(&front_matter.engine).unwrap(); + + assert!(env.contains(&( + ADO_AW_MODEL_DETECTION_COPILOT.to_string(), + format!("$({ADO_AW_MODEL_DETECTION_COPILOT})") + ))); + assert!(env.contains(&( + ADO_AW_DEFAULT_MODEL_COPILOT.to_string(), + format!("$({ADO_AW_DEFAULT_MODEL_COPILOT})") + ))); + assert!(!env.iter().any(|(key, _)| key == ADO_AW_MODEL_AGENT_COPILOT)); + } + + #[test] + fn copilot_detection_env_omits_runtime_model_vars_for_explicit_model() { + let (front_matter, _) = parse_markdown( + "---\nname: test\ndescription: test\nsafe-outputs:\n threat-detection:\n engine:\n model: detector-model\n---\n", + ) + .unwrap(); + let threat_detection = front_matter.threat_detection_config().unwrap(); + let detection_engine = front_matter.effective_detection_engine(&threat_detection); + let env = copilot_detection_env(&detection_engine).unwrap(); + + assert!(!env.iter().any(|(key, _)| key == ADO_AW_MODEL_DETECTION_COPILOT)); + assert!(!env.iter().any(|(key, _)| key == ADO_AW_DEFAULT_MODEL_COPILOT)); + } + + #[test] + fn copilot_engine_env_rejects_user_runtime_model_var_override() { + let (front_matter, _) = parse_markdown( + "---\nname: test\ndescription: test\nengine:\n id: copilot\n env:\n ADO_AW_MODEL_AGENT_COPILOT: gpt-5\n---\n", + ) + .unwrap(); + let err = Engine::Copilot + .env(&front_matter.engine) + .unwrap_err() + .to_string(); + + assert!(err.contains("compiler-controlled environment variable")); + assert!(err.contains(ADO_AW_MODEL_AGENT_COPILOT)); + } + #[test] fn copilot_engine_env_sources_github_token_from_app_token_var_when_configured() { let src = "---\nname: test\ndescription: test\nengine:\n id: copilot\n \ @@ -1542,29 +1781,34 @@ mod tests { "---\nname: test\ndescription: test\nengine:\n id: copilot\n command: /usr/local/bin/my-copilot\n---\n", ).unwrap(); let result = Engine::Copilot - .invocation( + .invocation_request( &fm, &declarations_for(&fm), - "/tmp/prompt.md", - Some("/tmp/mcp.json"), + CopilotInvocationContext::new( + RuntimeModelRole::Agent, + "/tmp/prompt.md", + Some("/tmp/mcp.json"), + ), ) .unwrap(); - assert!(result.starts_with("/usr/local/bin/my-copilot ")); - assert!(!result.contains("/tmp/awf-tools/copilot")); + assert_eq!(result.command, "/usr/local/bin/my-copilot"); } #[test] fn engine_command_default_uses_awf_path() { let (fm, _) = parse_markdown("---\nname: test\ndescription: test\n---\n").unwrap(); let result = Engine::Copilot - .invocation( + .invocation_request( &fm, &declarations_for(&fm), - "/tmp/prompt.md", - Some("/tmp/mcp.json"), + CopilotInvocationContext::new( + RuntimeModelRole::Agent, + "/tmp/prompt.md", + Some("/tmp/mcp.json"), + ), ) .unwrap(); - assert!(result.starts_with("/tmp/awf-tools/copilot ")); + assert_eq!(result.command, "/tmp/awf-tools/copilot"); } #[test] @@ -1572,14 +1816,21 @@ mod tests { let (fm, _) = parse_markdown( "---\nname: test\ndescription: test\nengine:\n id: copilot\n command: \"/tmp/copilot; rm -rf /\"\n---\n", ).unwrap(); - let result = - Engine::Copilot.invocation(&fm, &declarations_for(&fm), "/tmp/prompt.md", None); + let result = Engine::Copilot.invocation_request( + &fm, + &declarations_for(&fm), + CopilotInvocationContext::new( + RuntimeModelRole::Agent, + "/tmp/prompt.md", + None, + ), + ); assert!(result.is_err()); assert!( result .unwrap_err() .to_string() - .contains("invalid characters") + .contains("is invalid") ); } @@ -1588,8 +1839,15 @@ mod tests { let (fm, _) = parse_markdown( "---\nname: test\ndescription: test\nengine:\n id: copilot\n command: \"/tmp/co'pilot\"\n---\n", ).unwrap(); - let result = - Engine::Copilot.invocation(&fm, &declarations_for(&fm), "/tmp/prompt.md", None); + let result = Engine::Copilot.invocation_request( + &fm, + &declarations_for(&fm), + CopilotInvocationContext::new( + RuntimeModelRole::Agent, + "/tmp/prompt.md", + None, + ), + ); assert!(result.is_err()); } diff --git a/src/validate.rs b/src/validate.rs index a12fb5c7e..a3059478e 100644 --- a/src/validate.rs +++ b/src/validate.rs @@ -42,12 +42,24 @@ pub fn is_safe_path_segment(s: &str) -> bool { .all(|c| c.is_ascii_alphanumeric() || matches!(c, '-' | '_' | '.')) } -/// Characters allowed in engine.command paths (absolute path chars only). -/// Prevents shell injection when the path is embedded in AWF single-quoted commands. +/// Validate an engine command as either a bare executable name or an absolute +/// container path with no empty, dot, or traversal segments. pub fn is_valid_command_path(s: &str) -> bool { - !s.is_empty() - && s.chars() + if s.is_empty() + || !s + .chars() .all(|c| c.is_ascii_alphanumeric() || matches!(c, '.' | '_' | '/' | '-')) + { + return false; + } + if !s.contains('/') { + return s != "." && s != ".."; + } + s.starts_with('/') + && !s.ends_with('/') + && s[1..] + .split('/') + .all(|segment| !segment.is_empty() && segment != "." && segment != "..") } /// Characters allowed in engine.agent and engine.model identifiers. @@ -1003,6 +1015,12 @@ mod tests { assert!(is_valid_command_path("/tmp/awf-tools/copilot")); assert!(is_valid_command_path("copilot")); assert!(is_valid_command_path("/usr/local/bin/my-tool_v2")); + assert!(!is_valid_command_path("bin/copilot")); + assert!(!is_valid_command_path(".")); + assert!(!is_valid_command_path("..")); + assert!(!is_valid_command_path("/tmp/../copilot")); + assert!(!is_valid_command_path("/tmp//copilot")); + assert!(!is_valid_command_path("/tmp/copilot/")); assert!(!is_valid_command_path("")); assert!(!is_valid_command_path("/tmp/copilot; rm -rf /")); assert!(!is_valid_command_path("/tmp/copilot'")); diff --git a/tests/awf-copilot-safeoutputs/run.sh b/tests/awf-copilot-safeoutputs/run.sh index b25c15a08..a0fb7b459 100644 --- a/tests/awf-copilot-safeoutputs/run.sh +++ b/tests/awf-copilot-safeoutputs/run.sh @@ -6,9 +6,12 @@ umask 077 : "${ADO_AW_BIN:?ADO_AW_BIN is required}" : "${AWF_BIN:?AWF_BIN is required}" : "${COPILOT_BIN:?COPILOT_BIN is required}" +: "${COPILOT_CONTROLLER_BUNDLE:?COPILOT_CONTROLLER_BUNDLE is required}" +: "${COPILOT_RUNNER_BUNDLE:?COPILOT_RUNNER_BUNDLE is required}" : "${AWF_VERSION:?AWF_VERSION is required}" : "${MCPG_VERSION:?MCPG_VERSION is required}" : "${ADO_AW_COPILOT_CLI_ARTIFACT_DIR:?ADO_AW_COPILOT_CLI_ARTIFACT_DIR is required}" +: "${ADO_AW_COPILOT_CLI_CONTROL_DIR:?ADO_AW_COPILOT_CLI_CONTROL_DIR is required}" : "${COPILOT_GITHUB_TOKEN:?COPILOT_GITHUB_TOKEN is required}" readonly CONTRACT_CONTEXT="awf-copilot-safeoutputs-contract" @@ -17,6 +20,7 @@ readonly MCP_GATEWAY_CONTAINER="awmg-mcpg" readonly MCPG_IMAGE="ghcr.io/github/gh-aw-mcpg:v${MCPG_VERSION}" readonly SAFEOUTPUTS_IMAGE="ghcr.io/github/gh-aw-firewall/agent:${AWF_VERSION}" readonly ARTIFACT_DIR="${ADO_AW_COPILOT_CLI_ARTIFACT_DIR}" +readonly CONTROL_DIR="${ADO_AW_COPILOT_CLI_CONTROL_DIR}" RUNTIME_DIR="$(mktemp -d /tmp/ado-aw-awf-contract.XXXXXX)" SAFE_OUTPUTS_DIR="$(mktemp -d /tmp/ado-aw-safeoutputs.XXXXXX)" @@ -24,6 +28,7 @@ TOOLS_DIR="/tmp/awf-tools" MCPG_PID="" mkdir -p "${ARTIFACT_DIR}" "${SAFE_OUTPUTS_DIR}" "${TOOLS_DIR}" +install -d -m 0700 "${CONTROL_DIR}" cleanup() { local status=$? @@ -37,7 +42,7 @@ cleanup() { if [[ -f "${SAFE_OUTPUTS_DIR}/safe_outputs.ndjson" ]]; then cp "${SAFE_OUTPUTS_DIR}/safe_outputs.ndjson" "${ARTIFACT_DIR}/safe_outputs.ndjson" fi - rm -rf "${RUNTIME_DIR}" "${SAFE_OUTPUTS_DIR}" + rm -rf "${RUNTIME_DIR}" "${SAFE_OUTPUTS_DIR}" "${CONTROL_DIR}" return "${status}" } trap cleanup EXIT @@ -71,6 +76,14 @@ for binary in "${ADO_AW_BIN}" "${AWF_BIN}" "${COPILOT_BIN}"; do exit 1 } done +[[ -f "${COPILOT_CONTROLLER_BUNDLE}" ]] || { + echo "Copilot controller bundle is missing: ${COPILOT_CONTROLLER_BUNDLE}" >&2 + exit 1 +} +[[ -f "${COPILOT_RUNNER_BUNDLE}" ]] || { + echo "Copilot runner bundle is missing: ${COPILOT_RUNNER_BUNDLE}" >&2 + exit 1 +} MCP_GATEWAY_API_KEY="$(openssl rand -base64 45 | tr -d '/+=')" install -m 0755 "${ADO_AW_BIN}" "${TOOLS_DIR}/ado-aw" @@ -223,15 +236,46 @@ jq \ chmod 600 "${TOOLS_DIR}/mcp-config.json" install -m 0755 "${COPILOT_BIN}" "${TOOLS_DIR}/copilot" +mkdir -p /tmp/ado-aw-scripts/ado-script +install -m 0644 "${COPILOT_RUNNER_BUNDLE}" \ + /tmp/ado-aw-scripts/ado-script/copilot-runner.js +install -m 0500 "${COPILOT_CONTROLLER_BUNDLE}" \ + "${CONTROL_DIR}/copilot-controller.js" cat >"${TOOLS_DIR}/agent-prompt.md" <"${CONTROL_DIR}/invocation-request.json" +chmod 600 "${CONTROL_DIR}/invocation-request.json" +node "${CONTROL_DIR}/copilot-controller.js" prepare \ + "${CONTROL_DIR}/invocation-request.json" \ + "${TOOLS_DIR}/copilot-invocation.json" \ + "${CONTROL_DIR}/invocation-result.json" readonly ALLOWED_DOMAINS="api.business.githubcopilot.com,api.enterprise.githubcopilot.com,api.github.com,api.githubcopilot.com,api.individual.githubcopilot.com,config.edge.skype.com,copilot-proxy.githubusercontent.com,github.com,telemetry.enterprise.githubcopilot.com,*.copilot.github.com,*.githubcopilot.com" # shellcheck disable=SC2016 # AWF expands the engine command inside the sandbox. -readonly ENGINE_RUN='export NO_PROXY="${NO_PROXY:+$NO_PROXY,}awmg-mcpg"; export no_proxy="$NO_PROXY"; /tmp/awf-tools/copilot --prompt="$(cat /tmp/awf-tools/agent-prompt.md)" --additional-mcp-config @/tmp/awf-tools/mcp-config.json --model gpt-5-mini --disable-builtin-mcps --no-ask-user --allow-all-tools --allow-tool safeoutputs --allow-all-paths' +readonly ENGINE_RUN='export NO_PROXY="${NO_PROXY:+$NO_PROXY,}awmg-mcpg"; export no_proxy="$NO_PROXY"; exec node /tmp/ado-aw-scripts/ado-script/copilot-runner.js run /tmp/awf-tools/copilot-invocation.json' set +e "${AWF_BIN}" \ @@ -254,6 +298,23 @@ if [[ "${AWF_STATUS}" -ne 0 ]]; then exit "${AWF_STATUS}" fi +REQUESTED_MODEL="$( + node "${CONTROL_DIR}/copilot-controller.js" \ + read-result "${CONTROL_DIR}/invocation-result.json" agent +)" +[[ "${REQUESTED_MODEL}" == "gpt-5-mini" ]] || { + echo "Unexpected requested model from controller: ${REQUESTED_MODEL}" >&2 + exit 1 +} +[[ ! -e /tmp/ado-aw-scripts/ado-script/copilot-runner.js ]] || { + echo "Copilot runner did not remove itself before child execution" >&2 + exit 1 +} +[[ ! -e "${TOOLS_DIR}/copilot-invocation.json" ]] || { + echo "Prepared invocation document was not removed before child execution" >&2 + exit 1 +} + NDJSON_PATH="${SAFE_OUTPUTS_DIR}/safe_outputs.ndjson" for _ in $(seq 1 30); do [[ -s "${NDJSON_PATH}" ]] && break diff --git a/tests/compiler_tests.rs b/tests/compiler_tests.rs index 179337dd7..af8f1d225 100644 --- a/tests/compiler_tests.rs +++ b/tests/compiler_tests.rs @@ -8,6 +8,21 @@ fn compiled_has_enabled_tool(compiled: &str, tool: &str) -> bool { }) } +fn extract_mcpg_config(compiled: &str) -> &str { + let marker = "cat > \"$AGENT_TEMP/staging/mcpg-config.json\" << '"; + let tail = compiled + .split_once(marker) + .map(|(_, tail)| tail) + .expect("compiled pipeline must stage an MCPG config"); + let (sentinel, payload) = tail + .split_once("'\n") + .expect("MCPG config heredoc must have a quoted sentinel"); + payload + .split_once(sentinel) + .map(|(config, _)| config) + .expect("MCPG config heredoc must terminate") +} + // `assert_required_markers`, `assert_pool_config`, `assert_compiler_download`, // `assert_awf_download`, `assert_mcpg_integration`, and `test_compiled_yaml_structure` // validated the legacy `src/data/base.yml` template. The standalone target @@ -2049,33 +2064,31 @@ Call the noop tool exactly once. let detection = extract_job_block(&compiled, "Detection").expect("Detection job should exist"); assert!( - agent.contains( - "/tmp/awf-tools/copilot --prompt=\"$(cat /tmp/awf-tools/agent-prompt.md)\" \ - --additional-mcp-config @/tmp/awf-tools/mcp-config.json" - ), - "agent job should pass compiler-emitted MCP config to Copilot CLI: {agent}" + agent.contains(r#""prompt_path":"/tmp/awf-tools/agent-prompt.md""#) + && agent.contains(r#""mcp_config_path":"/tmp/awf-tools/mcp-config.json""#), + "agent invocation document should reference the prompt and MCP config: {agent}" ); assert!( - !agent.contains("--prompt \"$(cat "), - "agent job should not pass prompt as a separate option value: {agent}" + agent.contains("copilot-runner.js run") + && !agent.contains("/tmp/awf-tools/copilot --prompt"), + "agent job should execute only the fixed sandbox runner command: {agent}" ); assert!( - detection.contains( - "/tmp/awf-tools/copilot --prompt=\"$(cat \ - /tmp/awf-tools/threat-analysis-prompt.md)\"" - ), - "detection job should pass prompt using attached form: {detection}" + detection.contains(r#""prompt_path":"/tmp/awf-tools/threat-analysis-prompt.md""#), + "detection invocation document should reference the threat prompt: {detection}" ); assert!( - !detection.contains("--prompt \"$(cat "), - "detection job should not pass prompt as a separate option value: {detection}" + detection.contains("copilot-runner.js run") + && !detection.contains("/tmp/awf-tools/copilot --prompt"), + "detection job should execute only the fixed sandbox runner command: {detection}" ); assert!( agent.contains("--allow-all-tools"), "default unrestricted tools path should emit --allow-all-tools: {agent}" ); assert!( - !detection.contains("--additional-mcp-config"), + detection.contains(r#""mcp_config_path":null"#) + && !detection.contains(r#""mcp_config_path":"/tmp/awf-tools/mcp-config.json""#), "detection job should not receive the SafeOutputs MCP config: {detection}" ); assert!( @@ -2119,19 +2132,17 @@ fn test_runtime_import_frontmatter_prompt_uses_attached_copilot_prompt_flag() { "agent prompt should runtime-import the full markdown fixture: {compiled}" ); assert!( - agent.contains("/tmp/awf-tools/copilot --prompt=\"$(cat /tmp/awf-tools/agent-prompt.md)\""), - "agent job should pass prompt using attached form: {agent}" + agent.contains(r#""prompt_path":"/tmp/awf-tools/agent-prompt.md""#), + "agent invocation document should reference the resolved prompt file: {agent}" ); assert!( - detection.contains( - "/tmp/awf-tools/copilot --prompt=\"$(cat \ - /tmp/awf-tools/threat-analysis-prompt.md)\"" - ), - "detection job should pass prompt using attached form: {detection}" + detection.contains(r#""prompt_path":"/tmp/awf-tools/threat-analysis-prompt.md""#), + "detection invocation document should reference the threat prompt: {detection}" ); assert!( - !compiled.contains("--prompt \"$(cat "), - "compiled pipeline should not pass prompt as a separate option value: {compiled}" + compiled.contains("copilot-runner.js run") + && !compiled.contains("/tmp/awf-tools/copilot --prompt"), + "compiled pipeline should use the fixed sandbox runner command: {compiled}" ); exercise_attached_prompt_with_pinned_copilot_cli(&fixture); @@ -2171,19 +2182,19 @@ Call the noop tool exactly once. "restricted bash path should not emit --allow-all-tools: {agent}" ); assert!( - agent.contains("--allow-tool safeoutputs"), + agent.contains(r#""--allow-tool","safeoutputs""#), "restricted bash path must explicitly allow the SafeOutputs MCP server: {agent}" ); assert!( - agent.contains("--allow-tool \"shell(echo)\""), + agent.contains(r#""--allow-tool","shell(echo)""#), "restricted bash path must emit the configured bash allowlist: {agent}" ); assert!( - agent.contains("--agent my-custom-agent"), + agent.contains(r#""--agent","my-custom-agent""#), "engine.agent should flow through the compiled Copilot CLI invocation: {agent}" ); assert!( - agent.contains("--api-target api.example.com"), + agent.contains(r#""--api-target","api.example.com""#), "engine.api-target should flow through the compiled Copilot CLI invocation: {agent}" ); assert!( @@ -2191,7 +2202,7 @@ Call the noop tool exactly once. "engine.args should append additive Copilot CLI arguments: {agent}" ); assert!( - agent.contains("--additional-mcp-config @/tmp/awf-tools/mcp-config.json"), + agent.contains(r#""mcp_config_path":"/tmp/awf-tools/mcp-config.json""#), "restricted tools path should still use the compiler-emitted MCP config: {agent}" ); } @@ -2279,7 +2290,7 @@ fn permissions_read_enables_proxy_and_wrapped_az_without_mcp() { "displayName: Install az wrapper (ado-proxy)", "displayName: Detect Azure CLI on host (for AWF mount)", "--topology-attach \"awmg-ado-proxy\"", - "--allow-tool \"shell(az)\"", + r#""--allow-tool","shell(az)""#, ] { assert!( compiled.contains(required), @@ -2423,6 +2434,7 @@ fn test_fixture_azure_devops_mcp_compiled_output() { ); let compiled = fs::read_to_string(&output_path).expect("Should read compiled output"); + let mcpg_config = extract_mcpg_config(&compiled); // The policy document is now carried by the `POLICY` binding, which // `Binding::document` renders as a quoted heredoc in the generated @@ -2490,7 +2502,7 @@ fn test_fixture_azure_devops_mcp_compiled_output() { "MCPG config should have entrypointArgs field" ); assert!( - !compiled.contains("\"command\""), + !mcpg_config.contains("\"command\""), "MCPG config should NOT use command field" ); @@ -2596,6 +2608,7 @@ fn test_mcpg_config_container_based_mcp() { ); let compiled = fs::read_to_string(&output_path).unwrap(); + let mcpg_config = extract_mcpg_config(&compiled); assert!(compiled.contains("\"container\": \"ghcr.io/example/my-tool:latest\"")); assert!(compiled.contains("\"entrypoint\": \"my-tool\"")); @@ -2604,7 +2617,7 @@ fn test_mcpg_config_container_based_mcp() { assert!(compiled.contains("/host/data:/app/data:ro")); assert!(compiled.contains("\"API_KEY\": \"test-key\"")); assert!(compiled.contains("\"tool_a\"")); - assert!(!compiled.contains("\"command\"")); + assert!(!mcpg_config.contains("\"command\"")); let _ = fs::remove_dir_all(&temp_dir); } @@ -2724,11 +2737,12 @@ fn test_mcpg_config_http_based_mcp() { ); let compiled = fs::read_to_string(&output_path).unwrap(); + let mcpg_config = extract_mcpg_config(&compiled); assert!(compiled.contains("\"url\": \"https://mcp.dev.azure.com/myorg\"")); assert!(compiled.contains("\"X-MCP-Toolsets\": \"repos,wit\"")); assert!(compiled.contains("\"wit_get_work_item\"")); - assert!(!compiled.contains("\"command\"")); + assert!(!mcpg_config.contains("\"command\"")); let _ = fs::remove_dir_all(&temp_dir); } @@ -5310,8 +5324,8 @@ fn test_1es_compiled_output_is_valid_yaml() { "1ES output should contain SafeOutputs references" ); assert!( - compiled.contains("copilot --prompt="), - "1ES output should contain copilot invocation (engine_run substituted)" + compiled.contains("copilot-runner.js run"), + "1ES output should contain the fixed Copilot runner command" ); assert!( compiled.contains("threat-analysis"), @@ -5849,11 +5863,10 @@ fn extract_job_block<'a>(yaml: &'a str, name: &str) -> Option<&'a str> { Some(&yaml[start..end]) } -/// Per-job download placement: gate-only pipeline must put the download in -/// Setup and NOT in Agent. ADO jobs run on isolated VMs, so the gate's -/// install/download has to land in the same job as the gate step. +/// Gate-only pipelines stage the bundle in Setup for the gate and in both +/// Copilot jobs for the controller and runner. #[test] -fn test_gate_only_pipeline_downloads_bundle_in_setup_job_not_agent() { +fn test_gate_only_pipeline_downloads_bundle_in_all_consuming_jobs() { let yaml = compile_fixture("dedupe_gate_only.md"); let setup = extract_job_block(&yaml, "Setup").expect("Setup job should exist"); let agent = extract_job_block(&yaml, "Agent").expect("Agent job should exist"); @@ -5862,9 +5875,8 @@ fn test_gate_only_pipeline_downloads_bundle_in_setup_job_not_agent() { "Setup job is missing the script bundle download (gate consumer lives here)" ); assert!( - !agent.contains("Download ado-aw scripts"), - "Agent job should NOT have the script bundle download (gate-only, no runtime imports). \ - Agent block contents: {}", + agent.contains("Download ado-aw scripts"), + "Agent job must stage the Copilot controller/runner bundles. Agent block contents: {}", agent ); } @@ -5890,9 +5902,8 @@ fn test_imports_only_pipeline_downloads_bundle_in_agent_job_not_setup() { } } -/// Per-job download placement: when both gate and runtime imports are active, -/// the bundle is downloaded twice — once per consuming job. ADO's VM -/// isolation makes this correct architecture, not duplication waste. +/// When both gate and runtime imports are active, each isolated consuming job +/// stages the bundle: Setup, Agent, and Detection. #[test] fn test_both_features_active_downloads_bundle_in_both_jobs() { let yaml = compile_fixture("dedupe_both.md"); @@ -5908,23 +5919,20 @@ fn test_both_features_active_downloads_bundle_in_both_jobs() { ); assert_eq!( yaml.matches("Download ado-aw scripts").count(), - 2, - "Expected exactly two downloads — one per consuming job (Setup + Agent)" + 3, + "Expected exactly three downloads — Setup, Agent, and Detection" ); } -/// Per-job download placement: with neither gate nor runtime imports active, -/// no Node install or script-bundle download should appear anywhere. +/// Even with no gate or runtime imports, Agent and Detection stage the +/// controller and runner. #[test] -fn test_neither_feature_active_emits_no_node_or_download_anywhere() { +fn test_neither_feature_active_stages_copilot_bundles_in_copilot_jobs() { let yaml = compile_fixture("dedupe_neither.md"); - assert!( - !yaml.contains("UseNode@1"), - "No UseNode@1 expected when neither gate nor runtime imports are active" - ); - assert!( - !yaml.contains("Download ado-aw scripts"), - "No script bundle download expected when neither gate nor runtime imports are active" + assert_eq!( + yaml.matches("Download ado-aw scripts").count(), + 2, + "Agent and Detection must each stage the controller/runner bundles" ); } @@ -5993,12 +6001,11 @@ fn test_node_runtime_install_orders_after_ado_script_so_user_version_wins() { ado-script idx = {ado_script_install_idx}, user idx = {user_runtime_install_idx}" ); - // Both downloads of ado-script.zip remain unaffected (still exactly one - // in the Agent job in this fixture — no filters, so no Setup-side download). + // Agent and Detection each stage ado-script.zip; no Setup download exists. assert_eq!( yaml.matches("Download ado-aw scripts").count(), - 1, - "Expected exactly one ado-script.zip download (Agent job only; no gate active)" + 2, + "Expected exactly two ado-script.zip downloads (Agent + Detection)" ); } @@ -8175,12 +8182,23 @@ safe-outputs: let agent = job_block(&compiled, "Agent"); let detection = job_block(&compiled, "Detection"); - assert!(agent.contains("--model agent-model"), "{agent}"); + assert!( + agent.contains(r#""explicit_model":"agent-model""#), + "{agent}" + ); assert!(agent.contains("--reasoning-effort=high"), "{agent}"); - assert!(!agent.contains("--model detection-model"), "{agent}"); + assert!(!agent.contains("--model"), "{agent}"); + assert!( + !agent.contains(r#""explicit_model":"detection-model""#), + "{agent}" + ); assert!(!agent.contains("DETECTION_ENV"), "{agent}"); - assert!(detection.contains("--model detection-model"), "{detection}"); + assert!( + detection.contains(r#""explicit_model":"detection-model""#), + "{detection}" + ); + assert!(!detection.contains("--model"), "{detection}"); assert!(detection.contains("--reasoning-effort=low"), "{detection}"); assert!( !detection.contains("--reasoning-effort=high"), @@ -8204,6 +8222,107 @@ safe-outputs: assert!(detection.contains("2.0.2"), "{detection}"); } +#[test] +fn runtime_model_controls_compile_across_all_targets() { + for target in ["standalone", "1es", "job", "stage"] { + let target_field = if target == "standalone" { + String::new() + } else { + format!("target: {target}\n") + }; + let source = format!( + "---\nname: Runtime Model {target}\ndescription: Runtime model target coverage\n\ + {target_field}safe-outputs:\n noop: {{}}\n threat-detection: true\n---\n\n## Agent\n" + ); + let (ok, compiled, stderr) = + compile_inline_source(&format!("runtime-model-{target}"), &source); + assert!(ok, "{target} should compile: {stderr}"); + assert!( + compiled.contains( + "ADO_AW_MODEL_AGENT_COPILOT: $(ADO_AW_MODEL_AGENT_COPILOT)" + ), + "{target}: missing Agent runtime model env mapping" + ); + assert!( + compiled.contains( + "ADO_AW_MODEL_DETECTION_COPILOT: $(ADO_AW_MODEL_DETECTION_COPILOT)" + ), + "{target}: missing Detection runtime model env mapping" + ); + assert!( + compiled.contains("copilot-runner.js run"), + "{target}: fixed sandbox runner command must be emitted" + ); + assert!( + compiled.contains(r#""schema_version":2,"document_kind":"request","role":"agent""#) + && compiled.contains( + r#""schema_version":2,"document_kind":"request","role":"detection""# + ), + "{target}: Agent and Detection schema-v2 requests must be emitted" + ); + assert!( + compiled.contains( + r#"TRUSTED_CONTROLLER_DIR="$AGENT_TEMP/ado-aw-copilot-controller""# + ) && compiled.contains(r#"node "$TRUSTED_CONTROLLER_PATH" prepare"#) + && compiled.contains(r#"node "$TRUSTED_CONTROLLER_PATH" read-result"#), + "{target}: trusted controller preparation and result validation must use the host-private directory" + ); + assert!( + compiled.contains(r#"rm -f "$COPILOT_CONTROLLER_SOURCE_PATH""#), + "{target}: sandbox-visible controller source must be removed before AWF" + ); + assert!( + !compiled.contains(r#""result_path""#) + && !compiled.contains( + "/tmp/awf-tools/copilot-invocation-result.json" + ) + && !compiled.contains("copilot-controller.js run"), + "{target}: sandbox-visible requests and commands must not carry authoritative result/controller capabilities" + ); + assert!( + !compiled.contains("ADO_AW_EFFECTIVE_MODEL"), + "{target}: runtime model shell resolver must be absent" + ); + assert!( + !compiled.contains("--model"), + "{target}: compiler-generated model flags must be absent" + ); + } +} + +#[test] +fn runtime_model_control_resolves_after_prior_agent_step() { + let source = r###"--- +name: Runtime Model Set Variable +description: Runtime model task-scope resolution +steps: + - bash: | + echo "##vso[task.setvariable variable=ADO_AW_MODEL_AGENT_COPILOT]gpt-runtime" +safe-outputs: + threat-detection: false +--- + +## Agent +"###; + let (ok, compiled, stderr) = compile_inline_source("runtime-model-set-variable", source); + assert!(ok, "pipeline should compile: {stderr}"); + let agent = job_block(&compiled, "Agent"); + let producer = agent + .find("task.setvariable variable=ADO_AW_MODEL_AGENT_COPILOT") + .expect("model variable producer should be emitted"); + let consumer = agent + .find("Run copilot (AWF network isolated)") + .expect("Copilot controller/runner step should be emitted"); + assert!( + producer < consumer, + "trusted variable producer must run before the controller/runner task: {agent}" + ); + assert!( + agent.contains("ADO_AW_MODEL_AGENT_COPILOT: $(ADO_AW_MODEL_AGENT_COPILOT)"), + "controller task must resolve the model through its typed env mapping: {agent}" + ); +} + #[test] fn threat_detection_disabled_preserves_outputs_artifacts_and_manual_review() { let source = r#"--- @@ -10124,10 +10243,8 @@ fn test_github_app_token_hyphenated_private_key_variable() { /// When another ado-script bundle feature is active in the Agent job (here a /// safe-output activates the approval-summary bundle download), the mint step -/// must NOT trigger a second bundle download in that job — it reuses the -/// already-staged bundle. Proven by a delta: adding `github-app-token` to an -/// otherwise-identical workflow adds exactly ONE bundle download (the -/// Detection job, which has no extension-prepare phase), never two. +/// must NOT trigger another bundle download in either Copilot job because both +/// already stage the controller/runner bundles. #[test] fn test_github_app_token_reuses_staged_bundle_in_agent() { fn count_downloads(compiled: &str) -> usize { @@ -10150,14 +10267,11 @@ fn test_github_app_token_reuses_staged_bundle_in_agent() { ); assert_github_app_token_wiring(&with); - // Adding github-app-token stages the bundle only in Detection (Agent - // reuses its already-staged copy), so the download count grows by exactly 1. + // Adding github-app-token reuses the always-staged Agent and Detection bundles. assert_eq!( count_downloads(&with), - count_downloads(&without) + 1, - "github-app-token must add exactly one bundle download (Detection), \ - proving the Agent job reuses its staged bundle rather than \ - double-downloading. without={}, with={}", + count_downloads(&without), + "github-app-token must not add bundle downloads. without={}, with={}", count_downloads(&without), count_downloads(&with), ); diff --git a/tests/smoke/README.md b/tests/smoke/README.md index d6ddf30c7..910939ceb 100644 --- a/tests/smoke/README.md +++ b/tests/smoke/README.md @@ -35,7 +35,7 @@ credential class: | Lane | Secrets / service connections | Cases | | --- | --- | --- | -| `agentic` | `GITHUB_TOKEN`, `agent-playground-read`/`-write` | canary, ado-proxy, noop-target, custom-safe-output, multi-repo, janitor | +| `agentic` | `GITHUB_TOKEN`, `agent-playground-read`/`-write` | canary, ado-proxy, noop-target, custom-safe-output, multi-repo, runtime-model-queue, runtime-model-set-variable, janitor | | `infra` | none | *(reserved for AWF and the ado-proxy sidecar)* | No case currently files GitHub issues, so the lane holds no GitHub PAT beyond @@ -139,7 +139,10 @@ GitHub. "lane": "agentic", // must already exist; a NEW lane costs a registration "kind": "compiled", // or "raw" for hand-written YAML "modes": ["candidate", "released"], - "source": "tests/safe-outputs/my-case.md" + "source": "tests/safe-outputs/my-case.md", + "queueVariables": { // optional, non-secret ADO queue-time variables + "ADO_AW_MODEL_AGENT_COPILOT": "gpt-6-luna" + } } ``` @@ -178,10 +181,19 @@ Optional per-case assertions, so novel checks stay out of the harness code: "required": ["displayName: Start ado-proxy policy engine"], "forbidden": ["--network host"] }, - "requiredBuildTags": ["ado-aw-custom-job-{buildId}"] + "requiredBuildTags": ["ado-aw-custom-job-{buildId}"], + "requestedModels": { "agent": "gpt-6-luna" } } ``` +`requestedModels` runs `ado-aw audit` against the completed child and compares +the requested Agent and/or Detection model recorded in `overview.aw_info`. +Queue variables use the Build Queue API's `parameters` JSON string, matching +the encoding used by `az pipelines run --variables`, and therefore exercise +the same runtime source as variables supplied in the Azure DevOps Run Pipeline +UI. A lane definition may need the variable predeclared with +`allowOverride=true` when the project restricts queue-time variables. + ### `kind: raw` For pipelines that aren't compiled from front matter — for example a future @@ -246,6 +258,7 @@ malformed or mis-laned manifest fails locally rather than in ADO. | 6 | Exactly one ref per case is created, and every ref is deleted | both | | 7 | Each build ran its lane definition on that case's own ref | both | | 8 | Queued build count equals case count (no ref push CI-triggered a lane) | both | +| 9 | Queue-time and preceding same-job variables select the requested Agent model recorded by audit | candidate | ## Fork security boundary diff --git a/tests/smoke/REGISTERED.md b/tests/smoke/REGISTERED.md index 9249482cf..0fcd806fe 100644 --- a/tests/smoke/REGISTERED.md +++ b/tests/smoke/REGISTERED.md @@ -21,6 +21,12 @@ an explicitly supplied case ref. `infra` carries no cases yet, and a lane with no case in the running mode is never resolved, so it needs no definition until the first `infra` case lands. +The `agentic` lane declares an empty, non-secret +`ADO_AW_MODEL_AGENT_COPILOT` variable with `allowOverride=true`. The empty +definition value preserves the normal Copilot default for other cases, while +`runtime-model-queue` supplies a concrete value through the Build REST API and +verifies the resulting audit metadata. + **Only `agentic` needs registering at cutover** — one definition for the whole suite. diff --git a/tests/smoke/cases.json b/tests/smoke/cases.json index 79cffd7fa..3a4195844 100644 --- a/tests/smoke/cases.json +++ b/tests/smoke/cases.json @@ -73,6 +73,33 @@ "modes": ["candidate"], "source": "tests/smoke/multi-repo.md" }, + { + "id": "runtime-model-queue", + "lane": "agentic", + "kind": "compiled", + "modes": ["candidate"], + "source": "tests/smoke/runtime-model-queue.md", + "queueVariables": { + "ADO_AW_MODEL_AGENT_COPILOT": "gpt-6-luna" + }, + "assertions": { + "requestedModels": { + "agent": "gpt-6-luna" + } + } + }, + { + "id": "runtime-model-set-variable", + "lane": "agentic", + "kind": "compiled", + "modes": ["candidate"], + "source": "tests/smoke/runtime-model-set-variable.md", + "assertions": { + "requestedModels": { + "agent": "gpt-6-luna" + } + } + }, { "id": "janitor", "lane": "agentic", diff --git a/tests/smoke/runtime-model-queue.md b/tests/smoke/runtime-model-queue.md new file mode 100644 index 000000000..c1c42124e --- /dev/null +++ b/tests/smoke/runtime-model-queue.md @@ -0,0 +1,17 @@ +--- +name: "Candidate compiler smoke: queue-time runtime model" +description: "Proves an ADO queue-time variable selects the Agent Copilot model" +target: standalone +pool: + name: AZS-1ES-L-Playground-ubuntu-22.04 +engine: + id: copilot + timeout-minutes: 15 +safe-outputs: + noop: {} +--- + +## Queue-time runtime model smoke + +Call the `noop` safe-output tool exactly once with context +`runtime-model-queue-$(Build.BuildId)`, then stop. diff --git a/tests/smoke/runtime-model-set-variable.md b/tests/smoke/runtime-model-set-variable.md new file mode 100644 index 000000000..8728d0722 --- /dev/null +++ b/tests/smoke/runtime-model-set-variable.md @@ -0,0 +1,22 @@ +--- +name: "Candidate compiler smoke: task-setvariable runtime model" +description: "Proves a preceding same-job task.setvariable selects the Agent Copilot model" +target: standalone +pool: + name: AZS-1ES-L-Playground-ubuntu-22.04 +engine: + id: copilot + timeout-minutes: 15 +steps: + - bash: | + set -euo pipefail + echo "##vso[task.setvariable variable=ADO_AW_MODEL_AGENT_COPILOT]gpt-6-luna" + displayName: Select Agent runtime model +safe-outputs: + noop: {} +--- + +## Same-job runtime model smoke + +Call the `noop` safe-output tool exactly once with context +`runtime-model-set-variable-$(Build.BuildId)`, then stop.