diff --git a/agent/README.md b/agent/README.md index a5f98fcc3..46fbb04d1 100644 --- a/agent/README.md +++ b/agent/README.md @@ -481,3 +481,14 @@ agent/ The container **CMD** runs the app under `opentelemetry-instrument` with **uvicorn** using the **asyncio** event loop (not uvloop), avoiding known subprocess issues with uvloop. **Diagnostics:** `scripts/diagnostics/` holds optional smoke tests for local AgentCore debugging. They are not copied into the production Docker image. + +### MicroVM optional service configuration + +The shared `contracts/constants.json` allowlist transports `tool_gateway_url` → +`ABCA_TOOL_GATEWAY_URL`, `linear_vault_enabled` → `LINEAR_VAULT_ENABLED`, and +`linear_workload_identity_name` → `LINEAR_WORKLOAD_IDENTITY_NAME` in the MicroVM +`platform_config` payload. These optional values come from the deployed Gateway +and vault configuration. They contain identifiers, not credential values. +Rebuild the MicroVM snapshot from this checkout before enabling those features; +an older image does not recognize the new keys. AgentCore and ECS receive the +same settings through their deployment environment. diff --git a/agent/tests/test_server.py b/agent/tests/test_server.py index 067e01c6a..0ac6990ea 100644 --- a/agent/tests/test_server.py +++ b/agent/tests/test_server.py @@ -1824,6 +1824,9 @@ def test_wire_contract_is_exactly_the_documented_key_set(self): # wrong model that the IAM grant does not cover on any non-``global`` # deployment, surfacing as AccessDenied at turn 0. "anthropic_model": "ANTHROPIC_MODEL", + "tool_gateway_url": "ABCA_TOOL_GATEWAY_URL", + "linear_vault_enabled": "LINEAR_VAULT_ENABLED", + "linear_workload_identity_name": "LINEAR_WORKLOAD_IDENTITY_NAME", } def test_required_subset_is_exactly_the_four_run_blocking_keys(self): @@ -2005,7 +2008,7 @@ def test_every_allowlisted_key_is_installable(self, env_guard): key: _platform_config_value(key) for key in server.MICROVM_PLATFORM_CONFIG_ENV_BY_KEY } installed = server._install_platform_config(full) - assert installed == sorted(server.MICROVM_PLATFORM_CONFIG_ENV_BY_KEY.values()) + assert sorted(installed) == sorted(server.MICROVM_PLATFORM_CONFIG_ENV_BY_KEY.values()) for key, env_name in server.MICROVM_PLATFORM_CONFIG_ENV_BY_KEY.items(): assert os.environ[env_name] == _platform_config_value(key) diff --git a/cdk/AGENTS.md b/cdk/AGENTS.md index e328711e1..af40b7181 100644 --- a/cdk/AGENTS.md +++ b/cdk/AGENTS.md @@ -91,6 +91,7 @@ beforeAll(() => { ## Common mistakes +- **API Gateway Lambda permission growth** — Pass `allowTestInvoke: false` on every new `LambdaIntegration` and keep `scopePermissionToMethod` at its default `true`. `test/synthesis/deployment.test.ts` checks permissions and template budgets across the full deployment profile product, including nested stacks. - **Lambda bundling in unit tests** — `Template.fromStack()` synths the stack but bundling is disabled via `CDK_CONTEXT_JSON`. Do not re-enable globally; opt in per-test with `postCliContext` only when asserting on bundle output. Details: `test/setup/disable-bundling.ts`, #366. - **Cedar engine drift** — `@cedar-policy/cedar-wasm` and `cedarpy` share a Rust core. Bump both + parity fixtures in one commit. See `docs/design/CEDAR_HITL_GATES.md` §15.6 and `mise.toml` parity banner. - **Types out of sync** — `cdk/src/handlers/shared/types.ts` and `cli/src/types.ts` must match; CI runs `check-types-sync`. diff --git a/cdk/mise.toml b/cdk/mise.toml index 0ec5f7dde..58c5987fd 100644 --- a/cdk/mise.toml +++ b/cdk/mise.toml @@ -40,6 +40,11 @@ run = "yarn jest --coverage=false --runInBand" description = "cdk synth" run = ["mkdir -p $TMPDIR", "yarn synth"] +[tasks.census] +description = "Measure named CDK profiles and optionally audit synthesis stability (offline, unbundled)" +depends = [":compile"] +run = "yarn census" + [tasks."synth:quiet"] description = "cdk synth (quiet)" depends = [":compile"] diff --git a/cdk/package.json b/cdk/package.json index bccf3df3e..28f6cc43a 100644 --- a/cdk/package.json +++ b/cdk/package.json @@ -11,7 +11,8 @@ "test": "jest --maxWorkers=${JEST_MAX_WORKERS:-25%}", "eslint": "eslint --fix src test", "synth": "npx cdk synth", - "synth:quiet": "npx cdk synth -q" + "synth:quiet": "npx cdk synth -q", + "census": "node -r ts-node/register/transpile-only src/synthesis/cli.ts" }, "dependencies": { "@aws-cdk/aws-bedrock-alpha": "2.260.0-alpha.0", diff --git a/cdk/src/blueprints/definitions.ts b/cdk/src/blueprints/definitions.ts new file mode 100644 index 000000000..1d6f93ccb --- /dev/null +++ b/cdk/src/blueprints/definitions.ts @@ -0,0 +1,58 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import type { Node } from 'constructs'; +import type { BlueprintProps } from '../constructs/blueprint'; + +/** Repository configuration before it is bound to a RepoTable or a stack. */ +export interface BlueprintDefinition extends Omit { + /** Stable construct ID for this repository's provisioning controller. */ + readonly id: string; +} + +/** Resolve once so repository provisioning and the network use the same inputs. */ +export function resolveBlueprintDefinitions( + node: Node, + environment: NodeJS.ProcessEnv = process.env, +): readonly BlueprintDefinition[] { + const definitions: BlueprintDefinition[] = [{ + id: 'AgentPluginsBlueprint', + repo: environment.BLUEPRINT_REPO ?? node.tryGetContext('blueprintRepo') ?? 'awslabs/agent-plugins', + }]; + // Optional per-repository registry assets (#246); preserve the deployed IDs + // and environment/context precedence while moving configuration out of a stack. + const forkRepo = environment.FORK_BLUEPRINT_REPO ?? node.tryGetContext('forkBlueprintRepo'); + if (forkRepo) { + definitions.push({ + id: 'ForkBlueprint', + repo: forkRepo, + assets: { + mcpServers: ['registry://mcp_server/acme/aws-knowledge@^1.0.0'], + cedarPolicyModules: ['registry://cedar_policy_module/acme/guard@^1.0.0'], + skills: ['registry://skill/acme/readme-helper@^1.0.0'], + }, + }); + } + return definitions; +} + +/** Aggregate plain domain strings without referring to repository resources. */ +export function blueprintEgressDomains(definitions: readonly BlueprintDefinition[]): string[] { + return [...new Set(definitions.flatMap(definition => definition.networking?.egressAllowlist ?? []))]; +} diff --git a/cdk/src/constructs/agent-session-role.ts b/cdk/src/constructs/agent-session-role.ts index 1602b734d..94ed12a0a 100644 --- a/cdk/src/constructs/agent-session-role.ts +++ b/cdk/src/constructs/agent-session-role.ts @@ -18,12 +18,12 @@ */ import * as bedrock from '@aws-cdk/aws-bedrock-alpha'; -import { Duration } from 'aws-cdk-lib'; +import { AspectPriority, Aspects, Duration, Lazy } from 'aws-cdk-lib'; import * as dynamodb from 'aws-cdk-lib/aws-dynamodb'; import * as iam from 'aws-cdk-lib/aws-iam'; import * as s3 from 'aws-cdk-lib/aws-s3'; import { NagSuppressions } from 'cdk-nag'; -import { Construct } from 'constructs'; +import { Construct, IConstruct } from 'constructs'; /** S3 key prefixes the agent writes/reads, scoped per tenant. */ const TRACE_KEY_PREFIX = 'traces'; @@ -42,7 +42,9 @@ export interface AgentSessionRoleProps { * the trust surface. Both run the same trusted agent code, which sources the * `{user_id, repo, task_id}` tag values from the resolved TaskConfig. */ - readonly assumingRoles: iam.IRole[]; + readonly assumingRoles?: iam.IRole[]; + /** Admit deployed backends with admitComputeRole after constructing this role. */ + readonly deferComputeRoleBinding?: boolean; /** * The four task-scoped DynamoDB tables, all partitioned by `task_id`. The @@ -125,11 +127,12 @@ export class AgentSessionRole extends Construct { /** The SessionRole. Assumed by the agent at task startup. */ public readonly role: iam.Role; + private readonly admittedRoles = new Set(); constructor(scope: Construct, id: string, props: AgentSessionRoleProps) { super(scope, id); - if (props.assumingRoles.length === 0) { + if (!props.assumingRoles?.length && !props.deferComputeRoleBinding) { // A SessionRole no principal can assume is dead weight and would // synthesize an empty/invalid trust policy. Fail at synth instead. throw new Error( @@ -137,12 +140,22 @@ export class AgentSessionRole extends Construct { ); } - const [firstAssumingRole] = props.assumingRoles; + if (props.assumingRoles?.length && props.deferComputeRoleBinding) { + throw new Error('Specify assumingRoles or deferComputeRoleBinding, not both'); + } + this.node.addValidation({ validate: () => this.admittedRoles.size ? [] : ['AgentSessionRole requires an admitted compute role before synthesis'] }); + const firstAssumingRoleArn = props.assumingRoles?.[0]?.roleArn ?? Lazy.string({ + produce: () => { + const first = this.admittedRoles.values().next().value; + if (!first) throw new Error('AgentSessionRole requires an admitted compute role before synthesis'); + return first.roleArn; + }, + }); // CDK requires assumedBy; additional principals are admitted via // admitComputeRole so trust + grant always wire together. this.role = new iam.Role(this, 'Role', { - assumedBy: new iam.ArnPrincipal(firstAssumingRole.roleArn), + assumedBy: new iam.ArnPrincipal(firstAssumingRoleArn), description: 'Per-task scoped credentials for ABCA agent tenant-data access ' + '(DynamoDB task rows + S3 trace/attachment objects), constrained by ' @@ -226,30 +239,34 @@ export class AgentSessionRole extends Construct { invokable.grantInvoke(this.role); } - // The object-level prefix conditions above already constrain access to the - // session's own tenant prefix; the remaining wildcard is the per-object - // suffix (task_id/attachment_id/filename), which is the intended scope. - NagSuppressions.addResourceSuppressions( - this.role, - [ - { + // Model-list overrides can spill these grants into managed policies created + // during synthesis. Visit every policy before cdk-nag, including that late + // overflow, and allow only the wildcard shapes this construct requires. + Aspects.of(this.role).add({ + visit(node: IConstruct): void { + if (!(node instanceof iam.CfnRole || node instanceof iam.CfnPolicy || node instanceof iam.CfnManagedPolicy)) return; + NagSuppressions.addResourceSuppressions(node, [{ id: 'AwsSolutions-IAM5', reason: 'Resource wildcards are the per-object suffix under a tenant-scoped ' + 'prefix (traces/${aws:PrincipalTag/user_id}/*, ' + 'attachments/${aws:PrincipalTag/user_id}/*, ' - + 'artifacts/${aws:PrincipalTag/task_id}/*) and the DynamoDB item ' - + 'set gated by a dynamodb:LeadingKeys = ${aws:PrincipalTag/task_id} ' - + 'condition — narrower than the compute role this replaces. Bedrock ' - + 'InvokeModel resources are the explicit model + inference-profile ' - + 'ARNs from grantInvoke (cross-region profiles fan out to per-region ' - + 'foundation-model ARNs), matching the compute role grant (#215).', - }, - ], - true, - ); + + 'artifacts/${aws:PrincipalTag/task_id}/*). Bedrock grantInvoke uses ' + + 'InvokeModel* for synchronous/streaming invocation and a region ' + + 'wildcard for each literal foundation-model ID routed by a ' + + 'cross-region inference profile, matching the compute role (#215).', + appliesTo: [ + // cdk-nag renders policy variables as in finding IDs. + { regex: '/^Resource::[^*?]+/(traces|attachments)//\\*$/' }, + { regex: '/^Resource::[^*?]+/artifacts//\\*$/' }, + 'Action::bedrock:InvokeModel*', + { regex: '/^Resource::arn:[^*?]+:bedrock:\\*::foundation-model/[^*?]+$/' }, + ], + }]); + }, + }, { priority: AspectPriority.MUTATING }); - for (const computeRole of props.assumingRoles) { + for (const computeRole of props.assumingRoles ?? []) { this.admitComputeRole(computeRole); } } @@ -260,6 +277,8 @@ export class AgentSessionRole extends Construct { * `sts:AssumeRole`/`sts:TagSession` on the compute role's identity policy. */ public admitComputeRole(computeRole: iam.IRole): void { + if (this.admittedRoles.has(computeRole)) return; + this.admittedRoles.add(computeRole); this.addTrustForComputeRole(computeRole); this.grantAssumeToComputeRole(computeRole); } diff --git a/cdk/src/constructs/agent-vpc.ts b/cdk/src/constructs/agent-vpc.ts index 91e06d6f2..d5f10507f 100644 --- a/cdk/src/constructs/agent-vpc.ts +++ b/cdk/src/constructs/agent-vpc.ts @@ -17,7 +17,7 @@ * SOFTWARE. */ -import { RemovalPolicy } from 'aws-cdk-lib'; +import { RemovalPolicy, Tags } from 'aws-cdk-lib'; import * as ec2 from 'aws-cdk-lib/aws-ec2'; import * as logs from 'aws-cdk-lib/aws-logs'; import { NagSuppressions } from 'cdk-nag'; @@ -37,6 +37,26 @@ const DEFAULT_AGENT_VPC_AZS = 2; /** AgentCore high-availability floor: at least two zones. */ const MIN_AGENT_VPC_AZS = 2; +const MAX_RESERVED_NETWORK_AZS = 6; + +/** Reserve unused AZ address slots without allocating subnets or other resources. */ +export function resolveNetworkReservedAzs(value: unknown): number { + if (value === undefined) return 0; + const count = typeof value === 'string' && value.trim() !== '' ? Number(value) : value; + // Six slots cover the largest current regional AZ count and bound CDK's + // placeholder allocation. The normal three-to-two-zone reduction needs one. + if (typeof count !== 'number' || !Number.isInteger(count) || count < 0 || count > MAX_RESERVED_NETWORK_AZS) { + throw new Error(`networkReservedAzs must be an integer from 0 to ${MAX_RESERVED_NETWORK_AZS}`); + } + return count; +} + +/** The references consumed by any compute backend, regardless of stack ownership. */ +export interface AgentNetwork { + readonly vpc: ec2.IVpc; + readonly runtimeSecurityGroup: ec2.ISecurityGroup; +} + /** * Properties for the AgentVpc construct. */ @@ -94,6 +114,13 @@ export interface AgentVpcProps { * @default RemovalPolicy.DESTROY */ readonly removalPolicy?: RemovalPolicy; + + /** + * Original AgentVpc construct path used for generated Name tags and endpoint + * security-group descriptions. Keeps service properties stable across a move. + * @default - this construct's current path + */ + readonly resourcePath?: string; } /** @@ -103,7 +130,7 @@ export interface AgentVpcProps { * and NAT for internet egress (GitHub and package registries). * Flow logs are enabled for audit. */ -export class AgentVpc extends Construct { +export class AgentVpc extends Construct implements AgentNetwork { /** The VPC where the Runtime will be deployed. */ public readonly vpc: ec2.Vpc; @@ -134,14 +161,20 @@ export class AgentVpc extends Construct { const maxAzs = props.maxAzs ?? DEFAULT_AGENT_VPC_AZS; const natGateways = props.natGateways ?? 1; const removalPolicy = props.removalPolicy ?? RemovalPolicy.DESTROY; + const reservedAzs = resolveNetworkReservedAzs(this.node.tryGetContext('networkReservedAzs')); // --- VPC --- // When explicit AZs are provided (to target AgentCore-supported physical // zones), pass them directly and omit maxAzs — CDK does not allow both. this.vpc = new ec2.Vpc(this, 'Vpc', { ...(pinnedAzs?.length - ? { availabilityZones: pinnedAzs } + // CDK appends reserved placeholders to this array; keep the caller's + // real AZ selection intact for application wiring and diagnostics. + ? { availabilityZones: [...pinnedAzs] } : { maxAzs }), + // Keep active + reserved slots constant during an AZ reduction so CDK's + // private subnet CIDRs do not shift and force subnet replacements. + reservedAzs, natGateways, restrictDefaultSecurityGroup: true, subnetConfiguration: [ @@ -158,16 +191,27 @@ export class AgentVpc extends Construct { ], }); + const resourceVpcPath = `${props.resourcePath ?? this.node.path}/Vpc`; + if (props.resourcePath !== undefined) { + // CDK gives the VPC and each subnet their own inherited Name tag. Preserve + // those scopes so routes, NAT, endpoints and the IGW keep their old names. + Tags.of(this.vpc).add('Name', resourceVpcPath); + for (const subnet of [...this.vpc.publicSubnets, ...this.vpc.privateSubnets]) { + Tags.of(subnet).add('Name', `${resourceVpcPath}${subnet.node.path.slice(this.vpc.node.path.length)}`); + } + } + // --- Flow logs (satisfies AwsSolutions-VPC7) --- const flowLogGroup = new logs.LogGroup(this, 'FlowLogGroup', { retention: logs.RetentionDays.ONE_MONTH, removalPolicy, }); - this.vpc.addFlowLog('FlowLog', { + const flowLog = this.vpc.addFlowLog('FlowLog', { destination: ec2.FlowLogDestination.toCloudWatchLogs(flowLogGroup), trafficType: ec2.FlowLogTrafficType.ALL, }); + if (props.resourcePath !== undefined) Tags.of(flowLog).add('Name', `${resourceVpcPath}/FlowLog`); NagSuppressions.addResourceSuppressions(this.vpc, [ { @@ -210,11 +254,17 @@ export class AgentVpc extends Construct { ]; for (const ep of interfaceEndpoints) { - this.vpc.addInterfaceEndpoint(ep.id, { + const endpoint = this.vpc.addInterfaceEndpoint(ep.id, { service: ep.service, privateDnsEnabled: true, subnets: { subnetType: ec2.SubnetType.PRIVATE_WITH_EGRESS }, }); + if (props.resourcePath !== undefined) { + // GroupDescription is replacement-sensitive. The CDK default includes + // the current stack path, so explicitly keep the pre-extraction value. + const group = endpoint.node.findChild('SecurityGroup').node.defaultChild as ec2.CfnSecurityGroup; + group.groupDescription = `${resourceVpcPath}/${ep.id}/SecurityGroup`; + } } } } diff --git a/cdk/src/constructs/ecs-agent-cluster.ts b/cdk/src/constructs/ecs-agent-cluster.ts index 7e487f2bc..8a0ba3799 100644 --- a/cdk/src/constructs/ecs-agent-cluster.ts +++ b/cdk/src/constructs/ecs-agent-cluster.ts @@ -322,6 +322,7 @@ export class EcsAgentCluster extends Construct { public readonly securityGroup: ec2.SecurityGroup; public readonly containerName: string; public readonly taskRoleArn: string; + public readonly logGroup: logs.LogGroup; public readonly executionRoleArn: string; constructor(scope: Construct, id: string, props: EcsAgentClusterProps) { @@ -349,7 +350,7 @@ export class EcsAgentCluster extends Construct { ); // CloudWatch log group for agent task output - const logGroup = new logs.LogGroup(this, 'TaskLogGroup', { + const logGroup = this.logGroup = new logs.LogGroup(this, 'TaskLogGroup', { retention: logs.RetentionDays.THREE_MONTHS, removalPolicy: RemovalPolicy.DESTROY, }); diff --git a/cdk/src/constructs/linear-identity-vault.ts b/cdk/src/constructs/linear-identity-vault.ts index a176945fe..10715bbb1 100644 --- a/cdk/src/constructs/linear-identity-vault.ts +++ b/cdk/src/constructs/linear-identity-vault.ts @@ -28,13 +28,13 @@ // framework with a bundled `onEvent` handler (mirrors registry.ts). Workload- // identity create/delete are synchronous, so no `isComplete` poller is needed. import * as path from 'path'; -import { ArnFormat, CustomResource, Duration, Stack } from 'aws-cdk-lib'; +import { ArnFormat, AspectPriority, Aspects, CustomResource, Duration, Stack } from 'aws-cdk-lib'; import * as iam from 'aws-cdk-lib/aws-iam'; import { Architecture, Runtime } from 'aws-cdk-lib/aws-lambda'; import * as lambda from 'aws-cdk-lib/aws-lambda-nodejs'; import * as cr from 'aws-cdk-lib/custom-resources'; import { NagSuppressions } from 'cdk-nag'; -import { Construct } from 'constructs'; +import { Construct, IConstruct } from 'constructs'; const PROVISION_TIMEOUT_SECONDS = 60; const PROVISION_MEMORY_MB = 256; @@ -73,6 +73,8 @@ export interface LinearIdentityVaultProps { * (webhook processor, orchestrator, agent session role). */ export class LinearIdentityVault extends Construct { + private readonly annotatedMintGrantees = new WeakSet(); + /** The provisioned workload identity name (stable natural id). */ public readonly workloadName: string; @@ -317,18 +319,40 @@ export class LinearIdentityVault extends Construct { // Live-verified rather than reasoned: under the scoped grant a Linear mint for a // consented workspace succeeds, and `GetSecretValue` on the GitHub provider's // secret is denied. The trailing `*` covers the id suffix the service appends. + const credentialSecretArn = stack.formatArn({ + service: 'secretsmanager', + resource: 'secret', + arnFormat: ArnFormat.COLON_RESOURCE_NAME, + resourceName: `bedrock-agentcore-identity!default/oauth2/${LINEAR_CREDENTIAL_PROVIDER_PREFIX}*`, + }); grantee.grantPrincipal.addToPrincipalPolicy( new iam.PolicyStatement({ actions: ['secretsmanager:GetSecretValue'], - resources: [ - stack.formatArn({ - service: 'secretsmanager', - resource: 'secret', - arnFormat: ArnFormat.COLON_RESOURCE_NAME, - resourceName: `bedrock-agentcore-identity!default/oauth2/${LINEAR_CREDENTIAL_PROVIDER_PREFIX}*`, - }), - ], + resources: [credentialSecretArn], }), ); + this.annotateMintGrant(grantee); + } + + /** Follow these specific grant resources into CDK's lazily created overflow policies. */ + private annotateMintGrant(grantee: iam.IGrantable): void { + if (!Construct.isConstruct(grantee) || this.annotatedMintGrantees.has(grantee)) return; + this.annotatedMintGrantees.add(grantee); + Aspects.of(grantee).add({ + visit(node: IConstruct): void { + if (!(node instanceof iam.CfnPolicy || node instanceof iam.CfnManagedPolicy)) return; + NagSuppressions.addResourceSuppressions(node, [{ + id: 'AwsSolutions-IAM5', + reason: 'Linear OAuth providers are created per workspace after deployment. Minting requires the Linear-only provider prefix and its service-owned OAuth secret suffix; unrelated providers and secrets remain excluded.', + // Account/region/partition can render as literals or pseudo-parameter + // references, but must not contain wildcards. Only the provider/secret + // suffix varies; widening the partition, account or region must fail. + appliesTo: [ + { regex: `/^Resource::arn:[^*?]+:bedrock-agentcore:[^*?]+:token-vault/default/oauth2credentialprovider/${LINEAR_CREDENTIAL_PROVIDER_PREFIX}\\*$/` }, + { regex: `/^Resource::arn:[^*?]+:secretsmanager:[^*?]+:secret:bedrock-agentcore-identity!default/oauth2/${LINEAR_CREDENTIAL_PROVIDER_PREFIX}\\*$/` }, + ], + }]); + }, + }, { priority: AspectPriority.MUTATING }); } } diff --git a/cdk/src/constructs/solution-ua-aspect.ts b/cdk/src/constructs/solution-ua-aspect.ts index 26bd766ad..bcde54d5e 100644 --- a/cdk/src/constructs/solution-ua-aspect.ts +++ b/cdk/src/constructs/solution-ua-aspect.ts @@ -17,7 +17,7 @@ * SOFTWARE. */ -import { IAspect } from 'aws-cdk-lib'; +import { CfnResource, IAspect } from 'aws-cdk-lib'; import * as lambda from 'aws-cdk-lib/aws-lambda'; import { IConstruct } from 'constructs'; @@ -101,6 +101,11 @@ export class SolutionUaAspect implements IAspect { } if (node instanceof lambda.Function) { node.addEnvironment('AWS_SDK_UA_APP_ID', this.appId); + } else if (CfnResource.isCfnResource(node) && node.cfnResourceType === 'AWS::Lambda::Function' + && !(node.node.scope instanceof lambda.Function)) { + // Core CDK providers (e.g. default-SG restriction and S3 auto-delete) use + // generic CfnResource directly, bypassing the L2 environment API above. + node.addPropertyOverride('Environment.Variables.AWS_SDK_UA_APP_ID', this.appId); } } } diff --git a/cdk/src/constructs/task-api.ts b/cdk/src/constructs/task-api.ts index be735cad2..577f95f63 100644 --- a/cdk/src/constructs/task-api.ts +++ b/cdk/src/constructs/task-api.ts @@ -868,8 +868,8 @@ export class TaskApi extends Construct { // API Gateway console's "TEST" button, which nothing here invokes. Real traffic is // unaffected: `scopePermissionToMethod` stays at its default `true`, so each route // keeps its own narrowly-scoped `SourceArn`. Keep new routes consistent; - // `test/stacks/agent.test.ts` asserts no `test-invoke-stage` permission is ever - // emitted. + // `test/synthesis/deployment.test.ts` checks method-scoped permissions and + // rejects `test-invoke-stage` grants in every parent/nested deployment template. // --- API resource tree: /tasks --- const tasks = this.api.root.addResource('tasks'); diff --git a/cdk/src/constructs/task-dashboard.ts b/cdk/src/constructs/task-dashboard.ts index 6f4e40f7a..ce2cf782c 100644 --- a/cdk/src/constructs/task-dashboard.ts +++ b/cdk/src/constructs/task-dashboard.ts @@ -39,7 +39,7 @@ export interface TaskDashboardProps { * The ARN of the AgentCore runtime, used as the ``Resource`` dimension * for native CloudWatch metrics under the ``AWS/Bedrock`` namespace. */ - readonly runtimeArn: string; + readonly runtimeArn?: string; } /** @@ -255,72 +255,74 @@ export class TaskDashboard extends Construct { }), ); - // --- Row 7: AgentCore Runtime native metrics --- - // Namespace AWS/Bedrock, dimensions { Service, Resource } scoped to this - // runtime. Metrics are batched at 1-minute intervals by the runtime. - const metricDimensions = { - Service: 'AgentCore.Runtime', - Resource: props.runtimeArn, - }; + if (props.runtimeArn) { + // --- Row 7: AgentCore Runtime native metrics --- + // Namespace AWS/Bedrock, dimensions { Service, Resource } scoped to this + // runtime. Metrics are batched at 1-minute intervals by the runtime. + const metricDimensions = { + Service: 'AgentCore.Runtime', + Resource: props.runtimeArn, + }; - this.dashboard.addWidgets( - new cloudwatch.GraphWidget({ - title: 'Runtime Invocations', - left: [ - new cloudwatch.Metric({ - namespace: 'AWS/Bedrock', - metricName: 'Invocations', - dimensionsMap: metricDimensions, - statistic: 'Sum', - period: Duration.hours(1), - }), - ], - width: 8, - height: 6, - }), - new cloudwatch.GraphWidget({ - title: 'Runtime Errors', - left: [ - new cloudwatch.Metric({ - namespace: 'AWS/Bedrock', - metricName: 'SystemErrors', - dimensionsMap: metricDimensions, - statistic: 'Sum', - period: Duration.hours(1), - }), - new cloudwatch.Metric({ - namespace: 'AWS/Bedrock', - metricName: 'UserErrors', - dimensionsMap: metricDimensions, - statistic: 'Sum', - period: Duration.hours(1), - }), - ], - width: 8, - height: 6, - }), - new cloudwatch.GraphWidget({ - title: 'Runtime Latency (p50 / p99)', - left: [ - new cloudwatch.Metric({ - namespace: 'AWS/Bedrock', - metricName: 'Latency', - dimensionsMap: metricDimensions, - statistic: 'p50', - period: Duration.hours(1), - }), - new cloudwatch.Metric({ - namespace: 'AWS/Bedrock', - metricName: 'Latency', - dimensionsMap: metricDimensions, - statistic: 'p99', - period: Duration.hours(1), - }), - ], - width: 8, - height: 6, - }), - ); + this.dashboard.addWidgets( + new cloudwatch.GraphWidget({ + title: 'Runtime Invocations', + left: [ + new cloudwatch.Metric({ + namespace: 'AWS/Bedrock', + metricName: 'Invocations', + dimensionsMap: metricDimensions, + statistic: 'Sum', + period: Duration.hours(1), + }), + ], + width: 8, + height: 6, + }), + new cloudwatch.GraphWidget({ + title: 'Runtime Errors', + left: [ + new cloudwatch.Metric({ + namespace: 'AWS/Bedrock', + metricName: 'SystemErrors', + dimensionsMap: metricDimensions, + statistic: 'Sum', + period: Duration.hours(1), + }), + new cloudwatch.Metric({ + namespace: 'AWS/Bedrock', + metricName: 'UserErrors', + dimensionsMap: metricDimensions, + statistic: 'Sum', + period: Duration.hours(1), + }), + ], + width: 8, + height: 6, + }), + new cloudwatch.GraphWidget({ + title: 'Runtime Latency (p50 / p99)', + left: [ + new cloudwatch.Metric({ + namespace: 'AWS/Bedrock', + metricName: 'Latency', + dimensionsMap: metricDimensions, + statistic: 'p50', + period: Duration.hours(1), + }), + new cloudwatch.Metric({ + namespace: 'AWS/Bedrock', + metricName: 'Latency', + dimensionsMap: metricDimensions, + statistic: 'p99', + period: Duration.hours(1), + }), + ], + width: 8, + height: 6, + }), + ); + } // --- Row 8+9: Cedar HITL approval widgets (§11.3, IMPL-28) -------------- // diff --git a/cdk/src/constructs/task-orchestrator.ts b/cdk/src/constructs/task-orchestrator.ts index ba3398883..ddd429ce2 100644 --- a/cdk/src/constructs/task-orchestrator.ts +++ b/cdk/src/constructs/task-orchestrator.ts @@ -73,7 +73,12 @@ export interface TaskOrchestratorProps { /** * ARN of the AgentCore runtime. */ - readonly runtimeArn: string; + readonly runtimeArn?: string; + /** + * Backends deployed by this stack; the first is the repository default. + * Omit only for legacy composition. + */ + readonly deployedComputeTypes?: ReadonlyArray<'agentcore' | 'ecs' | 'lambda-microvm'>; /** * The DynamoDB repo config table. When provided, the orchestrator loads @@ -279,6 +284,8 @@ export interface TaskOrchestratorProps { * no per-repo override failed at turn 0 with AccessDenied. */ readonly anthropicModel: string; + /** Optional SigV4 tool gateway, forwarded to the MicroVM guest. */ + readonly toolGatewayUrl?: string; }; /** @@ -388,6 +395,18 @@ export class TaskOrchestrator extends Construct { constructor(scope: Construct, id: string, props: TaskOrchestratorProps) { super(scope, id); + if (props.deployedComputeTypes) { + const backends = props.deployedComputeTypes; + const agentcore = backends.includes('agentcore'); + const ecs = backends.includes('ecs'); + if (agentcore !== Boolean(props.runtimeArn) + || (!agentcore && props.additionalRuntimeArns?.length) + || ecs !== Boolean(props.ecsConfig) + || (!backends.includes('lambda-microvm') && props.microvmConfig)) { + throw new Error(`TaskOrchestrator configuration must match the deployed '${backends.join(', ')}' backends`); + } + } + if (props.guardrailId && !props.guardrailVersion) { throw new Error('guardrailVersion is required when guardrailId is provided'); } @@ -446,7 +465,8 @@ export class TaskOrchestrator extends Construct { TASK_TABLE_NAME: props.taskTable.tableName, TASK_EVENTS_TABLE_NAME: props.taskEventsTable.tableName, USER_CONCURRENCY_TABLE_NAME: props.userConcurrencyTable.tableName, - RUNTIME_ARN: props.runtimeArn, + ...(props.runtimeArn && { RUNTIME_ARN: props.runtimeArn }), + ...(props.deployedComputeTypes && { DEPLOYED_COMPUTE_TYPE: props.deployedComputeTypes.join(',') }), MAX_CONCURRENT_TASKS_PER_USER: String(maxConcurrent), TASK_RETENTION_DAYS: String(props.taskRetentionDays ?? DEFAULT_TASK_RETENTION_DAYS), ...(props.repoTable && { REPO_TABLE_NAME: props.repoTable.tableName }), @@ -517,6 +537,7 @@ export class TaskOrchestrator extends Construct { // the backend that depends on this block, and it fell back to the Python // literal in `agent/src/config.py` regardless of the deployed geography. ANTHROPIC_MODEL: props.agentPlatformConfig.anthropicModel, + ...(props.agentPlatformConfig.toolGatewayUrl && { ABCA_TOOL_GATEWAY_URL: props.agentPlatformConfig.toolGatewayUrl }), }), }, bundling: orchestratorBundling, @@ -577,16 +598,18 @@ export class TaskOrchestrator extends Construct { // `BedrockAgentCoreContext.get_workload_access_token()` returns // non-None). Without this grant, `InvokeAgentRuntimeCommand` with // `runtimeUserId` set fails with AccessDenied. - const runtimeArns = [props.runtimeArn, ...(props.additionalRuntimeArns ?? [])]; + const runtimeArns = [...(props.runtimeArn ? [props.runtimeArn] : []), ...(props.additionalRuntimeArns ?? [])]; const runtimeResources = runtimeArns.flatMap(arn => [arn, `${arn}/*`]); - this.fn.addToRolePolicy(new iam.PolicyStatement({ - actions: [ - 'bedrock-agentcore:InvokeAgentRuntime', - 'bedrock-agentcore:InvokeAgentRuntimeForUser', - 'bedrock-agentcore:StopRuntimeSession', - ], - resources: runtimeResources, - })); + if (runtimeResources.length) { + this.fn.addToRolePolicy(new iam.PolicyStatement({ + actions: [ + 'bedrock-agentcore:InvokeAgentRuntime', + 'bedrock-agentcore:InvokeAgentRuntimeForUser', + 'bedrock-agentcore:StopRuntimeSession', + ], + resources: runtimeResources, + })); + } // Registry (#246): read-only access so the orchestrator can resolve the // Blueprint's registry:// asset refs at task start. Scoped to THIS registry diff --git a/cdk/src/constructs/tool-gateway.ts b/cdk/src/constructs/tool-gateway.ts index 0ffa87beb..6ddcd7f57 100644 --- a/cdk/src/constructs/tool-gateway.ts +++ b/cdk/src/constructs/tool-gateway.ts @@ -18,11 +18,11 @@ */ import * as path from 'path'; -import { Duration } from 'aws-cdk-lib'; +import { Duration, Stack } from 'aws-cdk-lib'; import * as agentcore from 'aws-cdk-lib/aws-bedrockagentcore'; import * as dynamodb from 'aws-cdk-lib/aws-dynamodb'; import type { IGrantable } from 'aws-cdk-lib/aws-iam'; -import { Architecture, Runtime } from 'aws-cdk-lib/aws-lambda'; +import { Architecture, CfnFunction, Runtime } from 'aws-cdk-lib/aws-lambda'; import * as lambda from 'aws-cdk-lib/aws-lambda-nodejs'; import { NagSuppressions } from 'cdk-nag'; import { Construct } from 'constructs'; @@ -146,6 +146,16 @@ export class ToolGateway extends Construct { ]), }); + // The L2's grantInvoke includes this function's qualified versions. + // Keep its binding/permission contract, with an exception for only that + // generated ARN family. Other gateway wildcards must still fail the audit. + const functionId = Stack.of(this).getLogicalId(this.repoConfigFn.node.defaultChild as CfnFunction); + NagSuppressions.addResourceSuppressions(this.gateway.role, [{ + id: 'AwsSolutions-IAM5', + reason: 'CDK Lambda target binding grants invoke on the single RepoConfig function and its qualified versions; it cannot invoke other functions.', + appliesTo: [`Resource::<${functionId}.Arn>:*`], + }], true); + // grantReadData → dynamodb:GetItem/Query/... on the table AND index/* ARNs. NagSuppressions.addResourceSuppressions(this.repoConfigFn, [ { diff --git a/cdk/src/handlers/shared/compute-backend.ts b/cdk/src/handlers/shared/compute-backend.ts new file mode 100644 index 000000000..14134180d --- /dev/null +++ b/cdk/src/handlers/shared/compute-backend.ts @@ -0,0 +1,55 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +/** A deployment selects one or more backends; the first listed is the repository default. */ +export type ComputeBackend = 'agentcore' | 'ecs' | 'lambda-microvm'; + +export function resolveComputeBackend(value: unknown = 'agentcore'): ComputeBackend { + if (value === 'agentcore' || value === 'ecs' || value === 'lambda-microvm') return value; + throw new Error(`compute_type must be agentcore, ecs or lambda-microvm; received '${String(value)}'`); +} + +/** + * Resolve the deployed backends from `compute_types` (comma list or array). + * Without it, a legacy `compute_type=ecs|lambda-microvm` keeps the additive + * shape `main` deploys today: AgentCore (still the repository default) plus + * that backend. + */ +export function resolveComputeBackends(computeTypes: unknown, legacyComputeType?: unknown): ComputeBackend[] { + if (computeTypes === undefined) { + const legacy = resolveComputeBackend(legacyComputeType); + return legacy === 'agentcore' ? ['agentcore'] : ['agentcore', legacy]; + } + const raw: unknown[] = Array.isArray(computeTypes) ? computeTypes + : typeof computeTypes === 'string' ? computeTypes.split(',') : []; + if (raw.length === 0 || raw.some(value => typeof value !== 'string' || !value.trim())) { + throw new Error('compute_types must be a non-empty comma-separated list or array of agentcore, ecs or lambda-microvm'); + } + return [...new Set(raw.map(value => resolveComputeBackend((value as string).trim())))]; +} + +/** Legacy deployments without a selection retain their per-repository routing. */ +export function resolveRepositoryBackend(override: unknown, deployed: string | undefined): ComputeBackend { + const backends = deployed === undefined ? undefined : resolveComputeBackends(deployed); + const effective = resolveComputeBackend(override ?? backends?.[0]); + if (backends && !backends.includes(effective)) { + throw new Error(`Repository compute_type '${effective}' is not deployed; this stack deploys only '${backends.join(', ')}'. Update the repository configuration before submitting tasks.`); + } + return effective; +} diff --git a/cdk/src/handlers/shared/orchestrator.ts b/cdk/src/handlers/shared/orchestrator.ts index 81685e6d7..1dc290a1e 100644 --- a/cdk/src/handlers/shared/orchestrator.ts +++ b/cdk/src/handlers/shared/orchestrator.ts @@ -20,6 +20,7 @@ import { S3Client } from '@aws-sdk/client-s3'; import { GetCommand, PutCommand, UpdateCommand } from '@aws-sdk/lib-dynamodb'; import { ulid } from 'ulid'; +import { resolveRepositoryBackend } from './compute-backend'; import type { SessionHandle, SessionStatus } from './compute-strategy'; import { AttachmentBudgetExceededError, AttachmentConfigurationError, AttachmentResolutionError, hydrateContext, resolveGitHubToken } from './context-hydration'; import { logger, type Logger } from './logger'; @@ -582,19 +583,10 @@ export async function loadBlueprintConfig(task: TaskRecord): Promise= 33 chars; UUID v4 is 36 chars. const sessionId = randomUUID(); const runtimeArn = input.blueprintConfig.runtime_arn; + if (!runtimeArn) throw new Error('AgentCore compute requires a configured runtime ARN'); // `runtimeUserId` triggers AgentCore Identity's workload-access-token // injection: when set, AgentCore exchanges the caller's identity for diff --git a/cdk/src/main.ts b/cdk/src/main.ts index 7637e5fea..f3ca9ba00 100644 --- a/cdk/src/main.ts +++ b/cdk/src/main.ts @@ -17,8 +17,9 @@ * SOFTWARE. */ -import { App, AppProps, AspectPriority, Aspects, Tags } from 'aws-cdk-lib'; +import { App, AppProps, AspectPriority, Aspects, STACK_RESOURCE_LIMIT_CONTEXT, Tags } from 'aws-cdk-lib'; import { AwsSolutionsChecks } from 'cdk-nag'; +import { BlueprintDefinition, blueprintEgressDomains, resolveBlueprintDefinitions } from './blueprints/definitions'; import { applyAgentCoreAzDiagnostics, DescribeAzsFn, @@ -26,7 +27,10 @@ import { resolveAgentCoreAzs, } from './constructs/agentcore-azs'; import { buildAppId, SolutionUaAspect } from './constructs/solution-ua-aspect'; +import { resolveComputeBackends } from './handlers/shared/compute-backend'; import { AgentStack } from './stacks/agent'; +import { NetworkStack, resolveNetworkTopology } from './stacks/network'; +import { DEFAULT_BUDGETS } from './synthesis/budgets'; // for development, use account/region from cdk cli const devEnv = { @@ -46,6 +50,8 @@ export interface BuildAppOptions { readonly describeAzs?: DescribeAzsFn; /** Injectable caller-account lookup so tests need no AWS access. */ readonly resolveCallerAccount?: ResolveCallerAccountFn; + /** Repository configuration shared by provisioning and DNS policy. */ + readonly blueprints?: readonly BlueprintDefinition[]; } /** @@ -62,21 +68,54 @@ export interface BuildAppOptions { */ export async function buildApp(options: BuildAppOptions = {}): Promise { const app = new App(options.appProps); + // Never silently downgrade an explicitly configured migration prototype to + // the original Blueprint provider or guardrail implementation. + for (const key of ['blueprintProvisioning', 'guardrailVersionMigration']) { + if (app.node.tryGetContext(key) !== undefined) { + throw new Error( + `Context '${key}' belongs to the deferred migration prototype and is no longer supported. ` + + 'A stack deployed with that prototype needs a separate recovery plan; do not drop this setting and deploy over it.', + ); + } + } + // Apply to every parent and nested template, including newly extracted stacks. + app.node.setContext('@aws-cdk/core:suppressTemplateIndentation', true); + // Enforce the same ceiling on actual deploy inputs, including operator overrides + // outside the census. CDK applies this context to parent and nested stacks. + const configuredLimit: unknown = app.node.tryGetContext(STACK_RESOURCE_LIMIT_CONTEXT); + const resourceLimit = configuredLimit === undefined ? DEFAULT_BUDGETS.resources + : typeof configuredLimit === 'string' ? Number(configuredLimit) : configuredLimit; + if (typeof resourceLimit !== 'number' || !Number.isInteger(resourceLimit) + || resourceLimit < 1 || resourceLimit > DEFAULT_BUDGETS.resources) { + throw new Error( + `Context '${STACK_RESOURCE_LIMIT_CONTEXT}' must be an integer from 1 to ${DEFAULT_BUDGETS.resources}. ` + + 'The ABCA resource budget can be tightened but not raised. Use networkTopology=split for more ' + + 'application headroom; existing deployments require an explicit network migration.', + ); + } + app.node.setContext(STACK_RESOURCE_LIMIT_CONTEXT, resourceLimit); Aspects.of(app).add(new AwsSolutionsChecks()); const stackName = app.node.tryGetContext('stackName') ?? 'backgroundagent-dev'; + const networkTopology = resolveNetworkTopology(app.node.tryGetContext('networkTopology')); + const blueprints = options.blueprints ?? resolveBlueprintDefinitions(app.node); const env = { account: options.account ?? devEnv.account, region: options.region ?? devEnv.region, }; + // Preserve existing VPC placement across backend selection changes. The shared + // network continues to use the established AgentCore-compatible AZ policy. // Auto-pin the VPC to AgentCore-supported AZs (or honor the validated // `agentcore:availabilityZones` override). `zones` undefined => CDK default // selection; `diagnostics` are attached to the stack below, because CDK only // collects annotations that hang off a stack's tree — App-node metadata would // be silently dropped, which is how a failed lookup used to pass unnoticed. + // Tag values allow '+', not ','; the tag records every deployed backend. + const computeType = resolveComputeBackends( + app.node.tryGetContext('compute_types'), app.node.tryGetContext('compute_type')).join('+'); const azResolution = await resolveAgentCoreAzs({ node: app.node, account: env.account, @@ -85,33 +124,33 @@ export async function buildApp(options: BuildAppOptions = {}): Promise { resolveCallerAccount: options.resolveCallerAccount, }); + const network = networkTopology === 'split' ? new NetworkStack(app, `${stackName}-network`, { + env, + applicationStackName: stackName, + agentCoreAvailabilityZones: azResolution.zones, + additionalAllowedDomains: blueprintEgressDomains(blueprints), + description: 'ABCA network infrastructure (uksb-wt64nei4u6)', + }) : undefined; + const stack = new AgentStack( app, stackName, { env, agentCoreAvailabilityZones: azResolution.zones, + network, + blueprints, description: 'ABCA Development Stack (uksb-wt64nei4u6)', - // Emit compact JSON for a CloudFormation 1 MB template-body ceiling. - suppressTemplateIndentation: true, }, ); - applyAgentCoreAzDiagnostics(stack, azResolution); + applyAgentCoreAzDiagnostics(network ?? stack, azResolution); // Outbound SDK solution attribution (#319): set AWS_SDK_UA_APP_ID on every // Lambda so the SDK emits `app/uksb-wt64nei4u6#{stackName}` natively. One // Aspect covers current and future functions structurally. Override via // `-c sdkUaAppId=...`; `-c sdkUaAppId=''` opts out (no app/ segment anywhere). const sdkUaAppIdOverride = app.node.tryGetContext('sdkUaAppId') as string | undefined; - // MUTATING priority so the env var is set before cdk-nag (priority 500) - // inspects the synthesized functions — matches the agent stack's aspects. - Aspects.of(stack).add(new SolutionUaAspect(buildAppId(stackName, sdkUaAppIdOverride)), { - priority: AspectPriority.MUTATING, - }); - - const computeType = app.node.tryGetContext('compute_type') ?? 'agentcore'; - // Route53 Resolver resources where tag changes trigger replacement cascades. // Config: treats ANY property change (including tags) as requiring replacement. // Association: depends on Config's physical ID; if Config is replaced, the @@ -121,17 +160,6 @@ export async function buildApp(options: BuildAppOptions = {}): Promise { 'AWS::Route53Resolver::ResolverQueryLoggingConfigAssociation', ]; - // TODO(#645): with three backends this single-valued tag is no longer an honest - // statement of what a stack runs — a `--context compute_type=lambda-microvm` - // deploy still provisions the AgentCore runtime, so every resource gets tagged - // `compute_type=lambda-microvm` including the AgentCore ones. ADR-021 - // sub-decision 4 flags revisiting the semantics (e.g. a `compute_types` list). - // Deliberately NOT changed here: retagging every resource in the stack is a - // replacement-risk change of its own, and MicroVM spend is already attributable - // through the per-resource `abca:compute-backend` tags the - // LambdaMicrovmCompute construct applies. - Tags.of(stack).add('compute_type', computeType, { excludeResourceTypes }); - const githubTagKeys = [ 'sha', 'ref', @@ -148,9 +176,16 @@ export async function buildApp(options: BuildAppOptions = {}): Promise { 'clean', ] as const; - for (const key of githubTagKeys) { - const value = app.node.tryGetContext(`github:${key}`); - Tags.of(stack).add(`github:${key}`, value || 'none', { excludeResourceTypes }); + for (const deploymentStack of network ? [network, stack] : [stack]) { + // Keep the application deployment identity on both stacks' SDK calls. + Aspects.of(deploymentStack).add(new SolutionUaAspect(buildAppId(stackName, sdkUaAppIdOverride)), { + priority: AspectPriority.MUTATING, + }); + Tags.of(deploymentStack).add('compute_type', computeType, { excludeResourceTypes }); + for (const key of githubTagKeys) { + const value = app.node.tryGetContext(`github:${key}`); + Tags.of(deploymentStack).add(`github:${key}`, value || 'none', { excludeResourceTypes }); + } } return app; diff --git a/cdk/src/stacks/agent.ts b/cdk/src/stacks/agent.ts index 308431789..b640a76f8 100644 --- a/cdk/src/stacks/agent.ts +++ b/cdk/src/stacks/agent.ts @@ -29,10 +29,11 @@ import * as secretsmanager from 'aws-cdk-lib/aws-secretsmanager'; import * as cr from 'aws-cdk-lib/custom-resources'; import { NagSuppressions } from 'cdk-nag'; import { Construct, IConstruct } from 'constructs'; +import { BlueprintDefinition, blueprintEgressDomains, resolveBlueprintDefinitions } from '../blueprints/definitions'; import { AdmissionQueuePickup } from '../constructs/admission-queue-pickup'; import { AgentMemory } from '../constructs/agent-memory'; import { AgentSessionRole } from '../constructs/agent-session-role'; -import { AgentVpc } from '../constructs/agent-vpc'; +import { AgentNetwork, AgentVpc } from '../constructs/agent-vpc'; import { ApiKeyTable } from '../constructs/api-key-table'; import { ApprovalMetricsPublisherConsumer } from '../constructs/approval-metrics-publisher-consumer'; import { AttachmentsBucket } from '../constructs/attachments-bucket'; @@ -85,6 +86,7 @@ import { ToolGateway } from '../constructs/tool-gateway'; import { TraceArtifactsBucket } from '../constructs/trace-artifacts-bucket'; import { UserConcurrencyTable } from '../constructs/user-concurrency-table'; import { WebhookTable } from '../constructs/webhook-table'; +import { resolveComputeBackends } from '../handlers/shared/compute-backend'; /** Max length of the Bedrock Guardrail name (CloudFormation constraint). */ const GUARDRAIL_NAME_MAX_LENGTH = 50; @@ -151,6 +153,10 @@ export interface AgentStackProps extends StackProps { * values under one name in one class is a trap for `Stack.of(x)` callers. */ readonly agentCoreAvailabilityZones?: string[]; + /** Network owned by a separate stack. Omit to preserve the inline topology. */ + readonly network?: AgentNetwork; + /** Shared plain configuration, resolved before network/application construction. */ + readonly blueprints?: readonly BlueprintDefinition[]; } export class AgentStack extends Stack { @@ -180,9 +186,11 @@ export class AgentStack extends Stack { // changes. Pattern lifted from ``merge/akw-integration``. const repoRoot = path.join(__dirname, '..', '..', '..'); - const artifact = agentcore.AgentRuntimeArtifact.fromAsset(repoRoot, { - file: 'agent/Dockerfile', - }); + const computeTypes = resolveComputeBackends( + this.node.tryGetContext('compute_types'), this.node.tryGetContext('compute_type')); + const agentCoreEnabled = computeTypes.includes('agentcore'); + const ecsEnabled = computeTypes.includes('ecs'); + const lambdaMicrovmEnabled = computeTypes.includes('lambda-microvm'); // Task state persistence const taskTable = new TaskTable(this, 'TaskTable'); @@ -273,59 +281,11 @@ export class AgentStack extends Stack { ]); // --- Repository onboarding --- - const blueprintRepo = process.env.BLUEPRINT_REPO ?? this.node.tryGetContext('blueprintRepo') ?? 'awslabs/agent-plugins'; - const agentPluginsBlueprint = new Blueprint(this, 'AgentPluginsBlueprint', { - repo: blueprintRepo, - repoTable: repoTable.table, - }); - - const blueprints = [agentPluginsBlueprint]; - - // Optional per-repo blueprint pinning registry assets (#246), opt-in via - // context/env so it does not hardcode a specific fork for other contributors. - // Set ``forkBlueprintRepo`` (e.g. ``--context forkBlueprintRepo=owner/repo``) - // to onboard a repo with the AWS Knowledge MCP asset pinned. - const forkBlueprintRepo = process.env.FORK_BLUEPRINT_REPO ?? this.node.tryGetContext('forkBlueprintRepo'); - if (forkBlueprintRepo) { - blueprints.push(new Blueprint(this, 'ForkBlueprint', { - repo: forkBlueprintRepo, - repoTable: repoTable.table, - assets: { - mcpServers: ['registry://mcp_server/acme/aws-knowledge@^1.0.0'], - cedarPolicyModules: ['registry://cedar_policy_module/acme/guard@^1.0.0'], - skills: ['registry://skill/acme/readme-helper@^1.0.0'], - }, - })); + const blueprintDefinitions = props.blueprints ?? resolveBlueprintDefinitions(this.node); + for (const { id: blueprintId, ...definition } of blueprintDefinitions) { + new Blueprint(this, blueprintId, { ...definition, repoTable: repoTable.table }); } - // The AwsCustomResource singleton Lambda used by Blueprint constructs - NagSuppressions.addResourceSuppressionsByPath(this, [ - `${this.stackName}/AWS679f53fac002430cb0da5b7982bd2287/ServiceRole/Resource`, - `${this.stackName}/AWS679f53fac002430cb0da5b7982bd2287/Resource`, - ], [ - { - id: 'AwsSolutions-IAM4', - reason: 'AwsCustomResource singleton Lambda uses AWS managed AWSLambdaBasicExecutionRole — required by CDK custom-resources framework', - }, - { - id: 'AwsSolutions-L1', - reason: 'AwsCustomResource singleton Lambda runtime is managed by the CDK custom-resources framework', - }, - ]); - - // Log groups (created before runtime so we can reference the name in env vars) - const applicationLogGroup = new logs.LogGroup(this, 'RuntimeApplicationLogGroup', { - logGroupName: `/aws/vendedlogs/bedrock-agentcore/runtime/APPLICATION_LOGS/${this.stackName}`, - retention: logs.RetentionDays.THREE_MONTHS, - removalPolicy: RemovalPolicy.DESTROY, - }); - - const usageLogGroup = new logs.LogGroup(this, 'RuntimeUsageLogGroup', { - logGroupName: `/aws/vendedlogs/bedrock-agentcore/runtime/USAGE_LOGS/${this.stackName}`, - retention: logs.RetentionDays.THREE_MONTHS, - removalPolicy: RemovalPolicy.DESTROY, - }); - // GitHub token stored in Secrets Manager — agent fetches at startup via ARN const githubTokenSecret = new secretsmanager.Secret(this, 'GitHubTokenSecret', { description: 'GitHub personal access token for the background agent', @@ -339,16 +299,6 @@ export class AgentStack extends Stack { }, ]); - // --- Compute-backend deploy gate (read early) --- - // Which optional compute substrate this deploy provisions, from the - // ``compute_type`` deploy context (default 'agentcore' — the AgentCore - // runtime is always present, the other backends are additive). Read HERE, - // well above the constructs it gates, because TaskApi is instantiated - // before them and needs to know whether to wire the cancel Lambda's - // MicroVM termination grant (ADR-021 sub-decision 4). - const computeType = this.node.tryGetContext('compute_type') ?? 'agentcore'; - const lambdaMicrovmEnabled = computeType === 'lambda-microvm'; - // --- Tool-federation Gateway deploy gate (ADR-019 P1) --- // Whether to provision the AgentCore Gateway that federates the agent's MCP // tools (P1: one read-only Lambda target, ``abca_repo_config``). OFF by @@ -371,20 +321,6 @@ export class AgentStack extends Stack { // second chance to disagree. const linearVaultWorkload = linearVaultWorkloadName(this); - // Fail here, naming both flags, rather than 500 resources later. The two features - // together synthesize 505 resources against CloudFormation's hard 500 limit (MicroVM - // alone 496, the vault alone 488), so the combination is not deployable today. Left to - // the resource counter, the operator gets a per-type census and no hint that two - // context flags are the cause. - if (linearIdentityVaultEnabled && computeType === 'lambda-microvm') { - throw new Error( - 'enableLinearIdentityVault cannot be combined with compute_type=lambda-microvm: the two ' - + 'together exceed CloudFormation\'s 500-resource limit for this stack (505). Deploy the ' - + 'vault on the agentcore or ecs substrate, or omit enableLinearIdentityVault. See ' - + 'docs/design/ADR-016 and the LINEAR_SETUP_GUIDE.', - ); - } - // The operator-supplied MicroVM image inputs, resolved HERE (pure context // reads, no construct dependency) rather than at the construct's call site // below, because TaskApi — created well before the MicroVM construct — needs @@ -430,19 +366,20 @@ export class AgentStack extends Stack { // override, else auto-selected from the account's AgentCore-supported zones // when synth has a concrete account/region. Left undefined otherwise, so the // construct keeps CDK's default AZ selection. See constructs/agentcore-azs.ts. - const agentVpc = new AgentVpc(this, 'AgentVpc', { + const agentVpc = props.network ?? new AgentVpc(this, 'AgentVpc', { ...(props.agentCoreAvailabilityZones?.length ? { availabilityZones: props.agentCoreAvailabilityZones } : {}), }); - // DNS Firewall — domain-level egress filtering (observation mode for initial deployment) - const additionalDomains = [...new Set(blueprints.flatMap(b => b.egressAllowlist))]; - new DnsFirewall(this, 'DnsFirewall', { - vpc: agentVpc.vpc, - additionalAllowedDomains: additionalDomains, - observationMode: true, - }); + if (!props.network) { + // The default topology retains the original ownership and construct paths. + new DnsFirewall(this, 'DnsFirewall', { + vpc: agentVpc.vpc, + additionalAllowedDomains: blueprintEgressDomains(blueprintDefinitions), + observationMode: true, + }); + } // --- AgentCore Memory (cross-task learning) --- const agentMemory = new AgentMemory(this, 'AgentMemory'); @@ -515,6 +452,13 @@ export class AgentStack extends Stack { }, }); + let ecsClusterArnHolder: string | undefined; + const lazyEcsClusterArn = Lazy.string({ + produce: () => { + if (!ecsClusterArnHolder) throw new Error('ECS cluster ARN was accessed before compute was created'); + return ecsClusterArnHolder; + }, + }); // --- Task API (REST API + Cognito + Lambda handlers) --- const taskApi = new TaskApi(this, 'TaskApi', { taskTable: taskTable.table, @@ -528,7 +472,8 @@ export class AgentStack extends Stack { orchestratorFunctionArn: lazyOrchestratorArn, guardrailId: inputGuardrail.guardrailId, guardrailVersion: inputGuardrail.guardrailVersion, - agentCoreStopSessionRuntimeArn: lazyRuntimeArn, + ...(agentCoreEnabled && { agentCoreStopSessionRuntimeArn: lazyRuntimeArn }), + ...(ecsEnabled && { ecsClusterArn: lazyEcsClusterArn }), traceArtifactsBucket: traceArtifactsBucket.bucket, attachmentsBucket: attachmentsBucket.bucket, userConcurrencyTable: userConcurrencyTable.table, @@ -581,191 +526,211 @@ export class AgentStack extends Stack { // geography's profiles while telling the agent to call another's. const bedrockGeoRegion = resolveBedrockGeoRegion(this.node); - const runtimeEnvironmentVariables = { - GITHUB_TOKEN_SECRET_ARN: githubTokenSecret.secretArn, - AWS_REGION: process.env.AWS_REGION ?? 'us-east-1', - CLAUDE_CODE_USE_BEDROCK: '1', - ANTHROPIC_LOG: 'debug', - // Cross-region inference-profile ids (geo prefix), NOT bare foundation-model - // ids: Claude 4.x can't be invoked on-demand by bare id (400 "on-demand - // throughput isn't supported"). Both are derived from `bedrockGeoRegion` rather - // than hardcoded, so neither can silently split from the granted profiles on a - // non-default deploy, and the model ids come from the same constants the grant - // list interpolates. - // - // The MAIN model is set here deliberately, and was previously absent: only the - // auxiliary var was injected, so the main model fell through to a literal in - // agent/src/config.py that a geography change does not touch. A deploy with a - // different `bedrockGeoRegion` therefore granted one geography's profiles while - // the agent asked for another's, and every task with no per-repo override failed - // at turn 0 with AccessDenied. - // - // The lambda-microvm `platform_config` block below derives the same two values - // from the same geography. runner.py re-sets both at spawn time; a per-repo - // `model_id` still overrides. - ANTHROPIC_MODEL: inferenceProfileId(bedrockGeoRegion, PLATFORM_DEFAULT_MODEL_ID), - ANTHROPIC_DEFAULT_HAIKU_MODEL: inferenceProfileId(bedrockGeoRegion, PLATFORM_DEFAULT_AUX_MODEL_ID), - TASK_TABLE_NAME: taskTable.table.tableName, - TASK_EVENTS_TABLE_NAME: taskEventsTable.table.tableName, - NUDGES_TABLE_NAME: taskNudgesTable.table.tableName, - // Cedar HITL approval gates (§6.5). Agent's task_state primitives - // use this to write PENDING rows + transition tasks to - // AWAITING_APPROVAL; absent → hook fails closed with - // ``approval_write_failed`` (the `ApprovalTablesUnavailable` path). - TASK_APPROVALS_TABLE_NAME: taskApprovalsTable.table.tableName, - // Hint for the hook's remaining-maxLifetime calculation (§6.5 - // pseudocode line 793). Kept in sync with the AgentCore - // lifecycle configuration below so drift is visible. 8 hours. - AGENTCORE_MAX_LIFETIME_S: '28800', - USER_CONCURRENCY_TABLE_NAME: userConcurrencyTable.table.tableName, - // Per-task SessionRole: the agent assumes this with session tags - // {user_id, repo, task_id} and uses the scoped creds for tenant-data - // (DDB/S3) access. Resolved lazily — the role lists runtime.role as an - // assuming principal, so it is created after the runtime. - AGENT_SESSION_ROLE_ARN: lazySessionRoleArn, - // --trace artifact store (§10.1). The agent writes the JSONL - // trajectory to ``traces//.jsonl.gz`` on - // terminal state when the submit payload enabled ``trace``. - TRACE_ARTIFACTS_BUCKET_NAME: traceArtifactsBucket.bucket.bucketName, - // Repo-less deliverable artifacts: a deliver_artifact step - // uploads its product to ``artifacts//`` in the same bucket. - ARTIFACTS_BUCKET_NAME: traceArtifactsBucket.bucket.bucketName, - LOG_GROUP_NAME: applicationLogGroup.logGroupName, - MEMORY_ID: agentMemory.memory.memoryId, - MAX_TURNS: '100', - // Session storage: the S3-backed FUSE mount at /mnt/workspace does NOT - // support flock(). Only caches whose tools never call flock() go there. - // Everything else stays on local ephemeral disk. - // - // Local disk (tools use flock): - // AGENT_WORKSPACE — omitted, defaults to /workspace - // MISE_DATA_DIR — mise's pipx backend sets UV_TOOL_DIR inside installs/, - // and uv flocks that directory → must be local. - MISE_DATA_DIR: '/tmp/mise-data', - UV_CACHE_DIR: '/tmp/uv-cache', - // Persistent mount (no flock): - CLAUDE_CONFIG_DIR: '/mnt/workspace/.claude-config', - npm_config_cache: '/mnt/workspace/.npm-cache', - // ENABLE_CLI_TELEMETRY: '1', - // Outbound SDK solution attribution (#319): botocore reads - // AWS_SDK_UA_APP_ID natively → `app/uksb-wt64nei4u6#{stack}`. The - // Lambda-only Aspect can't reach this runtime, so set it explicitly. - ...(sdkUaAppId ? { AWS_SDK_UA_APP_ID: sdkUaAppId } : {}), - // ADR-019 P1: the federated-tool Gateway URL (context-gated). Present only - // when ``--context enableToolGateway=true``; the agent's in-process SigV4 - // MCP bridge (gateway_tools.build_gateway_server) reads it to register the - // ``abca_gateway`` SDK server. Absent → no gateway tool, unchanged. - ...(toolGateway ? { ABCA_TOOL_GATEWAY_URL: toolGateway.gatewayUrl } : {}), - // RFC #249 Phase 1 (context-gated `enableLinearIdentityVault`): tell the - // agent's Linear token resolver to mint via the AgentCore Token Vault when - // a task carries a provider name. Absent → the agent stays on the - // Secrets-Manager path. The workload name is the stack-derived value computed - // above and passed INTO the construct — the construct has no default of its own, - // and a rename orphans every consent already given. - ...(linearIdentityVaultEnabled - ? { LINEAR_VAULT_ENABLED: 'true', LINEAR_WORKLOAD_IDENTITY_NAME: linearVaultWorkload } - : {}), - }; + // Keep the AgentCore log groups owned across explicit backend switches, + // with the existing destroy policy so rollback and same-name reinstall work. + const applicationLogGroup = new logs.LogGroup(this, 'RuntimeApplicationLogGroup', { + logGroupName: `/aws/vendedlogs/bedrock-agentcore/runtime/APPLICATION_LOGS/${this.stackName}`, + retention: logs.RetentionDays.THREE_MONTHS, + removalPolicy: RemovalPolicy.DESTROY, + }); - const runtimeNetworkConfig = agentcore.RuntimeNetworkConfiguration.usingVpc(this, { - vpc: agentVpc.vpc, - vpcSubnets: { subnetType: ec2.SubnetType.PRIVATE_WITH_EGRESS }, - securityGroups: [agentVpc.runtimeSecurityGroup], + const usageLogGroup = new logs.LogGroup(this, 'RuntimeUsageLogGroup', { + logGroupName: `/aws/vendedlogs/bedrock-agentcore/runtime/USAGE_LOGS/${this.stackName}`, + retention: logs.RetentionDays.THREE_MONTHS, + removalPolicy: RemovalPolicy.DESTROY, }); - // LifecycleConfiguration — both timers set to the AgentCore 8h maximum so - // long-running tasks (approval waits, heavy builds) are not evicted. - const lifecycleConfiguration: agentcore.LifecycleConfiguration = { - idleRuntimeSessionTimeout: Duration.hours(RUNTIME_SESSION_TIMEOUT_HOURS), - maxLifetime: Duration.hours(RUNTIME_SESSION_TIMEOUT_HOURS), - }; + let runtime: agentcore.Runtime | undefined; + let agentLogGroup: logs.ILogGroup | undefined; + if (agentCoreEnabled) { + const artifact = agentcore.AgentRuntimeArtifact.fromAsset(repoRoot, { file: 'agent/Dockerfile' }); + const runtimeEnvironmentVariables = { + GITHUB_TOKEN_SECRET_ARN: githubTokenSecret.secretArn, + AWS_REGION: process.env.AWS_REGION ?? 'us-east-1', + CLAUDE_CODE_USE_BEDROCK: '1', + ANTHROPIC_LOG: 'debug', + // Cross-region inference-profile ids (geo prefix), NOT bare foundation-model + // ids: Claude 4.x can't be invoked on-demand by bare id (400 "on-demand + // throughput isn't supported"). Both are derived from `bedrockGeoRegion` rather + // than hardcoded, so neither can silently split from the granted profiles on a + // non-default deploy, and the model ids come from the same constants the grant + // list interpolates. + // + // The MAIN model is set here deliberately, and was previously absent: only the + // auxiliary var was injected, so the main model fell through to a literal in + // agent/src/config.py that a geography change does not touch. A deploy with a + // different `bedrockGeoRegion` therefore granted one geography's profiles while + // the agent asked for another's, and every task with no per-repo override failed + // at turn 0 with AccessDenied. + // + // The lambda-microvm `platform_config` block below derives the same two values + // from the same geography. runner.py re-sets both at spawn time; a per-repo + // `model_id` still overrides. + ANTHROPIC_MODEL: inferenceProfileId(bedrockGeoRegion, PLATFORM_DEFAULT_MODEL_ID), + ANTHROPIC_DEFAULT_HAIKU_MODEL: inferenceProfileId(bedrockGeoRegion, PLATFORM_DEFAULT_AUX_MODEL_ID), + TASK_TABLE_NAME: taskTable.table.tableName, + TASK_EVENTS_TABLE_NAME: taskEventsTable.table.tableName, + NUDGES_TABLE_NAME: taskNudgesTable.table.tableName, + // Cedar HITL approval gates (§6.5). Agent's task_state primitives + // use this to write PENDING rows + transition tasks to + // AWAITING_APPROVAL; absent → hook fails closed with + // ``approval_write_failed`` (the `ApprovalTablesUnavailable` path). + TASK_APPROVALS_TABLE_NAME: taskApprovalsTable.table.tableName, + // Hint for the hook's remaining-maxLifetime calculation (§6.5 + // pseudocode line 793). Kept in sync with the AgentCore + // lifecycle configuration below so drift is visible. 8 hours. + AGENTCORE_MAX_LIFETIME_S: '28800', + USER_CONCURRENCY_TABLE_NAME: userConcurrencyTable.table.tableName, + // Per-task SessionRole: the agent assumes this with session tags + // {user_id, repo, task_id} and uses the scoped creds for tenant-data + // (DDB/S3) access. Resolved lazily — the role lists runtime.role as an + // assuming principal, so it is created after the runtime. + AGENT_SESSION_ROLE_ARN: lazySessionRoleArn, + // --trace artifact store (§10.1). The agent writes the JSONL + // trajectory to ``traces//.jsonl.gz`` on + // terminal state when the submit payload enabled ``trace``. + TRACE_ARTIFACTS_BUCKET_NAME: traceArtifactsBucket.bucket.bucketName, + // Repo-less deliverable artifacts: a deliver_artifact step + // uploads its product to ``artifacts//`` in the same bucket. + ARTIFACTS_BUCKET_NAME: traceArtifactsBucket.bucket.bucketName, + LOG_GROUP_NAME: applicationLogGroup.logGroupName, + MEMORY_ID: agentMemory.memory.memoryId, + MAX_TURNS: '100', + // Session storage: the S3-backed FUSE mount at /mnt/workspace does NOT + // support flock(). Only caches whose tools never call flock() go there. + // Everything else stays on local ephemeral disk. + // + // Local disk (tools use flock): + // AGENT_WORKSPACE — omitted, defaults to /workspace + // MISE_DATA_DIR — mise's pipx backend sets UV_TOOL_DIR inside installs/, + // and uv flocks that directory → must be local. + MISE_DATA_DIR: '/tmp/mise-data', + UV_CACHE_DIR: '/tmp/uv-cache', + // Persistent mount (no flock): + CLAUDE_CONFIG_DIR: '/mnt/workspace/.claude-config', + npm_config_cache: '/mnt/workspace/.npm-cache', + // ENABLE_CLI_TELEMETRY: '1', + // Outbound SDK solution attribution (#319): botocore reads + // AWS_SDK_UA_APP_ID natively → `app/uksb-wt64nei4u6#{stack}`. The + // Lambda-only Aspect can't reach this runtime, so set it explicitly. + ...(sdkUaAppId ? { AWS_SDK_UA_APP_ID: sdkUaAppId } : {}), + // ADR-019 P1: the federated-tool Gateway URL (context-gated). Present only + // when ``--context enableToolGateway=true``; the agent's in-process SigV4 + // MCP bridge (gateway_tools.build_gateway_server) reads it to register the + // ``abca_gateway`` SDK server. Absent → no gateway tool, unchanged. + ...(toolGateway ? { ABCA_TOOL_GATEWAY_URL: toolGateway.gatewayUrl } : {}), + // RFC #249 Phase 1 (context-gated `enableLinearIdentityVault`): tell the + // agent's Linear token resolver to mint via the AgentCore Token Vault when + // a task carries a provider name. Absent → the agent stays on the + // Secrets-Manager path. The workload name is the stack-derived value computed + // above and passed INTO the construct — the construct has no default of its own, + // and a rename orphans every consent already given. + ...(linearIdentityVaultEnabled + ? { LINEAR_VAULT_ENABLED: 'true', LINEAR_WORKLOAD_IDENTITY_NAME: linearVaultWorkload } + : {}), + }; - // Construct id 'Runtime' is load-bearing — renaming it forces CFN to - // CREATE the new resource before DELETING the old one, violating - // AgentCore's account-level runtimeName uniqueness and triggering an - // UPDATE_ROLLBACK. - const runtime = new agentcore.Runtime(this, 'Runtime', { - agentRuntimeArtifact: artifact, - networkConfiguration: runtimeNetworkConfig, - environmentVariables: runtimeEnvironmentVariables, - lifecycleConfiguration: lifecycleConfiguration, - loggingConfigs: [ - { - logType: agentcore.LogType.APPLICATION_LOGS, - destination: agentcore.LoggingDestination.cloudWatchLogs(applicationLogGroup), - }, - { - logType: agentcore.LogType.USAGE_LOGS, - destination: agentcore.LoggingDestination.cloudWatchLogs(usageLogGroup), - }, - ], - }); + const runtimeNetworkConfig = agentcore.RuntimeNetworkConfiguration.usingVpc(this, { + vpc: agentVpc.vpc, + vpcSubnets: { subnetType: ec2.SubnetType.PRIVATE_WITH_EGRESS }, + securityGroups: [agentVpc.runtimeSecurityGroup], + }); - runtimeArnHolder = runtime.agentRuntimeArn; + // LifecycleConfiguration — both timers set to the AgentCore 8h maximum so + // long-running tasks (approval waits, heavy builds) are not evicted. + const lifecycleConfiguration: agentcore.LifecycleConfiguration = { + idleRuntimeSessionTimeout: Duration.hours(RUNTIME_SESSION_TIMEOUT_HOURS), + maxLifetime: Duration.hours(RUNTIME_SESSION_TIMEOUT_HOURS), + }; + + // Construct id 'Runtime' is load-bearing — renaming it forces CFN to + // CREATE the new resource before DELETING the old one, violating + // AgentCore's account-level runtimeName uniqueness and triggering an + // UPDATE_ROLLBACK. + runtime = new agentcore.Runtime(this, 'Runtime', { + agentRuntimeArtifact: artifact, + networkConfiguration: runtimeNetworkConfig, + environmentVariables: runtimeEnvironmentVariables, + lifecycleConfiguration: lifecycleConfiguration, + loggingConfigs: [ + { + logType: agentcore.LogType.APPLICATION_LOGS, + destination: agentcore.LoggingDestination.cloudWatchLogs(applicationLogGroup), + }, + { + logType: agentcore.LogType.USAGE_LOGS, + destination: agentcore.LoggingDestination.cloudWatchLogs(usageLogGroup), + }, + ], + }); - // --- AgentCore log delivery: the library owns these logical ids, not us --- - // - // The Runtime above creates a DeliverySource / DeliveryDestination / Delivery - // trio per loggingConfig, naming them from its own construct path. We - // deliberately leave those names alone, and that is load-bearing. - // - // Renaming any of them is fatal on an existing stack. A DeliverySource is - // unique per (resource ARN, log type) account-wide and the runtime ARN does not - // change, so CloudFormation's create-before-delete produces a second source for - // the same runtime, CloudWatch Logs rejects it as already existing, and the - // whole update rolls back. Note what that implies: no choice of name avoids the - // collision, because the conflict is on the ARN they point at rather than on - // their own names. Only leaving a logical id untouched updates in place. - // - // An earlier version overrode them from a table of ids recorded off a live - // stack, keyed by stack NAME. Stack name is not a proxy for deployed state: - // two accounts running a stack of the same name had diverged, so the table was - // correct for one and caused the rename on the other. No set of literals can - // describe every account, so we hold none and let the library generate them — - // deterministically, identically in every account, with nothing to keep in sync. - // - // The one cost: a stack deployed before a library-side rename, or held on older - // ids by that table, converges once and needs a one-time operator step first, - // since the old resources must be gone before the new ones are created. That - // step, and how to tell whether a stack needs it, are in - // docs/design/OBSERVABILITY.md ("AgentCore log delivery"). - - // --- Session storage (preview) --- - // The L2 construct does not yet expose filesystemConfigurations; use the - // CFN escape hatch. /mnt/workspace mount backs the persistent cache - // shared across tasks in the same repo. - const cfnRuntime = runtime.node.defaultChild as CfnResource; - cfnRuntime.addPropertyOverride('FilesystemConfigurations', [ - { - SessionStorage: { - MountPath: '/mnt/workspace', + runtimeArnHolder = runtime.agentRuntimeArn; + agentLogGroup = applicationLogGroup; + + // --- AgentCore log delivery: the library owns these logical ids, not us --- + // + // The Runtime above creates a DeliverySource / DeliveryDestination / Delivery + // trio per loggingConfig, naming them from its own construct path. We + // deliberately leave those names alone, and that is load-bearing. + // + // Renaming any of them is fatal on an existing stack. A DeliverySource is + // unique per (resource ARN, log type) account-wide and the runtime ARN does not + // change, so CloudFormation's create-before-delete produces a second source for + // the same runtime, CloudWatch Logs rejects it as already existing, and the + // whole update rolls back. Note what that implies: no choice of name avoids the + // collision, because the conflict is on the ARN they point at rather than on + // their own names. Only leaving a logical id untouched updates in place. + // + // An earlier version overrode them from a table of ids recorded off a live + // stack, keyed by stack NAME. Stack name is not a proxy for deployed state: + // two accounts running a stack of the same name had diverged, so the table was + // correct for one and caused the rename on the other. No set of literals can + // describe every account, so we hold none and let the library generate them — + // deterministically, identically in every account, with nothing to keep in sync. + // + // The one cost: a stack deployed before a library-side rename, or held on older + // ids by that table, converges once and needs a one-time operator step first, + // since the old resources must be gone before the new ones are created. That + // step, and how to tell whether a stack needs it, are in + // docs/design/OBSERVABILITY.md ("AgentCore log delivery"). + + // --- Session storage (preview) --- + // The L2 construct does not yet expose filesystemConfigurations; use the + // CFN escape hatch. /mnt/workspace mount backs the persistent cache + // shared across tasks in the same repo. + const cfnRuntime = runtime.node.defaultChild as CfnResource; + cfnRuntime.addPropertyOverride('FilesystemConfigurations', [ + { + SessionStorage: { + MountPath: '/mnt/workspace', + }, }, - }, - ]); + ]); - // --- IAM grants --- - // Per-session IAM scoping: tenant-data access (the four - // task_id-partitioned tables + the agent's trace/attachment S3 objects) - // is NOT granted to the runtime ExecutionRole. Instead the agent assumes a - // per-task SessionRole (created below) with session tags - // {user_id, repo, task_id}, and that role carries the tenant-data grants - // constrained by aws:PrincipalTag conditions. The runtime role keeps only - // non-tenant / shared access: - // - UserConcurrencyTable: user-scoped counter (agent path does not write - // it today; left here for the reconciler/orchestrator parity). - // - GitHub PAT secret: read once at startup, before the agent assumes the - // SessionRole. - // - CloudWatch Logs + AgentCore Memory: shared/non-tenant. - userConcurrencyTable.table.grantReadWriteData(runtime); - githubTokenSecret.grantRead(runtime); - applicationLogGroup.grantWrite(runtime); - agentMemory.grantReadWrite(runtime); - - // ADR-019 P1 (context-gated): let the runtime SigV4-invoke the tool Gateway - // (``bedrock-agentcore:InvokeGateway``). No-op unless the gateway is - // provisioned. The ECS task role gets the parallel grant via the - // EcsAgentCluster prop below (substrate parity). - toolGateway?.grantInvoke(runtime); + // --- IAM grants --- + // Per-session IAM scoping: tenant-data access (the four + // task_id-partitioned tables + the agent's trace/attachment S3 objects) + // is NOT granted to the runtime ExecutionRole. Instead the agent assumes a + // per-task SessionRole (created below) with session tags + // {user_id, repo, task_id}, and that role carries the tenant-data grants + // constrained by aws:PrincipalTag conditions. The runtime role keeps only + // non-tenant / shared access: + // - UserConcurrencyTable: user-scoped counter (agent path does not write + // it today; left here for the reconciler/orchestrator parity). + // - GitHub PAT secret: read once at startup, before the agent assumes the + // SessionRole. + // - CloudWatch Logs + AgentCore Memory: shared/non-tenant. + userConcurrencyTable.table.grantReadWriteData(runtime); + githubTokenSecret.grantRead(runtime); + applicationLogGroup.grantWrite(runtime); + agentMemory.grantReadWrite(runtime); + + // ADR-019 P1 (context-gated): let the runtime SigV4-invoke the tool Gateway + // (``bedrock-agentcore:InvokeGateway``). No-op unless the gateway is + // provisioned. The ECS task role gets the parallel grant via the + // EcsAgentCluster prop below (substrate parity). + toolGateway?.grantInvoke(runtime); + } // Grant the runtime invoke on each configured foundation model + its // cross-Region inference profile in the configured geography @@ -816,8 +781,10 @@ export class AgentStack extends Stack { geoRegion: bedrockGeoRegion, model: foundationModel, }); - foundationModel.grantInvoke(runtime); - crossRegionProfile.grantInvoke(runtime); + if (runtime) { + foundationModel.grantInvoke(runtime); + crossRegionProfile.grantInvoke(runtime); + } invokableBedrockModels.push(foundationModel, crossRegionProfile); } @@ -827,10 +794,10 @@ export class AgentStack extends Stack { // by aws:PrincipalTag conditions so a compromised session reaches only its // own task's data. The agent assumes this with refreshable credentials // (1h role-chaining cap, tasks run to 8h). Trust admits the runtime - // ExecutionRole as the assuming principal; the ECS task role is added in - // the ECS block below when that backend is enabled. + // roles of the deployed backends as assuming principals. ECS and MicroVM + // admit their role during construction below. const agentSessionRole = new AgentSessionRole(this, 'AgentSessionRole', { - assumingRoles: [runtime.role], + ...(runtime ? { assumingRoles: [runtime.role] } : { deferComputeRoleBinding: true }), taskScopedTables: [ taskTable.table, taskEventsTable.table, @@ -850,12 +817,14 @@ export class AgentStack extends Stack { // which needs CloudWatch Logs resource policy propagation. Re-enable via // tracingEnabled: true once resolved. - NagSuppressions.addResourceSuppressions(runtime, [ - { - id: 'AwsSolutions-IAM5', - reason: 'AgentCore runtime requires wildcard permissions for CloudWatch Logs, Bedrock model invocation, and cross-region inference profiles — generated by CDK L2 construct grants', - }, - ], true); + if (runtime) { + NagSuppressions.addResourceSuppressions(runtime, [ + { + id: 'AwsSolutions-IAM5', + reason: 'AgentCore runtime requires wildcard permissions for CloudWatch Logs, Bedrock model invocation, and cross-region inference profiles — generated by CDK L2 construct grants', + }, + ], true); + } // Chunk 10 deploy-prep: the Cedar HITL additions (TaskApprovalsTable // grant + extra env vars) pushed the runtime @@ -909,10 +878,12 @@ export class AgentStack extends Stack { // ``main.ts``) and the suppression would arrive too late. Aspects.of(this).add(overflowSuppressionAspect, { priority: AspectPriority.MUTATING }); - new CfnOutput(this, 'RuntimeArn', { - value: runtime.agentRuntimeArn, - description: 'ARN of the AgentCore runtime', - }); + if (runtime) { + new CfnOutput(this, 'RuntimeArn', { + value: runtime.agentRuntimeArn, + description: 'ARN of the AgentCore runtime', + }); + } new CfnOutput(this, 'TaskTableName', { value: taskTable.table.tableName, @@ -992,18 +963,15 @@ export class AgentStack extends Stack { // gives a bigger, tunable task (see EcsAgentCluster for the exact vCPU/memory // sizing and the measurements behind it — a 32 GB task was OOM-killed by a // fully parallel build, which is why the build tier serialises with MISE_JOBS=1) - // for repos that set ``compute_type: 'ecs'``. GATED on the ``compute_type`` deploy context - // (default 'agentcore') — ECS resources only synthesize when you deploy with - // ``--context compute_type=ecs``, so the default synth (and the - // bootstrap-coverage test that synths with default context) stays - // agentcore-only, matching how other optional constructs are context-gated. - // (``computeType`` is read near the top of the constructor — TaskApi needs it - // for the conditional MicroVM cancel grant.) + // for repositories selecting ECS. Resources synthesize when `compute_types` + // includes ecs or the legacy `compute_type=ecs` context is used. Default + // synthesis remains AgentCore-only. The backend list is resolved near the + // top of this constructor so TaskApi can apply the same cancellation gates. // Ephemeral bucket for ECS task payloads — the orchestrator writes the // payload here (it exceeds the 8 KB RunTask containerOverrides limit) and // passes only an S3 URI pointer; the container fetches it on boot, the // orchestrator deletes it at finalize. Only synthesized under the ecs gate. - const ecsPayloadBucket = computeType === 'ecs' + const ecsPayloadBucket = ecsEnabled ? new EcsPayloadBucket(this, 'EcsPayloadBucket') : undefined; if (ecsPayloadBucket) { @@ -1018,8 +986,8 @@ export class AgentStack extends Stack { // deliberately modest so an adopter who changes nothing does not pay for the // Fargate ceiling — but a large monorepo genuinely needs more, so the knobs // have to be reachable WITHOUT editing the construct. Same shape as - // ``compute_type`` above: - // cdk deploy -c compute_type=ecs -c ecsBuildTaskCpu=16384 \ + // ``compute_types`` above: + // cdk deploy -c compute_types=agentcore,ecs -c ecsBuildTaskCpu=16384 \ // -c ecsBuildTaskMemoryMiB=122880 -c ecsBuildTaskEphemeralStorageGiB=100 // cdk deploy -c ecsExtraBuildEnv='{"MISE_JOBS":"8"}' const ecsTaskSizing = resolveEcsTaskSizing(this.node); @@ -1082,7 +1050,7 @@ export class AgentStack extends Stack { } } - const ecsCluster = computeType === 'ecs' + const ecsCluster = ecsEnabled ? new EcsAgentCluster(this, 'EcsAgentCluster', { ...(ecsTaskSizing !== undefined && { taskSizing: ecsTaskSizing }), ...(linearIdentityVault && { linearIdentityVault }), @@ -1128,10 +1096,9 @@ export class AgentStack extends Stack { // task parked on a HITL approval gate stops billing compute while keeping // its cloned repo and warm build caches in memory. // - // Gated exactly like the ECS backend above: resources synthesize only under - // ``--context compute_type=lambda-microvm``, so the default synth — and the - // bootstrap-coverage test that synths with default context — stays - // agentcore-only. The construct itself enforces the ADR's Region gate, so a + // Resources synthesize when `compute_types` includes lambda-microvm or the + // legacy `compute_type=lambda-microvm` context is used. Default synthesis + // remains AgentCore-only. The construct enforces the ADR's Region gate, so a // deploy into a Region without Lambda MicroVMs fails at synth rather than on // the first task. const lambdaMicrovm = lambdaMicrovmEnabled @@ -1149,14 +1116,6 @@ export class AgentStack extends Stack { // AZ describe) need no stack input and are wired inside the construct. githubTokenSecret, agentMemory, - // ADR-021 P2-F4: the SAME log group whose name travels to the guest in - // `agentPlatformConfig.logGroupName` below (→ `LOG_GROUP_NAME`). P2 - // delivered the name without the grant, so the agent's structured per-task - // lines and its METRICS_REPORT were AccessDenied on - // logs:CreateLogStream and the platform's canonical observability streams - // were empty on this backend. Passing the construct (not the name) keeps the - // grant and the delivered value derived from one object. - applicationLogGroup, // Resolved above TaskApi — see `microvmImageInputs`. ...microvmImageInputs, }) @@ -1169,19 +1128,30 @@ export class AgentStack extends Stack { // unconfigured one never asks for it. microvmImageArnHolder = lambdaMicrovm?.imageArn; - // Advertise which compute substrate this deploy actually provisioned, so the - // CLI can refuse to onboard a repo as ``compute_type: ecs`` when the ECS gate - // wasn't on (``--context compute_type=ecs``) — otherwise that mismatch only - // surfaces per-task as "ECS compute strategy requires ECS_CLUSTER_ARN…" at - // runtime. ``ecs`` implies the AgentCore runtime is ALSO available (the ECS - // gate is additive), so an agentcore repo works on either substrate — and the - // same holds for ``lambda-microvm`` (ADR-021). + const computeRoles = [runtime?.role, ecsCluster?.taskDefinition.taskRole, lambdaMicrovm?.executionRole] + .filter((role): role is iam.IRole => role !== undefined); + const selectedComputeRole = computeRoles[0]; + ecsClusterArnHolder = ecsCluster?.cluster.clusterArn; + agentLogGroup ??= ecsCluster?.logGroup ?? lambdaMicrovm?.logGroup; + if (!selectedComputeRole || !agentLogGroup) throw new Error('Selected compute backend did not provide its role and logs'); + if (lambdaMicrovm) { + agentLogGroup.grantWrite(lambdaMicrovm.executionRole); + toolGateway?.grantInvoke(lambdaMicrovm.executionRole); + } + + // Additive stacks keep the comma-list contract existing CLIs parse; exclusive + // stacks name their only backend. new CfnOutput(this, 'ComputeSubstrate', { - value: ecsCluster ? 'ecs' : (lambdaMicrovm ? 'lambda-microvm' : 'agentcore'), - description: 'Compute substrate provisioned by this deploy: "agentcore" (default), "ecs" ' - + '(deployed with --context compute_type=ecs; adds the Fargate substrate alongside AgentCore) ' - + 'or "lambda-microvm" (--context compute_type=lambda-microvm; adds the Lambda MicroVMs ' - + 'substrate alongside AgentCore).', + value: computeTypes.join(','), + description: 'Deployed compute backends; with ComputeDeploymentMode=exclusive, the only one.', + }); + new CfnOutput(this, 'ComputeTypes', { + value: computeTypes.join(','), + description: 'Every deployed compute backend in order; the first is the repository default.', + }); + new CfnOutput(this, 'ComputeDeploymentMode', { + value: computeTypes.length === 1 ? 'exclusive' : 'additive', + description: 'exclusive: ComputeSubstrate is the only deployed backend. additive: see ComputeTypes.', }); // Both outputs are consumed by `platform doctor` and `repo onboard --model` to @@ -1248,7 +1218,8 @@ export class AgentStack extends Stack { userConcurrencyTable: userConcurrencyTable.table, maxConcurrentTasksPerUser, repoTable: repoTable.table, - runtimeArn: runtime.agentRuntimeArn, + deployedComputeTypes: computeTypes, + runtimeArn: runtime?.agentRuntimeArn, githubTokenSecretArn: githubTokenSecret.secretArn, memoryId: agentMemory.memory.memoryId, guardrailId: inputGuardrail.guardrailId, @@ -1271,7 +1242,7 @@ export class AgentStack extends Stack { agentPlatformConfig: { taskApprovalsTableName: taskApprovalsTable.table.tableName, nudgesTableName: taskNudgesTable.table.tableName, - logGroupName: applicationLogGroup.logGroupName, + logGroupName: agentLogGroup.logGroupName, // INTENTIONAL, not a wiring bug: both keys resolve to the SAME bucket // (`traceArtifactsBucket`), exactly as `ARTIFACTS_BUCKET_NAME` and // `TRACE_ARTIFACTS_BUCKET_NAME` do in the AgentCore runtime env block above @@ -1301,21 +1272,10 @@ export class AgentStack extends Stack { // `bedrockGeoRegion` granted one geography while the agent asked for another, // and every task with no per-repo override failed at turn 0 with AccessDenied. anthropicModel: inferenceProfileId(bedrockGeoRegion, PLATFORM_DEFAULT_MODEL_ID), - // Substrate parity for the Identity vault: the AgentCore runtime gets these - // as env and the ECS container via EcsAgentCluster, so a MicroVM guest must - // receive them too or its agent skips vault minting and falls back to a - // Secrets-Manager token a vault-managed workspace does not have — losing - // reactions and state transitions on work that otherwise succeeds. Forwarded - // as platform_config because a snapshot must not bake configuration in. - ...(linearIdentityVault - ? { - linearVaultEnabled: 'true', - linearWorkloadIdentityName: linearVaultWorkload, - } - : {}), + ...(toolGateway && { toolGatewayUrl: toolGateway.gatewayUrl }), }, // Route ``compute_type: 'ecs'`` repos to the Fargate cluster above — - // only when the cluster was synthesized (deploy --context compute_type=ecs). + // only when ECS is included in the deployment's backend list. ...(ecsCluster && { ecsConfig: { clusterArn: ecsCluster.cluster.clusterArn, @@ -1420,8 +1380,8 @@ export class AgentStack extends Stack { // --- Operator dashboard --- new TaskDashboard(this, 'TaskDashboard', { - applicationLogGroup, - runtimeArn: runtime.agentRuntimeArn, + applicationLogGroup: agentLogGroup, + runtimeArn: runtime?.agentRuntimeArn, }); // --- Slack integration (always deployed — secrets populated post-deploy) --- @@ -1506,7 +1466,7 @@ export class AgentStack extends Stack { // ambient credentials. The ECS task-role grant is wired inside // EcsAgentCluster, and the webhook-processor grant inside LinearIntegration. if (linearIdentityVault) { - linearIdentityVault.grantMintToken(runtime.role); + if (runtime) linearIdentityVault.grantMintToken(runtime.role); // MicroVM parity: the guest self-mints with its AMBIENT identity, which is the // compute's execution role — not the tenant-scoped session role. Without this // the platform_config above would tell the agent to use the vault and the call @@ -1703,17 +1663,19 @@ export class AgentStack extends Stack { // For a 24h Linear access-token TTL, the practical impact is that // a stale token in the cache forces the agent's next call to fail // closed — preferable to a trust gap. - runtime.role.addToPrincipalPolicy(new iam.PolicyStatement({ - actions: ['secretsmanager:GetSecretValue'], - resources: [ - Stack.of(this).formatArn({ - service: 'secretsmanager', - resource: 'secret', - arnFormat: ArnFormat.COLON_RESOURCE_NAME, - resourceName: 'bgagent-linear-oauth-*', - }), - ], - })); + for (const role of computeRoles) { + role.addToPrincipalPolicy(new iam.PolicyStatement({ + actions: ['secretsmanager:GetSecretValue'], + resources: [ + Stack.of(this).formatArn({ + service: 'secretsmanager', + resource: 'secret', + arnFormat: ArnFormat.COLON_RESOURCE_NAME, + resourceName: 'bgagent-linear-oauth-*', + }), + ], + })); + } // Phase 2.0b-O2: pipe the workspace registry table + per-workspace // OAuth-secret-prefix grant into the orchestrator so the concurrency-cap @@ -1837,17 +1799,19 @@ export class AgentStack extends Stack { // any tenant's OAuth bundle. Lambdas (trusted code in this stack) // own the in-place refresh path; the agent proceeds with whatever // token Lambdas have most-recently written. - runtime.role.addToPrincipalPolicy(new iam.PolicyStatement({ - actions: ['secretsmanager:GetSecretValue'], - resources: [ - Stack.of(this).formatArn({ - service: 'secretsmanager', - resource: 'secret', - arnFormat: ArnFormat.COLON_RESOURCE_NAME, - resourceName: 'bgagent-jira-oauth-*', - }), - ], - })); + for (const role of computeRoles) { + role.addToPrincipalPolicy(new iam.PolicyStatement({ + actions: ['secretsmanager:GetSecretValue'], + resources: [ + Stack.of(this).formatArn({ + service: 'secretsmanager', + resource: 'secret', + arnFormat: ArnFormat.COLON_RESOURCE_NAME, + resourceName: 'bgagent-jira-oauth-*', + }), + ], + })); + } // Pipe the workspace registry table + per-tenant OAuth-secret-prefix // grant into the orchestrator so the concurrency-cap rejection path @@ -2183,6 +2147,22 @@ export class AgentStack extends Stack { ]), }); + // Apply after all consumers exist: with no Blueprint definitions, DNS or + // model logging creates the shared AwsCustomResource provider instead. + NagSuppressions.addResourceSuppressionsByPath(this, [ + `${this.stackName}/AWS679f53fac002430cb0da5b7982bd2287/ServiceRole/Resource`, + `${this.stackName}/AWS679f53fac002430cb0da5b7982bd2287/Resource`, + ], [ + { + id: 'AwsSolutions-IAM4', + reason: 'AwsCustomResource singleton Lambda uses AWS managed AWSLambdaBasicExecutionRole — required by CDK custom-resources framework', + }, + { + id: 'AwsSolutions-L1', + reason: 'AwsCustomResource singleton Lambda runtime is managed by the CDK custom-resources framework', + }, + ]); + NagSuppressions.addResourceSuppressions(invocationLogging, [ { id: 'AwsSolutions-IAM5', diff --git a/cdk/src/stacks/network.ts b/cdk/src/stacks/network.ts new file mode 100644 index 000000000..a29f9c60f --- /dev/null +++ b/cdk/src/stacks/network.ts @@ -0,0 +1,92 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import { Stack, StackProps } from 'aws-cdk-lib'; +import { NagSuppressions } from 'cdk-nag'; +import { Construct } from 'constructs'; +import { AgentNetwork, AgentVpc } from '../constructs/agent-vpc'; +import { DnsFirewall } from '../constructs/dns-firewall'; + +export type NetworkTopology = 'inline' | 'split'; + +/** Existing deployments keep their resource ownership until explicitly migrated. */ +export function resolveNetworkTopology(value: unknown): NetworkTopology { + if (value === undefined || value === 'inline') return 'inline'; + if (value === 'split') return 'split'; + throw new Error('networkTopology must be inline or split'); +} + +export interface NetworkStackProps extends StackProps { + /** Original application stack name used in generated network service properties. */ + readonly applicationStackName: string; + /** Account-specific names selected by the shared AgentCore AZ policy. */ + readonly agentCoreAvailabilityZones?: string[]; + /** Plain Blueprint configuration, resolved before constructing either stack. */ + readonly additionalAllowedDomains?: string[]; +} + +/** Owns VPC and DNS resources; application consumers only reference this stack. */ +export class NetworkStack extends Stack implements AgentNetwork { + public readonly vpc: AgentNetwork['vpc']; + public readonly runtimeSecurityGroup: AgentNetwork['runtimeSecurityGroup']; + + constructor(scope: Construct, id: string, props: NetworkStackProps) { + super(scope, id, props); + // Keep construct IDs below the stack unchanged for explicit ownership moves. + const network = new AgentVpc(this, 'AgentVpc', { + resourcePath: `${props.applicationStackName}/AgentVpc`, + ...(props.agentCoreAvailabilityZones?.length + ? { availabilityZones: props.agentCoreAvailabilityZones } + : {}), + }); + this.vpc = network.vpc; + this.runtimeSecurityGroup = network.runtimeSecurityGroup; + new DnsFirewall(this, 'DnsFirewall', { + vpc: this.vpc, + additionalAllowedDomains: props.additionalAllowedDomains, + observationMode: true, + }); + + // Export the complete interface even when a backend does not use every + // value. Otherwise switching AgentCore <-> ECS/MicroVM tries to remove an + // export while the old application still imports it, blocking the deploy. + // AZ removal additionally needs an application-only deployment to release + // imports, with networkReservedAzs preserving the remaining subnet CIDRs. + // See the split-network AZ reduction procedure in DEPLOYMENT_GUIDE.md. + this.exportValue(this.vpc.vpcId); + this.exportValue(this.runtimeSecurityGroup.securityGroupId); + for (const subnet of this.vpc.privateSubnets) this.exportValue(subnet.subnetId); + + // DNS fail-open configuration uses the CDK AwsCustomResource singleton. + // Scope these framework exceptions to its role/function, as in AgentStack. + NagSuppressions.addResourceSuppressionsByPath(this, [ + `${this.node.path}/AWS679f53fac002430cb0da5b7982bd2287/ServiceRole/Resource`, + `${this.node.path}/AWS679f53fac002430cb0da5b7982bd2287/Resource`, + ], [ + { + id: 'AwsSolutions-IAM4', + reason: 'AwsCustomResource singleton Lambda uses AWS managed AWSLambdaBasicExecutionRole — required by CDK custom-resources framework', + }, + { + id: 'AwsSolutions-L1', + reason: 'AwsCustomResource singleton Lambda runtime is managed by the CDK custom-resources framework', + }, + ]); + } +} diff --git a/cdk/src/synthesis/assembly.ts b/cdk/src/synthesis/assembly.ts new file mode 100644 index 000000000..82b6c99b8 --- /dev/null +++ b/cdk/src/synthesis/assembly.ts @@ -0,0 +1,296 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import { createHash } from 'node:crypto'; +import { readFileSync, realpathSync } from 'node:fs'; +import * as path from 'node:path'; +import { canonicalJson, Json } from '../utils/canonical-json'; + +export { canonicalJson }; + +type JsonObject = { [key: string]: Json }; + +export interface TemplateCensus { + readonly file: string; + readonly resources: number; + readonly bytes: number; + readonly parameters: number; + readonly outputs: number; + readonly types: Readonly>; + readonly sha256: string; + readonly semanticSha256: string; + readonly inventory: readonly { + logicalId: string; + type: string; + constructPath?: string; + /** Explicit template policies; null leaves CloudFormation's type-specific default unspecified. */ + deletionPolicy: Json; + updateReplacePolicy: Json; + }[]; +} + +export interface AssemblyCensus { + readonly templates: readonly TemplateCensus[]; + readonly nestedEdges: readonly { parent: string; child: string }[]; + /** Direct stack dependencies, identified by assembly-relative template paths. Asset artifacts are excluded. */ + readonly stackDependencies: Readonly>; + readonly errors: readonly string[]; + /** Sum over distinct template artifacts, not the number of deployed instances. */ + readonly totalResources: number; +} + +function object(value: Json | undefined, label: string): JsonObject { + if (!value || typeof value !== 'object' || Array.isArray(value)) throw new Error(`Expected object: ${label}`); + return value; +} + +function readJson(file: string): JsonObject { + return object(JSON.parse(readFileSync(file, 'utf8')) as Json, file); +} + +function sha256(text: string): string { + return createHash('sha256').update(text).digest('hex'); +} + +function childPath(directory: string, relative: Json | undefined): string { + if (typeof relative !== 'string') throw new Error(`Missing local template/assembly path in ${directory}`); + const base = realpathSync(directory); + const resolved = path.resolve(base, relative); + if (!resolved.startsWith(`${base}${path.sep}`)) { + throw new Error(`Template/assembly path escapes its directory: ${relative}`); + } + const physical = realpathSync(resolved); + if (!physical.startsWith(`${base}${path.sep}`)) { + throw new Error(`Template/assembly path escapes its directory through a symlink: ${relative}`); + } + return physical; +} + +interface TemplateAsset { + readonly file: string; + readonly objectKey: string; +} + +function stringValues(value: Json | undefined): string[] { + if (typeof value === 'string') return [value]; + if (value && typeof value === 'object') return Object.values(value).flatMap(stringValues); + return []; +} + +function nestedTemplate(resource: JsonObject, directory: string, assets: readonly TemplateAsset[]): string { + const metadata = object(resource.Metadata ?? {}, 'nested stack metadata'); + if (metadata['aws:asset:path']) return childPath(directory, metadata['aws:asset:path']); + const url = object(resource.Properties ?? {}, 'nested stack properties').TemplateURL; + const strings = stringValues(url); + const matches = new Set(assets.filter(asset => + strings.some(value => value === asset.objectKey || value.endsWith(`/${asset.objectKey}`)), + ).map(asset => asset.file)); + if (matches.size !== 1) throw new Error(`Missing local template path or ambiguous nested template asset in ${directory}`); + return childPath(directory, path.relative(directory, [...matches][0])); +} + +/** Follow assembly/asset manifests and nested-template metadata, never a directory glob. */ +export function inspectAssembly(directory: string): AssemblyCensus { + const root = realpathSync(directory); + const templates = new Map(); + const nestedEdges: { parent: string; child: string }[] = []; + const stackDependencies: Record = {}; + const errors: string[] = []; + const visiting = new Set(); + + function visitTemplate(file: string, assets: readonly TemplateAsset[]): void { + const name = path.relative(root, file); + if (visiting.has(file)) throw new Error(`Nested template cycle at ${name}`); + if (templates.has(name)) return; + visiting.add(file); + const raw = readFileSync(file, 'utf8'); + const template = object(JSON.parse(raw) as Json, name); + const resources = object(template.Resources ?? {}, `${name}/Resources`); + const types: Record = {}; + const inventory = Object.entries(resources).map(([logicalId, value]) => { + const resource = object(value, logicalId); + if (typeof resource.Type !== 'string') throw new Error(`Missing resource type: ${name}/${logicalId}`); + types[resource.Type] = (types[resource.Type] ?? 0) + 1; + const metadata = object(resource.Metadata ?? {}, `${logicalId}/Metadata`); + if (resource.Type === 'AWS::CloudFormation::Stack') { + const child = nestedTemplate(resource, path.dirname(file), assets); + nestedEdges.push({ parent: name, child: path.relative(root, child) }); + visitTemplate(child, assets); + } + return { + logicalId, + type: resource.Type, + ...(typeof metadata['aws:cdk:path'] === 'string' ? { constructPath: metadata['aws:cdk:path'] } : {}), + deletionPolicy: resource.DeletionPolicy ?? null, + updateReplacePolicy: resource.UpdateReplacePolicy ?? null, + }; + }); + templates.set(name, { + file: name, + resources: inventory.length, + bytes: Buffer.byteLength(raw), + parameters: Object.keys(object(template.Parameters ?? {}, `${name}/Parameters`)).length, + outputs: Object.keys(object(template.Outputs ?? {}, `${name}/Outputs`)).length, + types, + sha256: sha256(raw), + semanticSha256: sha256(canonicalJson(template)), + inventory, + }); + visiting.delete(file); + } + + function visitManifest(dir: string): void { + const manifest = readJson(path.join(dir, 'manifest.json')); + const artifacts = object(manifest.artifacts, 'artifacts'); + const missing = manifest.missing ?? []; + if (!Array.isArray(missing)) throw new Error('Invalid missing-context entries'); + for (const value of missing) { + const lookup = object(value, 'missing context'); + if (typeof lookup.key !== 'string' || typeof lookup.provider !== 'string') throw new Error('Invalid missing-context entry'); + errors.push(`Unresolved CDK context in ${path.relative(root, dir) || '.'}: ${lookup.key} (${lookup.provider})`); + } + const assetsByArtifact = new Map(); + for (const [assetManifestId, artifactValue] of Object.entries(artifacts)) { + const artifact = object(artifactValue, 'artifact'); + if (artifact.type !== 'cdk:asset-manifest') continue; + const assets: TemplateAsset[] = []; + assetsByArtifact.set(assetManifestId, assets); + const properties = object(artifact.properties, 'asset manifest properties'); + const assetFile = childPath(dir, properties.file); + const assetManifest = readJson(assetFile); + for (const assetValue of Object.values(object(assetManifest.files ?? {}, 'file assets'))) { + const asset = object(assetValue, 'asset'); + const source = object(asset.source, 'asset source'); + if (source.packaging !== 'file') continue; + if (typeof source.path !== 'string') throw new Error('Missing asset source path'); + for (const destinationValue of Object.values(object(asset.destinations, 'asset destinations'))) { + const destination = object(destinationValue, 'destination'); + if (typeof destination.objectKey !== 'string') throw new Error('Missing asset object key'); + // Unstaged file assets may live outside the assembly. Only validate + // containment when the asset is actually a referenced nested template. + assets.push({ file: path.resolve(path.dirname(assetFile), source.path), objectKey: destination.objectKey }); + } + } + } + for (const [id, value] of Object.entries(artifacts)) { + const artifact = object(value, id); + const properties = object(artifact.properties ?? {}, `${id}/properties`); + const metadataSources = [object(artifact.metadata ?? {}, `${id}/metadata`)]; + if (artifact.additionalMetadataFile) metadataSources.push(readJson(childPath(dir, artifact.additionalMetadataFile))); + for (const metadata of metadataSources) { + for (const [scope, entries] of Object.entries(metadata)) { + if (!Array.isArray(entries)) throw new Error(`Expected metadata array: ${scope}`); + for (const entry of entries) { + const annotation = object(entry, scope); + if (annotation.type === 'aws:cdk:error') errors.push(`${scope}: ${String(annotation.data)}`); + } + } + } + if (artifact.type === 'aws:cloudformation:stack') { + const file = childPath(dir, properties.templateFile); + const dependencies = artifact.dependencies ?? []; + if (!Array.isArray(dependencies) || dependencies.some(v => typeof v !== 'string')) { + throw new Error(`Invalid dependencies: ${id}`); + } + const dependencyIds = dependencies as string[]; + const name = path.relative(root, file); + if (Object.hasOwn(stackDependencies, name)) throw new Error(`Multiple stack artifacts reference ${name}`); + stackDependencies[name] = [...new Set(dependencyIds.flatMap(dependency => { + if (!Object.hasOwn(artifacts, dependency)) { + throw new Error(`Unknown dependency of ${id}: ${dependency}`); + } + const target = object(artifacts[dependency], dependency); + if (target.type !== 'aws:cloudformation:stack') return []; + const targetProperties = object(target.properties, `${dependency}/properties`); + return [path.relative(root, childPath(dir, targetProperties.templateFile))]; + }))].sort(); + // Different stacks can publish identical child contents under the same + // asset hash. Resolve only through this stack's own asset manifests. + visitTemplate(file, dependencyIds.flatMap(dependency => assetsByArtifact.get(dependency) ?? [])); + } else if (artifact.type === 'cdk:cloud-assembly') { + visitManifest(childPath(dir, properties.directoryName)); + } + } + } + + visitManifest(root); + if (!templates.size) throw new Error(`No stack templates found in ${root}`); + const checked = new Set(); + function checkDependencies(file: string): void { + if (visiting.has(file)) throw new Error(`Stack dependency cycle at ${file}`); + if (checked.has(file)) return; + visiting.add(file); + for (const dependency of stackDependencies[file]) checkDependencies(dependency); + visiting.delete(file); + checked.add(file); + } + for (const file of Object.keys(stackDependencies)) checkDependencies(file); + const measured = [...templates.values()].sort((a, b) => a.file.localeCompare(b.file)); + return { + templates: measured, + nestedEdges, + stackDependencies, + errors, + totalResources: measured.reduce((sum, template) => sum + template.resources, 0), + }; +} + +export interface AssemblyDifference { + readonly kind: 'template' | 'stack-dependencies'; + readonly file: string; + /** JSON pointers within a template; empty for dependencies stored in assembly manifests. */ + readonly paths: readonly string[]; + readonly totalDifferences: number; +} + +/** JSON-pointer paths only: diagnostics do not print resource property values. */ +export function compareAssemblies(first: string, second: string): readonly AssemblyDifference[] { + const a = inspectAssembly(first); + const b = inspectAssembly(second); + const names = [...new Set([...a.templates, ...b.templates].map(t => t.file))].sort(); + const differences: AssemblyDifference[] = names.flatMap(file => { + const left = a.templates.find(t => t.file === file); + const right = b.templates.find(t => t.file === file); + if (left?.semanticSha256 === right?.semanticSha256) return []; + if (!left || !right) return [{ kind: 'template' as const, file, paths: ['/'], totalDifferences: 1 }]; + const paths: string[] = []; + let totalDifferences = 0; + function visit(x: Json | undefined, y: Json | undefined, pointer: string): void { + if (x !== undefined && y !== undefined && canonicalJson(x) === canonicalJson(y)) return; + if (x && y && typeof x === 'object' && typeof y === 'object' && !Array.isArray(x) && !Array.isArray(y)) { + for (const key of [...new Set([...Object.keys(x), ...Object.keys(y)])].sort()) { + visit(x[key], y[key], `${pointer}/${key.replace(/~/g, '~0').replace(/\//g, '~1')}`); + } + } else { + totalDifferences++; + if (paths.length < 100) paths.push(pointer || '/'); + } + } + visit(readJson(path.join(first, file)), readJson(path.join(second, file)), ''); + return [{ kind: 'template' as const, file, paths, totalDifferences }]; + }); + for (const file of [...new Set([...Object.keys(a.stackDependencies), ...Object.keys(b.stackDependencies)])].sort()) { + const left = a.stackDependencies[file]; + const right = b.stackDependencies[file]; + if (!left || !right || canonicalJson([...left]) !== canonicalJson([...right])) { + differences.push({ kind: 'stack-dependencies', file, paths: [], totalDifferences: 1 }); + } + } + return differences; +} diff --git a/cdk/src/synthesis/audit.ts b/cdk/src/synthesis/audit.ts new file mode 100644 index 000000000..fd4bd92d0 --- /dev/null +++ b/cdk/src/synthesis/audit.ts @@ -0,0 +1,87 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import { AssemblyCensus, AssemblyDifference, compareAssemblies } from './assembly'; +import type { Budgets } from './budgets'; +import { SynthesisProfile } from './profiles'; + +export { DEFAULT_BUDGETS } from './budgets'; +export type { Budgets } from './budgets'; + +export type WorkerResult = { kind: 'synthesized'; census: AssemblyCensus } | { kind: 'rejected'; error: string }; +export type Worker = (profile: SynthesisProfile, directory: string) => WorkerResult; + +export interface ProfileAudit { + readonly profile: SynthesisProfile; + readonly first?: WorkerResult; + readonly second?: WorkerResult; + readonly differences: readonly AssemblyDifference[]; + readonly failures: readonly string[]; +} + +function resultFailures(profile: SynthesisProfile, result: WorkerResult, budgets: Budgets): string[] { + if (result.kind === 'rejected') { + const expected = profile.expectedError; + if (!expected) return [result.error]; + if (typeof expected === 'string') return result.error.startsWith(expected) ? [] : [result.error]; + const match = /^Number of resources in stack '([^']+)': (\d+) is greater than allowed maximum of (\d+):/.exec(result.error); + return match && match[1] === expected.stackName && Number(match[2]) > expected.resourceLimit + && Number(match[3]) === expected.resourceLimit ? [] : [result.error]; + } + const failures = [...result.census.errors]; + if (profile.expectedError) failures.push(`Expected rejection was not raised: ${JSON.stringify(profile.expectedError)}`); + for (const template of result.census.templates) { + for (const metric of ['resources', 'bytes', 'parameters', 'outputs'] as const) { + if (template[metric] > budgets[metric]) { + failures.push(`${template.file}: ${template[metric]} ${metric} exceeds ${budgets[metric]}`); + } + } + } + return failures; +} + +/** Apply identical acceptance rules to both independent runs, including expected rejections. */ +export function auditProfile( + profile: SynthesisProfile, + directory: string, + budgets: Budgets, + checkStability: boolean, + worker: Worker, +): ProfileAudit { + const failures: string[] = []; + let first: WorkerResult | undefined; + let second: WorkerResult | undefined; + let differences: readonly AssemblyDifference[] = []; + try { + first = worker(profile, directory); + failures.push(...resultFailures(profile, first, budgets)); + if (checkStability) { + const repeatDirectory = `${directory}-repeat`; + second = worker(profile, repeatDirectory); + failures.push(...resultFailures(profile, second, budgets).map(failure => `Repeat: ${failure}`)); + if (first.kind === 'synthesized' && second.kind === 'synthesized') { + differences = compareAssemblies(directory, repeatDirectory); + if (differences.length) failures.push(`Unstable assembly: ${differences.map(d => `${d.file} (${d.kind})`).join(', ')}`); + } + } + } catch (error) { + failures.push(error instanceof Error ? error.message : String(error)); + } + return { profile, first, second, differences, failures }; +} diff --git a/cdk/src/synthesis/budgets.ts b/cdk/src/synthesis/budgets.ts new file mode 100644 index 000000000..c5261565d --- /dev/null +++ b/cdk/src/synthesis/budgets.ts @@ -0,0 +1,23 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +export type Budgets = Readonly>; + +/** Shared by production synthesis and the census; reserve CloudFormation headroom. */ +export const DEFAULT_BUDGETS: Budgets = { resources: 490, bytes: 800_000, parameters: 200, outputs: 200 }; diff --git a/cdk/src/synthesis/cli.ts b/cdk/src/synthesis/cli.ts new file mode 100644 index 000000000..1b4db1d30 --- /dev/null +++ b/cdk/src/synthesis/cli.ts @@ -0,0 +1,192 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import { spawnSync } from 'node:child_process'; +import { mkdirSync, readFileSync, writeFileSync } from 'node:fs'; +import * as path from 'node:path'; +import { parseArgs } from 'node:util'; +import bedrockPackage from '@aws-cdk/aws-bedrock-alpha/package.json'; +import cdkPackage from 'aws-cdk-lib/package.json'; +import { buildApp } from '../main'; +import { inspectAssembly } from './assembly'; +import { auditProfile, DEFAULT_BUDGETS, ProfileAudit, WorkerResult } from './audit'; +import { FIXTURE, STRUCTURAL_CONTEXT, synthesisEnvironment, synthesisProfiles, SynthesisProfile } from './profiles'; +import { createOutputDirectory, projectContext, sourceProvenance } from './workspace'; + +const PROCESS_OUTPUT_LIMIT = 8_388_608; +const MAX_TEMPLATE_BYTES = 1_000_000; +const CHECKOUT = path.resolve(__dirname, '../../..'); + +const HELP = `Usage: mise //cdk:census -- [options] + + --list List named structural profiles + --profile NAME Select a profile (repeatable; default: all) + --output DIRECTORY New output directory (default: a temporary directory) + --check-stability Synthesize twice in independent processes; fail on differences + --max-resources NUMBER Per-template audit ceiling (default/maximum: ${DEFAULT_BUDGETS.resources}) + --max-template-bytes NUMBER Per-template ceiling (default: 800000) + --help Show this help + +Uses the production buildApp with fixed account/AZ/context inputs, metadata enabled, +and bundling/staging disabled. No AWS credentials or network lookups are required. +This is structural evidence, not a deploy or bundled-release validation. +Production synthesis always enforces the ${DEFAULT_BUDGETS.resources}-resource ceiling; audit options cannot raise it. +Reports, source fingerprints, templates, and per-profile logs stay in the output +directory. The stability check preserves timestamps, logical IDs, metadata, and +stack dependencies. Missing CDK context fails the audit instead of triggering lookups. +`; + +function writeJson(file: string, value: unknown): void { + writeFileSync(file, `${JSON.stringify(value, null, 2)}\n`); +} + +function errorMessage(error: unknown): string { + return error instanceof Error ? error.message : String(error); +} + +async function synthesize(profile: SynthesisProfile, directory: string): Promise { + let result: WorkerResult; + try { + // Only workers construct the app; the parent never synthesizes with its environment. + const app = await buildApp({ + account: FIXTURE.account, + region: FIXTURE.region, + describeAzs: async () => [...FIXTURE.zones], + resolveCallerAccount: async () => FIXTURE.account, + appProps: { + outdir: directory, + autoSynth: false, + context: { ...projectContext(CHECKOUT), ...profile.context }, + postCliContext: STRUCTURAL_CONTEXT, + }, + }); + app.synth(); + result = { kind: 'synthesized', census: inspectAssembly(directory) }; + } catch (error) { + result = { kind: 'rejected', error: errorMessage(error) }; + } + writeJson(path.join(directory, 'result.json'), result); +} + +function runWorker(profile: SynthesisProfile, directory: string): WorkerResult { + mkdirSync(directory); + const child = spawnSync(process.execPath, [ + '-r', require.resolve('ts-node/register/transpile-only'), __filename, + '--worker', '--profile', profile.name, '--output', directory, + ], { + cwd: path.resolve(__dirname, '../..'), + env: synthesisEnvironment(process.env), + encoding: 'utf8', + maxBuffer: PROCESS_OUTPUT_LIMIT, + timeout: 120_000, + }); + writeFileSync(path.join(directory, 'synth.log'), `${child.stdout ?? ''}${child.stderr ?? ''}`); + if (child.error || child.status !== 0) { + throw new Error(`Synthesis process failed (${child.signal ?? child.status}): ${child.error?.message ?? `see ${path.join(directory, 'synth.log')}`}`); + } + return JSON.parse(readFileSync(path.join(directory, 'result.json'), 'utf8')) as WorkerResult; +} + +function ceiling(value: string | undefined, fallback: number, maximum: number, name: string): number { + const parsed = value === undefined ? fallback : Number(value); + if (!Number.isInteger(parsed) || parsed < 1 || parsed > maximum) throw new Error(`${name} must be an integer from 1 to ${maximum}`); + return parsed; +} + +async function main(): Promise { + const { values } = parseArgs({ + options: { + 'help': { type: 'boolean' }, + 'list': { type: 'boolean' }, + 'profile': { type: 'string', multiple: true }, + 'output': { type: 'string' }, + 'check-stability': { type: 'boolean' }, + 'max-resources': { type: 'string' }, + 'max-template-bytes': { type: 'string' }, + 'worker': { type: 'boolean' }, + }, + }); + if (values.help) { process.stdout.write(HELP); return; } + const all = synthesisProfiles(); + if (values.list) { + for (const profile of all) process.stdout.write(`${profile.name}${profile.expectedError ? ' [expected rejection]' : ''}\n`); + return; + } + const names = values.profile ?? all.map(p => p.name); + const selected = [...new Set(names)].map(name => { + const found = all.find(profile => profile.name === name); + if (!found) throw new Error(`Unknown profile '${name}'; use --list`); + return found; + }); + if (values.worker) { + if (selected.length !== 1 || !values.output) throw new Error('Worker requires one profile and an output directory'); + await synthesize(selected[0], path.resolve(values.output)); + return; + } + const resourceLimit = ceiling(values['max-resources'], DEFAULT_BUDGETS.resources, DEFAULT_BUDGETS.resources, 'max-resources'); + const byteLimit = ceiling(values['max-template-bytes'], DEFAULT_BUDGETS.bytes, MAX_TEMPLATE_BYTES, 'max-template-bytes'); + const directory = createOutputDirectory(CHECKOUT, values.output); + const before = sourceProvenance(CHECKOUT); + const baseContext = projectContext(CHECKOUT); + const budgets = { ...DEFAULT_BUDGETS, resources: resourceLimit, bytes: byteLimit }; + const results: ProfileAudit[] = []; + let failed = false; + for (const profile of selected) { + const firstDir = path.join(directory, profile.name); + const result = auditProfile(profile, firstDir, budgets, !!values['check-stability'], runWorker); + const { first, failures } = result; + failed ||= failures.length > 0; + results.push(result); + const status = failures.length ? 'FAIL' : first?.kind === 'rejected' ? 'EXPECTED REJECTION' : 'PASS'; + process.stdout.write(`${status} ${profile.name}${first?.kind === 'synthesized' ? `: ${first.census.totalResources} template resources across ${first.census.templates.length} templates` : ''}\n`); + for (const failure of failures) process.stderr.write(` ${failure}\n`); + } + const after = sourceProvenance(CHECKOUT); + const sourceChanged = before.sourceSha256 !== after.sourceSha256; + failed ||= sourceChanged; + writeJson(path.join(directory, 'report.json'), { + schemaVersion: 2, + mode: 'structural-unbundled', + provenance: before, + sourceChangedDuringRun: sourceChanged, + versions: { + node: process.version, + cdk: cdkPackage.version, + bedrockAlpha: bedrockPackage.version, + }, + fixture: FIXTURE, + workerEnvironment: synthesisEnvironment(process.env), + projectContext: baseContext, + contextOverrides: STRUCTURAL_CONTEXT, + budgets, + stabilityChecked: !!values['check-stability'], + results, + }); + if (sourceChanged) process.stderr.write('Source inputs changed during the census; rerun against a stable checkout.\n'); + process.stdout.write(`Report: ${path.join(directory, 'report.json')}\n`); + process.exitCode = failed ? 1 : 0; +} + +/* istanbul ignore next -- exercised through the command-line smoke checks */ +if (require.main === module) { + void main().catch(error => { + process.stderr.write(`${errorMessage(error)}\n`); + process.exitCode = 1; + }); +} diff --git a/cdk/src/synthesis/profiles.ts b/cdk/src/synthesis/profiles.ts new file mode 100644 index 000000000..69a4f2bee --- /dev/null +++ b/cdk/src/synthesis/profiles.ts @@ -0,0 +1,219 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import { DISABLE_ASSET_STAGING_CONTEXT } from 'aws-cdk-lib/cx-api'; +import { DEFAULT_BUDGETS } from './budgets'; +import { AGENTCORE_AZS_CONTEXT_KEY } from '../constructs/agentcore-azs'; +import { DEFAULT_BEDROCK_MODEL_IDS } from '../handlers/shared/bedrock-model-constants'; + +/** Enough additional model grants to exercise the SessionRole's lazy policy split. */ +const EXTRA_MODEL_COUNT = 8; + +export type Compute = 'agentcore' | 'ecs' | 'lambda-microvm'; +export type Image = 'none' | 'managed' | 'external'; +export type Context = Readonly>; + +/** Structural profiles describe provisioned resources, not live backend readiness. */ +export interface SynthesisProfile { + readonly name: string; + readonly context: Context; + readonly microvmImageConfigured: boolean; + /** An error prefix or the specific resource ceiling that must reject this profile. */ + readonly expectedError?: string | { readonly stackName: string; readonly resourceLimit: number }; +} + +export const FIXTURE = { + account: '123456789012', + region: 'us-east-1', + zones: [ + { zoneName: 'us-east-1a', zoneId: 'use1-az2' }, + { zoneName: 'us-east-1b', zoneId: 'use1-az4' }, + { zoneName: 'us-east-1c', zoneId: 'use1-az1' }, + ], +} as const; + +/** CDK's own context lookup is separate from buildApp's injected AWS lookup functions. */ +export const STRUCTURAL_CONTEXT: Context = { + 'aws:cdk:version-reporting': true, + 'aws:cdk:enable-path-metadata': true, + [DISABLE_ASSET_STAGING_CONTEXT]: true, + [`availability-zones:account=${FIXTURE.account}:region=${FIXTURE.region}`]: FIXTURE.zones.map(zone => zone.zoneName), +}; + +function profile(compute: Compute, gateway: boolean, registry: boolean, vault: boolean, image: Image): SynthesisProfile { + return { + name: `${compute}-gw${+gateway}-reg${+registry}-vault${+vault}-${image}`, + microvmImageConfigured: compute === 'lambda-microvm' && image !== 'none', + context: { + stackName: 'backgroundagent-dev', + networkTopology: 'inline', + blueprintRepo: 'awslabs/agent-plugins', + bedrockGeoRegion: 'global', + compute_types: compute, + enableToolGateway: gateway, + enableAgentRegistry: registry, + enableLinearIdentityVault: vault, + ...(image === 'managed' ? { + microvm_base_image_arn: 'arn:aws:lambda:us-east-1:aws:microvm-image:al2023-1', + microvm_base_image_version: '1', + } : {}), + ...(image === 'external' ? { + microvm_image_identifier: 'arn:aws:lambda:us-east-1:123456789012:microvm-image:census-image', + microvm_image_version: '1', + } : {}), + }, + }; +} + +/** One profile product shared by the CLI and its coverage assertions. */ +export function synthesisProfiles(): readonly SynthesisProfile[] { + const profiles: SynthesisProfile[] = []; + for (const compute of ['agentcore', 'ecs', 'lambda-microvm'] as const) { + for (const gateway of [false, true]) { + for (const registry of [true, false]) { + for (const vault of [false, true]) { + const images: readonly Image[] = compute === 'lambda-microvm' ? ['none', 'managed', 'external'] : ['none']; + for (const image of images) profiles.push(profile(compute, gateway, registry, vault, image)); + } + } + } + } + + // Probe supplemental options together for every backend's widest profile: + // IAM policy overflow means their effects cannot be added to default counts. + const supplemental: SynthesisProfile[] = [ + profile('agentcore', false, true, false, 'none'), + profile('agentcore', true, true, true, 'none'), + profile('ecs', true, true, true, 'none'), + profile('lambda-microvm', true, true, true, 'managed'), + ].map(base => ({ + ...base, + name: `${base.name}-email-fork`, + context: { ...base.context, alertEmail: 'census@example.com', forkBlueprintRepo: 'example/census-blueprints' }, + })); + profiles.push(...supplemental); + const externalConsent = profile('ecs', true, true, true, 'none'); + profiles.push({ + ...externalConsent, + name: `${externalConsent.name}-external-consent`, + context: { ...externalConsent.context, linearVaultHostedReturnUrl: 'https://example.com/consent' }, + }); + // Keeping the named AgentCore log groups across backend switches pushes + // this two-zone MicroVM combination over budget as well. + const widestInlineMicrovm = 'lambda-microvm-gw1-reg1-vault1-managed-email-fork'; + const topologies: SynthesisProfile[] = [...profiles.map(candidate => candidate.name === widestInlineMicrovm + ? { ...candidate, expectedError: { stackName: 'backgroundagent-dev', resourceLimit: DEFAULT_BUDGETS.resources } } + : candidate), ...profiles.map(candidate => ({ + ...candidate, + name: `${candidate.name}-split`, + context: { ...candidate.context, networkTopology: 'split' }, + }))]; + // Auto-pin still selects two zones. Explicit pins use every requested zone, + // adding eight resources that the original two-zone product could not expose. + for (const base of supplemental.filter(candidate => candidate.context.enableToolGateway)) { + for (const networkTopology of ['inline', 'split'] as const) { + const overBudget = networkTopology === 'inline'; + topologies.push({ + ...base, + name: `${base.name}-az3${networkTopology === 'split' ? '-split' : ''}`, + context: { + ...base.context, + networkTopology, + [AGENTCORE_AZS_CONTEXT_KEY]: FIXTURE.zones.map(zone => zone.zoneName), + }, + ...(overBudget ? { + expectedError: { stackName: 'backgroundagent-dev', resourceLimit: DEFAULT_BUDGETS.resources }, + } : {}), + }); + } + } + // Model grants can overflow IAM policies without first exceeding the resource + // budget. Synthetic IDs exercise policy size only, not live model availability. + const expandedModels = [ + ...DEFAULT_BEDROCK_MODEL_IDS, + ...Array.from({ length: EXTRA_MODEL_COUNT }, (_, index) => `anthropic.claude-census-${index}-v1:0`), + ]; + for (const base of supplemental.filter(candidate => candidate.context.enableToolGateway)) { + for (const networkTopology of ['inline', 'split'] as const) { + topologies.push({ + ...base, + name: `${base.name}-expanded-models${networkTopology === 'split' ? '-split' : ''}`, + context: { ...base.context, networkTopology, bedrockModels: expandedModels }, + ...(networkTopology === 'inline' && base.context.compute_types === 'lambda-microvm' ? { + expectedError: { stackName: 'backgroundagent-dev', resourceLimit: DEFAULT_BUDGETS.resources }, + } : {}), + }); + } + } + // Additive probes: several backends in one stack (`compute_types`). Measured + // in both topologies; the first listed backend is the repository default. + const ALL_BACKENDS = 3; + const additive: Array = [ + ['agentcore', 'lambda-microvm'], ['agentcore', 'ecs'], ['ecs', 'lambda-microvm'], + ['agentcore', 'ecs', 'lambda-microvm'], + ]; + for (const backends of additive) { + for (const wide of [false, true]) { + const microvm = backends.includes('lambda-microvm'); + const base = profile(microvm ? 'lambda-microvm' : backends[backends.length - 1], wide, true, wide, microvm ? 'managed' : 'none'); + for (const networkTopology of ['inline', 'split'] as const) { + // Inline, only the lighter two-backend stacks fit; split fits every combination. + const overBudget = networkTopology === 'inline' && (wide || backends.length === ALL_BACKENDS); + topologies.push({ + ...base, + name: `additive-${backends.join('+')}-${wide ? 'widest' : 'default'}-${networkTopology}`, + context: { + ...base.context, + compute_types: backends.join(','), + networkTopology, + ...(wide ? { alertEmail: 'census@example.com', forkBlueprintRepo: 'example/census-blueprints' } : {}), + }, + ...(overBudget ? { + expectedError: { stackName: 'backgroundagent-dev', resourceLimit: DEFAULT_BUDGETS.resources }, + } : {}), + }); + } + } + } + // The original selector remains additive: exercise real legacy inputs rather + // than relying on the equivalent explicit list to protect upgrade behavior. + for (const compute of ['ecs', 'lambda-microvm'] as const) { + const base = profile(compute, false, true, false, compute === 'lambda-microvm' ? 'managed' : 'none'); + const { compute_types: _explicit, ...context } = base.context; + for (const networkTopology of ['inline', 'split'] as const) { + topologies.push({ + ...base, + name: `legacy-${compute}-${networkTopology}`, + context: { ...context, compute_type: compute, networkTopology }, + }); + } + } + return topologies; +} + +/** Never inherit deploy context, credentials, NODE_OPTIONS, or blueprint overrides. */ +export function synthesisEnvironment(parent: NodeJS.ProcessEnv): NodeJS.ProcessEnv { + return { + ...(parent.PATH ? { PATH: parent.PATH } : {}), + ...(parent.TMPDIR ? { TMPDIR: parent.TMPDIR } : {}), + AWS_REGION: FIXTURE.region, + AWS_EC2_METADATA_DISABLED: 'true', + CDK_CONTEXT_JSON: JSON.stringify({ 'aws:cdk:bundling-stacks': [] }), + }; +} diff --git a/cdk/src/synthesis/workspace.ts b/cdk/src/synthesis/workspace.ts new file mode 100644 index 000000000..650bffcb7 --- /dev/null +++ b/cdk/src/synthesis/workspace.ts @@ -0,0 +1,106 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import { execFileSync } from 'node:child_process'; +import { createHash } from 'node:crypto'; +import { lstatSync, mkdirSync, mkdtempSync, readFileSync, readlinkSync, realpathSync } from 'node:fs'; +import { tmpdir } from 'node:os'; +import * as path from 'node:path'; + +const GIT_OUTPUT_LIMIT = 8_388_608; +const EXECUTABLE_PERMISSION_BITS = 0o111; + +export interface SourceProvenance { + readonly commit: string; + readonly dirty: boolean; + readonly sourceSha256: string; + readonly lockfileSha256: string; + readonly fingerprintFormat: 'git-visible-v2'; + readonly fileCount: number; +} + +function hash(value: Buffer | string): string { + return createHash('sha256').update(value).digest('hex'); +} + +/** Fingerprint Git-visible files and symlink identities, including uncommitted changes. + * Ignored build output/dependencies and external symlink targets are not release attestations. + */ +export function sourceProvenance(root: string): SourceProvenance { + // Git hook variables (e.g. GIT_INDEX_FILE) must not redirect the checkout being measured. + const env = Object.fromEntries(Object.entries(process.env).filter(([key]) => !key.startsWith('GIT_'))); + const git = (...args: string[]) => execFileSync('git', args, { + cwd: root, env, encoding: 'utf8', maxBuffer: GIT_OUTPUT_LIMIT, + }); + const files = [...new Set(git('ls-files', '--cached', '--others', '--exclude-standard', '-z').split('\0').filter(Boolean))].sort(); + const digest = createHash('sha256').update('git-visible-v2\n'); + for (const file of files) { + const absolute = path.join(root, file); + let entry: object; + try { + const stat = lstatSync(absolute); + if (stat.isSymbolicLink()) { + entry = { file, kind: 'symlink', target: readlinkSync(absolute) }; + } else if (stat.isFile()) { + // eslint-disable-next-line no-bitwise -- POSIX file modes encode executable permissions as bits. + const executable = (stat.mode & EXECUTABLE_PERMISSION_BITS) !== 0; + entry = { file, kind: 'file', executable, sha256: hash(readFileSync(absolute)) }; + } else { + throw new Error(`Unsupported Git-visible source entry: ${file}`); + } + } catch (error) { + if ((error as NodeJS.ErrnoException).code !== 'ENOENT') throw error; + entry = { file, kind: 'deleted' }; + } + // Typed, framed records distinguish deletions, file bytes, symlinks, and modes. + digest.update(JSON.stringify(entry)).update('\n'); + } + return { + commit: git('rev-parse', '--verify', 'HEAD').trim(), + dirty: git('status', '--porcelain').trim().length > 0, + sourceSha256: digest.digest('hex'), + lockfileSha256: hash(readFileSync(path.join(root, 'yarn.lock'))), + fingerprintFormat: 'git-visible-v2', + fileCount: files.length, + }; +} + +/** Use versioned CDK defaults, then let the named profile and structural overrides win. */ +export function projectContext(root: string): Record { + const config = JSON.parse(readFileSync(path.join(root, 'cdk/cdk.json'), 'utf8')); + const context: unknown = config.context ?? {}; + if (!context || typeof context !== 'object' || Array.isArray(context)) throw new Error('cdk.json context must be an object'); + return context as Record; +} + +/** Allocate a fresh directory, checking real paths before writing even through symlinked parents. */ +export function createOutputDirectory(root: string, output?: string, temporaryRoot = tmpdir()): string { + const checkout = realpathSync(root); + const requested = output === undefined ? undefined : path.resolve(output); + const parent = realpathSync(requested ? path.dirname(requested) : temporaryRoot); + const target = requested ? path.join(parent, path.basename(requested)) : parent; + if (target === checkout || target.startsWith(`${checkout}${path.sep}`)) { + throw new Error('Census output must be outside the checkout to avoid recursive asset fingerprinting'); + } + if (requested) { + mkdirSync(target); // Reject existing directories and symlinks, including dangling symlinks. + return target; + } + return mkdtempSync(path.join(parent, 'abca-census-')); +} diff --git a/cdk/src/utils/canonical-json.ts b/cdk/src/utils/canonical-json.ts new file mode 100644 index 000000000..53c33c83c --- /dev/null +++ b/cdk/src/utils/canonical-json.ts @@ -0,0 +1,29 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +export type Json = null | boolean | number | string | Json[] | { [key: string]: Json }; + +/** Object key order is irrelevant; array order and every value are preserved. */ +export function canonicalJson(value: Json): string { + if (Array.isArray(value)) return `[${value.map(canonicalJson).join(',')}]`; + if (value !== null && typeof value === 'object') { + return `{${Object.keys(value).sort().map(key => `${JSON.stringify(key)}:${canonicalJson(value[key])}`).join(',')}}`; + } + return JSON.stringify(value); +} diff --git a/cdk/test/blueprints/definitions.test.ts b/cdk/test/blueprints/definitions.test.ts new file mode 100644 index 000000000..ba586bd8e --- /dev/null +++ b/cdk/test/blueprints/definitions.test.ts @@ -0,0 +1,75 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import { App } from 'aws-cdk-lib'; +import { BlueprintDefinition, blueprintEgressDomains, resolveBlueprintDefinitions } from '../../src/blueprints/definitions'; + +describe('Blueprint configuration before stack construction', () => { + test('preserves the default repository and provisioning construct ID', () => { + expect(resolveBlueprintDefinitions(new App().node, {})).toEqual([ + { id: 'AgentPluginsBlueprint', repo: 'awslabs/agent-plugins' }, + ]); + }); + + test('preserves context repositories and the fork registry assets', () => { + const app = new App({ context: { blueprintRepo: 'example/plugins', forkBlueprintRepo: 'example/fork' } }); + expect(resolveBlueprintDefinitions(app.node, {})).toEqual([ + { id: 'AgentPluginsBlueprint', repo: 'example/plugins' }, + { + id: 'ForkBlueprint', + repo: 'example/fork', + assets: { + mcpServers: ['registry://mcp_server/acme/aws-knowledge@^1.0.0'], + cedarPolicyModules: ['registry://cedar_policy_module/acme/guard@^1.0.0'], + skills: ['registry://skill/acme/readme-helper@^1.0.0'], + }, + }, + ]); + }); + + test('keeps environment precedence, including an explicit empty fork override', () => { + const node = new App({ context: { blueprintRepo: 'context/plugins', forkBlueprintRepo: 'context/fork' } }).node; + expect(resolveBlueprintDefinitions(node, { BLUEPRINT_REPO: 'env/plugins', FORK_BLUEPRINT_REPO: 'env/fork' }) + .map(definition => definition.repo)).toEqual(['env/plugins', 'env/fork']); + expect(resolveBlueprintDefinitions(node, { FORK_BLUEPRINT_REPO: '' })).toEqual([ + { id: 'AgentPluginsBlueprint', repo: 'context/plugins' }, + ]); + }); + + test('creates independent definitions for each app', () => { + const node = new App({ context: { forkBlueprintRepo: 'example/fork' } }).node; + const first = resolveBlueprintDefinitions(node, {}); + first[1].assets!.skills!.push('registry://skill/example/custom@^1.0.0'); + expect(resolveBlueprintDefinitions(node, {})[1].assets!.skills).toEqual(['registry://skill/acme/readme-helper@^1.0.0']); + }); + + test('unions domains without mutating repository configuration or constructing resources', () => { + const definitions: BlueprintDefinition[] = [ + { id: 'First', repo: 'example/first', networking: { egressAllowlist: ['one.example.com', '*.example.org'] } }, + { id: 'Second', repo: 'example/second', networking: { egressAllowlist: ['one.example.com', 'two.example.com'] } }, + { id: 'Third', repo: 'example/third' }, + ]; + const before = structuredClone(definitions); + const domains = blueprintEgressDomains(definitions); + expect(domains).toEqual(['one.example.com', '*.example.org', 'two.example.com']); + domains.push('new.example.com'); + expect(definitions).toEqual(before); + expect(blueprintEgressDomains([])).toEqual([]); + }); +}); diff --git a/cdk/test/constructs/agent-session-role.test.ts b/cdk/test/constructs/agent-session-role.test.ts index 676b34f65..c4022350d 100644 --- a/cdk/test/constructs/agent-session-role.test.ts +++ b/cdk/test/constructs/agent-session-role.test.ts @@ -18,11 +18,12 @@ */ import * as bedrock from '@aws-cdk/aws-bedrock-alpha'; -import { App, Stack } from 'aws-cdk-lib'; -import { Template, Match } from 'aws-cdk-lib/assertions'; +import { App, Aspects, Stack } from 'aws-cdk-lib'; +import { Annotations, Template, Match } from 'aws-cdk-lib/assertions'; import * as dynamodb from 'aws-cdk-lib/aws-dynamodb'; import * as iam from 'aws-cdk-lib/aws-iam'; import * as s3 from 'aws-cdk-lib/aws-s3'; +import { AwsSolutionsChecks } from 'cdk-nag'; import { AgentSessionRole } from '../../src/constructs/agent-session-role'; function createStack() { @@ -280,3 +281,112 @@ describe('AgentSessionRole construct', () => { expect(stsGrant).toBeDefined(); }); }); + +describe('deferred compute-role binding', () => { + function fixture() { + const app = new App(); + const stack = new Stack(app, 'Deferred'); + const session = new AgentSessionRole(stack, 'Session', { + deferComputeRoleBinding: true, + taskScopedTables: [], + traceArtifactsBucket: new s3.Bucket(stack, 'Traces'), + attachmentsBucket: new s3.Bucket(stack, 'Attachments'), + }); + return { app, stack, session }; + } + test('rejects an unbound role before synthesis', () => { + const { app } = fixture(); + expect(() => app.synth()).toThrow(/admitted compute role/); + }); + test('uses the admitted role for initial trust and grants AssumeRole with tags', () => { + const { stack, session } = fixture(); + const role = new iam.Role(stack, 'Selected', { assumedBy: new iam.ServicePrincipal('ecs-tasks.amazonaws.com') }); + session.admitComputeRole(role); + session.admitComputeRole(role); + const template = Template.fromStack(stack); + const trust = Object.entries(template.findResources('AWS::IAM::Role')).find(([id]) => id.startsWith('SessionRole'))![1]; + expect(JSON.stringify(trust.Properties.AssumeRolePolicyDocument)).toContain('Selected'); + template.hasResourceProperties('AWS::IAM::Policy', { + PolicyDocument: { + Statement: Match.arrayWith([ + Match.objectLike({ Action: ['sts:AssumeRole', 'sts:TagSession'] }), + ]), + }, + }); + }); +}); + +describe.each([false, true])('session-role IAM audit with concrete environment %p', concreteEnvironment => { + function fixture(addUnrelatedWildcards: boolean) { + const stack = new Stack(new App(), 'SessionAudit', { + ...(concreteEnvironment ? { env: { account: '123456789012', region: 'us-east-1' } } : {}), + }); + Aspects.of(stack).add(new AwsSolutionsChecks()); + const computeRole = new iam.Role(stack, 'Compute', { + assumedBy: new iam.ServicePrincipal('bedrock-agentcore.amazonaws.com'), + }); + // Let CDK split the actual model grants during synthesis. Do not create an + // overflow policy in the fixture: the late creation caused the regression. + const invokableModels = Array.from({ length: 16 }, (_, index) => { + const model = new bedrock.BedrockFoundationModel(`anthropic.session-audit-${index}-v1:0`, { + supportsCrossRegion: true, + }); + return [model, bedrock.CrossRegionInferenceProfile.fromConfig({ + geoRegion: bedrock.CrossRegionInferenceProfileRegion.GLOBAL, model, + })]; + }).flat(); + const session = new AgentSessionRole(stack, 'Session', { + assumingRoles: [computeRole], + taskScopedTables: [], + traceArtifactsBucket: new s3.Bucket(stack, 'Traces'), + attachmentsBucket: s3.Bucket.fromBucketArn(stack, 'Attachments', 'arn:aws:s3:::session-audit-attachments'), + invokableModels, + }); + if (addUnrelatedWildcards) { + session.role.addToPrincipalPolicy(new iam.PolicyStatement({ + actions: ['s3:*'], resources: ['*'], + })); + session.role.addToPrincipalPolicy(new iam.PolicyStatement({ + actions: ['bedrock:InvokeModel'], + resources: [ + 'arn:aws:bedrock:*::foundation-model/unrelated-*', + 'arn:*:bedrock:*::foundation-model/unrelated-literal', + ], + })); + session.role.addToPrincipalPolicy(new iam.PolicyStatement({ + actions: ['s3:GetObject'], resources: ['arn:aws:s3:::session-audit-attachments/attachments/*'], + })); + } + const template = Template.fromStack(stack); + const errors = Annotations.fromStack(stack).findError('*', Match.stringLikeRegexp('AwsSolutions-IAM5')) + .filter(finding => finding.id.includes('/Session/')) + .map(finding => String(finding.entry.data)); + return { template, errors }; + } + + let clean: ReturnType; + let unrelated: ReturnType; + beforeAll(() => { + clean = fixture(false); + unrelated = fixture(true); + }); + + test('audits tenant prefixes and explicit model grants through lazy policy overflow', () => { + const policies = clean.template.findResources('AWS::IAM::ManagedPolicy'); + expect(Object.keys(policies).some(id => id.includes('SessionRoleOverflowPolicy'))).toBe(true); + expect(clean.errors).toEqual([]); + }); + + test('still reports wildcard actions, all-resource grants, model patterns and unscoped S3 prefixes', () => { + expect(unrelated.errors).toHaveLength(5); + for (const finding of [ + '[Action::s3:*]', + '[Resource::*]', + '[Resource::arn:aws:bedrock:*::foundation-model/unrelated-*]', + '[Resource::arn:*:bedrock:*::foundation-model/unrelated-literal]', + '[Resource::arn:aws:s3:::session-audit-attachments/attachments/*]', + ]) { + expect(unrelated.errors.some(error => error.includes(finding))).toBe(true); + } + }); +}); diff --git a/cdk/test/constructs/iam-grant-audit.test.ts b/cdk/test/constructs/iam-grant-audit.test.ts new file mode 100644 index 000000000..f95e5e5c8 --- /dev/null +++ b/cdk/test/constructs/iam-grant-audit.test.ts @@ -0,0 +1,105 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import { App, Aspects, Stack } from 'aws-cdk-lib'; +import { Annotations, Match, Template } from 'aws-cdk-lib/assertions'; +import * as dynamodb from 'aws-cdk-lib/aws-dynamodb'; +import * as iam from 'aws-cdk-lib/aws-iam'; +import { AwsSolutionsChecks } from 'cdk-nag'; +import { LinearIdentityVault } from '../../src/constructs/linear-identity-vault'; +import { ToolGateway } from '../../src/constructs/tool-gateway'; + +const unscopedLinearResources = [ + 'arn:*:bedrock-agentcore:us-east-1:123456789012:token-vault/default/oauth2credentialprovider/bgagent-linear-oauth-*', + 'arn:aws:bedrock-agentcore:*:123456789012:token-vault/default/oauth2credentialprovider/bgagent-linear-oauth-*', + 'arn:aws:bedrock-agentcore:us-east-1:*:token-vault/default/oauth2credentialprovider/bgagent-linear-oauth-*', + 'arn:*:secretsmanager:us-east-1:123456789012:secret:bedrock-agentcore-identity!default/oauth2/bgagent-linear-oauth-*', + 'arn:aws:secretsmanager:*:123456789012:secret:bedrock-agentcore-identity!default/oauth2/bgagent-linear-oauth-*', + 'arn:aws:secretsmanager:us-east-1:*:secret:bedrock-agentcore-identity!default/oauth2/bgagent-linear-oauth-*', +]; + +function fixture(addUnrelatedWildcard: boolean, concreteEnvironment: boolean) { + const stack = new Stack(new App(), 'Audit', concreteEnvironment + ? { env: { account: '123456789012', region: 'us-east-1' } } : {}); + const table = new dynamodb.Table(stack, 'Repos', { partitionKey: { name: 'repo', type: dynamodb.AttributeType.STRING } }); + const gateway = new ToolGateway(stack, 'Gateway', { repoTable: table }); + const vault = new LinearIdentityVault(stack, 'Vault', { + workloadName: 'linear-audit', allowedReturnUrls: ['http://localhost/callback'], + }); + const consumer = new iam.Role(stack, 'Consumer', { assumedBy: new iam.ServicePrincipal('lambda.amazonaws.com') }); + // Force CDK to create overflow policies during prepare, after the grant helper + // runs. Distinct conditions prevent statement merging from hiding the split. + for (let index = 0; index < 60; index++) { + consumer.addToPrincipalPolicy(new iam.PolicyStatement({ + actions: ['s3:GetObject'], + resources: [`arn:aws:s3:::fixture-${index}-${'x'.repeat(100)}/object`], + conditions: { StringEquals: { 'aws:ResourceTag/fixture': String(index) } }, + })); + } + vault.grantMintToken(consumer); + if (addUnrelatedWildcard) { + consumer.addToPrincipalPolicy(new iam.PolicyStatement({ + actions: ['secretsmanager:GetSecretValue'], + resources: ['arn:aws:secretsmanager:us-east-1:123456789012:secret:unrelated-*'], + })); + for (const resource of unscopedLinearResources) { + consumer.addToPrincipalPolicy(new iam.PolicyStatement({ + actions: [resource.includes(':secretsmanager:') ? 'secretsmanager:GetSecretValue' : 'bedrock-agentcore:GetResourceOauth2Token'], + resources: [resource], + })); + } + gateway.gateway.role.addToPrincipalPolicy(new iam.PolicyStatement({ + actions: ['lambda:InvokeFunction'], resources: ['*'], + })); + } + Aspects.of(stack).add(new AwsSolutionsChecks()); + const template = Template.fromStack(stack); + const errors = Annotations.fromStack(stack).findError('*', Match.stringLikeRegexp('AwsSolutions-IAM5')) + .filter(error => error.id.includes('/Consumer/') || error.id.includes('/Gateway/Gateway/ServiceRole/')) + .map(error => String(error.entry.data)); + return { template, errors }; +} + +describe.each([false, true])('grant-specific IAM audit exceptions (concrete environment: %s)', concreteEnvironment => { + let clean: ReturnType; + let unrelated: ReturnType; + beforeAll(() => { + clean = fixture(false, concreteEnvironment); + unrelated = fixture(true, concreteEnvironment); + }); + + test('known Lambda/version and Linear-prefix grants pass, including lazy overflow policies', () => { + const policies = clean.template.findResources('AWS::IAM::ManagedPolicy'); + expect(Object.keys(policies).some(id => id.includes('ConsumerOverflowPolicy'))).toBe(true); + expect(clean.errors).toEqual([]); + expect(JSON.stringify(policies)).toContain('bgagent-linear-oauth-'); + }); + + test('the same principals still fail for unrelated wildcard resources', () => { + expect(unrelated.errors).toHaveLength(2 + unscopedLinearResources.length); + expect(unrelated.errors.some(error => error.includes('secret:unrelated-*'))).toBe(true); + expect(unrelated.errors.some(error => error.includes('[Resource::*]'))).toBe(true); + }); + + test('Linear prefixes do not hide wildcard partitions, regions or accounts', () => { + for (const resource of unscopedLinearResources) { + expect(unrelated.errors.some(error => error.includes(`[Resource::${resource}]`))).toBe(true); + } + }); +}); diff --git a/cdk/test/constructs/solution-ua-aspect.test.ts b/cdk/test/constructs/solution-ua-aspect.test.ts index 8d68668ca..be7241706 100644 --- a/cdk/test/constructs/solution-ua-aspect.test.ts +++ b/cdk/test/constructs/solution-ua-aspect.test.ts @@ -17,7 +17,7 @@ * SOFTWARE. */ -import { App, Aspects, Stack } from 'aws-cdk-lib'; +import { App, Aspects, CfnResource, Stack } from 'aws-cdk-lib'; import { Template } from 'aws-cdk-lib/assertions'; import * as lambda from 'aws-cdk-lib/aws-lambda'; import { buildAppId, ComponentUaAspect, SolutionUaAspect } from '../../src/constructs/solution-ua-aspect'; @@ -100,6 +100,32 @@ describe('SolutionUaAspect', () => { expect(vars.AWS_SDK_UA_APP_ID).toBe('uksb-wt64nei4u6#dev'); }); + test.each(['uksb-wt64nei4u6#dev', undefined])('handles raw provider functions and opt-out (%p)', appId => { + const stack = new Stack(new App(), 'RawProvider'); + new lambda.CfnFunction(stack, 'Handler', { + code: { zipFile: 'exports.handler = async () => {};' }, + handler: 'index.handler', + runtime: 'nodejs22.x', + role: 'arn:aws:iam::123456789012:role/provider', + environment: { variables: { EXISTING: 'preserved' } }, + }); + new CfnResource(stack, 'CoreProvider', { + type: 'AWS::Lambda::Function', + properties: { + Code: { ZipFile: 'exports.handler = async () => {};' }, + Handler: 'index.handler', + Runtime: 'nodejs22.x', + Role: 'arn:aws:iam::123456789012:role/provider', + Environment: { Variables: { EXISTING: 'preserved' } }, + }, + }); + Aspects.of(stack).add(new SolutionUaAspect(appId)); + const template = Template.fromStack(stack); + template.resourcePropertiesCountIs('AWS::Lambda::Function', { + Environment: { Variables: { EXISTING: 'preserved', ...(appId ? { AWS_SDK_UA_APP_ID: appId } : {}) } }, + }, 2); + }); + test('undefined appId (opt-out) sets nothing', () => { const vars = envVarsOfFirstFunction((s) => Aspects.of(s).add(new SolutionUaAspect(undefined))); expect(vars.AWS_SDK_UA_APP_ID).toBeUndefined(); diff --git a/cdk/test/handlers/orchestrate-task.test.ts b/cdk/test/handlers/orchestrate-task.test.ts index be9901081..b289944da 100644 --- a/cdk/test/handlers/orchestrate-task.test.ts +++ b/cdk/test/handlers/orchestrate-task.test.ts @@ -1625,3 +1625,21 @@ describe('finalizeTask — memory fallback', () => { expect(mockWriteMinimalEpisode).toHaveBeenCalled(); }); }); + +describe('exclusive deployment routing', () => { + const original = process.env.DEPLOYED_COMPUTE_TYPE; + afterEach(() => { + if (original === undefined) delete process.env.DEPLOYED_COMPUTE_TYPE; + else process.env.DEPLOYED_COMPUTE_TYPE = original; + }); + test.each(['agentcore', 'ecs', 'lambda-microvm'])('inherits %s for unpinned and repo-less tasks', async backend => { + process.env.DEPLOYED_COMPUTE_TYPE = backend; + expect((await loadBlueprintConfig(baseTask as any)).compute_type).toBe(backend); + expect((await loadBlueprintConfig({ ...baseTask, repo: undefined } as any)).compute_type).toBe(backend); + }); + test('rejects a stored backend override that is no longer deployed', async () => { + process.env.DEPLOYED_COMPUTE_TYPE = 'ecs'; + mockLoadRepoConfig.mockResolvedValueOnce({ compute_type: 'agentcore' }); + await expect(loadBlueprintConfig(baseTask as any)).rejects.toThrow(/is not deployed/); + }); +}); diff --git a/cdk/test/handlers/shared/compute-backend.test.ts b/cdk/test/handlers/shared/compute-backend.test.ts new file mode 100644 index 000000000..ca2aa98c2 --- /dev/null +++ b/cdk/test/handlers/shared/compute-backend.test.ts @@ -0,0 +1,62 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import { resolveComputeBackend, resolveComputeBackends, resolveRepositoryBackend } from '../../../src/handlers/shared/compute-backend'; + +test('defaults to AgentCore only when the selector is absent', () => { + expect(resolveComputeBackend()).toBe('agentcore'); +}); +test.each(['', 'fargate', 'AGENTCORE', null, false, ['ecs']])('rejects invalid selector %p', value => { + expect(() => resolveComputeBackend(value)).toThrow(/compute_type must be/); +}); +test.each(['agentcore', 'ecs', 'lambda-microvm'])('inherits and enforces deployed backend %s', backend => { + expect(resolveRepositoryBackend(undefined, backend)).toBe(backend); + expect(resolveRepositoryBackend(backend, backend)).toBe(backend); + const other = backend === 'agentcore' ? 'ecs' : 'agentcore'; + expect(() => resolveRepositoryBackend(other, backend)).toThrow(/is not deployed/); +}); +test('preserves legacy routing without a deployed selector', () => { + expect(resolveRepositoryBackend('ecs', undefined)).toBe('ecs'); +}); +test.each([ + [undefined, undefined, ['agentcore']], + [undefined, 'agentcore', ['agentcore']], + [undefined, 'ecs', ['agentcore', 'ecs']], + [undefined, 'lambda-microvm', ['agentcore', 'lambda-microvm']], + ['lambda-microvm', 'ecs', ['lambda-microvm']], + ['agentcore, lambda-microvm,agentcore', undefined, ['agentcore', 'lambda-microvm']], + [['ecs', 'agentcore'], undefined, ['ecs', 'agentcore']], +])('resolves compute_types %p with legacy compute_type %p', (list, legacy, expected) => { + expect(resolveComputeBackends(list, legacy)).toEqual(expected); +}); +test.each(['agentcore,fargate', ',', '', ' ', 'ecs,', null, false, 0, {}, [], [['ecs']], ['ecs', null]])('rejects invalid compute_types %p', value => { + expect(() => resolveComputeBackends(value)).toThrow(/compute_type must be|compute_types must be/); +}); +test('enforces membership on additive deployments and defaults to the first backend', () => { + expect(resolveRepositoryBackend(undefined, 'agentcore,lambda-microvm')).toBe('agentcore'); + expect(resolveRepositoryBackend('lambda-microvm', 'agentcore,lambda-microvm')).toBe('lambda-microvm'); + expect(() => resolveRepositoryBackend('ecs', 'agentcore,lambda-microvm')).toThrow(/deploys only 'agentcore, lambda-microvm'/); + expect(resolveRepositoryBackend(undefined, 'ecs,agentcore')).toBe('ecs'); + expect(resolveRepositoryBackend(undefined, 'lambda-microvm,ecs')).toBe('lambda-microvm'); + expect(() => resolveRepositoryBackend('agentcore', 'ecs,lambda-microvm')).toThrow(/is not deployed/); +}); + +test('fails closed on a blank deployed selector instead of inventing an AgentCore deployment', () => { + expect(() => resolveRepositoryBackend(undefined, '')).toThrow(/compute_types must be/); +}); diff --git a/cdk/test/handlers/shared/strategies/lambda-microvm-strategy.test.ts b/cdk/test/handlers/shared/strategies/lambda-microvm-strategy.test.ts index 9d94f7cae..087758077 100644 --- a/cdk/test/handlers/shared/strategies/lambda-microvm-strategy.test.ts +++ b/cdk/test/handlers/shared/strategies/lambda-microvm-strategy.test.ts @@ -73,6 +73,9 @@ for (const optional of [ 'AWS_SDK_UA_APP_ID', 'ANTHROPIC_DEFAULT_HAIKU_MODEL', 'ANTHROPIC_MODEL', + 'ABCA_TOOL_GATEWAY_URL', + 'LINEAR_VAULT_ENABLED', + 'LINEAR_WORKLOAD_IDENTITY_NAME', ]) { delete process.env[optional]; } @@ -1237,6 +1240,9 @@ describe('buildMicrovmPlatformConfig — the MicroVM substitute for a deploy-tim AWS_SDK_UA_APP_ID: 'uksb-wt64nei4u6#backgroundagent-dev', ANTHROPIC_DEFAULT_HAIKU_MODEL: 'us.anthropic.claude-haiku-4-5-20251001-v1:0', ANTHROPIC_MODEL: 'us.anthropic.claude-opus-5', + ABCA_TOOL_GATEWAY_URL: 'https://gateway.example/mcp', + LINEAR_VAULT_ENABLED: 'true', + LINEAR_WORKLOAD_IDENTITY_NAME: 'abca_linear_oauth_test', }; test('sources the wire key allow-list from the cross-language contract', () => { @@ -1276,8 +1282,11 @@ describe('buildMicrovmPlatformConfig — the MicroVM substitute for a deploy-tim // `agent/src/config.py` — wrong on any deployment whose geography is not // `global`, and wrong in a way that surfaces only as AccessDenied at turn 0. 'anthropic_model', + 'tool_gateway_url', + 'linear_vault_enabled', + 'linear_workload_identity_name', ]); - expect(MICROVM_PLATFORM_CONFIG_KEYS).toHaveLength(14); + expect(MICROVM_PLATFORM_CONFIG_KEYS).toHaveLength(17); // snake_case on the wire, matching every other key in the /run envelope. for (const key of MICROVM_PLATFORM_CONFIG_KEYS) { expect(key).toMatch(/^[a-z][a-z0-9_]*$/); @@ -1298,7 +1307,7 @@ describe('buildMicrovmPlatformConfig — the MicroVM substitute for a deploy-tim } }); - test('emits all fourteen keys, in declaration order, from a full environment', () => { + test('emits all seventeen keys, in declaration order, from a full environment', () => { const config = buildMicrovmPlatformConfig(FULL_ENV); expect(Object.keys(config)).toEqual([...MICROVM_PLATFORM_CONFIG_KEYS]); expect(config.task_table_name).toBe('tasks'); @@ -1310,6 +1319,9 @@ describe('buildMicrovmPlatformConfig — the MicroVM substitute for a deploy-tim // a `us` deployment a missing value is not an error anywhere — it is a wrong // model that the IAM grant does not cover, surfacing as AccessDenied at turn 0. expect(config.anthropic_model).toBe('us.anthropic.claude-opus-5'); + expect(config.tool_gateway_url).toBe('https://gateway.example/mcp'); + expect(config.linear_vault_enabled).toBe('true'); + expect(config.linear_workload_identity_name).toBe('abca_linear_oauth_test'); }); test('OMITS optional keys the orchestrator does not carry (no `undefined` placeholders)', () => { @@ -1445,6 +1457,9 @@ describe('buildMicrovmPlatformConfig — the MicroVM substitute for a deploy-tim 'ARTIFACTS_BUCKET_NAME', 'TRACE_ARTIFACTS_BUCKET_NAME', 'LINEAR_OAUTH_SECRET_ARN', 'JIRA_OAUTH_SECRET_ARN', 'AWS_SDK_UA_APP_ID', 'ANTHROPIC_DEFAULT_HAIKU_MODEL', 'ANTHROPIC_MODEL', + 'ABCA_TOOL_GATEWAY_URL', + 'LINEAR_VAULT_ENABLED', + 'LINEAR_WORKLOAD_IDENTITY_NAME', ]) { delete env[optional]; } diff --git a/cdk/test/main.test.ts b/cdk/test/main.test.ts index d40ac5d42..7d3e7be32 100644 --- a/cdk/test/main.test.ts +++ b/cdk/test/main.test.ts @@ -18,7 +18,7 @@ */ import * as fs from 'fs'; -import { App, Stack } from 'aws-cdk-lib'; +import { App, CfnResource, NestedStack, STACK_RESOURCE_LIMIT_CONTEXT, Stack } from 'aws-cdk-lib'; import { Annotations, Match, Template } from 'aws-cdk-lib/assertions'; import { AGENTCORE_AZS_CONTEXT_KEY, @@ -129,6 +129,18 @@ describe('buildApp — AgentCore AZ wiring', () => { expect(errors[0].entry.data).toContain('Could not resolve AgentCore-supported availability zones'); }); + it('attaches AZ lookup errors to the network stack in the split topology', async () => { + const built = await app({ + appProps: { context: { networkTopology: 'split' } }, + describeAzs: async () => { throw new Error('AccessDeniedException'); }, + }); + const assembly = built.synth(); + const errors = assembly.getStackByName(`${STACK_NAME}-network`).messages.filter(message => message.level === 'error'); + expect(errors).toHaveLength(1); + expect(errors[0].entry.data).toContain('Could not resolve AgentCore-supported availability zones'); + expect(assembly.getStackByName(STACK_NAME).messages.filter(message => message.level === 'error')).toEqual([]); + }); + it('surfaces the unpinned env-agnostic case as a stack-artifact WARNING', async () => { const built = await buildApp({ account: undefined, region: undefined }); Annotations.fromStack(stackOf(built)).hasWarning( @@ -155,20 +167,24 @@ describe('buildApp — AgentCore AZ wiring', () => { }); }); -describe('buildApp — CloudFormation template-body budget within 80% 1Mb budget', () => { - // CloudFormation caps a template body at 1 MB; CDK checks against - // `TEMPLATE_BODY_MAXIMUM_SIZE = 1e6` and only - // raises an `@aws-cdk/core:Stack.templateSize` *warning* — so this ceiling fails - // **open**. A template can grow past it, synthesize cleanly, and fail at deploy. - // These assertions read the emitted artifact rather than `Template.fromStack`, - // because indentation is the thing under test and `Template` has already parsed - // it away. - const CDK_TEMPLATE_BODY_MAXIMUM_SIZE = 1_000_000; - // Budget to the point CDK starts warning, so a regression trips a readable assertion - // instead of silently riding the warning band up to the hard limit. - const WARNING_THRESHOLD = 0.8; - const TEMPLATE_BODY_BUDGET = CDK_TEMPLATE_BODY_MAXIMUM_SIZE * WARNING_THRESHOLD; +describe('buildApp — deferred migration settings', () => { + test.each([ + { blueprintProvisioning: 'legacy' }, + { blueprintProvisioning: 'prepare' }, + { blueprintProvisioning: 'adopt' }, + { blueprintProvisioning: 'managed' }, + { guardrailVersionMigration: { logicalId: 'ExistingVersion', configurationHash: 'a'.repeat(64) } }, + ])('refuses to ignore experimental context %j before resolving AWS inputs', async context => { + const lookup = jest.fn(okZones); + await expect(app({ describeAzs: lookup, appProps: { context } })) + .rejects.toThrow('needs a separate recovery plan; do not drop this setting and deploy over it'); + expect(lookup).not.toHaveBeenCalled(); + }); +}); +describe('buildApp — compact template output', () => { + // Read the emitted artifact: parsing with Template.fromStack loses indentation. + // synthesis/deployment.test.ts owns byte budgets across the full profile product. let templateText: string; beforeAll(async () => { @@ -184,8 +200,43 @@ describe('buildApp — CloudFormation template-body budget within 80% 1Mb budget // not need re-baselining every time a resource is added. expect(templateText).not.toContain('\n '); }); +}); + +describe('buildApp — production resource ceiling', () => { + describe.each(['parent', 'nested'] as const)('%s template', kind => { + test.each([490, 491])('enforces the boundary at %i resources without the census', async count => { + const built = await app(); + const parent = new Stack(built, 'BudgetProbe', { analyticsReporting: false }); + const scope = kind === 'nested' ? new NestedStack(parent, 'Child') : parent; + for (let i = 0; i < count; i++) { + new CfnResource(scope, `Handle${i}`, { type: 'AWS::CloudFormation::WaitConditionHandle' }); + } + if (count === 490) { + expect(() => built.synth()).not.toThrow(); + } else { + expect(() => built.synth()).toThrow(/491 is greater than allowed maximum of 490:/); + } + }); + }); - it('stays inside the template-body budget', () => { - expect(Buffer.byteLength(templateText, 'utf8')).toBeLessThan(TEMPLATE_BODY_BUDGET); + test.each([480, '480'])('honors a stricter numeric or CLI-string ceiling: %s', async limit => { + const built = await app({ appProps: { context: { [STACK_RESOURCE_LIMIT_CONTEXT]: limit } } }); + const probe = new Stack(built, 'BudgetProbe', { analyticsReporting: false }); + for (let i = 0; i < 481; i++) { + new CfnResource(probe, `Handle${i}`, { type: 'AWS::CloudFormation::WaitConditionHandle' }); + } + expect(() => built.synth()).toThrow(/481 is greater than allowed maximum of 480:/); }); + + test.each([500, '500', 0, -1, 490.5, null, true, 'invalid', ''])( + 'rejects an invalid or weakened resource ceiling before resolving AWS inputs: %s', + async limit => { + const lookup = jest.fn(okZones); + await expect(app({ + describeAzs: lookup, + appProps: { context: { [STACK_RESOURCE_LIMIT_CONTEXT]: limit } }, + })).rejects.toThrow(`Context '${STACK_RESOURCE_LIMIT_CONTEXT}' must be an integer from 1 to 490`); + expect(lookup).not.toHaveBeenCalled(); + }, + ); }); diff --git a/cdk/test/stacks/agent.test.ts b/cdk/test/stacks/agent.test.ts index 8c32cefe8..dedd325e0 100644 --- a/cdk/test/stacks/agent.test.ts +++ b/cdk/test/stacks/agent.test.ts @@ -505,7 +505,13 @@ describe('AgentStack', () => { 'inference-profile/us.anthropic.claude-sonnet-4-6', ]; - const serialized = JSON.stringify(template.findResources('AWS::IAM::Policy')); + // Audit actual policy documents, including overflow, rather than cdk-nag + // metadata whose finding patterns also name foundation-model resources. + const policies = { + ...template.findResources('AWS::IAM::Policy'), + ...template.findResources('AWS::IAM::ManagedPolicy'), + }; + const serialized = JSON.stringify(Object.values(policies).map(policy => policy.Properties.PolicyDocument)); const found = [...new Set( serialized.match(/(?:foundation-model|inference-profile)\/[^"]+/g) ?? [], )].sort(); @@ -1133,41 +1139,14 @@ describe('AgentStack with the ECS substrate gate (--context compute_type=ecs)', let template: Template; beforeAll(() => { - // Deploying with the gate on provisions the Fargate substrate alongside the - // always-present AgentCore runtime; the ComputeSubstrate output flips to 'ecs'. - const app = new App({ context: { compute_type: 'ecs' } }); + // Selecting ECS provisions the Fargate backend and emits ComputeSubstrate=ecs. + const app = new App({ context: { compute_types: 'ecs' } }); const stack = new AgentStack(app, 'TestAgentStackEcs', { env: { account: '123456789012', region: 'us-east-1' }, }); template = Template.fromStack(stack); }); - /** - * CloudFormation refuses a template over 1 MB, and refuses it at CHANGESET CREATION — - * after synth succeeds and every asset is pushed. The message names no resource, and - * the stack's own status stays at whatever the previous deploy left, so checking stack - * status instead of the deploy's exit code reads as success. - * - * Asserted on the ECS template because that is the substrate deployments use, and it is - * the larger of the two: 894,261 bytes here versus 858,062 for the default at the time - * of writing. Measured the way the CDK CLI WRITES the template (`null, 2`), which is how - * CloudFormation counts it — compact serialization of the same template is ~300 KB - * smaller, so a budget checked against compact bytes passes while the deploy fails. - * - * These in-test figures are lower than what `cdk synth` writes to disk, because the CLI - * resolves asset hashes and account/region tokens that `Template.fromStack` leaves - * symbolic. The budget is therefore a trend guard on the relative number, not a - * prediction of the byte count CloudFormation will receive. - * - * Reuses the template this describe already synthesizes; no extra synth. - */ - test('stays inside a deployable template budget (CloudFormation hard-fails at 1 MB)', () => { - const bytes = Buffer.byteLength(JSON.stringify(template.toJSON(), null, 2), 'utf8'); - expect(bytes).toBeLessThan(1_000_000); - // 5% under, so this fires while there is still room to land the change that trips it. - expect(bytes).toBeLessThan(950_000); - }); - test('provisions an ECS cluster + both Fargate task definitions (build + planning)', () => { template.resourceCountIs('AWS::ECS::Cluster', 1); // Two task defs — the 64 GB build def and the 8 GB read-only planning def @@ -1211,7 +1190,7 @@ describe('AgentStack with the ECS substrate gate (--context compute_type=ecs)', // This asserts the whole path: context -> resolver -> construct -> template. const app = new App({ context: { - compute_type: 'ecs', + compute_types: 'ecs', ecsBuildTaskCpu: '16384', ecsBuildTaskMemoryMiB: '122880', ecsBuildTaskEphemeralStorageGiB: '100', @@ -1247,7 +1226,7 @@ describe('AgentStack with the Lambda MicroVMs substrate gate (--context compute_ // exists once an image identifier is available. const app = new App({ context: { - compute_type: 'lambda-microvm', + compute_types: 'lambda-microvm', microvm_base_image_arn: BASE_IMAGE_ARN, microvm_base_image_version: '1', }, @@ -1441,7 +1420,7 @@ describe('AgentStack with the Lambda MicroVMs substrate gate (--context compute_ .filter(([id]) => id.includes('LambdaMicrovmComputeExecutionRole')), ); expect(policies).toContain('logs:CreateLogStream'); - expect(policies).toContain('RuntimeApplicationLogGroup'); + expect(policies).toContain('LambdaMicrovmComputeMicrovmLogGroup'); // ...and the orchestrator delivers that group's NAME as LOG_GROUP_NAME. const orchestrator = Object.entries(template.findResources('AWS::Lambda::Function')) @@ -1449,7 +1428,7 @@ describe('AgentStack with the Lambda MicroVMs substrate gate (--context compute_ const logGroupEnv = JSON.stringify( orchestrator[1].Properties.Environment.Variables.LOG_GROUP_NAME, ); - expect(logGroupEnv).toContain('RuntimeApplicationLogGroup'); + expect(logGroupEnv).toContain('LambdaMicrovmComputeMicrovmLogGroup'); }); test('MicroVM resources carry the backend cost-allocation tag', () => { @@ -1472,7 +1451,7 @@ describe('AgentStack with the Lambda MicroVMs substrate gate (--context compute_ beforeAll(() => { const app = new App({ context: { - compute_type: 'lambda-microvm', + compute_types: 'lambda-microvm', microvm_region_override: true, microvm_base_image_arn: BASE_IMAGE_ARN, microvm_base_image_version: '1', @@ -1484,7 +1463,7 @@ describe('AgentStack with the Lambda MicroVMs substrate gate (--context compute_ }); test('fails synth when the stack Region has no Lambda MicroVMs', () => { - const app = new App({ context: { compute_type: 'lambda-microvm' } }); + const app = new App({ context: { compute_types: 'lambda-microvm' } }); expect(() => new AgentStack(app, 'TestAgentStackMicrovmBadRegion', { env: { account: '123456789012', region: 'eu-central-1' }, })).toThrow(/AWS Lambda MicroVMs are not available in eu-central-1/); @@ -1538,7 +1517,7 @@ describe('AgentStack with the MicroVM gate on but no image configured (first dep // but no image yet. Exercises the false branch of the shared // `isLambdaMicrovmImageConfigured` predicate that gates BOTH the // orchestrator's MICROVM_* wiring and the cancel Lambda's grant. - const app = new App({ context: { compute_type: 'lambda-microvm' } }); + const app = new App({ context: { compute_types: 'lambda-microvm' } }); const stack = new AgentStack(app, 'TestAgentStackMicrovmNoImage', { env: { account: '123456789012', region: 'us-east-1' }, }); @@ -1582,7 +1561,7 @@ describe('AgentStack MicroVM image ARN invariant', () => { const configuredSpy = jest.spyOn(lambdaMicrovmCompute, 'isLambdaMicrovmImageConfigured') .mockReturnValue(true); try { - const app = new App({ context: { compute_type: 'lambda-microvm' } }); + const app = new App({ context: { compute_types: 'lambda-microvm' } }); const stack = new AgentStack(app, 'TestAgentStackMicrovmInvariant', { env: { account: '123456789012', region: 'us-east-1' }, }); @@ -1723,7 +1702,7 @@ describe('AgentStack tool-gateway gate (ADR-019 P1)', () => { beforeAll(() => { const app = new App({ - context: { enableToolGateway: true, compute_type: 'ecs' }, + context: { enableToolGateway: true, compute_types: 'ecs' }, }); const stack = new AgentStack(app, 'GatewayEcsStack', { env: { account: '123456789012', region: 'us-east-1' }, @@ -1917,30 +1896,16 @@ describe('AgentStack Linear identity vault gate (#809)', () => { expect([...workloadNames]).toEqual(['abca_linear_oauth_LinearVaultWritersStack']); }); - test('MicroVM + vault is REFUSED by name, not left to the resource counter', () => { - // Pinning a limitation, not a behaviour. The vault IS wired for the MicroVM substrate - // — platform_config carries the workload name and the guest's execution role gets the - // mint grant — but the two cannot be enabled together today: 505 resources against a - // HARD limit of 500 (microvm alone 496, the vault alone 488). Claiming MicroVM support - // without saying so would be false. - // - // The stack refuses the combination itself rather than letting the counter throw, - // because the counter's message is a per-type census that never mentions either flag — - // the operator cannot tell from it what to change. - // - // Reclaiming room means nesting a subsystem. MicroVM (+19 resources) is the cheapest - // candidate and currently deployed nowhere, but nesting it needs the session-role trust - // wiring to stop referencing a child resource (it creates a parent↔child cycle today). - // - // When the room is found, this test should be replaced by a real parity assertion. - const app = new App({ - context: { enableLinearIdentityVault: true, compute_type: 'lambda-microvm' }, - }); - expect(() => Template.fromStack( - new AgentStack(app, 'LinearVaultMicrovmStack', { - env: { account: '123456789012', region: 'us-east-1' }, - }), - )).toThrow(/enableLinearIdentityVault cannot be combined with compute_type=lambda-microvm/); + test('MicroVM + vault fits with only the MicroVM compute backend deployed', () => { + const app = new App({ context: { enableLinearIdentityVault: true, compute_types: 'lambda-microvm' } }); + const template = Template.fromStack(new AgentStack(app, 'LinearVaultMicrovmStack', { + env: { account: '123456789012', region: 'us-east-1' }, + })); + template.resourceCountIs('AWS::BedrockAgentCore::Runtime', 0); + const policy = JSON.stringify(Object.entries(template.findResources('AWS::IAM::Policy')) + .filter(([id]) => id.includes('LambdaMicrovmComputeExecutionRole'))); + expect(policy).toContain('bedrock-agentcore:GetResourceOauth2Token'); + expect(Object.keys(template.toJSON().Resources).length).toBeLessThanOrEqual(500); }); test('the source graph names no Linear-minting handler that is unwired', () => { @@ -1998,66 +1963,6 @@ describe('AgentStack Linear identity vault gate (#809)', () => { }); }); -describe('AgentStack CloudFormation resource budget 500 with cushion', () => { - // The 500-resource limit is a hard, non-adjustable CloudFormation template quota, - // and CDK enforces it by *throwing* `TooManyResourcesInStack` during synth. - // Every deploy-gate cell is covered, not only the widest, so a regression confined - // to one substrate cannot hide behind the others. The gap this closes: no test had - // ever constructed `compute_type` and `enableToolGateway` *together*, so the widest - // cell could exceed the quota — unable to synthesize at all — with CI still green. - // Budget is deliberately below the quota so this fails as a readable assertion with a - // named remedy before synth starts throwing. - const MAX_RESOURCE_BUDGET = 500; - const CUSHION = 10; - const RESOURCE_BUDGET = MAX_RESOURCE_BUDGET - CUSHION; - - // `Template.fromStack` counts one fewer than `cdk synth`, which also emits - // `AWS::CDK::Metadata`. Budget the synthesized number, so add that resource back. - const SYNTH_ONLY_RESOURCES = 1; - - const COMPUTE_TYPES = ['agentcore', 'ecs', 'lambda-microvm']; - const CELLS = COMPUTE_TYPES.flatMap(computeType => - [false, true].map(enableToolGateway => ({ computeType, enableToolGateway })), - ); - - describe.each(CELLS)( - 'compute_type=$computeType enableToolGateway=$enableToolGateway', - ({ computeType, enableToolGateway }) => { - let template: Template; - - beforeAll(() => { - const app = new App({ context: { compute_type: computeType, enableToolGateway } }); - const stack = new AgentStack(app, 'BudgetStack', { - env: { account: '123456789012', region: 'us-east-1' }, - }); - // Throws `TooManyResourcesInStack` if this cell is over the hard quota, so - // reaching the assertions below is itself part of the guard. - template = Template.fromStack(stack); - }); - - test('stays inside the resource budget', () => { - const resourceCount = Object.keys(template.toJSON().Resources ?? {}).length; - expect(resourceCount + SYNTH_ONLY_RESOURCES).toBeLessThanOrEqual(RESOURCE_BUDGET); - }); - - test('emits no Lambda permission for the API Gateway console test-invoke stage', () => { - // Every `LambdaIntegration` in this app passes `allowTestInvoke: false`. Left at - // its default `true`, CDK emits a second `AWS::Lambda::Permission` per method - // scoped to `method.testMethodArn` — removing the API Gateway console's "TEST" - // button, which nothing in this solution invokes, and its extra - // `lambda:InvokeFunction` grant. - // Naming the offending logical IDs makes a regressed call site point straight - // at its own construct. - const offenders = Object.entries(template.findResources('AWS::Lambda::Permission')) - .filter(([, resource]) => JSON.stringify(resource).includes('test-invoke-stage')) - .map(([logicalId]) => logicalId); - - expect(offenders).toEqual([]); - }); - }, - ); -}); - describe('AgentStack Agent Registry gate', () => { test.each([undefined, true, 'true'])( 'enableAgentRegistry=%p includes the registry by default or explicit enablement', diff --git a/cdk/test/stacks/compute-selection.test.ts b/cdk/test/stacks/compute-selection.test.ts new file mode 100644 index 000000000..1237110fb --- /dev/null +++ b/cdk/test/stacks/compute-selection.test.ts @@ -0,0 +1,194 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import { App } from 'aws-cdk-lib'; +import { Template } from 'aws-cdk-lib/assertions'; +import { AgentStack } from '../../src/stacks/agent'; + +describe.each(['agentcore', 'ecs', 'lambda-microvm'])('exclusive %s deployment', backend => { + let template: Template; + beforeAll(() => { + const app = new App({ + context: { + compute_types: backend, + enableToolGateway: true, + enableLinearIdentityVault: true, + ...(backend === 'lambda-microvm' ? { + microvm_image_identifier: 'arn:aws:lambda:us-east-1:123456789012:microvm-image:test-image', + microvm_image_version: '1', + } : {}), + }, + }); + template = Template.fromStack(new AgentStack(app, 'ComputeSelection', { + env: { account: '123456789012', region: 'us-east-1' }, + })); + }); + + test('provisions only the selected compute backend and advertises its default', () => { + template.resourceCountIs('AWS::BedrockAgentCore::Runtime', backend === 'agentcore' ? 1 : 0); + template.resourceCountIs('AWS::ECS::Cluster', backend === 'ecs' ? 1 : 0); + template.resourceCountIs('AWS::Lambda::NetworkConnector', backend === 'lambda-microvm' ? 2 : 0); + template.hasOutput('ComputeSubstrate', { Value: backend }); + template.hasOutput('ComputeDeploymentMode', { Value: 'exclusive' }); + expect(!!template.toJSON().Outputs.RuntimeArn).toBe(backend === 'agentcore'); + template.resourceCountIs('AWS::BedrockAgentCore::Memory', 1); + template.resourceCountIs('AWS::BedrockAgentCore::Gateway', 1); + }); + + test('keeps AgentCore logs owned across backend switches with the existing destroy policy', () => { + const groups = Object.fromEntries(Object.entries(template.findResources('AWS::Logs::LogGroup')) + .filter(([id]) => id.startsWith('RuntimeApplicationLogGroup') || id.startsWith('RuntimeUsageLogGroup')) + .map(([id, resource]) => [id, { + name: resource.Properties.LogGroupName, + retention: resource.Properties.RetentionInDays, + deletion: resource.DeletionPolicy, + replacement: resource.UpdateReplacePolicy, + }])); + expect(groups).toEqual({ + RuntimeApplicationLogGroupCCD512EC: { + name: '/aws/vendedlogs/bedrock-agentcore/runtime/APPLICATION_LOGS/ComputeSelection', + retention: 90, + deletion: 'Delete', + replacement: 'Delete', + }, + RuntimeUsageLogGroup3193D914: { + name: '/aws/vendedlogs/bedrock-agentcore/runtime/USAGE_LOGS/ComputeSelection', + retention: 90, + deletion: 'Delete', + replacement: 'Delete', + }, + }); + }); + + test('allows fixed-name logs to be cleaned up on destroy and failed creation', () => { + const groups = Object.values(template.findResources('AWS::Logs::LogGroup')); + const names = [ + '/aws/bedrock/model-invocation-logs/ComputeSelection', + '/aws/vendedlogs/bedrock-agentcore/runtime/APPLICATION_LOGS/ComputeSelection', + '/aws/vendedlogs/bedrock-agentcore/runtime/USAGE_LOGS/ComputeSelection', + ...(backend === 'lambda-microvm' ? ['/aws/lambda-microvms/ComputeSelection-abca-agent'] : []), + ]; + for (const name of names) { + expect(groups.find(resource => resource.Properties.LogGroupName === name)).toMatchObject({ + DeletionPolicy: 'Delete', + UpdateReplacePolicy: 'Delete', + }); + } + }); + + test('dispatch and cancellation target the selected backend', () => { + const fns = Object.entries(template.findResources('AWS::Lambda::Function')); + const orchestrator = fns.find(([id]) => id.startsWith('TaskOrchestratorOrchestratorFn'))![1]; + const env = orchestrator.Properties.Environment.Variables; + expect(env.DEPLOYED_COMPUTE_TYPE).toBe(backend); + expect(!!env.RUNTIME_ARN).toBe(backend === 'agentcore'); + expect(env.LINEAR_VAULT_ENABLED).toBe('true'); + expect(env.LINEAR_WORKLOAD_IDENTITY_NAME).toBeDefined(); + expect(env.ABCA_TOOL_GATEWAY_URL).toBeDefined(); + expect(!!env.ECS_CLUSTER_ARN).toBe(backend === 'ecs'); + const cancel = fns.find(([id]) => id.startsWith('TaskApiCancelTaskFn'))![1]; + expect(!!cancel.Properties.Environment.Variables.RUNTIME_ARN).toBe(backend === 'agentcore'); + expect(!!cancel.Properties.Environment.Variables.ECS_CLUSTER_ARN).toBe(backend === 'ecs'); + const policies = JSON.stringify(template.findResources('AWS::IAM::Policy')); + expect(policies.includes('bedrock-agentcore:InvokeAgentRuntime')).toBe(backend === 'agentcore'); + expect(policies.includes('bedrock-agentcore:StopRuntimeSession')).toBe(backend === 'agentcore'); + expect(policies.includes('ecs:StopTask')).toBe(backend === 'ecs'); + expect(policies.includes('lambda:TerminateMicrovm')).toBe(backend === 'lambda-microvm'); + }); + + test('session trust contains only the selected compute role', () => { + const role = Object.entries(template.findResources('AWS::IAM::Role')) + .find(([id]) => id.startsWith('AgentSessionRole'))![1]; + const trust = JSON.stringify(role.Properties.AssumeRolePolicyDocument); + expect(trust.includes('RuntimeExecutionRole')).toBe(backend === 'agentcore'); + expect(trust.includes('EcsAgentClusterTaskRole')).toBe(backend === 'ecs'); + expect(trust.includes('LambdaMicrovmComputeExecutionRole')).toBe(backend === 'lambda-microvm'); + expect(trust).toContain('sts:TagSession'); + const prefix = backend === 'agentcore' ? 'RuntimeExecutionRole' : backend === 'ecs' ? 'EcsAgentClusterTaskRole' : 'LambdaMicrovmComputeExecutionRole'; + const policies = JSON.stringify(Object.entries(template.findResources('AWS::IAM::Policy')).filter(([id]) => id.startsWith(prefix))); + expect(policies).toContain('bedrock-agentcore:InvokeGateway'); + expect(policies).toContain('bedrock-agentcore:GetResourceOauth2Token'); + }); +}); + +describe.each([ + { label: 'explicit AgentCore/MicroVM', selector: { compute_types: 'agentcore,lambda-microvm' }, backends: ['agentcore', 'lambda-microvm'] }, + { label: 'legacy MicroVM', selector: { compute_type: 'lambda-microvm' }, backends: ['agentcore', 'lambda-microvm'] }, + { label: 'legacy ECS', selector: { compute_type: 'ecs' }, backends: ['agentcore', 'ecs'] }, + { label: 'ECS default with AgentCore', selector: { compute_types: ['ecs', 'agentcore'] }, backends: ['ecs', 'agentcore'] }, + { label: 'MicroVM default without AgentCore', selector: { compute_types: 'lambda-microvm,ecs' }, backends: ['lambda-microvm', 'ecs'] }, + { label: 'all backends', selector: { compute_types: 'agentcore,ecs,lambda-microvm' }, backends: ['agentcore', 'ecs', 'lambda-microvm'] }, +])('additive deployment from $label', ({ selector, backends }) => { + let template: Template; + beforeAll(() => { + const app = new App({ + context: { + ...selector, + microvm_image_identifier: 'arn:aws:lambda:us-east-1:123456789012:microvm-image:test-image', + microvm_image_version: '1', + }, + }); + template = Template.fromStack(new AgentStack(app, 'ComputeSelection', { + env: { account: '123456789012', region: 'us-east-1' }, + })); + }); + + test('provisions every listed backend and preserves its declared default', () => { + template.resourceCountIs('AWS::BedrockAgentCore::Runtime', backends.includes('agentcore') ? 1 : 0); + template.resourceCountIs('AWS::Lambda::NetworkConnector', backends.includes('lambda-microvm') ? 2 : 0); + template.resourceCountIs('AWS::ECS::Cluster', backends.includes('ecs') ? 1 : 0); + // Existing CLIs parse a comma list here on non-exclusive stacks. + template.hasOutput('ComputeSubstrate', { Value: backends.join(',') }); + template.hasOutput('ComputeTypes', { Value: backends.join(',') }); + template.hasOutput('ComputeDeploymentMode', { Value: 'additive' }); + const orchestrator = Object.entries(template.findResources('AWS::Lambda::Function')) + .find(([id]) => id.startsWith('TaskOrchestratorOrchestratorFn'))![1]; + const env = orchestrator.Properties.Environment.Variables; + expect(env.DEPLOYED_COMPUTE_TYPE).toBe(backends.join(',')); + expect(!!env.RUNTIME_ARN).toBe(backends.includes('agentcore')); + expect(!!env.ECS_CLUSTER_ARN).toBe(backends.includes('ecs')); + }); + + test('session trust admits every deployed compute role', () => { + const role = Object.entries(template.findResources('AWS::IAM::Role')) + .find(([id]) => id.startsWith('AgentSessionRole'))![1]; + const trust = JSON.stringify(role.Properties.AssumeRolePolicyDocument); + expect(trust.includes('RuntimeExecutionRole')).toBe(backends.includes('agentcore')); + expect(trust.includes('EcsAgentClusterTaskRole')).toBe(backends.includes('ecs')); + expect(trust.includes('LambdaMicrovmComputeExecutionRole')).toBe(backends.includes('lambda-microvm')); + }); + + test('grants Linear and Jira OAuth reads to every deployed compute role', () => { + const prefixes: Record = { + 'agentcore': 'RuntimeExecutionRole', + 'ecs': 'EcsAgentClusterTaskRole', + 'lambda-microvm': 'LambdaMicrovmComputeExecutionRole', + }; + const policies = { + ...template.findResources('AWS::IAM::Policy'), + ...template.findResources('AWS::IAM::ManagedPolicy'), + }; + for (const backend of backends) { + const grants = JSON.stringify(Object.entries(policies).filter(([id]) => id.startsWith(prefixes[backend]))); + expect(grants).toContain('secretsmanager:GetSecretValue'); + expect(grants).toContain('bgagent-linear-oauth-*'); + expect(grants).toContain('bgagent-jira-oauth-*'); + } + }); +}); diff --git a/cdk/test/stacks/network.test.ts b/cdk/test/stacks/network.test.ts new file mode 100644 index 000000000..7e2397ad4 --- /dev/null +++ b/cdk/test/stacks/network.test.ts @@ -0,0 +1,371 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import { mkdtempSync, readFileSync, rmSync } from 'node:fs'; +import { tmpdir } from 'node:os'; +import * as path from 'node:path'; +import { BlueprintDefinition } from '../../src/blueprints/definitions'; +import { resolveNetworkReservedAzs } from '../../src/constructs/agent-vpc'; +import { AGENTCORE_AZS_CONTEXT_KEY } from '../../src/constructs/agentcore-azs'; +import { buildApp } from '../../src/main'; +import { NetworkTopology, resolveNetworkTopology } from '../../src/stacks/network'; +import { AssemblyCensus, inspectAssembly } from '../../src/synthesis/assembly'; +import { FIXTURE, STRUCTURAL_CONTEXT } from '../../src/synthesis/profiles'; + +const APP_NAME = 'backgroundagent-dev'; +const NETWORK_NAME = `${APP_NAME}-network`; +const BLUEPRINTS: readonly BlueprintDefinition[] = [ + { id: 'AgentPluginsBlueprint', repo: 'example/plugins', networking: { egressAllowlist: ['packages.example.com'] } }, + { + id: 'ForkBlueprint', + repo: 'example/fork', + networking: { egressAllowlist: ['packages.example.com', '*.internal.example.org'] }, + }, +]; + +type TemplateJson = Record; +interface Deployment { + readonly directory: string; + readonly census: AssemblyCensus; + readonly application: TemplateJson; + readonly network?: TemplateJson; +} + +function withoutMetadata(resource: TemplateJson): TemplateJson { + const { Metadata: _metadata, ...definition } = resource; + return definition; +} + +function isNetworkResource(id: string): boolean { + return id.startsWith('AgentVpc') || id.startsWith('DnsFirewall'); +} + +function importedExports(template: TemplateJson): Set { + const imports = new Set(); + function visit(value: any): void { + if (!value || typeof value !== 'object') return; + if (typeof value['Fn::ImportValue'] === 'string') imports.add(value['Fn::ImportValue']); + for (const child of Object.values(value)) visit(child); + } + visit(template); + return imports; +} + +describe('network topology selection', () => { + test('defaults to the existing inline ownership', () => { + expect(resolveNetworkTopology(undefined)).toBe('inline'); + expect(resolveNetworkTopology('inline')).toBe('inline'); + expect(resolveNetworkTopology('split')).toBe('split'); + }); + + test.each([[undefined, 0], [0, 0], ['0', 0], [1, 1], ['1', 1], [6, 6]])( + 'accepts reserved AZ slots %p as %p', + (value, expected) => { expect(resolveNetworkReservedAzs(value)).toBe(expected); }, + ); + + test.each(['', ' ', 'typo', true, false, null, -1, 0.5, 7, Infinity])( + 'rejects invalid reserved AZ slots %p', + value => { expect(() => resolveNetworkReservedAzs(value)).toThrow('networkReservedAzs must be an integer from 0 to 6'); }, + ); + + test.each(['', 'typo', true, false, null, 1])('rejects invalid topology %p before an AWS lookup', async value => { + const describeAzs = jest.fn(); + const resolveCallerAccount = jest.fn(); + await expect(buildApp({ + appProps: { context: { networkTopology: value } }, describeAzs, resolveCallerAccount, + })).rejects.toThrow('networkTopology must be inline or split'); + expect(describeAzs).not.toHaveBeenCalled(); + expect(resolveCallerAccount).not.toHaveBeenCalled(); + }); +}); + +describe.each(['agentcore', 'ecs', 'lambda-microvm'] as const)('%s network extraction', compute => { + const directories: string[] = []; + let inline: Deployment; + let split: Deployment; + let threeZones: Deployment; + let reducedZones: Deployment; + let network: TemplateJson; + + async function synthesize(topology: NetworkTopology, zones?: readonly string[], reservedAzs = '0'): Promise { + const directory = mkdtempSync(path.join(tmpdir(), 'network-extraction-')); + directories.push(directory); + const app = await buildApp({ + account: FIXTURE.account, + region: FIXTURE.region, + describeAzs: async () => [...FIXTURE.zones], + resolveCallerAccount: async () => FIXTURE.account, + blueprints: BLUEPRINTS, + appProps: { + outdir: directory, + autoSynth: false, + context: { + 'stackName': APP_NAME, + 'networkTopology': topology, + 'networkReservedAzs': reservedAzs, + 'compute_types': compute, + 'bedrockGeoRegion': 'global', + 'enableToolGateway': true, + 'enableAgentRegistry': true, + 'enableLinearIdentityVault': true, + 'github:sha': 'fixture-revision', + ...(zones ? { [AGENTCORE_AZS_CONTEXT_KEY]: zones } : {}), + ...(compute === 'lambda-microvm' ? { + microvm_image_identifier: 'arn:aws:lambda:us-east-1:123456789012:microvm-image:fixture-image', + microvm_image_version: '1', + } : {}), + }, + postCliContext: STRUCTURAL_CONTEXT, + }, + }); + const assembly = app.synth(); + return { + directory, + census: inspectAssembly(directory), + application: assembly.getStackByName(APP_NAME).template, + ...(topology === 'split' ? { network: assembly.getStackByName(NETWORK_NAME).template } : {}), + }; + } + + beforeAll(async () => { + // Blueprint timestamps are intentionally unchanged from main. Fix the clock + // only for this comparison so it isolates the network ownership change. + jest.useFakeTimers({ now: new Date('2026-10-02T00:00:00Z'), doNotFake: ['nextTick', 'setImmediate'] }); + inline = await synthesize('inline'); + split = await synthesize('split'); + threeZones = await synthesize('split', FIXTURE.zones.map(zone => zone.zoneName)); + reducedZones = await synthesize('split', FIXTURE.zones.slice(0, 2).map(zone => zone.zoneName), '1'); + network = split.network!; + }, 60_000); + + afterAll(() => { + jest.useRealTimers(); + for (const directory of directories) rmSync(directory, { recursive: true, force: true }); + }); + + test('synthesizes two stacks with only application-to-network dependencies and no nag errors', () => { + expect(inline.census.stackDependencies).toEqual({ [`${APP_NAME}.template.json`]: [] }); + expect(split.census.stackDependencies).toEqual({ + [`${APP_NAME}.template.json`]: [`${NETWORK_NAME}.template.json`], + [`${NETWORK_NAME}.template.json`]: [], + }); + expect(inline.census.errors).toEqual([]); + expect(split.census.errors).toEqual([]); + expect(JSON.stringify(network)).not.toContain('Fn::ImportValue'); + expect(JSON.stringify(split.application)).toContain('Fn::ImportValue'); + expect(Object.keys(split.application.Resources).length).toBeLessThan(Object.keys(inline.application.Resources).length - 45); + }); + + test('exports the complete network interface even when this backend leaves a value unused', () => { + const resources = Object.entries(network.Resources as Record); + const vpc = resources.find(([, resource]) => resource.Type === 'AWS::EC2::VPC')!; + const runtimeGroup = resources.find(([, resource]) => resource.Type === 'AWS::EC2::SecurityGroup' + && resource.Properties.GroupDescription === 'AgentCore Runtime - egress TCP 443 only')!; + const privateSubnets = resources.filter(([, resource]) => resource.Type === 'AWS::EC2::Subnet' + && resource.Properties.Tags.some((tag: { Key: string; Value: string }) => tag.Key === 'aws-cdk:subnet-type' && tag.Value === 'Private')); + expect(privateSubnets).toHaveLength(2); + const expected = [ + { Ref: vpc[0] }, + { 'Fn::GetAtt': [runtimeGroup[0], 'GroupId'] }, + ...privateSubnets.map(([id]) => ({ Ref: id })), + ]; + const outputs = Object.values(network.Outputs as Record); + expect(outputs.map(output => JSON.stringify(output.Value)).sort()).toEqual(expected.map(value => JSON.stringify(value)).sort()); + for (const output of outputs) expect(output.Export.Name).toMatch(`${NETWORK_NAME}:ExportsOutput`); + }); + + test('can release the third subnet export by deploying only the two-zone application first', () => { + const oldExports = new Map(Object.values(threeZones.network!.Outputs as Record) + .map(output => [output.Export.Name, output.Value])); + const reducedNetwork = reducedZones.network!; + const newExports = new Map(Object.values(reducedNetwork.Outputs as Record) + .map(output => [output.Export.Name, output.Value])); + const removed = [...oldExports.keys()].filter(name => !newExports.has(name)); + expect(removed).toHaveLength(1); + expect(importedExports(threeZones.application).has(removed[0])).toBe(true); + + // Stage one: every import in the target application still resolves in the + // deployed three-zone network. --exclusively keeps that network unchanged. + const targetImports = importedExports(reducedZones.application); + expect(targetImports.has(removed[0])).toBe(false); + for (const name of targetImports) expect(oldExports.get(name)).toEqual(newExports.get(name)); + + // Stage two: the network can drop the unused export and subnet. Remaining + // exported resources keep their identities and service properties. + const removedSubnetId = oldExports.get(removed[0]).Ref; + expect(threeZones.network!.Resources[removedSubnetId].Type).toBe('AWS::EC2::Subnet'); + expect(reducedNetwork.Resources).not.toHaveProperty(removedSubnetId); + for (const [name, reference] of newExports) { + expect(oldExports.get(name)).toEqual(reference); + const resourceId = reference.Ref ?? reference['Fn::GetAtt'][0]; + expect(withoutMetadata(reducedNetwork.Resources[resourceId])) + .toEqual(withoutMetadata(threeZones.network!.Resources[resourceId])); + } + const subnets = Object.entries(reducedNetwork.Resources as Record) + .filter(([, resource]) => resource.Type === 'AWS::EC2::Subnet'); + expect(subnets).toHaveLength(4); + for (const [id, subnet] of subnets) { + expect(subnet.Properties).toEqual(threeZones.network!.Resources[id].Properties); + } + expect(reducedZones.census.errors).toEqual([]); + }); + + test('moves the VPC and DNS definitions with the same logical IDs and service properties', () => { + const moved = Object.entries(inline.application.Resources).filter(([id]) => isNetworkResource(id)); + expect(moved.length).toBeGreaterThan(45); + for (const [id, original] of moved) { + expect(split.application.Resources).not.toHaveProperty(id); + expect({ [id]: withoutMetadata(network.Resources[id]) }).toEqual({ [id]: withoutMetadata(original as TemplateJson) }); + } + }); + + test('preserves every application data resource and its lifecycle policies', () => { + const data = Object.entries(inline.application.Resources as Record) + .filter(([id, resource]) => ['AWS::DynamoDB::Table', 'AWS::S3::Bucket', 'AWS::SecretsManager::Secret', + 'AWS::Cognito::UserPool', 'AWS::KMS::Key', 'AWS::Logs::LogGroup', 'AWS::BedrockAgentCore::Memory'] + .includes(resource.Type) && !isNetworkResource(id)); + expect(data.length).toBeGreaterThan(20); + for (const [id, original] of data) { + expect({ [id]: split.application.Resources[id] }).toEqual({ [id]: original }); + } + const logs = Object.values(network.Resources as Record).filter(resource => resource.Type === 'AWS::Logs::LogGroup'); + expect(logs).toHaveLength(2); + for (const resource of logs) expect(resource).toMatchObject({ DeletionPolicy: 'Delete', UpdateReplacePolicy: 'Delete' }); + }); + + test('keeps shared API routes, CORS, authorizers, permissions and deployment dependencies in the application', () => { + const apiResources = (template: TemplateJson): TemplateJson => Object.fromEntries( + Object.entries(template.Resources as Record) + .filter(([, resource]) => resource.Type.startsWith('AWS::ApiGateway::') || resource.Type === 'AWS::Lambda::Permission'), + ); + expect(apiResources(split.application)).toEqual(apiResources(inline.application)); + expect(apiResources(network)).toEqual({}); + expect(split.application.Outputs).toEqual(inline.application.Outputs); + }); + + test('resolves network imports to the same references without changing application service properties', () => { + const exports = new Map(Object.values(network.Outputs as Record) + .map(output => [JSON.stringify(output.Export.Name), output.Value])); + const imports = new Set(); + const versionOf = (template: TemplateJson): [string, TemplateJson] => { + const versions = Object.entries(template.Resources as Record) + .filter(([id, resource]) => resource.Type === 'AWS::Lambda::Version' && id.startsWith('TaskOrchestratorOrchestratorFnCurrentVersion')); + expect(versions).toHaveLength(1); + return versions[0]; + }; + const [beforeVersionId, beforeVersion] = versionOf(inline.application); + const [afterVersionId, afterVersion] = versionOf(split.application); + expect(withoutMetadata(afterVersion)).toEqual(withoutMetadata(beforeVersion)); + // The existing alpha Guardrail hashes unresolved tokens, so its version ID + // (and the orchestrator version consuming it) can change between syntheses. + // Compare their definitions before mapping only these immutable version IDs. + // The census stability check reports the real churn without normalization. + const guardrailVersion = (template: TemplateJson): [string, TemplateJson] => { + const versions = Object.entries(template.Resources as Record) + .filter(([, resource]) => resource.Type === 'AWS::Bedrock::GuardrailVersion'); + expect(versions).toHaveLength(1); + return versions[0]; + }; + const [beforeGuardrailId, beforeGuardrail] = guardrailVersion(inline.application); + const [afterGuardrailId, afterGuardrail] = guardrailVersion(split.application); + expect(withoutMetadata(afterGuardrail)).toEqual(withoutMetadata(beforeGuardrail)); + const versionIds = new Map([[afterVersionId, beforeVersionId], [afterGuardrailId, beforeGuardrailId]]); + const originalId = (id: string): string => versionIds.get(id) ?? id; + function normalize(value: any): any { + if (Array.isArray(value)) return value.map(normalize); + if (value && typeof value === 'object') { + if (Object.hasOwn(value, 'Fn::ImportValue')) { + const name = JSON.stringify(value['Fn::ImportValue']); + expect(exports.has(name)).toBe(true); + imports.add(name); + return exports.get(name); + } + return Object.fromEntries(Object.entries(value).filter(([key]) => key !== 'Metadata') + .map(([key, child]) => [key, normalize(child)])); + } + return typeof value === 'string' ? originalId(value) : value; + } + for (const [id, resource] of Object.entries(split.application.Resources as Record) + .filter(([, value]) => value.Type !== 'AWS::CDK::Metadata')) { + const key = originalId(id); + expect({ [key]: normalize(resource) }).toEqual({ [key]: normalize(inline.application.Resources[key]) }); + } + expect(imports.size).toBeGreaterThan(0); + expect(imports.size).toBeLessThanOrEqual(exports.size); + const applicationIds = new Set(Object.keys(split.application.Resources).map(originalId)); + for (const [id, resource] of Object.entries(inline.application.Resources as Record) + .filter(([key]) => !applicationIds.has(key))) { + expect({ [id]: withoutMetadata(network.Resources[id]) }).toEqual({ [id]: withoutMetadata(resource) }); + } + }); + + test('feeds the same Blueprint domain configuration into DNS and repository provisioning', () => { + const additional = Object.values(network.Resources as Record) + .find(resource => resource.Type === 'AWS::Route53Resolver::FirewallDomainList' && resource.Properties.Name === 'blueprint-additional'); + expect(additional!.Properties.Domains).toEqual(['packages.example.com', '*.internal.example.org']); + const repositories = Object.entries(split.application.Resources as Record) + .filter(([id, resource]) => resource.Type === 'Custom::AWS' + && BLUEPRINTS.some(blueprint => id.startsWith(`${blueprint.id}RepoConfigCR`))) + .map(([, resource]) => { + const create = resource.Properties.Create; + const serialized = typeof create === 'string' ? create + : create['Fn::Join'][1].map((part: unknown) => typeof part === 'string' ? part : 'table-name').join(''); + return JSON.parse(serialized).parameters.Item; + }); + expect(repositories).toHaveLength(2); + for (const blueprint of BLUEPRINTS) { + const row = repositories.find(item => item.repo.S === blueprint.repo)!; + expect(row.egress_allowlist).toEqual({ + L: blueprint.networking!.egressAllowlist!.map(S => ({ S })), + }); + } + }); + + test('keeps deployment attribution on both stacks without tagging replacement-sensitive DNS logging resources', () => { + for (const template of [split.application, network]) { + const functions = Object.entries(template.Resources as Record) + .filter(([, resource]) => resource.Type === 'AWS::Lambda::Function'); + for (const [id, fn] of functions) { + expect(fn.Properties.Environment?.Variables?.AWS_SDK_UA_APP_ID).toBe(`uksb-wt64nei4u6#${APP_NAME}`); + // Core CDK providers use generic CfnResource without a TagManager; + // require parity with their existing tags as well as attributed SDK calls. + expect(fn.Properties.Tags).toEqual(inline.application.Resources[id].Properties.Tags); + } + const tagged = functions.filter(([, fn]) => fn.Properties.Tags); + expect(tagged.length).toBeGreaterThan(0); + for (const [, fn] of tagged) { + expect(fn.Properties.Tags).toEqual(expect.arrayContaining([ + { Key: 'github:sha', Value: 'fixture-revision' }, + { Key: 'compute_type', Value: compute }, + ])); + } + } + for (const resource of Object.values(network.Resources as Record) + .filter(candidate => ['AWS::Route53Resolver::ResolverQueryLoggingConfig', + 'AWS::Route53Resolver::ResolverQueryLoggingConfigAssociation'].includes(candidate.Type))) { + expect(resource.Properties).not.toHaveProperty('Tags'); + } + }); + + test('emits compact JSON for both top-level stacks and every nested template', () => { + for (const template of split.census.templates) { + expect(readFileSync(path.join(split.directory, template.file), 'utf8')).not.toContain('\n '); + } + }); +}); diff --git a/cdk/test/synthesis/assembly.test.ts b/cdk/test/synthesis/assembly.test.ts new file mode 100644 index 000000000..55259847d --- /dev/null +++ b/cdk/test/synthesis/assembly.test.ts @@ -0,0 +1,356 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import { mkdtempSync, mkdirSync, readFileSync, rmSync, symlinkSync, writeFileSync } from 'node:fs'; +import { tmpdir } from 'node:os'; +import * as path from 'node:path'; +import { Annotations, App, CfnResource, NestedStack, Stack } from 'aws-cdk-lib'; +import { StringParameter } from 'aws-cdk-lib/aws-ssm'; +import { canonicalJson, compareAssemblies, inspectAssembly } from '../../src/synthesis/assembly'; + +describe('CloudFormation assembly census', () => { + let directory: string; + beforeEach(() => { directory = mkdtempSync(path.join(tmpdir(), 'assembly-census-')); }); + afterEach(() => { rmSync(directory, { recursive: true, force: true }); }); + + function json(file: string, value: unknown): void { + const target = path.join(directory, file); + mkdirSync(path.dirname(target), { recursive: true }); + writeFileSync(target, JSON.stringify(value)); + } + + function assembly(prefix = ''): void { + json(`${prefix}manifest.json`, { + artifacts: { + Api: { + type: 'aws:cloudformation:stack', + properties: { templateFile: 'api.template.json' }, + dependencies: ['Assets'], + metadata: { '/Api': [{ type: 'aws:cdk:warning', data: 'warning is not an error' }] }, + }, + Assets: { type: 'cdk:asset-manifest', properties: { file: 'assets.json' } }, + }, + }); + json(`${prefix}assets.json`, { files: {} }); + json(`${prefix}api.template.json`, { + Resources: { + Metadata: { Type: 'AWS::CDK::Metadata' }, + Child: { Type: 'AWS::CloudFormation::Stack', Metadata: { 'aws:asset:path': 'child.template.json' } }, + }, + Parameters: { Config: { Type: 'String' } }, + Outputs: { Value: { Value: 'value' } }, + }); + json(`${prefix}child.template.json`, { + Resources: { + Table: { + Type: 'AWS::DynamoDB::Table', + Metadata: { 'aws:cdk:path': 'Api/Child/Table' }, + DeletionPolicy: 'Retain', + }, + }, + }); + } + + test('counts parent and nested templates separately, includes metadata, ignores orphan files', () => { + assembly(); + json('stale.template.json', { Resources: { Orphan: { Type: 'AWS::S3::Bucket' } } }); + const result = inspectAssembly(directory); + expect(result.totalResources).toBe(3); + expect(result.templates).toHaveLength(2); + expect(result.templates[0]).toMatchObject({ + file: 'api.template.json', + resources: 2, + parameters: 1, + outputs: 1, + types: { 'AWS::CDK::Metadata': 1, 'AWS::CloudFormation::Stack': 1 }, + bytes: readFileSync(path.join(directory, 'api.template.json')).length, + }); + expect(result.templates[1].inventory).toEqual([{ + logicalId: 'Table', + type: 'AWS::DynamoDB::Table', + constructPath: 'Api/Child/Table', + deletionPolicy: 'Retain', + updateReplacePolicy: null, + }]); + expect(result.nestedEdges).toEqual([{ parent: 'api.template.json', child: 'child.template.json' }]); + expect(result.errors).toEqual([]); + }); + + test('traverses stage assemblies and retains error annotations', () => { + assembly('stage/'); + json('manifest.json', { + artifacts: { + Stage: { type: 'cdk:cloud-assembly', properties: { directoryName: 'stage' } }, + Broken: { type: 'tree', metadata: { '/Child': [{ type: 'aws:cdk:error', data: 'missing permission' }] } }, + }, + }); + const result = inspectAssembly(directory); + expect(result.templates.map(t => t.file)).toEqual(['stage/api.template.json', 'stage/child.template.json']); + expect(result.errors).toEqual(['/Child: missing permission']); + }); + + test('reads actual CDK asset manifests and metadata sidecars, including nested errors', () => { + const app = new App({ outdir: directory }); + const stack = new Stack(app, 'Root', { env: { account: '123456789012', region: 'us-east-1' } }); + const child = new NestedStack(stack, 'Child'); + new CfnResource(child, 'Bucket', { type: 'AWS::S3::Bucket' }); + Annotations.of(child).addError('nested error must reach the census'); + app.synth(); + const result = inspectAssembly(directory); + expect(result.templates).toHaveLength(2); + expect(result.nestedEdges).toHaveLength(1); + expect(result.errors).toContain('/Root/Child: nested error must reach the census'); + expect(result.templates.flatMap(t => t.inventory).filter(r => r.type === 'AWS::S3::Bucket')).toHaveLength(1); + }); + + test('resolves identical nested template hashes separately for each owning top-level stack', () => { + const app = new App({ outdir: directory, autoSynth: false }); + for (const id of ['First', 'Second']) { + const stack = new Stack(app, id, { env: { account: '123456789012', region: 'us-east-1' } }); + new CfnResource(new NestedStack(stack, 'Child'), 'Bucket', { type: 'AWS::S3::Bucket' }); + } + app.synth(); + const result = inspectAssembly(directory); + expect(result.templates).toHaveLength(4); + expect(result.nestedEdges).toEqual(expect.arrayContaining([ + { parent: 'First.template.json', child: expect.stringMatching(/^FirstChild.*nested.template.json$/) }, + { parent: 'Second.template.json', child: expect.stringMatching(/^SecondChild.*nested.template.json$/) }, + ])); + const children = result.templates.filter(template => template.file.endsWith('.nested.template.json')); + expect(children).toHaveLength(2); + expect(children[0].semanticSha256).toBe(children[1].semanticSha256); + }); + + test('flags unresolved CDK lookups even when app.synth emits templates without error annotations', () => { + const app = new App({ outdir: directory, autoSynth: false }); + const stack = new Stack(app, 'Root', { env: { account: '123456789012', region: 'us-east-1' } }); + new CfnResource(stack, 'Bucket', { + type: 'AWS::S3::Bucket', + properties: { BucketName: StringParameter.valueFromLookup(stack, '/fixture/bucket') }, + }); + app.synth(); + const result = inspectAssembly(directory); + expect(result.templates).toHaveLength(1); + expect(result.errors).toEqual([expect.stringContaining('Unresolved CDK context')]); + expect(result.errors[0]).toContain('parameterName=/fixture/bucket'); + }); + + test('retains missing-context failures inside stage assemblies', () => { + assembly('stage/'); + const file = path.join(directory, 'stage/manifest.json'); + const manifest = JSON.parse(readFileSync(file, 'utf8')); + manifest.missing = [{ key: 'fixture', provider: 'ssm', props: {} }]; + json('stage/manifest.json', manifest); + json('manifest.json', { artifacts: { Stage: { type: 'cdk:cloud-assembly', properties: { directoryName: 'stage' } } } }); + expect(inspectAssembly(directory).errors).toEqual(['Unresolved CDK context in stage: fixture (ssm)']); + }); + + test('rejects template and stage symlinks that escape or revisit their assembly', () => { + assembly('inside/'); + json('outside.json', { Resources: {} }); + symlinkSync('../outside.json', path.join(directory, 'inside/escape.json')); + json('inside/api.template.json', { + Resources: { Child: { Type: 'AWS::CloudFormation::Stack', Metadata: { 'aws:asset:path': 'escape.json' } } }, + }); + expect(() => inspectAssembly(path.join(directory, 'inside'))).toThrow(/symlink/); + symlinkSync('.', path.join(directory, 'cycle')); + json('manifest.json', { artifacts: { Stage: { type: 'cdk:cloud-assembly', properties: { directoryName: 'cycle' } } } }); + expect(() => inspectAssembly(directory)).toThrow(/symlink/); + }); + + test('compares real CDK stack dependencies even when every template is unchanged', () => { + for (const [name, dependent] of [['first', false], ['second', true]] as const) { + const app = new App({ outdir: path.join(directory, name), autoSynth: false }); + const producer = new Stack(app, 'Producer'); + new CfnResource(producer, 'Bucket', { type: 'AWS::S3::Bucket' }); + const consumer = new Stack(app, 'Consumer'); + new CfnResource(consumer, 'Queue', { type: 'AWS::SQS::Queue' }); + if (dependent) consumer.addDependency(producer); + app.synth(); + } + expect(inspectAssembly(path.join(directory, 'second')).stackDependencies).toEqual({ + 'Producer.template.json': [], + 'Consumer.template.json': ['Producer.template.json'], + }); + expect(compareAssemblies(path.join(directory, 'first'), path.join(directory, 'second'))).toEqual([{ + kind: 'stack-dependencies', file: 'Consumer.template.json', paths: [], totalDifferences: 1, + }]); + }); + + test('qualifies same-named dependencies by stage and treats dependency order as irrelevant', () => { + for (const prefix of ['first/one/', 'first/two/', 'second/one/', 'second/two/']) { + assembly(prefix); + const file = path.join(directory, prefix, 'manifest.json'); + const manifest = JSON.parse(readFileSync(file, 'utf8')); + manifest.artifacts.Producer = { type: 'aws:cloudformation:stack', properties: { templateFile: 'producer.template.json' } }; + manifest.artifacts.Api.dependencies = prefix.startsWith('first') ? ['Producer', 'Assets'] : ['Assets', 'Producer']; + json(`${prefix}manifest.json`, manifest); + json(`${prefix}producer.template.json`, { Resources: { Bucket: { Type: 'AWS::S3::Bucket' } } }); + } + for (const prefix of ['first/', 'second/']) { + json(`${prefix}manifest.json`, { + artifacts: { + One: { type: 'cdk:cloud-assembly', properties: { directoryName: 'one' } }, + Two: { type: 'cdk:cloud-assembly', properties: { directoryName: 'two' } }, + }, + }); + } + expect(inspectAssembly(path.join(directory, 'first')).stackDependencies).toMatchObject({ + 'one/api.template.json': ['one/producer.template.json'], + 'two/api.template.json': ['two/producer.template.json'], + }); + expect(compareAssemblies(path.join(directory, 'first'), path.join(directory, 'second'))).toEqual([]); + }); + + test('rejects unresolved and cyclic stack dependencies', () => { + assembly(); + const file = path.join(directory, 'manifest.json'); + const manifest = JSON.parse(readFileSync(file, 'utf8')); + manifest.artifacts.Api.dependencies = ['Missing']; + json('manifest.json', manifest); + expect(() => inspectAssembly(directory)).toThrow(/Unknown dependency/); + manifest.artifacts.Api.dependencies = ['Api']; + json('manifest.json', manifest); + expect(() => inspectAssembly(directory)).toThrow(/Stack dependency cycle/); + }); + + test('does not count a referenced template twice', () => { + assembly(); + json('api.template.json', { + Resources: { + First: { Type: 'AWS::CloudFormation::Stack', Metadata: { 'aws:asset:path': 'child.template.json' } }, + Second: { Type: 'AWS::CloudFormation::Stack', Metadata: { 'aws:asset:path': 'child.template.json' } }, + }, + }); + expect(inspectAssembly(directory).templates).toHaveLength(2); + }); + + test('allows unrelated external file assets but rejects nested templates outside the assembly', () => { + assembly(); + json('manifest.json', { + artifacts: { + Api: { type: 'aws:cloudformation:stack', properties: { templateFile: 'api.template.json' }, dependencies: ['Assets'] }, + Assets: { type: 'cdk:asset-manifest', properties: { file: 'assets.json' } }, + }, + }); + json('assets.json', { + files: { + Unrelated: { + source: { path: '../outside.json', packaging: 'file' }, + destinations: { Fixture: { objectKey: 'outside.json' } }, + }, + }, + }); + expect(inspectAssembly(directory).templates).toHaveLength(2); + json('api.template.json', { + Resources: { + Child: { + Type: 'AWS::CloudFormation::Stack', + Properties: { TemplateURL: 'https://example.com/outside.json' }, + }, + }, + }); + expect(() => inspectAssembly(directory)).toThrow(/escapes/); + }); + + test('preserves absent retention policies without assuming a resource-specific default', () => { + assembly(); + json('api.template.json', { Resources: { Database: { Type: 'AWS::RDS::DBCluster' } } }); + expect(inspectAssembly(directory).templates[0].inventory[0]).toMatchObject({ + deletionPolicy: null, updateReplacePolicy: null, + }); + }); + + test.each([ + ['missing local metadata', { Type: 'AWS::CloudFormation::Stack' }, /Missing local/], + ['escaping path', { Type: 'AWS::CloudFormation::Stack', Metadata: { 'aws:asset:path': '../outside.json' } }, /escapes/], + ['cycle', { Type: 'AWS::CloudFormation::Stack', Metadata: { 'aws:asset:path': 'api.template.json' } }, /cycle/], + ['missing type', {}, /Missing resource type/], + ['invalid resource', null, /Expected object/], + ])('fails closed for %s', (_name, resource, message) => { + assembly(); + json('api.template.json', { Resources: { Broken: resource } }); + expect(() => inspectAssembly(directory)).toThrow(message); + }); + + test.each([ + ['empty assembly', { artifacts: {} }, /No stack templates/], + ['invalid dependencies', { + artifacts: { + Api: { + type: 'aws:cloudformation:stack', properties: { templateFile: 'api.template.json' }, dependencies: [1], + }, + }, + }, /Invalid dependencies/], + ['invalid metadata', { artifacts: { Api: { metadata: { '/Api': {} } } } }, /Expected metadata array/], + ])('rejects %s', (_name, manifest, message) => { + assembly(); + json('manifest.json', manifest); + expect(() => inspectAssembly(directory)).toThrow(message); + }); + + test('JSON equality ignores object order but retains arrays and all meaningful values', () => { + expect(canonicalJson({ b: [true, null, 1], a: 'x' })).toBe(canonicalJson({ a: 'x', b: [true, null, 1] })); + expect(canonicalJson([1, 2])).not.toBe(canonicalJson([2, 1])); + }); + + test('compares nested templates, preserves timestamp differences, and escapes JSON pointers', () => { + assembly('first/'); + assembly('second/'); + json('first/child.template.json', { + Resources: { + Table: { Type: 'AWS::DynamoDB::Table', Properties: { 'a/b~c': { timestamp: 'first' } } }, + }, + }); + json('second/child.template.json', { + Resources: { + Table: { Properties: { 'a/b~c': { timestamp: 'second' } }, Type: 'AWS::DynamoDB::Table' }, + }, + }); + expect(compareAssemblies(path.join(directory, 'first'), path.join(directory, 'second'))).toEqual([{ + kind: 'template', + file: 'child.template.json', + paths: ['/Resources/Table/Properties/a~1b~0c/timestamp'], + totalDifferences: 1, + }]); + }); + + test('reports removed templates and bounds diagnostics without losing the difference count', () => { + assembly('first/'); + assembly('second/'); + json('second/api.template.json', { Resources: {}, Description: 'changed' }); + const first = path.join(directory, 'first'); + const second = path.join(directory, 'second'); + expect(compareAssemblies(first, second)).toContainEqual({ + kind: 'template', file: 'child.template.json', paths: ['/'], totalDifferences: 1, + }); + json('first/api.template.json', { Resources: {}, ...Object.fromEntries(Array.from({ length: 120 }, (_, i) => [`k${i}`, i])) }); + json('second/api.template.json', { Resources: {}, ...Object.fromEntries(Array.from({ length: 120 }, (_, i) => [`k${i}`, i + 1])) }); + expect(compareAssemblies(first, second)[0]).toMatchObject({ totalDifferences: 120 }); + expect(compareAssemblies(first, second)[0].paths).toHaveLength(100); + }); + + test('formatting changes do not become semantic changes', () => { + assembly('first/'); + assembly('second/'); + const file = path.join(directory, 'second/api.template.json'); + writeFileSync(file, JSON.stringify(JSON.parse(readFileSync(file, 'utf8')), null, 4)); + expect(compareAssemblies(path.join(directory, 'first'), path.join(directory, 'second'))).toEqual([]); + }); +}); diff --git a/cdk/test/synthesis/audit.test.ts b/cdk/test/synthesis/audit.test.ts new file mode 100644 index 000000000..69283cb24 --- /dev/null +++ b/cdk/test/synthesis/audit.test.ts @@ -0,0 +1,139 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import { mkdirSync, mkdtempSync, rmSync, writeFileSync } from 'node:fs'; +import { tmpdir } from 'node:os'; +import * as path from 'node:path'; +import { inspectAssembly } from '../../src/synthesis/assembly'; +import { auditProfile, Budgets, WorkerResult } from '../../src/synthesis/audit'; +import { synthesisProfiles } from '../../src/synthesis/profiles'; + +describe('profile acceptance rules', () => { + let directory: string; + const profile = synthesisProfiles()[0]; + const rejected = { + ...profile, + name: 'invalid-compute', + context: { ...profile.context, compute_types: 'unsupported' }, + expectedError: 'compute_type must be agentcore, ecs or lambda-microvm', + }; + const budgets: Budgets = { resources: 500, bytes: 800_000, parameters: 200, outputs: 200 }; + beforeEach(() => { directory = mkdtempSync(path.join(tmpdir(), 'profile-audit-')); }); + afterEach(() => { rmSync(directory, { recursive: true, force: true }); }); + + const retained = { DeletionPolicy: 'Retain', UpdateReplacePolicy: 'Retain' }; + function synthesize(target: string, padding = '', template: object = { Resources: { Bucket: { Type: 'AWS::S3::Bucket', ...retained } } }): WorkerResult { + mkdirSync(target); + writeFileSync(path.join(target, 'manifest.json'), JSON.stringify({ + artifacts: { Api: { type: 'aws:cloudformation:stack', properties: { templateFile: 'api.template.json' } } }, + })); + writeFileSync(path.join(target, 'api.template.json'), `${JSON.stringify(template)}${padding}`); + return { kind: 'synthesized', census: inspectAssembly(target) }; + } + + test('accepts two unchanged independent assemblies', () => { + const worker = jest.fn((_profile, target: string) => synthesize(target)); + const audit = auditProfile(profile, path.join(directory, 'first'), budgets, true, worker); + expect(worker).toHaveBeenCalledTimes(2); + expect(audit.failures).toEqual([]); + expect(audit.differences).toEqual([]); + }); + + test('checks repeat byte limits even if only JSON whitespace changed', () => { + let calls = 0; + const worker = jest.fn((_profile, target: string) => synthesize(target, calls++ ? ' '.repeat(1000) : '')); + const audit = auditProfile(profile, path.join(directory, 'first'), { ...budgets, bytes: 128 }, true, worker); + expect(audit.differences).toEqual([]); + expect(audit.failures).toEqual([expect.stringMatching(/^Repeat: api.template.json: \d+ bytes exceeds 128$/)]); + }); + + test('enforces resource, byte, parameter, and output limits', () => { + const worker = (_profile: unknown, target: string) => synthesize(target, '', { + Resources: { Bucket: { Type: 'AWS::S3::Bucket', ...retained }, Queue: { Type: 'AWS::SQS::Queue', ...retained } }, + Parameters: { A: { Type: 'String' }, B: { Type: 'String' } }, + Outputs: { A: { Value: 'a' }, B: { Value: 'b' } }, + }); + const audit = auditProfile(profile, path.join(directory, 'first'), { resources: 1, bytes: 1, parameters: 1, outputs: 1 }, false, worker); + expect(audit.failures).toHaveLength(4); + for (const metric of ['resources', 'bytes', 'parameters', 'outputs']) { + expect(audit.failures).toContainEqual(expect.stringContaining(`${metric} exceeds 1`)); + } + }); + + test('checks expected rejection in both processes when stability is requested', () => { + const worker = jest.fn((): WorkerResult => ({ kind: 'rejected', error: `${rejected.expectedError} fixture reason` })); + const audit = auditProfile(rejected, directory, budgets, true, worker); + expect(worker).toHaveBeenCalledTimes(2); + expect(audit.first?.kind).toBe('rejected'); + expect(audit.second?.kind).toBe('rejected'); + expect(audit.failures).toEqual([]); + }); + + test('does not mistake an unrelated repeat failure for the expected guard', () => { + const worker = jest.fn() + .mockReturnValueOnce({ kind: 'rejected', error: `${rejected.expectedError} fixture reason` }) + .mockReturnValueOnce({ kind: 'rejected', error: 'unrelated synthesis failure' }); + expect(auditProfile(rejected, directory, budgets, true, worker).failures).toEqual(['Repeat: unrelated synthesis failure']); + }); + + test('accepts only the named stack and production ceiling for a resource-budget rejection', () => { + const limited = { + ...profile, + expectedError: { stackName: 'backgroundagent-dev', resourceLimit: 490 }, + }; + const expected = "Number of resources in stack 'backgroundagent-dev': 497 is greater than allowed maximum of 490: fixture"; + for (const message of [expected, expected.replace('497', '498')]) { + expect(auditProfile(limited, directory, budgets, false, () => ({ kind: 'rejected', error: message })).failures).toEqual([]); + } + for (const message of [ + expected.replace('maximum of 490', 'maximum of 500'), + expected.replace('497', '490'), + expected.replace('backgroundagent-dev', 'unrelated'), + `Unrelated failure: ${expected}`, + ]) { + expect(auditProfile(limited, directory, budgets, false, () => ({ kind: 'rejected', error: message })).failures).toEqual([message]); + } + }); + + test('fails if an invalid profile unexpectedly synthesizes', () => { + const worker = (_profile: unknown, target: string) => synthesize(target); + const audit = auditProfile(rejected, path.join(directory, 'first'), budgets, false, worker); + expect(audit.failures).toEqual([expect.stringContaining('Expected rejection was not raised')]); + }); + + test('retains completed evidence when a repeat worker fails', () => { + const worker = jest.fn((_profile, target: string) => synthesize(target)) + .mockImplementationOnce((_profile, target: string) => synthesize(target)) + .mockImplementationOnce(() => { throw new Error('worker timeout'); }); + const audit = auditProfile(profile, path.join(directory, 'first'), budgets, true, worker); + expect(audit.first?.kind).toBe('synthesized'); + expect(audit.second).toBeUndefined(); + expect(audit.failures).toEqual(['worker timeout']); + }); + + test('carries unresolved-context diagnostics into profile failure', () => { + const worker = (_profile: unknown, target: string): WorkerResult => { + const result = synthesize(target); + if (result.kind !== 'synthesized') throw new Error('fixture must synthesize'); + return { kind: 'synthesized', census: { ...result.census, errors: ['Unresolved CDK context: fixture'] } }; + }; + expect(auditProfile(profile, path.join(directory, 'first'), budgets, false, worker).failures) + .toEqual(['Unresolved CDK context: fixture']); + }); +}); diff --git a/cdk/test/synthesis/deployment.test.ts b/cdk/test/synthesis/deployment.test.ts new file mode 100644 index 000000000..99ea4894e --- /dev/null +++ b/cdk/test/synthesis/deployment.test.ts @@ -0,0 +1,138 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import { mkdtempSync, readFileSync, rmSync } from 'node:fs'; +import { tmpdir } from 'node:os'; +import * as path from 'node:path'; +import { Template } from 'aws-cdk-lib/assertions'; +import type { CloudAssembly } from 'aws-cdk-lib/cx-api'; +import { AGENTCORE_AZS_CONTEXT_KEY, AUTO_PIN_AZ_COUNT } from '../../src/constructs/agentcore-azs'; +import { resolveComputeBackends } from '../../src/handlers/shared/compute-backend'; +import { buildApp } from '../../src/main'; +import { AssemblyCensus, inspectAssembly } from '../../src/synthesis/assembly'; +import { auditProfile, DEFAULT_BUDGETS, WorkerResult } from '../../src/synthesis/audit'; +import { FIXTURE, STRUCTURAL_CONTEXT, synthesisProfiles } from '../../src/synthesis/profiles'; +import { projectContext } from '../../src/synthesis/workspace'; + +// Exercise the same full gate product as the offline census in the normal build. +// Each configuration is synthesized once, including real CDK metadata and +// parent/nested templates for all quota checks. +describe.each(synthesisProfiles())('$name deployment', profile => { + let directory: string; + let census: AssemblyCensus; + let assembly: CloudAssembly; + let result: WorkerResult; + let apiPermissions: readonly { id: string; sourceArn: string }[]; + let subnetZones: string[]; + beforeAll(async () => { + directory = mkdtempSync(path.join(tmpdir(), 'deployment-profile-')); + try { + const app = await buildApp({ + account: FIXTURE.account, + region: FIXTURE.region, + describeAzs: async () => [...FIXTURE.zones], + resolveCallerAccount: async () => FIXTURE.account, + appProps: { + outdir: directory, + autoSynth: false, + context: { ...projectContext(path.resolve(__dirname, '../../..')), ...profile.context }, + postCliContext: STRUCTURAL_CONTEXT, + }, + }); + assembly = app.synth(); + census = inspectAssembly(directory); + result = { kind: 'synthesized', census }; + const templates = census.templates.map(({ file }) => ({ + file, template: Template.fromJSON(JSON.parse(readFileSync(path.join(directory, file), 'utf8'))), + })); + apiPermissions = templates.flatMap(({ file, template }) => Object.entries(template.findResources('AWS::Lambda::Permission')) + .filter(([, resource]) => resource.Properties?.Principal === 'apigateway.amazonaws.com') + .map(([logicalId, resource]) => ({ + id: `${file}/${logicalId}`, + sourceArn: JSON.stringify(resource.Properties?.SourceArn ?? null), + }))); + subnetZones = templates.flatMap(({ template }) => Object.values(template.findResources('AWS::EC2::Subnet')) + .map(resource => resource.Properties.AvailabilityZone as string)); + } catch (error) { + if (!profile.expectedError) throw error; + result = { kind: 'rejected', error: error instanceof Error ? error.message : String(error) }; + } + }, 60_000); + afterAll(() => { if (directory) rmSync(directory, { recursive: true, force: true }); }); + + test(profile.expectedError ? 'rejects the over-budget configuration at production synthesis' + : 'keeps every template within budget', () => { + const audit = auditProfile(profile, directory, DEFAULT_BUDGETS, false, () => result); + expect(audit.failures).toEqual([]); + }); + + // A deliberate rejection has no assembly to inspect. The audit above verifies + // the exact guard and also fails if the configuration unexpectedly synthesizes. + if (profile.expectedError) return; + + if (profile.context.bedrockModels) { + test('exercises the SessionRole model grants after CDK creates an overflow policy', () => { + const resources = census.templates.flatMap(template => template.inventory); + expect(resources.some(resource => resource.type === 'AWS::IAM::ManagedPolicy' + && resource.constructPath?.includes('/AgentSessionRole/Role/OverflowPolicy'))).toBe(true); + }); + } + + test('keeps auto-pin at two zones and honors every explicitly pinned zone', () => { + const override = profile.context[AGENTCORE_AZS_CONTEXT_KEY]; + const expected = Array.isArray(override) ? override + : FIXTURE.zones.slice(0, AUTO_PIN_AZ_COUNT).map(zone => zone.zoneName); + expect(subnetZones.sort()).toEqual(expected.flatMap(zone => [zone, zone]).sort()); + }); + + test('emits no CDK template-size warnings, including nested stacks', () => { + const warnings = assembly.stacks.flatMap(stack => stack.messages + .filter(message => message.level === 'warning' && String(message.entry.data).includes('Template size')) + .map(message => `${stack.stackName}/${message.id}: ${String(message.entry.data)}`)); + expect(warnings).toEqual([]); + }); + + test('keeps API Gateway Lambda permissions method-scoped without console test-invoke grants', () => { + expect(apiPermissions.length).toBeGreaterThan(0); + const offenders = apiPermissions.filter(({ sourceArn }) => { + const methodScoped = /\/(GET|POST|PUT|DELETE|PATCH|HEAD|OPTIONS)\//.test(sourceArn); + const specificAuthorizer = sourceArn.includes('/authorizers/') && !sourceArn.includes('/authorizers/*'); + return sourceArn.includes('test-invoke-stage') || (!methodScoped && !specificAuthorizer); + }); + expect(offenders).toEqual([]); + }); + + test('keeps stack dependencies one-way for the selected topology', () => { + const application = 'backgroundagent-dev.template.json'; + const network = 'backgroundagent-dev-network.template.json'; + expect(census.stackDependencies).toEqual(profile.context.networkTopology === 'split' + ? { [application]: [network], [network]: [] } + : { [application]: [] }); + }); + + test('provisions only the selected compute backends across the assembly', () => { + const resources = census.templates.flatMap(template => template.inventory); + const count = (type: string): number => resources.filter(resource => resource.type === type).length; + const backends = resolveComputeBackends(profile.context.compute_types, profile.context.compute_type); + expect(count('AWS::BedrockAgentCore::Runtime')).toBe(backends.includes('agentcore') ? 1 : 0); + expect(count('AWS::ECS::Cluster')).toBe(backends.includes('ecs') ? 1 : 0); + expect(count('AWS::Lambda::NetworkConnector')).toBe(backends.includes('lambda-microvm') ? 2 : 0); + expect(count('AWS::CDK::Metadata')).toBeGreaterThan(0); + }); +}); diff --git a/cdk/test/synthesis/profiles.test.ts b/cdk/test/synthesis/profiles.test.ts new file mode 100644 index 000000000..d3ed84bb7 --- /dev/null +++ b/cdk/test/synthesis/profiles.test.ts @@ -0,0 +1,199 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import { readdirSync } from 'node:fs'; +import { App, AssetStaging, Stack } from 'aws-cdk-lib'; +import { AGENTCORE_AZS_CONTEXT_KEY, AGENTCORE_SUPPORTED_AZ_IDS } from '../../src/constructs/agentcore-azs'; +import { DEFAULT_BEDROCK_MODEL_IDS } from '../../src/handlers/shared/bedrock-model-constants'; +import { FIXTURE, STRUCTURAL_CONTEXT, synthesisEnvironment, synthesisProfiles } from '../../src/synthesis/profiles'; + +describe('structural synthesis profiles', () => { + const profiles = synthesisProfiles(); + const matrix = profiles.filter(p => /-(none|managed|external)(-split)?$/.test(p.name)); + + test.each(['inline', 'split'])('enumerates the real 40-cell product for the %s topology', topology => { + const topologyMatrix = matrix.filter(profile => profile.context.networkTopology === topology); + expect(matrix).toHaveLength(80); + expect(topologyMatrix).toHaveLength(40); + expect(profiles).toHaveLength(122); + expect(new Set(profiles.map(p => p.name)).size).toBe(profiles.length); + for (const compute of ['agentcore', 'ecs', 'lambda-microvm']) { + for (const gateway of [false, true]) { + for (const registry of [false, true]) { + for (const vault of [false, true]) { + const matches = topologyMatrix.filter(p => + p.context.compute_types === compute && p.context.enableToolGateway === gateway && + p.context.enableAgentRegistry === registry && p.context.enableLinearIdentityVault === vault, + ); + expect(matches).toHaveLength(compute === 'lambda-microvm' ? 3 : 1); + } + } + } + } + }); + + test('expects all backend and optional-service combinations to synthesize', () => { + expect(matrix.filter(p => p.expectedError)).toHaveLength(0); + }); + + test( + 'rejects the widest two-zone inline MicroVM profile while keeping its split counterpart', + () => { + const inline = profiles.find(profile => profile.name === 'lambda-microvm-gw1-reg1-vault1-managed-email-fork')!; + expect(inline.expectedError).toEqual({ stackName: 'backgroundagent-dev', resourceLimit: 490 }); + expect(inline.context).not.toHaveProperty(AGENTCORE_AZS_CONTEXT_KEY); + expect(profiles.find(profile => profile.name === `${inline.name}-split`)!.expectedError).toBeUndefined(); + }, + ); + + test( + 'covers three-zone pins and their budget rejections', + () => { + const pinned = profiles.filter(profile => profile.context[AGENTCORE_AZS_CONTEXT_KEY]); + expect(pinned).toHaveLength(6); + expect(FIXTURE.zones).toHaveLength(3); + for (const zone of FIXTURE.zones) { + expect(AGENTCORE_SUPPORTED_AZ_IDS[FIXTURE.region]).toContain(zone.zoneId); + } + for (const compute of ['agentcore', 'ecs', 'lambda-microvm']) { + for (const topology of ['inline', 'split']) { + const matches = pinned.filter(profile => profile.context.compute_types === compute + && profile.context.networkTopology === topology); + expect(matches).toHaveLength(1); + expect(matches[0].context).toMatchObject({ + [AGENTCORE_AZS_CONTEXT_KEY]: ['us-east-1a', 'us-east-1b', 'us-east-1c'], + enableToolGateway: true, + enableAgentRegistry: true, + enableLinearIdentityVault: true, + alertEmail: 'census@example.com', + forkBlueprintRepo: 'example/census-blueprints', + }); + expect(!!matches[0].expectedError).toBe(topology === 'inline'); + } + } + }, + ); + + test('covers every additive backend set at default and widest settings in both topologies', () => { + const additive = profiles.filter(profile => profile.name.startsWith('additive-')); + expect(additive).toHaveLength(16); + for (const backends of ['agentcore,ecs', 'agentcore,lambda-microvm', 'ecs,lambda-microvm', 'agentcore,ecs,lambda-microvm']) { + expect(additive.filter(profile => profile.context.compute_types === backends)).toHaveLength(4); + } + expect(additive.filter(profile => profile.context.networkTopology === 'split') + .every(profile => !profile.expectedError)).toBe(true); + }); + + test('covers expanded model grants on every widest backend in both topologies', () => { + const expanded = profiles.filter(profile => profile.context.bedrockModels); + expect(expanded).toHaveLength(6); + for (const compute of ['agentcore', 'ecs', 'lambda-microvm']) { + for (const topology of ['inline', 'split']) { + const matches = expanded.filter(profile => profile.context.compute_types === compute + && profile.context.networkTopology === topology); + expect(matches).toHaveLength(1); + expect(matches[0].context).toMatchObject({ + enableToolGateway: true, + enableAgentRegistry: true, + enableLinearIdentityVault: true, + alertEmail: 'census@example.com', + forkBlueprintRepo: 'example/census-blueprints', + bedrockModels: expect.arrayContaining([...DEFAULT_BEDROCK_MODEL_IDS]), + }); + expect(matches[0].context.bedrockModels).toHaveLength(DEFAULT_BEDROCK_MODEL_IDS.length + 8); + expect(!!matches[0].expectedError).toBe(topology === 'inline' && compute === 'lambda-microvm'); + } + } + }); + + test('protects both legacy additive selectors without setting compute_types', () => { + const legacy = profiles.filter(profile => profile.name.startsWith('legacy-')); + expect(legacy).toHaveLength(4); + for (const compute of ['ecs', 'lambda-microvm']) { + expect(legacy.filter(profile => profile.context.compute_type === compute)).toHaveLength(2); + } + for (const profile of legacy) { + expect(profile.context).not.toHaveProperty('compute_types'); + expect(profile.expectedError).toBeUndefined(); + } + }); + + test('distinguishes configured images from provisioning-only MicroVM profiles', () => { + const microvm = matrix.filter(p => p.context.compute_types === 'lambda-microvm' && !p.expectedError); + expect(microvm.filter(p => p.microvmImageConfigured)).toHaveLength(32); + for (const p of matrix) { + expect(p.microvmImageConfigured).toBe(!!(p.context.microvm_base_image_arn || p.context.microvm_image_identifier)); + expect(!!p.context.microvm_base_image_arn && !!p.context.microvm_image_identifier).toBe(false); + } + }); + + test('satisfies the CDK availability-zone lookup from the same explicit fixture', () => { + const app = new App({ autoSynth: false, postCliContext: STRUCTURAL_CONTEXT }); + const stack = new Stack(app, 'Fixture', { env: FIXTURE }); + expect(stack.availabilityZones).toEqual(FIXTURE.zones.map(zone => zone.zoneName)); + expect(app.synth().manifest.missing ?? []).toEqual([]); + }); + + test.each(['agentcore', 'ecs', 'lambda-microvm'])('exercises supplemental resources together on the widest %s profile', compute => { + expect(profiles).toContainEqual(expect.objectContaining({ + context: expect.objectContaining({ + compute_types: compute, + enableToolGateway: true, + enableAgentRegistry: true, + enableLinearIdentityVault: true, + alertEmail: 'census@example.com', + forkBlueprintRepo: 'example/census-blueprints', + }), + })); + expect(profiles.some(p => p.context.linearVaultHostedReturnUrl)).toBe(true); + }); + + test('disables real CDK asset copying so a census does not duplicate dependency archives per profile', () => { + const app = new App({ autoSynth: false, postCliContext: STRUCTURAL_CONTEXT }); + const stack = new Stack(app, 'AssetFixture'); + const asset = new AssetStaging(stack, 'Source', { sourcePath: __dirname }); + expect(asset.stagedPath).toBe(__dirname); + expect(readdirSync(app.synth().directory).filter(name => name.startsWith('asset.'))).toEqual([]); + }); + + test('isolates worker configuration and credentials while keeping metadata/bundling explicit', () => { + const environment = synthesisEnvironment({ + PATH: '/fixture/bin', + TMPDIR: '/fixture/tmp', + HOME: '/operator', + BLUEPRINT_REPO: 'operator/override', + FORK_BLUEPRINT_REPO: 'operator/fork', + AWS_REGION: 'eu-west-1', + AWS_PROFILE: 'production', + AWS_ACCESS_KEY_ID: 'not-a-credential', + CDK_CONTEXT_JSON: '{"compute_type":"ecs"}', + CDK_DEFAULT_ACCOUNT: '999999999999', + NODE_OPTIONS: '--require=unexpected.js', + }); + expect(environment).toEqual({ + PATH: '/fixture/bin', + TMPDIR: '/fixture/tmp', + AWS_REGION: FIXTURE.region, + AWS_EC2_METADATA_DISABLED: 'true', + CDK_CONTEXT_JSON: '{"aws:cdk:bundling-stacks":[]}', + }); + expect(synthesisEnvironment({})).not.toHaveProperty('PATH'); + expect(synthesisEnvironment({})).not.toHaveProperty('TMPDIR'); + }); +}); diff --git a/cdk/test/synthesis/workspace.test.ts b/cdk/test/synthesis/workspace.test.ts new file mode 100644 index 000000000..f263c8221 --- /dev/null +++ b/cdk/test/synthesis/workspace.test.ts @@ -0,0 +1,153 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import { execFileSync } from 'node:child_process'; +import { chmodSync, existsSync, mkdirSync, mkdtempSync, readFileSync, readdirSync, realpathSync, rmSync, symlinkSync, writeFileSync } from 'node:fs'; +import { devNull, tmpdir } from 'node:os'; +import * as path from 'node:path'; +import { createOutputDirectory, projectContext, sourceProvenance } from '../../src/synthesis/workspace'; + +describe('census workspace evidence', () => { + let directory: string; + let checkout: string; + function createCheckout(root: string): void { + mkdirSync(root); + // Under a Git hook, GIT_DIR and friends point at the developer's repository; + // without stripping them the fixture commands would write there instead. + const env = Object.fromEntries(Object.entries(process.env).filter(([key]) => !key.startsWith('GIT_'))); + const git = (...args: string[]) => execFileSync('git', [ + '-c', `core.hooksPath=${devNull}`, '-c', 'commit.gpgSign=false', + '-c', 'user.name=Census Test', '-c', 'user.email=census@example.com', ...args, + ], { cwd: root, env, stdio: 'pipe' }); + git('init', '--quiet', '-b', 'census-fixture'); + writeFileSync(path.join(root, 'yarn.lock'), 'fixture lock'); + writeFileSync(path.join(root, '.gitignore'), 'build/\n'); + writeFileSync(path.join(root, 'input.txt'), ''); + git('add', '.'); + git('commit', '--quiet', '-m', 'fixture'); + } + beforeEach(() => { + directory = mkdtempSync(path.join(tmpdir(), 'census-workspace-')); + checkout = path.join(directory, 'checkout'); + createCheckout(checkout); + }); + afterEach(() => { rmSync(directory, { recursive: true, force: true }); }); + + test('fixture creation under Git hook variables cannot change another repository or its index', () => { + const before = sourceProvenance(checkout); + const indexPath = path.join(checkout, '.git/index'); + const index = readFileSync(indexPath); + const inherited = { + GIT_DIR: path.join(checkout, '.git'), + GIT_COMMON_DIR: path.join(checkout, '.git'), + GIT_WORK_TREE: checkout, + GIT_INDEX_FILE: indexPath, + }; + const saved = Object.fromEntries(Object.keys(inherited).map(key => [key, process.env[key]])); + try { + for (const [key, value] of Object.entries(inherited)) process.env[key] = value; + const fixture = path.join(directory, 'hook-fixture'); + createCheckout(fixture); + expect(existsSync(path.join(fixture, '.git/HEAD'))).toBe(true); + expect(sourceProvenance(fixture).dirty).toBe(false); + expect(sourceProvenance(checkout)).toEqual(before); + expect(readFileSync(indexPath)).toEqual(index); + } finally { + for (const [key, value] of Object.entries(saved)) { + if (value === undefined) delete process.env[key]; + else process.env[key] = value; + } + } + }); + + test('detects tracked deletion even when the former bytes equal the old deletion sentinel', () => { + const before = sourceProvenance(checkout); + rmSync(path.join(checkout, 'input.txt')); + const after = sourceProvenance(checkout); + expect(after.sourceSha256).not.toBe(before.sourceSha256); + expect(before.dirty).toBe(false); + expect(after.dirty).toBe(true); + }); + + test('distinguishes a symlink target from identical ordinary file bytes', () => { + const before = sourceProvenance(checkout); + rmSync(path.join(checkout, 'input.txt')); + symlinkSync('', path.join(checkout, 'input.txt')); + expect(sourceProvenance(checkout).sourceSha256).not.toBe(before.sourceSha256); + }); + + test('detects executable changes and untracked input changes, but excludes ignored build output', () => { + const input = path.join(checkout, 'input.txt'); + chmodSync(input, 0o644); + const before = sourceProvenance(checkout); + chmodSync(input, 0o755); + const executable = sourceProvenance(checkout); + expect(executable.sourceSha256).not.toBe(before.sourceSha256); + writeFileSync(path.join(checkout, 'new\ninput.ts'), 'new input'); + const untracked = sourceProvenance(checkout); + expect(untracked.sourceSha256).not.toBe(executable.sourceSha256); + expect(untracked.fileCount).toBe(executable.fileCount + 1); + mkdirSync(path.join(checkout, 'build')); + writeFileSync(path.join(checkout, 'build/output.js'), 'ignored'); + expect(sourceProvenance(checkout).sourceSha256).toBe(untracked.sourceSha256); + }); + + test('ignores an inherited alternate Git index', () => { + const before = sourceProvenance(checkout); + const saved = process.env.GIT_INDEX_FILE; + process.env.GIT_INDEX_FILE = path.join(directory, 'foreign-index'); + try { + expect(sourceProvenance(checkout)).toEqual(before); + } finally { + if (saved === undefined) delete process.env.GIT_INDEX_FILE; + else process.env.GIT_INDEX_FILE = saved; + } + }); + + test('rejects output through a symlink to the checkout before writing anything', () => { + const alias = path.join(directory, 'alias'); + symlinkSync(checkout, alias); + const before = readdirSync(checkout); + expect(() => createOutputDirectory(checkout, path.join(alias, 'output'))).toThrow(/outside the checkout/); + expect(() => createOutputDirectory(alias, path.join(checkout, 'output'))).toThrow(/outside the checkout/); + expect(() => createOutputDirectory(checkout, undefined, alias)).toThrow(/outside the checkout/); + expect(readdirSync(checkout)).toEqual(before); + }); + + test('allocates fresh external output and refuses to reuse existing directories or symlinks', () => { + const output = createOutputDirectory(checkout, path.join(directory, 'output')); + expect(output).toBe(realpathSync(path.join(directory, 'output'))); + expect(() => createOutputDirectory(checkout, output)).toThrow(/EEXIST/); + const alias = path.join(directory, 'existing-link'); + symlinkSync(checkout, alias); + expect(() => createOutputDirectory(checkout, alias)).toThrow(/EEXIST/); + const temporary = createOutputDirectory(checkout, undefined, directory); + expect(existsSync(temporary)).toBe(true); + expect(temporary).not.toBe(output); + }); + + test('reads versioned CDK context and rejects malformed context', () => { + mkdirSync(path.join(checkout, 'cdk')); + const file = path.join(checkout, 'cdk/cdk.json'); + writeFileSync(file, JSON.stringify({ context: { '@aws-cdk/core:fixture': true, 'bedrockGeoRegion': 'global' } })); + expect(projectContext(checkout)).toEqual({ '@aws-cdk/core:fixture': true, 'bedrockGeoRegion': 'global' }); + writeFileSync(file, JSON.stringify({ context: [] })); + expect(() => projectContext(checkout)).toThrow(/context must be an object/); + }); +}); diff --git a/cli/src/commands/repo.ts b/cli/src/commands/repo.ts index 86ef9fe5e..7f50b3b0c 100644 --- a/cli/src/commands/repo.ts +++ b/cli/src/commands/repo.ts @@ -18,7 +18,7 @@ */ import { Command } from 'commander'; -import { assertComputeSubstrateDeployed } from '../compute-substrate'; +import { assertComputeSubstrateDeployed, defaultComputeType } from '../compute-substrate'; import { CliError } from '../errors'; import { assertModelIdUsable } from '../model-id'; import { DEFAULT_STACK_NAME, redactSecretArn, resolveOperatorContext } from '../operator-context'; @@ -124,13 +124,17 @@ export function makeRepoCommand(): Command { } const config = await loadRepoConfig(region, tableName, repoId); - const [platformTokenArn, runtimeArn] = await Promise.all([ + const [platformTokenArn, runtimeArn, computeSubstrate, computeDeploymentMode, computeTypes] = await Promise.all([ getStackOutput(region, stackName, 'GitHubTokenSecretArn'), getStackOutput(region, stackName, 'RuntimeArn'), + getStackOutput(region, stackName, 'ComputeSubstrate'), + getStackOutput(region, stackName, 'ComputeDeploymentMode'), + getStackOutput(region, stackName, 'ComputeTypes'), ]); const display = formatRepoConfigForDisplay(config, { githubTokenSecretArn: platformTokenArn, runtimeArn, + deployment: { stackName, computeSubstrate, computeDeploymentMode, computeTypes }, }); if (opts.output === 'json') { @@ -173,7 +177,7 @@ export function makeRepoCommand(): Command { const { region, stackName } = resolveOperatorContext(opts); const [ tableName, platformRuntimeArn, platformGithubTokenSecretArn, computeSubstrate, deployedGeo, - grantedModelIds, + grantedModelIds, computeDeploymentMode, computeTypes, ] = await Promise.all([ getStackOutput(region, stackName, 'RepoTableName'), getStackOutput(region, stackName, 'RuntimeArn'), @@ -181,33 +185,17 @@ export function makeRepoCommand(): Command { getStackOutput(region, stackName, 'ComputeSubstrate'), getStackOutput(region, stackName, 'BedrockGeoRegion'), getStackOutput(region, stackName, 'BedrockModelIds'), + getStackOutput(region, stackName, 'ComputeDeploymentMode'), + getStackOutput(region, stackName, 'ComputeTypes'), ]); if (!tableName) { throw new CliError( `Stack '${stackName}' is missing output 'RepoTableName'. Re-deploy the CDK stack.`, ); } - // Refuse to onboard a repo onto a compute backend the deployed stack did - // NOT provision — otherwise every task on this repo fails at session - // start ("ECS compute strategy requires ECS_CLUSTER_ARN…" for ecs, or the - // MicroVM strategy's "deployed without the Lambda MicroVMs substrate" for - // lambda-microvm). Catch it here, at config time, with a fixable message. - // - // ORDERING — this runs BEFORE `onboardRepo`, and that is deliberate: - // `onboardRepo` performs the live `ListManagedMicrovmImages` regional - // availability probe for lambda-microvm. The substrate gate is both - // CHEAPER (it reuses the `ComputeSubstrate` output already fetched in the - // Promise.all above — zero extra API calls, no extra IAM) and MORE - // SPECIFIC (a stack with no MicroVM substrate cannot run the backend even - // in a Region that supports it, whereas the reverse cannot happen: the - // synth-time Region gate means a stack carrying the MicroVM substrate is - // already in a supported Region). Reporting "this stack has no MicroVM - // substrate" beats reporting "MicroVMs are unavailable in this Region" - // when both are true — the first names the actual fix. - // - // See `assertComputeSubstrateDeployed` for the ComputeSubstrate output's - // exact semantics (single-valued today, list-tolerant by construction). - assertComputeSubstrateDeployed({ stackName, computeType: opts.computeType, computeSubstrate }); + const deployment = { stackName, computeSubstrate, computeDeploymentMode, computeTypes }; + // Check explicit input early; onboardRepo also checks any stored override. + assertComputeSubstrateDeployed({ ...deployment, computeType: opts.computeType }); // Same reasoning as the substrate gate above: reuse an output already // fetched, and fail here rather than let a task die at turn 0 with an @@ -231,6 +219,7 @@ export function makeRepoCommand(): Command { const config = await onboardRepo(region, tableName, repoId, { computeType: opts.computeType, + deployment, runtimeArn: opts.runtimeArn, modelId, githubTokenSecretArn: opts.tokenSecretArn, @@ -241,6 +230,7 @@ export function makeRepoCommand(): Command { config, platformRuntimeArn, platformGithubTokenSecretArn, + defaultComputeType: defaultComputeType(deployment), }); if (opts.output === 'json') { diff --git a/cli/src/commands/runtime.ts b/cli/src/commands/runtime.ts index 78088cb85..650ef6aa7 100644 --- a/cli/src/commands/runtime.ts +++ b/cli/src/commands/runtime.ts @@ -41,9 +41,12 @@ export function makeRuntimeCommand(): Command { .action(async (opts) => { if (opts.repo) assertRepoFormat(opts.repo); const { region, stackName } = resolveOperatorContext(opts); - const [repoTableName, platformRuntimeArn] = await Promise.all([ + const [repoTableName, platformRuntimeArn, computeSubstrate, computeDeploymentMode, computeTypes] = await Promise.all([ getStackOutput(region, stackName, 'RepoTableName'), getStackOutput(region, stackName, 'RuntimeArn'), + getStackOutput(region, stackName, 'ComputeSubstrate'), + getStackOutput(region, stackName, 'ComputeDeploymentMode'), + getStackOutput(region, stackName, 'ComputeTypes'), ]); if (!repoTableName) { throw new CliError( @@ -55,7 +58,7 @@ export function makeRuntimeCommand(): Command { region, repoTableName, platformRuntimeArn, - { repo: opts.repo }, + { repo: opts.repo, deployment: { stackName, computeSubstrate, computeDeploymentMode, computeTypes } }, ); if (opts.output === 'json') { @@ -64,7 +67,12 @@ export function makeRuntimeCommand(): Command { } console.log('Runtime status is resolved per blueprint (RepoTable) with platform defaults.'); - console.log(`Platform default RuntimeArn: ${platformRuntimeArn ?? '(stack output missing)'}`); + const selectedComputeType = report.compute_deployment.default_compute_type; + console.log(`Platform default compute: ${selectedComputeType}`); + console.log(`Compute deployment mode: ${report.compute_deployment.compute_deployment_mode ?? 'legacy additive'}`); + console.log(`ComputeSubstrate: ${report.compute_deployment.compute_substrate ?? '(stack output missing)'}`); + console.log(`ComputeTypes: ${report.compute_deployment.compute_types?.join(', ') ?? '(legacy stack output missing)'}`); + if (selectedComputeType === 'agentcore') console.log(`Platform default RuntimeArn: ${platformRuntimeArn ?? '(stack output missing)'}`); console.log(); if (report.blueprints.length === 0) { @@ -72,19 +80,21 @@ export function makeRuntimeCommand(): Command { return; } - console.log('Per-blueprint effective compute:'); + console.log('Per-blueprint compute configuration:'); console.log( `${'REPO'.padEnd(REPO_WIDTH)} ${'STATUS'.padEnd(10)} ` + `${'COMPUTE'.padEnd(COMPUTE_WIDTH)} RUNTIME_ARN (source)`, ); for (const b of report.blueprints) { - const runtimeLabel = b.runtime_arn - ? `${b.runtime_arn} (${b.runtime_arn_source})` - : b.compute_type === 'ecs' - ? '(n/a — ECS uses platform cluster)' - : b.compute_type === 'lambda-microvm' - ? '(n/a — Lambda MicroVMs are platform-managed)' - : '(missing)'; + const runtimeLabel = !b.compute_available + ? `UNAVAILABLE: ${b.configuration_error}` + : b.runtime_arn + ? `${b.runtime_arn} (${b.runtime_arn_source})` + : b.compute_type === 'ecs' + ? '(n/a — ECS uses platform cluster)' + : b.compute_type === 'lambda-microvm' + ? '(n/a — Lambda MicroVMs are platform-managed)' + : '(missing)'; console.log( `${b.repo.padEnd(REPO_WIDTH)} ${b.status.padEnd(10)} ` + `${b.compute_type.padEnd(COMPUTE_WIDTH)} ${runtimeLabel}`, diff --git a/cli/src/compute-substrate.ts b/cli/src/compute-substrate.ts index 7572d0989..a8252aa42 100644 --- a/cli/src/compute-substrate.ts +++ b/cli/src/compute-substrate.ts @@ -19,125 +19,106 @@ import { CliError } from './errors'; -/** Per-repo compute backend, mirrored from `cdk/src/handlers/shared/repo-config.ts`. */ +/** Mirrored from cdk/src/handlers/shared/compute-backend.ts. */ export type OnboardComputeType = 'agentcore' | 'ecs' | 'lambda-microvm'; -/** - * Parse the stack's `ComputeSubstrate` output into the set of OPTIONAL substrates - * the deploy provisioned. - * - * ## What the output actually contains today: ONE value - * - * `cdk/src/stacks/agent.ts` emits - * `ecsCluster ? 'ecs' : (lambdaMicrovm ? 'lambda-microvm' : 'agentcore')`, and both - * constructs are gated on the SAME single-valued `compute_type` deploy context - * (`--context compute_type=…`). So the two optional backends are **mutually - * exclusive today** — a mixed `ecs` + `lambda-microvm` deploy is not - * expressible, which is why that ternary can never have to arbitrate, and why - * the CDK test asserts "does NOT provision the ECS substrate (the gates are - * mutually exclusive)" on a MicroVM stack. - * - * The three reachable values are therefore: - * - * | Output | Substrates available to tasks | - * |---|---| - * | `agentcore` | AgentCore only | - * | `ecs` | AgentCore **and** ECS (the optional backends are additive) | - * | `lambda-microvm` | AgentCore **and** Lambda MicroVMs | - * - * ## Why this parses a LIST anyway - * - * ADR-021 sub-decision 4 explicitly flags the single-value tag as "already - * imprecise with two backends, wrong with three" and names a `compute_types` list - * as the intended follow-up. If that lands, a stack would emit - * `ecs,lambda-microvm` — and an `!== 'ecs'` equality check would then start - * REFUSING valid onboardings, silently, in the safe-looking direction. Splitting - * on commas makes that future value work correctly with no change here, while - * being byte-identical in behaviour for the single values above. - * - * @param raw - the raw `ComputeSubstrate` output value, or null when absent. - * @returns the provisioned substrate names, or `undefined` when the output is - * missing/blank (an older stack predating the output — "unknown", not "none"). - */ -export function parseComputeSubstrateOutput( - raw: string | null | undefined, -): readonly string[] | undefined { - if (raw === null || raw === undefined) { - return undefined; - } - const values = raw.split(',').map((value) => value.trim()).filter(Boolean); - return values.length > 0 ? values : undefined; +export interface ComputeDeployment { + readonly stackName: string; + readonly computeSubstrate: string | null | undefined; + readonly computeDeploymentMode?: string | null; + /** Complete ordered list; its first entry is the repository default. */ + readonly computeTypes?: string | null; } -/** Backend-specific remedy copy for {@link assertComputeSubstrateDeployed}. */ -const SUBSTRATE_REMEDIES: Record, { - readonly label: string; - readonly context: string; - readonly adds: string; - readonly runtimeFailure: string; -}> = { - 'ecs': { - label: 'ECS', - context: '`--context compute_type=ecs`', - adds: 'adds the Fargate substrate alongside AgentCore', - runtimeFailure: 'fail at task start', - }, - 'lambda-microvm': { - label: 'Lambda MicroVMs', - context: '`--context compute_type=lambda-microvm`', - adds: 'adds the Lambda MicroVMs substrate alongside AgentCore', - // More specific than the ECS wording because the MicroVM failure surfaces - // later and less legibly: the strategy's own env-var guard fires first if no - // image is configured, and otherwise RunMicrovm rejects the call. - runtimeFailure: 'fail at session start (no MICROVM_* configuration on the orchestrator)', - }, -}; +export interface ComputeDeploymentStatus { + readonly stack_name: string; + readonly compute_substrate: string | null; + readonly compute_deployment_mode: string | null; + readonly compute_types: readonly OnboardComputeType[] | null; + readonly default_compute_type: OnboardComputeType; +} -/** - * Refuse to onboard a repo onto a compute backend the deployed stack never - * provisioned. - * - * Without this the row is written happily and every task on that repo dies at - * session start — for `ecs` with "ECS compute strategy requires ECS_CLUSTER_ARN…", - * for `lambda-microvm` with the strategy's "deployed without the Lambda MicroVMs - * substrate" error (or, if an image somehow IS configured, a `RunMicrovm` - * rejection). Catching it here turns a per-task runtime failure into one - * config-time message with a fixable remedy. - * - * Two deliberate non-strictnesses, both carried over from the original ECS check: - * - * - **`agentcore` is never gated.** The AgentCore runtime is unconditional; the - * other two backends are additive on top of it. - * - **An absent output means "unknown", not "none".** Stacks deployed before - * `ComputeSubstrate` existed return null, and hard-blocking there would break - * onboarding against a perfectly good older deploy. The runtime error remains - * the backstop in that case. - * - * @param args.stackName - stack the outputs were read from, for the message. - * @param args.computeType - the backend the operator asked for. - * @param args.computeSubstrate - raw `ComputeSubstrate` stack output (or null). - * @throws CliError when the requested backend is definitely not deployed. - */ -export function assertComputeSubstrateDeployed(args: { - stackName: string; - computeType: OnboardComputeType | undefined; - computeSubstrate: string | null | undefined; -}): void { - const { stackName, computeType, computeSubstrate } = args; - if (!computeType || computeType === 'agentcore') { - return; +/** Availability describes the deployment contract, not live backend health. */ +export interface RepositoryComputeBinding { + readonly compute_type: string; + readonly compute_available: boolean; + readonly configuration_error?: string; +} + +/** Older deployments advertised optional backends as a comma-separated list. */ +export function parseComputeSubstrateOutput(raw: string | null | undefined): readonly string[] | undefined { + const values = raw?.split(',').map(value => value.trim()).filter(Boolean); + return values?.length ? values : undefined; +} + +type ComputeOutputs = Pick; + +/** Explicit outputs are authoritative; only stacks without them imply AgentCore. */ +function declaredComputeTypes(deployment: ComputeOutputs): readonly OnboardComputeType[] | undefined { + const mode = deployment.computeDeploymentMode; + if (mode != null && mode !== 'exclusive' && mode !== 'additive') { + throw new CliError(`Unknown ComputeDeploymentMode '${mode}'. Update the CLI or re-deploy the CDK stack.`); + } + if (deployment.computeTypes == null && mode == null) return undefined; + const output = deployment.computeTypes != null ? 'ComputeTypes' : 'ComputeSubstrate'; + const raw = deployment.computeTypes ?? deployment.computeSubstrate; + const values = raw?.split(',').map(value => value.trim()); + if (!values?.length || values.some(value => !['agentcore', 'ecs', 'lambda-microvm'].includes(value)) + || new Set(values).size !== values.length + || (mode === 'exclusive' && values.length !== 1) + || (mode === 'additive' && values.length < 2)) { + throw new CliError(`Compute deployment has an invalid or missing ${output} output. Re-deploy the CDK stack.`); } + if (deployment.computeTypes != null && deployment.computeSubstrate != null + && values.join(',') !== deployment.computeSubstrate.split(',').map(value => value.trim()).join(',')) { + throw new CliError('ComputeTypes and ComputeSubstrate outputs disagree. Re-deploy the CDK stack before changing repository configuration.'); + } + return values as OnboardComputeType[]; +} - const provisioned = parseComputeSubstrateOutput(computeSubstrate); - if (!provisioned || provisioned.includes(computeType)) { - return; +export function defaultComputeType(deployment: ComputeOutputs): OnboardComputeType { + return declaredComputeTypes(deployment)?.[0] ?? 'agentcore'; +} + +export function describeComputeDeployment(deployment: ComputeDeployment): ComputeDeploymentStatus { + return { + stack_name: deployment.stackName, + compute_substrate: deployment.computeSubstrate ?? null, + compute_deployment_mode: deployment.computeDeploymentMode ?? null, + compute_types: declaredComputeTypes(deployment) ?? null, + default_compute_type: defaultComputeType(deployment), + }; +} + +function computeConfigurationError(args: ComputeDeployment & { computeType: string | undefined }): string | undefined { + const declared = declaredComputeTypes(args); + const selected = declared?.[0] ?? 'agentcore'; + const requested = args.computeType ?? selected; + if (!['agentcore', 'ecs', 'lambda-microvm'].includes(requested)) { + return `Unsupported repository compute_type '${requested}'. Choose agentcore, ecs or lambda-microvm.`; } + if (declared) { + if (declared.includes(requested as OnboardComputeType)) return; + return `Stack '${args.stackName}' deploys only '${declared.join(', ')}' (ComputeSubstrate=${args.computeSubstrate}); --compute-type ${requested} is unavailable. Use --compute-type ${selected}, or add the backend with --context compute_types=${[...declared, requested].join(',')} and review the deployment change set.`; + } + const provisioned = parseComputeSubstrateOutput(args.computeSubstrate); + if (requested === 'agentcore' || !provisioned || provisioned.includes(requested)) return; + const label = requested === 'ecs' ? 'ECS' : 'Lambda MicroVMs'; + return `Stack '${args.stackName}' was deployed without the ${label} substrate (ComputeSubstrate=${args.computeSubstrate}), so a repo onboarded as --compute-type ${requested} would fail at session start. Redeploy the stack with --context compute_type=${requested} first, then re-run this — or onboard with --compute-type agentcore.`; +} + +/** Report each stale pin without preventing inspection of other repositories. */ +export function resolveRepositoryCompute(deployment: ComputeDeployment, computeType?: string): RepositoryComputeBinding { + const configurationError = computeConfigurationError({ ...deployment, computeType }); + return { + compute_type: computeType ?? defaultComputeType(deployment), + compute_available: configurationError === undefined, + ...(configurationError ? { configuration_error: configurationError } : {}), + }; +} - const remedy = SUBSTRATE_REMEDIES[computeType]; - throw new CliError( - `Stack '${stackName}' was deployed without the ${remedy.label} substrate ` - + `(ComputeSubstrate=${computeSubstrate}), so a repo onboarded as --compute-type ${computeType} ` - + `would ${remedy.runtimeFailure}. Redeploy the stack with ${remedy.context} first ` - + `(${remedy.adds}), then re-run this — or onboard with --compute-type agentcore.`, - ); +/** Reject incompatible repository pins before writing RepoTable or probing MicroVM availability. */ +export function assertComputeSubstrateDeployed(args: ComputeDeployment & { computeType: string | undefined }): void { + const configurationError = computeConfigurationError(args); + if (configurationError) throw new CliError(configurationError); } diff --git a/cli/src/repo-display.ts b/cli/src/repo-display.ts index 58810c554..768656ab8 100644 --- a/cli/src/repo-display.ts +++ b/cli/src/repo-display.ts @@ -17,6 +17,12 @@ * SOFTWARE. */ +import { + describeComputeDeployment, + resolveRepositoryCompute, + type ComputeDeployment, + type ComputeDeploymentStatus, +} from './compute-substrate'; import type { GithubTokenSecretSource } from './github-token'; import { redactSecretArn } from './operator-context'; import { RepoConfigRow } from './repo-lookup'; @@ -25,6 +31,7 @@ export type FieldSource = 'blueprint' | 'platform'; /** Stack outputs + constants used to resolve platform defaults for display. */ export interface PlatformStackContext { + readonly deployment: ComputeDeployment; readonly runtimeArn: string | null; readonly githubTokenSecretArn: string | null; } @@ -60,9 +67,12 @@ export interface RepoConfigDisplay { readonly status: RepoConfigRow['status']; readonly onboarded_at?: string; readonly updated_at?: string; + readonly compute_deployment: ComputeDeploymentStatus; + readonly compute_available: boolean; + readonly configuration_error?: string; /** Raw RepoTable values (absent when the Blueprint did not override). */ readonly blueprint_overrides: Record; - /** Values used at task time after merging with platform defaults. */ + /** Resolved configuration; dispatch requires compute_available to be true. */ readonly effective: { readonly compute_type: string; readonly runtime_arn?: string; @@ -128,15 +138,20 @@ export function formatRepoConfigForDisplay( } } + const compute = resolveRepositoryCompute(platform.deployment, config.compute_type); return { repo: config.repo, status: config.status, onboarded_at: config.onboarded_at, updated_at: config.updated_at, + compute_deployment: describeComputeDeployment(platform.deployment), + compute_available: compute.compute_available, + configuration_error: compute.configuration_error, blueprint_overrides: blueprintOverrides, effective: { - compute_type: config.compute_type ?? PLATFORM_REPO_DEFAULTS.compute_type, - runtime_arn: config.runtime_arn ?? platform.runtimeArn ?? undefined, + compute_type: compute.compute_type, + runtime_arn: compute.compute_available && compute.compute_type === 'agentcore' + ? config.runtime_arn ?? platform.runtimeArn ?? undefined : undefined, model_id: config.model_id ?? PLATFORM_REPO_DEFAULTS.model_id, max_turns: config.max_turns ?? PLATFORM_REPO_DEFAULTS.max_turns, max_budget_usd: config.max_budget_usd !== undefined @@ -166,15 +181,23 @@ export function buildRepoShowLines(display: RepoConfigDisplay): RepoShowLine[] { { key: 'status', text: display.status }, { key: 'onboarded_at', text: display.onboarded_at ?? '-' }, { key: 'updated_at', text: display.updated_at ?? '-' }, + { + key: 'deployed_compute_types', + text: display.compute_deployment.compute_types?.join(', ') ?? '(legacy additive deployment)', + }, { key: 'compute_type', - text: formatSourcedValue(display.effective.compute_type, display.field_sources.compute_type), + text: `${display.compute_available ? '' : 'UNAVAILABLE — '}${formatSourcedValue(display.effective.compute_type, display.field_sources.compute_type)}`, }, { key: 'runtime_arn', - text: display.effective.runtime_arn - ? formatSourcedValue(display.effective.runtime_arn, display.field_sources.runtime_arn) - : '(platform default — RuntimeArn stack output not found)', + text: !display.compute_available + ? '(unavailable — repository compute configuration is incompatible)' + : display.effective.runtime_arn + ? formatSourcedValue(display.effective.runtime_arn, display.field_sources.runtime_arn) + : display.effective.compute_type === 'agentcore' + ? '(platform default — RuntimeArn stack output not found)' + : `(not applicable — ${display.effective.compute_type} uses platform compute)`, }, { key: 'model_id', @@ -206,6 +229,10 @@ export function buildRepoShowLines(display: RepoConfigDisplay): RepoShowLine[] { }, ]; + if (display.configuration_error) { + lines.push({ key: 'configuration_error', text: display.configuration_error }); + } + if (typeof display.blueprint_overrides.system_prompt_overrides === 'string') { lines.push({ key: 'system_prompt_overrides', diff --git a/cli/src/repo-onboard-notes.ts b/cli/src/repo-onboard-notes.ts index f6c1e9c3c..6a1249b9f 100644 --- a/cli/src/repo-onboard-notes.ts +++ b/cli/src/repo-onboard-notes.ts @@ -17,9 +17,11 @@ * SOFTWARE. */ +import type { OnboardComputeType } from './compute-substrate'; import { RepoConfigRow } from './repo-lookup'; export interface RepoOnboardNotesInput { + readonly defaultComputeType?: OnboardComputeType; readonly config: RepoConfigRow; readonly platformRuntimeArn: string | null; readonly platformGithubTokenSecretArn: string | null; @@ -29,12 +31,13 @@ export interface RepoOnboardNotesInput { export function buildRepoOnboardNotes(input: RepoOnboardNotesInput): readonly string[] { const notes: string[] = [ 'This command writes RepoTable only. With no per-repo overrides, tasks inherit the ' - + 'platform RuntimeArn and GitHubTokenSecretArn (IAM for those is granted at CDK deploy).', + + `platform compute backend (${input.defaultComputeType ?? 'agentcore'}) and GitHubTokenSecretArn (IAM for those is granted at CDK deploy).`, 'For Cedar policies, egress rules, custom runtime/token IAM, and durable lifecycle, ' + 'prefer a CDK Blueprint construct and `mise //cdk:deploy`.', ]; - const customRuntime = input.config.runtime_arn; + const computeType = input.config.compute_type ?? input.defaultComputeType ?? 'agentcore'; + const customRuntime = computeType === 'agentcore' ? input.config.runtime_arn : undefined; if (customRuntime && customRuntime !== input.platformRuntimeArn) { notes.push( 'WARNING: A custom runtime_arn is stored. The orchestrator Lambda must be granted ' @@ -52,14 +55,14 @@ export function buildRepoOnboardNotes(input: RepoOnboardNotesInput): readonly st ); } - if (input.config.compute_type === 'ecs') { + if (computeType === 'ecs') { notes.push( 'NOTE: compute_type=ecs requires ECS wired into the stack (TaskOrchestrator ecsConfig). ' + 'Verify your CDK stack before submitting tasks.', ); } - if (input.config.compute_type === 'lambda-microvm') { + if (computeType === 'lambda-microvm') { // Mirrors the ECS note, plus the one thing that has no ECS analogue: the // substrate can be fully deployed and still carry no IMAGE (ADR-021's // three-state table — the artifact bucket must exist before the artifact can diff --git a/cli/src/repo-onboard.ts b/cli/src/repo-onboard.ts index 4e30009fd..7f4246030 100644 --- a/cli/src/repo-onboard.ts +++ b/cli/src/repo-onboard.ts @@ -18,7 +18,9 @@ */ import { PutCommand, UpdateCommand } from '@aws-sdk/lib-dynamodb'; +import { assertComputeSubstrateDeployed, defaultComputeType, type ComputeDeployment } from './compute-substrate'; import { documentClient } from './dynamo-clients'; +import { CliError } from './errors'; import { LambdaMicrovmProbeClientFactory, requireLambdaMicrovmAvailability, @@ -35,6 +37,7 @@ import { export const REMOVED_REPO_TTL_DAYS = 30; export interface OnboardRepoOptions { + readonly deployment?: ComputeDeployment; readonly computeType?: 'agentcore' | 'ecs' | 'lambda-microvm'; readonly runtimeArn?: string; readonly modelId?: string; @@ -73,7 +76,14 @@ export async function onboardRepo( existing = undefined; } - const effectiveComputeType = options.computeType ?? existing?.compute_type ?? 'agentcore'; + const effectiveComputeType = options.computeType ?? existing?.compute_type + ?? (options.deployment ? defaultComputeType(options.deployment) : 'agentcore'); + if (options.deployment) { + assertComputeSubstrateDeployed({ ...options.deployment, computeType: effectiveComputeType }); + } + if (options.runtimeArn && effectiveComputeType !== 'agentcore') { + throw new CliError('--runtime-arn applies only to the agentcore backend'); + } if (effectiveComputeType === 'lambda-microvm') { await requireLambdaMicrovmAvailability(region, dependencies.lambdaMicrovmClientFactory); } diff --git a/cli/src/runtime-status.ts b/cli/src/runtime-status.ts index 53e2776de..bea40b74b 100644 --- a/cli/src/runtime-status.ts +++ b/cli/src/runtime-status.ts @@ -21,14 +21,19 @@ import { BedrockAgentCoreControlClient, GetAgentRuntimeCommand, } from '@aws-sdk/client-bedrock-agentcore-control'; -import { PLATFORM_REPO_DEFAULTS } from './repo-display'; +import { + describeComputeDeployment, + resolveRepositoryCompute, + type ComputeDeployment, + type ComputeDeploymentStatus, + type RepositoryComputeBinding, +} from './compute-substrate'; import { listRepoConfigs, RepoConfigRow } from './repo-lookup'; import { makeClient } from './ua'; -interface BlueprintRuntimeBinding { +interface BlueprintRuntimeBinding extends RepositoryComputeBinding { readonly repo: string; readonly status: RepoConfigRow['status']; - readonly compute_type: string; readonly runtime_arn?: string; readonly runtime_arn_source: 'blueprint' | 'platform'; } @@ -72,6 +77,7 @@ interface LambdaMicrovmSubstrateSummary { } export interface RuntimeStatusReport { + readonly compute_deployment: ComputeDeploymentStatus; readonly platform_default_runtime_arn: string | null; readonly blueprints: readonly BlueprintRuntimeBinding[]; readonly agentcore_runtimes: readonly RuntimeProbeResult[]; @@ -95,17 +101,18 @@ export function parseAgentRuntimeArn(runtimeArn: string): { agentRuntimeId: stri function bindingForRepo( config: RepoConfigRow, platformRuntimeArn: string | null, + deployment: ComputeDeployment, ): BlueprintRuntimeBinding { - const computeType = config.compute_type ?? PLATFORM_REPO_DEFAULTS.compute_type; + const compute = resolveRepositoryCompute(deployment, config.compute_type); const hasBlueprintRuntime = config.runtime_arn !== undefined; - const runtimeArn = computeType === 'agentcore' + const runtimeArn = compute.compute_available && compute.compute_type === 'agentcore' ? hasBlueprintRuntime ? config.runtime_arn : platformRuntimeArn ?? undefined : undefined; return { repo: config.repo, status: config.status, - compute_type: computeType, + ...compute, runtime_arn: runtimeArn, runtime_arn_source: hasBlueprintRuntime ? 'blueprint' : 'platform', }; @@ -153,21 +160,22 @@ export async function buildRuntimeStatusReport( region: string, repoTableName: string, platformRuntimeArn: string | null, - options: { readonly repo?: string } = {}, + options: { readonly repo?: string; readonly deployment: ComputeDeployment }, ): Promise { + const computeDeployment = describeComputeDeployment(options.deployment); let repos = await listRepoConfigs(region, repoTableName); if (options.repo) { repos = repos.filter((r) => r.repo === options.repo); } - const blueprints = repos.map((r) => bindingForRepo(r, platformRuntimeArn)); + const blueprints = repos.map((r) => bindingForRepo(r, platformRuntimeArn, options.deployment)); const agentcoreMap = new Map(); const ecsRepos: string[] = []; const lambdaMicrovmRepos: string[] = []; for (const binding of blueprints) { - if (binding.status !== 'active') continue; + if (binding.status !== 'active' || !binding.compute_available) continue; if (binding.compute_type === 'ecs') { ecsRepos.push(binding.repo); continue; @@ -207,6 +215,7 @@ export async function buildRuntimeStatusReport( : []; return { + compute_deployment: computeDeployment, platform_default_runtime_arn: platformRuntimeArn, blueprints, agentcore_runtimes, diff --git a/cli/test/commands/repo-display.test.ts b/cli/test/commands/repo-display.test.ts index 6d45cf7e0..b7b3da595 100644 --- a/cli/test/commands/repo-display.test.ts +++ b/cli/test/commands/repo-display.test.ts @@ -27,6 +27,7 @@ import { } from '../../src/repo-display'; const PLATFORM = { + deployment: { stackName: 'backgroundagent-dev', computeSubstrate: null }, runtimeArn: 'arn:aws:bedrock:us-east-1:123456789012:runtime/test', githubTokenSecretArn: 'arn:aws:secretsmanager:us-east-1:123456789012:secret:GitHubTokenSecret-AbCdEf', }; @@ -46,6 +47,48 @@ describe('formatRepoConfigForDisplay', () => { expect(Object.keys(display.blueprint_overrides)).toHaveLength(0); }); + test.each(['agentcore', 'ecs', 'lambda-microvm'] as const)( + 'checks every repository pin against the exclusive %s deployment', + backend => { + const platform = { + ...PLATFORM, + deployment: { stackName: 'backgroundagent-dev', computeSubstrate: backend, computeDeploymentMode: 'exclusive' }, + }; + const inherited = formatRepoConfigForDisplay({ repo: 'acme/default', status: 'active' }, platform); + expect(inherited.effective.compute_type).toBe(backend); + expect(inherited.compute_available).toBe(true); + expect(inherited.compute_deployment).toEqual({ + stack_name: 'backgroundagent-dev', + compute_substrate: backend, + compute_deployment_mode: 'exclusive', + compute_types: [backend], + default_compute_type: backend, + }); + + for (const requested of ['agentcore', 'ecs', 'lambda-microvm'] as const) { + const display = formatRepoConfigForDisplay({ + repo: 'acme/pinned', + status: 'active', + compute_type: requested, + runtime_arn: 'arn:aws:bedrock-agentcore:us-east-1:123:runtime/custom', + }, platform); + expect(display.blueprint_overrides.compute_type).toBe(requested); + expect(display.compute_available).toBe(requested === backend); + const lines = buildRepoShowLines(display); + if (requested === backend) { + expect(display.configuration_error).toBeUndefined(); + expect(lines.find(line => line.key === 'compute_type')?.text).not.toContain('UNAVAILABLE'); + } else { + expect(display.configuration_error).toContain(`deploys only '${backend}'`); + expect(display.effective.runtime_arn).toBeUndefined(); + expect(lines.find(line => line.key === 'compute_type')?.text).toContain('UNAVAILABLE'); + expect(lines.find(line => line.key === 'runtime_arn')?.text).toContain('unavailable'); + expect(lines.find(line => line.key === 'configuration_error')?.text).toBe(display.configuration_error); + } + } + }, + ); + test('marks blueprint override when github_token_secret_arn is set', () => { const display = formatRepoConfigForDisplay( { @@ -160,7 +203,7 @@ describe('buildRepoShowLines', () => { test('warns when platform stack output is missing', () => { const display = formatRepoConfigForDisplay( { repo: 'awslabs/agent-plugins', status: 'active' }, - { runtimeArn: null, githubTokenSecretArn: null }, + { ...PLATFORM, runtimeArn: null, githubTokenSecretArn: null }, ); expect(formatGithubTokenSecretLine(display)) diff --git a/cli/test/commands/repo-onboard.test.ts b/cli/test/commands/repo-onboard.test.ts index d6114e3a7..8f0d9f2b9 100644 --- a/cli/test/commands/repo-onboard.test.ts +++ b/cli/test/commands/repo-onboard.test.ts @@ -132,6 +132,45 @@ describe('repo onboard/offboard', () => { expect(ddbSend).not.toHaveBeenCalled(); }); + test.each(['lambda-microvm', 'lambda-microvm,ecs'])('inherits %s and probes availability without persisting a default pin', async computeTypes => { + const send = jest.fn().mockResolvedValue({ images: [] }); + const config = await onboardRepo('us-east-1', 'RepoTable', 'acme/a', { + deployment: { + stackName: 'test', + computeTypes, + computeSubstrate: computeTypes, + computeDeploymentMode: computeTypes.includes(',') ? 'additive' : 'exclusive', + }, + }, { lambdaMicrovmClientFactory: () => ({ send }) }); + expect(send).toHaveBeenCalledTimes(1); + expect(config.compute_type).toBeUndefined(); + }); + + test('rejects an AgentCore pin when the additive deployment omits AgentCore', async () => { + const { loadRepoConfig } = jest.requireMock('../../src/repo-lookup') as { loadRepoConfig: jest.Mock }; + loadRepoConfig.mockResolvedValueOnce({ repo: 'acme/a', status: 'active', compute_type: 'agentcore' }); + await expect(onboardRepo('us-east-1', 'RepoTable', 'acme/a', { + deployment: { + stackName: 'test', + computeTypes: 'ecs,lambda-microvm', + computeSubstrate: 'ecs,lambda-microvm', + computeDeploymentMode: 'additive', + }, + })).rejects.toThrow(/deploys only 'ecs, lambda-microvm'/); + expect(ddbSend).not.toHaveBeenCalled(); + }); + + test('rejects stale stored overrides before a probe or write', async () => { + const { loadRepoConfig } = jest.requireMock('../../src/repo-lookup') as { loadRepoConfig: jest.Mock }; + loadRepoConfig.mockResolvedValueOnce({ repo: 'acme/a', status: 'active', compute_type: 'lambda-microvm' }); + const send = jest.fn(); + await expect(onboardRepo('us-east-1', 'RepoTable', 'acme/a', { + deployment: { stackName: 'test', computeSubstrate: 'ecs', computeDeploymentMode: 'exclusive' }, + }, { lambdaMicrovmClientFactory: () => ({ send }) })).rejects.toThrow(/deploys only 'ecs'/); + expect(send).not.toHaveBeenCalled(); + expect(ddbSend).not.toHaveBeenCalled(); + }); + test('offboardRepo sets removed status and TTL', async () => { await offboardRepo('us-east-1', 'RepoTable', 'acme/a'); diff --git a/cli/test/commands/repo.test.ts b/cli/test/commands/repo.test.ts index 0dfcd2c61..791d1ab4d 100644 --- a/cli/test/commands/repo.test.ts +++ b/cli/test/commands/repo.test.ts @@ -96,7 +96,8 @@ describe('repo command JSON output', () => { beforeEach(() => { ddbSend.mockReset(); consoleSpy = jest.spyOn(console, 'log').mockImplementation(); - getStackOutputMock.mockReset().mockResolvedValue('RepoTable-dev'); + getStackOutputMock.mockReset().mockImplementation(async (_r: string, _s: string, key: string) => + key === 'RepoTableName' ? 'RepoTable-dev' : null); onboardRepoMock.mockReset(); offboardRepoMock.mockReset(); }); @@ -147,6 +148,31 @@ describe('repo command JSON output', () => { expect(out).toContain('****'); }); + test.each(['text', 'json'])('repo show reports an incompatible pin on an exclusive stack in %s', async format => { + getStackOutputMock.mockImplementation(async (_r: string, _s: string, key: string) => ({ + RepoTableName: 'RepoTable-dev', + ComputeSubstrate: 'ecs', + ComputeDeploymentMode: 'exclusive', + } as Record)[key] ?? null); + ddbSend.mockResolvedValueOnce({ + Item: { repo: 'acme/a', status: 'active', compute_type: 'lambda-microvm' }, + }); + await makeRepoCommand().parseAsync(['node', 'test', 'show', 'acme/a', '--region', 'us-east-1', '--output', format]); + if (format === 'json') { + const display = JSON.parse(consoleSpy.mock.calls[0][0] as string); + expect(display.compute_available).toBe(false); + expect(display.configuration_error).toContain("deploys only 'ecs'"); + expect(display.compute_deployment).toMatchObject({ + compute_substrate: 'ecs', compute_deployment_mode: 'exclusive', default_compute_type: 'ecs', + }); + } else { + const output = consoleSpy.mock.calls.map(call => call[0]).join('\n'); + expect(output).toContain('UNAVAILABLE'); + expect(output).toContain("deploys only 'ecs'"); + expect(output).not.toContain('lambda-microvm uses platform compute'); + } + }); + test('repo onboard --output json redacts the per-repo secret ARN', async () => { onboardRepoMock.mockResolvedValue({ repo: 'acme/a', @@ -166,11 +192,39 @@ describe('repo command JSON output', () => { expect(payload.repo.github_token_secret_arn).toContain('****'); }); + test('repo show reads ComputeTypes and inherits a non-AgentCore additive default', async () => { + getStackOutputMock.mockImplementation(async (_r: string, _s: string, key: string) => ({ + RepoTableName: 'RepoTable-dev', + ComputeSubstrate: 'ecs,lambda-microvm', + ComputeTypes: 'ecs,lambda-microvm', + ComputeDeploymentMode: 'additive', + } as Record)[key] ?? null); + ddbSend.mockResolvedValueOnce({ Item: { repo: 'acme/a', status: 'active' } }); + await makeRepoCommand().parseAsync(['node', 'test', 'show', 'acme/a', '--region', 'us-east-1', '--output', 'json']); + const payload = JSON.parse(consoleSpy.mock.calls[0][0] as string); + expect(payload.effective.compute_type).toBe('ecs'); + expect(payload.compute_available).toBe(true); + expect(payload.compute_deployment.compute_types).toEqual(['ecs', 'lambda-microvm']); + }); + + test('onboard reads ComputeTypes before writing and refuses an omitted AgentCore backend', async () => { + getStackOutputMock.mockImplementation(async (_r: string, _s: string, key: string) => ({ + RepoTableName: 'RepoTable-dev', + ComputeSubstrate: 'ecs,lambda-microvm', + ComputeTypes: 'ecs,lambda-microvm', + ComputeDeploymentMode: 'additive', + } as Record)[key] ?? null); + await expect(makeRepoCommand().parseAsync([ + 'node', 'test', 'onboard', 'acme/a', '--region', 'us-east-1', '--compute-type', 'agentcore', + ])).rejects.toThrow(/deploys only 'ecs, lambda-microvm'/); + expect(onboardRepoMock).not.toHaveBeenCalled(); + }); + test('onboard --compute-type ecs is REFUSED when the stack has no ECS substrate', async () => { // Per-key outputs: RepoTableName present, ComputeSubstrate=agentcore (deployed // without --context compute_type=ecs). getStackOutputMock.mockReset().mockImplementation((_r: string, _s: string, key: string) => - Promise.resolve(key === 'ComputeSubstrate' ? 'agentcore' : 'RepoTable-dev')); + Promise.resolve(key === 'ComputeSubstrate' ? 'agentcore' : key === 'RepoTableName' ? 'RepoTable-dev' : null)); const cmd = makeRepoCommand(); await expect(cmd.parseAsync([ @@ -182,7 +236,7 @@ describe('repo command JSON output', () => { test('onboard --compute-type ecs is ALLOWED when the stack provisioned ECS', async () => { getStackOutputMock.mockReset().mockImplementation((_r: string, _s: string, key: string) => - Promise.resolve(key === 'ComputeSubstrate' ? 'ecs' : 'RepoTable-dev')); + Promise.resolve(key === 'ComputeSubstrate' ? 'ecs' : key === 'RepoTableName' ? 'RepoTable-dev' : null)); onboardRepoMock.mockResolvedValue({ repo: 'acme/a', status: 'active', compute_type: 'ecs' }); const cmd = makeRepoCommand(); @@ -197,7 +251,7 @@ describe('repo command JSON output', () => { // Back-compat: pre-output stacks return null for ComputeSubstrate; don't hard-block // (the runtime error is the backstop there). getStackOutputMock.mockReset().mockImplementation((_r: string, _s: string, key: string) => - Promise.resolve(key === 'ComputeSubstrate' ? null : 'RepoTable-dev')); + Promise.resolve(key === 'ComputeSubstrate' ? null : key === 'RepoTableName' ? 'RepoTable-dev' : null)); onboardRepoMock.mockResolvedValue({ repo: 'acme/a', status: 'active', compute_type: 'ecs' }); const cmd = makeRepoCommand(); @@ -209,7 +263,7 @@ describe('repo command JSON output', () => { test('onboard --compute-type agentcore is unaffected by ComputeSubstrate', async () => { getStackOutputMock.mockReset().mockImplementation((_r: string, _s: string, key: string) => - Promise.resolve(key === 'ComputeSubstrate' ? 'agentcore' : 'RepoTable-dev')); + Promise.resolve(key === 'ComputeSubstrate' ? 'agentcore' : key === 'RepoTableName' ? 'RepoTable-dev' : null)); onboardRepoMock.mockResolvedValue({ repo: 'acme/a', status: 'active', compute_type: 'agentcore' }); const cmd = makeRepoCommand(); @@ -227,7 +281,7 @@ describe('repo command JSON output', () => { test('onboard --compute-type lambda-microvm is REFUSED when the stack has no MicroVM substrate', async () => { getStackOutputMock.mockReset().mockImplementation((_r: string, _s: string, key: string) => - Promise.resolve(key === 'ComputeSubstrate' ? 'agentcore' : 'RepoTable-dev')); + Promise.resolve(key === 'ComputeSubstrate' ? 'agentcore' : key === 'RepoTableName' ? 'RepoTable-dev' : null)); const cmd = makeRepoCommand(); await expect(cmd.parseAsync([ @@ -243,7 +297,7 @@ describe('repo command JSON output', () => { // first. `onboardRepo` owns the ListManagedMicrovmImages probe, so "the probe // did not run" is exactly "onboardRepo was never called". getStackOutputMock.mockReset().mockImplementation((_r: string, _s: string, key: string) => - Promise.resolve(key === 'ComputeSubstrate' ? 'agentcore' : 'RepoTable-dev')); + Promise.resolve(key === 'ComputeSubstrate' ? 'agentcore' : key === 'RepoTableName' ? 'RepoTable-dev' : null)); onboardRepoMock.mockReset(); const cmd = makeRepoCommand(); @@ -255,7 +309,7 @@ describe('repo command JSON output', () => { test('onboard --compute-type lambda-microvm is ALLOWED when the stack provisioned it', async () => { getStackOutputMock.mockReset().mockImplementation((_r: string, _s: string, key: string) => - Promise.resolve(key === 'ComputeSubstrate' ? 'lambda-microvm' : 'RepoTable-dev')); + Promise.resolve(key === 'ComputeSubstrate' ? 'lambda-microvm' : key === 'RepoTableName' ? 'RepoTable-dev' : null)); onboardRepoMock.mockResolvedValue({ repo: 'acme/a', status: 'active', compute_type: 'lambda-microvm', }); @@ -275,7 +329,7 @@ describe('repo command JSON output', () => { // The two optional backends are mutually exclusive today, so an ecs stack is // a real (not hypothetical) way to get this wrong. getStackOutputMock.mockReset().mockImplementation((_r: string, _s: string, key: string) => - Promise.resolve(key === 'ComputeSubstrate' ? 'ecs' : 'RepoTable-dev')); + Promise.resolve(key === 'ComputeSubstrate' ? 'ecs' : key === 'RepoTableName' ? 'RepoTable-dev' : null)); const cmd = makeRepoCommand(); await expect(cmd.parseAsync([ @@ -287,7 +341,7 @@ describe('repo command JSON output', () => { test('onboard --compute-type lambda-microvm proceeds against an OLDER stack lacking ComputeSubstrate', async () => { // Same back-compat posture as the ECS gate: null → "unknown", not "none". getStackOutputMock.mockReset().mockImplementation((_r: string, _s: string, key: string) => - Promise.resolve(key === 'ComputeSubstrate' ? null : 'RepoTable-dev')); + Promise.resolve(key === 'ComputeSubstrate' ? null : key === 'RepoTableName' ? 'RepoTable-dev' : null)); onboardRepoMock.mockResolvedValue({ repo: 'acme/a', status: 'active', compute_type: 'lambda-microvm', }); @@ -303,7 +357,7 @@ describe('repo command JSON output', () => { // The effective compute type may come from the existing row, which the gate // cannot see — `onboardRepo` resolves that and runs its own probe. getStackOutputMock.mockReset().mockImplementation((_r: string, _s: string, key: string) => - Promise.resolve(key === 'ComputeSubstrate' ? 'agentcore' : 'RepoTable-dev')); + Promise.resolve(key === 'ComputeSubstrate' ? 'agentcore' : key === 'RepoTableName' ? 'RepoTable-dev' : null)); onboardRepoMock.mockResolvedValue({ repo: 'acme/a', status: 'active' }); const cmd = makeRepoCommand(); diff --git a/cli/test/commands/runtime-status.test.ts b/cli/test/commands/runtime-status.test.ts index bae2c7d55..a90335e22 100644 --- a/cli/test/commands/runtime-status.test.ts +++ b/cli/test/commands/runtime-status.test.ts @@ -21,6 +21,7 @@ import { listRepoConfigs } from '../../src/repo-lookup'; import { buildRuntimeStatusReport } from '../../src/runtime-status'; const controlPlaneSend = jest.fn(); +const LEGACY_DEPLOYMENT = { stackName: 'backgroundagent-dev', computeSubstrate: null }; jest.mock('../../src/repo-lookup'); jest.mock('@aws-sdk/client-bedrock-agentcore-control', () => ({ @@ -68,6 +69,7 @@ describe('buildRuntimeStatusReport', () => { 'us-east-1', 'RepoTable', 'arn:aws:bedrock-agentcore:us-east-1:123:runtime/platform', + { deployment: LEGACY_DEPLOYMENT }, ); expect(report.agentcore_runtimes).toHaveLength(2); @@ -83,6 +85,117 @@ describe('buildRuntimeStatusReport', () => { expect(controlPlaneSend).toHaveBeenCalledTimes(2); }); + test.each(['ecs', 'lambda-microvm'] as const)('inherits %s without probing AgentCore', async backend => { + (listRepoConfigs as jest.Mock).mockResolvedValue([{ repo: 'acme/a', status: 'active' }]); + const report = await buildRuntimeStatusReport('us-east-1', 'RepoTable', null, { + deployment: { ...LEGACY_DEPLOYMENT, computeSubstrate: backend, computeDeploymentMode: 'exclusive' }, + }); + expect(report.blueprints[0].compute_type).toBe(backend); + expect(report.blueprints[0].runtime_arn).toBeUndefined(); + expect(controlPlaneSend).not.toHaveBeenCalled(); + }); + + test.each(['agentcore', 'ecs', 'lambda-microvm'] as const)( + 'reports incompatible pins and probes only the exclusive %s deployment', + async backend => { + (listRepoConfigs as jest.Mock).mockResolvedValue( + ['agentcore', 'ecs', 'lambda-microvm'].map(compute_type => ({ + repo: `acme/${compute_type}`, + status: 'active', + compute_type, + runtime_arn: 'arn:aws:bedrock-agentcore:us-east-1:123:runtime/custom', + })), + ); + const report = await buildRuntimeStatusReport('us-east-1', 'RepoTable', null, { + deployment: { ...LEGACY_DEPLOYMENT, computeSubstrate: backend, computeDeploymentMode: 'exclusive' }, + }); + expect(report.compute_deployment).toEqual({ + stack_name: 'backgroundagent-dev', + compute_substrate: backend, + compute_deployment_mode: 'exclusive', + compute_types: [backend], + default_compute_type: backend, + }); + expect(report.blueprints).toHaveLength(3); + for (const binding of report.blueprints) { + expect(binding.compute_available).toBe(binding.compute_type === backend); + if (binding.compute_available) { + expect(binding.configuration_error).toBeUndefined(); + } else { + expect(binding.configuration_error).toContain(`deploys only '${backend}'`); + expect(binding.runtime_arn).toBeUndefined(); + } + } + expect(report.ecs_substrates).toHaveLength(backend === 'ecs' ? 1 : 0); + expect(report.lambda_microvm_substrates).toHaveLength(backend === 'lambda-microvm' ? 1 : 0); + expect(report.agentcore_runtimes).toHaveLength(backend === 'agentcore' ? 1 : 0); + expect(controlPlaneSend).toHaveBeenCalledTimes(backend === 'agentcore' ? 1 : 0); + }, + ); + + test('preserves the legacy additive contract while identifying an undeployed optional backend', async () => { + (listRepoConfigs as jest.Mock).mockResolvedValue([ + { repo: 'acme/default', status: 'active' }, + { repo: 'acme/ecs', status: 'active', compute_type: 'ecs' }, + { repo: 'acme/microvm', status: 'active', compute_type: 'lambda-microvm' }, + ]); + const report = await buildRuntimeStatusReport('us-east-1', 'RepoTable', + 'arn:aws:bedrock-agentcore:us-east-1:123:runtime/platform', { + deployment: { ...LEGACY_DEPLOYMENT, computeSubstrate: 'ecs' }, + }); + expect(report.compute_deployment.default_compute_type).toBe('agentcore'); + expect(report.compute_deployment.compute_deployment_mode).toBeNull(); + expect(report.blueprints.map(binding => binding.compute_available)).toEqual([true, true, false]); + expect(report.ecs_substrates).toHaveLength(1); + expect(report.agentcore_runtimes).toHaveLength(1); + expect(report.lambda_microvm_substrates).toEqual([]); + }); + + test('uses the ordered additive default and skips unavailable AgentCore pins', async () => { + (listRepoConfigs as jest.Mock).mockResolvedValue([ + { repo: 'acme/default', status: 'active' }, + { repo: 'acme/ecs', status: 'active', compute_type: 'ecs' }, + { repo: 'acme/stale', status: 'active', compute_type: 'agentcore' }, + ]); + const report = await buildRuntimeStatusReport('us-east-1', 'RepoTable', null, { + deployment: { + ...LEGACY_DEPLOYMENT, + computeTypes: 'lambda-microvm,ecs', + computeSubstrate: 'lambda-microvm,ecs', + computeDeploymentMode: 'additive', + }, + }); + expect(report.compute_deployment.default_compute_type).toBe('lambda-microvm'); + expect(report.compute_deployment.compute_types).toEqual(['lambda-microvm', 'ecs']); + expect(report.blueprints.map(binding => binding.compute_available)).toEqual([true, true, false]); + expect(report.blueprints[0].compute_type).toBe('lambda-microvm'); + expect(report.agentcore_runtimes).toEqual([]); + expect(report.ecs_substrates).toHaveLength(1); + expect(report.lambda_microvm_substrates).toHaveLength(1); + expect(controlPlaneSend).not.toHaveBeenCalled(); + }); + + test('does not probe an unsupported repository compute type', async () => { + (listRepoConfigs as jest.Mock).mockResolvedValue([{ + repo: 'acme/invalid', + status: 'active', + compute_type: 'unknown', + runtime_arn: 'arn:aws:bedrock-agentcore:us-east-1:123:runtime/custom', + }]); + const report = await buildRuntimeStatusReport('us-east-1', 'RepoTable', null, { deployment: LEGACY_DEPLOYMENT }); + expect(report.blueprints[0].compute_available).toBe(false); + expect(report.blueprints[0].configuration_error).toContain('Unsupported repository compute_type'); + expect(controlPlaneSend).not.toHaveBeenCalled(); + }); + + test('rejects malformed exclusive outputs even when there are no repositories', async () => { + (listRepoConfigs as jest.Mock).mockResolvedValue([]); + await expect(buildRuntimeStatusReport('us-east-1', 'RepoTable', null, { + deployment: { ...LEGACY_DEPLOYMENT, computeDeploymentMode: 'exclusive' }, + })).rejects.toThrow('invalid or missing ComputeSubstrate'); + expect(controlPlaneSend).not.toHaveBeenCalled(); + }); + test('records probe errors without failing the report', async () => { controlPlaneSend.mockRejectedValue(new Error('AccessDenied')); (listRepoConfigs as jest.Mock).mockResolvedValue([{ @@ -96,6 +209,7 @@ describe('buildRuntimeStatusReport', () => { 'us-east-1', 'RepoTable', 'arn:aws:bedrock-agentcore:us-east-1:123:runtime/platform', + { deployment: LEGACY_DEPLOYMENT }, ); expect(report.agentcore_runtimes[0].probe_status).toBe('error'); @@ -112,7 +226,7 @@ describe('buildRuntimeStatusReport', () => { 'us-east-1', 'RepoTable', 'arn:aws:bedrock-agentcore:us-east-1:123:runtime/platform', - { repo: 'acme/b' }, + { repo: 'acme/b', deployment: LEGACY_DEPLOYMENT }, ); expect(report.blueprints).toHaveLength(1); @@ -137,6 +251,7 @@ describe('buildRuntimeStatusReport', () => { 'us-east-1', 'RepoTable', 'arn:aws:bedrock-agentcore:us-east-1:123:runtime/platform', + { deployment: LEGACY_DEPLOYMENT }, ); expect(report.agentcore_runtimes[0].last_updated_at).toBe('2026-01-01T00:00:00.000Z'); @@ -160,6 +275,7 @@ describe('buildRuntimeStatusReport', () => { 'us-east-1', 'RepoTable', 'arn:aws:bedrock-agentcore:us-east-1:123:runtime/platform', + { deployment: LEGACY_DEPLOYMENT }, ); expect(report.agentcore_runtimes[0].failure_reason).toBe('image pull failed'); @@ -172,7 +288,7 @@ describe('buildRuntimeStatusReport', () => { compute_type: 'agentcore', }]); - const report = await buildRuntimeStatusReport('us-east-1', 'RepoTable', null); + const report = await buildRuntimeStatusReport('us-east-1', 'RepoTable', null, { deployment: LEGACY_DEPLOYMENT }); expect(report.agentcore_runtimes).toHaveLength(0); expect(report.blueprints[0].runtime_arn).toBeUndefined(); diff --git a/cli/test/commands/runtime.test.ts b/cli/test/commands/runtime.test.ts index 11cb69627..b11e5a8bc 100644 --- a/cli/test/commands/runtime.test.ts +++ b/cli/test/commands/runtime.test.ts @@ -22,7 +22,18 @@ import { buildRuntimeStatusReport } from '../../src/runtime-status'; import { getStackOutput } from '../../src/stack-outputs'; jest.mock('../../src/runtime-status'); -jest.mock('../../src/stack-outputs'); +jest.mock('../../src/stack-outputs', () => ({ + ...jest.requireActual('../../src/stack-outputs'), + getStackOutput: jest.fn(), +})); + +const COMPUTE_DEPLOYMENT = { + stack_name: 'backgroundagent-dev', + compute_substrate: null, + compute_deployment_mode: null, + compute_types: null, + default_compute_type: 'agentcore', +}; describe('runtime status command', () => { let consoleSpy: jest.SpiedFunction; @@ -35,12 +46,14 @@ describe('runtime status command', () => { return null; }); (buildRuntimeStatusReport as jest.Mock).mockResolvedValue({ + compute_deployment: COMPUTE_DEPLOYMENT, platform_default_runtime_arn: 'arn:aws:bedrock-agentcore:us-east-1:123:runtime/platform', blueprints: [{ repo: 'acme/a', status: 'active', compute_type: 'agentcore', runtime_arn: 'arn:aws:bedrock-agentcore:us-east-1:123:runtime/platform', + compute_available: true, runtime_arn_source: 'platform', }], agentcore_runtimes: [{ @@ -64,19 +77,21 @@ describe('runtime status command', () => { await cmd.parseAsync(['node', 'test', 'status', '--region', 'us-east-1']); const output = consoleSpy.mock.calls.map((c) => c[0]).join('\n'); - expect(output).toContain('Per-blueprint effective compute'); + expect(output).toContain('Per-blueprint compute configuration'); expect(output).toContain('acme/a'); expect(output).toContain('READY'); }); test('prints ECS blueprint without runtime ARN', async () => { (buildRuntimeStatusReport as jest.Mock).mockResolvedValue({ + compute_deployment: COMPUTE_DEPLOYMENT, platform_default_runtime_arn: 'arn:aws:bedrock-agentcore:us-east-1:123:runtime/platform', blueprints: [{ repo: 'acme/ecs', status: 'active', compute_type: 'ecs', runtime_arn: undefined, + compute_available: true, runtime_arn_source: 'platform', }], agentcore_runtimes: [], @@ -97,12 +112,14 @@ describe('runtime status command', () => { test('prints ECS substrate note', async () => { (buildRuntimeStatusReport as jest.Mock).mockResolvedValue({ + compute_deployment: COMPUTE_DEPLOYMENT, platform_default_runtime_arn: 'arn:aws:bedrock-agentcore:us-east-1:123:runtime/platform', blueprints: [{ repo: 'acme/ecs', status: 'active', compute_type: 'ecs', runtime_arn: undefined, + compute_available: true, runtime_arn_source: 'platform', }], agentcore_runtimes: [], @@ -124,12 +141,14 @@ describe('runtime status command', () => { test('prints Lambda MicroVM substrate note without runtime probing', async () => { (buildRuntimeStatusReport as jest.Mock).mockResolvedValue({ + compute_deployment: COMPUTE_DEPLOYMENT, platform_default_runtime_arn: 'arn:aws:bedrock-agentcore:us-east-1:123:runtime/platform', blueprints: [{ repo: 'acme/microvm', status: 'active', compute_type: 'lambda-microvm', runtime_arn: undefined, + compute_available: true, runtime_arn_source: 'platform', }], agentcore_runtimes: [], @@ -163,6 +182,7 @@ describe('runtime status command', () => { test('reports empty blueprint set', async () => { (buildRuntimeStatusReport as jest.Mock).mockResolvedValue({ + compute_deployment: COMPUTE_DEPLOYMENT, platform_default_runtime_arn: null, blueprints: [], agentcore_runtimes: [], @@ -178,12 +198,14 @@ describe('runtime status command', () => { test('shows successful probe metadata', async () => { (buildRuntimeStatusReport as jest.Mock).mockResolvedValue({ + compute_deployment: COMPUTE_DEPLOYMENT, platform_default_runtime_arn: 'arn:aws:bedrock-agentcore:us-east-1:123:runtime/platform', blueprints: [{ repo: 'acme/a', status: 'active', compute_type: 'agentcore', runtime_arn: 'arn:aws:bedrock-agentcore:us-east-1:123:runtime/platform', + compute_available: true, runtime_arn_source: 'platform', }], agentcore_runtimes: [{ @@ -210,12 +232,14 @@ describe('runtime status command', () => { test('shows probe error details', async () => { (buildRuntimeStatusReport as jest.Mock).mockResolvedValue({ + compute_deployment: COMPUTE_DEPLOYMENT, platform_default_runtime_arn: 'arn:aws:bedrock-agentcore:us-east-1:123:runtime/platform', blueprints: [{ repo: 'acme/a', status: 'active', compute_type: 'agentcore', runtime_arn: 'arn:aws:bedrock-agentcore:us-east-1:123:runtime/platform', + compute_available: true, runtime_arn_source: 'platform', }], agentcore_runtimes: [{ @@ -242,5 +266,82 @@ describe('runtime status command', () => { const payload = JSON.parse(consoleSpy.mock.calls[0][0] as string); expect(payload.blueprints).toHaveLength(1); + expect(payload.compute_deployment).toEqual(COMPUTE_DEPLOYMENT); + expect(payload.blueprints[0].compute_available).toBe(true); + }); + + test.each(['text', 'json'])('passes the exclusive deployment contract and reports a stale pin in %s', async format => { + (getStackOutput as jest.Mock).mockImplementation(async (_r: string, _s: string, key: string) => ({ + RepoTableName: 'RepoTable', + ComputeSubstrate: 'ecs', + ComputeDeploymentMode: 'exclusive', + } as Record)[key] ?? null); + const configurationError = "Stack 'backgroundagent-dev' deploys only 'ecs'; lambda-microvm is unavailable."; + (buildRuntimeStatusReport as jest.Mock).mockResolvedValue({ + compute_deployment: { + ...COMPUTE_DEPLOYMENT, + compute_substrate: 'ecs', + compute_deployment_mode: 'exclusive', + default_compute_type: 'ecs', + }, + platform_default_runtime_arn: null, + blueprints: [{ + repo: 'acme/stale', + status: 'active', + compute_type: 'lambda-microvm', + compute_available: false, + configuration_error: configurationError, + runtime_arn_source: 'platform', + }], + agentcore_runtimes: [], + ecs_substrates: [], + lambda_microvm_substrates: [], + }); + await makeRuntimeCommand().parseAsync(['node', 'test', 'status', '--region', 'us-east-1', '--output', format]); + expect(buildRuntimeStatusReport).toHaveBeenLastCalledWith('us-east-1', 'RepoTable', null, { + repo: undefined, + deployment: { stackName: 'backgroundagent-dev', computeSubstrate: 'ecs', computeDeploymentMode: 'exclusive', computeTypes: null }, + }); + if (format === 'json') { + const report = JSON.parse(consoleSpy.mock.calls[0][0] as string); + expect(report.blueprints[0].compute_available).toBe(false); + expect(report.blueprints[0].configuration_error).toBe(configurationError); + expect(report.compute_deployment.default_compute_type).toBe('ecs'); + } else { + const output = consoleSpy.mock.calls.map(call => call[0]).join('\n'); + expect(output).toContain(`UNAVAILABLE: ${configurationError}`); + expect(output).toContain('Compute deployment mode: exclusive'); + expect(output).not.toContain('Lambda MicroVMs are platform-managed'); + } + }); + + test('passes the full ordered ComputeTypes output into runtime reporting', async () => { + (getStackOutput as jest.Mock).mockImplementation(async (_r: string, _s: string, key: string) => ({ + RepoTableName: 'RepoTable', + ComputeSubstrate: 'ecs,agentcore', + ComputeTypes: 'ecs,agentcore', + ComputeDeploymentMode: 'additive', + } as Record)[key] ?? null); + (buildRuntimeStatusReport as jest.Mock).mockResolvedValue({ + compute_deployment: { + ...COMPUTE_DEPLOYMENT, + compute_substrate: 'ecs,agentcore', + compute_deployment_mode: 'additive', + compute_types: ['ecs', 'agentcore'], + default_compute_type: 'ecs', + }, + blueprints: [], + }); + await makeRuntimeCommand().parseAsync(['node', 'test', 'status', '--region', 'us-east-1']); + expect(buildRuntimeStatusReport).toHaveBeenLastCalledWith('us-east-1', 'RepoTable', null, { + repo: undefined, + deployment: { + stackName: 'backgroundagent-dev', + computeSubstrate: 'ecs,agentcore', + computeTypes: 'ecs,agentcore', + computeDeploymentMode: 'additive', + }, + }); + expect(consoleSpy.mock.calls.map(call => call[0]).join('\n')).toContain('ComputeTypes: ecs, agentcore'); }); }); diff --git a/cli/test/compute-substrate.test.ts b/cli/test/compute-substrate.test.ts index 2260e925d..044ec994f 100644 --- a/cli/test/compute-substrate.test.ts +++ b/cli/test/compute-substrate.test.ts @@ -19,7 +19,10 @@ import { assertComputeSubstrateDeployed, + defaultComputeType, + describeComputeDeployment, parseComputeSubstrateOutput, + resolveRepositoryCompute, } from '../src/compute-substrate'; import { CliError } from '../src/errors'; @@ -38,14 +41,11 @@ describe('parseComputeSubstrateOutput', () => { ['agentcore', ['agentcore']], ['ecs', ['ecs']], ['lambda-microvm', ['lambda-microvm']], - ])('parses the single value %s that the stack emits today', (raw, expected) => { + ])('parses the single value %s from legacy or single-backend stacks', (raw, expected) => { expect(parseComputeSubstrateOutput(raw)).toEqual(expected); }); - test('tolerates a comma list, so a future compute_types output cannot silently over-refuse', () => { - // ADR-021 sub-decision 4 names a `compute_types` list as the intended - // follow-up to the single-valued tag. An `!== 'ecs'` equality check would - // start rejecting valid onboardings the day that lands. + test('parses complete comma-separated backend lists', () => { expect(parseComputeSubstrateOutput('ecs,lambda-microvm')).toEqual(['ecs', 'lambda-microvm']); expect(parseComputeSubstrateOutput(' ecs , lambda-microvm ')).toEqual(['ecs', 'lambda-microvm']); }); @@ -60,8 +60,8 @@ describe('parseComputeSubstrateOutput', () => { ); }); -describe('assertComputeSubstrateDeployed', () => { - test('never gates agentcore — the runtime is unconditional', () => { +describe('assertComputeSubstrateDeployed with legacy outputs', () => { + test('preserves the unconditional AgentCore backend of legacy stacks', () => { expect(assertFor('agentcore', 'agentcore')).not.toThrow(); expect(assertFor('agentcore', 'ecs')).not.toThrow(); expect(assertFor('agentcore', 'lambda-microvm')).not.toThrow(); @@ -102,10 +102,10 @@ describe('assertComputeSubstrateDeployed', () => { expect(assertFor('lambda-microvm', 'agentcore')).toThrow(/--context compute_type=lambda-microvm/); // The MicroVM remedy is more specific than ECS's about WHERE it fails, // because the strategy's env-var guard fires before any AWS call. - expect(assertFor('lambda-microvm', 'agentcore')).toThrow(/MICROVM_\*/); + expect(assertFor('lambda-microvm', 'agentcore')).toThrow(/fail at session start/); }); - test('refuses each optional backend on the OTHER one (they are mutually exclusive today)', () => { + test('refuses an optional backend absent from the legacy output', () => { expect(assertFor('lambda-microvm', 'ecs')).toThrow(/without the Lambda MicroVMs substrate/); expect(assertFor('ecs', 'lambda-microvm')).toThrow(/without the ECS substrate/); }); @@ -114,10 +114,63 @@ describe('assertComputeSubstrateDeployed', () => { expect(assertFor('lambda-microvm', 'agentcore')).toThrow(new RegExp(`'${STACK}'`)); }); - test('allows both optional backends against a hypothetical multi-substrate output', () => { - // Behavioural proof of the list tolerance above: this must NOT throw, or the - // `compute_types` follow-up would break onboarding for both backends. + test('allows each optional backend listed in ComputeSubstrate', () => { expect(assertFor('ecs', 'ecs,lambda-microvm')).not.toThrow(); expect(assertFor('lambda-microvm', 'ecs,lambda-microvm')).not.toThrow(); }); }); + +describe('exclusive backend output', () => { + test.each(['agentcore', 'ecs', 'lambda-microvm'] as const)('inherits %s and rejects every other backend', backend => { + const deployment = { stackName: 'test', computeSubstrate: backend, computeDeploymentMode: 'exclusive' }; + expect(() => assertComputeSubstrateDeployed({ ...deployment, computeType: undefined })).not.toThrow(); + for (const requested of ['agentcore', 'ecs', 'lambda-microvm'] as const) { + const check = () => assertComputeSubstrateDeployed({ ...deployment, computeType: requested }); + if (requested === backend) expect(check).not.toThrow(); + else expect(check).toThrow(/deploys only/); + } + }); + test.each([null, '', 'ecs,lambda-microvm', 'unknown'])('rejects malformed exclusive output %p', computeSubstrate => { + expect(() => assertComputeSubstrateDeployed({ stackName: 'test', computeSubstrate, computeDeploymentMode: 'exclusive', computeType: undefined })).toThrow(/invalid or missing/); + }); +}); + +describe('ordered ComputeTypes output', () => { + const deployment = { + stackName: STACK, + computeTypes: 'lambda-microvm,ecs', + computeSubstrate: 'lambda-microvm,ecs', + computeDeploymentMode: 'additive', + }; + + test('inherits the first entry and enforces membership without assuming AgentCore', () => { + expect(defaultComputeType(deployment)).toBe('lambda-microvm'); + expect(resolveRepositoryCompute(deployment)).toEqual({ compute_type: 'lambda-microvm', compute_available: true }); + expect(resolveRepositoryCompute(deployment, 'ecs')).toEqual({ compute_type: 'ecs', compute_available: true }); + expect(resolveRepositoryCompute(deployment, 'agentcore')).toMatchObject({ + compute_type: 'agentcore', + compute_available: false, + configuration_error: expect.stringContaining('compute_types=lambda-microvm,ecs,agentcore'), + }); + expect(describeComputeDeployment(deployment)).toMatchObject({ + compute_types: ['lambda-microvm', 'ecs'], + default_compute_type: 'lambda-microvm', + }); + }); + + test('supports the complete additive substrate output when ComputeTypes is absent', () => { + expect(defaultComputeType({ ...deployment, computeTypes: null })).toBe('lambda-microvm'); + expect(resolveRepositoryCompute({ ...deployment, computeTypes: null }, 'agentcore').compute_available).toBe(false); + }); + + test.each(['', ' ', ',', 'ecs,', 'ecs,unknown', 'ecs,ecs'])('rejects malformed ComputeTypes %p', computeTypes => { + expect(() => defaultComputeType({ ...deployment, computeTypes })).toThrow(/invalid or missing ComputeTypes/); + }); + + test('refuses contradictory outputs and an unknown deployment mode', () => { + expect(() => defaultComputeType({ ...deployment, computeSubstrate: 'ecs' })).toThrow(/outputs disagree/); + expect(() => defaultComputeType({ ...deployment, computeDeploymentMode: 'exclusive' })).toThrow(/invalid or missing/); + expect(() => defaultComputeType({ ...deployment, computeTypes: 'ecs', computeSubstrate: 'ecs' })).toThrow(/invalid or missing/); + expect(() => defaultComputeType({ ...deployment, computeDeploymentMode: 'unknown' })).toThrow(/Unknown ComputeDeploymentMode/); + }); +}); diff --git a/contracts/constants.json b/contracts/constants.json index 07a624ba3..174812d40 100644 --- a/contracts/constants.json +++ b/contracts/constants.json @@ -46,7 +46,10 @@ "agent_session_role_arn": "AGENT_SESSION_ROLE_ARN", "aws_sdk_ua_app_id": "AWS_SDK_UA_APP_ID", "anthropic_default_haiku_model": "ANTHROPIC_DEFAULT_HAIKU_MODEL", - "anthropic_model": "ANTHROPIC_MODEL" + "anthropic_model": "ANTHROPIC_MODEL", + "tool_gateway_url": "ABCA_TOOL_GATEWAY_URL", + "linear_vault_enabled": "LINEAR_VAULT_ENABLED", + "linear_workload_identity_name": "LINEAR_WORKLOAD_IDENTITY_NAME" }, "required": [ "task_table_name", diff --git a/docs/decisions/ADR-016-pluggable-identity-and-auth.md b/docs/decisions/ADR-016-pluggable-identity-and-auth.md index 757f2bcc6..b24aa23cc 100644 --- a/docs/decisions/ADR-016-pluggable-identity-and-auth.md +++ b/docs/decisions/ADR-016-pluggable-identity-and-auth.md @@ -160,7 +160,7 @@ Per-user `McpCredential` selection requires the Gateway to know *which task-user | P6 | **Trusted task-user identity propagation** (prerequisite for per-user MCP on the general plane, P5): specify + validate a user-scoped inbound identity the Gateway authorizer trusts, replacing the M2M JWT for per-user credential selection. | **Blocks per-user `McpCredential`.** Until done, MCP credentials are workspace-scoped at best. | | P7 | Jira + Slack `ChannelCredential` (same shape as P1); GitHub `GithubOauth2` behind a flag, retire the shared PAT; OBO `act`-claim delegation feeding #237. | Flag-gated; per-surface. | -**Substrate exception — `lambda-microvm` (2026-09-02):** P1's vault cannot be enabled on the MicroVM substrate. The two together synthesize 505 resources against CloudFormation's hard 500-resource limit (MicroVM alone 496, the vault alone 488), so `AgentStack` refuses the combination at synth, naming both context flags. The MicroVM wiring itself is complete — `platform_config` carries the workload name and the guest execution role holds the mint grant — so this is a capacity limit, not a design gap, and it lifts as soon as a subsystem moves into a nested stack. +**Substrate update — `lambda-microvm` (#852):** The optional split network provides headroom for MicroVM plus the vault, including additive deployments that retain AgentCore. Explicit `compute_types=lambda-microvm` also permits a MicroVM-only deployment. The production 490-resource guard validates the complete configuration. The guest execution role has the mint grant; the shared `platform_config` contract now forwards vault enablement and workload identity. A rebuilt compatible MicroVM image and live consent/mint rehearsal are required before using the combination. Lambda MicroVMs remains experimental. **Substrate independence (verified 2026-07-21, both proven live):** the vault path works on any compute. AgentCore Runtime injects the Workload Access Token as the `WorkloadAccessToken` header; ECS/Fargate/Lambda bootstrap it via `GetWorkloadAccessTokenForJWT(workloadName, userToken=)` against a **standalone** (non-service-linked) workload identity, then call `GetResourceOauth2Token`. Runtime-managed (service-linked) workload identities cannot self-vend, so the ECS path needs a manually-created workload identity. The runtime execution role today has `GetWorkloadAccessToken*` but **not** `GetResourceOauth2Token` — P1 adds it, plus `GetSecretValue` scoped to that surface's providers (`bedrock-agentcore-identity!default/oauth2/*`; see the implementation notes below for why this is narrower than the wildcard first anticipated here). diff --git a/docs/decisions/ADR-021-lambda-microvms-compute-backend.md b/docs/decisions/ADR-021-lambda-microvms-compute-backend.md index 49bf51482..79b984ffc 100644 --- a/docs/decisions/ADR-021-lambda-microvms-compute-backend.md +++ b/docs/decisions/ADR-021-lambda-microvms-compute-backend.md @@ -326,7 +326,7 @@ Two networking facts the construct has to encode, both established live: `lambda:CreateMicrovmAuthToken` is granted to no role in P1–P3 (no JWE consumer exists; see sub-decision 3). -**Cost attribution.** `cdk/src/main.ts` currently tags the whole stack with a single `compute_type` context value (default `agentcore`) — already imprecise with two backends, wrong with three. P1 must add backend-identifying cost-allocation tags on the MicroVM-specific resources (images, payload/artifact bucket wiring, log groups) and revisit the stack-level tag semantics (e.g. a `compute_types` list), keeping attribution consistent with [#645](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/645)'s cost/attribution acceptance criterion. +**Cost attribution update (#852).** The `compute_types` context selects an ordered backend list; legacy `compute_type` retains its additive meaning. The stack-level `compute_type` tag records all deployed names separated by `+` (for example, `agentcore+lambda-microvm`), which is valid in AWS tag values. MicroVM-specific resources retain their `abca:compute-backend` tags. Shared optional services remain independent of the compute selection. An unchanged legacy context preserves AgentCore; deliberate backend removal requires a drained transition; see [Compute](../design/COMPUTE.md#selecting-and-changing-the-backend). - Where a deployment enables the `lambda-microvm` backend, MicroVM-specific resources shall carry backend-identifying cost-allocation tags. diff --git a/docs/decisions/ADR-023-cloudformation-stack-boundaries.md b/docs/decisions/ADR-023-cloudformation-stack-boundaries.md new file mode 100644 index 000000000..fb3ffdb99 --- /dev/null +++ b/docs/decisions/ADR-023-cloudformation-stack-boundaries.md @@ -0,0 +1,60 @@ +# ADR-023: Optional network stack and deployment budgets + +**Status:** proposed +**Date:** 2026-09-21 +**Last-updated:** 2026-10-05 +**Issue:** [#852](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/852) + +Per the [ADR lifecycle](./README.md#lifecycle), this decision remains proposed while its implementing PR is in review and becomes accepted when that PR merges. This record does not approve or waive the existing-stack migration criteria in #852. + +## Context + +The application stack approaches CloudFormation's 500-resource limit. [#851](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/851) records the incident; approved issue #852 covers resource budgets, template bytes and stack boundaries. [#735](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/735) is related byte-limit evidence, not a separate approval for this implementation. + +The Task API owns most resources, but its integrations share one API Gateway RestApi and deployment. The #852 proof of concept showed that extracting an integration can lose deployment dependencies, CORS, solution attribution and tags even when synthesis passes. Networking has a smaller interface and an independent lifecycle. + +Live review of an earlier #912 revision found that broad retention blocks failed-create retries and same-name redeploys, while Blueprint ownership handoff can lose repository settings and orphan PITR-enabled ledger tables. Those changes are removed from this PR. The reviewed scope is the optional network stack, deployment budgets and compatible compute selection. + +## Decision + +1. Keep the Task API, authorizers, deployment, stage and route integrations together in the application stack. Preserve existing nested stacks for Registry, RegistryApi and hosted consent pages. +2. Offer `networkTopology=split` for new installations; `inline` remains the default. Move AgentVpc and DnsFirewall into `${stackName}-network`, resolve Blueprint egress definitions before either stack is constructed, and permit application-to-network references only. Keep VPC/subnet/security-group exports present across backend changes and preserve network properties, attribution and provenance tags. +3. Select one or more backends with `compute_types`; its first entry is the repository default. Preserve legacy `compute_type` behavior, including AgentCore alongside ECS or MicroVM. Removing a backend requires an explicit list that omits it. Publish the complete ordered list in `ComputeTypes` and `ComputeSubstrate`; the CLI and orchestrator enforce membership. Shared optional services remain independent. +4. Enforce CDK's `@aws-cdk/core:stackResourceLimit` at 490 before constructing stacks, including nested stacks and operator configurations outside the census. Operators may tighten but cannot raise it. Share the census profile product with the normal build, checking every template against resource, byte, parameter and output budgets and preserving method-scoped API permissions. Default byte budget: 800,000. +5. Keep current resource removal policies, Blueprint provisioning and guardrail versioning. Existing inline-to-split migration, broad retention and Blueprint controller handoff are deferred. Local template comparisons do not satisfy the populated refactor/import and rollback rehearsal required by #852. + +## Validation scope + +The 122-profile product covers all single-backend/service/image combinations in both topologies, supplemental alert/fork/consent options, explicit three-AZ pins, expanded model grants, every multi-backend set at default and widest settings, and legacy additive selectors. The expanded-model profiles add eight synthetic model IDs to the platform defaults to exercise IAM policy overflow; they do not assert live model availability. The SessionRole's audit exceptions follow its generated overflow policies and remain scoped to tenant object prefixes and literal model grants. Expected over-budget profiles must fail at the production 490-resource ceiling for the application stack; an unrelated error or unexpected success fails the gate. Real CDK metadata is included. The census records per-template measurements and source provenance with bundling and asset staging disabled; it does not measure a deployed stack. + +Network tests compare moved logical IDs and service properties, the complete export interface, application data resources and lifecycle policies, shared API routes and permissions, attribution and one-way dependencies. Their comparisons control the clock and account for the existing alpha guardrail and orchestrator version IDs. The independent-process `--check-stability` diagnostic keeps timestamps, IDs and asset hashes intact and reports existing churn; passing budget checks does not imply deterministic synthesis. + +`networkReservedAzs` preserves unused address slots when removing trailing AZs. Tests verify that the target application can use the old network, release the removed export, and keep every remaining subnet's properties. The [application-first procedure](../guides/DEPLOYMENT_GUIDE.md#reducing-azs-in-an-existing-split-network) is separate from moving an inline network into another stack. Physical-ID preservation and rollback in AWS still require live verification. + +## Deferred migration work + +The following remain under #852 and need separate review and releases before a supported existing-stack migration: + +- **Retention lifecycle:** use appropriate per-resource policies, including `RetainExceptOnCreate` where retaining established data is needed without orphaning a failed first create. Verify fixed-name log groups and external registries can be recovered, imported or cleaned up before a same-name reinstall. Retaining an S3 bucket alone must not leave a destructive cleanup callback active. +- **Blueprint downgrade protection:** refuse an unsafe return from managed ownership to the legacy writer even when a context flag is omitted. Prove `max_turns`, `compute_type`, `onboarded_at` and other CLI overrides survive updates and rollback. A green deployment must not hide row replacement or tombstoning. +- **Ownership release and re-onboarding:** adoption must have a defined release path. Removing a repository must not block legitimate re-onboarding for the tombstone TTL. Test retries, owner changes and out-of-order callbacks. +- **Ledger lifecycle and bootstrap coverage:** establish a bounded cleanup/recovery plan for PITR-enabled ownership tables across mode changes, failures and destroy. Any future `Custom::BlueprintRepoConfig` must be represented in the bootstrap resource-action map and its coverage tests before it ships. +- **Guardrail and image normalization:** review stable version binding and Docker build-context changes separately from network ownership. They are not prerequisites for reporting truthful census differences. +- **Populated migration rehearsal:** verify refactor/import eligibility for each moved type and provider, preserve physical IDs and data, test networking and API behavior, and execute rollback. Retention must be installed on source resources before a transfer; an ordinary topology flag change is not a move. The criterion remains open, without an author-only waiver. + +The [teardown guidance](../guides/DEPLOYMENT_GUIDE.md#teardown-blocked-by-agentcore-network-interfaces) covers the separately observed AgentCore ENI cleanup delay, which also occurs on `main`. + +## Consequences + +- New split installations gain application headroom without widening API Gateway permissions or changing repository ownership. +- Existing legacy compute contexts retain AgentCore. Ordered lists let repositories choose among deployed backends; older CLIs require AgentCore to remain present and first on additive stacks. +- Exports constrain later network changes. Application consumers must release an export before the network removes it. +- Resource and template budgets are checked locally; live deployment, migration and rollback remain separate evidence. + +## References + +- [Developer guide: synthesis budgets](../guides/DEVELOPER_GUIDE.md#stack-decomposition-and-synthesis-budgets) +- [Deployment guide: network topology](../guides/DEPLOYMENT_GUIDE.md#network-stack-topology) +- [Compute selection](../design/COMPUTE.md#selecting-and-changing-the-backend) +- [CDK best practices](https://docs.aws.amazon.com/cdk/v2/guide/best-practices.html) +- [CloudFormation quotas](https://docs.aws.amazon.com/AWSCloudFormation/latest/UserGuide/cloudformation-limits.html) diff --git a/docs/design/ARCHITECTURE.md b/docs/design/ARCHITECTURE.md index 2d188b878..f73387750 100644 --- a/docs/design/ARCHITECTURE.md +++ b/docs/design/ARCHITECTURE.md @@ -37,9 +37,20 @@ The orchestrator and agent are deliberately separated. The orchestrator handles For the full orchestrator design, see [ORCHESTRATOR.md](./ORCHESTRATOR.md). For the API contract, see [API_CONTRACT.md](./API_CONTRACT.md). +## Deployment boundaries + +`AgentStack` keeps the shared Task API, its route integrations, data stores and selected compute backends together. Registry, RegistryApi and the hosted Linear consent page retain their existing nested boundaries. Network ownership is selected by `networkTopology`: + +| Topology | Network ownership | Stack dependencies | +|---|---|---| +| `inline` (default) | AgentVpc and DnsFirewall inside the application stack | Existing parent/nested structure | +| `split` | Separate `${stackName}-network` stack | Application imports network references; network has no application references | + +The split gives networking an independent deployment lifecycle and reduces the application template's resource count. It preserves AZ selection, DNS observation mode and security rules. All stacks receive solution attribution and provenance tags, while existing removal policies remain in effect. The split is available for new installations; existing inline-to-split migration is deferred pending a populated rehearsal; see [deployment guidance](../guides/DEPLOYMENT_GUIDE.md#network-stack-topology) and [ADR-023](../decisions/ADR-023-cloudformation-stack-boundaries.md). Live migration has not been validated. + ## Repository onboarding -Onboarding is CDK-based. Each repository is an instance of the `Blueprint` construct in the stack. The construct writes a `RepoConfig` record to DynamoDB; the orchestrator reads it at task time. +Onboarding is CDK-based. Plain repository definitions in `cdk/src/blueprints/definitions.ts` feed both network egress policy and the `Blueprint` constructs in the application stack. Each construct writes a `RepoConfig` record to DynamoDB; the orchestrator reads it at task time. Resolving configuration before stack construction keeps the optional network stack independent of repository resources. Blueprints configure how the orchestrator executes steps for each repo: compute strategy, model selection, turn limits, GitHub token, and optional custom steps. See [REPO_ONBOARDING.md](./REPO_ONBOARDING.md) for the full design. diff --git a/docs/design/COMPUTE.md b/docs/design/COMPUTE.md index 48670a6df..b74ee18ce 100644 --- a/docs/design/COMPUTE.md +++ b/docs/design/COMPUTE.md @@ -7,7 +7,7 @@ Every task runs in an isolated cloud compute environment. Nothing runs on the us ## Compute options -The default runtime is **Amazon Bedrock AgentCore Runtime**, which runs each session in a Firecracker MicroVM with per-session isolation, managed lifecycle, and built-in health monitoring. For repos that exceed AgentCore's constraints (2 GB image limit, no GPU), the `ComputeStrategy` interface allows switching to alternative backends per repo. +The default runtime is **Amazon Bedrock AgentCore Runtime**, which runs each session in a Firecracker MicroVM with per-session isolation, managed lifecycle, and built-in health monitoring. The `compute_types` CDK context selects one or more backends: `agentcore`, `ecs`, and `lambda-microvm`. Its first entry is the default for repositories; a repository can explicitly select any deployed backend. The `ComputeStrategy` interface dispatches each task to its resolved backend. | | AgentCore Runtime | ECS on Fargate | **Lambda MicroVMs** | ECS on EC2 | EKS | AWS Batch | Lambda (functions) | Custom EC2 + Firecracker | |---|---|---|---|---|---|---|---|---| @@ -23,7 +23,45 @@ The default runtime is **Amazon Bedrock AgentCore Runtime**, which runs each ses > **Lambda MicroVMs are not Lambda functions.** They are a different compute primitive, so the functions column's 15-minute cap and poor-fit verdict do not apply. See [ADR-021](../decisions/ADR-021-lambda-microvms-compute-backend.md). -The backend is selected per repo via `compute_type` in the Blueprint config. The orchestrator resolves the strategy and delegates session start, polling, and termination to the strategy implementation. See [REPO_ONBOARDING.md](./REPO_ONBOARDING.md) for the `ComputeStrategy` interface. +Repositories without a `compute_type` override inherit the first entry in the deployed list. An explicit Blueprint or RepoTable override must name a deployed backend; mismatches fail before concurrency admission. Memory, Tool Gateway, Agent Registry and Linear Identity Vault are separate service choices, so selecting ECS or MicroVM does not disable them. The orchestrator resolves the strategy and delegates session start, polling, and termination to the strategy implementation. See [REPO_ONBOARDING.md](./REPO_ONBOARDING.md) for the `ComputeStrategy` interface. + +## Selecting and changing the backend + +Set `compute_types` in `cdk/cdk.json` as an array, or pass a comma-separated list to the deployment task: + +```bash +# AgentCore and MicroVM, with AgentCore as the repository default. +MISE_EXPERIMENTAL=1 mise //cdk:deploy -- -c compute_types=agentcore,lambda-microvm + +# Only ECS; removing an existing backend requires the transition below. +MISE_EXPERIMENTAL=1 mise //cdk:deploy -- -c compute_types=ecs +``` + +`compute_types` takes precedence over the legacy `compute_type` context. Empty lists, non-string entries and unsupported names fail synthesis. Duplicate names are collapsed while preserving order. Without `compute_types`, the previous deployment behavior is preserved: + +| Legacy context | Deployed backends | Repository default | +|---|---|---| +| Absent, or `compute_type=agentcore` | AgentCore | AgentCore | +| `compute_type=ecs` | AgentCore and ECS | AgentCore | +| `compute_type=lambda-microvm` | AgentCore and Lambda MicroVMs | AgentCore | + +An upgrade with the same legacy context keeps AgentCore, its role and its log delivery. Selecting a single optional backend explicitly, such as `compute_types=ecs`, opts into removing AgentCore. The bootstrap `ComputeTypes` CloudFormation parameter is a separate permission allowlist; enable every backend you plan to deploy there too. + +`ComputeTypes` and `ComputeSubstrate` both publish the complete ordered comma list. `ComputeDeploymentMode` is `exclusive` for one backend and `additive` for several. `RuntimeArn` exists whenever AgentCore is included. The orchestrator receives the same list in `DEPLOYED_COMPUTE_TYPE`. Stack-level `compute_type` tags join the deployed names with `+`, a tag-safe separator; backend-specific resources keep their own attribution. + +Use the updated CLI when the first backend is not AgentCore, or the list omits AgentCore. Older CLIs assume AgentCore is present and is the default on additive stacks. The updated CLI reads `ComputeTypes` for onboarding, `repo show`, and `runtime status`. Stacks without the new outputs retain the legacy interpretation. + +`bgagent repo show` and `bgagent runtime status` mark incompatible stored pins as **UNAVAILABLE** and explain how to reconcile them. JSON includes the ordered `compute_deployment.compute_types` array, `default_compute_type`, `compute_available`, and `configuration_error` (per repository in runtime status). Incompatible repositories are excluded from backend summaries and AgentCore probes. This checks deployment configuration; MicroVM image readiness and live backend health remain separate checks. + +Before deliberately removing a backend or changing the first entry: + +1. Record deployed context, templates, image identifiers, repository overrides and active sessions. Pause task submissions and let running or suspended work finish, or cancel it with the existing deployment. +2. Reconcile stored overrides with the target list. Omitted `compute_type` inherits its first entry; explicit pins to other deployed backends remain valid. Remove obsolete `runtime_arn` overrides when leaving AgentCore. +3. Prepare images and bootstrap permissions. MicroVM needs a compatible snapshot; rebuild it from this checkout before enabling Gateway or the vault because their optional settings travel through `platform_config`. Infrastructure without an image cannot run tasks. +4. Review the full change set. Removing AgentCore removes its Runtime and delivery resources. The two named AgentCore log groups stay owned by the application across backend switches, with their existing destroy policies and retention periods. Shared data stores keep their existing lifecycle policies. A network ownership transfer is a separate, deferred migration. +5. Rehearse deployment and rollback, then verify a task on each backend, repository-less work, cancellation, logs and enabled integrations before resuming submissions. Re-adding a backend cannot resume deleted sessions or recover runtime storage. + +The resource budget applies to the complete combination. The split topology fits the sampled combinations, including all three backends; wider inline combinations fail at the 490-resource ceiling. See [Network stack topology](../guides/DEPLOYMENT_GUIDE.md#network-stack-topology). Local synthesis verifies wiring and budget behavior; it does not establish a live migration guarantee or change MicroVM's experimental status. ## What runs in the session @@ -77,7 +115,7 @@ See [ORCHESTRATOR.md](./ORCHESTRATOR.md) for how the orchestrator handles these ## Lambda MicroVMs backend -Lambda MicroVMs are an opt-in third backend, selected per repository with `compute_type: lambda-microvm`; AgentCore remains the default. Image configuration has three states: a managed base-image ARN and version creates the snapshot image in CDK; an external image identifier uses a snapshot built out of band; and supplying neither provisions only the roles, buckets, and connectors needed for the bootstrap deploy. `cdk/scripts/package-microvm-artifact.sh` packages the agent as zip + Dockerfile, uploads it to the artifact bucket, and can create the external image. Lambda MicroVMs are available in five launch regions (us-east-1, us-east-2, us-west-2, eu-west-1, ap-northeast-1) and will expand; the platform enforces regional availability in layers via a synth-time constant, onboarding live probes, and orchestration-time classification. +Lambda MicroVMs are an opt-in third backend. Include `lambda-microvm` in `compute_types`; repositories inherit the first listed backend or choose a deployed backend through an override. Legacy `compute_type=lambda-microvm` deploys AgentCore and MicroVM, with AgentCore as the default. Image configuration has three states: a managed base-image ARN and version creates the snapshot image in CDK; an external image identifier uses a snapshot built out of band; and supplying neither provisions only the roles, buckets, and connectors needed for the bootstrap deploy. `cdk/scripts/package-microvm-artifact.sh` packages the agent as zip + Dockerfile, uploads it to the artifact bucket, and can create the external image. Lambda MicroVMs are available in five launch regions (us-east-1, us-east-2, us-west-2, eu-west-1, ap-northeast-1) and will expand; the platform enforces regional availability in layers via a synth-time constant, onboarding live probes, and orchestration-time classification. Because a snapshot freezes its build-time environment, deployment-specific, non-secret identifiers travel in the `/run` hook's `platform_config` block instead. The strategy sends the canonical inline envelope or, when that envelope exceeds the verified 4,096-byte `runHookPayload` limit, an S3-pointer envelope with the configuration also merged into the uploaded payload. The agent accepts only allowlisted keys and installs them before pipeline initialization; [ADR-021 §3](../decisions/ADR-021-lambda-microvms-compute-backend.md#3-packaging-same-agent-image-source-new-build-path) defines the exact wire shapes and validation rules. diff --git a/docs/design/REPO_ONBOARDING.md b/docs/design/REPO_ONBOARDING.md index 3abb263d9..3a2ca7ab2 100644 --- a/docs/design/REPO_ONBOARDING.md +++ b/docs/design/REPO_ONBOARDING.md @@ -45,7 +45,7 @@ interface BlueprintProps { repo: string; // "owner/repo" repoTable: dynamodb.ITable; compute?: { - type?: 'agentcore' | 'ecs'; // default: 'agentcore' + type?: 'agentcore' | 'ecs' | 'lambda-microvm'; // inherits first deployed backend runtimeArn?: string; config?: Record; }; @@ -118,7 +118,7 @@ From lowest to highest priority: | Field | Default | Source | |---|---|---| -| `compute_type` | `agentcore` | Platform constant | +| `compute_type` | First deployed backend (`agentcore` with legacy context) | Ordered `DEPLOYED_COMPUTE_TYPE` list on the orchestrator; `ComputeTypes`, `ComputeSubstrate` and `ComputeDeploymentMode` outputs for CLI discovery | | `runtime_arn` | Stack-level env var | CDK stack props | | `model_id` | `global.anthropic.claude-opus-5` | injected by the stack as `ANTHROPIC_MODEL` from `bedrockGeoRegion`; `agent/src/config.py` holds the no-env fallback — see [Model configuration](../guides/DEVELOPER_GUIDE.md#model-configuration) | | `max_turns` | 100 | Platform constant | @@ -239,7 +239,7 @@ interface ComputeStrategy { } ``` -The `agentcore` strategy implements `startSession` via `invoke_agent_runtime`, `pollSession` via re-invocation with sticky routing, and `stopSession` via `stop_runtime_session`. Alternative strategies (e.g. `ecs`) implement the same interface. The backend is selected per repo via `compute_type` in the Blueprint. +The `agentcore` strategy implements `startSession` via `invoke_agent_runtime`, `pollSession` via re-invocation with sticky routing, and `stopSession` via `stop_runtime_session`. Alternative strategies (e.g. `ecs`) implement the same interface. The deployment selects one or more backends with `compute_types`. A Blueprint `compute_type` override must name one of them; omit the override to inherit the first backend in the list. Legacy `compute_type` deployment contexts continue to include AgentCore as the default. See [Compute](./COMPUTE.md#selecting-and-changing-the-backend) before changing an existing deployment. ## Re-onboarding diff --git a/docs/guides/DEPLOYMENT_GUIDE.md b/docs/guides/DEPLOYMENT_GUIDE.md index 9c66085df..e2f4997dc 100644 --- a/docs/guides/DEPLOYMENT_GUIDE.md +++ b/docs/guides/DEPLOYMENT_GUIDE.md @@ -4,7 +4,7 @@ This guide covers deploying ABCA into an AWS account, including compute backend ## Architecture overview -ABCA deploys as a **single CDK stack** (`backgroundagent-dev`) containing all platform resources. The stack uses a `ComputeStrategy` interface to support three compute backends within the same stack: +ABCA deploys from the `backgroundagent-dev` application stack with nested stacks for selected subsystems. Networking stays in that stack by default; `networkTopology=split` gives it a separate top-level stack. A deployment can provision one or more compute backends: | Aspect | AgentCore (default) | ECS Fargate (opt-in) | Lambda MicroVMs (experimental) | |--------|--------------------|--------------------|--------------------| @@ -17,16 +17,87 @@ ABCA deploys as a **single CDK stack** (`backgroundagent-dev`) containing all pl All backends are orchestrated by the same durable Lambda function. The `ComputeStrategy` interface abstracts `startSession()`, `pollSession()`, and `stopSession()` -- the ECS strategy calls `ecs:RunTask` / `ecs:DescribeTasks` / `ecs:StopTask` directly from the Lambda. No Step Functions are used. -ECS Fargate is currently **opt-in** -- the `EcsAgentCluster` construct is present in the stack code but commented out. To enable it, uncomment the ECS blocks in `cdk/src/stacks/agent.ts`. +AgentCore is the default. `compute_types` selects a comma-separated list (or a JSON array in `cdk/cdk.json`); its first entry is the repository default. For example, `-c compute_types=agentcore,ecs` deploys both, while `-c compute_types=ecs` deploys only ECS. Repository overrides can select any deployed backend. Memory, Gateway, Registry and the Linear vault remain independently configurable. + +Without `compute_types`, legacy `compute_type=ecs` or `compute_type=lambda-microvm` keeps AgentCore alongside that backend, including on upgrade. An unchanged legacy context therefore does not remove AgentCore. Use the updated CLI for lists whose default is not AgentCore or that omit AgentCore; it reads the ordered `ComputeTypes` output. See the [backend transition procedure](../design/COMPUTE.md#selecting-and-changing-the-backend) before deliberately removing a backend or changing the default. + +### Network stack topology + +`networkTopology=inline` is the default. Use `networkTopology=split` for a new environment to put the VPC, subnets, endpoints, flow logs and DNS firewall in `${stackName}-network`. The application stack retains its name, data stores, compute resources and shared Task API. It depends on network exports, so CDK deploys the network first. The complete VPC/subnet/security-group export set stays present across compute-backend changes. Keep the same topology context on subsequent synth, diff and deploy commands. + +Every local and pipeline synthesis enforces a **490-resource ceiling per parent or nested template**, including operator configurations outside the census. CDK fails synthesis with the stack name, resource count and ceiling when a template exceeds it. `@aws-cdk/core:stackResourceLimit` accepts a stricter integer from 1 to 490, as either a JSON number or CLI string; it cannot raise the production ceiling. + +The budget covers complete configurations, including multiple backends, Gateway, Registry, the Linear vault, managed MicroVM images, alert email, a fork Blueprint, explicit three-zone pins and expanded model allowlists. Wider inline combinations exceed 490 and fail synthesis; their split counterparts fit in the sampled matrix. The budget guard never changes topology automatically. The [offline census](./DEVELOPER_GUIDE.md#stack-decomposition-and-synthesis-budgets) reports counts for each supported profile. + +For a **new installation**: + +```bash +MISE_EXPERIMENTAL=1 mise //cdk:deploy -- --all \ + -c networkTopology=split -c compute_types=agentcore +``` + +Replace the compute list as needed and supply image settings for MicroVM. Persist the selected context in `cdk/cdk.json` for subsequent synth, diff and deploy commands. The split preserves the supported-AZ policy, HTTPS egress, endpoints and DNS observation mode. Configure additional Blueprint domains in `cdk/src/blueprints/definitions.ts`; both stacks consume those inputs before repository resources are constructed. + +**Existing inline-to-split migration is deferred.** Do not flip `networkTopology` on an existing inline deployment. An ordinary deploy creates a new VPC and removes old resources; identical logical IDs in different stacks do not preserve physical identity. Keep the existing topology until a separately reviewed migration has established resource-type eligibility, source retention and cleanup behavior, an ownership mapping, physical-ID/data preservation, and rollback on a populated disposable deployment. The `cdk refactor --unstable=refactor` criterion in #852 remains open; it is not waived or satisfied by synthesis tests. + +This change does not install broad retention, change the Blueprint provider, create an ownership ledger, or replace the guardrail versioning scheme. Those prerequisites belong in separate releases. CloudFormation exports still prevent removing a network used by the application; switching back to `inline` is not an automatic rollback. See [ADR-023's deferred work](../decisions/ADR-023-cloudformation-stack-boundaries.md#deferred-migration-work). + +Stacks used to test earlier revisions with `blueprintProvisioning=prepare|adopt|managed` or the broad retention aspect require a separate recovery plan before adopting this narrowed version. Returning those stacks directly to the original Blueprint provider can overwrite or soft-delete repository rows. Synthesis rejects the removed `blueprintProvisioning` and `guardrailVersionMigration` context keys instead of silently ignoring them; removing those keys does not make an experimental stack safe to update. Preserve its deployed templates, repository data and ledger inventory; the removed experimental modes are not an upgrade path supported by this release. + +#### Reducing AZs in an existing split network + +A normal `--all` deployment updates the network first, so removing an AZ can fail because the old application still imports its private-subnet export. Reducing the count also shifts CDK's private-subnet CIDRs unless the vacated address slot stays reserved. Use the following staged procedure for a **three-to-two-zone reduction that keeps the first two existing AZs in their original order**. Replacing or reordering AZs requires a separate network migration. + +1. Pause automated deployments, task submissions, webhooks and scheduled work. Drain running and suspended sessions. Record the deployed templates, AZ order, subnet CIDRs, physical IDs and network exports. Keep the same account, region, stack identity, backend, image and Blueprint configuration throughout. +2. Persist the target AZ list in the existing `cdk/cdk.json` context, keeping `networkTopology=split`. Increase `networkReservedAzs` by the number of removed trailing AZs, so active plus reserved slots remains constant. For three active zones with no reservations, the target is two active zones and one reserved slot. These example names must match the deployment's first two AZs: + + ```json + "agentcore:availabilityZones": ["us-east-1a", "us-east-1b"], + "networkReservedAzs": 1 + ``` + + `networkReservedAzs` accepts an integer from 0 to 6, as a JSON number or CLI string; the default is 0. It reserves address space and creates no AWS resources. Keep this setting in every subsequent synth/deploy, including automation. +3. Set `APP_STACK` to the existing application stack name and review both target templates. The remaining subnets must keep their logical IDs, CIDRs and AZs; the target application must stop importing the removed subnet. Stop if the diff changes a retained subnet or any unrelated configuration. + + ```bash + APP_STACK=backgroundagent-dev + MISE_EXPERIMENTAL=1 mise //cdk:diff -- --all --method template + ``` + +4. Deploy **only the application**, leaving the existing three-zone network in place. `--exclusively` prevents CDK from deploying its network dependency: + + ```bash + MISE_EXPERIMENTAL=1 mise //cdk:deploy -- "$APP_STACK" --exclusively + ``` + +5. Copy each removed private-subnet export's exact name from the deployed network outputs. Verify that `list-imports` returns `[]` before changing the network. Any additional consumer stack must also release that export. + + ```bash + aws cloudformation describe-stacks --stack-name "${APP_STACK}-network" \ + --query 'Stacks[0].Outputs[].{ExportName:ExportName,Value:OutputValue}' --output table + REMOVED_SUBNET_EXPORT='' + aws cloudformation list-imports --export-name "$REMOVED_SUBNET_EXPORT" \ + --query Imports --output json + ``` + +6. Deploy both stacks with the same persisted target context. The network can now remove the unused export and AZ resources. Verify the remaining subnet physical IDs and CIDRs, DNS/egress behavior and a task before resuming producers and automation. + + ```bash + MISE_EXPERIMENTAL=1 mise //cdk:deploy -- --all + ``` + +To restore the previous three-zone layout, restore its AZ list and reservation count, then deploy the network before the application (`--all` uses this order). The network must recreate the third subnet and export before the application imports it again. Keep active plus reserved slots constant during this recovery too. + +Local synthesis tests verify the import ordering and unchanged remaining subnet properties for all three backends. This procedure still requires a disposable AWS rehearsal before a production network update. ### Lambda MicroVMs backend (experimental) > **Not for production.** `lambda-microvm` carries no smoke-parity guarantee for an unattended deployment. Keep production repositories on `agentcore` or `ecs`. Synth emits an unsuppressible warning to this effect whenever the backend is selected. Design detail: [COMPUTE.md](../design/COMPUTE.md) and [ADR-021](../decisions/ADR-021-lambda-microvms-compute-backend.md). -Selecting it is a synth-time context flag: +Include it in the deployment's backend list, preserving any backends already in use: ```bash -mise //cdk:deploy -- --context compute_type=lambda-microvm +mise //cdk:deploy -- --context compute_types=agentcore,lambda-microvm ``` **You must re-bootstrap first.** This is the single most common way this backend fails, and the failure does not look like a configuration problem: @@ -69,7 +140,7 @@ Blueprints without `registry://` asset references continue to work. A remaining The string form is case-sensitive: use lowercase `true` or `false`. Any other value fails synthesis with an actionable validation error. -This context is an infrastructure switch, not a pause control. Applying it to an existing enabled deployment deletes the CloudFormation-managed registry and its records; re-enabling creates an empty registry that must be republished. See [REGISTRY.md](../design/REGISTRY.md) for the catalog migration and runtime behavior. +This context is an infrastructure switch, not a pause control. Applying it to an existing enabled deployment deletes the CloudFormation-managed registry and its records; re-enabling creates an empty registry that must be republished. The optional network split preserves that lifecycle; registry retention remains deferred. See [REGISTRY.md](../design/REGISTRY.md) for the catalog migration and runtime behavior. ## Bedrock inference geography @@ -135,7 +206,7 @@ At public US East (N. Virginia) first-tier list rates verified in August 2026, t |---------|---------|---------------| | Bedrock AgentCore Runtime (MicroVMs) | Agent sessions (default) | Yes | | ECS Fargate (when enabled) | Agent sessions (opt-in) | Yes | -| AWS Lambda MicroVMs (when enabled) | Agent sessions (experimental, `--context compute_type=lambda-microvm`) | Yes | +| AWS Lambda MicroVMs (when enabled) | Agent sessions (experimental, include `lambda-microvm` in `compute_types`) | Yes | | Lambda (Node.js 24, ARM64) | Orchestrator, API handlers, fanout consumer, reconcilers, custom resources | Yes | ### AI/ML @@ -309,6 +380,8 @@ AGENTCORE_AVAILABILITY_ZONES = ["us-east-1b","us-east-1c"] The override is validated at synth time, and both the JSON-array and `-c` string forms behave identically. Synth fails with a message naming the key when the value is not an array, has an empty/non-string entry, lists fewer than two **distinct** zones, contains zone *IDs* instead of names (`use1-az2` — a common column mix-up), or names zones outside the target region. When the account's mapping is knowable, the override is additionally cross-checked against the supported set, and unsupported or nonexistent zones fail synth. +Auto-pin selects two supported zones; an explicit override uses all the zones supplied. Pins above two zones remain subject to the 490-resource production ceiling. Configurations that exceed it need split networking; see [Network stack topology](#network-stack-topology) for the measured boundaries and migration requirements. + **Upgrading an existing stack.** Auto-pin is on by default, so a local `cdk deploy` against a stack created before this change may select different zones than the deployed subnets use. `Subnet.AvailabilityZone` is create-only, so that is a **replacement** of the subnets and the resources bound to them (route tables, NAT gateway/EIP, VPC endpoints). Run `mise //cdk:diff` first. If the diff shows subnet replacement and you would rather keep the current topology, pin the override to the zones already deployed: ```bash @@ -316,7 +389,28 @@ aws ec2 describe-subnets --filters "Name=vpc-id,Values=" \ --query 'Subnets[].[SubnetId,AvailabilityZone,AvailabilityZoneId]' --output text ``` -Be aware that destroying a VPC whose subnets held AgentCore ENIs can take 20–40 minutes while AWS reclaims them (see the `DELETE_FAILED` note in the [quick start](./QUICK_START.mdx) troubleshooting table). +### Teardown blocked by AgentCore network interfaces + +Deleting an AgentCore Runtime does not immediately release its service-managed network interfaces. The live review of [#912](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/912) observed `agentic_ai` interfaces attached by `amazon-aws` still in use after five hours. They can block RuntimeSG, subnet and VPC deletion and leave a stack in `DELETE_FAILED`. This also affects ordinary destroys from `main`; it is not a fixed 20–40 minute delay. + +Inspect the failed stack and its VPC before retrying: + +```bash +aws cloudformation list-stack-resources --stack-name \ + --query 'StackResourceSummaries[?ResourceStatus==`DELETE_FAILED`].[LogicalResourceId,PhysicalResourceId,ResourceType]' \ + --output table +aws ec2 describe-network-interfaces --filters Name=vpc-id,Values= +``` + +Let AgentCore release its interfaces; do not try to force-detach interfaces owned by `amazon-aws`. To finish deleting a stack already in `DELETE_FAILED`, inventory the blocked network resources and their dependencies, then retain those exact **logical IDs** in a deletion retry: + +```bash +aws cloudformation delete-stack --stack-name \ + --retain-resources +``` + +For split deployments, identify whether the failed resources belong to the application or the `-network` stack and target that stack. Keep physical IDs for every retained resource and track their cost. After the service releases the ENIs, clean up retained security groups, subnets and other VPC dependencies before deleting the VPC. `--retain-resources` does not perform that later cleanup or make the resources reusable by a same-name reinstall. Reconcile any retained resources from earlier experimental builds separately. + ### DNS Query Log Config replacement cascade (upgrading from pre-v0.5) diff --git a/docs/guides/DEVELOPER_GUIDE.md b/docs/guides/DEVELOPER_GUIDE.md index eb9a6dfcd..10cbce3cd 100644 --- a/docs/guides/DEVELOPER_GUIDE.md +++ b/docs/guides/DEVELOPER_GUIDE.md @@ -59,21 +59,21 @@ The default is `awslabs/agent-plugins`. For a quick end-to-end test, fork that r ### Multiple repositories -To onboard additional repositories, add more `Blueprint` constructs in `cdk/src/stacks/agent.ts` and append them to the `blueprints` array (used to aggregate DNS egress allowlists): +To onboard additional repositories, add entries to `resolveBlueprintDefinitions` in `cdk/src/blueprints/definitions.ts`. The app resolves these plain inputs before constructing either stack, so repository provisioning and DNS egress policy use the same configuration: ```typescript -new Blueprint(this, 'MyServiceBlueprint', { +definitions.push({ + id: 'MyServiceBlueprint', repo: 'acme/my-service', - repoTable: repoTable.table, }); ``` -Each Blueprint supports per-repo overrides grouped into nested props (`BlueprintProps` in `cdk/src/constructs/blueprint.ts`): +Each entry supports the per-repo overrides from `BlueprintProps` in `cdk/src/constructs/blueprint.ts`, without a table reference. Keep its `id` stable across releases: ```typescript -new Blueprint(this, 'MyServiceBlueprint', { +definitions.push({ + id: 'MyServiceBlueprint', repo: 'acme/my-service', - repoTable: repoTable.table, compute: { runtimeArn: '...' }, // override the default runtime ARN agent: { modelId: 'global.anthropic.claude-opus-5', // foundation model override @@ -99,6 +99,24 @@ The command defaults to **`mise run build`** / **`mise run lint`**. A repo that Redeploy after changing Blueprints: `mise //cdk:deploy`. +### Stack decomposition and synthesis budgets + +`networkTopology` defaults to `inline`, preserving existing network ownership. For a new installation, `split` puts AgentVpc and DnsFirewall in `${stackName}-network`; the application consumes VPC, subnet and security-group exports. The network has no application dependencies. Plain Blueprint definitions are resolved once in `cdk/src/blueprints/definitions.ts` so DNS and repository provisioning use the same domains without cross-stack coupling. Task API routes, authorizers, permissions, CORS and deployment remain together in `AgentStack`. + +The CDK build and offline census share 122 named profiles: the 40-cell single-backend/service/image product in both topologies, supplemental email/fork/consent cases, explicit three-AZ pins, expanded `bedrockModels` grants, every multi-backend set at default and widest settings, and legacy additive selectors. Successful profiles check every parent and nested template against 490 resources, 800,000 bytes and 200 parameters/outputs. They also verify method-scoped API permissions, absence of console test-invoke grants and template-size warnings, backend membership, and two-zone auto-pin versus explicit pins. The model-expansion cases add eight synthetic IDs to the platform defaults and assert that the SessionRole's generated IAM overflow policies pass the audit; these fixtures test policy size, not model availability. Expected resource-budget failures must identify the application stack and the production ceiling; unrelated errors cannot satisfy the gate. + +```bash +MISE_EXPERIMENTAL=1 mise //cdk:census -- --output /tmp/stack-census +``` + +Use `--list` to see the profiles and `--profile NAME` to select them. The census runs the production app with fixed account/AZ inputs, CDK metadata enabled and bundling/staging disabled. It records template inventories, counts, bytes, dependencies and source provenance. The production app sets CDK's `@aws-cdk/core:stackResourceLimit` before any stack is constructed, so actual operator configurations outside the census also fail above 490. Context overrides may tighten but cannot raise the limit. `--max-resources` can only tighten the census audit ceiling. + +`--check-stability` runs each selected profile twice in independent processes and fails on differences. It does not normalize away timestamps, logical IDs or asset hashes. The existing Blueprint callbacks embed synthesis-time timestamps, and the alpha Bedrock guardrail uses token-derived version IDs; unchanged source can therefore fail this optional diagnostic. Deterministic Blueprint provisioning, guardrail versioning and Docker build-context changes are deferred. Passing the budget gate is not a claim of repeatable synthesis or live resource preservation. + +Network tests compare the moved definitions, generated Name tags, endpoint security-group descriptions, exports, application service properties, shared API resources, solution attribution and provenance tags. Comparison tests fix the clock and account for existing immutable guardrail/orchestrator version IDs; the census reports those real differences. `networkReservedAzs` reserves unused address slots so removing a trailing AZ need not shift remaining subnet CIDRs. Follow the [staged AZ reduction procedure](./DEPLOYMENT_GUIDE.md#reducing-azs-in-an-existing-split-network) to release old imports before changing the network. + +Existing inline-to-split migration remains deferred under [#852](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/852). A populated `cdk refactor`/import and rollback rehearsal is still required before a supported migration procedure can be published. Retention and Blueprint ownership handoff must be separately reviewed and released; this change keeps existing removal policies and the existing Blueprint provider. The concrete follow-up requirements are recorded in [ADR-023](../decisions/ADR-023-cloudformation-stack-boundaries.md#deferred-migration-work). See [Network stack topology](./DEPLOYMENT_GUIDE.md#network-stack-topology) for fresh-install guidance and migration limits. + ### Customizing the agent image The default image (`agent/Dockerfile`) includes Python, Node 24 (LTS), `git`, `gh`, Claude Code CLI, and `mise`. If your repositories need additional runtimes (Java, Go, native libs), extend the Dockerfile. A normal `cdk deploy` rebuilds the image asset. diff --git a/docs/guides/LINEAR_SETUP_GUIDE.md b/docs/guides/LINEAR_SETUP_GUIDE.md index a788fa297..90488fb9e 100644 --- a/docs/guides/LINEAR_SETUP_GUIDE.md +++ b/docs/guides/LINEAR_SETUP_GUIDE.md @@ -24,7 +24,7 @@ One of two places, chosen automatically at setup time: | **AgentCore Identity vault** | The stack was deployed with `--context enableLinearIdentityVault=true` | Nothing long-lived. AgentCore holds the refresh token and mints short-lived access tokens on demand. | | **Secrets Manager** | Otherwise — including regions where AgentCore Identity isn't available | An OAuth token bundle in `bgagent-linear-oauth-`, refreshed and rotated by ABCA. | -The vault is unavailable on the `lambda-microvm` substrate — see [Not available with `compute_type=lambda-microvm`](#not-available-with-compute_typelambda-microvm) below. +The vault can be configured with any selected backend. Lambda MicroVMs remains experimental; its image must include the current shared configuration contract. `bgagent linear setup` picks whichever the deployment supports and tells you which one it used. There is no flag. If the vault isn't available it prints one line and continues on Secrets Manager: @@ -36,9 +36,9 @@ A workspace that started on Secrets Manager and later moves to the vault **keeps When a workspace's authorization dies, ABCA records it on the registry row and publishes to the stack's operational alert topic. That topic has **no subscribers unless you deployed with `alertEmail`**, so set it if you want to hear about a dead workspace rather than discover it from `bgagent platform doctor`. -#### Not available with `compute_type=lambda-microvm` +#### Using the vault with Lambda MicroVMs -The vault and the Lambda MicroVMs substrate cannot be enabled on the same stack. Together they synthesize 505 CloudFormation resources against a hard limit of 500 (MicroVM alone is 496, the vault alone 488), so `cdk deploy` refuses the combination by name at synth rather than failing partway through. Use the vault on the `agentcore` or `ecs` substrate; a MicroVM stack stays on Secrets Manager until the stack reclaims room. +Use split networking when the full MicroVM-plus-vault configuration exceeds the 490-resource budget. `compute_types` can include MicroVM alongside AgentCore or ECS, or select MicroVM alone; an unchanged legacy `compute_type=lambda-microvm` keeps AgentCore. The guest execution role receives the mint grant, and the `/run` configuration carries `LINEAR_VAULT_ENABLED` and `LINEAR_WORKLOAD_IDENTITY_NAME`. Rebuild the MicroVM image from this checkout before enabling the vault. Verify consent and token minting in a live rehearsal; synthesis alone does not qualify this experimental backend. #### One workload identity per stack diff --git a/docs/src/content/docs/architecture/Architecture.md b/docs/src/content/docs/architecture/Architecture.md index fa87f3b06..88a176092 100644 --- a/docs/src/content/docs/architecture/Architecture.md +++ b/docs/src/content/docs/architecture/Architecture.md @@ -41,9 +41,20 @@ The orchestrator and agent are deliberately separated. The orchestrator handles For the full orchestrator design, see [ORCHESTRATOR.md](/sample-autonomous-cloud-coding-agents/architecture/orchestrator). For the API contract, see [API_CONTRACT.md](/sample-autonomous-cloud-coding-agents/architecture/api-contract). +## Deployment boundaries + +`AgentStack` keeps the shared Task API, its route integrations, data stores and selected compute backends together. Registry, RegistryApi and the hosted Linear consent page retain their existing nested boundaries. Network ownership is selected by `networkTopology`: + +| Topology | Network ownership | Stack dependencies | +|---|---|---| +| `inline` (default) | AgentVpc and DnsFirewall inside the application stack | Existing parent/nested structure | +| `split` | Separate `${stackName}-network` stack | Application imports network references; network has no application references | + +The split gives networking an independent deployment lifecycle and reduces the application template's resource count. It preserves AZ selection, DNS observation mode and security rules. All stacks receive solution attribution and provenance tags, while existing removal policies remain in effect. The split is available for new installations; existing inline-to-split migration is deferred pending a populated rehearsal; see [deployment guidance](/sample-autonomous-cloud-coding-agents/getting-started/deployment-guide#network-stack-topology) and [ADR-023](/sample-autonomous-cloud-coding-agents/architecture/adr-023-cloudformation-stack-boundaries). Live migration has not been validated. + ## Repository onboarding -Onboarding is CDK-based. Each repository is an instance of the `Blueprint` construct in the stack. The construct writes a `RepoConfig` record to DynamoDB; the orchestrator reads it at task time. +Onboarding is CDK-based. Plain repository definitions in `cdk/src/blueprints/definitions.ts` feed both network egress policy and the `Blueprint` constructs in the application stack. Each construct writes a `RepoConfig` record to DynamoDB; the orchestrator reads it at task time. Resolving configuration before stack construction keeps the optional network stack independent of repository resources. Blueprints configure how the orchestrator executes steps for each repo: compute strategy, model selection, turn limits, GitHub token, and optional custom steps. See [REPO_ONBOARDING.md](/sample-autonomous-cloud-coding-agents/architecture/repo-onboarding) for the full design. diff --git a/docs/src/content/docs/architecture/Compute.md b/docs/src/content/docs/architecture/Compute.md index c574d6350..d9a6603fa 100644 --- a/docs/src/content/docs/architecture/Compute.md +++ b/docs/src/content/docs/architecture/Compute.md @@ -11,7 +11,7 @@ Every task runs in an isolated cloud compute environment. Nothing runs on the us ## Compute options -The default runtime is **Amazon Bedrock AgentCore Runtime**, which runs each session in a Firecracker MicroVM with per-session isolation, managed lifecycle, and built-in health monitoring. For repos that exceed AgentCore's constraints (2 GB image limit, no GPU), the `ComputeStrategy` interface allows switching to alternative backends per repo. +The default runtime is **Amazon Bedrock AgentCore Runtime**, which runs each session in a Firecracker MicroVM with per-session isolation, managed lifecycle, and built-in health monitoring. The `compute_types` CDK context selects one or more backends: `agentcore`, `ecs`, and `lambda-microvm`. Its first entry is the default for repositories; a repository can explicitly select any deployed backend. The `ComputeStrategy` interface dispatches each task to its resolved backend. | | AgentCore Runtime | ECS on Fargate | **Lambda MicroVMs** | ECS on EC2 | EKS | AWS Batch | Lambda (functions) | Custom EC2 + Firecracker | |---|---|---|---|---|---|---|---|---| @@ -27,7 +27,45 @@ The default runtime is **Amazon Bedrock AgentCore Runtime**, which runs each ses > **Lambda MicroVMs are not Lambda functions.** They are a different compute primitive, so the functions column's 15-minute cap and poor-fit verdict do not apply. See [ADR-021](/sample-autonomous-cloud-coding-agents/architecture/adr-021-lambda-microvms-compute-backend). -The backend is selected per repo via `compute_type` in the Blueprint config. The orchestrator resolves the strategy and delegates session start, polling, and termination to the strategy implementation. See [REPO_ONBOARDING.md](/sample-autonomous-cloud-coding-agents/architecture/repo-onboarding) for the `ComputeStrategy` interface. +Repositories without a `compute_type` override inherit the first entry in the deployed list. An explicit Blueprint or RepoTable override must name a deployed backend; mismatches fail before concurrency admission. Memory, Tool Gateway, Agent Registry and Linear Identity Vault are separate service choices, so selecting ECS or MicroVM does not disable them. The orchestrator resolves the strategy and delegates session start, polling, and termination to the strategy implementation. See [REPO_ONBOARDING.md](/sample-autonomous-cloud-coding-agents/architecture/repo-onboarding) for the `ComputeStrategy` interface. + +## Selecting and changing the backend + +Set `compute_types` in `cdk/cdk.json` as an array, or pass a comma-separated list to the deployment task: + +```bash +# AgentCore and MicroVM, with AgentCore as the repository default. +MISE_EXPERIMENTAL=1 mise //cdk:deploy -- -c compute_types=agentcore,lambda-microvm + +# Only ECS; removing an existing backend requires the transition below. +MISE_EXPERIMENTAL=1 mise //cdk:deploy -- -c compute_types=ecs +``` + +`compute_types` takes precedence over the legacy `compute_type` context. Empty lists, non-string entries and unsupported names fail synthesis. Duplicate names are collapsed while preserving order. Without `compute_types`, the previous deployment behavior is preserved: + +| Legacy context | Deployed backends | Repository default | +|---|---|---| +| Absent, or `compute_type=agentcore` | AgentCore | AgentCore | +| `compute_type=ecs` | AgentCore and ECS | AgentCore | +| `compute_type=lambda-microvm` | AgentCore and Lambda MicroVMs | AgentCore | + +An upgrade with the same legacy context keeps AgentCore, its role and its log delivery. Selecting a single optional backend explicitly, such as `compute_types=ecs`, opts into removing AgentCore. The bootstrap `ComputeTypes` CloudFormation parameter is a separate permission allowlist; enable every backend you plan to deploy there too. + +`ComputeTypes` and `ComputeSubstrate` both publish the complete ordered comma list. `ComputeDeploymentMode` is `exclusive` for one backend and `additive` for several. `RuntimeArn` exists whenever AgentCore is included. The orchestrator receives the same list in `DEPLOYED_COMPUTE_TYPE`. Stack-level `compute_type` tags join the deployed names with `+`, a tag-safe separator; backend-specific resources keep their own attribution. + +Use the updated CLI when the first backend is not AgentCore, or the list omits AgentCore. Older CLIs assume AgentCore is present and is the default on additive stacks. The updated CLI reads `ComputeTypes` for onboarding, `repo show`, and `runtime status`. Stacks without the new outputs retain the legacy interpretation. + +`bgagent repo show` and `bgagent runtime status` mark incompatible stored pins as **UNAVAILABLE** and explain how to reconcile them. JSON includes the ordered `compute_deployment.compute_types` array, `default_compute_type`, `compute_available`, and `configuration_error` (per repository in runtime status). Incompatible repositories are excluded from backend summaries and AgentCore probes. This checks deployment configuration; MicroVM image readiness and live backend health remain separate checks. + +Before deliberately removing a backend or changing the first entry: + +1. Record deployed context, templates, image identifiers, repository overrides and active sessions. Pause task submissions and let running or suspended work finish, or cancel it with the existing deployment. +2. Reconcile stored overrides with the target list. Omitted `compute_type` inherits its first entry; explicit pins to other deployed backends remain valid. Remove obsolete `runtime_arn` overrides when leaving AgentCore. +3. Prepare images and bootstrap permissions. MicroVM needs a compatible snapshot; rebuild it from this checkout before enabling Gateway or the vault because their optional settings travel through `platform_config`. Infrastructure without an image cannot run tasks. +4. Review the full change set. Removing AgentCore removes its Runtime and delivery resources. The two named AgentCore log groups stay owned by the application across backend switches, with their existing destroy policies and retention periods. Shared data stores keep their existing lifecycle policies. A network ownership transfer is a separate, deferred migration. +5. Rehearse deployment and rollback, then verify a task on each backend, repository-less work, cancellation, logs and enabled integrations before resuming submissions. Re-adding a backend cannot resume deleted sessions or recover runtime storage. + +The resource budget applies to the complete combination. The split topology fits the sampled combinations, including all three backends; wider inline combinations fail at the 490-resource ceiling. See [Network stack topology](/sample-autonomous-cloud-coding-agents/getting-started/deployment-guide#network-stack-topology). Local synthesis verifies wiring and budget behavior; it does not establish a live migration guarantee or change MicroVM's experimental status. ## What runs in the session @@ -81,7 +119,7 @@ See [ORCHESTRATOR.md](/sample-autonomous-cloud-coding-agents/architecture/orches ## Lambda MicroVMs backend -Lambda MicroVMs are an opt-in third backend, selected per repository with `compute_type: lambda-microvm`; AgentCore remains the default. Image configuration has three states: a managed base-image ARN and version creates the snapshot image in CDK; an external image identifier uses a snapshot built out of band; and supplying neither provisions only the roles, buckets, and connectors needed for the bootstrap deploy. `cdk/scripts/package-microvm-artifact.sh` packages the agent as zip + Dockerfile, uploads it to the artifact bucket, and can create the external image. Lambda MicroVMs are available in five launch regions (us-east-1, us-east-2, us-west-2, eu-west-1, ap-northeast-1) and will expand; the platform enforces regional availability in layers via a synth-time constant, onboarding live probes, and orchestration-time classification. +Lambda MicroVMs are an opt-in third backend. Include `lambda-microvm` in `compute_types`; repositories inherit the first listed backend or choose a deployed backend through an override. Legacy `compute_type=lambda-microvm` deploys AgentCore and MicroVM, with AgentCore as the default. Image configuration has three states: a managed base-image ARN and version creates the snapshot image in CDK; an external image identifier uses a snapshot built out of band; and supplying neither provisions only the roles, buckets, and connectors needed for the bootstrap deploy. `cdk/scripts/package-microvm-artifact.sh` packages the agent as zip + Dockerfile, uploads it to the artifact bucket, and can create the external image. Lambda MicroVMs are available in five launch regions (us-east-1, us-east-2, us-west-2, eu-west-1, ap-northeast-1) and will expand; the platform enforces regional availability in layers via a synth-time constant, onboarding live probes, and orchestration-time classification. Because a snapshot freezes its build-time environment, deployment-specific, non-secret identifiers travel in the `/run` hook's `platform_config` block instead. The strategy sends the canonical inline envelope or, when that envelope exceeds the verified 4,096-byte `runHookPayload` limit, an S3-pointer envelope with the configuration also merged into the uploaded payload. The agent accepts only allowlisted keys and installs them before pipeline initialization; [ADR-021 §3](/sample-autonomous-cloud-coding-agents/architecture/adr-021-lambda-microvms-compute-backend#3-packaging-same-agent-image-source-new-build-path) defines the exact wire shapes and validation rules. diff --git a/docs/src/content/docs/architecture/Repo-onboarding.md b/docs/src/content/docs/architecture/Repo-onboarding.md index 03d04de47..80ef5373c 100644 --- a/docs/src/content/docs/architecture/Repo-onboarding.md +++ b/docs/src/content/docs/architecture/Repo-onboarding.md @@ -49,7 +49,7 @@ interface BlueprintProps { repo: string; // "owner/repo" repoTable: dynamodb.ITable; compute?: { - type?: 'agentcore' | 'ecs'; // default: 'agentcore' + type?: 'agentcore' | 'ecs' | 'lambda-microvm'; // inherits first deployed backend runtimeArn?: string; config?: Record; }; @@ -122,7 +122,7 @@ From lowest to highest priority: | Field | Default | Source | |---|---|---| -| `compute_type` | `agentcore` | Platform constant | +| `compute_type` | First deployed backend (`agentcore` with legacy context) | Ordered `DEPLOYED_COMPUTE_TYPE` list on the orchestrator; `ComputeTypes`, `ComputeSubstrate` and `ComputeDeploymentMode` outputs for CLI discovery | | `runtime_arn` | Stack-level env var | CDK stack props | | `model_id` | `global.anthropic.claude-opus-5` | injected by the stack as `ANTHROPIC_MODEL` from `bedrockGeoRegion`; `agent/src/config.py` holds the no-env fallback — see [Model configuration](/sample-autonomous-cloud-coding-agents/developer-guide/model-configuration) | | `max_turns` | 100 | Platform constant | @@ -243,7 +243,7 @@ interface ComputeStrategy { } ``` -The `agentcore` strategy implements `startSession` via `invoke_agent_runtime`, `pollSession` via re-invocation with sticky routing, and `stopSession` via `stop_runtime_session`. Alternative strategies (e.g. `ecs`) implement the same interface. The backend is selected per repo via `compute_type` in the Blueprint. +The `agentcore` strategy implements `startSession` via `invoke_agent_runtime`, `pollSession` via re-invocation with sticky routing, and `stopSession` via `stop_runtime_session`. Alternative strategies (e.g. `ecs`) implement the same interface. The deployment selects one or more backends with `compute_types`. A Blueprint `compute_type` override must name one of them; omit the override to inherit the first backend in the list. Legacy `compute_type` deployment contexts continue to include AgentCore as the default. See [Compute](/sample-autonomous-cloud-coding-agents/architecture/compute#selecting-and-changing-the-backend) before changing an existing deployment. ## Re-onboarding diff --git a/docs/src/content/docs/decisions/Adr-016-pluggable-identity-and-auth.md b/docs/src/content/docs/decisions/Adr-016-pluggable-identity-and-auth.md index d30b13960..39d2589e2 100644 --- a/docs/src/content/docs/decisions/Adr-016-pluggable-identity-and-auth.md +++ b/docs/src/content/docs/decisions/Adr-016-pluggable-identity-and-auth.md @@ -164,7 +164,7 @@ Per-user `McpCredential` selection requires the Gateway to know *which task-user | P6 | **Trusted task-user identity propagation** (prerequisite for per-user MCP on the general plane, P5): specify + validate a user-scoped inbound identity the Gateway authorizer trusts, replacing the M2M JWT for per-user credential selection. | **Blocks per-user `McpCredential`.** Until done, MCP credentials are workspace-scoped at best. | | P7 | Jira + Slack `ChannelCredential` (same shape as P1); GitHub `GithubOauth2` behind a flag, retire the shared PAT; OBO `act`-claim delegation feeding #237. | Flag-gated; per-surface. | -**Substrate exception — `lambda-microvm` (2026-09-02):** P1's vault cannot be enabled on the MicroVM substrate. The two together synthesize 505 resources against CloudFormation's hard 500-resource limit (MicroVM alone 496, the vault alone 488), so `AgentStack` refuses the combination at synth, naming both context flags. The MicroVM wiring itself is complete — `platform_config` carries the workload name and the guest execution role holds the mint grant — so this is a capacity limit, not a design gap, and it lifts as soon as a subsystem moves into a nested stack. +**Substrate update — `lambda-microvm` (#852):** The optional split network provides headroom for MicroVM plus the vault, including additive deployments that retain AgentCore. Explicit `compute_types=lambda-microvm` also permits a MicroVM-only deployment. The production 490-resource guard validates the complete configuration. The guest execution role has the mint grant; the shared `platform_config` contract now forwards vault enablement and workload identity. A rebuilt compatible MicroVM image and live consent/mint rehearsal are required before using the combination. Lambda MicroVMs remains experimental. **Substrate independence (verified 2026-07-21, both proven live):** the vault path works on any compute. AgentCore Runtime injects the Workload Access Token as the `WorkloadAccessToken` header; ECS/Fargate/Lambda bootstrap it via `GetWorkloadAccessTokenForJWT(workloadName, userToken=)` against a **standalone** (non-service-linked) workload identity, then call `GetResourceOauth2Token`. Runtime-managed (service-linked) workload identities cannot self-vend, so the ECS path needs a manually-created workload identity. The runtime execution role today has `GetWorkloadAccessToken*` but **not** `GetResourceOauth2Token` — P1 adds it, plus `GetSecretValue` scoped to that surface's providers (`bedrock-agentcore-identity!default/oauth2/*`; see the implementation notes below for why this is narrower than the wildcard first anticipated here). diff --git a/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md b/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md index fec83f5a4..fa42d0108 100644 --- a/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md +++ b/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md @@ -330,7 +330,7 @@ Two networking facts the construct has to encode, both established live: `lambda:CreateMicrovmAuthToken` is granted to no role in P1–P3 (no JWE consumer exists; see sub-decision 3). -**Cost attribution.** `cdk/src/main.ts` currently tags the whole stack with a single `compute_type` context value (default `agentcore`) — already imprecise with two backends, wrong with three. P1 must add backend-identifying cost-allocation tags on the MicroVM-specific resources (images, payload/artifact bucket wiring, log groups) and revisit the stack-level tag semantics (e.g. a `compute_types` list), keeping attribution consistent with [#645](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/645)'s cost/attribution acceptance criterion. +**Cost attribution update (#852).** The `compute_types` context selects an ordered backend list; legacy `compute_type` retains its additive meaning. The stack-level `compute_type` tag records all deployed names separated by `+` (for example, `agentcore+lambda-microvm`), which is valid in AWS tag values. MicroVM-specific resources retain their `abca:compute-backend` tags. Shared optional services remain independent of the compute selection. An unchanged legacy context preserves AgentCore; deliberate backend removal requires a drained transition; see [Compute](/sample-autonomous-cloud-coding-agents/architecture/compute#selecting-and-changing-the-backend). - Where a deployment enables the `lambda-microvm` backend, MicroVM-specific resources shall carry backend-identifying cost-allocation tags. diff --git a/docs/src/content/docs/decisions/Adr-023-cloudformation-stack-boundaries.md b/docs/src/content/docs/decisions/Adr-023-cloudformation-stack-boundaries.md new file mode 100644 index 000000000..b077c8671 --- /dev/null +++ b/docs/src/content/docs/decisions/Adr-023-cloudformation-stack-boundaries.md @@ -0,0 +1,64 @@ +--- +title: Adr 023 cloudformation stack boundaries +--- + +# ADR-023: Optional network stack and deployment budgets + +**Status:** proposed +**Date:** 2026-09-21 +**Last-updated:** 2026-10-05 +**Issue:** [#852](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/852) + +Per the [ADR lifecycle](/sample-autonomous-cloud-coding-agents/architecture/readme#lifecycle), this decision remains proposed while its implementing PR is in review and becomes accepted when that PR merges. This record does not approve or waive the existing-stack migration criteria in #852. + +## Context + +The application stack approaches CloudFormation's 500-resource limit. [#851](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/851) records the incident; approved issue #852 covers resource budgets, template bytes and stack boundaries. [#735](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/735) is related byte-limit evidence, not a separate approval for this implementation. + +The Task API owns most resources, but its integrations share one API Gateway RestApi and deployment. The #852 proof of concept showed that extracting an integration can lose deployment dependencies, CORS, solution attribution and tags even when synthesis passes. Networking has a smaller interface and an independent lifecycle. + +Live review of an earlier #912 revision found that broad retention blocks failed-create retries and same-name redeploys, while Blueprint ownership handoff can lose repository settings and orphan PITR-enabled ledger tables. Those changes are removed from this PR. The reviewed scope is the optional network stack, deployment budgets and compatible compute selection. + +## Decision + +1. Keep the Task API, authorizers, deployment, stage and route integrations together in the application stack. Preserve existing nested stacks for Registry, RegistryApi and hosted consent pages. +2. Offer `networkTopology=split` for new installations; `inline` remains the default. Move AgentVpc and DnsFirewall into `${stackName}-network`, resolve Blueprint egress definitions before either stack is constructed, and permit application-to-network references only. Keep VPC/subnet/security-group exports present across backend changes and preserve network properties, attribution and provenance tags. +3. Select one or more backends with `compute_types`; its first entry is the repository default. Preserve legacy `compute_type` behavior, including AgentCore alongside ECS or MicroVM. Removing a backend requires an explicit list that omits it. Publish the complete ordered list in `ComputeTypes` and `ComputeSubstrate`; the CLI and orchestrator enforce membership. Shared optional services remain independent. +4. Enforce CDK's `@aws-cdk/core:stackResourceLimit` at 490 before constructing stacks, including nested stacks and operator configurations outside the census. Operators may tighten but cannot raise it. Share the census profile product with the normal build, checking every template against resource, byte, parameter and output budgets and preserving method-scoped API permissions. Default byte budget: 800,000. +5. Keep current resource removal policies, Blueprint provisioning and guardrail versioning. Existing inline-to-split migration, broad retention and Blueprint controller handoff are deferred. Local template comparisons do not satisfy the populated refactor/import and rollback rehearsal required by #852. + +## Validation scope + +The 122-profile product covers all single-backend/service/image combinations in both topologies, supplemental alert/fork/consent options, explicit three-AZ pins, expanded model grants, every multi-backend set at default and widest settings, and legacy additive selectors. The expanded-model profiles add eight synthetic model IDs to the platform defaults to exercise IAM policy overflow; they do not assert live model availability. The SessionRole's audit exceptions follow its generated overflow policies and remain scoped to tenant object prefixes and literal model grants. Expected over-budget profiles must fail at the production 490-resource ceiling for the application stack; an unrelated error or unexpected success fails the gate. Real CDK metadata is included. The census records per-template measurements and source provenance with bundling and asset staging disabled; it does not measure a deployed stack. + +Network tests compare moved logical IDs and service properties, the complete export interface, application data resources and lifecycle policies, shared API routes and permissions, attribution and one-way dependencies. Their comparisons control the clock and account for the existing alpha guardrail and orchestrator version IDs. The independent-process `--check-stability` diagnostic keeps timestamps, IDs and asset hashes intact and reports existing churn; passing budget checks does not imply deterministic synthesis. + +`networkReservedAzs` preserves unused address slots when removing trailing AZs. Tests verify that the target application can use the old network, release the removed export, and keep every remaining subnet's properties. The [application-first procedure](/sample-autonomous-cloud-coding-agents/getting-started/deployment-guide#reducing-azs-in-an-existing-split-network) is separate from moving an inline network into another stack. Physical-ID preservation and rollback in AWS still require live verification. + +## Deferred migration work + +The following remain under #852 and need separate review and releases before a supported existing-stack migration: + +- **Retention lifecycle:** use appropriate per-resource policies, including `RetainExceptOnCreate` where retaining established data is needed without orphaning a failed first create. Verify fixed-name log groups and external registries can be recovered, imported or cleaned up before a same-name reinstall. Retaining an S3 bucket alone must not leave a destructive cleanup callback active. +- **Blueprint downgrade protection:** refuse an unsafe return from managed ownership to the legacy writer even when a context flag is omitted. Prove `max_turns`, `compute_type`, `onboarded_at` and other CLI overrides survive updates and rollback. A green deployment must not hide row replacement or tombstoning. +- **Ownership release and re-onboarding:** adoption must have a defined release path. Removing a repository must not block legitimate re-onboarding for the tombstone TTL. Test retries, owner changes and out-of-order callbacks. +- **Ledger lifecycle and bootstrap coverage:** establish a bounded cleanup/recovery plan for PITR-enabled ownership tables across mode changes, failures and destroy. Any future `Custom::BlueprintRepoConfig` must be represented in the bootstrap resource-action map and its coverage tests before it ships. +- **Guardrail and image normalization:** review stable version binding and Docker build-context changes separately from network ownership. They are not prerequisites for reporting truthful census differences. +- **Populated migration rehearsal:** verify refactor/import eligibility for each moved type and provider, preserve physical IDs and data, test networking and API behavior, and execute rollback. Retention must be installed on source resources before a transfer; an ordinary topology flag change is not a move. The criterion remains open, without an author-only waiver. + +The [teardown guidance](/sample-autonomous-cloud-coding-agents/getting-started/deployment-guide#teardown-blocked-by-agentcore-network-interfaces) covers the separately observed AgentCore ENI cleanup delay, which also occurs on `main`. + +## Consequences + +- New split installations gain application headroom without widening API Gateway permissions or changing repository ownership. +- Existing legacy compute contexts retain AgentCore. Ordered lists let repositories choose among deployed backends; older CLIs require AgentCore to remain present and first on additive stacks. +- Exports constrain later network changes. Application consumers must release an export before the network removes it. +- Resource and template budgets are checked locally; live deployment, migration and rollback remain separate evidence. + +## References + +- [Developer guide: synthesis budgets](/sample-autonomous-cloud-coding-agents/developer-guide/introduction#stack-decomposition-and-synthesis-budgets) +- [Deployment guide: network topology](/sample-autonomous-cloud-coding-agents/getting-started/deployment-guide#network-stack-topology) +- [Compute selection](/sample-autonomous-cloud-coding-agents/architecture/compute#selecting-and-changing-the-backend) +- [CDK best practices](https://docs.aws.amazon.com/cdk/v2/guide/best-practices.html) +- [CloudFormation quotas](https://docs.aws.amazon.com/AWSCloudFormation/latest/UserGuide/cloudformation-limits.html) diff --git a/docs/src/content/docs/developer-guide/Repository-preparation.md b/docs/src/content/docs/developer-guide/Repository-preparation.md index 0877d39b1..9190be278 100644 --- a/docs/src/content/docs/developer-guide/Repository-preparation.md +++ b/docs/src/content/docs/developer-guide/Repository-preparation.md @@ -31,21 +31,21 @@ The default is `awslabs/agent-plugins`. For a quick end-to-end test, fork that r ### Multiple repositories -To onboard additional repositories, add more `Blueprint` constructs in `cdk/src/stacks/agent.ts` and append them to the `blueprints` array (used to aggregate DNS egress allowlists): +To onboard additional repositories, add entries to `resolveBlueprintDefinitions` in `cdk/src/blueprints/definitions.ts`. The app resolves these plain inputs before constructing either stack, so repository provisioning and DNS egress policy use the same configuration: ```typescript -new Blueprint(this, 'MyServiceBlueprint', { +definitions.push({ + id: 'MyServiceBlueprint', repo: 'acme/my-service', - repoTable: repoTable.table, }); ``` -Each Blueprint supports per-repo overrides grouped into nested props (`BlueprintProps` in `cdk/src/constructs/blueprint.ts`): +Each entry supports the per-repo overrides from `BlueprintProps` in `cdk/src/constructs/blueprint.ts`, without a table reference. Keep its `id` stable across releases: ```typescript -new Blueprint(this, 'MyServiceBlueprint', { +definitions.push({ + id: 'MyServiceBlueprint', repo: 'acme/my-service', - repoTable: repoTable.table, compute: { runtimeArn: '...' }, // override the default runtime ARN agent: { modelId: 'global.anthropic.claude-opus-5', // foundation model override @@ -71,6 +71,24 @@ The command defaults to **`mise run build`** / **`mise run lint`**. A repo that Redeploy after changing Blueprints: `mise //cdk:deploy`. +### Stack decomposition and synthesis budgets + +`networkTopology` defaults to `inline`, preserving existing network ownership. For a new installation, `split` puts AgentVpc and DnsFirewall in `${stackName}-network`; the application consumes VPC, subnet and security-group exports. The network has no application dependencies. Plain Blueprint definitions are resolved once in `cdk/src/blueprints/definitions.ts` so DNS and repository provisioning use the same domains without cross-stack coupling. Task API routes, authorizers, permissions, CORS and deployment remain together in `AgentStack`. + +The CDK build and offline census share 122 named profiles: the 40-cell single-backend/service/image product in both topologies, supplemental email/fork/consent cases, explicit three-AZ pins, expanded `bedrockModels` grants, every multi-backend set at default and widest settings, and legacy additive selectors. Successful profiles check every parent and nested template against 490 resources, 800,000 bytes and 200 parameters/outputs. They also verify method-scoped API permissions, absence of console test-invoke grants and template-size warnings, backend membership, and two-zone auto-pin versus explicit pins. The model-expansion cases add eight synthetic IDs to the platform defaults and assert that the SessionRole's generated IAM overflow policies pass the audit; these fixtures test policy size, not model availability. Expected resource-budget failures must identify the application stack and the production ceiling; unrelated errors cannot satisfy the gate. + +```bash +MISE_EXPERIMENTAL=1 mise //cdk:census -- --output /tmp/stack-census +``` + +Use `--list` to see the profiles and `--profile NAME` to select them. The census runs the production app with fixed account/AZ inputs, CDK metadata enabled and bundling/staging disabled. It records template inventories, counts, bytes, dependencies and source provenance. The production app sets CDK's `@aws-cdk/core:stackResourceLimit` before any stack is constructed, so actual operator configurations outside the census also fail above 490. Context overrides may tighten but cannot raise the limit. `--max-resources` can only tighten the census audit ceiling. + +`--check-stability` runs each selected profile twice in independent processes and fails on differences. It does not normalize away timestamps, logical IDs or asset hashes. The existing Blueprint callbacks embed synthesis-time timestamps, and the alpha Bedrock guardrail uses token-derived version IDs; unchanged source can therefore fail this optional diagnostic. Deterministic Blueprint provisioning, guardrail versioning and Docker build-context changes are deferred. Passing the budget gate is not a claim of repeatable synthesis or live resource preservation. + +Network tests compare the moved definitions, generated Name tags, endpoint security-group descriptions, exports, application service properties, shared API resources, solution attribution and provenance tags. Comparison tests fix the clock and account for existing immutable guardrail/orchestrator version IDs; the census reports those real differences. `networkReservedAzs` reserves unused address slots so removing a trailing AZ need not shift remaining subnet CIDRs. Follow the [staged AZ reduction procedure](/sample-autonomous-cloud-coding-agents/getting-started/deployment-guide#reducing-azs-in-an-existing-split-network) to release old imports before changing the network. + +Existing inline-to-split migration remains deferred under [#852](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/852). A populated `cdk refactor`/import and rollback rehearsal is still required before a supported migration procedure can be published. Retention and Blueprint ownership handoff must be separately reviewed and released; this change keeps existing removal policies and the existing Blueprint provider. The concrete follow-up requirements are recorded in [ADR-023](/sample-autonomous-cloud-coding-agents/architecture/adr-023-cloudformation-stack-boundaries#deferred-migration-work). See [Network stack topology](/sample-autonomous-cloud-coding-agents/getting-started/deployment-guide#network-stack-topology) for fresh-install guidance and migration limits. + ### Customizing the agent image The default image (`agent/Dockerfile`) includes Python, Node 24 (LTS), `git`, `gh`, Claude Code CLI, and `mise`. If your repositories need additional runtimes (Java, Go, native libs), extend the Dockerfile. A normal `cdk deploy` rebuilds the image asset. diff --git a/docs/src/content/docs/getting-started/Deployment-guide.md b/docs/src/content/docs/getting-started/Deployment-guide.md index ee8be30c0..832b21d0a 100644 --- a/docs/src/content/docs/getting-started/Deployment-guide.md +++ b/docs/src/content/docs/getting-started/Deployment-guide.md @@ -8,7 +8,7 @@ This guide covers deploying ABCA into an AWS account, including compute backend ## Architecture overview -ABCA deploys as a **single CDK stack** (`backgroundagent-dev`) containing all platform resources. The stack uses a `ComputeStrategy` interface to support three compute backends within the same stack: +ABCA deploys from the `backgroundagent-dev` application stack with nested stacks for selected subsystems. Networking stays in that stack by default; `networkTopology=split` gives it a separate top-level stack. A deployment can provision one or more compute backends: | Aspect | AgentCore (default) | ECS Fargate (opt-in) | Lambda MicroVMs (experimental) | |--------|--------------------|--------------------|--------------------| @@ -21,16 +21,87 @@ ABCA deploys as a **single CDK stack** (`backgroundagent-dev`) containing all pl All backends are orchestrated by the same durable Lambda function. The `ComputeStrategy` interface abstracts `startSession()`, `pollSession()`, and `stopSession()` -- the ECS strategy calls `ecs:RunTask` / `ecs:DescribeTasks` / `ecs:StopTask` directly from the Lambda. No Step Functions are used. -ECS Fargate is currently **opt-in** -- the `EcsAgentCluster` construct is present in the stack code but commented out. To enable it, uncomment the ECS blocks in `cdk/src/stacks/agent.ts`. +AgentCore is the default. `compute_types` selects a comma-separated list (or a JSON array in `cdk/cdk.json`); its first entry is the repository default. For example, `-c compute_types=agentcore,ecs` deploys both, while `-c compute_types=ecs` deploys only ECS. Repository overrides can select any deployed backend. Memory, Gateway, Registry and the Linear vault remain independently configurable. + +Without `compute_types`, legacy `compute_type=ecs` or `compute_type=lambda-microvm` keeps AgentCore alongside that backend, including on upgrade. An unchanged legacy context therefore does not remove AgentCore. Use the updated CLI for lists whose default is not AgentCore or that omit AgentCore; it reads the ordered `ComputeTypes` output. See the [backend transition procedure](/sample-autonomous-cloud-coding-agents/architecture/compute#selecting-and-changing-the-backend) before deliberately removing a backend or changing the default. + +### Network stack topology + +`networkTopology=inline` is the default. Use `networkTopology=split` for a new environment to put the VPC, subnets, endpoints, flow logs and DNS firewall in `${stackName}-network`. The application stack retains its name, data stores, compute resources and shared Task API. It depends on network exports, so CDK deploys the network first. The complete VPC/subnet/security-group export set stays present across compute-backend changes. Keep the same topology context on subsequent synth, diff and deploy commands. + +Every local and pipeline synthesis enforces a **490-resource ceiling per parent or nested template**, including operator configurations outside the census. CDK fails synthesis with the stack name, resource count and ceiling when a template exceeds it. `@aws-cdk/core:stackResourceLimit` accepts a stricter integer from 1 to 490, as either a JSON number or CLI string; it cannot raise the production ceiling. + +The budget covers complete configurations, including multiple backends, Gateway, Registry, the Linear vault, managed MicroVM images, alert email, a fork Blueprint, explicit three-zone pins and expanded model allowlists. Wider inline combinations exceed 490 and fail synthesis; their split counterparts fit in the sampled matrix. The budget guard never changes topology automatically. The [offline census](/sample-autonomous-cloud-coding-agents/developer-guide/introduction#stack-decomposition-and-synthesis-budgets) reports counts for each supported profile. + +For a **new installation**: + +```bash +MISE_EXPERIMENTAL=1 mise //cdk:deploy -- --all \ + -c networkTopology=split -c compute_types=agentcore +``` + +Replace the compute list as needed and supply image settings for MicroVM. Persist the selected context in `cdk/cdk.json` for subsequent synth, diff and deploy commands. The split preserves the supported-AZ policy, HTTPS egress, endpoints and DNS observation mode. Configure additional Blueprint domains in `cdk/src/blueprints/definitions.ts`; both stacks consume those inputs before repository resources are constructed. + +**Existing inline-to-split migration is deferred.** Do not flip `networkTopology` on an existing inline deployment. An ordinary deploy creates a new VPC and removes old resources; identical logical IDs in different stacks do not preserve physical identity. Keep the existing topology until a separately reviewed migration has established resource-type eligibility, source retention and cleanup behavior, an ownership mapping, physical-ID/data preservation, and rollback on a populated disposable deployment. The `cdk refactor --unstable=refactor` criterion in #852 remains open; it is not waived or satisfied by synthesis tests. + +This change does not install broad retention, change the Blueprint provider, create an ownership ledger, or replace the guardrail versioning scheme. Those prerequisites belong in separate releases. CloudFormation exports still prevent removing a network used by the application; switching back to `inline` is not an automatic rollback. See [ADR-023's deferred work](/sample-autonomous-cloud-coding-agents/architecture/adr-023-cloudformation-stack-boundaries#deferred-migration-work). + +Stacks used to test earlier revisions with `blueprintProvisioning=prepare|adopt|managed` or the broad retention aspect require a separate recovery plan before adopting this narrowed version. Returning those stacks directly to the original Blueprint provider can overwrite or soft-delete repository rows. Synthesis rejects the removed `blueprintProvisioning` and `guardrailVersionMigration` context keys instead of silently ignoring them; removing those keys does not make an experimental stack safe to update. Preserve its deployed templates, repository data and ledger inventory; the removed experimental modes are not an upgrade path supported by this release. + +#### Reducing AZs in an existing split network + +A normal `--all` deployment updates the network first, so removing an AZ can fail because the old application still imports its private-subnet export. Reducing the count also shifts CDK's private-subnet CIDRs unless the vacated address slot stays reserved. Use the following staged procedure for a **three-to-two-zone reduction that keeps the first two existing AZs in their original order**. Replacing or reordering AZs requires a separate network migration. + +1. Pause automated deployments, task submissions, webhooks and scheduled work. Drain running and suspended sessions. Record the deployed templates, AZ order, subnet CIDRs, physical IDs and network exports. Keep the same account, region, stack identity, backend, image and Blueprint configuration throughout. +2. Persist the target AZ list in the existing `cdk/cdk.json` context, keeping `networkTopology=split`. Increase `networkReservedAzs` by the number of removed trailing AZs, so active plus reserved slots remains constant. For three active zones with no reservations, the target is two active zones and one reserved slot. These example names must match the deployment's first two AZs: + + ```json + "agentcore:availabilityZones": ["us-east-1a", "us-east-1b"], + "networkReservedAzs": 1 + ``` + + `networkReservedAzs` accepts an integer from 0 to 6, as a JSON number or CLI string; the default is 0. It reserves address space and creates no AWS resources. Keep this setting in every subsequent synth/deploy, including automation. +3. Set `APP_STACK` to the existing application stack name and review both target templates. The remaining subnets must keep their logical IDs, CIDRs and AZs; the target application must stop importing the removed subnet. Stop if the diff changes a retained subnet or any unrelated configuration. + + ```bash + APP_STACK=backgroundagent-dev + MISE_EXPERIMENTAL=1 mise //cdk:diff -- --all --method template + ``` + +4. Deploy **only the application**, leaving the existing three-zone network in place. `--exclusively` prevents CDK from deploying its network dependency: + + ```bash + MISE_EXPERIMENTAL=1 mise //cdk:deploy -- "$APP_STACK" --exclusively + ``` + +5. Copy each removed private-subnet export's exact name from the deployed network outputs. Verify that `list-imports` returns `[]` before changing the network. Any additional consumer stack must also release that export. + + ```bash + aws cloudformation describe-stacks --stack-name "${APP_STACK}-network" \ + --query 'Stacks[0].Outputs[].{ExportName:ExportName,Value:OutputValue}' --output table + REMOVED_SUBNET_EXPORT='' + aws cloudformation list-imports --export-name "$REMOVED_SUBNET_EXPORT" \ + --query Imports --output json + ``` + +6. Deploy both stacks with the same persisted target context. The network can now remove the unused export and AZ resources. Verify the remaining subnet physical IDs and CIDRs, DNS/egress behavior and a task before resuming producers and automation. + + ```bash + MISE_EXPERIMENTAL=1 mise //cdk:deploy -- --all + ``` + +To restore the previous three-zone layout, restore its AZ list and reservation count, then deploy the network before the application (`--all` uses this order). The network must recreate the third subnet and export before the application imports it again. Keep active plus reserved slots constant during this recovery too. + +Local synthesis tests verify the import ordering and unchanged remaining subnet properties for all three backends. This procedure still requires a disposable AWS rehearsal before a production network update. ### Lambda MicroVMs backend (experimental) > **Not for production.** `lambda-microvm` carries no smoke-parity guarantee for an unattended deployment. Keep production repositories on `agentcore` or `ecs`. Synth emits an unsuppressible warning to this effect whenever the backend is selected. Design detail: [COMPUTE.md](/sample-autonomous-cloud-coding-agents/architecture/compute) and [ADR-021](/sample-autonomous-cloud-coding-agents/architecture/adr-021-lambda-microvms-compute-backend). -Selecting it is a synth-time context flag: +Include it in the deployment's backend list, preserving any backends already in use: ```bash -mise //cdk:deploy -- --context compute_type=lambda-microvm +mise //cdk:deploy -- --context compute_types=agentcore,lambda-microvm ``` **You must re-bootstrap first.** This is the single most common way this backend fails, and the failure does not look like a configuration problem: @@ -73,7 +144,7 @@ Blueprints without `registry://` asset references continue to work. A remaining The string form is case-sensitive: use lowercase `true` or `false`. Any other value fails synthesis with an actionable validation error. -This context is an infrastructure switch, not a pause control. Applying it to an existing enabled deployment deletes the CloudFormation-managed registry and its records; re-enabling creates an empty registry that must be republished. See [REGISTRY.md](/sample-autonomous-cloud-coding-agents/architecture/registry) for the catalog migration and runtime behavior. +This context is an infrastructure switch, not a pause control. Applying it to an existing enabled deployment deletes the CloudFormation-managed registry and its records; re-enabling creates an empty registry that must be republished. The optional network split preserves that lifecycle; registry retention remains deferred. See [REGISTRY.md](/sample-autonomous-cloud-coding-agents/architecture/registry) for the catalog migration and runtime behavior. ## Bedrock inference geography @@ -139,7 +210,7 @@ At public US East (N. Virginia) first-tier list rates verified in August 2026, t |---------|---------|---------------| | Bedrock AgentCore Runtime (MicroVMs) | Agent sessions (default) | Yes | | ECS Fargate (when enabled) | Agent sessions (opt-in) | Yes | -| AWS Lambda MicroVMs (when enabled) | Agent sessions (experimental, `--context compute_type=lambda-microvm`) | Yes | +| AWS Lambda MicroVMs (when enabled) | Agent sessions (experimental, include `lambda-microvm` in `compute_types`) | Yes | | Lambda (Node.js 24, ARM64) | Orchestrator, API handlers, fanout consumer, reconcilers, custom resources | Yes | ### AI/ML @@ -313,6 +384,8 @@ AGENTCORE_AVAILABILITY_ZONES = ["us-east-1b","us-east-1c"] The override is validated at synth time, and both the JSON-array and `-c` string forms behave identically. Synth fails with a message naming the key when the value is not an array, has an empty/non-string entry, lists fewer than two **distinct** zones, contains zone *IDs* instead of names (`use1-az2` — a common column mix-up), or names zones outside the target region. When the account's mapping is knowable, the override is additionally cross-checked against the supported set, and unsupported or nonexistent zones fail synth. +Auto-pin selects two supported zones; an explicit override uses all the zones supplied. Pins above two zones remain subject to the 490-resource production ceiling. Configurations that exceed it need split networking; see [Network stack topology](#network-stack-topology) for the measured boundaries and migration requirements. + **Upgrading an existing stack.** Auto-pin is on by default, so a local `cdk deploy` against a stack created before this change may select different zones than the deployed subnets use. `Subnet.AvailabilityZone` is create-only, so that is a **replacement** of the subnets and the resources bound to them (route tables, NAT gateway/EIP, VPC endpoints). Run `mise //cdk:diff` first. If the diff shows subnet replacement and you would rather keep the current topology, pin the override to the zones already deployed: ```bash @@ -320,7 +393,28 @@ aws ec2 describe-subnets --filters "Name=vpc-id,Values=" \ --query 'Subnets[].[SubnetId,AvailabilityZone,AvailabilityZoneId]' --output text ``` -Be aware that destroying a VPC whose subnets held AgentCore ENIs can take 20–40 minutes while AWS reclaims them (see the `DELETE_FAILED` note in the [quick start](./QUICK_START.mdx) troubleshooting table). +### Teardown blocked by AgentCore network interfaces + +Deleting an AgentCore Runtime does not immediately release its service-managed network interfaces. The live review of [#912](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/912) observed `agentic_ai` interfaces attached by `amazon-aws` still in use after five hours. They can block RuntimeSG, subnet and VPC deletion and leave a stack in `DELETE_FAILED`. This also affects ordinary destroys from `main`; it is not a fixed 20–40 minute delay. + +Inspect the failed stack and its VPC before retrying: + +```bash +aws cloudformation list-stack-resources --stack-name \ + --query 'StackResourceSummaries[?ResourceStatus==`DELETE_FAILED`].[LogicalResourceId,PhysicalResourceId,ResourceType]' \ + --output table +aws ec2 describe-network-interfaces --filters Name=vpc-id,Values= +``` + +Let AgentCore release its interfaces; do not try to force-detach interfaces owned by `amazon-aws`. To finish deleting a stack already in `DELETE_FAILED`, inventory the blocked network resources and their dependencies, then retain those exact **logical IDs** in a deletion retry: + +```bash +aws cloudformation delete-stack --stack-name \ + --retain-resources +``` + +For split deployments, identify whether the failed resources belong to the application or the `-network` stack and target that stack. Keep physical IDs for every retained resource and track their cost. After the service releases the ENIs, clean up retained security groups, subnets and other VPC dependencies before deleting the VPC. `--retain-resources` does not perform that later cleanup or make the resources reusable by a same-name reinstall. Reconcile any retained resources from earlier experimental builds separately. + ### DNS Query Log Config replacement cascade (upgrading from pre-v0.5) diff --git a/docs/src/content/docs/using/Linear-setup-guide.md b/docs/src/content/docs/using/Linear-setup-guide.md index 39d2ea5bf..38339dd90 100644 --- a/docs/src/content/docs/using/Linear-setup-guide.md +++ b/docs/src/content/docs/using/Linear-setup-guide.md @@ -28,7 +28,7 @@ One of two places, chosen automatically at setup time: | **AgentCore Identity vault** | The stack was deployed with `--context enableLinearIdentityVault=true` | Nothing long-lived. AgentCore holds the refresh token and mints short-lived access tokens on demand. | | **Secrets Manager** | Otherwise — including regions where AgentCore Identity isn't available | An OAuth token bundle in `bgagent-linear-oauth-`, refreshed and rotated by ABCA. | -The vault is unavailable on the `lambda-microvm` substrate — see [Not available with `compute_type=lambda-microvm`](#not-available-with-compute_typelambda-microvm) below. +The vault can be configured with any selected backend. Lambda MicroVMs remains experimental; its image must include the current shared configuration contract. `bgagent linear setup` picks whichever the deployment supports and tells you which one it used. There is no flag. If the vault isn't available it prints one line and continues on Secrets Manager: @@ -40,9 +40,9 @@ A workspace that started on Secrets Manager and later moves to the vault **keeps When a workspace's authorization dies, ABCA records it on the registry row and publishes to the stack's operational alert topic. That topic has **no subscribers unless you deployed with `alertEmail`**, so set it if you want to hear about a dead workspace rather than discover it from `bgagent platform doctor`. -#### Not available with `compute_type=lambda-microvm` +#### Using the vault with Lambda MicroVMs -The vault and the Lambda MicroVMs substrate cannot be enabled on the same stack. Together they synthesize 505 CloudFormation resources against a hard limit of 500 (MicroVM alone is 496, the vault alone 488), so `cdk deploy` refuses the combination by name at synth rather than failing partway through. Use the vault on the `agentcore` or `ecs` substrate; a MicroVM stack stays on Secrets Manager until the stack reclaims room. +Use split networking when the full MicroVM-plus-vault configuration exceeds the 490-resource budget. `compute_types` can include MicroVM alongside AgentCore or ECS, or select MicroVM alone; an unchanged legacy `compute_type=lambda-microvm` keeps AgentCore. The guest execution role receives the mint grant, and the `/run` configuration carries `LINEAR_VAULT_ENABLED` and `LINEAR_WORKLOAD_IDENTITY_NAME`. Rebuild the MicroVM image from this checkout before enabling the vault. Verify consent and token minting in a live rehearsal; synthesis alone does not qualify this experimental backend. #### One workload identity per stack