From 199047755ed8b6907b0d64a1323cf4048d68448a Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Sun, 13 Sep 2026 14:46:49 -0400 Subject: [PATCH 001/149] docs(645): review MicroVM readiness and plan P3 --- agent/README.md | 8 +- agent/src/server.py | 55 +- cdk/src/constructs/lambda-microvm-compute.ts | 32 +- cdk/src/constructs/task-orchestrator.ts | 17 +- cdk/src/handlers/orchestrate-task.ts | 11 +- cdk/src/handlers/shared/compute-strategy.ts | 11 +- cdk/src/handlers/shared/orchestrator.ts | 14 +- .../handlers/shared/session-start-retry.ts | 6 +- .../strategies/lambda-microvm-strategy.ts | 13 +- cdk/src/stacks/agent.ts | 14 +- cdk/test/stacks/agent.test.ts | 20 +- ...ADR-021-lambda-microvms-compute-backend.md | 50 +- docs/design/COMPUTE.md | 2 +- docs/design/ORCHESTRATOR.md | 6 +- docs/guides/DEPLOYMENT_GUIDE.md | 12 +- docs/guides/LINEAR_SETUP_GUIDE.md | 2 +- docs/src/content/docs/architecture/Compute.md | 2 +- .../content/docs/architecture/Orchestrator.md | 6 +- ...Adr-021-lambda-microvms-compute-backend.md | 50 +- .../docs/getting-started/Deployment-guide.md | 12 +- .../content/docs/using/Linear-setup-guide.md | 2 +- .../645-nesting-probe-results.json | 793 ++++++++++++++++++ .../645-p3-implementation-plan.md | 234 ++++++ docs/verification/645-p3-readiness-review.md | 123 +++ 24 files changed, 1309 insertions(+), 186 deletions(-) create mode 100644 docs/verification/645-nesting-probe-results.json create mode 100644 docs/verification/645-p3-implementation-plan.md create mode 100644 docs/verification/645-p3-readiness-review.md diff --git a/agent/README.md b/agent/README.md index 2b1ed2393..58902300c 100644 --- a/agent/README.md +++ b/agent/README.md @@ -236,11 +236,11 @@ The warm-up is the *primary* fix; the probe that failed is also now non-fatal. ` **`POST /aws/lambda-microvms/runtime/v1/validate`** — Build hook (P2). A **shallow self-check only**: server alive, every hook route registered, interpreter floor, `platform_config` contract loaded. Returns 200 with the individual check results, or 503 while still initialising (which fails the build if it never clears — the right outcome for a broken snapshot). That 503 branch is a **refactor tripwire**, not a state you can reach today: `_module_initialized` is set as the module's last statement and uvicorn accepts no request until the import completes, so it only becomes reachable once someone moves warm-up work behind the bind — and a hook with only a 200 path would then report a still-initialising snapshot as valid. -It runs under the **build role**, which deliberately holds no Bedrock / Secrets Manager / DynamoDB grants, so it **makes zero AWS API calls and must keep making zero** — including its own logging (both build hooks log to stdout via `_build_hook_log`, never through the CloudWatch writer). Two reasons: a Logs write under the build role can only fail, and each failure pollutes the shared `_debug_cw_failures` alarm signal; and `boto3.client(...)` populates `boto3.DEFAULT_SESSION`, a module global holding a resolved credential chain plus the build-time region, which the snapshot would then freeze in for every MicroVM launched from that image version. "Deeper warm-up assertions" (Bedrock reachability, Memory access, tool availability) are therefore *not* implementable here. The one member of that list that turned out to be partly implementable — proving the local `claude` binary execs — lives on `/ready` instead, as a side effect of warming it (above): a local `exec` is not an AWS call, and it belongs to the hook whose 200 gates the snapshot. +It runs under the **build role**, which has no Bedrock, Secrets Manager or DynamoDB runtime grants. Both build hooks make **zero AWS API calls**, including logging: `_build_hook_log` writes to stdout. The build role can write within its MicroVM log namespace, but an application log group may be outside that grant. More importantly, resolving credentials before taking the snapshot can preserve build-role credential state. Boto3 caches resolved credentials; environment-derived region settings are re-read for each new client. The `_debug_cw_failures` counter is currently unexported ([#810](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/810)), so it is not an alarm signal. Local binary warm-up belongs in `/ready`; runtime AWS access must be tested on a real task. Baked secrets are **reported, not enforced**: `warnings` lists the names (never values) of any credential-shaped env var present in the snapshot, because the build environment's own credentials may legitimately be in that env and failing here would fail every build. -**`POST /aws/lambda-microvms/runtime/v1/terminate`** — Runtime hook (P2). Best-effort: emits one final structured log line and returns 200 — always, inside the hook budget, even with nothing running, and for **any body**: malformed JSON, a wrong content-type, an empty body or no body at all. That is why the handler takes the raw request instead of a typed body model — FastAPI validates a typed body *before* the handler runs, so a truncated body would answer 422 and report a hook failure for a teardown that actually succeeded. It does **not** join the pipeline thread (that is `lifespan`'s job on graceful shutdown) and it **never writes terminal task status**: the orchestrator finalizes the task and *then* calls `TerminateMicrovm`, so a status write here would race that finalization. Nothing is buffered to flush — `ProgressWriter` does a synchronous `put_item` per event, so progress is already durable at call time. +**`POST /aws/lambda-microvms/runtime/v1/terminate`** — Runtime hook (P2). Best-effort: emits one final structured log line and returns 200 — always, inside the hook budget, even with nothing running, and for **any body**: malformed JSON, a wrong content-type, an empty body or no body at all. That is why the handler takes the raw request instead of a typed body model — FastAPI validates a typed body *before* the handler runs, so a truncated body would answer 422 and report a hook failure for a teardown that actually succeeded. It does **not** join the pipeline thread (that is `lifespan`'s job on graceful shutdown) and it **never writes terminal task status**: the orchestrator finalizes the task and *then* calls `TerminateMicrovm`, so a status write here would race that finalization. `ProgressWriter` writes each event synchronously, but catches and drops failures; returning from an event method is not a durability guarantee. P3 needs an acknowledged barrier for `/suspend`, not merely a call to this best-effort writer. `microvmId` is parsed defensively and **arrives empty in practice**: the service sends `""` here, unlike `/run` where it is populated (live-verified, ADR-021 P2-F8). So an empty id is expected-normal, not a degraded read — and this hook therefore **cannot** join the guest's record to the control-plane one. `/run`'s `hook accepted task_id=… microvm_id=…` line carries that correlation; `/terminate`'s value is the pipeline-state snapshot it reports. @@ -270,7 +270,7 @@ On AgentCore and ECS the agent's non-secret platform env arrives as runtime env { "platform_config": { "task_table_name": "…", "github_token_secret_arn": "arn:…" } } ``` -Each snake_case key installs into its UPPER_SNAKE env var, and a payload value **wins** over any image/pre-existing value (the payload describes the live deployment; the snapshot describes a past one). Installation happens **before** any credential or pipeline initialisation — the very next step resolves the GitHub token from `GITHUB_TOKEN_SECRET_ARN`. Everything the hook logs before that point goes to stdout only (`[server/run-pre-config]`), for the same reason the build hooks do: the CloudWatch writer resolves AWS credentials and pins `boto3.DEFAULT_SESSION` (region included), and until the install has run the only environment available is whatever the snapshot baked. The single AWS call allowed before the install is the S3 payload fetch, because the config is inside the object being fetched. The allowlist lives in `contracts/constants.json` → `microvm_platform_config` (produced by the orchestrator, consumed here; shape enforced by `mise run check:constants-sync`): +Each snake_case key installs into its UPPER_SNAKE env var, and a payload value **wins** over any image/pre-existing value (the payload describes the live deployment; the snapshot describes a past one). Installation happens **before** any credential or pipeline initialisation — the very next step resolves the GitHub token from `GITHUB_TOKEN_SECRET_ARN`. Everything the hook logs before that point goes to stdout only (`[server/run-pre-config]`), for the same reason the build hooks do: the CloudWatch writer resolves AWS credentials and can cache credential state in `boto3.DEFAULT_SESSION` (environment-derived region is re-read for new clients), and until the install has run the only environment available is whatever the snapshot baked. The single AWS call allowed before the install is the S3 payload fetch, because the config is inside the object being fetched. The allowlist lives in `contracts/constants.json` → `microvm_platform_config` (produced by the orchestrator, consumed here; shape enforced by `mise run check:constants-sync`): | Key | Env var | Required | |---|---|---| @@ -288,7 +288,7 @@ Each snake_case key installs into its UPPER_SNAKE env var, and a payload value * | `aws_sdk_ua_app_id` | `AWS_SDK_UA_APP_ID` | | | `anthropic_default_haiku_model` | `ANTHROPIC_DEFAULT_HAIKU_MODEL` | | -Values are **non-secret identifiers only** — secrets are still fetched at `/run` time from Secrets Manager using the ARNs delivered here, so the snapshot stays secret-free. The allowlist **fails closed**: these values land in `os.environ` of the process that spawns the agent's tool subprocesses, so an unrecognised key is an env-injection attempt (`LD_PRELOAD`, `AWS_ENDPOINT_URL`, …) and the whole run is rejected with nothing installed. Blank/`null` values for optional keys are skipped rather than clobbering an image value; blank required keys are rejected. An envelope with no `platform_config` at all is accepted with a loud warning (the image and the orchestrator deploy on independent cadences). +Values are **non-secret identifiers only** — secrets are still fetched at `/run` time from Secrets Manager using the ARNs delivered here, so task secrets need not be baked into the snapshot. The build-hook warning is not proof that a hand-built image contains no secrets. The allowlist **fails closed**: these values land in `os.environ` of the process that spawns the agent's tool subprocesses, so an unrecognised key is an env-injection attempt (`LD_PRELOAD`, `AWS_ENDPOINT_URL`, …) and the whole run is rejected with nothing installed. Blank/`null` values for optional keys are skipped rather than clobbering an image value; blank required keys are rejected. An envelope with no `platform_config` is accepted with a warning **only if the effective environment already contains all required identifiers**. Otherwise `/run` rejects it as incomplete. Control characters and inconsistent ARN account/partition fields are also rejected; the ARN check does not establish that an identifier belongs to this deployment ([#817](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/817)). Rejections are structured so they are readable in the MicroVM log group: `400 MICROVM_RUN_PAYLOAD_INVALID` (unusable envelope — retrying the same body cannot help), `500 MICROVM_RUN_PAYLOAD_UNREADABLE` (the S3 fetch failed), `400 MICROVM_RUN_PLATFORM_CONFIG_INVALID` (key off the allowlist, non-object block, or non-string value — fix the producer), `400 MICROVM_RUN_PLATFORM_CONFIG_INCOMPLETE` (a required key missing or blank — fix the deployment wiring), `400 TASK_RECORD_INCOMPLETE` (same validator and vocabulary as `/invocations`). diff --git a/agent/src/server.py b/agent/src/server.py index 50942d8f8..153c77c02 100644 --- a/agent/src/server.py +++ b/agent/src/server.py @@ -36,11 +36,9 @@ from shared_constants import SHARED_CONSTANTS # --- _debug_cw / _warn_cw failure counter ------------------------------- -# Shared counter for BOTH the debug and warn CloudWatch writers. AgentCore -# doesn't forward container stdout to APPLICATION_LOGS, so a broken writer -# is invisible except for this metric. Single counter = single alarm -# surface — the trade-off is that the alarm can't distinguish which writer -# is broken (see Chunk 7c review notes). Defined BEFORE any function that +# Shared counter for BOTH the debug and warn CloudWatch writers. It is not +# exported or read yet (#810), so it cannot currently reveal a broken writer +# to an operator. Defined BEFORE any function that # references it (including ``_debug_cw`` / ``_warn_cw``) so the ordering is # import-time safe: a daemon thread spawned from a write-blocking function # can never race with module-level globals still being assigned. @@ -143,7 +141,7 @@ def _warn_cw(msg: str, *, task_id: str | None = None) -> None: and the ``capfd``-based unit tests still observe the line. CloudWatch delivery is fire-and-forget — failures bump the shared ``_debug_cw_failures`` counter via ``_warn_cw_write_blocking`` - so a silently broken writer still surfaces via that single metric. + which is currently unexported (#810); it is not an observable metric yet. """ # Redact cached credentials and emit via the same os.write path as # ``_debug_cw``: warn messages can embed payload fragments, so they @@ -173,8 +171,8 @@ def _warn_cw_write_blocking(log_group: str, task_id: str | None, stamped: str) - Mirrors ``_debug_cw_write_blocking`` but writes to the ``server_warn/`` stream so warn-level traffic is easy to alarm on independently of debug breadcrumbs. Failures bump the - shared ``_debug_cw_failures`` counter — a single alarm surface - covers both writers. + shared ``_debug_cw_failures`` counter. Nothing exports that counter + yet (#810), so it does not currently provide an alarm surface. """ try: from aws_session import platform_client @@ -264,8 +262,8 @@ def filter(self, record: logging.LogRecord) -> bool: _last_ping_status: str = "" # Heartbeat cadence for the TaskTable ``agent_heartbeat_at`` writer thread. -# Each live pipeline bumps the heartbeat every N seconds so operators can -# distinguish a stuck pipeline from a healthy long-running one. +# The independent worker reports process/writer liveness. A stuck pipeline can +# leave this thread running, so a fresh heartbeat does not prove work is progressing. _HEARTBEAT_INTERVAL_SECONDS = 45 @@ -1172,9 +1170,10 @@ def _reject_foreign_arns(resolved: dict[str, str]) -> None: Region is deliberately NOT compared. Secrets Manager and IAM ARNs legitimately differ on that axis in this system — IAM is global (empty region field), and a cross-Region secret is a supported deployment shape — so requiring agreement - would reject valid configurations while adding nothing: the execution role's - grants are account-scoped, so an in-account cross-Region ARN reaches nothing - the in-Region one does not. Partition + account is the boundary this checks. + would reject valid configurations. IAM grants separately constrain accessible + resources. Partition + account consistency does not bind these identifiers + to this deployment or prevent selecting another workspace's same-account + secret where IAM permits it (#817). """ anchor_value = resolved.get(MICROVM_PLATFORM_CONFIG_ACCOUNT_ANCHOR_KEY, "") # The anchor is contract-guaranteed REQUIRED (asserted at import), so by the @@ -1368,9 +1367,8 @@ def _build_hook_log(msg: str) -> None: 1. The build role's Logs grant is scoped to the service's own ``/aws/lambda-microvms/*`` namespace, so a write to any OTHER ``LOG_GROUP_NAME`` — e.g. an APPLICATION_LOGS group baked into a legacy or - hand-built image — can only FAIL. That failure bumps the shared - ``_debug_cw_failures`` counter, i.e. such an image build would poison the - "debug path is blind" signal with a false positive. + hand-built image — can fail. The counter incremented by such a failure + is currently unexported (#810), so it provides no alarm signal. 2. ``boto3.DEFAULT_SESSION`` created during ``/ready`` freezes the BUILD role's CREDENTIALS into the snapshot, where every launched MicroVM would inherit them (region is re-resolved per client; credentials are not — see @@ -1389,11 +1387,11 @@ def _pre_config_log(msg: str) -> None: ``_install_platform_config`` has run, ``LOG_GROUP_NAME`` is whatever the snapshot happens to carry — normally nothing, but a legacy or hand-built image could bake it, and then a ``_debug_cw`` on this path would resolve credentials - and pin ``boto3.DEFAULT_SESSION`` *before* ``AGENT_SESSION_ROLE_ARN`` is in the - environment — memoizing the UNSCOPED compute-role credentials for the life of - the process, where the whole point of that variable is that every later client - is tenant-scoped. (Region and ``AWS_SDK_UA_APP_ID`` are re-resolved per client - and so are NOT at risk here; the credentials are the exposure.) The one AWS + and create ``boto3.DEFAULT_SESSION`` before configuration is installed. + Tenant-data clients use the separate, tag-scoped session in ``aws_session``; + setting ``AGENT_SESSION_ROLE_ARN`` does not scope boto3's default session. + Platform clients intentionally retain compute-role credentials. Region and + ``AWS_SDK_UA_APP_ID`` are re-resolved per client. The one AWS call this phase is allowed to make is the S3 payload fetch, because ``platform_config`` is inside the object it fetches. @@ -1895,8 +1893,8 @@ def microvm_validate(): It must also not touch credential resolution: ``platform_config`` has not arrived yet (it comes with ``/run``), and any client built here would leave a - resolved boto3 session — with the build role's credentials and the build - region — frozen in the snapshot for every MicroVM launched from it. Hence + resolved boto3 session with cached build-role credentials in the snapshot + for every MicroVM launched from it (region is re-resolved per client). Hence ``_build_hook_log`` instead of ``_debug_cw``, and no import of ``aws_session``. @@ -2010,13 +2008,10 @@ async def microvm_terminate(request: Request): requires awaiting it. Safe on the event loop: the work is a JSON parse, a thread-count read and a fire-and-forget log — no blocking AWS call. - On flushing: there is nothing buffered to flush. ``_ProgressWriter`` performs - a synchronous DynamoDB ``put_item`` per event, and ``task_state`` writes - inline, so every progress/status write is already durable at call time — this - hook has no queue to drain, which is why it is a log-and-acknowledge rather - than a flush loop. (ADR-021 sub-decision 2's "flush progress events before - returning 200" applies to ``/suspend`` in P3 for the same reason: durability - is per-write, so the hook only has to observe it.) + There is no progress queue to drain. ``ProgressWriter`` writes synchronously + but catches and drops failures; a return from its event method is not proof + of durability. This hook only logs and acknowledges teardown. P3's + ``/suspend`` needs a separate acknowledged durability barrier. """ raw = b"" try: diff --git a/cdk/src/constructs/lambda-microvm-compute.ts b/cdk/src/constructs/lambda-microvm-compute.ts index e429b0728..a0d7bc716 100644 --- a/cdk/src/constructs/lambda-microvm-compute.ts +++ b/cdk/src/constructs/lambda-microvm-compute.ts @@ -47,12 +47,11 @@ import { LAMBDA_MICROVM_SUPPORTED_REGIONS, isLambdaMicrovmRegionSupported } from /** * Lifecycle expiry for MicroVM `/run` hook payloads, in days. * - * Mirrors {@link ECS_PAYLOAD_TTL_DAYS} in shape but NOT in role: on ECS the - * orchestrator deletes the payload at finalize and the rule is only a crash - * backstop, whereas the MicroVM strategy never deletes (ADR-021 sub-decision 3 - * / `microvmPayloadKey`) — so on this backend the lifecycle rule is the ONLY - * reaper. Payloads carry the hydrated prompt context, so the TTL stays as tight - * as the read-once-at-`/run` access pattern allows. + * Mirrors {@link ECS_PAYLOAD_TTL_DAYS}. Finalization attempts to delete the + * object identified by `microvmPayloadKey`, but the orchestrator is still + * missing its DeleteObject grant (#817). Lifecycle expiry is the current + * fallback; S3 processes expiry asynchronously, not exactly 24 hours after + * upload. Payloads carry hydrated prompt context and are read once at `/run`. */ export const MICROVM_PAYLOAD_TTL_DAYS = 1; @@ -693,7 +692,7 @@ export interface LambdaMicrovmComputeProps extends LambdaMicrovmImageInputs { * fails fast with the strategy's own "stack deployed without the MicroVM * substrate" error, which names the remedy. * - * ## ⚠️ A P2 substrate is fully wired, but still NOT smoke-verified + * ## ⚠️ P2 smoke succeeded with an IAM workaround; clean verification is pending * * Reaching state 1 or 2 provisions a complete substrate, a buildable image, and * a payload-deliverable `/run` path: P1 declares AND the agent serves `/ready` @@ -714,11 +713,12 @@ export interface LambdaMicrovmComputeProps extends LambdaMicrovmImageInputs { * what it can assert) and `/terminate` (in-guest teardown breadcrumb — * {@link TERMINATE_HOOK_TIMEOUT_SECONDS}). * - * What is still unverified is the thing no amount of wiring can assert: an - * end-to-end clone → change → PR run on this substrate, plus egress specifics and - * heartbeat/progress behaviour from a live MicroVM. That is what the - * `abca:microvm-image-p1-smoke-unverified` warning below says, and it is repeated - * in `cdk/scripts/package-microvm-artifact.sh`. Only `/suspend` and `/resume` + * The 2026-08-07 smoke completed clone → change → PR with live progress and + * heartbeats, but required a manual IAM workaround. The permanent PassRole + * fixes still need a clean rerun after re-bootstrap to policy bundle >=1.6.0, + * including runtime log verification. The stable + * `abca:microvm-image-p1-smoke-unverified` warning below records that remaining + * work, as does `cdk/scripts/package-microvm-artifact.sh`. Only `/suspend` and `/resume` * remain undeclared, until P3 implements them: a hook the service calls but * nothing answers fails the corresponding lifecycle transition. * @@ -1374,11 +1374,9 @@ export class LambdaMicrovmCompute extends Construct { if (this.imageIdentifier) { // Emitted on EVERY deploy that configures an image, in both image states. - // Not a throw and not suppressible: the substrate now looks like a working - // backend in every observable way — the image builds, launches, receives a - // payload, and the execution role holds the full runtime permission set — - // while nothing has exercised clone → change → PR on it. The warning's job is - // to keep "deploy succeeded" from reading as "backend works". + // A successful deploy does not establish a clean end-to-end run. The + // successful smoke used a manual IAM workaround; the warning identifies + // the source fixes and re-bootstrap/live verification still outstanding. // // The id is deliberately UNCHANGED across P1→P2 (operators grep for it, and a // rename would read as "the old warning is gone, so it must be fine"). diff --git a/cdk/src/constructs/task-orchestrator.ts b/cdk/src/constructs/task-orchestrator.ts index 2371824a8..b6c547587 100644 --- a/cdk/src/constructs/task-orchestrator.ts +++ b/cdk/src/constructs/task-orchestrator.ts @@ -338,9 +338,9 @@ export interface TaskOrchestratorProps { /** * Bucket for `/run` payloads that exceed the 4 KB `runHookPayload` cap — * i.e. nearly all of them, since a hydrated payload is bigger than that. - * The orchestrator gets **write only**: unlike the ECS payload bucket - * there is no finalize-time delete on this backend (the bucket's lifecycle - * rule is the reaper), so `grantDelete` would be an unused permission. + * The current grant is **write only**. Finalization also attempts deletion, + * so the missing DeleteObject permission is a defect tracked in #817; + * lifecycle expiry is the fallback until that grant is corrected. */ readonly payloadBucket: s3.IBucket; }; @@ -523,12 +523,9 @@ export class TaskOrchestrator extends Construct { props.ecsPayloadBucket.grantDelete(this.fn); } - // ADR-021: MicroVM payload bucket. WRITE only — the strategy uploads an - // oversized /run payload and never reads it back (the MicroVM execution - // role is the reader, with its own read-only grant), and it never deletes - // (the bucket's lifecycle rule reaps). No grantDelete, deliberately: the ECS - // path has one because the orchestrator deletes at finalize; this one does - // not, so the grant would be dead permission. + // ADR-021: upload oversized /run payloads; the execution role is the reader. + // Finalization calls deleteMicrovmPayload, but this grant is missing delete + // permission (#817). The lifecycle rule currently provides the fallback. if (props.microvmConfig) { props.microvmConfig.payloadBucket.grantPut(this.fn); } @@ -800,7 +797,7 @@ export class TaskOrchestrator extends Construct { }, { id: 'AwsSolutions-IAM5', - reason: 'DynamoDB index/* wildcards generated by CDK grantReadWriteData; AgentCore runtime/* required for sub-resource invocation; Secrets Manager wildcards generated by CDK grantRead; AgentCore Memory wildcards generated by CDK grantRead/grantWrite; ECS RunTask/DescribeTasks/StopTask conditioned on cluster ARN; iam:PassRole scoped to ECS task/execution roles and conditioned on ecs-tasks.amazonaws.com; S3 object/* wildcard from CDK grantPut on the dedicated MicroVM payload bucket; MicroVM lifecycle actions (RunMicrovm/GetMicrovm/TerminateMicrovm) are scoped to the single platform MicroVM image ARN plus a :* version-suffix sibling (every one of them authorizes against the image resource, not the per-session instance; no account-wide wildcard is used); lambda:PassNetworkConnector requires Resource:* because the action supports no resource-level permissions and the AWS-managed connectors live outside this account; iam:PassRole is scoped to the MicroVM execution role and conditioned on lambda.amazonaws.com; Agent Registry read scoped to the wired registry ARN, with a record/* suffix wildcard because record ids are server-assigned and unknown at synth (#246)', + reason: 'DynamoDB index/* wildcards generated by CDK grantReadWriteData; AgentCore runtime/* required for sub-resource invocation; Secrets Manager wildcards generated by CDK grantRead; AgentCore Memory wildcards generated by CDK grantRead/grantWrite; ECS RunTask/DescribeTasks/StopTask conditioned on cluster ARN; iam:PassRole scoped to ECS task/execution roles and conditioned on ecs-tasks.amazonaws.com; S3 object/* wildcard from CDK grantPut on the dedicated MicroVM payload bucket; MicroVM lifecycle actions (RunMicrovm/GetMicrovm/TerminateMicrovm) are scoped to the single platform MicroVM image ARN plus a :* version-suffix sibling (every one of them authorizes against the image resource, not the per-session instance; no account-wide wildcard is used); lambda:PassNetworkConnector requires Resource:* because the action supports no resource-level permissions and the AWS-managed connectors live outside this account; iam:PassRole is scoped to the exact MicroVM execution role without iam:PassedToService (ADR-021 P2r2-F10); Agent Registry read scoped to the wired registry ARN, with a record/* suffix wildcard because record ids are server-assigned and unknown at synth (#246)', }, ], true); } diff --git a/cdk/src/handlers/orchestrate-task.ts b/cdk/src/handlers/orchestrate-task.ts index 3f448c133..44311a083 100644 --- a/cdk/src/handlers/orchestrate-task.ts +++ b/cdk/src/handlers/orchestrate-task.ts @@ -461,8 +461,9 @@ const durableHandler: DurableExecutionHandler = asyn // bucket (the guest must read its object before any tenant identity exists), // keys are `/payload.json`, and the guest runs untrusted repo code — // so a TTL-only reaper left every finished task's hydrated prompt readable by - // any concurrently running MicroVM for up to ~24 h. See - // `deleteMicrovmPayload`. + // any concurrently running MicroVM until asynchronous lifecycle deletion. + // The current MicroVM delete grant is still missing (#817); task-scoped + // payload reads are a separate improvement (#700). if (blueprintConfig.compute_type === 'ecs') { await deleteEcsPayload(taskId); } else if (blueprintConfig.compute_type === 'lambda-microvm') { @@ -472,9 +473,9 @@ const durableHandler: DurableExecutionHandler = asyn // orchestrator shall call terminate-microvm (termination shall not rely on // any substrate timeout)." Without this the VM lingers until // `maximumDurationInSeconds` (8 h) expires — with `idlePolicy` omitted there - // is no tighter substrate bound — so we would keep paying for a full 8-hour - // reservation after every task, and every SUSPENDED/RUNNING VM keeps counting - // against the account memory quota that gates admission. + // is no tighter substrate bound. A leaked running VM can keep billing until + // that cap; suspended VMs retain snapshot charges. Whether suspended VMs + // consume the account memory quota is still unverified (ADR-021). // // `stopSession` is internally best-effort (it swallows and level-differentiates // every failure), so this cannot fail the finalize step or strand the task in diff --git a/cdk/src/handlers/shared/compute-strategy.ts b/cdk/src/handlers/shared/compute-strategy.ts index aec753f3a..4248f87db 100644 --- a/cdk/src/handlers/shared/compute-strategy.ts +++ b/cdk/src/handlers/shared/compute-strategy.ts @@ -78,13 +78,10 @@ export type SessionHandle = * Declared on all four variants for UNIFORMITY, though only ``completed`` and * ``failed`` are read today (``reconcileMicrovmSubstrateState`` returns early for * the other two). The wide union is deliberate rather than dead weight: - * ``suspended.reason`` has a named future consumer — P3's suspend/resume policy - * (ADR-021 sub-decision 2) has to distinguish an orchestrator-intended suspend - * during an approval wait from a substrate-side one, and ``stateReason`` is the - * only evidence the substrate offers for that. Narrowing the union now would mean - * widening it again there, and a per-variant union would invite call sites to - * branch on which variant carries a reason — the opposite of the opacity rule - * above. + * ``suspended.reason`` can explain an observation in P3 diagnostics. The policy + * must distinguish intended suspension through durable orchestrator intent and + * task/approval state, not by parsing this service-provided text. Keeping the + * field on every variant preserves the same diagnostic shape. */ export type SessionStatus = | { readonly status: 'running'; readonly reason?: string } diff --git a/cdk/src/handlers/shared/orchestrator.ts b/cdk/src/handlers/shared/orchestrator.ts index 81685e6d7..3a7a060ec 100644 --- a/cdk/src/handlers/shared/orchestrator.ts +++ b/cdk/src/handlers/shared/orchestrator.ts @@ -96,14 +96,12 @@ const AGENT_HEARTBEAT_STALE_SEC = 240; * `AgentCoreComputeStrategy.pollSession` is an explicit stub that always * reports `running`, so a crashed container is invisible without the heartbeat. * - `lambda-microvm` — yes, and it is the SECOND of two complementary signals - * (ADR-021 P2). The substrate `GetMicrovm` check catches a VM that DIED; it - * cannot catch a VM that is alive and healthy while the in-guest pipeline is - * hung, deadlocked, or OOM-killed inside the guest — nothing self-terminates on - * this substrate (live-verified: a MicroVM with a broken hook sat in `RUNNING` - * indefinitely with no `stateReason`). Without the heartbeat check such a task - * would burn the full ~8.5 h poll window, billing an 8-hour MicroVM - * reservation, before the safety net fired. So liveness here is substrate state - * AND agent heartbeat. + * (ADR-021 P2). `GetMicrovm` reports VM state; a stale heartbeat detects loss + * of the in-guest heartbeat writer even if the VM still reports RUNNING. + * The writer runs in a separate thread, so it can continue during a pipeline + * deadlock: neither signal proves that coding work is progressing. A failed + * run hook can cause service teardown; after an accepted hook, explicit + * termination and the maximum session duration remain cleanup safeguards. * - `ecs` — no, and this is a HARD CORRECTNESS CONSTRAINT, not a tuning * preference. The ECS boot command (`ecs-strategy.ts`) invokes * `run_task_from_payload` directly and "bypasses the uvicorn server entirely", diff --git a/cdk/src/handlers/shared/session-start-retry.ts b/cdk/src/handlers/shared/session-start-retry.ts index 2f7156e39..2e264d97b 100644 --- a/cdk/src/handlers/shared/session-start-retry.ts +++ b/cdk/src/handlers/shared/session-start-retry.ts @@ -23,9 +23,9 @@ * isolation (the handler's inline ``start-session`` step is never invoked by * the test suite). See #599 review B1/B2. * - * session-start is the ONE place a retry is idempotent by construction — no repo - * clone, no commits, no PR have happened yet, so re-invoking - * RunTask/InvokeAgentRuntime can't double-run work. A transient hiccup here (an + * This retries the start API, before the caller has a session handle. That does + * not guarantee idempotency: a lost success response can leave a live session + * behind, so each backend needs an idempotency/reconciliation policy. A transient hiccup here (an * ECS deploy-race "TaskDefinition is inactive", ENI/capacity delay, a * Bedrock/agentcore throttle) usually clears on a second attempt, so the first * transient failure is swallowed and retried once. A NON-transient failure (bad diff --git a/cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts b/cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts index 6e3e8a0ef..a31224d35 100644 --- a/cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts +++ b/cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts @@ -673,14 +673,11 @@ export class LambdaMicrovmComputeStrategy implements ComputeStrategy { // is present, so omission is the unambiguous disabled state. Suspension is // orchestrator-owned (P3) — do NOT reintroduce this field. // - // `clientToken` is also deliberately omitted. It is an idempotency token, - // and the one place a MicroVM start is retried is `startSessionWithRetry`, - // which retries precisely BECAUSE the first attempt FAILED. Passing a - // task-derived token there would ask the service to dedupe against that - // failed attempt and could replay its outcome instead of genuinely - // retrying — turning the auto-retry into a no-op. Session start is already - // idempotent by construction at the ABCA level (no clone, commit, or PR has - // happened yet), so the token buys nothing and risks the retry. + // No application-stable `clientToken` is supplied. The SDK may generate a + // token for one command, but `startSessionWithRetry` constructs another + // command on its next attempt. A failed response does not prove the first + // VM was never created. Lost-response reconciliation and attempt-scoped + // tokens need coverage before this can claim idempotent session starts. }); let result; diff --git a/cdk/src/stacks/agent.ts b/cdk/src/stacks/agent.ts index 4ce3142cb..20cb1cd04 100644 --- a/cdk/src/stacks/agent.ts +++ b/cdk/src/stacks/agent.ts @@ -369,15 +369,15 @@ export class AgentStack extends Stack { // second chance to disagree. const linearVaultWorkload = linearVaultWorkloadName(this); - // Fail here, naming both flags, rather than 500 resources later. The two features - // together synthesize 505 resources against CloudFormation's hard 500 limit (MicroVM - // alone 496, the vault alone 488), so the combination is not deployable today. Left to - // the resource counter, the operator gets a per-type census and no hint that two - // context flags are the cause. + // Keep the combination gated pending the bundled feature-matrix/deployment + // verification tracked in #857. The original 505-resource measurement predates + // #854 and later stack changes; the current offline measurements are recorded + // in docs/verification/645-p3-readiness-review.md. Do not treat that historical + // count as a current limit check. if (linearIdentityVaultEnabled && computeType === 'lambda-microvm') { throw new Error( - 'enableLinearIdentityVault cannot be combined with compute_type=lambda-microvm: the two ' - + 'together exceed CloudFormation\'s 500-resource limit for this stack (505). Deploy the ' + 'enableLinearIdentityVault cannot be combined with compute_type=lambda-microvm: this ' + + 'combination remains gated pending deployment verification (#857). Deploy the ' + 'vault on the agentcore or ecs substrate, or omit enableLinearIdentityVault. See ' + 'docs/design/ADR-016 and the LINEAR_SETUP_GUIDE.', ); diff --git a/cdk/test/stacks/agent.test.ts b/cdk/test/stacks/agent.test.ts index 731db7e56..14c001cdb 100644 --- a/cdk/test/stacks/agent.test.ts +++ b/cdk/test/stacks/agent.test.ts @@ -1737,21 +1737,11 @@ describe('AgentStack Linear identity vault gate (#809)', () => { }); test('MicroVM + vault is REFUSED by name, not left to the resource counter', () => { - // Pinning a limitation, not a behaviour. The vault IS wired for the MicroVM substrate - // — platform_config carries the workload name and the guest's execution role gets the - // mint grant — but the two cannot be enabled together today: 505 resources against a - // HARD limit of 500 (microvm alone 496, the vault alone 488). Claiming MicroVM support - // without saying so would be false. - // - // The stack refuses the combination itself rather than letting the counter throw, - // because the counter's message is a per-type census that never mentions either flag — - // the operator cannot tell from it what to change. - // - // Reclaiming room means nesting a subsystem. MicroVM (+19 resources) is the cheapest - // candidate and currently deployed nowhere, but nesting it needs the session-role trust - // wiring to stop referencing a child resource (it creates a parent↔child cycle today). - // - // When the room is found, this test should be replaced by a real parity assertion. + // Preserve the current #857 gate until bundled feature-matrix and deployment + // verification support removing it. The original 505-resource count is stale; + // current measurements and a cycle-free nested-stack prototype are documented + // in docs/verification/645-p3-readiness-review.md. Once the combination is + // supported, replace this refusal assertion with real parity/size assertions. const app = new App({ context: { enableLinearIdentityVault: true, compute_type: 'lambda-microvm' }, }); diff --git a/docs/decisions/ADR-021-lambda-microvms-compute-backend.md b/docs/decisions/ADR-021-lambda-microvms-compute-backend.md index 49bf51482..3bd397e5b 100644 --- a/docs/decisions/ADR-021-lambda-microvms-compute-backend.md +++ b/docs/decisions/ADR-021-lambda-microvms-compute-backend.md @@ -1,16 +1,16 @@ # ADR-021: AWS Lambda MicroVMs as a third ComputeStrategy backend -> Number: candidate ADR-021 (ADR-020 is the highest accepted on `main`; ADR-018 is claimed by open PR [#548](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/548), ADR-019 by open PR [#663](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/663)). Numbers are never reused. If a lower number frees before merge, renumber and coordinate with those PRs. +> **Implementation status (2026-09-13):** P1 ([#689](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/689)) and P2 ([#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733)) are merged. P2 completed real coding tasks with an IAM workaround; a clean rerun after re-bootstrap remains outstanding. P3 suspend/resume integration is not implemented. This ADR defines P1–P3, not a P4. See the [current review](../verification/645-p3-readiness-review.md) and [P3 implementation plan](../verification/645-p3-implementation-plan.md). **Status:** proposed **Date:** 2026-07-29 ## Context -ABCA selects a per-repo compute backend through the Blueprint's `compute_type` field (`cdk/src/handlers/shared/repo-config.ts`). Two backends exist today, resolved by `resolveComputeStrategy` (`cdk/src/handlers/shared/compute-strategy.ts`) behind a uniform `ComputeStrategy` interface (`startSession` / `pollSession` / `stopSession`): +ABCA selects a per-repo compute backend through the Blueprint's `compute_type` field (`cdk/src/handlers/shared/repo-config.ts`). Before this ADR, two backends existed, resolved by `resolveComputeStrategy` (`cdk/src/handlers/shared/compute-strategy.ts`) behind a uniform `ComputeStrategy` interface (`startSession` / `pollSession` / `stopSession`): - **AgentCore Runtime** (`agentcore`, default) — managed Firecracker MicroVM per session. Invoked via `InvokeAgentRuntime`; liveness is inferred from agent heartbeats in DynamoDB plus the FastAPI `/ping` endpoint (`agent/src/server.py`) — the strategy's `pollSession` is a stub that always reports `running`. Constraints: 2 GB image limit, no substrate-level suspend API exposed to the orchestrator. -- **ECS on Fargate** (`ecs`) — always-on Fargate task (16 vCPU / 120 GB, ARM64) for repos that exceed AgentCore's limits ([#596](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/596)). Invoked via `RunTask` in batch mode (bypassing the HTTP server); liveness via `DescribeTasks`. No suspend — an idle task blocked on an approval wait burns full compute the whole time. +- **ECS on Fargate** (`ecs`) — Fargate task (ARM64; current build default 4 vCPU / 16 GiB, configurable up to 16 vCPU / 120 GiB) for repos that exceed AgentCore's limits ([#596](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/596)). Invoked via `RunTask` in batch mode (bypassing the HTTP server); liveness via `DescribeTasks`. No suspend — an idle task blocked on an approval wait burns full compute the whole time. **AWS Lambda MicroVMs** (launched 2026-06-22) is a serverless Firecracker sandbox primitive that AWS positions explicitly for AI coding agents: VM-level isolation, snapshot-based near-instant launch, **suspend/resume with full memory + disk state preserved** (compute charges stop while suspended), a dedicated JWE-authenticated HTTPS endpoint per instance, lifecycle hooks (`/run`, `/suspend`, `/resume`, `/terminate`), and up to 8 hours per session. It is **not** classic Lambda: the 15-minute function cap does not apply, and [COMPUTE.md](../design/COMPUTE.md)'s "Lambda: poor fit" verdict refers to functions, not MicroVMs. @@ -23,7 +23,7 @@ ABCA selects a per-repo compute backend through the Blueprint's `compute_type` f | Isolation | MicroVM (managed) | Task-level (Firecracker) | MicroVM (Firecracker) | | Max duration | 8 h | No cap | 8 h (running + suspended) — **verified** (`L-B430C318 = 8 hours`) | | Suspend/resume | No orchestrator-visible API | No | **Yes** — explicit API + idle policy, state preserved, no compute charge while suspended. **Verified**: suspend reaches `SUSPENDED` in ~1 s with no `idlePolicy`; resume restores `RUNNING` in ~1 s with `microvmId` **and** `endpoint` byte-identical | -| Resources | AgentCore-managed | 16 vCPU / 120 GB / 20–200 GB disk | **Baseline 8 GiB RAM / 4 vCPU, auto-scaling to a 32 GiB / 16 vCPU peak**; 32 GB disk. `minimumMemoryInMiB` configures the BASELINE (max 8,192 MiB); the service scales vertically on demand — capacity is baseline-priced with 4× burst headroom | +| Resources | AgentCore-managed | Build default 4 vCPU / 16 GiB; up to 16 vCPU / 120 GiB; 20–200 GB disk | **Baseline 8 GiB RAM / 4 vCPU, auto-scaling to a 32 GiB / 16 vCPU peak**; 32 GB disk. `minimumMemoryInMiB` configures the BASELINE (max 8,192 MiB); the service scales vertically on demand — baseline capacity and additional burst usage are billed separately | | Packaging | ECR image ≤ 2 GB | ECR image, no hard cap | **Zip + Dockerfile in S3 → service-built snapshot image** (versioned, storage billed) | | Invocation | `InvokeAgentRuntime` (SigV4) | `RunTask` + container overrides | `RunMicrovm` (image **ARN** required — a bare name is rejected) → dedicated HTTPS endpoint + JWE token (`CreateMicrovmAuthToken`, ≤ 60 min TTL) | | Liveness | Agent heartbeat + `/ping` | `DescribeTasks` | MicroVM state (RUNNING / SUSPENDED / TERMINATED) via control-plane API **and** agent heartbeat (see sub-decision 1) | @@ -49,7 +49,7 @@ ABCA selects a per-repo compute backend through the Blueprint's `compute_type` f 1. **Idle detection is inbound-traffic-based; the ABCA agent is outbound-only.** MicroVM idle policies suspend when no traffic arrives at the *endpoint*. A busy agent running a 40-minute build receives no inbound traffic and would be suspended mid-work by a naive idle policy. Conversely, "no inbound traffic" is the agent's *normal* state. 2. **No self-suspend.** The agent cannot suspend its own MicroVM from inside; only an external `SuspendMicrovm` call can. Suspend decisions must be owned by the orchestrator — which aligns with the unified liveness model proposed in [#491](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/491). -3. **Snapshots bake state.** The image snapshot is captured once at build time; every MicroVM resumes from it. Secrets, tokens, and per-task identity must arrive at run time (`runHookPayload`, ≤ 4 KB, or fetched in the `/run` hook), never at image build. CSPRNG reseeding is a consequence of the same property and is scoped to **P3** — see the amended note under sub-decision 2 and the risk bullet, which record why the exposure is negligible today. +3. **Snapshots bake state.** The image snapshot is captured once at build time; every MicroVM resumes from it. Secrets, tokens, and per-task identity must arrive at run time (`runHookPayload`, ≤ 4 KB, or fetched in the `/run` hook), never at image build. Application PRNG reseeding is a consequence of the same property and is scoped to **P3** — see the amended note under sub-decision 2 and the risk bullet, which record why the exposure is negligible today. 4. **Auth tokens are short-lived.** JWE tokens max out at 60 minutes; any orchestrator→agent HTTP interaction over the endpoint needs token refresh, unlike AgentCore's SigV4 invoke or ECS's no-endpoint model. 5. **Identity delta — narrower than it looks.** Most AgentCore services ABCA uses are standalone and substrate-portable: Memory is already consumed from ECS via an IAM grant plus `MEMORY_ID` (`EcsAgentCluster`), and Gateway (ADR-019) is portable by design (SigV4 inbound). The genuinely Runtime-coupled piece is the workload-access-token **delivery mechanism** (`runtimeUserId` → `WorkloadAccessToken` request header → `BedrockAgentCoreContext`, used by `resolve_linear_api_token()`), which has no MicroVM equivalent. The ECS backend already lives with this delta (env-var token delivery); MicroVMs inherit the same posture until the pluggable identity work ([#249](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/249), ADR-016) redesigns the seam. 6. **The service's defaults are not our posture.** Two of them, both discovered live: `RunMicrovm` attaches a **public** `HTTP_INGRESS` connector (and mints a public `*.lambda-microvm..on.aws` endpoint) when `ingressNetworkConnectors` is omitted, and `CreateMicrovmImage` **requires** the `/ready` hook whenever any lifecycle hook is enabled. Neither posture can be reached by leaving a field out — each needs an explicit control (see sub-decisions 3 and 4). @@ -68,9 +68,9 @@ The `ComputeStrategy` interface gains **mandatory** `suspendSession(handle)` / ` **Poll semantics — the strategy reports, the orchestrator interprets.** `pollSession(handle)` receives only the session handle and cannot see task state, so the health rules must live where the DynamoDB status lives. `SessionStatus` gains a `'suspended'` variant; the strategy maps `GetMicrovm` state mechanically and the **orchestrator** cross-references against the task row — the same division of labor `finalPollState` already uses for ECS (substrate stopped + non-terminal DynamoDB status → failed) and `pollTaskStatus` uses for agentcore heartbeats: substrate `suspended` + task `AWAITING_APPROVAL` is healthy (orchestrator-intended suspend); `suspended` with any other task status is an anomaly to surface, not fail-fast; substrate terminal + non-terminal task status → classify failed. -**Liveness on this backend is substrate state AND agent heartbeat.** The substrate cross-check above answers one question — "is the VM still there?" — and P2 established that it is not sufficient on its own. `GetMicrovm` catches a MicroVM that *died*; it cannot catch a MicroVM that is alive and reporting `RUNNING` while the pipeline **inside the guest** is hung, deadlocked, or was OOM-killed. That state is not hypothetical or self-correcting on this substrate, and the P2 live run narrowed *why* without weakening the conclusion. The service does reap a VM whose run hook FAILS (a 4xx makes it terminate within ~12 s — see sub-decision 2), so the P1 evidence for this paragraph (a hook-less image sitting in `RUNNING` indefinitely with no `stateReason`) no longer describes an ABCA image. What the service reaps is a hook *result*; it has no view into the guest afterwards. So the surviving — and more realistic — hang case is a `/run` hook that returned **200** and a pipeline that then hung, deadlocked or was OOM-killed behind it: the substrate stays `RUNNING`, the service is satisfied, and nothing else notices. Left to the substrate check alone, such a task would burn the orchestrator's full ~8.5 h poll window — billing an 8-hour reservation — before the safety net fired. +**Liveness on this backend combines substrate state and the agent heartbeat.** `GetMicrovm` reports whether the VM is running or terminal. The independent `_heartbeat_worker` in `agent/src/server.py` refreshes the task timestamp every 45 seconds while the task is `RUNNING`; a stale timestamp detects loss of that writer, including a process crash or inability to write DynamoDB. It does **not** prove pipeline progress: a hung coding thread can coexist with a healthy heartbeat thread. A failed `/run` hook can trigger service teardown; after an accepted hook, ABCA still needs explicit termination and the maximum-duration backstop. A general progress watchdog remains separate work. -The in-guest half is already being written: the agent updates `agent_heartbeat_at` on the task row unconditionally, with no backend awareness, so the timestamp exists on every substrate. Only the orchestrator's *reaction* to it was backend-scoped — `pollTaskStatus` evaluated staleness for `agentcore` alone — which is the gap P2 closes by extending it to `lambda-microvm`. The grace and stale thresholds are the SAME on both: the timestamp is written by the same pipeline code at the same cadence, so a backend-specific window would encode a difference that does not exist. The two signals stay complementary rather than redundant — the substrate check is the crash detector, the heartbeat is the hang detector — and the check remains scoped to task status `RUNNING`, which is what keeps a deliberately suspended VM during an approval wait (sub-decision 2, P3) from being read as a dead one. `ecs` is deliberately left out: `DescribeTasks` reports a real container exit *with an exit code* (OOM-kill included) and the ECS poll block already interprets it with its own patience counters, so adding the heartbeat there would give one backend two independently-tuned kill paths for the same failure. +AgentCore and MicroVM tasks start the same periodic heartbeat worker, so they use the same 120-second grace and 240-second stale thresholds. ECS starts the pipeline directly, bypassing `server.py`; its pipeline writes a timestamp once but does not start this worker. Enabling the same stale check for ECS would therefore fail healthy tasks after roughly four minutes. The check applies only to `RUNNING`, not `AWAITING_APPROVAL`. The transition back to `RUNNING` currently leaves the old timestamp intact: after a long gate, a poll can see a stale heartbeat before the next worker tick. Refreshing it atomically on approval resume is a P3 prerequisite in the implementation plan. The service's `MicrovmState` enum has **six** members, not three, so the mapping is stated exhaustively (one line of rationale each, mirrored in the strategy's doc comment): @@ -78,7 +78,7 @@ The service's `MicrovmState` enum has **six** members, not three, so the mapping |---|---|---| | `PENDING` | `running` | Still booting; the same way ECS's `PENDING`/`PROVISIONING` map to `running`. | | `RUNNING` | `running` | — | -| `SUSPENDING` | `suspended` | Already on its way to frozen; reporting `running` would tell the orchestrator compute is still progressing when it is not. Both suspend states land on a report the orchestrator treats as benign-or-anomalous depending on task status, never as failure. **Never observable in practice** — suspend reached `SUSPENDED` in under 1 s live — so it is mapped for completeness and nothing may *wait* for it. | +| `SUSPENDING` | `suspended` | Already on its way to frozen; reporting `running` would tell the orchestrator compute is still progressing when it is not. Both suspend states land on a report the orchestrator treats as benign-or-anomalous depending on task status, never as failure. **Not observed in the recorded probe** — suspend reached `SUSPENDED` in under 1 s. Other runs may expose this state, so handle it without requiring it to appear. | | `SUSPENDED` | `suspended` | — | | `TERMINATING` | `completed` | Terminal-bound and carries no exit code, so "the substrate is gone" is all the strategy can honestly say. | | `TERMINATED` | `completed` | Success vs failure is the orchestrator's call — it cross-references the DynamoDB status. **This is the load-bearing terminal signal**, not `NotFound` (see below). | @@ -107,8 +107,8 @@ Normative requirements (EARS, per [ADR-020](./ADR-020-ears-requirements-syntax.m - If `GetMicrovm` reports that the MicroVM does not exist, then the strategy shall report `completed`. - If the strategy reports a terminal substrate state while the task's DynamoDB status is non-terminal, then the orchestrator shall re-read the task row and, if it is still non-terminal, classify the task as failed with a substrate-failure remedy. - If the strategy reports `suspended` while the task's DynamoDB status is not `AWAITING_APPROVAL`, then the orchestrator shall surface an anomaly event and shall not fail-fast the task. -- While a `lambda-microvm` task's DynamoDB status is `RUNNING`, if the task's `agent_heartbeat_at` is stale (or absent past the grace window) by the same thresholds the orchestrator applies to `agentcore`, then the orchestrator shall treat the session as unhealthy and stop polling — the substrate `GetMicrovm` check shall remain the crash detector, and the heartbeat shall be the in-guest hang detector. -- The task-detail API response shall include `agent_heartbeat_at`, and the CLI shall surface it while the task is non-terminal (P2r2-F11: the field drove the orchestrator's hang detector but was never projected, so no operator could observe the signal — and its invisibility produced a wrong verification conclusion). +- While a `lambda-microvm` task's DynamoDB status is `RUNNING`, if the task's `agent_heartbeat_at` is stale (or absent past the grace window) by the same thresholds the orchestrator applies to `agentcore`, then the orchestrator shall treat the session as unhealthy and stop polling — the substrate `GetMicrovm` check shall remain the crash detector, and the heartbeat shall detect loss of the in-guest heartbeat writer, without claiming to detect every pipeline hang. +- The task-detail API response shall include `agent_heartbeat_at`, and the CLI shall surface it while the task is non-terminal (P2r2-F11: the field drove the orchestrator's heartbeat check but was never projected, so no operator could observe the signal — and its invisibility produced a wrong verification conclusion). - The task-**summary** API response (`GET /v1/tasks`) shall also include `agent_heartbeat_at`, and `bgagent list` shall render it as an age column. Extending the field to the list response is a deliberate widening of the requirement above rather than an incidental one: the detail-only projection makes liveness a per-task question, and an operator checking a fleet of tasks one `bgagent status` at a time is exactly how the hung task P2r2-F11 describes went unnoticed. Same suppression rule as the detail view (terminal tasks and never-beaten tasks render a placeholder) so the two views cannot disagree. - If `suspendSession` or `resumeSession` is invoked on a strategy that does not support suspension, then the strategy shall return an explicit unsupported result. - When the agent process reaches a terminal state, the agent shall exit. @@ -123,10 +123,10 @@ The headline economic win is suspend during **HITL approval waits** (Cedar appro The handshake must respect the existing approval mechanics: the agent **discovers decisions itself** by polling DynamoDB (`_poll_for_decision`, monotonic timeout), the approve/deny Lambda writes only the decision rows, and `AWAITING_APPROVAL` holds the concurrency slot (Cedar decision #7). Nothing "delivers" an approval to the agent, and suspension freezes the agent's monotonic clock — so the design is: - **Suspend — orchestrator-owned.** The orchestrator's durable poll observes `AWAITING_APPROVAL` on a `lambda-microvm` task and calls `suspendSession` after a grace period, and only when the gate's remaining window exceeds grace + resume overhead (suspending a 30 s gate is pure loss). Suspend is a policy decision on a poll observation, not a user action. -- **Resume — inline in the approve/deny Lambdas, orchestrator poll as backstop.** After the transactional decision write commits, `ApproveTaskFn`/`DenyTaskFn` load the MicroVM handle from the task row's `compute_metadata` (persisted at session start — the same field `cancel-task.ts` reads ECS handles from) via a post-commit strongly-consistent `GetItem`, then call `resumeSession` **best-effort**: on failure they log a warning and write a resume-orphan task event; the decision response never fails on a compute error (the decision row is already durable). The orchestrator poll reconciles: decision row present + MicroVM still `SUSPENDED` → retry resume (idempotent). +- **Resume — inline in the approve/deny Lambdas, orchestrator poll as backstop.** After the transactional decision write commits, `ApproveTaskFn`/`DenyTaskFn` load the MicroVM handle from the task row's `compute_metadata` (persisted at session start — the same field `cancel-task.ts` reads ECS handles from) via a post-commit strongly-consistent `GetItem`, then call `resumeSession` **best-effort**: on failure they log a warning and write a resume-orphan task event; the decision response never fails on a compute error (the decision row is already durable). The orchestrator poll reconciles: current approval row APPROVED/DENIED + MicroVM still `SUSPENDED` → retry resume; a PENDING row alone must not trigger wake-up. *Why inline rather than poll-only — codebase precedent:* resume-on-approve is structurally identical to task cancellation — a user-initiated, latency-sensitive action whose purpose is an immediate compute-lifecycle side effect. `cancel-task.ts` already resolves this exact tension: the API-plane handler invokes ECS `StopTask` / AgentCore `StopRuntimeSession` inline, best-effort (a failed stop logs a warning and the state transition stands; a `task_cancel_compute_orphan` event is written when no stoppable compute handle exists, `reason: missing_runtime_handle`) — with the conditional IAM wired in `task-api.ts`. The resume path goes one step further than the precedent by also writing the orphan event on *failed* resume calls, because a failed resume strands a suspended VM awaiting a decision — a stronger liveness consequence than a failed stop of an already-cancelled task. The alternative (orchestrator-poll-only resume) preserves single-owner lifecycle purity but pays up to a full poll interval (~30 s) of latency on every approval, and the purity argument was already litigated and declined for cancel. `approve-task.ts` is deliberately minimal today (security-critical ownership comparison, Cedar finding #6); the resume call is therefore added *after* the transaction commits, cannot alter the decision outcome, and carries one conditional `lambda:ResumeMicrovm` grant — the same blast-radius trade the cancel handler accepted in review. -- **Timeout under freeze — the agent re-bases on the wall clock it already owns.** The agent's monotonic gate timer freezes while suspended, so resuming near the deadline is not enough: the frozen timer would still hold its remaining budget and fire the deny minutes *after* the user-visible window — colliding with the approval row's TTL (`created_at + timeout_s + 120s`) and triggering the "row reaped → stranded" fallback on a healthy gate. Instead, the gate expires at **`min(monotonic budget, created_at + timeout_s)`**, evaluated on each poll iteration and on `/resume`. This is not a new principle: Cedar decision #6 is already "min wins" for timeouts, the wall-clock deadline is already durable in the approval row the agent itself writes (`created_at` is in the agent's own clock domain — no skew), and §13.12's late-approval race fix already establishes that the durable row is authoritative over the agent's local timer. Deny authority stays agent-side (the conditional `TIMED_OUT` write + ConsistentRead re-read race protection is untouched); the orchestrator's resume at `deadline − margin` is purely the wake-up mechanism, with no correctness role. +- **Timeout under freeze — the agent re-bases on the wall clock it already owns.** The agent's monotonic gate timer freezes while suspended, so resuming near the deadline is not enough: the frozen timer would still hold its remaining budget and fire the deny minutes *after* the user-visible window — colliding with the approval row's TTL (`created_at + timeout_s + 120s`) and triggering the "row reaped → stranded" fallback on a healthy gate. Instead, the gate expires at **`min(monotonic budget, created_at + timeout_s)`**, evaluated on each poll iteration and on `/resume`. This is not a new principle: Cedar decision #6 is already "min wins" for timeouts, the wall-clock deadline is already durable in the approval row the agent itself writes (`created_at` is recorded by the agent; clock changes across suspend still need testing), and §13.12's late-approval race fix already establishes that the durable row is authoritative over the agent's local timer. Deny authority stays agent-side (the conditional `TIMED_OUT` write + ConsistentRead re-read race protection is untouched); the orchestrator's resume at `deadline − margin` is purely the wake-up mechanism, with no correctness role. - **Backstops, not mechanisms.** `maximumDurationInSeconds` (mandatory on every `RunMicrovm`, pinned at 28 800 s — see sub-decision 1) is the substrate kill switch bounding running **and** suspended time; the orchestrator's finalization `terminate-microvm` is the active cleanup path; the stranded-approval reconciler retains its role for orphaned waits. No `idlePolicy`-based bound is used in any phase — see sub-decision 1's omit-`idlePolicy` invariant. **The active terminate is still mandatory on the SUCCESS path, and P2 sharpened why.** P1 concluded flatly that "nothing self-terminates": a hook-less MicroVM reached `RUNNING` in 12 s and stayed there with no `stateReason` through every checkpoint. P2 refuted that *for the failure path only* — with `run: ENABLED`, a run hook that answers 4xx makes the **service** terminate the VM within ~12 s, `stateReason: "Run lifecycle hook returned HTTP status 400. Please check your hook endpoint and application logs for more details."`, after which `suspend-microvm` correctly refuses it. That is a real improvement in cost posture and a direct benefit of declaring hooks (see also the failure-path row in the phasing table, sub-decision 3). @@ -134,22 +134,22 @@ The handshake must respect the existing approval mechanics: the agent **discover It does **not** relieve the orchestrator of anything, because the two cases are disjoint. The service reaps a hook *result* it did not like; it has no view of the guest once the hook returned 200. So a task that starts normally — the overwhelming majority — has no service-side reaper at all, and a VM whose pipeline finished, crashed after `/run`, or hung is reaped by nobody but `TerminateMicrovm`. A leaked handle therefore remains a cost incident that bills until the 8 h cap; only the "the guest rejected its own payload" corner now cleans itself up. - **Concurrency slot stays held** during suspend. Cedar decision #7's rationale ("container alive, consuming memory") weakens under suspend, and the harder replacement rationale — "AWS counts `SUSPENDED` MicroVMs toward the account memory quota, so releasing ABCA's slot would not free real capacity" — is **undischarged**: the suspended VM stayed in `list-microvms` at every checkpoint, but that only proves *listed*. `L-CD1C0CC4` (1024 GB, account-scoped) exposes no `UsageMetric`, `AWS/Usage` carries only `CallCount` per API, and no MicroVM memory metric exists in any namespace, so consumption is **not observable safely** — proving it would need a large concurrent fleet. The conclusion (hold the slot) stands as the conservative choice, not as a verified fact. Size the arithmetic against the 32 GiB **peak** rather than the 8 GiB baseline: a busy fleet scales up, so peak is what actually competes for the account quota. -The agent's `/suspend` hook flushes progress events (durable writes before returning 200, within the 60 s hook budget); `/resume` reseeds CSPRNGs and refreshes cached credentials. +In P3, the agent's `/suspend` hook must acknowledge durable progress before returning 200 within its hook budget; `/resume` must refresh credentials and reseed application PRNG state from fresh OS entropy. These hooks are not implemented yet. A non-cryptographic PRNG such as Python `random` remains unsuitable for secrets even after reseeding; security-sensitive randomness must use an OS-backed cryptographic source. -**Amended (P2 review): CSPRNG reseeding is P3 scope, and the P2 exposure is negligible.** The original EARS requirement below implied a `/run`-time reseed had to land with P2; it did not, and shipping P2 with the requirement unmet-but-asserted was itself the defect. Measured exposure, which is what changed the scoping: +**Amended (P2 review): Application PRNG reseeding is P3 scope, and the P2 exposure is negligible.** The original EARS requirement below implied a `/run`-time reseed had to land with P2; it did not, and shipping P2 with the requirement unmet-but-asserted was itself the defect. Measured exposure, which is what changed the scoping: - The only consumer of a non-cryptographic PRNG in `agent/src` is `progress_writer.py`'s `getrandbits(80)`, used for the random half of a **ULID**. `os.urandom` / `secrets` are not seeded from the snapshot at all, and nothing in the agent derives a key, token, or nonce from `random`. - That ULID is a DynamoDB **sort key under a `task_id` partition**. A collision therefore needs two events in the *same task* at the *same millisecond* with the same 80 random bits — and identical PRNG state across two MicroVMs restored from one snapshot does not produce that, because the events are in different partitions. The worst case is a duplicate progress event within one task, not a security boundary. -- There is no credential exposure from the snapshot's PRNG state: build-role credentials are kept out of the snapshot structurally (the build hooks make zero AWS calls, so `boto3.DEFAULT_SESSION` is never populated — see `_aws_silent_log`), and per-task credentials arrive via `platform_config.agent_session_role_arn` at `/run`. +- Credential safety is a separate concern from the application PRNG. Build hooks avoid initializing AWS clients and their credential caches; their credential-environment warning does not prove that every possible image is secret-free. `platform_config.agent_session_role_arn` delivers a role identifier, not credentials. The running agent obtains credentials through its runtime provider and assumes the task-scoped role. -So the reseed moves to P3 alongside `/suspend` + `/resume`, where a *resumed* VM — which really does continue with the exact PRNG state it was frozen with, repeatedly — makes it load-bearing rather than theoretical. P3 must reseed on both `/run` and `/resume`, and must not treat the ULID as the only consumer: any future use of `random` for anything security-relevant needs the reseed in place first. +So the reseed moves to P3 alongside `/suspend` + `/resume`, where a *resumed* VM — which really does continue with the exact PRNG state it was frozen with, repeatedly — makes it load-bearing rather than theoretical. P3 must reseed on both `/run` and `/resume`, and must not treat the ULID as the only consumer: security-sensitive values must continue to use `secrets` or an equivalent cryptographic source, not `random`. Normative requirements (EARS): - While a `lambda-microvm` task is in `AWAITING_APPROVAL` and the gate's remaining window exceeds the configured grace period plus resume overhead, the orchestrator shall call `suspendSession` after the grace period. - When the approve or deny Lambda commits a decision for a `lambda-microvm` task, the Lambda shall load the MicroVM handle from `compute_metadata` and call `resumeSession` best-effort. - If the inline resume fails, then the Lambda shall record a resume-orphan task event and shall still return the decision outcome. -- While a decision row exists and the MicroVM remains `SUSPENDED`, the orchestrator shall retry `resumeSession`. +- While the current approval row is APPROVED or DENIED and the MicroVM remains `SUSPENDED`, the orchestrator shall retry `resumeSession`. A PENDING row alone is not a wake-up condition. - While a `lambda-microvm` task waits on an approval gate, the agent shall evaluate gate expiry as the earlier of its monotonic budget and the row's wall-clock deadline (`created_at + timeout_s`), on each poll iteration and on `/resume`. - If gate expiry is reached without a decision, then the agent shall deny. - If no decision arrives by the gate's wall-clock deadline minus the resume margin, then the orchestrator shall resume the MicroVM so the agent can evaluate expiry and fire the deny agent-side. @@ -171,10 +171,10 @@ The phasing is therefore: | `/ready` | **P1** (construct enables `hooks.microvmImageHooks.ready`) | **P1** | MANDATORY, not a quality nicety — see above. A 200 proves uvicorn is bound and `server` imported cleanly (pulling in `pipeline` → `runner` → the policy engine), so a missing policy file fails the BUILD instead of the first task. **Since P2-F5 it also WARMS the snapshot** — the hook's 200 is what the service waits for before capturing the snapshot, making this the only place a warm page can be created, and the 225 MiB `claude` binary was cold in it (see the P2-F5 correction below). A required warm-up failure answers 503, so a snapshot that cannot exec the agent's own CLI fails the image build instead of every task. Still makes ZERO AWS calls, logging included (a `--version` exec is neither an AWS call nor a network call). | | `/run` | **P1** (construct sets `hooks.microvmHooks.run`) | **P1** | The payload-delivery channel. Must be served in P1 because `/ready` forces hooks to exist at all, and a hook-less image cannot accept `runHookPayload`. Since P2 it is also the **platform-configuration** channel (see "Platform configuration delivery" below). | | `/validate` | **P2** (construct sets `hooks.microvmImageHooks.validate`) | **P2** | An **image** (build-time) hook, and a **shallow self-check only**: server alive, hook routes registered, interpreter + contract sanity. It runs under the BUILD role, which deliberately holds no Bedrock / Secrets Manager / DynamoDB grants, so it must make **zero AWS API calls** and must not touch credential resolution — the "deeper warm-up assertions (Bedrock reachability, Memory access, tool availability)" this ADR originally assigned here are **not implementable**: every one of them would `AccessDenied` and fail every image build. They belong to the first task's own error handling. 200 when the checks pass, 503 while still initialising. | -| `/terminate` | **P2** (construct sets `hooks.microvmHooks.terminate`) | **P2** | Best-effort final flush: a final structured log line, then 200 — always, inside the hook budget, even with nothing running. It must **not** write terminal task status (the orchestrator finalizes the task and *then* calls `TerminateMicrovm`, so a terminate hook that wrote a status would race that finalization and could clobber the real outcome) and must not join the pipeline thread. There is nothing buffered to flush: `_ProgressWriter` does a synchronous `put_item` per event, so durability is per-write. "Always 200" also covers the BODY: the handler reads the raw request rather than a typed model, because a typed body is validated before the handler runs and would answer 422 to malformed JSON — a reported hook failure on a successful teardown. Safe to declare because `TerminateMicrovm` removes the VM with or without in-guest cooperation. **Correction (P2-F8):** the service sends `microvmId: ""` on this hook, unlike `/run` where it is populated, so an empty id is expected-normal and this hook cannot join the guest record to the control-plane one — `/run`'s accepted line carries that correlation instead. | +| `/terminate` | **P2** (construct sets `hooks.microvmHooks.terminate`) | **P2** | Best-effort final flush: a final structured log line, then 200 — always, inside the hook budget, even with nothing running. It must **not** write terminal task status (the orchestrator finalizes the task and *then* calls `TerminateMicrovm`, so a terminate hook that wrote a status would race that finalization and could clobber the real outcome) and must not join the pipeline thread. `ProgressWriter` writes synchronously but drops failed events, so this hook cannot guarantee that every progress write succeeded. P3 needs an acknowledged durability barrier for suspension. "Always 200" also covers the BODY: the handler reads the raw request rather than a typed model, because a typed body is validated before the handler runs and would answer 422 to malformed JSON — a reported hook failure on a successful teardown. Safe to declare because `TerminateMicrovm` removes the VM with or without in-guest cooperation. **Correction (P2-F8):** the service sends `microvmId: ""` on this hook, unlike `/run` where it is populated, so an empty id is expected-normal and this hook cannot join the guest record to the control-plane one — `/run`'s accepted line carries that correlation instead. | | `/suspend`, `/resume` | **P3** | **P3** | Declaring a runtime hook the agent does not serve fails the corresponding lifecycle transition, so each is declared only in the phase that implements it. P1 termination is the orchestrator's `TerminateMicrovm`, which needs no in-guest cooperation. | -Consequence to state plainly, replacing the original "a P1-built MicroVM image is not runnable end to end": **a P1 image is creatable, launchable and payload-deliverable, but carries no smoke-parity guarantee.** P1 delivers the strategy, the construct, the roles/buckets/connectors, the image resource, the packaging script, and the `/ready` + `/run` endpoints — so a `lambda-microvm` task can start a MicroVM and hand it a payload. What P1 has **not** established is anything P2 owns: AgentCore Memory grants and `MEMORY_ID` delivery, the agent's non-secret env parity inside the snapshot, egress specifics from a running MicroVM, and heartbeat/progress behaviour end to end. No clone → change → PR run has happened on this substrate. P2 ("smoke parity") is the phase that closes that gap. The construct and the packaging script both surface exactly this at synth/run time (`abca:microvm-image-p1-smoke-unverified`) so an operator cannot mistake a launchable substrate for a verified one. +Consequence to state plainly, replacing the original "a P1-built MicroVM image is not runnable end to end": **a P1 image is creatable, launchable and payload-deliverable, but carries no smoke-parity guarantee.** P1 delivers the strategy, the construct, the roles/buckets/connectors, the image resource, the packaging script, and the `/ready` + `/run` endpoints — so a `lambda-microvm` task can start a MicroVM and hand it a payload. What P1 has **not** established is anything P2 owns: AgentCore Memory grants and `MEMORY_ID` delivery, the agent's non-secret env parity inside the snapshot, egress specifics from a running MicroVM, and heartbeat/progress behaviour end to end. At the P1 handoff, no clone → change → PR run had happened. P2 subsequently demonstrated that path with an IAM workaround; clean unattended verification is still pending. The construct and the packaging script both surface exactly this at synth/run time (`abca:microvm-image-p1-smoke-unverified`) so an operator cannot mistake a launchable substrate for a verified one. **The `AWS::Lambda::MicrovmImage` L1 enforces the API's enums (P2-F2, live 2026-08-06).** This closes the one item P1 left explicitly open, and it closes it against the construct's own stated reasoning. CloudFormation's generated types make `cpuConfigurations[].architecture` and all four `hooks.*` fields plain strings and document no allowed values, from which P1 concluded that the CloudFormation surface takes a *hook path* while the API takes an `ENABLED`/`DISABLED` flag, and that both were correct for their own surface. CloudFormation refused the change set at **early validation** — the stack was never touched, so there was no rollback and no runtime symptom to trace back — on five values: @@ -215,11 +215,11 @@ The agent's reader is deliberately more permissive than this contract: it also a **Platform configuration delivery (P2): payload-sourced, allowlisted, fail-closed.** The other two backends hand the agent its non-secret platform env at launch — AgentCore Runtime env vars, ECS container overrides — and there is no equivalent on this substrate: a MicroVM starts from a **snapshot**, so its process environment is whatever was frozen at *image build* time and is then replayed by every MicroVM launched from that image version. Baking the deployment's identifiers into the snapshot would make them **version-frozen**: a redeploy that renames a table, adds a bucket or rotates the session role would leave every existing image version describing a deployment that no longer exists, and the drift would surface as a task-time `ResourceNotFound` rather than a deploy-time error. So the values travel with the task instead: `platform_config` is a SIBLING of the payload — beside `agent_payload` in the inline branch, beside `agent_payload_s3_uri` in the pointer branch, and merged in beside the payload's own fields inside the S3 object (the canonical shapes above give each one exactly) — whose snake_case keys the agent installs into `os.environ` as their UPPER_SNAKE equivalents. A payload value therefore **wins** over any pre-existing/image value — the orchestrator is describing the live deployment, the snapshot is describing a past one. `platform_config` carries **non-secret identifiers only** (table and bucket names, secret ARNs, the session-role ARN); secrets are still fetched at `/run` time from Secrets Manager using those ARNs, so the snapshot-must-stay-secret-free requirement above is untouched. Per-task fields — `memory_id` and friends — stay inside `agent_payload`: `platform_config` configures the *process*, `agent_payload` describes the *task*. -Two rules make it safe. First, **the allowlist fails closed**: the agent installs a fixed set of keys and *rejects the entire run* (HTTP 400, nothing spawned, not one key installed) if the block carries anything else. These values become environment variables of the process that spawns the agent's tool subprocesses, so an unrecognised key is an attempt to set an arbitrary variable in the agent (`AWS_ENDPOINT_URL`, `LD_PRELOAD`, `PATH`, …) — an injection attempt, not a forward-compatibility gap, which is why unknown keys are refused rather than filtered out. Second, **installation happens before any credential or pipeline initialisation** on the hook path: the very next step reads `GITHUB_TOKEN_SECRET_ARN` to resolve the GitHub token and `AGENT_SESSION_ROLE_ARN` to scope the task's credentials, so installing later would silently resolve the whole task against the snapshot's frozen env. The one call that must precede installation is the S3 payload fetch (the config is *inside* the fetched object), which therefore runs on the ambient compute role via the attributed platform client — and it is the ONLY one: the same rule covers **logging**, so every `/run` log line before the install is stdout-only. The CloudWatch writer would otherwise resolve credentials and pin a boto3 default session (region included) off whatever a snapshot happened to bake, which is the build-hook defect one phase later. Nothing is lost — in the intended deployment there is no baked `LOG_GROUP_NAME`, so those lines would have gone to stdout anyway, and the reason for every pre-install rejection also travels in the structured 4xx/5xx body the service surfaces. A **required subset** (task table, task-events table, GitHub token secret ARN, session-role ARN) is rejected as `…_INCOMPLETE` when missing or blank — a distinct wire code from the `…_INVALID` allowlist rejection, because the remedies differ (deployment wiring vs. producer bug). A `/run` envelope with *no* `platform_config` at all is still accepted, loudly warned: the image snapshot and the orchestrator Lambda deploy on independent cadences, and a new image must not require a same-instant orchestrator. The key set is a cross-package contract in `contracts/constants.json` (`microvm_platform_config`), consumed by the agent's `/run` hook and produced by the orchestrator, with shape and required-subset invariants enforced by `scripts/check-constants-sync.ts`. +Two rules make it safe. First, **the allowlist fails closed**: the agent installs a fixed set of keys and *rejects the entire run* (HTTP 400, nothing spawned, not one key installed) if the block carries anything else. These values become environment variables of the process that spawns the agent's tool subprocesses, so an unrecognised key is an attempt to set an arbitrary variable in the agent (`AWS_ENDPOINT_URL`, `LD_PRELOAD`, `PATH`, …) — an injection attempt, not a forward-compatibility gap, which is why unknown keys are refused rather than filtered out. Second, **installation happens before any credential or pipeline initialisation** on the hook path: the very next step reads `GITHUB_TOKEN_SECRET_ARN` to resolve the GitHub token and `AGENT_SESSION_ROLE_ARN` to scope the task's credentials, so installing later would silently resolve the whole task against the snapshot's frozen env. The one call that must precede installation is the S3 payload fetch (the config is *inside* the fetched object), which therefore runs on the ambient compute role via the attributed platform client — and it is the ONLY one: the same rule covers **logging**, so every `/run` log line before the install is stdout-only. The CloudWatch writer would otherwise resolve credentials and cache credentials in a boto3 default session (environment-derived region is re-read for each new client) off whatever a snapshot happened to bake, which is the build-hook defect one phase later. Nothing is lost — in the intended deployment there is no baked `LOG_GROUP_NAME`, so those lines would have gone to stdout anyway, and the reason for every pre-install rejection also travels in the structured 4xx/5xx body the service surfaces. A **required subset** (task table, task-events table, GitHub token secret ARN, session-role ARN) is rejected as `…_INCOMPLETE` when missing or blank — a distinct wire code from the `…_INVALID` allowlist rejection, because the remedies differ (deployment wiring vs. producer bug). A `/run` envelope with *no* `platform_config` is accepted with a warning only when the effective environment already provides every required identifier; otherwise it is rejected as incomplete. The image and orchestrator can deploy on independent cadences without accepting an unusable environment. The key set is a cross-package contract in `contracts/constants.json` (`microvm_platform_config`), consumed by the agent's `/run` hook and produced by the orchestrator, with shape and required-subset invariants enforced by `scripts/check-constants-sync.ts`. **No orchestrator→agent HTTP path exists in P1–P3**: payload arrives through the `/run` hook, all agent work is outbound, and therefore **no JWE auth tokens are minted at all** — token minting (and its ≤ 60 min TTL refresh problem) is deferred until a real consumer exists (e.g. operator shell access, [#391](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/391)). The `endpoint` stays in the `SessionHandle` because it is genuinely per-session state that becomes load-bearing the day such a consumer appears. But note the service does not agree by default: omitting `ingressNetworkConnectors` on `RunMicrovm` attaches a **public** `HTTP_INGRESS` connector, so the strategy passes the Lambda-managed `NO_INGRESS` connector explicitly on every launch (see sub-decision 4's security table). -**Constraint accepted:** the configured **baseline is 8 GiB RAM / 4 vCPU** and the service scales vertically to a **32 GiB / 16 vCPU peak** on its own, with 32 GB of disk. So capacity is baseline-priced with 4× burst headroom — good for the bursty compile-and-test shape of an agent task — but the SUSTAINED ceiling is still 32 GiB, so repos that motivated the 120 GB ECS sizing stay on `ecs`. What the construct configures (and validates) is the baseline; the peak is not something a deployment asks for. +**Constraint accepted:** the configured **baseline is 8 GiB RAM / 4 vCPU** and the service scales vertically to a **32 GiB / 16 vCPU peak** on its own, with 32 GB of disk. So baseline capacity and additional burst usage are billed separately — good for the bursty compile-and-test shape of an agent task — but the SUSTAINED ceiling is still 32 GiB, so repos that motivated the 120 GB ECS sizing stay on `ecs`. What the construct configures (and validates) is the baseline; the peak is not something a deployment asks for. Normative requirements (EARS). **Each requirement's own `(Pn)` tag is authoritative**; there is no blanket phase for the list. The tags are per-requirement because the original list *was* split P1/P2 on the assumption that a hook could be declared in one phase and served in a later one — which the service does not permit (see the phasing table above), so the hook-serving requirements collapsed into P1 while the P2 items below arrived with the P2 hooks and `platform_config`: @@ -343,7 +343,7 @@ Two networking facts the construct has to encode, both established live: | IAM condition keys on the compute-role trust **and** on the `iam:PassRole` grants that hand it over | Trust pinned with `aws:SourceAccount`; `PassRole` under the allowlisted bootstrap statement | Trust pinned per-service; `PassRole` under the allowlisted bootstrap statement | **Neither is possible.** All three MicroVM-facing roles trust the bare `lambda.amazonaws.com` with no `aws:SourceAccount`/`aws:SourceArn`, **and** both `iam:PassRole` grants (orchestrator → execution role at `RunMicrovm`; CloudFormation → build role at `CreateMicrovmImage`) carry no `iam:PassedToService` — the service presents no usable value for any of those keys, and each condition is a hard blocker while present (live-verified, blocking, four times across two runs) | **Real, evidenced gap that does not close from our side, and it is wider than the trust policy alone.** `lambda.amazonaws.com` is shared with every other Lambda feature, so neither the account pin nor the passed-to-service pin is available on this path. Compensated per role (table in sub-decision 4): the **execution** role is passable by the **orchestrator only** (at `RunMicrovm`), restricted to its **exact ARN**; the **build** and **connector-operator** roles are passable by the CloudFormation deployment role under a new **conditional, per-backend, name-prefix-scoped** statement (`MicrovmPassRoles`, bootstrap ≥ 1.6.0) that deliberately excludes the execution role. The shared allowlisted `IAMPassRole` (`role/backgroundagent-dev-*`) is left intact to avoid widening the grant for ~30 other roles, so while it technically matches the execution role, only the orchestrator actively reaches for it. Resources are account-scoped by ARN apart from two justified `Resource: \'*\'` statements (`ec2:DescribeAvailabilityZones`; the operator role\'s ENI management — both carry cdk-nag IAM5 suppressions). No `iam:*`, no cross-account trust. Revisit if AWS ever documents the values the service presents; CloudTrail records no `lambda-microvms` events, so they cannot be read from logs | | Per-task observability writes | Runtime writes to the vended APPLICATION_LOGS group | Task role writes to the task log group | Execution role writes to the SAME APPLICATION_LOGS group, granted against the group `platform_config` names (P2-F4) | None — but only after P2-F4: the name was delivered a phase before the grant, so the agent attempted the write and every per-task line (and `METRICS_REPORT`) was `AccessDenied`, degrading silently to guest stdout | | Session isolation | MicroVM | Task-level | MicroVM (Firecracker) | None (≥ ECS) | -| State reuse | None | None | Snapshot shared across MicroVMs | New surface: CSPRNG reseed + credential refresh on `/run`/`/resume` — **P3 scope** (P2 exposure measured as negligible: sole `random` consumer is a ULID sort key under a `task_id` partition; `os.urandom`/`secrets` unaffected; no credential derives from `random`). Credential refresh IS in P2: per-task credentials arrive via `platform_config` at `/run` | +| State reuse | None | None | Snapshot shared across MicroVMs | New surface: application PRNG reseed + credential refresh on `/run`/`/resume` — **P3 scope** (P2 exposure measured as negligible: sole `random` consumer is a ULID sort key under a `task_id` partition; `os.urandom`/`secrets` unaffected; no credential derives from `random`). Credential refresh IS in P2: per-task credentials arrive via `platform_config` at `/run` | | Workload-token injection | Yes (Runtime-coupled) | No (env-var posture) | No (env-var posture) | Shared with ECS; deferred to [#249](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/249)/ADR-016 | | Operator shell access | No | No | Not enabled (`SHELL_INGRESS` omitted; candidate for [#391](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/391)) | None by default | | Auth-token minting | n/a | n/a | `CreateMicrovmAuthToken` granted to no role in any phase | Verified static-only: any principal holding the action *can* mint a working JWE, including against a `SUSPENDED` MicroVM, so the posture rests entirely on the grant being absent | @@ -385,7 +385,7 @@ Lambda MicroVMs launched in 5 regions (us-east-1/2, us-west-2, eu-west-1, ap-nor - (!) **Service defaults are not the desired posture.** Two live-caught cases (public `HTTP_INGRESS` by default; `/ready` mandatory) mean an omitted field on this backend does not mean "off" — it can mean "the service picks, and it picks wider than we want". Every new `RunMicrovm` / `CreateMicrovmImage` field should be assumed to have an opinionated default until checked. - (!) **Nothing self-terminates on the paths that matter** — superseding P1's unqualified version of this bullet. With `run: ENABLED` the service DOES reap a VM whose run hook returns 4xx (~12 s, `stateReason: "Run lifecycle hook returned HTTP status 400."`, live-verified), so a guest that rejects its own payload cleans itself up. That is the only self-cleaning case: the service reaps a hook *result*, and once `/run` has answered 200 it has no view of the guest. A VM whose task finished, crashed after `/run`, or hung stays `RUNNING` and billing until the 8 h cap, so the orchestrator's `TerminateMicrovm` on finalize remains the only cleanup for normal operation and a leaked handle is still a cost incident. - (!) **Snapshot uniqueness.** Shared memory snapshots mean every MicroVM restored from one image starts with identical PRNG state. **Re-scoped to P3 in the P2 review, with the exposure measured rather than assumed** (see the amendment under sub-decision 2): the sole `random` consumer in `agent/src` is a ULID sort key under a `task_id` partition, `os.urandom`/`secrets` are unaffected, and no credential or token derives from `random` — so the P2 exposure is a possible duplicate progress event, not a security boundary. It becomes load-bearing at P3, where a *resumed* VM continues from frozen state repeatedly; the reseed must land on `/run` and `/resume` together with those hooks. Asserting the requirement while leaving it unimplemented was the real defect, and this amendment is the fix. -- (−) **The agent stack template is at 98.6 % of CloudFormation's 1 MB limit** (985,886 bytes) and 486 of 500 resources with a MicroVM image configured — ~14 KB of headroom, i.e. roughly one more construct, and down from 98.4 % / ~16 KB one run earlier. Not caused by this backend (the MicroVM construct is ~6 KB of it) but reached by it, and it will block deploys for reasons that have nothing to do with MicroVMs. Tracked in [#735](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/735); the candidate remedies are `suppressTemplateIndentation` and a stack split. +- (−) **Root-stack resource headroom needs ongoing measurement.** The historical 985,886-byte / 486-resource measurement predates [#854](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/854), which reduced template size and resource count. The 2026-09-13 [offline review](../verification/645-p3-readiness-review.md) measures 472 root resources with a managed MicroVM image, and 489 with gateway and vault enabled in a scratch probe. Nesting can reduce that fullest root count to 474 in the prototype. These are unbundled synth measurements, not deployment validation; the vault combination remains guarded on main ([#857](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/857)). - (!) **Snapshot WARMTH is a first-class property, not an optimisation.** A snapshot inherits only the pages something touched before it was captured, so a large lazily-loaded artifact — the 225 MiB `claude` binary, and anything similar added later — pays its first-touch cost on the *first task* instead of at container start. That cost failed every task at turn 0 in the P2 smoke run (P2-F5). Anything heavyweight added to the image must be exec'd in `/ready`, and any timeout guarding a first touch must be sized for a cold page fault rather than for the work itself. - (!) **Regional availability (5 regions at launch, expanding)** — enforced in layers (synth-time static check, onboarding + doctor live probes, orchestration-time classification; see sub-decision 4). The static CDK constant is the one piece that rots as AWS expands; its update path and context-flag escape hatch are deliberate. - (!) **Workload-token injection delta persists** (shared with the ECS backend) until [#249](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/249)/ADR-016 land; document it in the security bar comparison rather than blocking on it. Memory and Gateway are explicitly *not* deltas — both are standalone services consumed via IAM from any substrate. @@ -395,7 +395,7 @@ Lambda MicroVMs launched in 5 regions (us-east-1/2, us-west-2, eu-west-1, ap-nor **P1 (start/poll/stop — no suspend):** - Unit tests for the strategy: start/poll/stop mapping (including `SessionStatus` `'suspended'` reported mechanically, without task-state interpretation), payload-size branching (inline vs S3 pointer, at the exact 4 096/4 097-byte boundary), the image-identifier-must-be-an-ARN guard, the explicit `NO_INGRESS` argument (including the blank-env-var fallback, which must never omit the field), error classification (`ServiceQuotaExceededException`, `ThrottlingException`, `ResourceNotFoundException`, regional-unavailability), the omit-`idlePolicy` invariant, and `maximumDurationInSeconds` fixed at 28 800. -- Agent tests: `/ready` returns 200 once the server is up and starts nothing; `/run` accepts both envelope shapes (inline and S3 pointer), starts the pipeline asynchronously through the same mapper `/invocations` uses, returns before the pipeline finishes, and rejects every unusable envelope with a named code before spawning; `/suspend` and `/resume` are NOT served (`/validate` and `/terminate` joined the served set in P2). +- Agent tests: `/ready` returns 200 after required local warm-up succeeds, without starting a task or calling AWS; `/run` accepts both envelope shapes (inline and S3 pointer), starts the pipeline asynchronously through the same mapper `/invocations` uses, returns before the pipeline finishes, and rejects every unusable envelope with a named code before spawning; `/suspend` and `/resume` are NOT served (`/validate` and `/terminate` joined the served set in P2). - Orchestrator tests: substrate-terminal + non-terminal task status → failed classification; `suspended` + non-`AWAITING_APPROVAL` status → anomaly event, no fail-fast; `compute_metadata` persisted with `microvmId`/`endpoint` after `startSession`. - CDK assertions: MicroVM resources present only when `ComputeTypes` includes the backend; synth failure for unsupported regions (plus the context-flag escape hatch); memory size validated against the accepted list at synth; the connector operator role and its trust; two connectors with the build-only one carrying port 80 and the runtime one not; the `/ready` + `/run` hook declaration and the absence of the others; IAM actions scoped as specified (orchestrator lifecycle set; no `CreateMicrovmAuthToken` anywhere); payload-bucket grants (execution role read-only); backend cost-allocation tags; types-sync check covers the widened `ComputeType`. - CLI tests: onboarding rejection with remedy when the availability probe fails; doctor check present when a blueprint selects the backend. @@ -404,7 +404,7 @@ Lambda MicroVMs launched in 5 regions (us-east-1/2, us-west-2, eu-west-1, ap-nor **P2 (smoke parity):** - Agent tests: `/validate` returns 200 with its individual check results, 503 while initialising, reports a missing hook route / unsupported interpreter, starts nothing, and makes **zero AWS calls even with `LOG_GROUP_NAME` set** (asserted by poisoning the boto3 and CloudWatch-writer seams — the same assertion covers `/ready`); `/terminate` returns 200 with no body at all, with a malformed / non-object / wrong-content-type / whitespace-only body, when the body read itself fails, with a pipeline still running (without joining it), and when its own best-effort step raises — and never calls `task_state.write_terminal`; a structural assertion that the route carries no typed body param keeps the 422 from being reintroduced. -- `platform_config` tests: the allowlist and required subset are read from `contracts/constants.json` (the wire key set is additionally asserted literally, as the agent-side tripwire on a contract edit); an unknown key rejects the whole block with nothing installed; a non-object block and a non-string value are rejected; a blank/`null` optional value is skipped without clobbering an image value while a blank required value is rejected; a payload value beats a pre-existing env value; installation is observed to happen before the GitHub-token resolver runs and before any pipeline thread exists; the config is picked up from the inline envelope, from beside the S3 pointer, and from inside the fetched object (inner wins); an envelope with no `platform_config` is still accepted with a warning. +- `platform_config` tests: the allowlist and required subset are read from `contracts/constants.json` (the wire key set is additionally asserted literally, as the agent-side tripwire on a contract edit); an unknown key rejects the whole block with nothing installed; a non-object block and a non-string value are rejected; a blank/`null` optional value is skipped without clobbering an image value while a blank required value is rejected; a payload value beats a pre-existing env value; installation is observed to happen before the GitHub-token resolver runs and before any pipeline thread exists; the config is picked up from the inline envelope, from beside the S3 pointer, and from inside the fetched object (inner wins); an envelope with no `platform_config` is accepted with a warning only if the effective environment already supplies every required identifier; otherwise it is rejected. - Snapshot credential hygiene: a subprocess probe asserts that importing `server` and serving `/ready` + `/validate` imports neither `boto3` nor `botocore`, caches no `aws_session` session, and spawns no CloudWatch writer thread — the property that keeps a build-role credential chain and the build-time region out of the snapshot. - `/run` pre-install silence: with a **baked `LOG_GROUP_NAME`** (the hostile case — without it the assertions pass vacuously) every AWS/credential seam (`boto3.client`/`Session`, the `aws_session` factories, `_debug_cw`/`_warn_cw`) is armed to raise until the install succeeds. Asserted on the accepted path, on all three rejection paths (bad envelope, `platform_config` invalid, `platform_config` incomplete) and on the failed-fetch 500 — where the seams stay armed for the whole request, because a rejected run installed nothing and so earns no AWS call. The permitted exception is asserted POSITIVELY: exactly one client is built pre-install, for `s3`, through the attributed factory. - `/ready` warm-up tests (P2-F5): the hook exec's each configured binary exactly once with a generous timeout; `claude` is the only REQUIRED entry; a timeout, a missing binary, a non-zero exit and an unexpected `OSError` each produce **503 with the reason logged to stdout** rather than a 200 or a 500; a best-effort failure still reports ready; the warm-up makes zero AWS calls with `LOG_GROUP_NAME` baked. Plus the backstop half: the `claude --version` probe's bound is asserted to be ≥ 60 s and to be applied to the *exec* rather than to the PATH lookup, and a missing CLI warns instead of raising. diff --git a/docs/design/COMPUTE.md b/docs/design/COMPUTE.md index 48670a6df..0ffc9921c 100644 --- a/docs/design/COMPUTE.md +++ b/docs/design/COMPUTE.md @@ -18,7 +18,7 @@ The default runtime is **Amazon Bedrock AgentCore Runtime**, which runs each ses | **Startup** | Service-managed | Slim images help | Snapshot resume | Warm ASGs + pre-pull | Karpenter + pre-pull | Backend-dependent | Provisioned concurrency | Snapshot pools (DIY) | | **GPU** | No | No | No | Yes | Yes | Yes (EC2/EKS backend) | No | Yes (with passthrough) | | **Ops burden** | Low (managed) | Low | Low (managed) | Medium | High | Low-Medium | Low | **Very high** | -| **Cost model** | vCPU-hrs + GB-hrs | vCPU + mem/sec | Baseline-priced (8 GiB / 4 vCPU) with 4× vertical burst (32 GiB / 16 vCPU peak); suspended time is storage-only | EC2 + EBS | EKS control + EC2 | Underlying compute | Request + duration | EC2 metal + your ops | +| **Cost model** | vCPU-hrs + GB-hrs | vCPU + mem/sec | Baseline compute (8 GiB / 4 vCPU) plus additional burst usage (up to 32 GiB / 16 vCPU); no suspended compute charge, but snapshot storage and read/write charges remain | EC2 + EBS | EKS control + EC2 | Underlying compute | Request + duration | EC2 metal + your ops | | **Fit** | **Default choice** | Repos > 2 GB image | Suspend/resume economics; approval-wait-heavy workloads; default-sized repos. Heavy sustained-memory builds stay on ECS | GPU, heavy toolchains | Max flexibility | Queued batch jobs | **Poor** (15 min cap) | Best potential, highest cost | > **Lambda MicroVMs are not Lambda functions.** They are a different compute primitive, so the functions column's 15-minute cap and poor-fit verdict do not apply. See [ADR-021](../decisions/ADR-021-lambda-microvms-compute-backend.md). diff --git a/docs/design/ORCHESTRATOR.md b/docs/design/ORCHESTRATOR.md index 188eea2bd..36a289c27 100644 --- a/docs/design/ORCHESTRATOR.md +++ b/docs/design/ORCHESTRATOR.md @@ -276,7 +276,7 @@ Liveness detection varies by compute backend. AgentCore sessions use DynamoDB he - **Grace period** (120s) - After entering `RUNNING`, the orchestrator waits before expecting heartbeats (covers container startup). - **Stale threshold** (240s) - If the heartbeat exists but is older than this, the session is treated as lost. -- **Early crash** - If no heartbeat is ever set after the combined window (360s), the agent died before the pipeline started. +- **Early crash** - If no heartbeat is ever set after the combined window (360s), the session is treated as lost; a process failure or failed DynamoDB writes can cause this. When the session is unhealthy, the task transitions to `FAILED` with "Agent session lost: no recent heartbeat." @@ -286,7 +286,7 @@ When the session is unhealthy, the task transitions to `FAILED` with "Agent sess - `suspended` is healthy only while the task is `AWAITING_APPROVAL`; in any other task state it emits an anomaly and keeps polling rather than failing recoverable work. - A terminal substrate report paired with a non-terminal task is a failure, but the orchestrator first re-reads the task row to confirm the agent did not write a terminal result between the original read and VM termination. -- Substrate state detects a dead VM; heartbeat staleness detects a hung, deadlocked, or OOM-killed pipeline inside a VM that still reports `RUNNING`. +- Substrate state detects a dead VM; heartbeat staleness detects loss of the heartbeat writer inside a VM that still reports `RUNNING`. The independent heartbeat thread can continue during a pipeline hang, so a fresh timestamp is not proof of progress. `TERMINATED` is the normal terminal signal and remains observable for at least 10 minutes. `ResourceNotFoundException` maps to completion only as a late fallback after the control-plane record is eventually reaped; polling does not wait for `NotFound`. @@ -314,7 +314,7 @@ Long-running distributed systems fail. The orchestrator is designed so that ever | Hydration | Guardrail API unavailable | Fail the task (fail-closed: unscreened content never reaches agent) | | Session start | Selected compute service throttled | Exponential backoff. Fail after retries exhausted. | | Session start | Session crashes immediately | AgentCore: heartbeat never set, detected after 360s grace window. ECS: `DescribeTasks` reports failure. Lambda MicroVMs: `GetMicrovm` reports terminal state or the heartbeat never appears. | -| Running | Agent crashes mid-task | AgentCore: heartbeat goes stale. ECS: `DescribeTasks` reports stopped task. Lambda MicroVMs: `GetMicrovm` detects VM death and heartbeat staleness detects an in-guest hang. Finalization inspects GitHub for partial work. | +| Running | Agent crashes mid-task | AgentCore: heartbeat goes stale. ECS: `DescribeTasks` reports stopped task. Lambda MicroVMs: `GetMicrovm` detects VM death and heartbeat staleness detects loss of the in-guest writer. Finalization inspects GitHub for partial work. | | Running | Agent hits turn or budget limit | Session ends normally. Finalize based on what was produced. | | Running | Idle for 15 min | AgentCore kills session. Task transitions to `TIMED_OUT`. | | Finalization | GitHub API down | Retry 3x. If still failing, mark `FAILED` with infrastructure reason. | diff --git a/docs/guides/DEPLOYMENT_GUIDE.md b/docs/guides/DEPLOYMENT_GUIDE.md index c002eedad..4d6629650 100644 --- a/docs/guides/DEPLOYMENT_GUIDE.md +++ b/docs/guides/DEPLOYMENT_GUIDE.md @@ -4,24 +4,24 @@ This guide covers deploying ABCA into an AWS account, including compute backend ## Architecture overview -ABCA deploys as a **single CDK stack** (`backgroundagent-dev`) containing all platform resources. The stack uses a `ComputeStrategy` interface to support three compute backends within the same stack: +ABCA deploys through a **root CDK stack** (`backgroundagent-dev`) and nested stacks for parts of the platform, including registry infrastructure. A `ComputeStrategy` interface supports three compute backends: | Aspect | AgentCore (default) | ECS Fargate (opt-in) | Lambda MicroVMs (experimental) | |--------|--------------------|--------------------|--------------------| | **Compute** | Bedrock AgentCore Runtime (Firecracker MicroVMs) | ECS Fargate containers | AWS Lambda MicroVMs | -| **Resources** | 2 vCPU, 8 GB RAM, 2 GB max image size | 2 vCPU, 4 GB RAM | 8 GB baseline / 32 GB peak memory | +| **Resources** | 2 vCPU, 8 GB RAM, 2 GB max image size | Build: 4 vCPU / 16 GiB; planning: 2 vCPU / 8 GiB (configurable) | 8 GiB baseline / 32 GiB peak memory | | **Orchestration** | Durable Lambda (checkpoint/replay) | Same durable Lambda via `ComputeStrategy` | Same durable Lambda via `ComputeStrategy` | | **Agent mode** | FastAPI server (HTTP invocation) | Batch (run-to-completion) | FastAPI server (lifecycle hooks) | | **Startup** | ~10s (warm MicroVM) | ~60-180s (Fargate cold start) | ~6s to `RUNNING` (live-measured) | | **Max duration** | 8 hours (AgentCore service limit) | 9 hours (orchestrator `executionTimeout`) | 8 hours (`maximumDurationInSeconds`) | -All backends are orchestrated by the same durable Lambda function. The `ComputeStrategy` interface abstracts `startSession()`, `pollSession()`, and `stopSession()` -- the ECS strategy calls `ecs:RunTask` / `ecs:DescribeTasks` / `ecs:StopTask` directly from the Lambda. No Step Functions are used. +All backends are orchestrated by the same durable Lambda function. The `ComputeStrategy` interface abstracts `startSession()`, `pollSession()`, and `stopSession()` -- the ECS strategy calls `ecs:RunTask` / `ecs:DescribeTasks` / `ecs:StopTask` directly from the Lambda. Task orchestration does not use Step Functions; other platform features may use them. -ECS Fargate is currently **opt-in** -- the `EcsAgentCluster` construct is present in the stack code but commented out. To enable it, uncomment the ECS blocks in `cdk/src/stacks/agent.ts`. +ECS Fargate is **opt-in**. Deploy with `--context compute_type=ecs`; the stack enables `EcsAgentCluster` from that context flag. ### Lambda MicroVMs backend (experimental) -> **Not for production.** `lambda-microvm` carries no smoke-parity guarantee for an unattended deployment. Keep production repositories on `agentcore` or `ecs`. Synth emits an unsuppressible warning to this effect whenever the backend is selected. Design detail: [COMPUTE.md](../design/COMPUTE.md) and [ADR-021](../decisions/ADR-021-lambda-microvms-compute-backend.md). +> **Not for production.** `lambda-microvm` carries no smoke-parity guarantee for an unattended deployment. Keep production repositories on `agentcore` or `ecs`. Synth emits a verification warning whenever a MicroVM image is configured; selecting the backend without an image emits a separate setup warning. Design detail: [COMPUTE.md](../design/COMPUTE.md) and [ADR-021](../decisions/ADR-021-lambda-microvms-compute-backend.md). Selecting it is a synth-time context flag: @@ -53,7 +53,7 @@ mise //cdk:deploy -- --context compute_type=lambda-microvm Operational notes specific to this backend: -- **Nothing self-terminates.** A MicroVM whose task finished, crashed, or hung stays `RUNNING` and billing until the 8-hour cap. The orchestrator calls `TerminateMicrovm` on finalize, and the heartbeat-staleness check catches a hung guest inside a healthy VM -- but a leaked handle is a cost incident. The one exception: the service reaps a VM whose `/run` hook returns 4xx (~12s). +- **Nothing self-terminates.** A MicroVM whose task finished, crashed, or hung stays `RUNNING` and billing until the 8-hour cap. The orchestrator calls `TerminateMicrovm` on finalize, and the heartbeat-staleness check detects loss of the in-guest heartbeat writer (a pipeline hang can leave that writer running) -- but a leaked handle is a cost incident. The one exception: the service reaps a VM whose `/run` hook returns 4xx (~12s). - **Logs** land in `/aws/lambda-microvms/`. Guest stdout goes there too, which is the fallback path when the agent cannot reach the application log group. - **Deployment identifiers are not baked into the image.** The snapshot carries no configuration; table names, secret ARNs, and the per-task session-role ARN arrive in the `/run` payload as a `platform_config` block. A version-skewed orchestrator that does not send it is refused rather than run with tenant scoping disabled. diff --git a/docs/guides/LINEAR_SETUP_GUIDE.md b/docs/guides/LINEAR_SETUP_GUIDE.md index 8c95b9449..c76c862af 100644 --- a/docs/guides/LINEAR_SETUP_GUIDE.md +++ b/docs/guides/LINEAR_SETUP_GUIDE.md @@ -38,7 +38,7 @@ When a workspace's authorization dies, ABCA records it on the registry row and p #### Not available with `compute_type=lambda-microvm` -The vault and the Lambda MicroVMs substrate cannot be enabled on the same stack. Together they synthesize 505 CloudFormation resources against a hard limit of 500 (MicroVM alone is 496, the vault alone 488), so `cdk deploy` refuses the combination by name at synth rather than failing partway through. Use the vault on the `agentcore` or `ecs` substrate; a MicroVM stack stays on Secrets Manager until the stack reclaims room. +The vault and the Lambda MicroVMs substrate remain gated from being enabled together ([#857](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/857)). The original 505-resource measurement predates stack-size reductions; the [current offline review](../verification/645-p3-readiness-review.md) finds room in its tested configurations, but bundled feature-matrix and deployment verification are still required before removing the guard. Use the vault on `agentcore` or `ecs`; MicroVM deployments use Secrets Manager while this gate remains. #### One workload identity per stack diff --git a/docs/src/content/docs/architecture/Compute.md b/docs/src/content/docs/architecture/Compute.md index c574d6350..a7ae76083 100644 --- a/docs/src/content/docs/architecture/Compute.md +++ b/docs/src/content/docs/architecture/Compute.md @@ -22,7 +22,7 @@ The default runtime is **Amazon Bedrock AgentCore Runtime**, which runs each ses | **Startup** | Service-managed | Slim images help | Snapshot resume | Warm ASGs + pre-pull | Karpenter + pre-pull | Backend-dependent | Provisioned concurrency | Snapshot pools (DIY) | | **GPU** | No | No | No | Yes | Yes | Yes (EC2/EKS backend) | No | Yes (with passthrough) | | **Ops burden** | Low (managed) | Low | Low (managed) | Medium | High | Low-Medium | Low | **Very high** | -| **Cost model** | vCPU-hrs + GB-hrs | vCPU + mem/sec | Baseline-priced (8 GiB / 4 vCPU) with 4× vertical burst (32 GiB / 16 vCPU peak); suspended time is storage-only | EC2 + EBS | EKS control + EC2 | Underlying compute | Request + duration | EC2 metal + your ops | +| **Cost model** | vCPU-hrs + GB-hrs | vCPU + mem/sec | Baseline compute (8 GiB / 4 vCPU) plus additional burst usage (up to 32 GiB / 16 vCPU); no suspended compute charge, but snapshot storage and read/write charges remain | EC2 + EBS | EKS control + EC2 | Underlying compute | Request + duration | EC2 metal + your ops | | **Fit** | **Default choice** | Repos > 2 GB image | Suspend/resume economics; approval-wait-heavy workloads; default-sized repos. Heavy sustained-memory builds stay on ECS | GPU, heavy toolchains | Max flexibility | Queued batch jobs | **Poor** (15 min cap) | Best potential, highest cost | > **Lambda MicroVMs are not Lambda functions.** They are a different compute primitive, so the functions column's 15-minute cap and poor-fit verdict do not apply. See [ADR-021](/sample-autonomous-cloud-coding-agents/architecture/adr-021-lambda-microvms-compute-backend). diff --git a/docs/src/content/docs/architecture/Orchestrator.md b/docs/src/content/docs/architecture/Orchestrator.md index f1bfc6dd9..1917cfee5 100644 --- a/docs/src/content/docs/architecture/Orchestrator.md +++ b/docs/src/content/docs/architecture/Orchestrator.md @@ -280,7 +280,7 @@ Liveness detection varies by compute backend. AgentCore sessions use DynamoDB he - **Grace period** (120s) - After entering `RUNNING`, the orchestrator waits before expecting heartbeats (covers container startup). - **Stale threshold** (240s) - If the heartbeat exists but is older than this, the session is treated as lost. -- **Early crash** - If no heartbeat is ever set after the combined window (360s), the agent died before the pipeline started. +- **Early crash** - If no heartbeat is ever set after the combined window (360s), the session is treated as lost; a process failure or failed DynamoDB writes can cause this. When the session is unhealthy, the task transitions to `FAILED` with "Agent session lost: no recent heartbeat." @@ -290,7 +290,7 @@ When the session is unhealthy, the task transitions to `FAILED` with "Agent sess - `suspended` is healthy only while the task is `AWAITING_APPROVAL`; in any other task state it emits an anomaly and keeps polling rather than failing recoverable work. - A terminal substrate report paired with a non-terminal task is a failure, but the orchestrator first re-reads the task row to confirm the agent did not write a terminal result between the original read and VM termination. -- Substrate state detects a dead VM; heartbeat staleness detects a hung, deadlocked, or OOM-killed pipeline inside a VM that still reports `RUNNING`. +- Substrate state detects a dead VM; heartbeat staleness detects loss of the heartbeat writer inside a VM that still reports `RUNNING`. The independent heartbeat thread can continue during a pipeline hang, so a fresh timestamp is not proof of progress. `TERMINATED` is the normal terminal signal and remains observable for at least 10 minutes. `ResourceNotFoundException` maps to completion only as a late fallback after the control-plane record is eventually reaped; polling does not wait for `NotFound`. @@ -318,7 +318,7 @@ Long-running distributed systems fail. The orchestrator is designed so that ever | Hydration | Guardrail API unavailable | Fail the task (fail-closed: unscreened content never reaches agent) | | Session start | Selected compute service throttled | Exponential backoff. Fail after retries exhausted. | | Session start | Session crashes immediately | AgentCore: heartbeat never set, detected after 360s grace window. ECS: `DescribeTasks` reports failure. Lambda MicroVMs: `GetMicrovm` reports terminal state or the heartbeat never appears. | -| Running | Agent crashes mid-task | AgentCore: heartbeat goes stale. ECS: `DescribeTasks` reports stopped task. Lambda MicroVMs: `GetMicrovm` detects VM death and heartbeat staleness detects an in-guest hang. Finalization inspects GitHub for partial work. | +| Running | Agent crashes mid-task | AgentCore: heartbeat goes stale. ECS: `DescribeTasks` reports stopped task. Lambda MicroVMs: `GetMicrovm` detects VM death and heartbeat staleness detects loss of the in-guest writer. Finalization inspects GitHub for partial work. | | Running | Agent hits turn or budget limit | Session ends normally. Finalize based on what was produced. | | Running | Idle for 15 min | AgentCore kills session. Task transitions to `TIMED_OUT`. | | Finalization | GitHub API down | Retry 3x. If still failing, mark `FAILED` with infrastructure reason. | diff --git a/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md b/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md index fec83f5a4..68e714ef9 100644 --- a/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md +++ b/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md @@ -4,17 +4,17 @@ title: Adr 021 lambda microvms compute backend # ADR-021: AWS Lambda MicroVMs as a third ComputeStrategy backend -> Number: candidate ADR-021 (ADR-020 is the highest accepted on `main`; ADR-018 is claimed by open PR [#548](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/548), ADR-019 by open PR [#663](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/663)). Numbers are never reused. If a lower number frees before merge, renumber and coordinate with those PRs. +> **Implementation status (2026-09-13):** P1 ([#689](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/689)) and P2 ([#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733)) are merged. P2 completed real coding tasks with an IAM workaround; a clean rerun after re-bootstrap remains outstanding. P3 suspend/resume integration is not implemented. This ADR defines P1–P3, not a P4. See the [current review](/sample-autonomous-cloud-coding-agents/architecture/645-p3-readiness-review) and [P3 implementation plan](/sample-autonomous-cloud-coding-agents/architecture/645-p3-implementation-plan). **Status:** proposed **Date:** 2026-07-29 ## Context -ABCA selects a per-repo compute backend through the Blueprint's `compute_type` field (`cdk/src/handlers/shared/repo-config.ts`). Two backends exist today, resolved by `resolveComputeStrategy` (`cdk/src/handlers/shared/compute-strategy.ts`) behind a uniform `ComputeStrategy` interface (`startSession` / `pollSession` / `stopSession`): +ABCA selects a per-repo compute backend through the Blueprint's `compute_type` field (`cdk/src/handlers/shared/repo-config.ts`). Before this ADR, two backends existed, resolved by `resolveComputeStrategy` (`cdk/src/handlers/shared/compute-strategy.ts`) behind a uniform `ComputeStrategy` interface (`startSession` / `pollSession` / `stopSession`): - **AgentCore Runtime** (`agentcore`, default) — managed Firecracker MicroVM per session. Invoked via `InvokeAgentRuntime`; liveness is inferred from agent heartbeats in DynamoDB plus the FastAPI `/ping` endpoint (`agent/src/server.py`) — the strategy's `pollSession` is a stub that always reports `running`. Constraints: 2 GB image limit, no substrate-level suspend API exposed to the orchestrator. -- **ECS on Fargate** (`ecs`) — always-on Fargate task (16 vCPU / 120 GB, ARM64) for repos that exceed AgentCore's limits ([#596](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/596)). Invoked via `RunTask` in batch mode (bypassing the HTTP server); liveness via `DescribeTasks`. No suspend — an idle task blocked on an approval wait burns full compute the whole time. +- **ECS on Fargate** (`ecs`) — Fargate task (ARM64; current build default 4 vCPU / 16 GiB, configurable up to 16 vCPU / 120 GiB) for repos that exceed AgentCore's limits ([#596](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/596)). Invoked via `RunTask` in batch mode (bypassing the HTTP server); liveness via `DescribeTasks`. No suspend — an idle task blocked on an approval wait burns full compute the whole time. **AWS Lambda MicroVMs** (launched 2026-06-22) is a serverless Firecracker sandbox primitive that AWS positions explicitly for AI coding agents: VM-level isolation, snapshot-based near-instant launch, **suspend/resume with full memory + disk state preserved** (compute charges stop while suspended), a dedicated JWE-authenticated HTTPS endpoint per instance, lifecycle hooks (`/run`, `/suspend`, `/resume`, `/terminate`), and up to 8 hours per session. It is **not** classic Lambda: the 15-minute function cap does not apply, and [COMPUTE.md](/sample-autonomous-cloud-coding-agents/architecture/compute)'s "Lambda: poor fit" verdict refers to functions, not MicroVMs. @@ -27,7 +27,7 @@ ABCA selects a per-repo compute backend through the Blueprint's `compute_type` f | Isolation | MicroVM (managed) | Task-level (Firecracker) | MicroVM (Firecracker) | | Max duration | 8 h | No cap | 8 h (running + suspended) — **verified** (`L-B430C318 = 8 hours`) | | Suspend/resume | No orchestrator-visible API | No | **Yes** — explicit API + idle policy, state preserved, no compute charge while suspended. **Verified**: suspend reaches `SUSPENDED` in ~1 s with no `idlePolicy`; resume restores `RUNNING` in ~1 s with `microvmId` **and** `endpoint` byte-identical | -| Resources | AgentCore-managed | 16 vCPU / 120 GB / 20–200 GB disk | **Baseline 8 GiB RAM / 4 vCPU, auto-scaling to a 32 GiB / 16 vCPU peak**; 32 GB disk. `minimumMemoryInMiB` configures the BASELINE (max 8,192 MiB); the service scales vertically on demand — capacity is baseline-priced with 4× burst headroom | +| Resources | AgentCore-managed | Build default 4 vCPU / 16 GiB; up to 16 vCPU / 120 GiB; 20–200 GB disk | **Baseline 8 GiB RAM / 4 vCPU, auto-scaling to a 32 GiB / 16 vCPU peak**; 32 GB disk. `minimumMemoryInMiB` configures the BASELINE (max 8,192 MiB); the service scales vertically on demand — baseline capacity and additional burst usage are billed separately | | Packaging | ECR image ≤ 2 GB | ECR image, no hard cap | **Zip + Dockerfile in S3 → service-built snapshot image** (versioned, storage billed) | | Invocation | `InvokeAgentRuntime` (SigV4) | `RunTask` + container overrides | `RunMicrovm` (image **ARN** required — a bare name is rejected) → dedicated HTTPS endpoint + JWE token (`CreateMicrovmAuthToken`, ≤ 60 min TTL) | | Liveness | Agent heartbeat + `/ping` | `DescribeTasks` | MicroVM state (RUNNING / SUSPENDED / TERMINATED) via control-plane API **and** agent heartbeat (see sub-decision 1) | @@ -53,7 +53,7 @@ ABCA selects a per-repo compute backend through the Blueprint's `compute_type` f 1. **Idle detection is inbound-traffic-based; the ABCA agent is outbound-only.** MicroVM idle policies suspend when no traffic arrives at the *endpoint*. A busy agent running a 40-minute build receives no inbound traffic and would be suspended mid-work by a naive idle policy. Conversely, "no inbound traffic" is the agent's *normal* state. 2. **No self-suspend.** The agent cannot suspend its own MicroVM from inside; only an external `SuspendMicrovm` call can. Suspend decisions must be owned by the orchestrator — which aligns with the unified liveness model proposed in [#491](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/491). -3. **Snapshots bake state.** The image snapshot is captured once at build time; every MicroVM resumes from it. Secrets, tokens, and per-task identity must arrive at run time (`runHookPayload`, ≤ 4 KB, or fetched in the `/run` hook), never at image build. CSPRNG reseeding is a consequence of the same property and is scoped to **P3** — see the amended note under sub-decision 2 and the risk bullet, which record why the exposure is negligible today. +3. **Snapshots bake state.** The image snapshot is captured once at build time; every MicroVM resumes from it. Secrets, tokens, and per-task identity must arrive at run time (`runHookPayload`, ≤ 4 KB, or fetched in the `/run` hook), never at image build. Application PRNG reseeding is a consequence of the same property and is scoped to **P3** — see the amended note under sub-decision 2 and the risk bullet, which record why the exposure is negligible today. 4. **Auth tokens are short-lived.** JWE tokens max out at 60 minutes; any orchestrator→agent HTTP interaction over the endpoint needs token refresh, unlike AgentCore's SigV4 invoke or ECS's no-endpoint model. 5. **Identity delta — narrower than it looks.** Most AgentCore services ABCA uses are standalone and substrate-portable: Memory is already consumed from ECS via an IAM grant plus `MEMORY_ID` (`EcsAgentCluster`), and Gateway (ADR-019) is portable by design (SigV4 inbound). The genuinely Runtime-coupled piece is the workload-access-token **delivery mechanism** (`runtimeUserId` → `WorkloadAccessToken` request header → `BedrockAgentCoreContext`, used by `resolve_linear_api_token()`), which has no MicroVM equivalent. The ECS backend already lives with this delta (env-var token delivery); MicroVMs inherit the same posture until the pluggable identity work ([#249](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/249), ADR-016) redesigns the seam. 6. **The service's defaults are not our posture.** Two of them, both discovered live: `RunMicrovm` attaches a **public** `HTTP_INGRESS` connector (and mints a public `*.lambda-microvm..on.aws` endpoint) when `ingressNetworkConnectors` is omitted, and `CreateMicrovmImage` **requires** the `/ready` hook whenever any lifecycle hook is enabled. Neither posture can be reached by leaving a field out — each needs an explicit control (see sub-decisions 3 and 4). @@ -72,9 +72,9 @@ The `ComputeStrategy` interface gains **mandatory** `suspendSession(handle)` / ` **Poll semantics — the strategy reports, the orchestrator interprets.** `pollSession(handle)` receives only the session handle and cannot see task state, so the health rules must live where the DynamoDB status lives. `SessionStatus` gains a `'suspended'` variant; the strategy maps `GetMicrovm` state mechanically and the **orchestrator** cross-references against the task row — the same division of labor `finalPollState` already uses for ECS (substrate stopped + non-terminal DynamoDB status → failed) and `pollTaskStatus` uses for agentcore heartbeats: substrate `suspended` + task `AWAITING_APPROVAL` is healthy (orchestrator-intended suspend); `suspended` with any other task status is an anomaly to surface, not fail-fast; substrate terminal + non-terminal task status → classify failed. -**Liveness on this backend is substrate state AND agent heartbeat.** The substrate cross-check above answers one question — "is the VM still there?" — and P2 established that it is not sufficient on its own. `GetMicrovm` catches a MicroVM that *died*; it cannot catch a MicroVM that is alive and reporting `RUNNING` while the pipeline **inside the guest** is hung, deadlocked, or was OOM-killed. That state is not hypothetical or self-correcting on this substrate, and the P2 live run narrowed *why* without weakening the conclusion. The service does reap a VM whose run hook FAILS (a 4xx makes it terminate within ~12 s — see sub-decision 2), so the P1 evidence for this paragraph (a hook-less image sitting in `RUNNING` indefinitely with no `stateReason`) no longer describes an ABCA image. What the service reaps is a hook *result*; it has no view into the guest afterwards. So the surviving — and more realistic — hang case is a `/run` hook that returned **200** and a pipeline that then hung, deadlocked or was OOM-killed behind it: the substrate stays `RUNNING`, the service is satisfied, and nothing else notices. Left to the substrate check alone, such a task would burn the orchestrator's full ~8.5 h poll window — billing an 8-hour reservation — before the safety net fired. +**Liveness on this backend combines substrate state and the agent heartbeat.** `GetMicrovm` reports whether the VM is running or terminal. The independent `_heartbeat_worker` in `agent/src/server.py` refreshes the task timestamp every 45 seconds while the task is `RUNNING`; a stale timestamp detects loss of that writer, including a process crash or inability to write DynamoDB. It does **not** prove pipeline progress: a hung coding thread can coexist with a healthy heartbeat thread. A failed `/run` hook can trigger service teardown; after an accepted hook, ABCA still needs explicit termination and the maximum-duration backstop. A general progress watchdog remains separate work. -The in-guest half is already being written: the agent updates `agent_heartbeat_at` on the task row unconditionally, with no backend awareness, so the timestamp exists on every substrate. Only the orchestrator's *reaction* to it was backend-scoped — `pollTaskStatus` evaluated staleness for `agentcore` alone — which is the gap P2 closes by extending it to `lambda-microvm`. The grace and stale thresholds are the SAME on both: the timestamp is written by the same pipeline code at the same cadence, so a backend-specific window would encode a difference that does not exist. The two signals stay complementary rather than redundant — the substrate check is the crash detector, the heartbeat is the hang detector — and the check remains scoped to task status `RUNNING`, which is what keeps a deliberately suspended VM during an approval wait (sub-decision 2, P3) from being read as a dead one. `ecs` is deliberately left out: `DescribeTasks` reports a real container exit *with an exit code* (OOM-kill included) and the ECS poll block already interprets it with its own patience counters, so adding the heartbeat there would give one backend two independently-tuned kill paths for the same failure. +AgentCore and MicroVM tasks start the same periodic heartbeat worker, so they use the same 120-second grace and 240-second stale thresholds. ECS starts the pipeline directly, bypassing `server.py`; its pipeline writes a timestamp once but does not start this worker. Enabling the same stale check for ECS would therefore fail healthy tasks after roughly four minutes. The check applies only to `RUNNING`, not `AWAITING_APPROVAL`. The transition back to `RUNNING` currently leaves the old timestamp intact: after a long gate, a poll can see a stale heartbeat before the next worker tick. Refreshing it atomically on approval resume is a P3 prerequisite in the implementation plan. The service's `MicrovmState` enum has **six** members, not three, so the mapping is stated exhaustively (one line of rationale each, mirrored in the strategy's doc comment): @@ -82,7 +82,7 @@ The service's `MicrovmState` enum has **six** members, not three, so the mapping |---|---|---| | `PENDING` | `running` | Still booting; the same way ECS's `PENDING`/`PROVISIONING` map to `running`. | | `RUNNING` | `running` | — | -| `SUSPENDING` | `suspended` | Already on its way to frozen; reporting `running` would tell the orchestrator compute is still progressing when it is not. Both suspend states land on a report the orchestrator treats as benign-or-anomalous depending on task status, never as failure. **Never observable in practice** — suspend reached `SUSPENDED` in under 1 s live — so it is mapped for completeness and nothing may *wait* for it. | +| `SUSPENDING` | `suspended` | Already on its way to frozen; reporting `running` would tell the orchestrator compute is still progressing when it is not. Both suspend states land on a report the orchestrator treats as benign-or-anomalous depending on task status, never as failure. **Not observed in the recorded probe** — suspend reached `SUSPENDED` in under 1 s. Other runs may expose this state, so handle it without requiring it to appear. | | `SUSPENDED` | `suspended` | — | | `TERMINATING` | `completed` | Terminal-bound and carries no exit code, so "the substrate is gone" is all the strategy can honestly say. | | `TERMINATED` | `completed` | Success vs failure is the orchestrator's call — it cross-references the DynamoDB status. **This is the load-bearing terminal signal**, not `NotFound` (see below). | @@ -111,8 +111,8 @@ Normative requirements (EARS, per [ADR-020](/sample-autonomous-cloud-coding-agen - If `GetMicrovm` reports that the MicroVM does not exist, then the strategy shall report `completed`. - If the strategy reports a terminal substrate state while the task's DynamoDB status is non-terminal, then the orchestrator shall re-read the task row and, if it is still non-terminal, classify the task as failed with a substrate-failure remedy. - If the strategy reports `suspended` while the task's DynamoDB status is not `AWAITING_APPROVAL`, then the orchestrator shall surface an anomaly event and shall not fail-fast the task. -- While a `lambda-microvm` task's DynamoDB status is `RUNNING`, if the task's `agent_heartbeat_at` is stale (or absent past the grace window) by the same thresholds the orchestrator applies to `agentcore`, then the orchestrator shall treat the session as unhealthy and stop polling — the substrate `GetMicrovm` check shall remain the crash detector, and the heartbeat shall be the in-guest hang detector. -- The task-detail API response shall include `agent_heartbeat_at`, and the CLI shall surface it while the task is non-terminal (P2r2-F11: the field drove the orchestrator's hang detector but was never projected, so no operator could observe the signal — and its invisibility produced a wrong verification conclusion). +- While a `lambda-microvm` task's DynamoDB status is `RUNNING`, if the task's `agent_heartbeat_at` is stale (or absent past the grace window) by the same thresholds the orchestrator applies to `agentcore`, then the orchestrator shall treat the session as unhealthy and stop polling — the substrate `GetMicrovm` check shall remain the crash detector, and the heartbeat shall detect loss of the in-guest heartbeat writer, without claiming to detect every pipeline hang. +- The task-detail API response shall include `agent_heartbeat_at`, and the CLI shall surface it while the task is non-terminal (P2r2-F11: the field drove the orchestrator's heartbeat check but was never projected, so no operator could observe the signal — and its invisibility produced a wrong verification conclusion). - The task-**summary** API response (`GET /v1/tasks`) shall also include `agent_heartbeat_at`, and `bgagent list` shall render it as an age column. Extending the field to the list response is a deliberate widening of the requirement above rather than an incidental one: the detail-only projection makes liveness a per-task question, and an operator checking a fleet of tasks one `bgagent status` at a time is exactly how the hung task P2r2-F11 describes went unnoticed. Same suppression rule as the detail view (terminal tasks and never-beaten tasks render a placeholder) so the two views cannot disagree. - If `suspendSession` or `resumeSession` is invoked on a strategy that does not support suspension, then the strategy shall return an explicit unsupported result. - When the agent process reaches a terminal state, the agent shall exit. @@ -127,10 +127,10 @@ The headline economic win is suspend during **HITL approval waits** (Cedar appro The handshake must respect the existing approval mechanics: the agent **discovers decisions itself** by polling DynamoDB (`_poll_for_decision`, monotonic timeout), the approve/deny Lambda writes only the decision rows, and `AWAITING_APPROVAL` holds the concurrency slot (Cedar decision #7). Nothing "delivers" an approval to the agent, and suspension freezes the agent's monotonic clock — so the design is: - **Suspend — orchestrator-owned.** The orchestrator's durable poll observes `AWAITING_APPROVAL` on a `lambda-microvm` task and calls `suspendSession` after a grace period, and only when the gate's remaining window exceeds grace + resume overhead (suspending a 30 s gate is pure loss). Suspend is a policy decision on a poll observation, not a user action. -- **Resume — inline in the approve/deny Lambdas, orchestrator poll as backstop.** After the transactional decision write commits, `ApproveTaskFn`/`DenyTaskFn` load the MicroVM handle from the task row's `compute_metadata` (persisted at session start — the same field `cancel-task.ts` reads ECS handles from) via a post-commit strongly-consistent `GetItem`, then call `resumeSession` **best-effort**: on failure they log a warning and write a resume-orphan task event; the decision response never fails on a compute error (the decision row is already durable). The orchestrator poll reconciles: decision row present + MicroVM still `SUSPENDED` → retry resume (idempotent). +- **Resume — inline in the approve/deny Lambdas, orchestrator poll as backstop.** After the transactional decision write commits, `ApproveTaskFn`/`DenyTaskFn` load the MicroVM handle from the task row's `compute_metadata` (persisted at session start — the same field `cancel-task.ts` reads ECS handles from) via a post-commit strongly-consistent `GetItem`, then call `resumeSession` **best-effort**: on failure they log a warning and write a resume-orphan task event; the decision response never fails on a compute error (the decision row is already durable). The orchestrator poll reconciles: current approval row APPROVED/DENIED + MicroVM still `SUSPENDED` → retry resume; a PENDING row alone must not trigger wake-up. *Why inline rather than poll-only — codebase precedent:* resume-on-approve is structurally identical to task cancellation — a user-initiated, latency-sensitive action whose purpose is an immediate compute-lifecycle side effect. `cancel-task.ts` already resolves this exact tension: the API-plane handler invokes ECS `StopTask` / AgentCore `StopRuntimeSession` inline, best-effort (a failed stop logs a warning and the state transition stands; a `task_cancel_compute_orphan` event is written when no stoppable compute handle exists, `reason: missing_runtime_handle`) — with the conditional IAM wired in `task-api.ts`. The resume path goes one step further than the precedent by also writing the orphan event on *failed* resume calls, because a failed resume strands a suspended VM awaiting a decision — a stronger liveness consequence than a failed stop of an already-cancelled task. The alternative (orchestrator-poll-only resume) preserves single-owner lifecycle purity but pays up to a full poll interval (~30 s) of latency on every approval, and the purity argument was already litigated and declined for cancel. `approve-task.ts` is deliberately minimal today (security-critical ownership comparison, Cedar finding #6); the resume call is therefore added *after* the transaction commits, cannot alter the decision outcome, and carries one conditional `lambda:ResumeMicrovm` grant — the same blast-radius trade the cancel handler accepted in review. -- **Timeout under freeze — the agent re-bases on the wall clock it already owns.** The agent's monotonic gate timer freezes while suspended, so resuming near the deadline is not enough: the frozen timer would still hold its remaining budget and fire the deny minutes *after* the user-visible window — colliding with the approval row's TTL (`created_at + timeout_s + 120s`) and triggering the "row reaped → stranded" fallback on a healthy gate. Instead, the gate expires at **`min(monotonic budget, created_at + timeout_s)`**, evaluated on each poll iteration and on `/resume`. This is not a new principle: Cedar decision #6 is already "min wins" for timeouts, the wall-clock deadline is already durable in the approval row the agent itself writes (`created_at` is in the agent's own clock domain — no skew), and §13.12's late-approval race fix already establishes that the durable row is authoritative over the agent's local timer. Deny authority stays agent-side (the conditional `TIMED_OUT` write + ConsistentRead re-read race protection is untouched); the orchestrator's resume at `deadline − margin` is purely the wake-up mechanism, with no correctness role. +- **Timeout under freeze — the agent re-bases on the wall clock it already owns.** The agent's monotonic gate timer freezes while suspended, so resuming near the deadline is not enough: the frozen timer would still hold its remaining budget and fire the deny minutes *after* the user-visible window — colliding with the approval row's TTL (`created_at + timeout_s + 120s`) and triggering the "row reaped → stranded" fallback on a healthy gate. Instead, the gate expires at **`min(monotonic budget, created_at + timeout_s)`**, evaluated on each poll iteration and on `/resume`. This is not a new principle: Cedar decision #6 is already "min wins" for timeouts, the wall-clock deadline is already durable in the approval row the agent itself writes (`created_at` is recorded by the agent; clock changes across suspend still need testing), and §13.12's late-approval race fix already establishes that the durable row is authoritative over the agent's local timer. Deny authority stays agent-side (the conditional `TIMED_OUT` write + ConsistentRead re-read race protection is untouched); the orchestrator's resume at `deadline − margin` is purely the wake-up mechanism, with no correctness role. - **Backstops, not mechanisms.** `maximumDurationInSeconds` (mandatory on every `RunMicrovm`, pinned at 28 800 s — see sub-decision 1) is the substrate kill switch bounding running **and** suspended time; the orchestrator's finalization `terminate-microvm` is the active cleanup path; the stranded-approval reconciler retains its role for orphaned waits. No `idlePolicy`-based bound is used in any phase — see sub-decision 1's omit-`idlePolicy` invariant. **The active terminate is still mandatory on the SUCCESS path, and P2 sharpened why.** P1 concluded flatly that "nothing self-terminates": a hook-less MicroVM reached `RUNNING` in 12 s and stayed there with no `stateReason` through every checkpoint. P2 refuted that *for the failure path only* — with `run: ENABLED`, a run hook that answers 4xx makes the **service** terminate the VM within ~12 s, `stateReason: "Run lifecycle hook returned HTTP status 400. Please check your hook endpoint and application logs for more details."`, after which `suspend-microvm` correctly refuses it. That is a real improvement in cost posture and a direct benefit of declaring hooks (see also the failure-path row in the phasing table, sub-decision 3). @@ -138,22 +138,22 @@ The handshake must respect the existing approval mechanics: the agent **discover It does **not** relieve the orchestrator of anything, because the two cases are disjoint. The service reaps a hook *result* it did not like; it has no view of the guest once the hook returned 200. So a task that starts normally — the overwhelming majority — has no service-side reaper at all, and a VM whose pipeline finished, crashed after `/run`, or hung is reaped by nobody but `TerminateMicrovm`. A leaked handle therefore remains a cost incident that bills until the 8 h cap; only the "the guest rejected its own payload" corner now cleans itself up. - **Concurrency slot stays held** during suspend. Cedar decision #7's rationale ("container alive, consuming memory") weakens under suspend, and the harder replacement rationale — "AWS counts `SUSPENDED` MicroVMs toward the account memory quota, so releasing ABCA's slot would not free real capacity" — is **undischarged**: the suspended VM stayed in `list-microvms` at every checkpoint, but that only proves *listed*. `L-CD1C0CC4` (1024 GB, account-scoped) exposes no `UsageMetric`, `AWS/Usage` carries only `CallCount` per API, and no MicroVM memory metric exists in any namespace, so consumption is **not observable safely** — proving it would need a large concurrent fleet. The conclusion (hold the slot) stands as the conservative choice, not as a verified fact. Size the arithmetic against the 32 GiB **peak** rather than the 8 GiB baseline: a busy fleet scales up, so peak is what actually competes for the account quota. -The agent's `/suspend` hook flushes progress events (durable writes before returning 200, within the 60 s hook budget); `/resume` reseeds CSPRNGs and refreshes cached credentials. +In P3, the agent's `/suspend` hook must acknowledge durable progress before returning 200 within its hook budget; `/resume` must refresh credentials and reseed application PRNG state from fresh OS entropy. These hooks are not implemented yet. A non-cryptographic PRNG such as Python `random` remains unsuitable for secrets even after reseeding; security-sensitive randomness must use an OS-backed cryptographic source. -**Amended (P2 review): CSPRNG reseeding is P3 scope, and the P2 exposure is negligible.** The original EARS requirement below implied a `/run`-time reseed had to land with P2; it did not, and shipping P2 with the requirement unmet-but-asserted was itself the defect. Measured exposure, which is what changed the scoping: +**Amended (P2 review): Application PRNG reseeding is P3 scope, and the P2 exposure is negligible.** The original EARS requirement below implied a `/run`-time reseed had to land with P2; it did not, and shipping P2 with the requirement unmet-but-asserted was itself the defect. Measured exposure, which is what changed the scoping: - The only consumer of a non-cryptographic PRNG in `agent/src` is `progress_writer.py`'s `getrandbits(80)`, used for the random half of a **ULID**. `os.urandom` / `secrets` are not seeded from the snapshot at all, and nothing in the agent derives a key, token, or nonce from `random`. - That ULID is a DynamoDB **sort key under a `task_id` partition**. A collision therefore needs two events in the *same task* at the *same millisecond* with the same 80 random bits — and identical PRNG state across two MicroVMs restored from one snapshot does not produce that, because the events are in different partitions. The worst case is a duplicate progress event within one task, not a security boundary. -- There is no credential exposure from the snapshot's PRNG state: build-role credentials are kept out of the snapshot structurally (the build hooks make zero AWS calls, so `boto3.DEFAULT_SESSION` is never populated — see `_aws_silent_log`), and per-task credentials arrive via `platform_config.agent_session_role_arn` at `/run`. +- Credential safety is a separate concern from the application PRNG. Build hooks avoid initializing AWS clients and their credential caches; their credential-environment warning does not prove that every possible image is secret-free. `platform_config.agent_session_role_arn` delivers a role identifier, not credentials. The running agent obtains credentials through its runtime provider and assumes the task-scoped role. -So the reseed moves to P3 alongside `/suspend` + `/resume`, where a *resumed* VM — which really does continue with the exact PRNG state it was frozen with, repeatedly — makes it load-bearing rather than theoretical. P3 must reseed on both `/run` and `/resume`, and must not treat the ULID as the only consumer: any future use of `random` for anything security-relevant needs the reseed in place first. +So the reseed moves to P3 alongside `/suspend` + `/resume`, where a *resumed* VM — which really does continue with the exact PRNG state it was frozen with, repeatedly — makes it load-bearing rather than theoretical. P3 must reseed on both `/run` and `/resume`, and must not treat the ULID as the only consumer: security-sensitive values must continue to use `secrets` or an equivalent cryptographic source, not `random`. Normative requirements (EARS): - While a `lambda-microvm` task is in `AWAITING_APPROVAL` and the gate's remaining window exceeds the configured grace period plus resume overhead, the orchestrator shall call `suspendSession` after the grace period. - When the approve or deny Lambda commits a decision for a `lambda-microvm` task, the Lambda shall load the MicroVM handle from `compute_metadata` and call `resumeSession` best-effort. - If the inline resume fails, then the Lambda shall record a resume-orphan task event and shall still return the decision outcome. -- While a decision row exists and the MicroVM remains `SUSPENDED`, the orchestrator shall retry `resumeSession`. +- While the current approval row is APPROVED or DENIED and the MicroVM remains `SUSPENDED`, the orchestrator shall retry `resumeSession`. A PENDING row alone is not a wake-up condition. - While a `lambda-microvm` task waits on an approval gate, the agent shall evaluate gate expiry as the earlier of its monotonic budget and the row's wall-clock deadline (`created_at + timeout_s`), on each poll iteration and on `/resume`. - If gate expiry is reached without a decision, then the agent shall deny. - If no decision arrives by the gate's wall-clock deadline minus the resume margin, then the orchestrator shall resume the MicroVM so the agent can evaluate expiry and fire the deny agent-side. @@ -175,10 +175,10 @@ The phasing is therefore: | `/ready` | **P1** (construct enables `hooks.microvmImageHooks.ready`) | **P1** | MANDATORY, not a quality nicety — see above. A 200 proves uvicorn is bound and `server` imported cleanly (pulling in `pipeline` → `runner` → the policy engine), so a missing policy file fails the BUILD instead of the first task. **Since P2-F5 it also WARMS the snapshot** — the hook's 200 is what the service waits for before capturing the snapshot, making this the only place a warm page can be created, and the 225 MiB `claude` binary was cold in it (see the P2-F5 correction below). A required warm-up failure answers 503, so a snapshot that cannot exec the agent's own CLI fails the image build instead of every task. Still makes ZERO AWS calls, logging included (a `--version` exec is neither an AWS call nor a network call). | | `/run` | **P1** (construct sets `hooks.microvmHooks.run`) | **P1** | The payload-delivery channel. Must be served in P1 because `/ready` forces hooks to exist at all, and a hook-less image cannot accept `runHookPayload`. Since P2 it is also the **platform-configuration** channel (see "Platform configuration delivery" below). | | `/validate` | **P2** (construct sets `hooks.microvmImageHooks.validate`) | **P2** | An **image** (build-time) hook, and a **shallow self-check only**: server alive, hook routes registered, interpreter + contract sanity. It runs under the BUILD role, which deliberately holds no Bedrock / Secrets Manager / DynamoDB grants, so it must make **zero AWS API calls** and must not touch credential resolution — the "deeper warm-up assertions (Bedrock reachability, Memory access, tool availability)" this ADR originally assigned here are **not implementable**: every one of them would `AccessDenied` and fail every image build. They belong to the first task's own error handling. 200 when the checks pass, 503 while still initialising. | -| `/terminate` | **P2** (construct sets `hooks.microvmHooks.terminate`) | **P2** | Best-effort final flush: a final structured log line, then 200 — always, inside the hook budget, even with nothing running. It must **not** write terminal task status (the orchestrator finalizes the task and *then* calls `TerminateMicrovm`, so a terminate hook that wrote a status would race that finalization and could clobber the real outcome) and must not join the pipeline thread. There is nothing buffered to flush: `_ProgressWriter` does a synchronous `put_item` per event, so durability is per-write. "Always 200" also covers the BODY: the handler reads the raw request rather than a typed model, because a typed body is validated before the handler runs and would answer 422 to malformed JSON — a reported hook failure on a successful teardown. Safe to declare because `TerminateMicrovm` removes the VM with or without in-guest cooperation. **Correction (P2-F8):** the service sends `microvmId: ""` on this hook, unlike `/run` where it is populated, so an empty id is expected-normal and this hook cannot join the guest record to the control-plane one — `/run`'s accepted line carries that correlation instead. | +| `/terminate` | **P2** (construct sets `hooks.microvmHooks.terminate`) | **P2** | Best-effort final flush: a final structured log line, then 200 — always, inside the hook budget, even with nothing running. It must **not** write terminal task status (the orchestrator finalizes the task and *then* calls `TerminateMicrovm`, so a terminate hook that wrote a status would race that finalization and could clobber the real outcome) and must not join the pipeline thread. `ProgressWriter` writes synchronously but drops failed events, so this hook cannot guarantee that every progress write succeeded. P3 needs an acknowledged durability barrier for suspension. "Always 200" also covers the BODY: the handler reads the raw request rather than a typed model, because a typed body is validated before the handler runs and would answer 422 to malformed JSON — a reported hook failure on a successful teardown. Safe to declare because `TerminateMicrovm` removes the VM with or without in-guest cooperation. **Correction (P2-F8):** the service sends `microvmId: ""` on this hook, unlike `/run` where it is populated, so an empty id is expected-normal and this hook cannot join the guest record to the control-plane one — `/run`'s accepted line carries that correlation instead. | | `/suspend`, `/resume` | **P3** | **P3** | Declaring a runtime hook the agent does not serve fails the corresponding lifecycle transition, so each is declared only in the phase that implements it. P1 termination is the orchestrator's `TerminateMicrovm`, which needs no in-guest cooperation. | -Consequence to state plainly, replacing the original "a P1-built MicroVM image is not runnable end to end": **a P1 image is creatable, launchable and payload-deliverable, but carries no smoke-parity guarantee.** P1 delivers the strategy, the construct, the roles/buckets/connectors, the image resource, the packaging script, and the `/ready` + `/run` endpoints — so a `lambda-microvm` task can start a MicroVM and hand it a payload. What P1 has **not** established is anything P2 owns: AgentCore Memory grants and `MEMORY_ID` delivery, the agent's non-secret env parity inside the snapshot, egress specifics from a running MicroVM, and heartbeat/progress behaviour end to end. No clone → change → PR run has happened on this substrate. P2 ("smoke parity") is the phase that closes that gap. The construct and the packaging script both surface exactly this at synth/run time (`abca:microvm-image-p1-smoke-unverified`) so an operator cannot mistake a launchable substrate for a verified one. +Consequence to state plainly, replacing the original "a P1-built MicroVM image is not runnable end to end": **a P1 image is creatable, launchable and payload-deliverable, but carries no smoke-parity guarantee.** P1 delivers the strategy, the construct, the roles/buckets/connectors, the image resource, the packaging script, and the `/ready` + `/run` endpoints — so a `lambda-microvm` task can start a MicroVM and hand it a payload. What P1 has **not** established is anything P2 owns: AgentCore Memory grants and `MEMORY_ID` delivery, the agent's non-secret env parity inside the snapshot, egress specifics from a running MicroVM, and heartbeat/progress behaviour end to end. At the P1 handoff, no clone → change → PR run had happened. P2 subsequently demonstrated that path with an IAM workaround; clean unattended verification is still pending. The construct and the packaging script both surface exactly this at synth/run time (`abca:microvm-image-p1-smoke-unverified`) so an operator cannot mistake a launchable substrate for a verified one. **The `AWS::Lambda::MicrovmImage` L1 enforces the API's enums (P2-F2, live 2026-08-06).** This closes the one item P1 left explicitly open, and it closes it against the construct's own stated reasoning. CloudFormation's generated types make `cpuConfigurations[].architecture` and all four `hooks.*` fields plain strings and document no allowed values, from which P1 concluded that the CloudFormation surface takes a *hook path* while the API takes an `ENABLED`/`DISABLED` flag, and that both were correct for their own surface. CloudFormation refused the change set at **early validation** — the stack was never touched, so there was no rollback and no runtime symptom to trace back — on five values: @@ -219,11 +219,11 @@ The agent's reader is deliberately more permissive than this contract: it also a **Platform configuration delivery (P2): payload-sourced, allowlisted, fail-closed.** The other two backends hand the agent its non-secret platform env at launch — AgentCore Runtime env vars, ECS container overrides — and there is no equivalent on this substrate: a MicroVM starts from a **snapshot**, so its process environment is whatever was frozen at *image build* time and is then replayed by every MicroVM launched from that image version. Baking the deployment's identifiers into the snapshot would make them **version-frozen**: a redeploy that renames a table, adds a bucket or rotates the session role would leave every existing image version describing a deployment that no longer exists, and the drift would surface as a task-time `ResourceNotFound` rather than a deploy-time error. So the values travel with the task instead: `platform_config` is a SIBLING of the payload — beside `agent_payload` in the inline branch, beside `agent_payload_s3_uri` in the pointer branch, and merged in beside the payload's own fields inside the S3 object (the canonical shapes above give each one exactly) — whose snake_case keys the agent installs into `os.environ` as their UPPER_SNAKE equivalents. A payload value therefore **wins** over any pre-existing/image value — the orchestrator is describing the live deployment, the snapshot is describing a past one. `platform_config` carries **non-secret identifiers only** (table and bucket names, secret ARNs, the session-role ARN); secrets are still fetched at `/run` time from Secrets Manager using those ARNs, so the snapshot-must-stay-secret-free requirement above is untouched. Per-task fields — `memory_id` and friends — stay inside `agent_payload`: `platform_config` configures the *process*, `agent_payload` describes the *task*. -Two rules make it safe. First, **the allowlist fails closed**: the agent installs a fixed set of keys and *rejects the entire run* (HTTP 400, nothing spawned, not one key installed) if the block carries anything else. These values become environment variables of the process that spawns the agent's tool subprocesses, so an unrecognised key is an attempt to set an arbitrary variable in the agent (`AWS_ENDPOINT_URL`, `LD_PRELOAD`, `PATH`, …) — an injection attempt, not a forward-compatibility gap, which is why unknown keys are refused rather than filtered out. Second, **installation happens before any credential or pipeline initialisation** on the hook path: the very next step reads `GITHUB_TOKEN_SECRET_ARN` to resolve the GitHub token and `AGENT_SESSION_ROLE_ARN` to scope the task's credentials, so installing later would silently resolve the whole task against the snapshot's frozen env. The one call that must precede installation is the S3 payload fetch (the config is *inside* the fetched object), which therefore runs on the ambient compute role via the attributed platform client — and it is the ONLY one: the same rule covers **logging**, so every `/run` log line before the install is stdout-only. The CloudWatch writer would otherwise resolve credentials and pin a boto3 default session (region included) off whatever a snapshot happened to bake, which is the build-hook defect one phase later. Nothing is lost — in the intended deployment there is no baked `LOG_GROUP_NAME`, so those lines would have gone to stdout anyway, and the reason for every pre-install rejection also travels in the structured 4xx/5xx body the service surfaces. A **required subset** (task table, task-events table, GitHub token secret ARN, session-role ARN) is rejected as `…_INCOMPLETE` when missing or blank — a distinct wire code from the `…_INVALID` allowlist rejection, because the remedies differ (deployment wiring vs. producer bug). A `/run` envelope with *no* `platform_config` at all is still accepted, loudly warned: the image snapshot and the orchestrator Lambda deploy on independent cadences, and a new image must not require a same-instant orchestrator. The key set is a cross-package contract in `contracts/constants.json` (`microvm_platform_config`), consumed by the agent's `/run` hook and produced by the orchestrator, with shape and required-subset invariants enforced by `scripts/check-constants-sync.ts`. +Two rules make it safe. First, **the allowlist fails closed**: the agent installs a fixed set of keys and *rejects the entire run* (HTTP 400, nothing spawned, not one key installed) if the block carries anything else. These values become environment variables of the process that spawns the agent's tool subprocesses, so an unrecognised key is an attempt to set an arbitrary variable in the agent (`AWS_ENDPOINT_URL`, `LD_PRELOAD`, `PATH`, …) — an injection attempt, not a forward-compatibility gap, which is why unknown keys are refused rather than filtered out. Second, **installation happens before any credential or pipeline initialisation** on the hook path: the very next step reads `GITHUB_TOKEN_SECRET_ARN` to resolve the GitHub token and `AGENT_SESSION_ROLE_ARN` to scope the task's credentials, so installing later would silently resolve the whole task against the snapshot's frozen env. The one call that must precede installation is the S3 payload fetch (the config is *inside* the fetched object), which therefore runs on the ambient compute role via the attributed platform client — and it is the ONLY one: the same rule covers **logging**, so every `/run` log line before the install is stdout-only. The CloudWatch writer would otherwise resolve credentials and cache credentials in a boto3 default session (environment-derived region is re-read for each new client) off whatever a snapshot happened to bake, which is the build-hook defect one phase later. Nothing is lost — in the intended deployment there is no baked `LOG_GROUP_NAME`, so those lines would have gone to stdout anyway, and the reason for every pre-install rejection also travels in the structured 4xx/5xx body the service surfaces. A **required subset** (task table, task-events table, GitHub token secret ARN, session-role ARN) is rejected as `…_INCOMPLETE` when missing or blank — a distinct wire code from the `…_INVALID` allowlist rejection, because the remedies differ (deployment wiring vs. producer bug). A `/run` envelope with *no* `platform_config` is accepted with a warning only when the effective environment already provides every required identifier; otherwise it is rejected as incomplete. The image and orchestrator can deploy on independent cadences without accepting an unusable environment. The key set is a cross-package contract in `contracts/constants.json` (`microvm_platform_config`), consumed by the agent's `/run` hook and produced by the orchestrator, with shape and required-subset invariants enforced by `scripts/check-constants-sync.ts`. **No orchestrator→agent HTTP path exists in P1–P3**: payload arrives through the `/run` hook, all agent work is outbound, and therefore **no JWE auth tokens are minted at all** — token minting (and its ≤ 60 min TTL refresh problem) is deferred until a real consumer exists (e.g. operator shell access, [#391](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/391)). The `endpoint` stays in the `SessionHandle` because it is genuinely per-session state that becomes load-bearing the day such a consumer appears. But note the service does not agree by default: omitting `ingressNetworkConnectors` on `RunMicrovm` attaches a **public** `HTTP_INGRESS` connector, so the strategy passes the Lambda-managed `NO_INGRESS` connector explicitly on every launch (see sub-decision 4's security table). -**Constraint accepted:** the configured **baseline is 8 GiB RAM / 4 vCPU** and the service scales vertically to a **32 GiB / 16 vCPU peak** on its own, with 32 GB of disk. So capacity is baseline-priced with 4× burst headroom — good for the bursty compile-and-test shape of an agent task — but the SUSTAINED ceiling is still 32 GiB, so repos that motivated the 120 GB ECS sizing stay on `ecs`. What the construct configures (and validates) is the baseline; the peak is not something a deployment asks for. +**Constraint accepted:** the configured **baseline is 8 GiB RAM / 4 vCPU** and the service scales vertically to a **32 GiB / 16 vCPU peak** on its own, with 32 GB of disk. So baseline capacity and additional burst usage are billed separately — good for the bursty compile-and-test shape of an agent task — but the SUSTAINED ceiling is still 32 GiB, so repos that motivated the 120 GB ECS sizing stay on `ecs`. What the construct configures (and validates) is the baseline; the peak is not something a deployment asks for. Normative requirements (EARS). **Each requirement's own `(Pn)` tag is authoritative**; there is no blanket phase for the list. The tags are per-requirement because the original list *was* split P1/P2 on the assumption that a hook could be declared in one phase and served in a later one — which the service does not permit (see the phasing table above), so the hook-serving requirements collapsed into P1 while the P2 items below arrived with the P2 hooks and `platform_config`: @@ -347,7 +347,7 @@ Two networking facts the construct has to encode, both established live: | IAM condition keys on the compute-role trust **and** on the `iam:PassRole` grants that hand it over | Trust pinned with `aws:SourceAccount`; `PassRole` under the allowlisted bootstrap statement | Trust pinned per-service; `PassRole` under the allowlisted bootstrap statement | **Neither is possible.** All three MicroVM-facing roles trust the bare `lambda.amazonaws.com` with no `aws:SourceAccount`/`aws:SourceArn`, **and** both `iam:PassRole` grants (orchestrator → execution role at `RunMicrovm`; CloudFormation → build role at `CreateMicrovmImage`) carry no `iam:PassedToService` — the service presents no usable value for any of those keys, and each condition is a hard blocker while present (live-verified, blocking, four times across two runs) | **Real, evidenced gap that does not close from our side, and it is wider than the trust policy alone.** `lambda.amazonaws.com` is shared with every other Lambda feature, so neither the account pin nor the passed-to-service pin is available on this path. Compensated per role (table in sub-decision 4): the **execution** role is passable by the **orchestrator only** (at `RunMicrovm`), restricted to its **exact ARN**; the **build** and **connector-operator** roles are passable by the CloudFormation deployment role under a new **conditional, per-backend, name-prefix-scoped** statement (`MicrovmPassRoles`, bootstrap ≥ 1.6.0) that deliberately excludes the execution role. The shared allowlisted `IAMPassRole` (`role/backgroundagent-dev-*`) is left intact to avoid widening the grant for ~30 other roles, so while it technically matches the execution role, only the orchestrator actively reaches for it. Resources are account-scoped by ARN apart from two justified `Resource: \'*\'` statements (`ec2:DescribeAvailabilityZones`; the operator role\'s ENI management — both carry cdk-nag IAM5 suppressions). No `iam:*`, no cross-account trust. Revisit if AWS ever documents the values the service presents; CloudTrail records no `lambda-microvms` events, so they cannot be read from logs | | Per-task observability writes | Runtime writes to the vended APPLICATION_LOGS group | Task role writes to the task log group | Execution role writes to the SAME APPLICATION_LOGS group, granted against the group `platform_config` names (P2-F4) | None — but only after P2-F4: the name was delivered a phase before the grant, so the agent attempted the write and every per-task line (and `METRICS_REPORT`) was `AccessDenied`, degrading silently to guest stdout | | Session isolation | MicroVM | Task-level | MicroVM (Firecracker) | None (≥ ECS) | -| State reuse | None | None | Snapshot shared across MicroVMs | New surface: CSPRNG reseed + credential refresh on `/run`/`/resume` — **P3 scope** (P2 exposure measured as negligible: sole `random` consumer is a ULID sort key under a `task_id` partition; `os.urandom`/`secrets` unaffected; no credential derives from `random`). Credential refresh IS in P2: per-task credentials arrive via `platform_config` at `/run` | +| State reuse | None | None | Snapshot shared across MicroVMs | New surface: application PRNG reseed + credential refresh on `/run`/`/resume` — **P3 scope** (P2 exposure measured as negligible: sole `random` consumer is a ULID sort key under a `task_id` partition; `os.urandom`/`secrets` unaffected; no credential derives from `random`). Credential refresh IS in P2: per-task credentials arrive via `platform_config` at `/run` | | Workload-token injection | Yes (Runtime-coupled) | No (env-var posture) | No (env-var posture) | Shared with ECS; deferred to [#249](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/249)/ADR-016 | | Operator shell access | No | No | Not enabled (`SHELL_INGRESS` omitted; candidate for [#391](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/391)) | None by default | | Auth-token minting | n/a | n/a | `CreateMicrovmAuthToken` granted to no role in any phase | Verified static-only: any principal holding the action *can* mint a working JWE, including against a `SUSPENDED` MicroVM, so the posture rests entirely on the grant being absent | @@ -389,7 +389,7 @@ Lambda MicroVMs launched in 5 regions (us-east-1/2, us-west-2, eu-west-1, ap-nor - (!) **Service defaults are not the desired posture.** Two live-caught cases (public `HTTP_INGRESS` by default; `/ready` mandatory) mean an omitted field on this backend does not mean "off" — it can mean "the service picks, and it picks wider than we want". Every new `RunMicrovm` / `CreateMicrovmImage` field should be assumed to have an opinionated default until checked. - (!) **Nothing self-terminates on the paths that matter** — superseding P1's unqualified version of this bullet. With `run: ENABLED` the service DOES reap a VM whose run hook returns 4xx (~12 s, `stateReason: "Run lifecycle hook returned HTTP status 400."`, live-verified), so a guest that rejects its own payload cleans itself up. That is the only self-cleaning case: the service reaps a hook *result*, and once `/run` has answered 200 it has no view of the guest. A VM whose task finished, crashed after `/run`, or hung stays `RUNNING` and billing until the 8 h cap, so the orchestrator's `TerminateMicrovm` on finalize remains the only cleanup for normal operation and a leaked handle is still a cost incident. - (!) **Snapshot uniqueness.** Shared memory snapshots mean every MicroVM restored from one image starts with identical PRNG state. **Re-scoped to P3 in the P2 review, with the exposure measured rather than assumed** (see the amendment under sub-decision 2): the sole `random` consumer in `agent/src` is a ULID sort key under a `task_id` partition, `os.urandom`/`secrets` are unaffected, and no credential or token derives from `random` — so the P2 exposure is a possible duplicate progress event, not a security boundary. It becomes load-bearing at P3, where a *resumed* VM continues from frozen state repeatedly; the reseed must land on `/run` and `/resume` together with those hooks. Asserting the requirement while leaving it unimplemented was the real defect, and this amendment is the fix. -- (−) **The agent stack template is at 98.6 % of CloudFormation's 1 MB limit** (985,886 bytes) and 486 of 500 resources with a MicroVM image configured — ~14 KB of headroom, i.e. roughly one more construct, and down from 98.4 % / ~16 KB one run earlier. Not caused by this backend (the MicroVM construct is ~6 KB of it) but reached by it, and it will block deploys for reasons that have nothing to do with MicroVMs. Tracked in [#735](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/735); the candidate remedies are `suppressTemplateIndentation` and a stack split. +- (−) **Root-stack resource headroom needs ongoing measurement.** The historical 985,886-byte / 486-resource measurement predates [#854](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/854), which reduced template size and resource count. The 2026-09-13 [offline review](/sample-autonomous-cloud-coding-agents/architecture/645-p3-readiness-review) measures 472 root resources with a managed MicroVM image, and 489 with gateway and vault enabled in a scratch probe. Nesting can reduce that fullest root count to 474 in the prototype. These are unbundled synth measurements, not deployment validation; the vault combination remains guarded on main ([#857](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/857)). - (!) **Snapshot WARMTH is a first-class property, not an optimisation.** A snapshot inherits only the pages something touched before it was captured, so a large lazily-loaded artifact — the 225 MiB `claude` binary, and anything similar added later — pays its first-touch cost on the *first task* instead of at container start. That cost failed every task at turn 0 in the P2 smoke run (P2-F5). Anything heavyweight added to the image must be exec'd in `/ready`, and any timeout guarding a first touch must be sized for a cold page fault rather than for the work itself. - (!) **Regional availability (5 regions at launch, expanding)** — enforced in layers (synth-time static check, onboarding + doctor live probes, orchestration-time classification; see sub-decision 4). The static CDK constant is the one piece that rots as AWS expands; its update path and context-flag escape hatch are deliberate. - (!) **Workload-token injection delta persists** (shared with the ECS backend) until [#249](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/249)/ADR-016 land; document it in the security bar comparison rather than blocking on it. Memory and Gateway are explicitly *not* deltas — both are standalone services consumed via IAM from any substrate. @@ -399,7 +399,7 @@ Lambda MicroVMs launched in 5 regions (us-east-1/2, us-west-2, eu-west-1, ap-nor **P1 (start/poll/stop — no suspend):** - Unit tests for the strategy: start/poll/stop mapping (including `SessionStatus` `'suspended'` reported mechanically, without task-state interpretation), payload-size branching (inline vs S3 pointer, at the exact 4 096/4 097-byte boundary), the image-identifier-must-be-an-ARN guard, the explicit `NO_INGRESS` argument (including the blank-env-var fallback, which must never omit the field), error classification (`ServiceQuotaExceededException`, `ThrottlingException`, `ResourceNotFoundException`, regional-unavailability), the omit-`idlePolicy` invariant, and `maximumDurationInSeconds` fixed at 28 800. -- Agent tests: `/ready` returns 200 once the server is up and starts nothing; `/run` accepts both envelope shapes (inline and S3 pointer), starts the pipeline asynchronously through the same mapper `/invocations` uses, returns before the pipeline finishes, and rejects every unusable envelope with a named code before spawning; `/suspend` and `/resume` are NOT served (`/validate` and `/terminate` joined the served set in P2). +- Agent tests: `/ready` returns 200 after required local warm-up succeeds, without starting a task or calling AWS; `/run` accepts both envelope shapes (inline and S3 pointer), starts the pipeline asynchronously through the same mapper `/invocations` uses, returns before the pipeline finishes, and rejects every unusable envelope with a named code before spawning; `/suspend` and `/resume` are NOT served (`/validate` and `/terminate` joined the served set in P2). - Orchestrator tests: substrate-terminal + non-terminal task status → failed classification; `suspended` + non-`AWAITING_APPROVAL` status → anomaly event, no fail-fast; `compute_metadata` persisted with `microvmId`/`endpoint` after `startSession`. - CDK assertions: MicroVM resources present only when `ComputeTypes` includes the backend; synth failure for unsupported regions (plus the context-flag escape hatch); memory size validated against the accepted list at synth; the connector operator role and its trust; two connectors with the build-only one carrying port 80 and the runtime one not; the `/ready` + `/run` hook declaration and the absence of the others; IAM actions scoped as specified (orchestrator lifecycle set; no `CreateMicrovmAuthToken` anywhere); payload-bucket grants (execution role read-only); backend cost-allocation tags; types-sync check covers the widened `ComputeType`. - CLI tests: onboarding rejection with remedy when the availability probe fails; doctor check present when a blueprint selects the backend. @@ -408,7 +408,7 @@ Lambda MicroVMs launched in 5 regions (us-east-1/2, us-west-2, eu-west-1, ap-nor **P2 (smoke parity):** - Agent tests: `/validate` returns 200 with its individual check results, 503 while initialising, reports a missing hook route / unsupported interpreter, starts nothing, and makes **zero AWS calls even with `LOG_GROUP_NAME` set** (asserted by poisoning the boto3 and CloudWatch-writer seams — the same assertion covers `/ready`); `/terminate` returns 200 with no body at all, with a malformed / non-object / wrong-content-type / whitespace-only body, when the body read itself fails, with a pipeline still running (without joining it), and when its own best-effort step raises — and never calls `task_state.write_terminal`; a structural assertion that the route carries no typed body param keeps the 422 from being reintroduced. -- `platform_config` tests: the allowlist and required subset are read from `contracts/constants.json` (the wire key set is additionally asserted literally, as the agent-side tripwire on a contract edit); an unknown key rejects the whole block with nothing installed; a non-object block and a non-string value are rejected; a blank/`null` optional value is skipped without clobbering an image value while a blank required value is rejected; a payload value beats a pre-existing env value; installation is observed to happen before the GitHub-token resolver runs and before any pipeline thread exists; the config is picked up from the inline envelope, from beside the S3 pointer, and from inside the fetched object (inner wins); an envelope with no `platform_config` is still accepted with a warning. +- `platform_config` tests: the allowlist and required subset are read from `contracts/constants.json` (the wire key set is additionally asserted literally, as the agent-side tripwire on a contract edit); an unknown key rejects the whole block with nothing installed; a non-object block and a non-string value are rejected; a blank/`null` optional value is skipped without clobbering an image value while a blank required value is rejected; a payload value beats a pre-existing env value; installation is observed to happen before the GitHub-token resolver runs and before any pipeline thread exists; the config is picked up from the inline envelope, from beside the S3 pointer, and from inside the fetched object (inner wins); an envelope with no `platform_config` is accepted with a warning only if the effective environment already supplies every required identifier; otherwise it is rejected. - Snapshot credential hygiene: a subprocess probe asserts that importing `server` and serving `/ready` + `/validate` imports neither `boto3` nor `botocore`, caches no `aws_session` session, and spawns no CloudWatch writer thread — the property that keeps a build-role credential chain and the build-time region out of the snapshot. - `/run` pre-install silence: with a **baked `LOG_GROUP_NAME`** (the hostile case — without it the assertions pass vacuously) every AWS/credential seam (`boto3.client`/`Session`, the `aws_session` factories, `_debug_cw`/`_warn_cw`) is armed to raise until the install succeeds. Asserted on the accepted path, on all three rejection paths (bad envelope, `platform_config` invalid, `platform_config` incomplete) and on the failed-fetch 500 — where the seams stay armed for the whole request, because a rejected run installed nothing and so earns no AWS call. The permitted exception is asserted POSITIVELY: exactly one client is built pre-install, for `s3`, through the attributed factory. - `/ready` warm-up tests (P2-F5): the hook exec's each configured binary exactly once with a generous timeout; `claude` is the only REQUIRED entry; a timeout, a missing binary, a non-zero exit and an unexpected `OSError` each produce **503 with the reason logged to stdout** rather than a 200 or a 500; a best-effort failure still reports ready; the warm-up makes zero AWS calls with `LOG_GROUP_NAME` baked. Plus the backstop half: the `claude --version` probe's bound is asserted to be ≥ 60 s and to be applied to the *exec* rather than to the PATH lookup, and a missing CLI warns instead of raising. diff --git a/docs/src/content/docs/getting-started/Deployment-guide.md b/docs/src/content/docs/getting-started/Deployment-guide.md index 12e753f41..50826dd95 100644 --- a/docs/src/content/docs/getting-started/Deployment-guide.md +++ b/docs/src/content/docs/getting-started/Deployment-guide.md @@ -8,24 +8,24 @@ This guide covers deploying ABCA into an AWS account, including compute backend ## Architecture overview -ABCA deploys as a **single CDK stack** (`backgroundagent-dev`) containing all platform resources. The stack uses a `ComputeStrategy` interface to support three compute backends within the same stack: +ABCA deploys through a **root CDK stack** (`backgroundagent-dev`) and nested stacks for parts of the platform, including registry infrastructure. A `ComputeStrategy` interface supports three compute backends: | Aspect | AgentCore (default) | ECS Fargate (opt-in) | Lambda MicroVMs (experimental) | |--------|--------------------|--------------------|--------------------| | **Compute** | Bedrock AgentCore Runtime (Firecracker MicroVMs) | ECS Fargate containers | AWS Lambda MicroVMs | -| **Resources** | 2 vCPU, 8 GB RAM, 2 GB max image size | 2 vCPU, 4 GB RAM | 8 GB baseline / 32 GB peak memory | +| **Resources** | 2 vCPU, 8 GB RAM, 2 GB max image size | Build: 4 vCPU / 16 GiB; planning: 2 vCPU / 8 GiB (configurable) | 8 GiB baseline / 32 GiB peak memory | | **Orchestration** | Durable Lambda (checkpoint/replay) | Same durable Lambda via `ComputeStrategy` | Same durable Lambda via `ComputeStrategy` | | **Agent mode** | FastAPI server (HTTP invocation) | Batch (run-to-completion) | FastAPI server (lifecycle hooks) | | **Startup** | ~10s (warm MicroVM) | ~60-180s (Fargate cold start) | ~6s to `RUNNING` (live-measured) | | **Max duration** | 8 hours (AgentCore service limit) | 9 hours (orchestrator `executionTimeout`) | 8 hours (`maximumDurationInSeconds`) | -All backends are orchestrated by the same durable Lambda function. The `ComputeStrategy` interface abstracts `startSession()`, `pollSession()`, and `stopSession()` -- the ECS strategy calls `ecs:RunTask` / `ecs:DescribeTasks` / `ecs:StopTask` directly from the Lambda. No Step Functions are used. +All backends are orchestrated by the same durable Lambda function. The `ComputeStrategy` interface abstracts `startSession()`, `pollSession()`, and `stopSession()` -- the ECS strategy calls `ecs:RunTask` / `ecs:DescribeTasks` / `ecs:StopTask` directly from the Lambda. Task orchestration does not use Step Functions; other platform features may use them. -ECS Fargate is currently **opt-in** -- the `EcsAgentCluster` construct is present in the stack code but commented out. To enable it, uncomment the ECS blocks in `cdk/src/stacks/agent.ts`. +ECS Fargate is **opt-in**. Deploy with `--context compute_type=ecs`; the stack enables `EcsAgentCluster` from that context flag. ### Lambda MicroVMs backend (experimental) -> **Not for production.** `lambda-microvm` carries no smoke-parity guarantee for an unattended deployment. Keep production repositories on `agentcore` or `ecs`. Synth emits an unsuppressible warning to this effect whenever the backend is selected. Design detail: [COMPUTE.md](/sample-autonomous-cloud-coding-agents/architecture/compute) and [ADR-021](/sample-autonomous-cloud-coding-agents/architecture/adr-021-lambda-microvms-compute-backend). +> **Not for production.** `lambda-microvm` carries no smoke-parity guarantee for an unattended deployment. Keep production repositories on `agentcore` or `ecs`. Synth emits a verification warning whenever a MicroVM image is configured; selecting the backend without an image emits a separate setup warning. Design detail: [COMPUTE.md](/sample-autonomous-cloud-coding-agents/architecture/compute) and [ADR-021](/sample-autonomous-cloud-coding-agents/architecture/adr-021-lambda-microvms-compute-backend). Selecting it is a synth-time context flag: @@ -57,7 +57,7 @@ mise //cdk:deploy -- --context compute_type=lambda-microvm Operational notes specific to this backend: -- **Nothing self-terminates.** A MicroVM whose task finished, crashed, or hung stays `RUNNING` and billing until the 8-hour cap. The orchestrator calls `TerminateMicrovm` on finalize, and the heartbeat-staleness check catches a hung guest inside a healthy VM -- but a leaked handle is a cost incident. The one exception: the service reaps a VM whose `/run` hook returns 4xx (~12s). +- **Nothing self-terminates.** A MicroVM whose task finished, crashed, or hung stays `RUNNING` and billing until the 8-hour cap. The orchestrator calls `TerminateMicrovm` on finalize, and the heartbeat-staleness check detects loss of the in-guest heartbeat writer (a pipeline hang can leave that writer running) -- but a leaked handle is a cost incident. The one exception: the service reaps a VM whose `/run` hook returns 4xx (~12s). - **Logs** land in `/aws/lambda-microvms/`. Guest stdout goes there too, which is the fallback path when the agent cannot reach the application log group. - **Deployment identifiers are not baked into the image.** The snapshot carries no configuration; table names, secret ARNs, and the per-task session-role ARN arrive in the `/run` payload as a `platform_config` block. A version-skewed orchestrator that does not send it is refused rather than run with tenant scoping disabled. diff --git a/docs/src/content/docs/using/Linear-setup-guide.md b/docs/src/content/docs/using/Linear-setup-guide.md index d2302413e..9791d3243 100644 --- a/docs/src/content/docs/using/Linear-setup-guide.md +++ b/docs/src/content/docs/using/Linear-setup-guide.md @@ -42,7 +42,7 @@ When a workspace's authorization dies, ABCA records it on the registry row and p #### Not available with `compute_type=lambda-microvm` -The vault and the Lambda MicroVMs substrate cannot be enabled on the same stack. Together they synthesize 505 CloudFormation resources against a hard limit of 500 (MicroVM alone is 496, the vault alone 488), so `cdk deploy` refuses the combination by name at synth rather than failing partway through. Use the vault on the `agentcore` or `ecs` substrate; a MicroVM stack stays on Secrets Manager until the stack reclaims room. +The vault and the Lambda MicroVMs substrate remain gated from being enabled together ([#857](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/857)). The original 505-resource measurement predates stack-size reductions; the [current offline review](/sample-autonomous-cloud-coding-agents/architecture/645-p3-readiness-review) finds room in its tested configurations, but bundled feature-matrix and deployment verification are still required before removing the guard. Use the vault on `agentcore` or `ecs`; MicroVM deployments use Secrets Manager while this gate remains. #### One workload identity per stack diff --git a/docs/verification/645-nesting-probe-results.json b/docs/verification/645-nesting-probe-results.json new file mode 100644 index 000000000..e2eee3f56 --- /dev/null +++ b/docs/verification/645-nesting-probe-results.json @@ -0,0 +1,793 @@ +{ + "source_commit": "5e10038c7e28179b302ac4de78b709795aeba3ce", + "date": "2026-09-13", + "method": "Offline TypeScript loader transforms only; no production nesting edits; bundling disabled; vault guard disabled only in probe; parent-role mode also derives names from parent stack; resource cycle validation uses Template.fromStack", + "results": [ + { + "name": "default", + "mode": "baseline", + "templates": [ + { + "file": "backgroundagent-dev.template.json", + "resources": 454, + "bytes": 476779, + "parameters": 1, + "outputs": 41, + "names": [], + "roles": [] + }, + { + "file": "backgroundagentdevAgentRegistryStackFF9DF145.nested.template.json", + "resources": 19, + "bytes": 36474, + "parameters": 0, + "outputs": 2, + "names": [], + "roles": [] + }, + { + "file": "backgroundagentdevRegistryApiF3ADADE2.nested.template.json", + "resources": 35, + "bytes": 47684, + "parameters": 3, + "outputs": 3, + "names": [], + "roles": [] + } + ] + }, + { + "name": "microvm", + "mode": "baseline", + "templates": [ + { + "file": "backgroundagent-dev.template.json", + "resources": 471, + "bytes": 507859, + "parameters": 1, + "outputs": 48, + "names": [ + { + "type": "AWS::Lambda::NetworkConnector", + "name": "backgroundagent-dev-microvm-egress" + }, + { + "type": "AWS::Lambda::NetworkConnector", + "name": "backgroundagent-dev-microvm-build-egress" + } + ], + "roles": [ + "LambdaMicrovmComputeConnectorOperatorRoleE206977E", + "LambdaMicrovmComputeBuildRoleF0A13AC7", + "LambdaMicrovmComputeExecutionRoleAA0C4A0D" + ] + }, + { + "file": "backgroundagentdevAgentRegistryStackFF9DF145.nested.template.json", + "resources": 19, + "bytes": 36474, + "parameters": 0, + "outputs": 2, + "names": [], + "roles": [] + }, + { + "file": "backgroundagentdevRegistryApiF3ADADE2.nested.template.json", + "resources": 35, + "bytes": 47684, + "parameters": 3, + "outputs": 3, + "names": [], + "roles": [] + } + ] + }, + { + "name": "microvm-image", + "mode": "baseline", + "templates": [ + { + "file": "backgroundagent-dev.template.json", + "resources": 472, + "bytes": 510894, + "parameters": 1, + "outputs": 48, + "names": [ + { + "type": "AWS::Lambda::NetworkConnector", + "name": "backgroundagent-dev-microvm-egress" + }, + { + "type": "AWS::Lambda::NetworkConnector", + "name": "backgroundagent-dev-microvm-build-egress" + }, + { + "type": "AWS::Lambda::MicrovmImage", + "name": "backgroundagent-dev-abca-agent" + } + ], + "roles": [ + "LambdaMicrovmComputeConnectorOperatorRoleE206977E", + "LambdaMicrovmComputeBuildRoleF0A13AC7", + "LambdaMicrovmComputeExecutionRoleAA0C4A0D" + ] + }, + { + "file": "backgroundagentdevAgentRegistryStackFF9DF145.nested.template.json", + "resources": 19, + "bytes": 36474, + "parameters": 0, + "outputs": 2, + "names": [], + "roles": [] + }, + { + "file": "backgroundagentdevRegistryApiF3ADADE2.nested.template.json", + "resources": 35, + "bytes": 47684, + "parameters": 3, + "outputs": 3, + "names": [], + "roles": [] + } + ] + }, + { + "name": "microvm-image-gateway", + "mode": "baseline", + "templates": [ + { + "file": "backgroundagent-dev.template.json", + "resources": 479, + "bytes": 516182, + "parameters": 1, + "outputs": 48, + "names": [ + { + "type": "AWS::Lambda::NetworkConnector", + "name": "backgroundagent-dev-microvm-egress" + }, + { + "type": "AWS::Lambda::NetworkConnector", + "name": "backgroundagent-dev-microvm-build-egress" + }, + { + "type": "AWS::Lambda::MicrovmImage", + "name": "backgroundagent-dev-abca-agent" + } + ], + "roles": [ + "LambdaMicrovmComputeConnectorOperatorRoleE206977E", + "LambdaMicrovmComputeBuildRoleF0A13AC7", + "LambdaMicrovmComputeExecutionRoleAA0C4A0D" + ] + }, + { + "file": "backgroundagentdevAgentRegistryStackFF9DF145.nested.template.json", + "resources": 19, + "bytes": 36474, + "parameters": 0, + "outputs": 2, + "names": [], + "roles": [] + }, + { + "file": "backgroundagentdevRegistryApiF3ADADE2.nested.template.json", + "resources": 35, + "bytes": 47684, + "parameters": 3, + "outputs": 3, + "names": [], + "roles": [] + } + ] + }, + { + "name": "microvm-image-gateway-vault", + "mode": "baseline", + "templates": [ + { + "file": "backgroundagent-dev.template.json", + "resources": 489, + "bytes": 536098, + "parameters": 1, + "outputs": 50, + "names": [ + { + "type": "AWS::Lambda::NetworkConnector", + "name": "backgroundagent-dev-microvm-egress" + }, + { + "type": "AWS::Lambda::NetworkConnector", + "name": "backgroundagent-dev-microvm-build-egress" + }, + { + "type": "AWS::Lambda::MicrovmImage", + "name": "backgroundagent-dev-abca-agent" + } + ], + "roles": [ + "LambdaMicrovmComputeConnectorOperatorRoleE206977E", + "LambdaMicrovmComputeBuildRoleF0A13AC7", + "LambdaMicrovmComputeExecutionRoleAA0C4A0D" + ] + }, + { + "file": "backgroundagentdevAgentRegistryStackFF9DF145.nested.template.json", + "resources": 19, + "bytes": 36474, + "parameters": 0, + "outputs": 2, + "names": [], + "roles": [] + }, + { + "file": "backgroundagentdevLinearVaultConsentPageStack671B3D9F.nested.template.json", + "resources": 12, + "bytes": 17737, + "parameters": 0, + "outputs": 1, + "names": [], + "roles": [] + }, + { + "file": "backgroundagentdevRegistryApiF3ADADE2.nested.template.json", + "resources": 35, + "bytes": 47684, + "parameters": 3, + "outputs": 3, + "names": [], + "roles": [] + } + ] + }, + { + "name": "default", + "mode": "raw-nested", + "templates": [ + { + "file": "backgroundagent-dev.template.json", + "resources": 454, + "bytes": 476779, + "parameters": 1, + "outputs": 41, + "names": [], + "roles": [] + }, + { + "file": "backgroundagentdevAgentRegistryStackFF9DF145.nested.template.json", + "resources": 19, + "bytes": 36474, + "parameters": 0, + "outputs": 2, + "names": [], + "roles": [] + }, + { + "file": "backgroundagentdevRegistryApiF3ADADE2.nested.template.json", + "resources": 35, + "bytes": 47684, + "parameters": 3, + "outputs": 3, + "names": [], + "roles": [] + } + ] + }, + { + "name": "microvm", + "mode": "raw-nested", + "templates": [ + { + "file": "backgroundagent-dev.template.json", + "resources": 455, + "bytes": 480295, + "parameters": 1, + "outputs": 48, + "names": [], + "roles": [] + }, + { + "file": "backgroundagentdevAgentRegistryStackFF9DF145.nested.template.json", + "resources": 19, + "bytes": 36474, + "parameters": 0, + "outputs": 2, + "names": [], + "roles": [] + }, + { + "file": "backgroundagentdevMicrovmNestedCD7454C8.nested.template.json", + "resources": 19, + "bytes": 41875, + "parameters": 7, + "outputs": 6, + "names": [ + { + "type": "AWS::Lambda::NetworkConnector", + "name": "--Token-TOKEN-5450---microvm-egress" + }, + { + "type": "AWS::Lambda::NetworkConnector", + "name": "--Token-TOKEN-5450---microvm-build-egress" + } + ], + "roles": [ + "LambdaMicrovmComputeConnectorOperatorRoleE206977E", + "LambdaMicrovmComputeBuildRoleF0A13AC7", + "LambdaMicrovmComputeExecutionRoleAA0C4A0D" + ] + }, + { + "file": "backgroundagentdevRegistryApiF3ADADE2.nested.template.json", + "resources": 35, + "bytes": 47684, + "parameters": 3, + "outputs": 3, + "names": [], + "roles": [] + } + ], + "cycleError": "Template is undeployable, these resources have a dependency cycle: TaskApiDeployment093F4F13e3bbe4ed71a6615db6cbe11f92059df4 -> TaskApijirawebhookPOST96251C09 -> JiraIntegrationWebhookFn7CE96A20 -> JiraIntegrationWebhookFnServiceRoleDefaultPolicy40DD76AF -> JiraIntegrationWebhookProcessorFnE5EB0D16 -> JiraIntegrationWebhookProcessorFnServiceRoleDefaultPolicy4E7A0586 -> TaskOrchestratorOrchestratorFnCurrentVersionAliaslive08F80EEB -> TaskOrchestratorOrchestratorFn8CE22C41 -> TaskOrchestratorOrchestratorFnServiceRoleDefaultPolicyDECF0D43 -> Runtime99E3DDFA -> RuntimeExecutionRoleDefaultPolicy2B020CFC -> AgentSessionRoleB6C61074 -> MicrovmNestedNestedStackMicrovmNestedNestedStackResource6BAFC965 -> AgentSessionRoleB6C61074:" + }, + { + "name": "microvm-image", + "mode": "raw-nested", + "templates": [ + { + "file": "backgroundagent-dev.template.json", + "resources": 455, + "bytes": 482894, + "parameters": 1, + "outputs": 48, + "names": [], + "roles": [] + }, + { + "file": "backgroundagentdevAgentRegistryStackFF9DF145.nested.template.json", + "resources": 19, + "bytes": 36474, + "parameters": 0, + "outputs": 2, + "names": [], + "roles": [] + }, + { + "file": "backgroundagentdevMicrovmNestedCD7454C8.nested.template.json", + "resources": 20, + "bytes": 44561, + "parameters": 7, + "outputs": 9, + "names": [ + { + "type": "AWS::Lambda::NetworkConnector", + "name": "--Token-TOKEN-8353---microvm-egress" + }, + { + "type": "AWS::Lambda::NetworkConnector", + "name": "--Token-TOKEN-8353---microvm-build-egress" + }, + { + "type": "AWS::Lambda::MicrovmImage", + "name": "--Token-TOKEN-8353---abca-agent" + } + ], + "roles": [ + "LambdaMicrovmComputeConnectorOperatorRoleE206977E", + "LambdaMicrovmComputeBuildRoleF0A13AC7", + "LambdaMicrovmComputeExecutionRoleAA0C4A0D" + ] + }, + { + "file": "backgroundagentdevRegistryApiF3ADADE2.nested.template.json", + "resources": 35, + "bytes": 47684, + "parameters": 3, + "outputs": 3, + "names": [], + "roles": [] + } + ], + "cycleError": "Template is undeployable, these resources have a dependency cycle: TaskApiDeployment093F4F13e3bbe4ed71a6615db6cbe11f92059df4 -> TaskApijirawebhookPOST96251C09 -> JiraIntegrationWebhookFn7CE96A20 -> JiraIntegrationWebhookFnServiceRoleDefaultPolicy40DD76AF -> JiraIntegrationWebhookProcessorFnE5EB0D16 -> JiraIntegrationWebhookProcessorFnServiceRoleDefaultPolicy4E7A0586 -> TaskOrchestratorOrchestratorFnCurrentVersionAliaslive08F80EEB -> TaskOrchestratorOrchestratorFn8CE22C41 -> TaskOrchestratorOrchestratorFnServiceRoleDefaultPolicyDECF0D43 -> MicrovmNestedNestedStackMicrovmNestedNestedStackResource6BAFC965 -> AgentSessionRoleB6C61074 -> MicrovmNestedNestedStackMicrovmNestedNestedStackResource6BAFC965:" + }, + { + "name": "microvm-image-gateway", + "mode": "raw-nested", + "templates": [ + { + "file": "backgroundagent-dev.template.json", + "resources": 462, + "bytes": 488182, + "parameters": 1, + "outputs": 48, + "names": [], + "roles": [] + }, + { + "file": "backgroundagentdevAgentRegistryStackFF9DF145.nested.template.json", + "resources": 19, + "bytes": 36474, + "parameters": 0, + "outputs": 2, + "names": [], + "roles": [] + }, + { + "file": "backgroundagentdevMicrovmNestedCD7454C8.nested.template.json", + "resources": 20, + "bytes": 44565, + "parameters": 7, + "outputs": 9, + "names": [ + { + "type": "AWS::Lambda::NetworkConnector", + "name": "--Token-TOKEN-11313---microvm-egress" + }, + { + "type": "AWS::Lambda::NetworkConnector", + "name": "--Token-TOKEN-11313---microvm-build-egress" + }, + { + "type": "AWS::Lambda::MicrovmImage", + "name": "--Token-TOKEN-11313---abca-agent" + } + ], + "roles": [ + "LambdaMicrovmComputeConnectorOperatorRoleE206977E", + "LambdaMicrovmComputeBuildRoleF0A13AC7", + "LambdaMicrovmComputeExecutionRoleAA0C4A0D" + ] + }, + { + "file": "backgroundagentdevRegistryApiF3ADADE2.nested.template.json", + "resources": 35, + "bytes": 47684, + "parameters": 3, + "outputs": 3, + "names": [], + "roles": [] + } + ], + "cycleError": "Template is undeployable, these resources have a dependency cycle: TaskApiDeployment093F4F13e3bbe4ed71a6615db6cbe11f92059df4 -> TaskApijirawebhookPOST96251C09 -> JiraIntegrationWebhookFn7CE96A20 -> JiraIntegrationWebhookFnServiceRoleDefaultPolicy40DD76AF -> JiraIntegrationWebhookProcessorFnE5EB0D16 -> JiraIntegrationWebhookProcessorFnServiceRoleDefaultPolicy4E7A0586 -> TaskOrchestratorOrchestratorFnCurrentVersionAliaslive08F80EEB -> TaskOrchestratorOrchestratorFn8CE22C41 -> TaskOrchestratorOrchestratorFnServiceRoleDefaultPolicyDECF0D43 -> MicrovmNestedNestedStackMicrovmNestedNestedStackResource6BAFC965 -> AgentSessionRoleB6C61074 -> MicrovmNestedNestedStackMicrovmNestedNestedStackResource6BAFC965:" + }, + { + "name": "microvm-image-gateway-vault", + "mode": "raw-nested", + "templates": [ + { + "file": "backgroundagent-dev.template.json", + "resources": 472, + "bytes": 506765, + "parameters": 1, + "outputs": 50, + "names": [], + "roles": [] + }, + { + "file": "backgroundagentdevAgentRegistryStackFF9DF145.nested.template.json", + "resources": 19, + "bytes": 36474, + "parameters": 0, + "outputs": 2, + "names": [], + "roles": [] + }, + { + "file": "backgroundagentdevLinearVaultConsentPageStack671B3D9F.nested.template.json", + "resources": 12, + "bytes": 17737, + "parameters": 0, + "outputs": 1, + "names": [], + "roles": [] + }, + { + "file": "backgroundagentdevMicrovmNestedCD7454C8.nested.template.json", + "resources": 20, + "bytes": 46955, + "parameters": 7, + "outputs": 9, + "names": [ + { + "type": "AWS::Lambda::NetworkConnector", + "name": "--Token-TOKEN-14382---microvm-egress" + }, + { + "type": "AWS::Lambda::NetworkConnector", + "name": "--Token-TOKEN-14382---microvm-build-egress" + }, + { + "type": "AWS::Lambda::MicrovmImage", + "name": "--Token-TOKEN-14382---abca-agent" + } + ], + "roles": [ + "LambdaMicrovmComputeConnectorOperatorRoleE206977E", + "LambdaMicrovmComputeBuildRoleF0A13AC7", + "LambdaMicrovmComputeExecutionRoleAA0C4A0D" + ] + }, + { + "file": "backgroundagentdevRegistryApiF3ADADE2.nested.template.json", + "resources": 35, + "bytes": 47684, + "parameters": 3, + "outputs": 3, + "names": [], + "roles": [] + } + ], + "cycleError": "Template is undeployable, these resources have a dependency cycle: TaskApiDeployment093F4F13e3bbe4ed71a6615db6cbe11f92059df4 -> TaskApijirawebhookPOST96251C09 -> JiraIntegrationWebhookFn7CE96A20 -> JiraIntegrationWebhookFnServiceRoleDefaultPolicy40DD76AF -> JiraIntegrationWebhookProcessorFnE5EB0D16 -> JiraIntegrationWebhookProcessorFnServiceRoleDefaultPolicy4E7A0586 -> TaskOrchestratorOrchestratorFnCurrentVersionAliaslive08F80EEB -> TaskOrchestratorOrchestratorFn8CE22C41 -> TaskOrchestratorOrchestratorFnServiceRoleDefaultPolicyDECF0D43 -> MicrovmNestedNestedStackMicrovmNestedNestedStackResource6BAFC965 -> AgentSessionRoleB6C61074 -> MicrovmNestedNestedStackMicrovmNestedNestedStackResource6BAFC965:" + }, + { + "name": "default", + "mode": "parent-role", + "templates": [ + { + "file": "backgroundagent-dev.template.json", + "resources": 454, + "bytes": 476779, + "parameters": 1, + "outputs": 41, + "names": [], + "roles": [] + }, + { + "file": "backgroundagentdevAgentRegistryStackFF9DF145.nested.template.json", + "resources": 19, + "bytes": 36474, + "parameters": 0, + "outputs": 2, + "names": [], + "roles": [] + }, + { + "file": "backgroundagentdevRegistryApiF3ADADE2.nested.template.json", + "resources": 35, + "bytes": 47684, + "parameters": 3, + "outputs": 3, + "names": [], + "roles": [] + } + ] + }, + { + "name": "microvm", + "mode": "parent-role", + "templates": [ + { + "file": "backgroundagent-dev.template.json", + "resources": 457, + "bytes": 488288, + "parameters": 1, + "outputs": 48, + "names": [], + "roles": [] + }, + { + "file": "backgroundagentdevAgentRegistryStackFF9DF145.nested.template.json", + "resources": 19, + "bytes": 36474, + "parameters": 0, + "outputs": 2, + "names": [], + "roles": [] + }, + { + "file": "backgroundagentdevMicrovmNestedCD7454C8.nested.template.json", + "resources": 17, + "bytes": 29503, + "parameters": 3, + "outputs": 6, + "names": [ + { + "type": "AWS::Lambda::NetworkConnector", + "name": "backgroundagent-dev-microvm-egress" + }, + { + "type": "AWS::Lambda::NetworkConnector", + "name": "backgroundagent-dev-microvm-build-egress" + } + ], + "roles": [ + "LambdaMicrovmComputeConnectorOperatorRoleE206977E", + "LambdaMicrovmComputeBuildRoleF0A13AC7" + ] + }, + { + "file": "backgroundagentdevRegistryApiF3ADADE2.nested.template.json", + "resources": 35, + "bytes": 47684, + "parameters": 3, + "outputs": 3, + "names": [], + "roles": [] + } + ] + }, + { + "name": "microvm-image", + "mode": "parent-role", + "templates": [ + { + "file": "backgroundagent-dev.template.json", + "resources": 457, + "bytes": 490655, + "parameters": 1, + "outputs": 48, + "names": [], + "roles": [] + }, + { + "file": "backgroundagentdevAgentRegistryStackFF9DF145.nested.template.json", + "resources": 19, + "bytes": 36474, + "parameters": 0, + "outputs": 2, + "names": [], + "roles": [] + }, + { + "file": "backgroundagentdevMicrovmNestedCD7454C8.nested.template.json", + "resources": 18, + "bytes": 31884, + "parameters": 3, + "outputs": 8, + "names": [ + { + "type": "AWS::Lambda::NetworkConnector", + "name": "backgroundagent-dev-microvm-egress" + }, + { + "type": "AWS::Lambda::NetworkConnector", + "name": "backgroundagent-dev-microvm-build-egress" + }, + { + "type": "AWS::Lambda::MicrovmImage", + "name": "backgroundagent-dev-abca-agent" + } + ], + "roles": [ + "LambdaMicrovmComputeConnectorOperatorRoleE206977E", + "LambdaMicrovmComputeBuildRoleF0A13AC7" + ] + }, + { + "file": "backgroundagentdevRegistryApiF3ADADE2.nested.template.json", + "resources": 35, + "bytes": 47684, + "parameters": 3, + "outputs": 3, + "names": [], + "roles": [] + } + ] + }, + { + "name": "microvm-image-gateway", + "mode": "parent-role", + "templates": [ + { + "file": "backgroundagent-dev.template.json", + "resources": 464, + "bytes": 495943, + "parameters": 1, + "outputs": 48, + "names": [], + "roles": [] + }, + { + "file": "backgroundagentdevAgentRegistryStackFF9DF145.nested.template.json", + "resources": 19, + "bytes": 36474, + "parameters": 0, + "outputs": 2, + "names": [], + "roles": [] + }, + { + "file": "backgroundagentdevMicrovmNestedCD7454C8.nested.template.json", + "resources": 18, + "bytes": 31884, + "parameters": 3, + "outputs": 8, + "names": [ + { + "type": "AWS::Lambda::NetworkConnector", + "name": "backgroundagent-dev-microvm-egress" + }, + { + "type": "AWS::Lambda::NetworkConnector", + "name": "backgroundagent-dev-microvm-build-egress" + }, + { + "type": "AWS::Lambda::MicrovmImage", + "name": "backgroundagent-dev-abca-agent" + } + ], + "roles": [ + "LambdaMicrovmComputeConnectorOperatorRoleE206977E", + "LambdaMicrovmComputeBuildRoleF0A13AC7" + ] + }, + { + "file": "backgroundagentdevRegistryApiF3ADADE2.nested.template.json", + "resources": 35, + "bytes": 47684, + "parameters": 3, + "outputs": 3, + "names": [], + "roles": [] + } + ] + }, + { + "name": "microvm-image-gateway-vault", + "mode": "parent-role", + "templates": [ + { + "file": "backgroundagent-dev.template.json", + "resources": 474, + "bytes": 515859, + "parameters": 1, + "outputs": 50, + "names": [], + "roles": [] + }, + { + "file": "backgroundagentdevAgentRegistryStackFF9DF145.nested.template.json", + "resources": 19, + "bytes": 36474, + "parameters": 0, + "outputs": 2, + "names": [], + "roles": [] + }, + { + "file": "backgroundagentdevLinearVaultConsentPageStack671B3D9F.nested.template.json", + "resources": 12, + "bytes": 17737, + "parameters": 0, + "outputs": 1, + "names": [], + "roles": [] + }, + { + "file": "backgroundagentdevMicrovmNestedCD7454C8.nested.template.json", + "resources": 18, + "bytes": 31884, + "parameters": 3, + "outputs": 8, + "names": [ + { + "type": "AWS::Lambda::NetworkConnector", + "name": "backgroundagent-dev-microvm-egress" + }, + { + "type": "AWS::Lambda::NetworkConnector", + "name": "backgroundagent-dev-microvm-build-egress" + }, + { + "type": "AWS::Lambda::MicrovmImage", + "name": "backgroundagent-dev-abca-agent" + } + ], + "roles": [ + "LambdaMicrovmComputeConnectorOperatorRoleE206977E", + "LambdaMicrovmComputeBuildRoleF0A13AC7" + ] + }, + { + "file": "backgroundagentdevRegistryApiF3ADADE2.nested.template.json", + "resources": 35, + "bytes": 47684, + "parameters": 3, + "outputs": 3, + "names": [], + "roles": [] + } + ] + } + ] +} diff --git a/docs/verification/645-p3-implementation-plan.md b/docs/verification/645-p3-implementation-plan.md new file mode 100644 index 000000000..8a4d6f121 --- /dev/null +++ b/docs/verification/645-p3-implementation-plan.md @@ -0,0 +1,234 @@ +# ADR-021 P3 implementation and completion plan + +Prepared 2026-09-13 from `main` `5e10038c7e28179b302ac4de78b709795aeba3ce`. Read the [review](./645-p3-readiness-review.md) for evidence and the beginner introduction. This document proposes work; it does not mark P3 as implemented. + +## The result we want + +When the coding agent asks a human for permission, its computer may go to sleep. The human can approve or deny while it sleeps. The computer wakes, reads the saved answer, and continues or refuses the action. If nobody answers, it wakes before the deadline and applies the existing timeout-as-denial rule. Its files, task identity, permissions, progress and deadline remain correct. + +**HITL** means “human in the loop.” **Cedar** is the policy system that decides which actions need permission. **DynamoDB** is where ABCA stores task and approval records. A **transaction** changes/checks related records together so competing actions cannot half-win. A **strongly consistent read** asks DynamoDB for the latest committed value. **Idempotent** means repeating an operation has the same intended effect as doing it once. A **reconciler** repeatedly compares what should be happening with what is actually happening and repairs differences. + +P3 coordinates the supervisor, approval records and the sleeping computer. + +For example, with a five-minute approval window and the suggested settings below: the question appears at **12:00**; after **12:00:30** the VM can sleep. If approval arrives at **12:02**, ABCA wakes it and it reads the saved answer. If nobody answers, ABCA wakes it around **12:04** so the agent can deny at **12:05**. Waking must not start a new five-minute timer. + +A few implementation words used below: + +| Term | Meaning here | +|---|---| +| API / SDK | An API is a service's set of commands; an SDK is a library for sending them from code. | +| Bootstrap | Install the deployment's foundation permissions and storage before deploying the application. | +| ARN | An AWS resource's full address, such as the address of one permission role or image. | +| Monotonic clock / wall clock | A stopwatch measuring elapsed time versus a clock showing the date/time. Sleep can affect them differently, so the saved deadline must still count. | +| Durable | Saved outside the VM, so putting the VM to sleep does not lose it. A failed save is not durable. | +| Barrier | A controlled door: coding cannot proceed until the required saves or credential refresh have succeeded. | +| Orphan | A VM that still exists but whose normal supervisor/wake-up path has lost track of it. | +| Replay / race | Replay repeats earlier work after recovery. A race happens when two actions, such as approval and cancellation, arrive almost together. | +| Regression test / fault injection | A test that prevents a known bug returning / a test that deliberately simulates a failure. | +| Change set / rollback | AWS's preview of deployment changes / the procedure to restore the previous working deployment. | + +## Fixed boundaries + +- Keep the eight-hour `maximumDurationInSeconds = 28800`, including suspended time. +- Keep `idlePolicy` absent. Traffic-based idleness would mistake outbound-only coding work for inactivity. +- Keep explicit `NO_INGRESS` and no `CreateMicrovmAuthToken` grant. No public agent-control endpoint or JWE token refresh system is needed for P3. +- Keep the 4,096-byte hook-payload boundary and S3 fallback, including registry-resolved assets. +- Only the orchestrator initiates suspension. Approval handlers may request resume after committing a decision; the orchestrator repairs missed resumes. +- The agent remains the authority that consumes a decision or times out a gate. Approval HTTP handlers must not simply mark the coding task RUNNING. +- Preserve tenant-scoped credentials, task/user/repository tags and fail-closed behavior. “Fail closed” means refusing an action when safe authorization cannot be established. +- Keep the ABCA concurrency slot while suspended; release it exactly once on terminal finalization. Suspended AWS memory-quota consumption remains unverified, not the justification for a new claim. +- General progress-based hang detection (#491), operator shell access and extended/off-hours approval windows are outside P3. Coordinate compatible seams without making those projects hard dependencies. + +## Delivery order + +| Work package | Depends on | Exit condition | +|---|---|---| +| 0. Freeze evidence and establish test baseline | Current review | Known baseline, reproducible checks and issue-to-test map | +| 1. Repair P2 correctness/security prerequisites | 0 | Critical defects have regression tests and are fixed | +| 2. Optional infrastructure nesting | 0; finish before final deployment validation if chosen | Safe boundary, current feature matrix, bootstrap/migration verified | +| 3. Prove clean P2 deployment | 1 and any chosen nesting changes | Managed image and task complete with no manual IAM workaround | +| 4. Define lifecycle contract and widen all strategies | 0; can develop alongside 1–3 | Typed interface, state/race contract and focused tests | +| 5. Implement safe agent hooks and clocks | 1's test/liveness fixes, 4 | Hooks, refresh barrier, deadlines and failure paths tested | +| 6. Add orchestrator and approval wake-up | 4–5 | Race-safe integration, bounded recovery, scoped grants | +| 7. Live P3 verification and completion | 3, 5–6 | Full acceptance matrix passes with retained evidence | + +Keep nesting in a separate change from lifecycle logic. Developing P3 locally need not wait for every live prerequisite, but claiming it complete does. + +## 0. Establish the baseline + +1. Record the source commit, dependency lockfiles, bootstrap bundle version, image version and enabled context flags. Track #645, #817, #700, #818, #841 and #857 with concrete remaining checkboxes. Do not reopen consolidated #813–816 as if they represented four separate finished fixes. +2. Land/review this comment cleanup independently. It deliberately does not repair runtime behavior. +3. Fix the #841 test fixture: every spawned pipeline thread must finish or be stopped/joined before mocks/environment are restored. Synchronize with events, not arbitrary sleeps; fail teardown if a thread remains alive. Keep test state isolated without discarding references to live threads. +4. Run focused baseline suites for server hooks, task-state approvals, credentials, strategy, orchestrator, approval handlers, construct IAM and bootstrap coverage. Record existing unrelated failures rather than quietly weakening assertions or thresholds. +5. Make the phase vocabulary explicit: this is P3 of ADR-021. Use different names for PR/work-package sequencing so “P1 priority” on an issue is not confused with phase P1. + +## 1. Repair the prerequisites + +### 1A. Finish #817 + +- **Deletion IAM:** grant the coordinator `s3:DeleteObject` on the dedicated MicroVM payload bucket's task-object prefix. The worker gets no delete permission. Update `task-orchestrator.test.ts` and `stacks/agent.test.ts`, which currently assert the wrong absence. Exercise upload → start → finalize → delete, failed deletion logging and lifecycle fallback. Keep deletion best-effort so it does not hide the task's real outcome. +- **Classifier:** separate a stable MicroVM failure category from untrusted/free-form AWS `stateReason`. Test host unavailable, capacity unavailable, true unsupported region, authorization/configuration, concurrency and hook HTTP 400 cases. Check both stored task classification and user-facing retry advice. Preserve raw reason text for diagnosis without letting it redefine the category. +- **Trusted configuration:** decide which data is trusted at `/run`. A same-payload account anchor cannot authenticate its siblings. Bind accepted deployment identifiers to trusted deployment configuration or an authenticated payload reference; preserve legitimate cross-region secrets. Reject another workspace's secret even when it has the same account number. Add exact contract key/ARN-key/anchor assertions plus the reverse assertion that newly introduced ARN fields cannot bypass validation. +- **S3 bad bytes:** feed truncated and invalid-encoding bytes through the real fetch/decode/envelope/route path. Expect a structured unreadable-payload response, no partially installed environment and no pipeline thread. Keep producer-schema mistakes distinguishable from transport failures. +- **Documentation:** verify all remaining contract/status changes update source docs and their generated copies through the sync script. + +### 1B. Narrow payload reads (#700) + +Choose task-scoped transport before describing the backend as suitable for untrusted multi-tenant tasks. The current role reads the payload **before** it establishes task-scoped identity, so merely moving an existing S3 grant onto the session role is not enough. + +Evaluate a short-lived, single-object signed URL or a trusted bootstrap envelope that safely establishes task identity first. For a signed URL, check whether the guest's network/DNS policy permits its host and whether retry duration fits the URL lifetime; keep the bearer URL out of logs. For role-based fetching, prove how the role/session tags are trusted before the object is read. Test wrong task, guessed key, expired reference, retries and maximum payload size. Preserve the ECS contract or document a deliberate staged rollout. Finalize-time deletion complements this fix; it does not prevent reads of other active tasks. + +### 1C. Fix approval/heartbeat ordering + +Change `task_state.transact_resume_from_approval` to refresh `agent_heartbeat_at` in the **same conditional update** that restores RUNNING. Keep the existing expected status and `awaiting_approval_request_id` conditions. + +Regression: task starts, waits over 240 seconds, then resumes. Force the orchestrator to poll after the transaction but before the heartbeat thread's next tick. It must remain healthy. Also test cancellation winning the race, wrong request ID, and ECS behavior; do not enable server-thread heartbeat enforcement on ECS. + +### 1D. Make session-start retries honest + +Fault-inject a successful service-side creation followed by a lost response. Observe a second application start attempt, not just retries inside one SDK command. Define a persisted logical start-attempt identity and stable client token for retries of an uncertain attempt. Mint a new attempt only when the old session is known terminal or the service's idempotency semantics require it. Test same-request replay, confirmed failed start, durable Lambda replay, cancellation and orphan cleanup. Check AWS token retention/conflict semantics before finalizing the policy; ECS's existing `clientToken: taskId` is useful precedent, not proof that MicroVM has identical semantics. + +### 1E. Make verification observable + +Resolve #810 by exposing a useful structured failure signal for CloudWatch writers or removing the dead counter and using another observable signal. Test failure of the logging system itself. For #818, document the shared runtime 443-only rule and test registry payload overflow; do not expand network ports just to satisfy an incorrect issue premise. + +## 2. Nest infrastructure if adopting the split + +1. Introduce `LambdaMicrovmStack` as a `NestedStack` wrapper and an explicit way to supply the MicroVM execution role from the parent. Keep shared session-role trust and runtime-role ownership in the parent. Preserve exact grant behavior. +2. Pass a stable deployment name for image and both connector names. Never sanitize unresolved CDK tokens. Preserve name/ARN resolution for imported images as well as managed images. +3. Return artifact/payload bucket, image, connector and build-role identifiers to existing parent consumers. Keep parent `Microvm*` outputs unchanged so packaging/CLI discovery still works. Check no-image bootstrap state and imported-image state, not just the managed-image case. +4. Inspect all generated role names against bootstrap allowlists. If permissions/names require a bundle update, regenerate artifacts, apply the repository's version rule, and run policy/golden/artifact-sync tests. Keep `PassRole` exact or narrowly prefixed; no broad all-roles shortcut or resurrected broken service condition. +5. Update assertions to inspect every child template. Assert no cycles with `Template.fromStack`, correct tags and solution user-agent propagation, supported region checks, hook values and runtime/build network separation. +6. Synthesize real bundled configurations: default, ECS, MicroVM without image, imported image and managed image; cross the relevant gateway/vault flags; validate the heaviest configuration and custom naming paths. Enforce each stack's resource/byte limits, parameter/output limits and the deployment's nested-operation limits. Do not require the numbers in this review to stay constant forever. +7. Remove or replace the #857 guard only when the supported combination actually passes the current matrix. Keep a size regression test; do not retain “505 resources” as a permanent error message. +8. Review the CloudFormation change set for replacement/deletion. For the first experimental rollout, prefer a dedicated test deployment. For migration of an existing one, stop new admissions, drain/terminate existing tasks, preserve required artifacts/state and use a service-supported migration procedure. Define rollback before applying the change set. Do not let `autoDeleteObjects` silently erase an active payload/artifact bucket. + +**Done:** cycle-free bundled templates, least-privilege bootstrap coverage, preserved outputs, safe migration/change-set evidence and a clean deployed image build. The local prototype alone meets none of the live gates. + +## 3. Re-prove P2 on the final infrastructure + +Use a supported Region and an isolated development repository/account deployment. Record the actual deployed bootstrap bundle (at least 1.6.0, or the newer bundle produced by nesting). Compare effective policies as well as the displayed version. Update bootstrap deliberately when required; a command that skips an already bootstrapped stack is not evidence of refresh. + +Follow `cdk/scripts/package-microvm-artifact.sh` and the P1/P2 runbooks: + +1. Deploy the no-image infrastructure if its artifact bucket does not exist. Package/upload the current source; create the managed image through CloudFormation with a pinned base image/version. Check `ARM_64`, hook `ENABLED` values and separate build/runtime connectors. Do not substitute the manual image path when validating the formerly broken managed path. +2. Verify `/ready` local warm-up and `/validate` self-check; no AWS credential initialization from build hooks. Review credential warnings without recording secret values. +3. Run a normal task through clone → change → tests → commit/push → PR. Observe `bgagent watch`, task details/list heartbeat, structured application logs and final task outcome. +4. Exercise the Memory path and any features claimed as parity. Absence of an error log alone is not proof a Memory write happened. +5. While the VM is **RUNNING**, verify logging with the narrowed runtime role, including whether removing CreateLogGroup still permits all required streams. Checking after teardown is not equivalent. +6. Verify payload deletion and active termination after success, failure and cancellation. Record service states, request IDs and timing. Negative-test public ingress and runtime port restrictions from suitable test locations. +7. No ad hoc IAM edits, console workarounds or privileged fallback. If one is needed, fix source/bootstrap and repeat the affected acceptance case on that final version. + +**Done:** the remaining P2 warning conditions are discharged by a dated runbook. Update warning text only to match the evidence; do not imply that P3 is finished. + +## 4. Define the P3 contract before wiring consumers + +### Strategy methods + +Add mandatory `suspendSession(handle)` and `resumeSession(handle)` to `ComputeStrategy` in the **same commit as all three implementations**. Use an explicit result such as `{ supported: false } | { supported: true }`; supported means the capability/request is supported, not that the VM is already in its final state. Operational failures must remain distinguishable from unsupported capability. + +AgentCore and ECS return explicit unsupported results. MicroVM issues `SuspendMicrovm`/`ResumeMicrovm` with `microvmIdentifier: handle.microvmId`. Test wrong handle variants, already-target-state requests, state-conflict races, missing/terminated VMs and retriable/permanent API errors. Verify actual AWS behavior before normalizing a conflict into success. Preserve the strategy's existing state mapping; keep `reason` diagnostic. + +### Durable intent and policy + +Store a small typed optional lifecycle record on the task row, separate from the existing string-only `compute_metadata` handle: active `request_id`, desired action (`suspend` or `resume`), request timestamp and gate deadline. Guard writes by task status, matching gate ID and matching MicroVM handle; do not overwrite another gate's intent. Add a generation/version condition if needed to prevent an older request from undoing a newer one. Reuse the task table, not a new coordination service. + +Persist failure counters/backoff and anomaly episode tracking in durable poll-loop state so Lambda replay does not reset them. Update shared/public types and sync guards only where that record is actually exposed. Store no credentials or bearer URLs in the lifecycle record. + +The policy combines task status, the **specific current approval row's status**, desired action, VM state and current time. A PENDING row already exists throughout every wait: “approval row exists” is not a reason to resume. Check APPROVED/DENIED, deadline proximity, missing-row recovery or other explicit wake conditions. + +Proposed initial tuning for the implementation: 30-second suspend grace, 60-second pre-deadline wake margin, and three consecutive poll failures before escalation. Treat these as measured/tunable policy values, not AWS facts. Only suspend after grace when enough time remains to pay for resume overhead and a useful sleep. Clamp polling/backoff to the next wake deadline and session lifetime; do not let a user-configured long poll interval oversleep it. + +### State/action table + +| Task and gate | Observed VM | Action | +|---|---|---| +| RUNNING, doing work | RUNNING | Check liveness; never suspend based on lack of inbound traffic | +| AWAITING_APPROVAL, current gate PENDING, before grace or too near deadline | RUNNING | Keep awake | +| Same, grace passed and sufficient remaining time | RUNNING | Conditionally record suspend intent, recheck gate, request suspend | +| PENDING, intentionally suspended, ample time remains | SUSPENDING/SUSPENDED | Wait; do not mark the stopped heartbeat as a crash | +| Current gate APPROVED/DENIED | SUSPENDING/SUSPENDED | Request/reschedule resume; let agent consume the committed answer | +| PENDING at deadline minus margin | SUSPENDING/SUSPENDED | Resume for agent-side expiry evaluation | +| RUNNING or a different gate, unexpectedly suspended | SUSPENDING/SUSPENDED | Emit one anomaly per episode and perform bounded recovery; do not label suspension itself a crash | +| Task terminal/cancelled | Any live or suspended state | Terminate; never resurrect task state | +| VM terminal, task nonterminal | Terminal/not found | Re-read task consistently, apply substrate-failure reconciliation, finalize once | +| Gate row missing or unreadable while asleep | SUSPENDED | Wake conservatively when possible; preserve existing missing-row/timeout safeguards and bounded API-error handling | + +## 5. Implement the agent's suspend/resume safety + +Files: `agent/src/server.py`, `hooks.py`, `task_state.py`, `aws_session.py`, `progress_writer.py`, credential-helper/cache owners, `contracts/constants.json`, and focused tests. + +1. Create a per-task lifecycle context holding task/VM identity, active approval gate, durable deadline and synchronization primitives. The current server does not hand `/resume` a live gate object automatically. Register/unregister this context with the pipeline and approval hook, including failures and teardown. +2. `/suspend` validates that the task is still parked on the intended gate. Wait for lifecycle/progress work already in progress to finish, establish an acknowledged durability barrier, and return success only within the hook budget. Existing best-effort event methods cannot establish that barrier. On timeout/write failure, report a hook failure so the coordinator keeps/reconciles the running VM. Test approval or cancellation arriving during this boundary. +3. `/resume` refreshes ambient/runtime credential providers as necessary, then ensures tenant-scoped assumed credentials are usable **with the same task/user/repo tags**. Inventory cached DynamoDB/S3/Memory/Logs clients and the Claude Bedrock credential helper; replacing one global session does not replace every already-created client or subprocess cache. Do not call the test-only `reset_session_cache()` and lose identity. Fail closed on refresh failure. +4. Keep the coding action blocked behind the resume barrier until refresh and gate reconciliation finish. Handle duplicate hook calls and concurrent lifecycle requests without deadlocks. Expired credentials or a slow AWS call must not hold the hook beyond its service budget. +5. Reseed the application PRNG from fresh OS entropy on **both `/run` and `/resume`**. Do not seed it with task IDs, timestamps or an image-fixed value. Continue using cryptographic randomness for secrets. Test resumed/sibling snapshot uniqueness where meaningful; do not claim that `random` becomes cryptographically safe. +6. Compute approval remaining time as `min(monotonic_deadline - monotonic_now, created_at + timeout_s - wall_now)`, clamped at zero. Pass the original recorded `created_at` into the gate context; do not create a fresh timeout on resume. Check it on every poll and at resume, then wake the existing agent-owned decision loop to apply its transaction rules. +7. Preserve the conditional TIMED_OUT write, strongly consistent reread when that write loses, and the late-approval winner behavior. A decision committed before timeout must not be overwritten because the VM woke late. Test forward/backward clock changes, frozen monotonic time, missing/TTL-reaped row, approval at the boundary and cancellation. TTL is asynchronous garbage collection, not a precise alarm clock. +8. Declare `/suspend` and `/resume` as enabled image hooks only when the same source version serves them. Add shared hook-budget constants and route/contract assertions. Keep `/ready` and `/validate` AWS-silent. An old image without the new hooks must not be eligible for automatic suspension; enable policy only after deploying a compatible pinned image, with explicit capability/version gating if mixed versions can coexist. + +## 6. Wire the supervisor and human decisions + +### Orchestrator + +Implement the state/action table in a small testable policy/reconciliation helper called by the durable poll loop. Read current gate identity/status/deadline consistently. Record intent before requesting suspension; reread after uncertain outcomes and after suspend success to catch an approval that won concurrently. API acknowledgement is not final VM state: poll it. + +An approval can arrive before a pending suspend finishes. Even if an inline resume sees “already running,” the orchestrator must later notice that the machine became suspended and wake it. Do not clear durable wake intent merely because one API call appeared successful. + +When `ResumeMicrovm` succeeds but the VM is still RESUMING, keep reconciling; grant a bounded recovery interval rather than immediately applying a pre-suspend stale heartbeat. When the agent restores task RUNNING, use the fresh timestamp from prerequisite 1C. Never exempt genuine crashed RUNNING tasks indefinitely. + +Add consecutive MicroVM poll-error tracking. Reset on successful observations; classify permanent failures separately from transient ones. At the chosen threshold, perform a final consistent task read, record an explicit infrastructure failure and finalize/terminate through the existing single-owner path. Emit recovery/orphan diagnostics when termination itself fails; do not silently lose the handle. Keep suspend failures distinguishable from lost compute: failure to save money can leave a task safely awake, whereas failure to wake threatens correctness and needs bounded escalation. + +### Approve and deny handlers + +After the existing authorization checks and decision transaction **commit**, use a shared helper to load `compute_metadata` with a strongly consistent task read. Validate compute type, complete handle and current task/gate identity. For MicroVM, request resume with a short bound. No HTTP call to the guest is necessary. + +Missing handle, read failure, wrong/terminal state or resume failure must produce a warning and a structured resume-orphan event (include task ID, gate ID, VM ID when known, stage, reason and safe AWS request ID). Audit-event failure is also best-effort. **None of these post-commit failures may turn a successful decision into a 500 or undo the transaction.** Preserve the current response/status and cross-tenant/expired/wrong-gate protections. The poll loop is the repair path. + +### IAM and deployment + +Grant orchestrator SuspendMicrovm/ResumeMicrovm on the exact configured image ARN and required version suffix, alongside its existing lifecycle actions. Grant approve/deny ResumeMicrovm, and GetMicrovm only if the shared wake helper uses it, with the same image scope. Do not grant these actions to the agent execution role. Add no token-minting, broad role-passing or network ingress permission. + +Check `task-api.ts`'s lazy image-ARN wiring and no-image branch, bootstrap deployment-role coverage, tests/suppressions and CloudFormation resource counts. Image-hook changes and runtime hook serving must deploy together; automatic suspension remains off until the compatible image is ready. Provide an operational disable switch that stops **new suspends while still allowing resume, timeout handling and termination** for already-sleeping tasks. + +## 7. Acceptance matrix and completion gates + +| Case | Required result | +|---|---| +| Ordinary task, no approval | Same successful workflow as clean P2; no suspend calls | +| Short gate or decision during grace | No unnecessary freeze | +| Long gate + approve | VM sleeps, decision commits, same workspace and identity resume, allowed action runs once | +| Long gate + deny | VM wakes, denied action does not execute | +| No decision | Wake before deadline; agent-owned timeout/deny occurs on the original deadline | +| Deadline already passed while frozen | No restarted timeout; conditional late-decision race remains correct | +| Inline resume read/API/event failure | Decision API still reports its committed success; orphan signal and orchestrator repair | +| Approval races with suspend | No permanently stranded approved/denied task | +| Cancel during suspend/resume | Terminal task never becomes RUNNING again; VM terminated; slot released once | +| Duplicate/replayed lifecycle calls | No duplicate task/action and no changed approval deadline | +| Credentials expire during sleep | Resume refreshes all relevant credential consumers without losing tenant tags | +| Resume refresh/durability failure | Fail closed or remain safely awake; no unbounded hidden wait | +| VM dies or API polling repeatedly fails | Specific failure, bounded recovery, finalization/cleanup with retained handle | +| Unexpected suspension during RUNNING | Anomaly emitted once per episode; bounded wake/recovery rather than immediate false failure | +| Payload/network negatives | Wrong-task reads denied, truncated bytes rejected, no public ingress, documented port behavior | +| Other backends | AgentCore/ECS explicit unsupported responses; normal tasks/cancellation unchanged | +| Deploy/migrate/rollback | Compatible pinned image, valid bootstrap grants, every stack within limits, no accidental data deletion | + +Use unit tests for timers/state transitions and fault injection, integration tests for cross-record races and credential/client ownership, and real AWS runs for hook ordering, snapshot behavior, networking and permissions. Do not use a real hour-long sleep as the only clock test: simulate expired credentials locally, then run a controlled live long-suspend case when the service/session limits permit it. Record what was actually verified. + +For live evidence, retain redacted task IDs, commit/image/bootstrap versions, context flags, timeline, VM state transitions, approval timestamps, CloudWatch/progress evidence, workspace checksums or sentinel files, and proof of final termination/payload cleanup. Measure resume latency to tune grace/wake margin. Never include tokens, signed URLs or secret values in runbooks. + +**P3 is complete only when:** the full strategy interface and agent hooks ship; approval/deadline races and credential refresh pass; clean P2 and live P3 gates pass on the final deployment; behavior-changing follow-ups have passing regressions; generated docs, bootstrap artifacts and compatibility controls match; no test VM is left running/suspended; and #645/ADR/runbook status is updated with the actual evidence. An unverified gate must be called out explicitly, not converted to a checked box because unit tests passed. + +## Validation of this review branch + +Completed locally on 2026-09-13. These checks concern this cleanup/review branch, not the future P3 acceptance tests. + +- **246 CDK tests passed** across `lambda-microvm-compute`, `task-orchestrator`, `orchestrate-task-microvm` and `lambda-microvm-strategy` suites (`npx jest --runInBand --coverage=false` with those four paths). +- **104 Python MicroVM server tests passed** (`pytest tests/test_server.py -k microvm --no-cov`); passing this focused run does not resolve #841's fixture isolation issue. +- **The vault/MicroVM guard regression passed** in `stacks/agent.test.ts`; only its wording changed, and it still rejects the same combination. +- **Code comparison passed:** TypeScript output with comments removed is unchanged after accounting for the cdk-nag explanation and guard error-text corrections. Python syntax trees are identical after removing docstrings. No lifecycle logic, IAM grants or deployment topology changed. +- **Python formatting, Markdown source link checks, new artifact relative links and `git diff --check` passed.** +- **Documentation sync and build passed: 77 pages.** The installed mise version could not expand the build task's `:sync` dependency (`':task' pattern should be expanded before matching`), so the declared steps were executed in order with `mise run sync` followed by `mise exec -- ./node_modules/.bin/astro build`. Existing Astro/Cedar highlighting/deprecation warnings remain; they did not fail the build. +- **Offline nesting probe:** measured five configurations in current, naive-nested and parent-role/stable-name modes. The naive mode failed cycle validation; the corrected prototype passed. Probe output is linked in the review. No AWS resources were created or modified. + +The live P2 rerun, production nested-stack migration and P3 implementation/acceptance matrix remain future work. This branch supplies the cleanup, evidence and plan. diff --git a/docs/verification/645-p3-readiness-review.md b/docs/verification/645-p3-readiness-review.md new file mode 100644 index 000000000..4519b9749 --- /dev/null +++ b/docs/verification/645-p3-readiness-review.md @@ -0,0 +1,123 @@ +# ADR-021 takeover review: P3 readiness + +Reviewed 2026-09-13 against `main` commit `5e10038c7e28179b302ac4de78b709795aeba3ce`. + +This review covers the existing MicroVM implementation, related P2 follow-ups, comment accuracy and nested-stack feasibility. It includes local experiments, not an AWS deployment or a new live smoke run. The associated [implementation plan](./645-p3-implementation-plan.md) turns the findings into ordered work. + +## Start here: the pieces in plain language + +A **MicroVM** is a small, isolated computer rented from AWS. **Firecracker** is the technology that keeps these small computers separate. A **backend** is the kind of rented computer ABCA chooses to run a coding task. + +A **snapshot** is a saved picture of the computer's memory and disk. An **image** is the prepared starting snapshot: installed tools, server and warm files, without a particular user's task. Suspending saves the current task's computer so it can continue later. It is like closing a laptop halfway through homework. Suspending stops compute charges, but snapshot storage and read/write charges remain. The eight-hour session limit includes time asleep. + +The **orchestrator** is ABCA's supervisor. It starts computers, checks tasks, and cleans up. A **hook** is a small HTTP handler AWS calls at a lifecycle event, such as “prepare the image” or “about to stop.” These lifecycle calls do not require opening the agent to the public internet. + +**CloudFormation** is AWS's deployment system. A **stack** is a group of resources it creates together. **CDK** is code that produces the deployment recipe, called a **template**. A **construct** is a code-organizing box; it does not give its resources a separate CloudFormation quota. A **nested stack** does: it is a smaller deployment managed through the parent deployment. Nesting infrastructure does **not** mean running a VM inside another VM. This review evaluates nested CloudFormation stacks, not nested hardware virtualization. + +An **IAM role** is a permission badge. A **trust policy** says who may wear that badge. **PassRole** lets a caller hand a specific badge to an AWS service. A **session role** is the more tightly limited badge for one task. Its **tags** carry task/user/repository identity so access can be restricted to that task's data. + +## What is finished? + +| Phase | Purpose | Current state | +|---|---|---| +| P1 | Build the computer, start it, deliver a task, check it and stop it | Merged in [#689](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/689). Strategy, infrastructure, bootstrap permissions, packaging, types, `/ready` and `/run` exist. | +| P2 | Make a real coding task work with configuration, permissions, logs and progress | Merged in [#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733). `/validate`, `/terminate`, warm-up, runtime grants and heartbeat support exist. Two recorded tasks completed and opened PRs on August 7, using a manual IAM workaround. Permanent fixes still need a clean rerun. | +| P3 | Sleep during a human approval wait, wake correctly, and keep deadlines/credentials safe | Not implemented. AWS's suspend API was exercised manually; ABCA's integrated pause/resume lifecycle is missing. | +| P4 | — | ADR-021 defines no P4. Verification runbooks have their own numbered phases; those are not extra ADR milestones. | + +The old unchecked checklist and the word “proposed” do not erase the merged work. Conversely, merged code is not proof that the final deployment path works unattended. + +## Can MicroVM infrastructure be nested? + +**Yes. A local prototype works with a deliberate permission boundary. A one-line wrap is undeployable.** + +The naive change puts the existing `LambdaMicrovmCompute` construct inside a `NestedStack`. Two things go wrong: + +1. The parent `AgentSessionRole` trusts the child's MicroVM execution role. The child also needs the parent session role's ARN (AWS resource address). Each needs the other created first. This is a **circular dependency**. `app.synth()` wrote templates, but `Template.fromStack()` rejected the result as undeployable with `AgentSessionRole → MicrovmNested → AgentSessionRole` in the cycle. Merely producing a template is insufficient verification. +2. The construct sanitizes `Stack.of(this).stackName` for image/connector names. A nested stack's generated name is an unresolved CDK **token**, a placeholder filled in later. String sanitization destroyed that placeholder, producing names such as `--Token-TOKEN-8353---abca-agent`. Names also varied with construct creation order. + +The successful prototype kept the MicroVM **execution role in the parent**, beside the shared session role, and derived resource names from the stable parent name. The child contained image/build/network/bucket/log resources. This removed the cycle; it is not yet a production refactor. + +```mermaid +flowchart TB + Parent["Parent: shared platform + session role + MicroVM execution role"] + Child["Child: MicroVM image, build role, connectors, buckets and logs"] + Parent -->|"VPC and subnet inputs"| Child + Child -->|"resource addresses returned as outputs"| Parent +``` + +The arrows show configuration passing, not a circular creation dependency: parent resources that depend on child outputs must be separate from the parent resources the child needs to create itself. + +### Fresh measurements + +Offline synth, bundling disabled, `backgroundagent-dev`, example account `123456789012`, `us-east-1`, `suppressTemplateIndentation: true`. “Image” means a managed image with both `microvm_base_image_arn` and `microvm_base_image_version`. The vault guard was disabled **only in the scratch experiment** to count that combination. + +| Configuration | Current root resources | Prototype root resources | MicroVM child resources | +|---|---:|---:|---:| +| Default AgentCore | 454 | 454 | — | +| MicroVM, no image yet | 471 | 457 | 17 | +| MicroVM + managed image | 472 | 457 | 18 | +| Above + tool gateway | 479 | 464 | 18 | +| Above + Linear identity vault | 489 | 474 | 18 | + +The fullest root template measured 536,098 bytes; the prototype root measured 515,859 bytes, with a 31,884-byte MicroVM child. Existing registry children had 19 and 35 resources; the optional consent-page child had 12. These are counts per stack, not totals for the whole deployment. The full prototype saves **15 root resources** and about **20 KB**, leaving 26 root resource slots under the 500-resource limit. + +The historical 505-resource refusal and 98.6%-full byte claim are stale for this source revision. The guard still exists in shipped code, and these unbundled experiments do not justify silently removing it. A production refactor must also measure real bundled templates and all supported feature combinations. See [probe results](./645-nesting-probe-results.json). + +### Conditions for a production split + +- Add an explicit execution-role injection point and stable deployment-name input. Do not put `nestedStackParent` lookups throughout production code simply because the scratch probe used one. +- Preserve parent `Microvm*` output keys consumed by the packaging helper and CLI; return child identifiers through outputs. +- Recheck bootstrap IAM. The current MicroVM PassRole patterns match `backgroundagent-dev-LambdaMicrovmComputeBuild*` and `...Connector*`. Moving resources changes generated physical role names, which may no longer match. Use narrowly scoped stable names/patterns, regenerate the bundle and bump its version if permissions change. +- Check every cross-boundary grant, tags, solution user-agent aspect and all nested templates. Parent-only resource assertions no longer cover the child. +- Plan migration. Moving a resource to another stack changes its identity to CloudFormation and can cause replacement. Artifact/payload buckets currently use destructive removal settings; named images/connectors can also collide with their old copies. Review the actual change set and choose a fresh experimental deployment or a supported resource-preserving migration. Never assume moving CDK code moves live resources safely. + +Nesting is useful preparation, especially for the vault combination, but it does not implement P3 and is not a fundamental prerequisite for an isolated MicroVM pause/resume prototype. + +## Behavior findings that need follow-up + +“Confirmed” below means visible in current source or reproduced locally. It does not mean reproduced on AWS during this review. + +| Finding | Evidence and consequence | Treatment | +|---|---|---| +| Finalize deletes without permission | `orchestrate-task.ts` calls `deleteMicrovmPayload`; `task-orchestrator.ts` grants only PutObject. Tests explicitly assert deletion permission is absent. Prompt data remains until lifecycle deletion when the call is denied. | Confirmed; [#817](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/817). Add narrowly scoped coordinator delete permission and reverse the wrong assertions. | +| Raw AWS reason text changes error classification | Executing the real classifier with a substrate-completed message plus `MicroVM host unavailable.` produces unsupported-region/non-retryable classification. Hook HTTP 400 is already classified correctly; that old example is stale. | Reproduced locally; #817. Use structured category/code for decisions and preserve service text as diagnostics. | +| ARN validation is consistency, not deployment identity | `_reject_foreign_arns` accepts another workspace's secret in the same account. Its anchor comes from the same supplied block. It checks partition/account agreement, not trusted provenance. | Local validator proof; #817. No forged `/run` reachability or credential theft demonstrated. Trusted deployment binding and negative ingress tests are needed. | +| Payload reader is broader than one task | Worker execution roles can read/list the payload bucket, so prompt data for other tasks may be accessible to untrusted task code. | Confirmed; [#700](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/700), shared with ECS. Task-scoped transport design is separate from deleting completed payloads. | +| A long approval wait leaves a stale heartbeat on resume | `write_heartbeat` only writes while status is RUNNING. `transact_resume_from_approval` restores RUNNING without refreshing the timestamp. A poll before the next 45-second tick can see an age over 240 seconds and fail a healthy task. | Source-proven race window, no live reproduction. Add atomic refresh plus an adversarial ordering test before P3. | +| Heartbeat is not a general progress watchdog | It runs on an independent thread. A stuck coding thread can coexist with fresh heartbeats. | Confirmed by call structure. Corrected comments; broader progress detection belongs with [#491](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/491). | +| Start retries have an uncertain-outcome window | `startSessionWithRetry` calls start again on transient errors. No application-stable MicroVM client token links those calls. Lost success response does not prove the first VM was never created. | Source risk, not a demonstrated live leak. Add response-loss fault injection and attempt-scoped idempotency/reconciliation design. Corrected the false guarantee. | +| Snapshot refresh must preserve task identity | `reset_session_cache()` clears tenant tags as well as the session; existing clients and the Claude credential helper can hold separate state. | Confirmed. It is a test reset, not a ready-made `/resume` implementation. | +| Existing progress writes do not provide a suspend barrier | `ProgressWriter._put_event` writes synchronously but catches/drops failures and can disable itself. There is no acknowledged queue-flush contract. | Confirmed. P3 must add a narrow durability barrier with a failure path. | +| Repeated MicroVM poll errors never escalate | The MicroVM branch logs and continues; ECS already has counters. | Confirmed P3 gap. Persist retry counters in durable poll state and bound recovery. | +| Logging failure count is invisible | `_debug_cw_failures` is incremented but never read/exported; `_DEBUG_CW_FAILURE_EMIT_EVERY` is unused. | Confirmed; [#810](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/810). Comments corrected; telemetry behavior remains work. | +| Server tests can leak background work | Test setup clears `_active_threads` without ensuring the threads finished; late work can use the next test's mocks. | Confirmed structure; [#841](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/841). Fix before trusting expanded hook tests. | + +The security findings are readiness work, not cosmetic cleanup. This review does not silently implement them or claim that their absence makes every existing deployment exploitable. + +### P2 issue map + +- **#817 is the primary tracker.** [#813](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/813), [#814](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/814), [#815](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/815) and [#816](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/816) closed because they were consolidated, not because all fixes landed. Its five tracks are classifier correctness, payload deletion IAM, contract/provenance checks, stale docs, and truncated-S3-body route coverage. `_PayloadFetchError` already distinguishes unreadable payloads, but the real bad-byte path needs a route-level regression. +- **[#818](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/818): registry tool networking and payload size.** Runtime networking is HTTPS/443-only on all three shipped backends, not uniquely MicroVM. Correct the premise, define supported tool ports, and test oversized resolved registry assets through the 4,096-byte/S3 path. +- **[#701](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/701): bootstrap refresh.** Checking the source version is insufficient; verify the deployed bundle and effective roles. [#867](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/867) fixed bootstrap template byte size, a different problem. +- **[#857](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/857): vault + MicroVM guard.** It is on main. [#854](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/854) reclaimed room; this review's newer measurements supersede old counts, not deployment verification. +- **[#811](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/811): ECS Haiku model environment parity.** Separate backend fix; not a reason to block MicroVM pause/resume. +- **[#702](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/702): teardown leaks.** Account for AgentCore ENIs (network attachments) and Memory deletion state when cleaning the test deployment. Separate platform operations work. +- **[#736](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/736): future IAM conditions.** Revisit when the service supports suitable context keys. Do not restore the previously broken `iam:PassedToService` condition just to make policies look tighter. + +## Cleanup applied in this branch + +Changes are comments, Python docstrings, documentation, one cdk-nag explanation string and the vault guard’s error wording. The error no longer claims a current 505-resource overflow; the guard still rejects exactly the same combination. The cdk-nag change affects template metadata, not IAM permissions. + +- Describe P2's successful workaround-assisted smoke and the remaining clean verification precisely; retain the stable warning ID. +- Remove the false no-delete-by-design explanation and point to the missing permission. +- Correct heartbeat, retry-idempotency, progress-durability and suspended-quota claims. +- Correct unexported logging-counter claims and the region/credential distinction. An offline probe on boto3/botocore 1.43.78 changed `AWS_DEFAULT_REGION`: new clients used the new region while the resolved credentials object remained cached. +- Explain that missing `platform_config` is accepted only when the effective environment already has required identifiers; describe the limits of ARN consistency validation. +- Fix stale ECS sizing/context instructions, root/nested deployment wording, old template counts, and PRNG terminology. Reseeding Python `random` does not make it suitable for secrets. + +## Review coverage and limits + +Reviewed the MicroVM construct, strategy, shared strategy interface, orchestrator start/poll/finalize paths, approval handlers and task API grant seams; the session role and bootstrap policy; agent hook dispatch/config installation, heartbeat, approval transactions/timers, credentials and progress writer; relevant tests/contracts; ADR, compute/orchestrator/deployment docs, packaging and recorded P1/P2 runbooks. + +Local checks and results are recorded in the implementation plan's review-validation section. Source review and mocked tests cannot establish AWS hook ordering, snapshot clock behavior, credential refresh after a long freeze, actual network reachability, or a safe migration of deployed resources. Those remain explicit live gates rather than claims of completion. From b5155928f2bf32e253368ffc400632b78b22c745 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Sun, 13 Sep 2026 14:55:11 -0400 Subject: [PATCH 002/149] fix(agent): drain server test threads before state reset (#841) --- agent/tests/test_server.py | 63 ++++++++++++++---- agent/tests/test_server_isolation.py | 98 ++++++++++++++++++++++++++++ 2 files changed, 150 insertions(+), 11 deletions(-) create mode 100644 agent/tests/test_server_isolation.py diff --git a/agent/tests/test_server.py b/agent/tests/test_server.py index b1bbf28cd..10e839037 100644 --- a/agent/tests/test_server.py +++ b/agent/tests/test_server.py @@ -19,15 +19,48 @@ import server -@pytest.fixture(autouse=True) -def reset_server_state(): - server._background_pipeline_failed = False +def _join_server_threads(timeout: float = 5.0) -> None: + """Reap tracked work before restoring test state; retain leaked handles on failure.""" + deadline = time.monotonic() + timeout + with server._threads_lock: + threads = list(server._active_threads) + # The pipeline may need _threads_lock to finish; never join while holding it. + for thread in threads: + if thread.is_alive(): + thread.join(timeout=max(0.0, deadline - time.monotonic())) with server._threads_lock: + leaked = [thread.name for thread in server._active_threads if thread.is_alive()] + if leaked: + pytest.fail(f"Server pipeline threads did not exit within {timeout}s: {leaked}") server._active_threads.clear() - yield + + +@pytest.fixture(autouse=True) +def reset_server_state(monkeypatch, env_guard): + # Dependencies force this teardown to precede mock/environment restoration. + # A still-starting pipeline resolves run_task from the module at call time. + _join_server_threads() server._background_pipeline_failed = False + try: + yield + finally: + _join_server_threads() + server._background_pipeline_failed = False + + +def test_server_thread_cleanup_keeps_leaked_handles_and_reports_their_names(): + release = threading.Event() + thread = threading.Thread(target=release.wait, name="deliberately-blocked-pipeline") + thread.start() with server._threads_lock: - server._active_threads.clear() + server._active_threads.append(thread) + try: + with pytest.raises(pytest.fail.Exception, match="deliberately-blocked-pipeline"): + _join_server_threads(timeout=0) + assert thread in server._active_threads + finally: + release.set() + thread.join(timeout=5) @pytest.fixture @@ -1748,7 +1781,7 @@ def baked_platform_env(env_guard): @pytest.fixture -def env_guard(): +def env_guard(monkeypatch): """Snapshot/restore ``os.environ`` around a test that installs into it. ``_install_platform_config`` writes to the REAL process environment (that is @@ -1756,6 +1789,10 @@ def env_guard(): without this, one platform_config test would leak table names and a bogus ``AGENT_SESSION_ROLE_ARN`` into every test that runs after it (the conftest ``_clean_env`` fixture only strips the subset it knows about). + + All server tests use this through reset_server_state. Depending on monkeypatch + keeps its original-value restoration last, after joining work and restoring + direct environment writes. """ before = dict(os.environ) yield @@ -2359,9 +2396,13 @@ def test_the_two_codes_are_distinct(self, env_guard): class TestMicrovmRunHookPlatformConfig: """``platform_config`` arrives on the ``/run`` hook as a SIBLING of ``agent_payload``.""" + @pytest.fixture(autouse=True) + def _task_identity(self, request): + self.task_id = f"t-pc-{request.node.name}" + def _payload(self, **extra) -> dict: return { - "task_id": "t-pc", + "task_id": self.task_id, "repo_url": "org/repo", "prompt": "do it", "github_token": "ghp_x", @@ -2950,14 +2991,14 @@ def test_an_unreadable_thread_count_reports_None_not_a_confident_zero( # COUNT is the only way to reach it, which is why patching `_debug_cw` (the # existing test above) cannot — by then `active` is already an int. class _UnreadableThreadList(list): - """Raises when COUNTED, but still clearable by the reset fixture.""" + """Simulate an unreadable registry only during the hook call.""" def __iter__(self): raise RuntimeError("thread registry read exploded") - monkeypatch.setattr(server, "_active_threads", _UnreadableThreadList()) - - r = client.post(TERMINATE_HOOK, json={"microvmId": "m-unknown"}) + with monkeypatch.context() as patch: + patch.setattr(server, "_active_threads", _UnreadableThreadList()) + r = client.post(TERMINATE_HOOK, json={"microvmId": "m-unknown"}) assert r.status_code == 200 body = r.json() diff --git a/agent/tests/test_server_isolation.py b/agent/tests/test_server_isolation.py new file mode 100644 index 000000000..0939d4aab --- /dev/null +++ b/agent/tests/test_server_isolation.py @@ -0,0 +1,98 @@ +"""Exercise server fixture teardown through a real, isolated pytest run.""" + +from __future__ import annotations + +import os +import subprocess +import sys +import textwrap +from pathlib import Path + + +def test_pipeline_finishes_before_test_mocks_and_environment_are_restored(tmp_path: Path): + """Force a pipeline to resolve run_task only when fixture teardown joins it.""" + test_file = tmp_path / "test_fixture_order.py" + test_file.write_text( + textwrap.dedent( + """ + import os + import threading + from unittest.mock import MagicMock + + import server + from test_server import env_guard, reset_server_state + + release = threading.Event() + entered = threading.Event() + first_run = MagicMock() + observed_env = [] + pipeline = None + original_join = None + + def test_first(monkeypatch, env_guard): + global pipeline, original_join + os.environ["ABCA_TEST_THREAD_ORIGIN"] = "first-test" + monkeypatch.setattr(server, "_debug_cw", lambda *a, **kw: None) + monkeypatch.setattr(server, "run_task", first_run) + + def delayed_lookup(**kwargs): + entered.set() + if release.wait(timeout=10): + observed_env.append(os.environ.get("ABCA_TEST_THREAD_ORIGIN")) + server.run_task(**kwargs) + + monkeypatch.setattr(server, "_run_task_background", delayed_lookup) + pipeline = server._spawn_background({"task_id": "fixture-first"}) + assert entered.wait(timeout=5) + original_join = pipeline.join + + def join_and_release(timeout=None): + release.set() + original_join(timeout=timeout) + + # Teardown must use this test's join and run_task before monkeypatch + # restores either. No timing-dependent sleep is needed. + monkeypatch.setattr(pipeline, "join", join_and_release) + + def test_second(monkeypatch): + next_run = MagicMock() + monkeypatch.setattr(server, "run_task", next_run) + try: + assert not pipeline.is_alive(), "prior test leaked a live pipeline" + first_run.assert_called_once_with(task_id="fixture-first") + assert observed_env == ["first-test"] + assert "ABCA_TEST_THREAD_ORIGIN" not in os.environ + assert server._active_threads == [] + finally: + # Also reap the deliberately blocked worker when testing the + # broken fixture, without ever invoking the real pipeline. + release.set() + original_join(timeout=5) + next_run.assert_not_called() + """ + ) + ) + agent_dir = Path(__file__).resolve().parents[1] + env: dict[str, str] = dict(os.environ) + env["PYTHONPATH"] = os.pathsep.join((str(agent_dir / "src"), str(agent_dir / "tests"))) + env.pop("ABCA_TEST_THREAD_ORIGIN", None) + result = subprocess.run( + [ + sys.executable, + "-m", + "pytest", + "-q", + "--no-cov", + "-c", + str(agent_dir / "pyproject.toml"), + str(test_file), + ], + cwd=agent_dir, + env=env, + text=True, + capture_output=True, + timeout=20, + check=False, + ) + assert result.returncode == 0, result.stdout + result.stderr + assert "2 passed" in result.stdout From 564ccc19233d9378362f6bf18344a5125c03d87c Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Sun, 13 Sep 2026 14:55:16 -0400 Subject: [PATCH 003/149] fix(compute): allow scoped MicroVM payload cleanup (#817) --- cdk/src/constructs/lambda-microvm-compute.ts | 7 +++--- cdk/src/constructs/task-orchestrator.ts | 18 +++++++++----- cdk/src/handlers/orchestrate-task.ts | 4 ++-- .../strategies/lambda-microvm-strategy.ts | 24 ++++++++----------- .../constructs/lambda-microvm-compute.test.ts | 2 +- cdk/test/constructs/task-orchestrator.test.ts | 15 ++++++++++-- cdk/test/stacks/agent.test.ts | 13 ++++++---- 7 files changed, 50 insertions(+), 33 deletions(-) diff --git a/cdk/src/constructs/lambda-microvm-compute.ts b/cdk/src/constructs/lambda-microvm-compute.ts index a0d7bc716..ab9e31d14 100644 --- a/cdk/src/constructs/lambda-microvm-compute.ts +++ b/cdk/src/constructs/lambda-microvm-compute.ts @@ -47,10 +47,9 @@ import { LAMBDA_MICROVM_SUPPORTED_REGIONS, isLambdaMicrovmRegionSupported } from /** * Lifecycle expiry for MicroVM `/run` hook payloads, in days. * - * Mirrors {@link ECS_PAYLOAD_TTL_DAYS}. Finalization attempts to delete the - * object identified by `microvmPayloadKey`, but the orchestrator is still - * missing its DeleteObject grant (#817). Lifecycle expiry is the current - * fallback; S3 processes expiry asynchronously, not exactly 24 hours after + * Mirrors {@link ECS_PAYLOAD_TTL_DAYS}. Finalization deletes the object + * identified by `microvmPayloadKey`; lifecycle expiry is the fallback if that + * step fails. S3 processes expiry asynchronously, not exactly 24 hours after * upload. Payloads carry hydrated prompt context and are read once at `/run`. */ export const MICROVM_PAYLOAD_TTL_DAYS = 1; diff --git a/cdk/src/constructs/task-orchestrator.ts b/cdk/src/constructs/task-orchestrator.ts index b6c547587..3243f4ddb 100644 --- a/cdk/src/constructs/task-orchestrator.ts +++ b/cdk/src/constructs/task-orchestrator.ts @@ -338,9 +338,9 @@ export interface TaskOrchestratorProps { /** * Bucket for `/run` payloads that exceed the 4 KB `runHookPayload` cap — * i.e. nearly all of them, since a hydrated payload is bigger than that. - * The current grant is **write only**. Finalization also attempts deletion, - * so the missing DeleteObject permission is a defect tracked in #817; - * lifecycle expiry is the fallback until that grant is corrected. + * The orchestrator uploads payloads and deletes `/payload.json` at + * finalization. It does not read them; the execution role is the reader. + * Lifecycle expiry is the fallback if finalization or deletion fails. */ readonly payloadBucket: s3.IBucket; }; @@ -524,10 +524,16 @@ export class TaskOrchestrator extends Construct { } // ADR-021: upload oversized /run payloads; the execution role is the reader. - // Finalization calls deleteMicrovmPayload, but this grant is missing delete - // permission (#817). The lifecycle rule currently provides the fallback. + // Match deleteMicrovmPayload's /payload.json key shape (#817). + // grantDelete would also grant DeleteObjectVersion; this unversioned bucket + // needs only DeleteObject, and no delete permission belongs on the worker. if (props.microvmConfig) { props.microvmConfig.payloadBucket.grantPut(this.fn); + this.fn.addToRolePolicy(new iam.PolicyStatement({ + sid: 'MicrovmDeletePayload', + actions: ['s3:DeleteObject'], + resources: [props.microvmConfig.payloadBucket.arnForObjects('*/payload.json')], + })); } // Durable execution managed policy @@ -797,7 +803,7 @@ export class TaskOrchestrator extends Construct { }, { id: 'AwsSolutions-IAM5', - reason: 'DynamoDB index/* wildcards generated by CDK grantReadWriteData; AgentCore runtime/* required for sub-resource invocation; Secrets Manager wildcards generated by CDK grantRead; AgentCore Memory wildcards generated by CDK grantRead/grantWrite; ECS RunTask/DescribeTasks/StopTask conditioned on cluster ARN; iam:PassRole scoped to ECS task/execution roles and conditioned on ecs-tasks.amazonaws.com; S3 object/* wildcard from CDK grantPut on the dedicated MicroVM payload bucket; MicroVM lifecycle actions (RunMicrovm/GetMicrovm/TerminateMicrovm) are scoped to the single platform MicroVM image ARN plus a :* version-suffix sibling (every one of them authorizes against the image resource, not the per-session instance; no account-wide wildcard is used); lambda:PassNetworkConnector requires Resource:* because the action supports no resource-level permissions and the AWS-managed connectors live outside this account; iam:PassRole is scoped to the exact MicroVM execution role without iam:PassedToService (ADR-021 P2r2-F10); Agent Registry read scoped to the wired registry ARN, with a record/* suffix wildcard because record ids are server-assigned and unknown at synth (#246)', + reason: 'DynamoDB index/* wildcards generated by CDK grantReadWriteData; AgentCore runtime/* required for sub-resource invocation; Secrets Manager wildcards generated by CDK grantRead; AgentCore Memory wildcards generated by CDK grantRead/grantWrite; ECS RunTask/DescribeTasks/StopTask conditioned on cluster ARN; iam:PassRole scoped to ECS task/execution roles and conditioned on ecs-tasks.amazonaws.com; S3 object/* wildcard from CDK grantPut on the dedicated MicroVM payload bucket; MicroVM cleanup DeleteObject restricted to */payload.json in that bucket; MicroVM lifecycle actions (RunMicrovm/GetMicrovm/TerminateMicrovm) are scoped to the single platform MicroVM image ARN plus a :* version-suffix sibling (every one of them authorizes against the image resource, not the per-session instance; no account-wide wildcard is used); lambda:PassNetworkConnector requires Resource:* because the action supports no resource-level permissions and the AWS-managed connectors live outside this account; iam:PassRole is scoped to the exact MicroVM execution role without iam:PassedToService (ADR-021 P2r2-F10); Agent Registry read scoped to the wired registry ARN, with a record/* suffix wildcard because record ids are server-assigned and unknown at synth (#246)', }, ], true); } diff --git a/cdk/src/handlers/orchestrate-task.ts b/cdk/src/handlers/orchestrate-task.ts index 44311a083..f80852981 100644 --- a/cdk/src/handlers/orchestrate-task.ts +++ b/cdk/src/handlers/orchestrate-task.ts @@ -462,8 +462,8 @@ const durableHandler: DurableExecutionHandler = asyn // keys are `/payload.json`, and the guest runs untrusted repo code — // so a TTL-only reaper left every finished task's hydrated prompt readable by // any concurrently running MicroVM until asynchronous lifecycle deletion. - // The current MicroVM delete grant is still missing (#817); task-scoped - // payload reads are a separate improvement (#700). + // Task-scoped payload reads are a separate improvement (#700); deleting + // completed payloads does not isolate other tasks that are still active. if (blueprintConfig.compute_type === 'ecs') { await deleteEcsPayload(taskId); } else if (blueprintConfig.compute_type === 'lambda-microvm') { diff --git a/cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts b/cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts index a31224d35..dc2dfb4d5 100644 --- a/cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts +++ b/cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts @@ -385,20 +385,16 @@ export function microvmPayloadKey(taskId: string): string { /** * Delete a task's MicroVM ``/run`` payload object. Best-effort: a failed delete - * must never fail the task — the bucket's 1-day lifecycle rule reaps it - * regardless. Called from the orchestrator's ``finalize`` step once the task is - * terminal. No-ops when the payload bucket isn't configured. - * - * WHY this exists rather than leaning on the TTL alone (ECS parity, and a real - * exposure delta): the execution role's payload-bucket grant is - * ``grantRead`` on the WHOLE bucket — it cannot be per-task scoped, because the - * MicroVM has to read its own object before any tenant identity is installed. - * The key shape is ``/payload.json``, so any running MicroVM that knows - * (or guesses) another task's id can read that task's HYDRATED PROMPT — issue - * body, comment thread, repo context. The MicroVM also runs untrusted repo code. - * Relying only on ``MICROVM_PAYLOAD_TTL_DAYS = 1`` left that window open for up - * to ~24 h; deleting at finalize closes it to the task's own lifetime, which is - * exactly the posture ``deleteEcsPayload`` already gives the ECS backend. + * must never fail the task; the bucket's 1-day lifecycle rule is the fallback. + * Called from the orchestrator's ``finalize`` step once the task is terminal. + * No-ops when the payload bucket isn't configured. + * + * The execution role currently reads the whole payload bucket before task-scoped + * credentials are established. Untrusted task code can therefore read other + * tasks' payloads where it knows their keys. Deleting completed payloads shortens + * that exposure; it does not isolate active tasks. #700 tracks task-scoped + * transport. S3 processes lifecycle expiry asynchronously, so it is not an exact + * 24-hour bound on retention if this deletion fails. * * ISSUED UNCONDITIONALLY, including for a task whose payload went INLINE (under * the {@link RUN_HOOK_PAYLOAD_LIMIT_BYTES} cap, so no object was ever written). diff --git a/cdk/test/constructs/lambda-microvm-compute.test.ts b/cdk/test/constructs/lambda-microvm-compute.test.ts index 133816282..d56f55336 100644 --- a/cdk/test/constructs/lambda-microvm-compute.test.ts +++ b/cdk/test/constructs/lambda-microvm-compute.test.ts @@ -575,7 +575,7 @@ describe('LambdaMicrovmCompute — image provisioned from a managed base image', } }); - test('payload bucket expires objects (the ONLY reaper on this backend)', () => { + test('payload bucket expires objects as a fallback for failed finalization', () => { template.hasResourceProperties('AWS::S3::Bucket', { LifecycleConfiguration: { Rules: Match.arrayWith([ diff --git a/cdk/test/constructs/task-orchestrator.test.ts b/cdk/test/constructs/task-orchestrator.test.ts index 6ebd6cc34..7cb64500f 100644 --- a/cdk/test/constructs/task-orchestrator.test.ts +++ b/cdk/test/constructs/task-orchestrator.test.ts @@ -803,7 +803,7 @@ describe('TaskOrchestrator with the Lambda MicroVMs backend (ADR-021)', () => { expect(actions.has('lambda:CreateMicrovmShellAuthToken')).toBe(false); }); - test('gets write on the payload bucket but NOT delete (lifecycle rule is the reaper)', () => { + test('can upload payloads and delete only task payload objects at finalize', () => { const payloadStatements = Object.values(template.findResources('AWS::IAM::Policy')) .flatMap(p => p.Properties.PolicyDocument.Statement as Array<{ Action: string | string[]; @@ -813,7 +813,18 @@ describe('TaskOrchestrator with the Lambda MicroVMs backend (ADR-021)', () => { const actions = payloadStatements.flatMap(s => Array.isArray(s.Action) ? s.Action : [s.Action]); expect(actions).toContain('s3:PutObject'); - expect(actions).not.toContain('s3:DeleteObject'); + expect(actions).toContain('s3:DeleteObject'); + expect(actions.some(action => action.startsWith('s3:Get') || action.startsWith('s3:List'))).toBe(false); + const deletes = payloadStatements.filter(s => + (Array.isArray(s.Action) ? s.Action : [s.Action]).some(action => action.startsWith('s3:Delete'))); + expect(deletes).toHaveLength(1); + expect(deletes[0]!.Action).toBe('s3:DeleteObject'); + expect(deletes[0]!.Resource).toEqual({ + 'Fn::Join': ['', [ + { 'Fn::GetAtt': [expect.stringMatching(/^MicrovmPayloadBucket/), 'Arn'] }, + '/*/payload.json', + ]], + }); }); test('adds no MicroVM statements when microvmConfig is omitted', () => { diff --git a/cdk/test/stacks/agent.test.ts b/cdk/test/stacks/agent.test.ts index 14c001cdb..0049b1c5b 100644 --- a/cdk/test/stacks/agent.test.ts +++ b/cdk/test/stacks/agent.test.ts @@ -1190,7 +1190,7 @@ describe('AgentStack with the Lambda MicroVMs substrate gate (--context compute_ expect(rendered).not.toContain('lambda:ConnectMicrovm'); }); - test('orchestrator may WRITE the payload bucket; nothing grants it delete', () => { + test('orchestrator may upload payloads and delete task payload objects at finalize', () => { const policies = Object.entries(template.findResources('AWS::IAM::Policy')) .filter(([id]) => id.includes('TaskOrchestrator')); const statements = policies.flatMap(([, p]) => p.Properties.PolicyDocument.Statement as Array<{ @@ -1202,9 +1202,14 @@ describe('AgentStack with the Lambda MicroVMs substrate gate (--context compute_ const actions = payloadStatements.flatMap(s => Array.isArray(s.Action) ? s.Action : [s.Action]); expect(actions).toContain('s3:PutObject'); - // The bucket's lifecycle rule is the reaper on this backend — unlike the ECS - // path the orchestrator never deletes, so the grant must not exist. - expect(actions).not.toContain('s3:DeleteObject'); + expect(actions).toContain('s3:DeleteObject'); + const deletion = payloadStatements.find(s => s.Action === 's3:DeleteObject'); + expect(deletion!.Resource).toEqual({ + 'Fn::Join': ['', [ + { 'Fn::GetAtt': [expect.stringMatching(/^LambdaMicrovmComputePayloadBucket/), 'Arn'] }, + '/*/payload.json', + ]], + }); }); test('cancel Lambda may terminate a MicroVM (and only terminate), image-scoped', () => { From 5490d6456e9094379b03781c660999d1946ab91c Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Sun, 13 Sep 2026 14:55:20 -0400 Subject: [PATCH 004/149] fix(agent): refresh heartbeat atomically after approval (#645) --- agent/src/task_state.py | 8 +++++- agent/tests/test_task_state.py | 22 +++++++++++++++ cdk/test/handlers/orchestrate-task.test.ts | 31 ++++++++++++++++++++++ 3 files changed, 60 insertions(+), 1 deletion(-) diff --git a/agent/src/task_state.py b/agent/src/task_state.py index bf40d712f..980ff7da1 100644 --- a/agent/src/task_state.py +++ b/agent/src/task_state.py @@ -746,6 +746,10 @@ def transact_resume_from_approval( - resuming with a stale request_id after a race with the reconciler / a concurrent approval. + Refresh the heartbeat in the same update: writes pause during approval waits, + so restoring RUNNING with the old timestamp could let an orchestrator poll + mark this healthy task lost before the next periodic heartbeat. + Raises ``ApprovalResumeError`` on ``TransactionCanceledException`` so the hook can emit ``approval_resume_failed`` + DENY. """ @@ -760,7 +764,8 @@ def transact_resume_from_approval( "TableName": task_table, "Key": {"task_id": {"S": task_id}}, "UpdateExpression": ( - "SET #s = :running REMOVE awaiting_approval_request_id" + "SET #s = :running, agent_heartbeat_at = :heartbeat " + "REMOVE awaiting_approval_request_id" ), "ConditionExpression": ( "#s = :awaiting AND awaiting_approval_request_id = :rid" @@ -770,6 +775,7 @@ def transact_resume_from_approval( ":running": {"S": _STATUS_RUNNING}, ":awaiting": {"S": _STATUS_AWAITING_APPROVAL}, ":rid": {"S": request_id}, + ":heartbeat": {"S": _now_iso()}, }, } } diff --git a/agent/tests/test_task_state.py b/agent/tests/test_task_state.py index b0eef03d9..04bc8aa25 100644 --- a/agent/tests/test_task_state.py +++ b/agent/tests/test_task_state.py @@ -807,6 +807,28 @@ def test_condition_failed_reason_includes_both_branches( class TestTransactResumeFromApproval: + def test_resume_refreshes_heartbeat_in_the_same_conditional_write( + self, approval_tables_env, monkeypatch + ): + """A poll immediately after a long approval wait must see a fresh heartbeat.""" + monkeypatch.setattr(task_state, "_now_iso", lambda: "2026-09-13T12:05:00Z") + client = MagicMock() + + task_state.transact_resume_from_approval("01KTASK", "01KREQ", client=client) + + client.transact_write_items.assert_called_once() + updates = client.transact_write_items.call_args.kwargs["TransactItems"] + assert len(updates) == 1 + update = updates[0]["Update"] + assert "agent_heartbeat_at = :heartbeat" in update["UpdateExpression"] + assert "#s = :running" in update["UpdateExpression"] + assert update["ExpressionAttributeValues"][":heartbeat"] == {"S": "2026-09-13T12:05:00Z"} + assert update["ConditionExpression"] == ( + "#s = :awaiting AND awaiting_approval_request_id = :rid" + ) + # An independent write would leave the old heartbeat visible after RUNNING. + client.update_item.assert_not_called() + def test_env_missing_raises(self, monkeypatch): monkeypatch.delenv("TASK_TABLE_NAME", raising=False) monkeypatch.delenv("TASK_APPROVALS_TABLE_NAME", raising=False) diff --git a/cdk/test/handlers/orchestrate-task.test.ts b/cdk/test/handlers/orchestrate-task.test.ts index be9901081..297833ff4 100644 --- a/cdk/test/handlers/orchestrate-task.test.ts +++ b/cdk/test/handlers/orchestrate-task.test.ts @@ -548,6 +548,37 @@ describe('hydrateAndTransition — Cedar HITL payload threading', () => { }); describe('pollTaskStatus', () => { + test.each(['agentcore', 'lambda-microvm'] as const)( + '%s stays healthy immediately after a long approval wait resumes', + async (computeType) => { + const beforeApproval = new Date(Date.now() - 600_000).toISOString(); + mockDdbSend.mockResolvedValueOnce({ + Item: { + status: 'AWAITING_APPROVAL', + session_id: 'sess-1', + started_at: beforeApproval, + agent_heartbeat_at: beforeApproval, + }, + }); + const waiting = await pollTaskStatus('TASK001', { attempts: 5 }, computeType); + expect(waiting.sessionUnhealthy).toBe(false); + + // The agent's conditional resume transaction refreshes this timestamp with + // status RUNNING. Poll before the independent 45-second worker ticks. + mockDdbSend.mockResolvedValueOnce({ + Item: { + status: 'RUNNING', + session_id: 'sess-1', + started_at: beforeApproval, + agent_heartbeat_at: new Date().toISOString(), + }, + }); + const resumed = await pollTaskStatus('TASK001', waiting, computeType); + expect(resumed.lastStatus).toBe('RUNNING'); + expect(resumed.sessionUnhealthy).toBe(false); + }, + ); + test('increments attempt count and reads status', async () => { mockDdbSend.mockResolvedValueOnce({ Item: { status: 'RUNNING' } }); const result = await pollTaskStatus('TASK001', { attempts: 5 }, 'agentcore'); From 3f6671bf4b95170452debd47d4b89b963209e40c Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Sun, 13 Sep 2026 14:57:07 -0400 Subject: [PATCH 005/149] docs(645): record completed MicroVM prerequisite fixes --- .../ADR-021-lambda-microvms-compute-backend.md | 2 +- docs/design/ORCHESTRATOR.md | 2 ++ .../content/docs/architecture/Orchestrator.md | 2 ++ .../Adr-021-lambda-microvms-compute-backend.md | 2 +- .../verification/645-p3-implementation-plan.md | 18 ++++++++++++++++-- docs/verification/645-p3-readiness-review.md | 2 ++ 6 files changed, 24 insertions(+), 4 deletions(-) diff --git a/docs/decisions/ADR-021-lambda-microvms-compute-backend.md b/docs/decisions/ADR-021-lambda-microvms-compute-backend.md index 3bd397e5b..764a02ef6 100644 --- a/docs/decisions/ADR-021-lambda-microvms-compute-backend.md +++ b/docs/decisions/ADR-021-lambda-microvms-compute-backend.md @@ -70,7 +70,7 @@ The `ComputeStrategy` interface gains **mandatory** `suspendSession(handle)` / ` **Liveness on this backend combines substrate state and the agent heartbeat.** `GetMicrovm` reports whether the VM is running or terminal. The independent `_heartbeat_worker` in `agent/src/server.py` refreshes the task timestamp every 45 seconds while the task is `RUNNING`; a stale timestamp detects loss of that writer, including a process crash or inability to write DynamoDB. It does **not** prove pipeline progress: a hung coding thread can coexist with a healthy heartbeat thread. A failed `/run` hook can trigger service teardown; after an accepted hook, ABCA still needs explicit termination and the maximum-duration backstop. A general progress watchdog remains separate work. -AgentCore and MicroVM tasks start the same periodic heartbeat worker, so they use the same 120-second grace and 240-second stale thresholds. ECS starts the pipeline directly, bypassing `server.py`; its pipeline writes a timestamp once but does not start this worker. Enabling the same stale check for ECS would therefore fail healthy tasks after roughly four minutes. The check applies only to `RUNNING`, not `AWAITING_APPROVAL`. The transition back to `RUNNING` currently leaves the old timestamp intact: after a long gate, a poll can see a stale heartbeat before the next worker tick. Refreshing it atomically on approval resume is a P3 prerequisite in the implementation plan. +AgentCore and MicroVM tasks start the same periodic heartbeat worker, so they use the same 120-second grace and 240-second stale thresholds. ECS starts the pipeline directly, bypassing `server.py`; its pipeline writes a timestamp once but does not start this worker. Enabling the same stale check for ECS would therefore fail healthy tasks after roughly four minutes. The check applies only to `RUNNING`, not `AWAITING_APPROVAL`. The transition back to `RUNNING` refreshes `agent_heartbeat_at` in the same conditional transaction that clears the matching approval request. A poll immediately after a long gate therefore sees a fresh heartbeat before the next worker tick. P3 still needs the separate suspend/resume lifecycle and recovery policy described below. The service's `MicrovmState` enum has **six** members, not three, so the mapping is stated exhaustively (one line of rationale each, mirrored in the strategy's doc comment): diff --git a/docs/design/ORCHESTRATOR.md b/docs/design/ORCHESTRATOR.md index 36a289c27..8e02ccd73 100644 --- a/docs/design/ORCHESTRATOR.md +++ b/docs/design/ORCHESTRATOR.md @@ -278,6 +278,8 @@ Liveness detection varies by compute backend. AgentCore sessions use DynamoDB he - **Stale threshold** (240s) - If the heartbeat exists but is older than this, the session is treated as lost. - **Early crash** - If no heartbeat is ever set after the combined window (360s), the session is treated as lost; a process failure or failed DynamoDB writes can cause this. +Approval waits suppress heartbeat writes. When the agent consumes a decision and restores `RUNNING`, its conditional transaction also refreshes `agent_heartbeat_at`, so the first poll after a long wait does not mistake the old timestamp for a crash. + When the session is unhealthy, the task transitions to `FAILED` with "Agent session lost: no recent heartbeat." **ECS task status polling (ECS only).** The orchestrator calls `computeStrategy.pollSession` (ECS `DescribeTasks`) on each poll cycle. Three failure modes are detected: container failure (immediate `FAILED`), container exit without DynamoDB terminal write (fail after 5 consecutive completed polls), and repeated API failures (fail after 3 consecutive errors). ECS does not have heartbeat-based hung-process detection; a hung but alive container polls for the full `MAX_POLL_ATTEMPTS` window (~8.5h) before timing out. diff --git a/docs/src/content/docs/architecture/Orchestrator.md b/docs/src/content/docs/architecture/Orchestrator.md index 1917cfee5..2e6e6288d 100644 --- a/docs/src/content/docs/architecture/Orchestrator.md +++ b/docs/src/content/docs/architecture/Orchestrator.md @@ -282,6 +282,8 @@ Liveness detection varies by compute backend. AgentCore sessions use DynamoDB he - **Stale threshold** (240s) - If the heartbeat exists but is older than this, the session is treated as lost. - **Early crash** - If no heartbeat is ever set after the combined window (360s), the session is treated as lost; a process failure or failed DynamoDB writes can cause this. +Approval waits suppress heartbeat writes. When the agent consumes a decision and restores `RUNNING`, its conditional transaction also refreshes `agent_heartbeat_at`, so the first poll after a long wait does not mistake the old timestamp for a crash. + When the session is unhealthy, the task transitions to `FAILED` with "Agent session lost: no recent heartbeat." **ECS task status polling (ECS only).** The orchestrator calls `computeStrategy.pollSession` (ECS `DescribeTasks`) on each poll cycle. Three failure modes are detected: container failure (immediate `FAILED`), container exit without DynamoDB terminal write (fail after 5 consecutive completed polls), and repeated API failures (fail after 3 consecutive errors). ECS does not have heartbeat-based hung-process detection; a hung but alive container polls for the full `MAX_POLL_ATTEMPTS` window (~8.5h) before timing out. diff --git a/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md b/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md index 68e714ef9..aed019a84 100644 --- a/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md +++ b/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md @@ -74,7 +74,7 @@ The `ComputeStrategy` interface gains **mandatory** `suspendSession(handle)` / ` **Liveness on this backend combines substrate state and the agent heartbeat.** `GetMicrovm` reports whether the VM is running or terminal. The independent `_heartbeat_worker` in `agent/src/server.py` refreshes the task timestamp every 45 seconds while the task is `RUNNING`; a stale timestamp detects loss of that writer, including a process crash or inability to write DynamoDB. It does **not** prove pipeline progress: a hung coding thread can coexist with a healthy heartbeat thread. A failed `/run` hook can trigger service teardown; after an accepted hook, ABCA still needs explicit termination and the maximum-duration backstop. A general progress watchdog remains separate work. -AgentCore and MicroVM tasks start the same periodic heartbeat worker, so they use the same 120-second grace and 240-second stale thresholds. ECS starts the pipeline directly, bypassing `server.py`; its pipeline writes a timestamp once but does not start this worker. Enabling the same stale check for ECS would therefore fail healthy tasks after roughly four minutes. The check applies only to `RUNNING`, not `AWAITING_APPROVAL`. The transition back to `RUNNING` currently leaves the old timestamp intact: after a long gate, a poll can see a stale heartbeat before the next worker tick. Refreshing it atomically on approval resume is a P3 prerequisite in the implementation plan. +AgentCore and MicroVM tasks start the same periodic heartbeat worker, so they use the same 120-second grace and 240-second stale thresholds. ECS starts the pipeline directly, bypassing `server.py`; its pipeline writes a timestamp once but does not start this worker. Enabling the same stale check for ECS would therefore fail healthy tasks after roughly four minutes. The check applies only to `RUNNING`, not `AWAITING_APPROVAL`. The transition back to `RUNNING` refreshes `agent_heartbeat_at` in the same conditional transaction that clears the matching approval request. A poll immediately after a long gate therefore sees a fresh heartbeat before the next worker tick. P3 still needs the separate suspend/resume lifecycle and recovery policy described below. The service's `MicrovmState` enum has **six** members, not three, so the mapping is stated exhaustively (one line of rationale each, mirrored in the strategy's doc comment): diff --git a/docs/verification/645-p3-implementation-plan.md b/docs/verification/645-p3-implementation-plan.md index 8a4d6f121..4bc3967dc 100644 --- a/docs/verification/645-p3-implementation-plan.md +++ b/docs/verification/645-p3-implementation-plan.md @@ -2,6 +2,20 @@ Prepared 2026-09-13 from `main` `5e10038c7e28179b302ac4de78b709795aeba3ce`. Read the [review](./645-p3-readiness-review.md) for evidence and the beginner introduction. This document proposes work; it does not mark P3 as implemented. +## Implementation progress + +First prerequisite batch completed locally on 2026-09-13, on `fix/645-microvm-readiness`: + +| Commit | Completed work | Proof | +|---|---|---| +| `b5155928` | #841: join test pipeline threads before restoring mocks/environment; retain and report timed-out handles; distinct task IDs | A deterministic two-test subprocess regression failed on the old fixture and now passes. All 234 server/isolation tests pass. | +| `564ccc19` | #817 deletion IAM: exact `s3:DeleteObject` on `*/payload.json` in the dedicated bucket, on the coordinator only | The IAM regression failed before the grant; construct and full-stack assertions now verify the action and resource scope. Worker read-only permissions are unchanged. | +| `5490d645` | Approval resume: refresh heartbeat in the same conditional write that restores RUNNING | Atomic-write regression failed before the change; immediate post-wait polling is tested for AgentCore and MicroVM. Cancellation/wrong-gate conditions remain intact. | + +Validation for this batch: `mise run quality` in `agent/` passed lint, formatting, type checks and **1,785 tests**, with **83.75%** coverage. Six relevant CDK suites passed **509 tests** via `mise run testf`; CDK ESLint and TypeScript compilation also passed. The original review/cleanup is commit `19904775`. + +These are source changes, not changes to deployed AWS resources. The IAM fix needs a normal stack deployment; the heartbeat fix needs an updated agent image. No bootstrap-policy change is required by this batch. #817's other tracks, #700, start-retry uncertainty, nesting, clean P2 verification and P3 lifecycle work remain open. The historical validation section at the end describes the earlier review commit only. + ## The result we want When the coding agent asks a human for permission, its computer may go to sleep. The human can approve or deny while it sleeps. The computer wakes, reads the saved answer, and continues or refuses the action. If nobody answers, it wakes before the deadline and applies the existing timeout-as-denial rule. Its files, task identity, permissions, progress and deadline remain correct. @@ -219,7 +233,7 @@ For live evidence, retain redacted task IDs, commit/image/bootstrap versions, co **P3 is complete only when:** the full strategy interface and agent hooks ship; approval/deadline races and credential refresh pass; clean P2 and live P3 gates pass on the final deployment; behavior-changing follow-ups have passing regressions; generated docs, bootstrap artifacts and compatibility controls match; no test VM is left running/suspended; and #645/ADR/runbook status is updated with the actual evidence. An unverified gate must be called out explicitly, not converted to a checked box because unit tests passed. -## Validation of this review branch +## Validation of the original review commit Completed locally on 2026-09-13. These checks concern this cleanup/review branch, not the future P3 acceptance tests. @@ -231,4 +245,4 @@ Completed locally on 2026-09-13. These checks concern this cleanup/review branch - **Documentation sync and build passed: 77 pages.** The installed mise version could not expand the build task's `:sync` dependency (`':task' pattern should be expanded before matching`), so the declared steps were executed in order with `mise run sync` followed by `mise exec -- ./node_modules/.bin/astro build`. Existing Astro/Cedar highlighting/deprecation warnings remain; they did not fail the build. - **Offline nesting probe:** measured five configurations in current, naive-nested and parent-role/stable-name modes. The naive mode failed cycle validation; the corrected prototype passed. Probe output is linked in the review. No AWS resources were created or modified. -The live P2 rerun, production nested-stack migration and P3 implementation/acceptance matrix remain future work. This branch supplies the cleanup, evidence and plan. +The live P2 rerun, production nested-stack migration and P3 implementation/acceptance matrix remain future work. This original review commit supplies the cleanup, evidence and plan; subsequent prerequisite fixes are tracked at the top of this document. diff --git a/docs/verification/645-p3-readiness-review.md b/docs/verification/645-p3-readiness-review.md index 4519b9749..b50d198e6 100644 --- a/docs/verification/645-p3-readiness-review.md +++ b/docs/verification/645-p3-readiness-review.md @@ -4,6 +4,8 @@ Reviewed 2026-09-13 against `main` commit `5e10038c7e28179b302ac4de78b709795aeba This review covers the existing MicroVM implementation, related P2 follow-ups, comment accuracy and nested-stack feasibility. It includes local experiments, not an AWS deployment or a new live smoke run. The associated [implementation plan](./645-p3-implementation-plan.md) turns the findings into ordered work. +**Implementation update (2026-09-13):** the subsequent prerequisite commits fix #841 thread isolation, #817 coordinator payload-deletion permission and the approval-resume heartbeat race locally. The findings below preserve the reviewed baseline; see [implementation progress](./645-p3-implementation-plan.md#implementation-progress) for commits and validation. Other findings and live verification remain open. + ## Start here: the pieces in plain language A **MicroVM** is a small, isolated computer rented from AWS. **Firecracker** is the technology that keeps these small computers separate. A **backend** is the kind of rented computer ABCA chooses to run a coding task. From 8c04dd34140d86150f9c557aa60b2d8225f09d22 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Sun, 13 Sep 2026 15:17:27 -0400 Subject: [PATCH 006/149] fix(agent): classify unreadable MicroVM payload streams (#817) --- agent/src/server.py | 27 ++++++++---------- agent/tests/test_server.py | 57 +++++++++++++++++++++++++++++++++++++- 2 files changed, 68 insertions(+), 16 deletions(-) diff --git a/agent/src/server.py b/agent/src/server.py index 153c77c02..d6e8733e2 100644 --- a/agent/src/server.py +++ b/agent/src/server.py @@ -1485,7 +1485,7 @@ class MicrovmRunHookRequest(BaseModel): class _PayloadFetchError(Exception): - """A ``/run`` payload the agent could not READ, as opposed to could not PARSE. + """A fetched ``/run`` payload that could not be read or decoded. Exists purely to be *not* a ``ValueError``, because the ``/run`` handler discriminates its 400 from its 500 on exactly that type and the two answers @@ -1496,13 +1496,13 @@ class _PayloadFetchError(Exception): * 500 ``MICROVM_RUN_PAYLOAD_UNREADABLE`` — "the payload could not be read; retrying CAN help." - A truncated, racing, or half-written S3 object is the SECOND kind, but its - natural exception is ``json.JSONDecodeError`` — a ``ValueError`` subclass — so - it landed in the 400 branch and told the operator the orchestrator was at - fault when the orchestrator was fine and the object was bad. Only the - *pre-fetch* URI-shape check legitimately raises ``ValueError`` on this path, - which is why a blanket ``except ValueError`` is the wrong discriminator and - this type exists. + Corrupt/truncated JSON or an interrupted body read is the SECOND kind, but + ``json.JSONDecodeError`` and a closed stream's error are ``ValueError`` + subclasses. Without this wrapper, the handler mistakes those fetch/decode + failures for malformed hook envelopes. Only the pre-fetch URI-shape check + should reach the handler's ``ValueError`` branch. + S3 publishes object writes atomically: a malformed stored object needs + replacement, and retrying the same bytes will not repair them. """ @@ -1539,17 +1539,14 @@ def _fetch_microvm_payload_from_s3(uri: str) -> dict: region = os.environ.get("AWS_REGION") or os.environ.get("AWS_DEFAULT_REGION") client = platform_client("s3", region_name=region) - body = client.get_object(Bucket=bucket, Key=key)["Body"].read() try: + body = client.get_object(Bucket=bucket, Key=key)["Body"].read() payload = json.loads(body) except ValueError as exc: - # ``JSONDecodeError`` IS a ``ValueError``, so without this re-raise a - # truncated or half-written object would be reported as an orchestrator - # envelope bug and marked non-retryable. Re-raised as the type the handler - # routes to its retryable 500. + # Decode failures and a closed response stream can both raise ValueError. + # Keep them out of the handler's malformed-envelope (400) branch. raise _PayloadFetchError( - f"S3 payload at {uri!r} is not valid JSON ({exc}); the object may be " - "truncated or still being written" + f"S3 payload at {uri!r} could not be read as JSON ({exc})" ) from exc if not isinstance(payload, dict): raise _PayloadFetchError( diff --git a/agent/tests/test_server.py b/agent/tests/test_server.py index 10e839037..b85cc661d 100644 --- a/agent/tests/test_server.py +++ b/agent/tests/test_server.py @@ -8,6 +8,7 @@ import sys import threading import time +from io import BytesIO from pathlib import Path from types import SimpleNamespace from typing import Any @@ -1465,6 +1466,59 @@ class TestMicrovmRunHookS3Payload: pointer travels in the hook body. """ + @pytest.mark.parametrize( + "body_bytes,stream_fault", + [ + (b'{"platform_config":{"task_table_name":"must-not-install"},"task_id":', None), + (b'{"task_id":"bad-encoding-\xff"}', None), + (b'{"task_id":"short-stream"}', "incomplete"), + (b'{"task_id":"closed-stream"}', "closed"), + (b"[1,2,3]", None), + ], + ids=["truncated-json", "invalid-encoding", "incomplete-stream", "closed-stream", "array"], + ) + def test_bad_s3_bytes_use_the_real_fetch_path_and_start_nothing( + self, client, monkeypatch, body_bytes, stream_fault + ): + from botocore.response import StreamingBody + + import aws_session + + raw = BytesIO(body_bytes) + if stream_fault == "closed": + raw.close() + body = StreamingBody(raw, len(body_bytes) + (stream_fault == "incomplete")) + s3 = MagicMock() + s3.get_object.return_value = {"Body": body} + factory = MagicMock(return_value=s3) + run_task = MagicMock() + install_config = MagicMock(wraps=server._install_platform_config) + monkeypatch.setattr(aws_session, "platform_client", factory) + monkeypatch.setattr(server, "run_task", run_task) + monkeypatch.setattr(server, "_install_platform_config", install_config) + before_env = dict(os.environ) + + response = client.post( + RUN_HOOK, + json=_run_hook_body( + { + "agent_payload_s3_uri": "s3://payload-bucket/t-bad/payload.json", + "platform_config": {"task_table_name": "outer-must-not-install"}, + } + ), + ) + + assert response.status_code == 500 + assert response.json()["code"] == "MICROVM_RUN_PAYLOAD_UNREADABLE" + assert response.json()["message"] + s3.get_object.assert_called_once_with(Bucket="payload-bucket", Key="t-bad/payload.json") + assert factory.call_args.args == ("s3",) + install_config.assert_not_called() + run_task.assert_not_called() + assert dict(os.environ) == before_env + with server._threads_lock: + assert server._active_threads == [] + def test_fetches_the_payload_from_s3_and_starts_the_pipeline( self, client, monkeypatch, baked_platform_env ): @@ -2681,7 +2735,8 @@ def test_a_nested_agent_payload_of_the_wrong_type_is_a_retryable_500(self, clien # Review N1: this is a problem with the FETCHED OBJECT, not with the envelope # the orchestrator built, so it belongs on the retryable 500 branch. The old # 400 told the operator "the orchestrator built a bad envelope; retrying - # cannot help" — both halves wrong for a racing or half-written S3 object. + # cannot help". A fetched-object failure is kept distinct from that + # producer-envelope error; invalid stored bytes still need replacement. monkeypatch.setattr( server, "_fetch_microvm_payload_from_s3", From 2e20d54ad382f460ff8ed857de56c6f5b21494ad Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Sun, 13 Sep 2026 15:20:03 -0400 Subject: [PATCH 007/149] fix(compute): stabilize MicroVM failure classification (#817) --- cdk/src/handlers/shared/compute-strategy.ts | 7 +- cdk/src/handlers/shared/error-classifier.ts | 179 +++++++++++------- cdk/src/handlers/shared/failure-reply.ts | 13 +- cdk/src/handlers/shared/orchestrator.ts | 11 +- .../handlers/shared/error-classifier.test.ts | 59 +++++- cdk/test/handlers/shared/orchestrator.test.ts | 58 ++++-- .../shared/session-start-retry.test.ts | 11 ++ 7 files changed, 233 insertions(+), 105 deletions(-) diff --git a/cdk/src/handlers/shared/compute-strategy.ts b/cdk/src/handlers/shared/compute-strategy.ts index 4248f87db..9ac48c5e3 100644 --- a/cdk/src/handlers/shared/compute-strategy.ts +++ b/cdk/src/handlers/shared/compute-strategy.ts @@ -71,9 +71,10 @@ export type SessionHandle = * ~12 s — reached the operator as the bare, and therefore fabricated, * ``"substrate state completed"``. * - * It is OPTIONAL and OPAQUE: no control flow may branch on its content (that would - * put substrate interpretation back in the strategy), and it is for the reconcile - * ``detail`` string and logs only. + * It is OPTIONAL and OPAQUE to the strategy. The orchestrator retains it as + * diagnostic detail and recognizes the service's documented run-hook 4xx shape + * to choose a stable failure code. Consumers classify that code, so arbitrary + * words in the reason cannot change the category or user-facing retry advice. * * Declared on all four variants for UNIFORMITY, though only ``completed`` and * ``failed`` are read today (``reconcileMicrovmSubstrateState`` returns early for diff --git a/cdk/src/handlers/shared/error-classifier.ts b/cdk/src/handlers/shared/error-classifier.ts index 8d6b29c14..f3275adb4 100644 --- a/cdk/src/handlers/shared/error-classifier.ts +++ b/cdk/src/handlers/shared/error-classifier.ts @@ -90,6 +90,67 @@ interface ErrorPattern { readonly classification: ErrorClassification; } +/** Stable codes written by the orchestrator; diagnostic text cannot override them. */ +const MICROVM_TERMINAL_CLASSIFICATIONS: Readonly> = { + MICROVM_RUN_HOOK_REJECTED: { + category: ErrorCategory.CONFIG, + title: 'The MicroVM rejected its own run payload', + description: + 'The agent\'s /run hook rejected the request with an HTTP 4xx response before reporting a task result. Common causes are a malformed payload, invalid platform configuration, or mismatched orchestrator and image versions.', + remedy: + 'Retrying as-is will not help — the guest will reject the identical payload again. ' + + 'Read the agent\'s structured response in the MicroVM log group (/aws/lambda-microvms/) for the MICROVM_RUN_* code and any missing environment variables. ' + + 'MICROVM_RUN_PLATFORM_CONFIG_INCOMPLETE means the orchestrator predates the platform_config contract — an admin should redeploy the stack so the orchestrator and image versions match.', + retryable: false, + errorClass: ErrorClass.SERVICE, + }, + MICROVM_SUBSTRATE_TERMINATED: { + category: ErrorCategory.COMPUTE, + title: 'The MicroVM stopped before the agent reported a result', + description: + 'The MicroVM ended before the agent reported a task result. Any explanation supplied by AWS is preserved in the original error message for diagnosis.', + remedy: + 'This is a compute-substrate fault, not a problem with your request — reply here to try again. ' + + 'If the message names a lifecycle-hook HTTP status, the guest rejected the run: check the MicroVM log group for the agent\'s structured response body. ' + + 'Otherwise check the MicroVM logs for the session and whether the task is exceeding the 8-hour session cap; ' + + 'a long-running repo may belong on --compute-type ecs.', + retryable: true, + errorClass: ErrorClass.TRANSIENT, + }, +}; +const MICROVM_TERMINAL_PREFIX = 'MicroVM substrate terminated before the agent wrote a terminal status: '; +const MICROVM_RUN_HOOK_4XX = /^Run lifecycle hook returned HTTP status 4\d{2}(?:\.|$)/i; + +/** + * Preserve the service reason for diagnosis, independently of the persisted code. + * GetMicrovm exposes hook status only in stateReason. Recognize that documented + * response at this boundary; unknown wording stays a generic terminal failure. + * The strategy itself continues to report state without applying health policy. + */ +export function formatMicrovmTerminalFailure(detail: string, stateReason?: string): string { + const code = stateReason && MICROVM_RUN_HOOK_4XX.test(stateReason) + ? 'MICROVM_RUN_HOOK_REJECTED' + : 'MICROVM_SUBSTRATE_TERMINATED'; + const reason = stateReason ? ` (${stateReason})` : ''; + return `${code}: ${MICROVM_TERMINAL_PREFIX}${detail}${reason}`; +} + +/** Classify persisted MicroVM terminal failures, including records without a code. */ +export function classifyMicrovmTerminalFailure(errorMessage?: string | null): ErrorClassification | null { + if (!errorMessage) return null; + const code = /^(MICROVM_RUN_HOOK_REJECTED|MICROVM_SUBSTRATE_TERMINATED): /.exec(errorMessage)?.[1]; + if (code) return MICROVM_TERMINAL_CLASSIFICATIONS[code]; + if (MICROVM_RUN_HOOK_4XX.test(errorMessage)) return MICROVM_TERMINAL_CLASSIFICATIONS.MICROVM_RUN_HOOK_REJECTED; + + // Old records have only the descriptive prefix. Do not let other words in + // their service reason override the known terminal failure. + if (errorMessage.startsWith(MICROVM_TERMINAL_PREFIX)) { + const hookRejected = /\(Run lifecycle hook returned HTTP status 4\d{2}(?:\.|$)/i.test(errorMessage); + return MICROVM_TERMINAL_CLASSIFICATIONS[hookRejected ? 'MICROVM_RUN_HOOK_REJECTED' : 'MICROVM_SUBSTRATE_TERMINATED']; + } + return null; +} + const PATTERNS: readonly ErrorPattern[] = [ // --- Auth --- { @@ -205,7 +266,7 @@ const PATTERNS: readonly ErrorPattern[] = [ // anchored on a `lambda`/`microvm` marker precisely so an AgentCore or ECS // endpoint failure cannot be hijacked into MicroVM copy — their endpoint // hosts are `bedrock-agentcore.*` / `ecs.*`. - pattern: /(?:UnknownEndpoint|Inaccessible host|Could not resolve endpoint).{0,120}lambda|(?:lambda[- ]?microvms?|microvms?).{0,60}(?:not available|not supported|unavailable)|(?:not available|not supported|unavailable).{0,60}(?:lambda[- ]?microvms?)/i, + pattern: /(?:UnknownEndpoint|Inaccessible host|Could not resolve endpoint).{0,120}lambda|(?:lambda[- ]?microvms?|microvms?)\s+(?:service\s+)?(?:(?:is|are)\s+)?(?:not available|not supported|unavailable)\s+in\s+[^.\n]{0,50}\bregion\b/i, classification: { category: ErrorCategory.CONFIG, title: 'Lambda MicroVMs is not available in this Region', @@ -224,7 +285,7 @@ const PATTERNS: readonly ErrorPattern[] = [ }, // --- Lambda MicroVMs (ADR-021) --- // - // SCOPING: the three SDK-exception entries below are anchored on the + // SCOPING: the SDK-exception entries below are anchored on the // ``MicroVM failed`` marker that // ``lambda-microvm-strategy.wrapMicrovmError`` puts on every error it lets // escape (see ``MICROVM_ERROR_MARKER``). The anchor is mandatory, not @@ -237,9 +298,8 @@ const PATTERNS: readonly ErrorPattern[] = [ // classified them before this section existed (a precise earlier pattern, or // UNKNOWN). // - // Both orders are accepted in each pattern so a future wrapper that puts the - // exception name ahead of the marker still matches; ``[\s\S]`` rather than - // ``.`` because SDK messages can span lines. + // ``[\s\S]`` rather than ``.`` because SDK messages can span lines. The older + // quota/throttle/not-found patterns also accept the reverse wrapper order. // // ORDERING: this section sits immediately ABOVE the generic // `Session start failed` catch-all. That is only safe BECAUSE of the marker: @@ -249,66 +309,6 @@ const PATTERNS: readonly ErrorPattern[] = [ // the MICROVM_* env vars to check) instead of "Check AgentCore Runtime or ECS // cluster health" — advice that names the wrong substrate entirely. If the // marker anchor is ever dropped, these MUST move back below the catch-all. - { - // A LIFECYCLE-HOOK 4xx, which the substrate reports in `stateReason` and - // `reconcileMicrovmSubstrateState` appends to the persisted message. - // - // ORDERING: this MUST stay above the generic `MicroVM substrate terminated` - // entry below, which matches the same message (both strings travel in one - // `error_message`) and would otherwise win with `retryable: true`. - // - // WHY it is a distinct, NON-retryable entry: every 4xx the guest can answer is - // a config/envelope fault that an identical retry cannot fix — - // `MICROVM_RUN_PAYLOAD_INVALID` (bad envelope), - // `MICROVM_RUN_PLATFORM_CONFIG_INVALID` (bad `platform_config` value) and - // `MICROVM_RUN_PLATFORM_CONFIG_INCOMPLETE` (missing required key / version - // skew), all in `agent/src/server.py`'s `microvm_run`. The one genuinely - // retryable guest answer, `MICROVM_RUN_PAYLOAD_UNREADABLE`, is deliberately a - // **500**, which is why this pattern is scoped to `4\d\d` and a 5xx still falls - // through to the retryable entry below. Without this, a permanently-skewed - // deployment rendered as TRANSIENT with the remedy "reply here to try again", - // i.e. an invitation to loop forever. - // - // WHAT THE MESSAGE DOES *NOT* CARRY, measured: the guest's structured response - // BODY does not reach `stateReason`. Live evidence - // (`docs/verification/645-p2-smoke-runbook.md` §6.1) shows the service supplies - // exactly `"Run lifecycle hook returned HTTP status 400. Please check your hook - // endpoint and application logs for more details."` — the status code and - // nothing else. So this entry anchors on the STATUS, which is the only - // discriminator that actually travels; the `code` is available to the operator - // only in the guest log group, which is what the remedy sends them to. If a - // future service release ever enriches `stateReason` with the body, a - // code-anchored entry becomes possible and would be strictly better. - pattern: /Run lifecycle hook returned HTTP status 4\d\d/i, - classification: { - category: ErrorCategory.CONFIG, - title: 'The MicroVM rejected its own run payload', - description: - 'The Lambda MicroVMs service delivered this task to the in-guest agent\'s /run lifecycle hook and the agent answered 4xx, so the service reaped the MicroVM within ~12 s without the task ever starting. The agent refuses a run for exactly three reasons, all of them wiring faults rather than transient ones: the orchestrator built a malformed envelope, a platform_config value was invalid, or a required platform_config key was missing (a version-skewed orchestrator paired with a current image). The agent writes a structured reason to its log group before answering.', - remedy: - 'Retrying as-is will not help — the guest will reject the identical payload again. ' - + 'Read the agent\'s own structured response body in the MicroVM log group (/aws/lambda-microvms/): it carries a MICROVM_RUN_* code that names which of the three faults occurred, and for a missing-config fault it lists the exact environment variables. ' - + 'MICROVM_RUN_PLATFORM_CONFIG_INCOMPLETE means the orchestrator predates the platform_config contract — an admin should redeploy the stack so the orchestrator and image versions match.', - retryable: false, - errorClass: ErrorClass.SERVICE, - }, - }, - { - pattern: /MicroVM substrate terminated before the agent wrote a terminal status/i, - classification: { - category: ErrorCategory.COMPUTE, - title: 'The MicroVM stopped before the agent reported a result', - description: - 'The Lambda MicroVM running this task reached a terminal state while the task was still mid-flight, so no result was ever written. When the substrate supplied a reason (GetMicrovm\'s stateReason) the orchestrator appends it in parentheses on the message above — read that first, because it distinguishes the common causes (a /run lifecycle-hook 4xx, which the service reaps in ~12 s) from the rarer ones (the session duration cap, a host fault, or an external terminate).', - remedy: - 'This is a compute-substrate fault, not a problem with your request — reply here to try again. ' - + 'If the message names a lifecycle-hook HTTP status, the guest rejected the run: check the MicroVM log group for the agent\'s structured response body. ' - + 'Otherwise check the MicroVM logs for the session and whether the task is exceeding the 8-hour session cap; ' - + 'a long-running repo may belong on --compute-type ecs.', - retryable: true, - errorClass: ErrorClass.TRANSIENT, - }, - }, { // A `platform_config` block the orchestrator could not even assemble: a // REQUIRED key is absent from the orchestrator Lambda's own environment, so @@ -335,18 +335,14 @@ const PATTERNS: readonly ErrorPattern[] = [ }, }, { - // Account-level MicroVM memory quota. Deliberately TRANSIENT rather than - // SERVICE: this quota is capacity-shaped, not configuration-shaped — it - // frees as running/suspended MicroVMs terminate (AWS counts SUSPENDED VMs - // toward the quota), which is the same "wait and retry" character as the - // existing per-user `concurrency limit` entry above. The remedy still names - // the quota-increase path for the case where the ceiling is genuinely too - // low, so a persistently failing deployment is not left guessing. + // Account-level capacity may become available as other sessions terminate. + // Suspended-session quota accounting still needs live verification; do not + // promise that suspending a session frees capacity. pattern: /MicroVM [\w ]+failed[\s\S]*ServiceQuotaExceededException|ServiceQuotaExceededException[\s\S]*MicroVM [\w ]+failed/i, classification: { category: ErrorCategory.COMPUTE, title: 'Couldn\'t start — the MicroVM compute quota is currently exhausted', - description: 'Starting the MicroVM was rejected because the account\'s Lambda MicroVMs quota (memory across running and suspended MicroVMs) is fully consumed.', + description: 'Starting the MicroVM was rejected because the account\'s Lambda MicroVMs compute quota is fully consumed.', remedy: 'Wait for in-flight tasks to finish and retry — the quota frees as MicroVMs terminate. ' + 'If the platform hits this routinely, request a Lambda MicroVMs quota increase in Service Quotas, ' @@ -387,6 +383,39 @@ const PATTERNS: readonly ErrorPattern[] = [ }, }, + { + pattern: /MicroVM [\w ]+failed[\s\S]*(?:AccessDeniedException|UnauthorizedException)/i, + classification: { + category: ErrorCategory.AUTH, + title: 'The platform is not authorized to run this MicroVM', + description: 'AWS denied the MicroVM lifecycle request with the current permissions.', + remedy: 'An admin should check the orchestrator role, iam:PassRole, and the MicroVM execution role against the deployed bootstrap bundle and stack. Correct the permissions before retrying.', + retryable: false, + errorClass: ErrorClass.SERVICE, + }, + }, + { + pattern: /MicroVM [\w ]+failed[\s\S]*(?:ValidationException|InvalidParameterValueException)/i, + classification: { + category: ErrorCategory.CONFIG, + title: 'The MicroVM request contains invalid configuration', + description: 'AWS rejected a value in the MicroVM request.', + remedy: 'An admin should check the image identifier/version, connector ARNs, execution role and hook payload against the service contract, then redeploy the corrected configuration.', + retryable: false, + errorClass: ErrorClass.SERVICE, + }, + }, + { + pattern: /MicroVM[\s\S]*(?:host|capacity)[\s\S]*(?:unavailable|not available)|MicroVM[\s\S]*InsufficientCapacityException/i, + classification: { + category: ErrorCategory.COMPUTE, + title: 'The MicroVM host or capacity is temporarily unavailable', + description: 'AWS could not provide the host or capacity needed for this MicroVM.', + remedy: 'Retry the task. If the failure persists, an admin should check AWS service health and the available MicroVM capacity.', + retryable: true, + errorClass: ErrorClass.TRANSIENT, + }, + }, { pattern: /Session start failed/i, classification: { @@ -867,6 +896,10 @@ export function classifyError(errorMessage: string | undefined | null): ErrorCla return null; } + // Read the orchestrator's code before scanning any appended service text. + const microvmFailure = classifyMicrovmTerminalFailure(errorMessage); + if (microvmFailure) return microvmFailure; + // Environmental blockers carry a canonical ``BLOCKED[]`` prefix // and an extractable resource — check them first so the remedy can name the // exact secret / host rather than falling through to a generic pattern. diff --git a/cdk/src/handlers/shared/failure-reply.ts b/cdk/src/handlers/shared/failure-reply.ts index ff65ad963..3fe70d513 100644 --- a/cdk/src/handlers/shared/failure-reply.ts +++ b/cdk/src/handlers/shared/failure-reply.ts @@ -41,7 +41,7 @@ * iteration on the same PR. Pure and deterministic; no I/O. */ -import { classifyError, retryGuidance } from './error-classifier'; +import { classifyError, classifyMicrovmTerminalFailure, retryGuidance } from './error-classifier'; import type { TaskStatusType } from '../../constructs/task-status'; /** Max chars of the raw agent error surfaced inline (the rest is in CloudWatch). */ @@ -96,6 +96,7 @@ const BUILD_GATE_TIMEOUT_RE = /agent_status=['"]?(success|end_turn)['"]?.*build_ * that surfaces build_passed directly). */ function isBuildFailure(input: Pick): boolean { + if (classifyMicrovmTerminalFailure(input.errorMessage)) return false; if (input.errorMessage && BUILD_GATE_FAILED_RE.test(input.errorMessage)) { return true; } @@ -104,13 +105,14 @@ function isBuildFailure(input: Pick): boolean { + if (classifyMicrovmTerminalFailure(input.errorMessage)) return false; return !!input.errorMessage && BUILD_GATE_TIMEOUT_RE.test(input.errorMessage); } -/** Collapse whitespace + clip to EXCERPT_MAX chars with an ellipsis. Strips the - * internal `[auto-retried]` marker (it drives the guidance, not user-facing text). */ +/** Collapse whitespace and clip the raw detail; hide internal classification and retry markers. */ function excerpt(raw: string): string { - const oneLine = raw.replace(/\s*\[auto-retried\]\s*/gi, ' ').replace(/\s+/g, ' ').trim(); + const oneLine = raw.replace(/^MICROVM_(?:RUN_HOOK_REJECTED|SUBSTRATE_TERMINATED): /, '') + .replace(/\s*\[auto-retried\]\s*/gi, ' ').replace(/\s+/g, ' ').trim(); return oneLine.length > EXCERPT_MAX ? `${oneLine.slice(0, EXCERPT_MAX)}…` : oneLine; } @@ -165,6 +167,9 @@ export function renderFailureReply(input: FailureReplyInput): string { /** True when the orchestrator marked this failure as already auto-retried once. */ function wasAutoRetried(errorMessage?: string | null): boolean { + // A terminal-state observation is not a session-start retry. Any matching + // text in its AWS diagnostic reason must not claim that ABCA retried it. + if (classifyMicrovmTerminalFailure(errorMessage)) return false; return !!errorMessage && /\[auto-retried\]/i.test(errorMessage); } diff --git a/cdk/src/handlers/shared/orchestrator.ts b/cdk/src/handlers/shared/orchestrator.ts index 3a7a060ec..f6233da33 100644 --- a/cdk/src/handlers/shared/orchestrator.ts +++ b/cdk/src/handlers/shared/orchestrator.ts @@ -22,6 +22,7 @@ import { GetCommand, PutCommand, UpdateCommand } from '@aws-sdk/lib-dynamodb'; import { ulid } from 'ulid'; import type { SessionHandle, SessionStatus } from './compute-strategy'; import { AttachmentBudgetExceededError, AttachmentConfigurationError, AttachmentResolutionError, hydrateContext, resolveGitHubToken } from './context-hydration'; +import { formatMicrovmTerminalFailure } from './error-classifier'; import { logger, type Logger } from './logger'; import { writeMinimalEpisode } from './memory'; import { coerceNumericOrNull } from './numeric'; @@ -503,10 +504,10 @@ export async function reconcileMicrovmSubstrateState(args: { // happened. With it the operator gets "substrate state completed (Run lifecycle // hook returned HTTP status 400…)", which points at the guest logs where the // agent's own structured 4xx body already is. - const substrateReason = substrate.reason ? ` (${substrate.reason})` : ''; const detail = substrate.status === 'failed' - ? `${substrate.error}${substrateReason}` - : `substrate state ${substrate.status}${substrateReason}`; + ? substrate.error + : `substrate state ${substrate.status}`; + const failureReason = formatMicrovmTerminalFailure(detail, substrate.reason); const reread = await loadTask(taskId); if (TERMINAL_STATUSES.includes(reread.status)) { @@ -524,14 +525,14 @@ export async function reconcileMicrovmSubstrateState(args: { log.error('MicroVM reached a terminal state before the agent wrote a terminal status', { microvm_id: microvmId, task_status: reread.status, - detail, + detail: failureReason, }); // `releaseConcurrency: false` — the finalize step sees the now-terminal task // and decrements, matching the ECS substrate-failure branch in orchestrate-task. await failTask( taskId, reread.status, - `MicroVM substrate terminated before the agent wrote a terminal status: ${detail}`, + failureReason, userId, false, repo, diff --git a/cdk/test/handlers/shared/error-classifier.test.ts b/cdk/test/handlers/shared/error-classifier.test.ts index e42a69886..83822d6d2 100644 --- a/cdk/test/handlers/shared/error-classifier.test.ts +++ b/cdk/test/handlers/shared/error-classifier.test.ts @@ -603,6 +603,59 @@ describe('classifyError', () => { 'MicroVM substrate terminated before the agent wrote a terminal status: ' + `substrate state completed (${reason})`; + test.each([ + 'MicroVM host unavailable.', + 'MicroVM capacity unavailable in this Availability Zone.', + 'MicroVM host unavailable in this region.', + ])('does not mistake "%s" for an unsupported region', (reason) => { + for (const message of [reason, reconciled(reason)]) { + const result = classifyError(message)!; + expect(result.category).toBe(ErrorCategory.COMPUTE); + expect(result.errorClass).toBe(ErrorClass.TRANSIENT); + expect(result.retryable).toBe(true); + expect(retryGuidance(result)).toMatch(/reply here to try again/i); + } + }); + + test.each([ + 'MicroVM host unavailable.', + 'MicroVM unavailable in this region.', + 'INSUFFICIENT_GITHUB_REPO_PERMISSIONS', + 'concurrency limit reached', + 'BLOCKED[missing_secret]: diagnostic text', + 'Run lifecycle hook returned HTTP status 400.', + ])('a stable terminal code cannot be reclassified by diagnostic text: %s', (reason) => { + const result = classifyError(`MICROVM_SUBSTRATE_TERMINATED: ${reconciled(reason)}`)!; + expect(result.title).toBe('The MicroVM stopped before the agent reported a result'); + expect(result.category).toBe(ErrorCategory.COMPUTE); + expect(result.errorClass).toBe(ErrorClass.TRANSIENT); + }); + + test('a stable hook-rejection code keeps configuration guidance despite other diagnostic words', () => { + const result = classifyError( + `MICROVM_RUN_HOOK_REJECTED: ${reconciled('MicroVM host unavailable; concurrency limit')}`, + )!; + expect(result.category).toBe(ErrorCategory.CONFIG); + expect(result.retryable).toBe(false); + expect(retryGuidance(result)).toMatch(/needs your ABCA admin/i); + }); + + test.each(['AccessDeniedException', 'UnauthorizedException'])( + 'does not retry a marked %s when starting a MicroVM', (name) => { + const result = classifyError(`Session start failed: MicroVM RunMicrovm failed: ${name}: denied`)!; + expect(result.category).toBe(ErrorCategory.AUTH); + expect(result.errorClass).toBe(ErrorClass.SERVICE); + expect(result.retryable).toBe(false); + }, + ); + + test.each(['ValidationException', 'InvalidParameterValueException'])('does not retry MicroVM %s', (name) => { + const result = classifyError(`Session start failed: MicroVM RunMicrovm failed: ${name}: invalid connector`)!; + expect(result.category).toBe(ErrorCategory.CONFIG); + expect(result.errorClass).toBe(ErrorClass.SERVICE); + expect(result.retryable).toBe(false); + }); + test.each([400, 403, 404, 422, 499])( 'classifies a lifecycle-hook %i as a NON-retryable config fault', (status) => { @@ -612,6 +665,7 @@ describe('classifyError', () => { // generic COMPUTE/TRANSIENT entry whose remedy is "reply here to try // again" — an invitation to loop forever on a version-skewed deployment. const result = classifyError(reconciled(hookReason(status)))!; + expect(classifyError(hookReason(status))).toEqual(result); expect(result.category).toBe(ErrorCategory.CONFIG); expect(result.retryable).toBe(false); expect(result.errorClass).toBe(ErrorClass.SERVICE); @@ -626,7 +680,7 @@ describe('classifyError', () => { test('a lifecycle-hook 5xx stays RETRYABLE — MICROVM_RUN_PAYLOAD_UNREADABLE is a 500', () => { // The scoping that makes the entry above safe. The agent answers 500 for a - // truncated/racing S3 payload, which a retry genuinely can fix, so the 4xx + // failed S3 read, which a retry may fix, so the 4xx // pattern must not swallow the 5xx family. const result = classifyError(reconciled(hookReason(500)))!; expect(result.category).toBe(ErrorCategory.COMPUTE); @@ -695,6 +749,9 @@ describe('classifyError', () => { ['ServiceQuotaExceededException: quota exceeded'], ['ResourceNotFoundException: Requested resource not found'], ['TooManyRequestsException: slow down'], + ['AccessDeniedException: denied'], + ['UnauthorizedException: denied'], + ['ValidationException: invalid'], ])('an unmarked "%s" still classifies as UNKNOWN, exactly as before', (message) => { const result = classifyError(message)!; expect(result.category).toBe(PRE_CHANGE_UNKNOWN.category); diff --git a/cdk/test/handlers/shared/orchestrator.test.ts b/cdk/test/handlers/shared/orchestrator.test.ts index b53e1671a..5b8759613 100644 --- a/cdk/test/handlers/shared/orchestrator.test.ts +++ b/cdk/test/handlers/shared/orchestrator.test.ts @@ -43,7 +43,9 @@ import type { SessionHandle, SessionStatus } from '../../../src/handlers/shared/ // The real classifier: the reason-append must not break the anchor the substrate // -failure classification keys on. import { classifyError } from '../../../src/handlers/shared/error-classifier'; +import { renderFailureReply, renderPanelFailureReason } from '../../../src/handlers/shared/failure-reply'; import { buildComputeMetadata, reconcileMicrovmSubstrateState } from '../../../src/handlers/shared/orchestrator'; +import { toTaskDetail, type TaskRecord } from '../../../src/handlers/shared/types'; const MICROVM_ID = 'mvm-0123456789abcdef'; const ENDPOINT = 'https://mvm-0123456789abcdef.microvm.lambda.us-east-1.amazonaws.com'; @@ -302,6 +304,35 @@ describe('reconcileMicrovmSubstrateState', () => { }); describe('terminal substrate', () => { + test.each([ + ['MicroVM host unavailable.', 'MICROVM_SUBSTRATE_TERMINATED', 'compute', true], + ['capacity unavailable in this Availability Zone.', 'MICROVM_SUBSTRATE_TERMINATED', 'compute', true], + ['MicroVM unavailable in this region.', 'MICROVM_SUBSTRATE_TERMINATED', 'compute', true], + ['INSUFFICIENT_GITHUB_REPO_PERMISSIONS', 'MICROVM_SUBSTRATE_TERMINATED', 'compute', true], + ['BLOCKED[missing_secret]: diagnostic text', 'MICROVM_SUBSTRATE_TERMINATED', 'compute', true], + ["agent_status='success', build_ok=False", 'MICROVM_SUBSTRATE_TERMINATED', 'compute', true], + ["agent_status='success', build_ok=timeout [auto-retried]", 'MICROVM_SUBSTRATE_TERMINATED', 'compute', true], + ['Run lifecycle hook returned HTTP status 400.', 'MICROVM_RUN_HOOK_REJECTED', 'config', false], + ['Run lifecycle hook returned HTTP status 500.', 'MICROVM_SUBSTRATE_TERMINATED', 'compute', true], + ])('persists a stable failure code and consistent user guidance for %s', async (reason, code, category, retryable) => { + primeReread(TaskStatus.RUNNING); + await reconcile({ status: 'completed', reason }, TaskStatus.RUNNING); + const values = commandsOfType('Update')[0].input.ExpressionAttributeValues as Record; + const errorMessage = String(values[':attr_error_message']); + expect(errorMessage).toMatch(new RegExp(`^${code}: `)); + expect(errorMessage).toContain(reason); + expect(toTaskDetail({ + task_id: 'TASK001', status: TaskStatus.FAILED, error_message: errorMessage, + } as TaskRecord).error_classification).toMatchObject({ category, retryable }); + + const input = { status: TaskStatus.FAILED, errorMessage, taskId: 'TASK001' }; + for (const reply of [renderFailureReply(input), renderPanelFailureReason(input)]) { + expect(reply).toMatch(retryable ? /reply here to try again/i : /needs your ABCA admin/i); + expect(reply).not.toContain('Lambda MicroVMs is not available in this Region'); + expect(reply).not.toContain('I automatically tried again'); + } + }); + test('fails the task when the re-read status is still non-terminal', async () => { primeReread(TaskStatus.RUNNING); @@ -321,7 +352,7 @@ describe('reconcileMicrovmSubstrateState', () => { // The reason string is what error-classifier keys the substrate-failure // classification on — keep the two in lockstep. expect(values[':attr_error_message']).toBe( - 'MicroVM substrate terminated before the agent wrote a terminal status: substrate state completed', + 'MICROVM_SUBSTRATE_TERMINATED: MicroVM substrate terminated before the agent wrote a terminal status: substrate state completed', ); // Plus the task_failed audit event. @@ -364,7 +395,7 @@ describe('reconcileMicrovmSubstrateState', () => { expect(result).toEqual({ taskFailed: true, suspendAnomalyReported: false }); const values = commandsOfType('Update')[0].input.ExpressionAttributeValues as Record; expect(values[':attr_error_message']).toBe( - 'MicroVM substrate terminated before the agent wrote a terminal status: host fault', + 'MICROVM_SUBSTRATE_TERMINATED: MicroVM substrate terminated before the agent wrote a terminal status: host fault', ); }); @@ -383,7 +414,7 @@ describe('reconcileMicrovmSubstrateState', () => { const values = commandsOfType('Update')[0].input.ExpressionAttributeValues as Record; expect(values[':attr_error_message']).toBe( - 'MicroVM substrate terminated before the agent wrote a terminal status: ' + 'MICROVM_RUN_HOOK_REJECTED: MicroVM substrate terminated before the agent wrote a terminal status: ' + `substrate state completed (${reason})`, ); }); @@ -398,32 +429,23 @@ describe('reconcileMicrovmSubstrateState', () => { const values = commandsOfType('Update')[0].input.ExpressionAttributeValues as Record; expect(values[':attr_error_message']).toBe( - 'MicroVM substrate terminated before the agent wrote a terminal status: ' + 'MICROVM_SUBSTRATE_TERMINATED: MicroVM substrate terminated before the agent wrote a terminal status: ' + 'host fault (hypervisor evicted the guest)', ); }); - test('renders unchanged when the substrate supplies no reason', async () => { - // A live-verified hung MicroVM reports no `stateReason` at all, so the - // reason-less string stays the baseline — and stays the one the classifier's - // `MicroVM substrate terminated…` pattern is anchored on. + test('keeps a stable code when the substrate supplies no reason', async () => { primeReread(TaskStatus.RUNNING); await reconcile({ status: 'completed' }, TaskStatus.RUNNING); const values = commandsOfType('Update')[0].input.ExpressionAttributeValues as Record; expect(values[':attr_error_message']).toBe( - 'MicroVM substrate terminated before the agent wrote a terminal status: substrate state completed', + 'MICROVM_SUBSTRATE_TERMINATED: MicroVM substrate terminated before the agent wrote a terminal status: substrate state completed', ); }); - test('a reason-carrying message still classifies, and a hook 4xx outranks the generic entry', async () => { - // The append must not break classification — that would trade a misleading - // remedy for no remedy. It now does BETTER than preserve the generic anchor: - // a hook-4xx reason reaches a dedicated NON-retryable entry, because every - // 4xx the guest can answer is a wiring fault an identical retry cannot fix. - // Both strings live in one `error_message`, so this is really an assertion - // about classifier ORDER. + test('a hook 4xx selects the non-retryable failure code', async () => { primeReread(TaskStatus.RUNNING); await reconcile({ status: 'completed', reason: 'Run lifecycle hook returned HTTP status 400.' }, TaskStatus.RUNNING); const values = commandsOfType('Update')[0].input.ExpressionAttributeValues as Record; @@ -435,9 +457,7 @@ describe('reconcileMicrovmSubstrateState', () => { }); test('a NON-hook reason keeps the generic retryable substrate-failure entry', async () => { - // The other half: the `MicroVM substrate terminated…` anchor must still be - // the answer for the reasons it was written for (duration cap, host fault, - // external terminate), so the entry above must not have swallowed them. + // An ordinary service reason must not select the hook-rejection code. primeReread(TaskStatus.RUNNING); await reconcile( { status: 'completed', reason: 'host fault (hypervisor evicted the guest)' }, diff --git a/cdk/test/handlers/shared/session-start-retry.test.ts b/cdk/test/handlers/shared/session-start-retry.test.ts index ee27474c1..75d9a122a 100644 --- a/cdk/test/handlers/shared/session-start-retry.test.ts +++ b/cdk/test/handlers/shared/session-start-retry.test.ts @@ -214,6 +214,17 @@ describe('startSessionWithRetry — Lambda MicroVM start failures (ADR-021)', () await expect(startSessionWithRetry({ startSession }, {} as never, d)).rejects.toBe(notFound); expect(startSession).toHaveBeenCalledTimes(1); }); + + it.each(['AccessDeniedException', 'UnauthorizedException', 'ValidationException', 'InvalidParameterValueException'])( + 'does NOT retry a MicroVM %s', async (name) => { + const error = new Error(`MicroVM RunMicrovm failed: ${name}: request rejected`); + const startSession = jest.fn().mockRejectedValueOnce(error); + const { d, emitReasons } = deps(); + await expect(startSessionWithRetry({ startSession }, {} as never, d)).rejects.toBe(error); + expect(startSession).toHaveBeenCalledTimes(1); + expect(emitReasons).toEqual([]); + }, + ); }); describe('startSessionWithRetry — other backends keep their pre-ADR-021 retry behaviour', () => { From cfe95c5e48e6cff2e95138f1444b6ed824bf43ec Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Sun, 13 Sep 2026 15:20:35 -0400 Subject: [PATCH 008/149] fix(contracts): require validation for MicroVM ARN fields (#817) --- agent/src/server.py | 5 ++ agent/tests/test_server.py | 60 +++++++++++++++++++ .../lambda-microvm-strategy.test.ts | 9 +++ cdk/test/scripts/check-constants-sync.test.ts | 23 +++++++ scripts/check-constants-sync.ts | 19 ++++-- 5 files changed, 110 insertions(+), 6 deletions(-) diff --git a/agent/src/server.py b/agent/src/server.py index d6e8733e2..71a0dcc75 100644 --- a/agent/src/server.py +++ b/agent/src/server.py @@ -1054,6 +1054,11 @@ def _validate_platform_config_contract() -> None: unknown_arn = sorted(MICROVM_PLATFORM_CONFIG_ARN_KEYS - set(MICROVM_PLATFORM_CONFIG_ENV_BY_KEY)) if unknown_arn: raise ValueError(f"{where}.arn_keys names key(s) absent from env_by_key: {unknown_arn}") + for key, env_name in MICROVM_PLATFORM_CONFIG_ENV_BY_KEY.items(): + if ( + key.endswith("_arn") or env_name.endswith("_ARN") + ) and key not in MICROVM_PLATFORM_CONFIG_ARN_KEYS: + raise ValueError(f"{where}: ARN-shaped key {key!r} is missing from arn_keys") anchor = MICROVM_PLATFORM_CONFIG_ACCOUNT_ANCHOR_KEY if anchor not in MICROVM_PLATFORM_CONFIG_ARN_KEYS: raise ValueError(f"{where}.account_anchor_key {anchor!r} must be one of arn_keys") diff --git a/agent/tests/test_server.py b/agent/tests/test_server.py index b85cc661d..af6b912fa 100644 --- a/agent/tests/test_server.py +++ b/agent/tests/test_server.py @@ -1867,8 +1867,68 @@ def test_allowlist_is_sourced_from_the_shared_contract(self): from shared_constants import SHARED_CONSTANTS contract = SHARED_CONSTANTS["microvm_platform_config"] + assert set(contract) == {"env_by_key", "required", "arn_keys", "account_anchor_key"} assert contract["env_by_key"] == server.MICROVM_PLATFORM_CONFIG_ENV_BY_KEY assert frozenset(contract["required"]) == server.MICROVM_PLATFORM_CONFIG_REQUIRED_KEYS + assert frozenset(contract["arn_keys"]) == server.MICROVM_PLATFORM_CONFIG_ARN_KEYS + assert contract["account_anchor_key"] == server.MICROVM_PLATFORM_CONFIG_ACCOUNT_ANCHOR_KEY + + def test_arn_fields_and_account_anchor_are_explicitly_reviewed(self): + assert { + "github_token_secret_arn", + "linear_oauth_secret_arn", + "jira_oauth_secret_arn", + "agent_session_role_arn", + } == server.MICROVM_PLATFORM_CONFIG_ARN_KEYS + assert server.MICROVM_PLATFORM_CONFIG_ACCOUNT_ANCHOR_KEY == "agent_session_role_arn" + + @pytest.mark.parametrize( + "key,env_name", + [("new_resource_arn", "NEW_RESOURCE"), ("new_resource", "NEW_RESOURCE_ARN")], + ) + def test_new_arn_fields_cannot_omit_arn_validation(self, monkeypatch, key, env_name): + monkeypatch.setattr( + server, + "MICROVM_PLATFORM_CONFIG_ENV_BY_KEY", + {**server.MICROVM_PLATFORM_CONFIG_ENV_BY_KEY, key: env_name}, + ) + with pytest.raises(ValueError, match=r"ARN-shaped key.*missing from arn_keys"): + server._validate_platform_config_contract() + + monkeypatch.setattr( + server, + "MICROVM_PLATFORM_CONFIG_ARN_KEYS", + server.MICROVM_PLATFORM_CONFIG_ARN_KEYS | {key}, + ) + server._validate_platform_config_contract() + + @pytest.mark.parametrize( + "constant,value,error", + [ + ("MICROVM_PLATFORM_CONFIG_ARN_KEYS", frozenset(), "arn_keys must not be empty"), + ( + "MICROVM_PLATFORM_CONFIG_ARN_KEYS", + frozenset({"unknown_arn"}), + "arn_keys names key.*absent from env_by_key", + ), + ( + "MICROVM_PLATFORM_CONFIG_ACCOUNT_ANCHOR_KEY", + "task_table_name", + "must be one of arn_keys", + ), + ( + "MICROVM_PLATFORM_CONFIG_ACCOUNT_ANCHOR_KEY", + "linear_oauth_secret_arn", + "must also be listed in .required", + ), + ], + ) + def test_invalid_arn_contract_fails_before_serving_tasks( + self, monkeypatch, constant, value, error + ): + monkeypatch.setattr(server, constant, value) + with pytest.raises(ValueError, match=error): + server._validate_platform_config_contract() def test_wire_contract_is_exactly_the_documented_key_set(self): # Spelled out on purpose: this is the wire contract Stage B's producer is diff --git a/cdk/test/handlers/shared/strategies/lambda-microvm-strategy.test.ts b/cdk/test/handlers/shared/strategies/lambda-microvm-strategy.test.ts index 9abf87a5a..43a062041 100644 --- a/cdk/test/handlers/shared/strategies/lambda-microvm-strategy.test.ts +++ b/cdk/test/handlers/shared/strategies/lambda-microvm-strategy.test.ts @@ -1250,6 +1250,15 @@ describe('buildMicrovmPlatformConfig — the MicroVM substitute for a deploy-tim .toEqual(Object.keys(sharedConstants.microvm_platform_config.env_by_key)); expect([...MICROVM_PLATFORM_CONFIG_REQUIRED_KEYS]) .toEqual(sharedConstants.microvm_platform_config.required); + expect(Object.keys(sharedConstants.microvm_platform_config).sort()) + .toEqual(['account_anchor_key', 'arn_keys', 'env_by_key', 'required']); + expect(sharedConstants.microvm_platform_config.arn_keys).toEqual([ + 'github_token_secret_arn', + 'linear_oauth_secret_arn', + 'jira_oauth_secret_arn', + 'agent_session_role_arn', + ]); + expect(sharedConstants.microvm_platform_config.account_anchor_key).toBe('agent_session_role_arn'); // Order is part of the contract: it is the serialization order, which the 4 KB // inline/S3 branch decision is computed against. diff --git a/cdk/test/scripts/check-constants-sync.test.ts b/cdk/test/scripts/check-constants-sync.test.ts index 8e251e1d5..73f791ddd 100644 --- a/cdk/test/scripts/check-constants-sync.test.ts +++ b/cdk/test/scripts/check-constants-sync.test.ts @@ -367,6 +367,29 @@ describe('check-constants-sync', () => { }); describe('ARN-pinning contract (review B5)', () => { + test.each([ + ['new_resource_arn', 'NEW_RESOURCE'], + ['new_resource', 'NEW_RESOURCE_ARN'], + ])('rejects an unvalidated ARN field %s → %s', (key, envName) => { + const result = runInMutatedRepo((root) => { + patchContract(root, (json) => { + json.microvm_platform_config.env_by_key[key] = envName; + }); + }); + expect(result.status).toBe(1); + expect(result.stderr).toContain(`ARN-shaped key "${key}" is missing from arn_keys`); + }); + + test('accepts a new ARN field when its validation is also declared', () => { + const result = runInMutatedRepo((root) => { + patchContract(root, (json) => { + json.microvm_platform_config.env_by_key.new_resource_arn = 'NEW_RESOURCE_ARN'; + json.microvm_platform_config.arn_keys.push('new_resource_arn'); + }); + }); + expect(result.status).toBe(0); + }); + test('rejects an arn_keys entry that is not a wire key', () => { const result = runInMutatedRepo((root) => { patchContract(root, (json) => { diff --git a/scripts/check-constants-sync.ts b/scripts/check-constants-sync.ts index 536427486..d4dc86afa 100644 --- a/scripts/check-constants-sync.ts +++ b/scripts/check-constants-sync.ts @@ -287,12 +287,11 @@ function main(): number { invariantErrors.push('microvm_platform_config.required contains a duplicate'); } - // ARN pinning (ADR-021 P2, review B5). `arn_keys` names the values the agent - // pins to its own partition/account before installing them into the env that - // resolves credentials and fetches secrets; `account_anchor_key` names the ARN - // that supplies the expected partition/account. Both are validated here as well - // as at agent import time, because a malformed entry would silently WIDEN what - // the agent accepts from a network payload. + // ARN consistency (ADR-021 P2, review B5). `arn_keys` names the values checked + // against the partition/account supplied by `account_anchor_key`. That anchor + // comes from the same payload; agreement does not establish deployment identity. + // Validate both here and at agent import time so a new ARN field cannot + // silently skip the existing consistency check. if (!Array.isArray(mpc.arn_keys) || mpc.arn_keys.length === 0) { invariantErrors.push('microvm_platform_config.arn_keys must be a non-empty array'); } else { @@ -306,6 +305,14 @@ function main(): number { if (new Set(mpc.arn_keys).size !== mpc.arn_keys.length) { invariantErrors.push('microvm_platform_config.arn_keys contains a duplicate'); } + for (const [key, envName] of Object.entries(envByKey)) { + if ((key.endsWith('_arn') || (typeof envName === 'string' && envName.endsWith('_ARN'))) + && !mpc.arn_keys.includes(key)) { + invariantErrors.push( + `microvm_platform_config: ARN-shaped key "${key}" is missing from arn_keys`, + ); + } + } if (!mpc.arn_keys.includes(mpc.account_anchor_key)) { invariantErrors.push( 'microvm_platform_config.account_anchor_key must be one of arn_keys', From f037efe0deb359b8ac85db077c48426fcf632142 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Sun, 13 Sep 2026 15:21:12 -0400 Subject: [PATCH 009/149] docs(645): track completed P2 error and contract fixes --- .../645-p3-implementation-plan.md | 44 +++++++++++++++---- docs/verification/645-p3-readiness-review.md | 2 +- 2 files changed, 36 insertions(+), 10 deletions(-) diff --git a/docs/verification/645-p3-implementation-plan.md b/docs/verification/645-p3-implementation-plan.md index 4bc3967dc..2ee6ad89c 100644 --- a/docs/verification/645-p3-implementation-plan.md +++ b/docs/verification/645-p3-implementation-plan.md @@ -4,7 +4,22 @@ Prepared 2026-09-13 from `main` `5e10038c7e28179b302ac4de78b709795aeba3ce`. Read ## Implementation progress -First prerequisite batch completed locally on 2026-09-13, on `fix/645-microvm-readiness`: +Prerequisite work is tracked here on `fix/645-microvm-readiness`. “Completed” means implemented and checked locally; AWS deployment and live verification have separate completion gates below. + +- [x] Review current implementation, clean up verified stale comments, and prototype nesting. +- [x] Fix server-test thread isolation (#841). +- [x] Grant scoped coordinator payload deletion (#817). +- [x] Refresh heartbeat atomically after approval. +- [x] Stabilize terminal failure classification and user retry guidance (#817). +- [x] Exercise real S3 bad-byte paths and fix closed-stream error classification (#817). +- [x] Require new ARN fields to participate in validation; pin contract fields and anchor (#817). +- [ ] Bind configuration to trusted deployment identity and restrict payload reads per task (#817 / #700). +- [ ] Resolve uncertain session-start retries without duplicate or orphan VMs. +- [ ] Finish logging-failure observability (#810) and registry-overflow coverage (#818). +- [ ] Implement production nesting if included, then verify a clean P2 deployment. +- [ ] Implement and verify the P3 sleep/wake lifecycle described below. + +First prerequisite batch completed locally on 2026-09-13: | Commit | Completed work | Proof | |---|---|---| @@ -14,7 +29,17 @@ First prerequisite batch completed locally on 2026-09-13, on `fix/645-microvm-re Validation for this batch: `mise run quality` in `agent/` passed lint, formatting, type checks and **1,785 tests**, with **83.75%** coverage. Six relevant CDK suites passed **509 tests** via `mise run testf`; CDK ESLint and TypeScript compilation also passed. The original review/cleanup is commit `19904775`. -These are source changes, not changes to deployed AWS resources. The IAM fix needs a normal stack deployment; the heartbeat fix needs an updated agent image. No bootstrap-policy change is required by this batch. #817's other tracks, #700, start-retry uncertainty, nesting, clean P2 verification and P3 lifecycle work remain open. The historical validation section at the end describes the earlier review commit only. +These are source changes, not changes to deployed AWS resources. The IAM fix needs a normal stack deployment; the heartbeat fix needs an updated agent image. No bootstrap-policy change is required by this batch. The historical validation section at the end describes the earlier review commit only. + +Second prerequisite batch completed locally on 2026-09-13: + +| Commit | Completed work | Proof | +|---|---|---| +| `8c04dd34` | #817: exercise real S3 body failures; classify a closed stream as unreadable payload | Five route cases cover bad JSON/encoding, incomplete/closed streams and a non-object body. The closed-stream case failed with HTTP 400 before the fix and now returns structured HTTP 500 without installing config or starting work. | +| `2e20d54a` | #817: stable terminal failure codes, precise region/auth/config guidance and consistent task/reply classification | New regressions reproduced the prior misclassification. Tests assert saved message → task API → channel/panel guidance, legacy records, hook 4xx/5xx and start-retry decisions. | +| `cfe95c5e` | #817: exact contract assertions and reverse ARN-validation guards | Negative mutations failed in both Python and the real constants-checker subprocess before the fix. New ARN fields pass only when added to the validation set. | + +Final checks for the second batch: `mise run quality` passed **1,797 Python tests**, lint, formatting and type checks (**83.87%** coverage). Eight relevant CDK suites passed **476 tests**; CDK ESLint, TypeScript compilation, constants-sync and Markdown link checks passed. These counts describe the selected suites for each batch, not additional disjoint tests. No cloud resources were changed. Deploy the orchestrator update and an updated agent image to use these fixes; trusted configuration provenance, task-scoped payload reads, uncertain-start retries and P3 remain open. ## The result we want @@ -72,7 +97,7 @@ Keep nesting in a separate change from lifecycle logic. Developing P3 locally ne 1. Record the source commit, dependency lockfiles, bootstrap bundle version, image version and enabled context flags. Track #645, #817, #700, #818, #841 and #857 with concrete remaining checkboxes. Do not reopen consolidated #813–816 as if they represented four separate finished fixes. 2. Land/review this comment cleanup independently. It deliberately does not repair runtime behavior. -3. Fix the #841 test fixture: every spawned pipeline thread must finish or be stopped/joined before mocks/environment are restored. Synchronize with events, not arbitrary sleeps; fail teardown if a thread remains alive. Keep test state isolated without discarding references to live threads. +3. **Completed locally — #841:** every tracked pipeline thread is joined before mocks/environment are restored; teardown reports surviving threads without discarding their handles. A deterministic subprocess regression verifies isolation across successive tests. 4. Run focused baseline suites for server hooks, task-state approvals, credentials, strategy, orchestrator, approval handlers, construct IAM and bootstrap coverage. Record existing unrelated failures rather than quietly weakening assertions or thresholds. 5. Make the phase vocabulary explicit: this is P3 of ADR-021. Use different names for PR/work-package sequencing so “P1 priority” on an issue is not confused with phase P1. @@ -80,10 +105,11 @@ Keep nesting in a separate change from lifecycle logic. Developing P3 locally ne ### 1A. Finish #817 -- **Deletion IAM:** grant the coordinator `s3:DeleteObject` on the dedicated MicroVM payload bucket's task-object prefix. The worker gets no delete permission. Update `task-orchestrator.test.ts` and `stacks/agent.test.ts`, which currently assert the wrong absence. Exercise upload → start → finalize → delete, failed deletion logging and lifecycle fallback. Keep deletion best-effort so it does not hide the task's real outcome. -- **Classifier:** separate a stable MicroVM failure category from untrusted/free-form AWS `stateReason`. Test host unavailable, capacity unavailable, true unsupported region, authorization/configuration, concurrency and hook HTTP 400 cases. Check both stored task classification and user-facing retry advice. Preserve raw reason text for diagnosis without letting it redefine the category. -- **Trusted configuration:** decide which data is trusted at `/run`. A same-payload account anchor cannot authenticate its siblings. Bind accepted deployment identifiers to trusted deployment configuration or an authenticated payload reference; preserve legitimate cross-region secrets. Reject another workspace's secret even when it has the same account number. Add exact contract key/ARN-key/anchor assertions plus the reverse assertion that newly introduced ARN fields cannot bypass validation. -- **S3 bad bytes:** feed truncated and invalid-encoding bytes through the real fetch/decode/envelope/route path. Expect a structured unreadable-payload response, no partially installed environment and no pipeline thread. Keep producer-schema mistakes distinguishable from transport failures. +- [x] **Deletion IAM:** the coordinator has exact `s3:DeleteObject` on `*/payload.json` in the dedicated bucket. Construct and stack tests pin that scope. The worker retains read-only permissions, and deletion failure does not hide the task outcome. +- [x] **Classifier:** reconciliation persists `MICROVM_SUBSTRATE_TERMINATED` or `MICROVM_RUN_HOOK_REJECTED` separately from the descriptive AWS reason. The known leading run-hook 4xx response selects the latter; arbitrary appended text cannot override the code. Legacy task records remain readable. Tests cover task API classification, channel/panel retry guidance, host/capacity failures, regional faults, authorization/configuration, concurrency words, hook 400/500 and other backends. +- [ ] **Trusted configuration:** decide which data is trusted at `/run`. A same-payload account anchor cannot authenticate its siblings. Bind accepted deployment identifiers to trusted deployment configuration or an authenticated payload reference; preserve legitimate cross-region secrets. Reject another workspace's secret even when it has the same account number. +- [x] **Contract guards:** pin exact contract fields, ARN fields and the account anchor in both languages. Python import-time validation and the constants checker reject newly added `*_arn` / `*_ARN` fields omitted from `arn_keys`. This prevents accidental validation gaps; it does not establish deployment identity. +- [x] **S3 bad bytes:** real `StreamingBody` tests cover truncated JSON, invalid encoding, short and closed streams, and non-object JSON through fetch/decode/envelope/route. They assert a structured unreadable response, no configuration installation/environment mutation, and no pipeline thread. A closed-stream `ValueError` now reaches the unreadable-payload 500 branch. Malformed hook envelopes retain the separate 400 response. S3 writes are atomic; invalid stored bytes need replacement, not an assumption that another identical read repairs them. - **Documentation:** verify all remaining contract/status changes update source docs and their generated copies through the sync script. ### 1B. Narrow payload reads (#700) @@ -94,9 +120,9 @@ Evaluate a short-lived, single-object signed URL or a trusted bootstrap envelope ### 1C. Fix approval/heartbeat ordering -Change `task_state.transact_resume_from_approval` to refresh `agent_heartbeat_at` in the **same conditional update** that restores RUNNING. Keep the existing expected status and `awaiting_approval_request_id` conditions. +**Completed locally:** `task_state.transact_resume_from_approval` refreshes `agent_heartbeat_at` in the **same conditional update** that restores RUNNING. The expected status and `awaiting_approval_request_id` conditions remain in place. -Regression: task starts, waits over 240 seconds, then resumes. Force the orchestrator to poll after the transaction but before the heartbeat thread's next tick. It must remain healthy. Also test cancellation winning the race, wrong request ID, and ECS behavior; do not enable server-thread heartbeat enforcement on ECS. +Regression coverage includes a task that waits over 240 seconds and resumes before the next heartbeat tick; immediate polling remains healthy for AgentCore and MicroVM. Existing cancellation/wrong-request conditions and ECS behavior are preserved. This is approval-state coverage, not yet a live frozen-MicroVM test. ### 1D. Make session-start retries honest diff --git a/docs/verification/645-p3-readiness-review.md b/docs/verification/645-p3-readiness-review.md index b50d198e6..24454696e 100644 --- a/docs/verification/645-p3-readiness-review.md +++ b/docs/verification/645-p3-readiness-review.md @@ -4,7 +4,7 @@ Reviewed 2026-09-13 against `main` commit `5e10038c7e28179b302ac4de78b709795aeba This review covers the existing MicroVM implementation, related P2 follow-ups, comment accuracy and nested-stack feasibility. It includes local experiments, not an AWS deployment or a new live smoke run. The associated [implementation plan](./645-p3-implementation-plan.md) turns the findings into ordered work. -**Implementation update (2026-09-13):** the subsequent prerequisite commits fix #841 thread isolation, #817 coordinator payload-deletion permission and the approval-resume heartbeat race locally. The findings below preserve the reviewed baseline; see [implementation progress](./645-p3-implementation-plan.md#implementation-progress) for commits and validation. Other findings and live verification remain open. +**Implementation update (2026-09-13):** subsequent prerequisite work fixes #841 thread isolation, #817 coordinator payload-deletion permission, terminal-failure classification, closed S3-stream handling, and the approval-resume heartbeat race locally. Real bad-byte route tests and ARN contract guards are also added. Trusted deployment identity, task-scoped payload reads and live verification remain open. The findings below preserve the reviewed baseline; see [implementation progress](./645-p3-implementation-plan.md#implementation-progress) for commits, checks and remaining work. ## Start here: the pieces in plain language From e9944475fe257712e795aa8bd2da68f8ee879180 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Sun, 13 Sep 2026 15:51:33 -0400 Subject: [PATCH 010/149] fix(compute): recover MicroVM starts across lost responses (#645) --- cdk/src/handlers/orchestrate-task.ts | 103 +++++- cdk/src/handlers/shared/error-classifier.ts | 27 ++ cdk/src/handlers/shared/microvm-start.ts | 191 +++++++++++ cdk/src/handlers/shared/orchestrator.ts | 7 +- .../handlers/shared/session-start-retry.ts | 31 +- .../strategies/lambda-microvm-strategy.ts | 131 +++++--- .../handlers/orchestrate-task-microvm.test.ts | 210 ++++++++++-- cdk/test/handlers/orchestrate-task.test.ts | 32 ++ .../handlers/shared/error-classifier.test.ts | 10 + .../shared/microvm-start-recovery.test.ts | 305 ++++++++++++++++++ .../lambda-microvm-strategy.test.ts | 9 + .../start-session-composition.test.ts | 27 +- docs/design/ORCHESTRATOR.md | 14 +- .../content/docs/architecture/Orchestrator.md | 14 +- 14 files changed, 999 insertions(+), 112 deletions(-) create mode 100644 cdk/src/handlers/shared/microvm-start.ts create mode 100644 cdk/test/handlers/shared/microvm-start-recovery.test.ts diff --git a/cdk/src/handlers/orchestrate-task.ts b/cdk/src/handlers/orchestrate-task.ts index f80852981..9ebd50285 100644 --- a/cdk/src/handlers/orchestrate-task.ts +++ b/cdk/src/handlers/orchestrate-task.ts @@ -20,6 +20,7 @@ import { withDurableExecution, type DurableExecutionHandler } from '@aws/durable-execution-sdk-js'; import { TaskStatus, TERMINAL_STATUSES } from '../constructs/task-status'; import { resolveComputeStrategy } from './shared/compute-strategy'; +import { MicrovmStartUncertainError } from './shared/error-classifier'; import { reportIssueFailure as reportJiraIssueFailure } from './shared/jira-feedback'; import { reportIssueFailure } from './shared/linear-feedback'; import { logger } from './shared/logger'; @@ -184,8 +185,9 @@ const durableHandler: DurableExecutionHandler = asyn // Returns the full SessionHandle (serializable) so ECS polling can use it in step 5. const sessionHandle = await context.step('start-session', async () => { let autoRetried = false; + let failureStatus: TaskRecord['status'] = TaskStatus.HYDRATING; // Hoisted out of the `try` so the catch can reap a MicroVM that STARTED but - // whose handle never made it into DynamoDB — see the catch block. + // whose registration did not complete — see the catch block. let strategy: ReturnType | undefined; let startedHandle: Awaited>['handle'] | undefined; try { @@ -223,6 +225,15 @@ const durableHandler: DurableExecutionHandler = asyn // can load the handle they resume from — ADR-021 sub-decision 2). const computeMetadata = buildComputeMetadata(handle); + const current = handle.strategyType === 'lambda-microvm' ? await loadTask(taskId, true) : undefined; + if (current && TERMINAL_STATUSES.includes(current.status)) { + await strategy.stopSession(handle); + return null; + } + const alreadyRegistered = current?.session_id === handle.sessionId + && (current.status === TaskStatus.RUNNING || current.status === TaskStatus.AWAITING_APPROVAL); + if (alreadyRegistered) return handle; + await transitionTask(taskId, TaskStatus.HYDRATING, TaskStatus.RUNNING, { session_id: handle.sessionId, started_at: new Date().toISOString(), @@ -230,10 +241,17 @@ const durableHandler: DurableExecutionHandler = asyn compute_metadata: computeMetadata, ...(handle.strategyType === 'agentcore' && { agent_runtime_arn: handle.runtimeArn }), }); - await emitTaskEvent(taskId, 'session_started', { - session_id: handle.sessionId, - strategy_type: handle.strategyType, - }, correlation); + try { + await emitTaskEvent(taskId, 'session_started', { + session_id: handle.sessionId, + strategy_type: handle.strategyType, + }, correlation); + } catch (emitErr) { + if (handle.strategyType !== 'lambda-microvm') throw emitErr; + log.warn('session_started event failed after the MicroVM was registered', { + session_id: handle.sessionId, error: String(emitErr), + }); + } log.info('Session started', { session_id: handle.sessionId, @@ -242,15 +260,39 @@ const durableHandler: DurableExecutionHandler = asyn return handle; } catch (err) { - // ORPHAN REAP (ADR-021). `RunMicrovm` may have already succeeded and left a - // MicroVM RUNNING — the throw could have come from `buildComputeMetadata`, - // the `transitionTask` write, or the `session_started` emit. Nothing - // self-terminates on this substrate (live-verified: a MicroVM with no - // working hook reached RUNNING in 12 s and stayed RUNNING with no - // stateReason), and the handle only ever existed in this Lambda's memory — - // once we throw, no poll and no finalize step will ever see it. So the VM - // would bill until `maximumDurationInSeconds` (8 h) expired while also - // holding account memory quota that gates admission for everyone else. + if (blueprintConfig.compute_type === 'lambda-microvm') { + try { + const current = await loadTask(taskId, true); + // A lost DynamoDB update response does not undo its committed result. + if (startedHandle && current.session_id === startedHandle.sessionId + && (current.status === TaskStatus.RUNNING || current.status === TaskStatus.AWAITING_APPROVAL)) { + return startedHandle; + } + if (TERMINAL_STATUSES.includes(current.status)) { + if (startedHandle && strategy) await strategy.stopSession(startedHandle); + if (!startedHandle && err instanceof MicrovmStartUncertainError) { + log.error('Task became terminal while its MicroVM start outcome is unknown', { + task_id: taskId, client_token: taskId, task_status: current.status, + }); + try { + await emitTaskEvent(taskId, 'microvm_start_outcome_unknown', { + client_token: taskId, task_status: current.status, + }, correlation); + } catch (eventErr) { + log.warn('Could not record the unknown MicroVM start event', { error: String(eventErr) }); + } + } + return null; + } + failureStatus = current.status; + } catch (readErr) { + log.warn('Could not reconcile MicroVM registration after start failure', { error: String(readErr) }); + } + } + // Registration failed and a strong read could not establish a committed + // RUNNING/approval-wait task. Reap the known computer. The start receipt + // retains its ID for recovery if cleanup itself fails; the service's + // eight-hour maximum duration remains the final lifetime bound. // // Best-effort in the strongest sense: `stopSession` is internally // non-throwing for this backend, and the extra try/catch guarantees that @@ -290,10 +332,39 @@ const durableHandler: DurableExecutionHandler = asyn // Without this, a double-transient failure was told "reply to retry" instead // of "I already retried" — the exact confusion the marker exists to prevent. const retriedNote = (autoRetried || isAutoRetried(err)) ? ' [auto-retried]' : ''; - await failTask(taskId, TaskStatus.HYDRATING, `Session start failed: ${String(err)}${retriedNote}`, task.user_id, true, task.repo); + const detail = err instanceof MicrovmStartUncertainError + ? `MICROVM_START_OUTCOME_UNKNOWN: ${String(err)}` + : String(err); + const errorMessage = `Session start failed: ${detail}${retriedNote}`; + if (blueprintConfig.compute_type === 'lambda-microvm') { + // The durable finalization step below owns the terminal event and slot + // release. Replaying this step after FAILED must not release it twice. + try { + await transitionTask(taskId, failureStatus, TaskStatus.FAILED, { + completed_at: new Date().toISOString(), error_message: errorMessage, + }); + } catch (transitionErr) { + const current = await loadTask(taskId, true); + if (!TERMINAL_STATUSES.includes(current.status)) throw transitionErr; + } + return null; + } + await failTask(taskId, failureStatus, errorMessage, task.user_id, true, task.repo); throw err; } - }); + }, blueprintConfig.compute_type === 'lambda-microvm' + ? { retryStrategy: () => ({ shouldRetry: false }) } + : undefined); + + if (!sessionHandle) { + // The task ended before registration, including a persisted start failure. + // Reach normal finalization without starting another VM. + await context.step('finalize-before-session', async () => { + await finalizeTask(taskId, { attempts: 0 }, task.user_id); + await deleteMicrovmPayload(taskId); + }); + return; + } // Resolve the compute strategy once and reuse it across poll iterations // instead of constructing a new instance on every cycle. diff --git a/cdk/src/handlers/shared/error-classifier.ts b/cdk/src/handlers/shared/error-classifier.ts index f3275adb4..3ee238479 100644 --- a/cdk/src/handlers/shared/error-classifier.ts +++ b/cdk/src/handlers/shared/error-classifier.ts @@ -41,6 +41,11 @@ export const ErrorCategory = { export type ErrorCategoryType = (typeof ErrorCategory)[keyof typeof ErrorCategory]; +/** A start was sent, but its response does not establish whether AWS created a VM. */ +export class MicrovmStartUncertainError extends Error { + override name = 'MicrovmStartUncertainError'; +} + /** * WHO should act, and whether retrying the SAME request can help — the axis a * channel reader needs to answer "just retry, or tell my admin?". Distinct from @@ -152,6 +157,17 @@ export function classifyMicrovmTerminalFailure(errorMessage?: string | null): Er } const PATTERNS: readonly ErrorPattern[] = [ + { + pattern: /MICROVM_START_(?:OUTCOME_UNKNOWN|INPUT_CHANGED|STATE_INVALID|TASK_CLOSED|RECEIPT_SAVE_FAILED):/, + classification: { + category: ErrorCategory.COMPUTE, + title: 'The MicroVM start requires reconciliation', + description: 'The platform cannot safely issue another start for this task.', + remedy: 'An admin should inspect the task\'s saved MicroVM start receipt and any recorded computer ID before submitting another task. An unknown start may still have created a MicroVM; do not assume that a missing response means it failed.', + retryable: false, + errorClass: ErrorClass.SERVICE, + }, + }, // --- Auth --- { pattern: /INSUFFICIENT_GITHUB_REPO_PERMISSIONS/i, @@ -416,6 +432,17 @@ const PATTERNS: readonly ErrorPattern[] = [ errorClass: ErrorClass.TRANSIENT, }, }, + { + pattern: /MicroVM [\w ]+failed[\s\S]*(?:TimeoutError|RequestTimeout|ECONNRESET|ETIMEDOUT|InternalServerException|ServiceException)/i, + classification: { + category: ErrorCategory.COMPUTE, + title: 'The MicroVM service response was interrupted', + description: 'The request timed out, the connection failed, or AWS returned a server error. The request may already have succeeded.', + remedy: 'The platform retries the same saved start request within its recovery window. If recovery fails, an admin should inspect the saved start receipt before creating a new task.', + retryable: true, + errorClass: ErrorClass.TRANSIENT, + }, + }, { pattern: /Session start failed/i, classification: { diff --git a/cdk/src/handlers/shared/microvm-start.ts b/cdk/src/handlers/shared/microvm-start.ts new file mode 100644 index 000000000..ad3f2aebe --- /dev/null +++ b/cdk/src/handlers/shared/microvm-start.ts @@ -0,0 +1,191 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import { createHash } from 'node:crypto'; +import { GetCommand, UpdateCommand } from '@aws-sdk/lib-dynamodb'; +import type { SessionHandle } from './compute-strategy'; +import { makeDocClient } from './ua'; +import { TaskStatus, TERMINAL_STATUSES } from '../../constructs/task-status'; + +type MicrovmHandle = Extract; + +/** + * Internal TaskTable attribute, deliberately not part of the task API. + * One task owns one logical start. Retrying a failed task creates a new task ID; + * neither an application retry nor durable replay may mint a replacement token. + */ +interface StartReceipt { + readonly clientToken: string; + readonly requestHash: string; + readonly createdAt: string; + readonly expiresAt: number; + readonly handle?: MicrovmHandle; +} + +interface StartRecord { + readonly user_id: string; + readonly status: string; + readonly session_id?: string; + readonly compute_type?: string; + readonly compute_metadata?: Record; + readonly microvm_start?: StartReceipt; +} + +export interface MicrovmStartClaim { + readonly clientToken: string; + readonly handle?: MicrovmHandle; + readonly closed: boolean; +} + +// A local retry limit, NOT a claim about AWS's undocumented token retention. +// After this window an unknown outcome requires investigation, never a fresh VM. +export const MICROVM_START_REPLAY_WINDOW_MS = 120_000; + +const ddb = makeDocClient(); +const TABLE_NAME = process.env.TASK_TABLE_NAME!; +const ACTIVE = new Set([TaskStatus.HYDRATING, TaskStatus.RUNNING, TaskStatus.AWAITING_APPROVAL]); + +/** Include the full S3 content, not just its URI; ignore object-key ordering. */ +export function microvmStartRequestHash(request: unknown, payload: unknown): string { + const canonical = JSON.stringify([request, payload], (_key, value: unknown) => + value && typeof value === 'object' && !Array.isArray(value) + ? Object.fromEntries(Object.entries(value).sort(([a], [b]) => a.localeCompare(b))) + : value); + return createHash('sha256').update(canonical).digest('hex'); +} + +function recordedHandle(record: StartRecord): MicrovmHandle | undefined { + const handle = record.microvm_start?.handle; + if (handle?.strategyType === 'lambda-microvm' && handle.microvmId && handle.endpoint + && handle.sessionId === handle.microvmId) return handle; + // Read handles written before start receipts existed, without starting again. + const metadata = record.compute_metadata; + if (record.compute_type === 'lambda-microvm' && record.session_id + && metadata?.microvmId === record.session_id && metadata.endpoint) { + return { + strategyType: 'lambda-microvm', + sessionId: record.session_id, + microvmId: metadata.microvmId, + endpoint: metadata.endpoint, + }; + } + return undefined; +} + +async function readStartRecord(taskId: string, userId: string): Promise { + const result = await ddb.send(new GetCommand({ + TableName: TABLE_NAME, + Key: { task_id: taskId }, + ConsistentRead: true, + })); + const record = result.Item as StartRecord | undefined; + if (!record || record.user_id !== userId) { + throw new Error('MICROVM_START_STATE_INVALID: task is missing or its owner does not match'); + } + return record; +} + +/** + * Establish identity and input immutability before S3 writes or RunMicrovm. + * Conditional creation makes competing invocations use the winning receipt. + */ +export async function claimMicrovmStart( + taskId: string, + userId: string, + requestHash: string, +): Promise { + for (let attempt = 0; attempt < 2; attempt++) { + const record = await readStartRecord(taskId, userId); + const handle = recordedHandle(record); + if (TERMINAL_STATUSES.some(status => status === record.status)) { + return { clientToken: taskId, handle, closed: true }; + } + if (!ACTIVE.has(record.status)) { + throw new Error(`MICROVM_START_STATE_INVALID: cannot start a task in ${record.status}`); + } + if (handle) return { clientToken: taskId, handle, closed: false }; + const receipt = record.microvm_start; + if (receipt) { + if (receipt.clientToken !== taskId || !Number.isFinite(receipt.expiresAt)) { + throw new Error('MICROVM_START_STATE_INVALID: invalid saved start receipt'); + } + if (receipt.requestHash !== requestHash) { + throw new Error('MICROVM_START_INPUT_CHANGED: refusing to overwrite or restart an earlier MicroVM request'); + } + if (Date.now() >= receipt.expiresAt) { + throw new Error('MICROVM_START_OUTCOME_UNKNOWN: replay window expired; inspect the original start before retrying'); + } + return { clientToken: receipt.clientToken, closed: false }; + } + if (record.status !== TaskStatus.HYDRATING) { + throw new Error('MICROVM_START_STATE_INVALID: active task has no recoverable start receipt or handle'); + } + const now = Date.now(); + const receiptToSave: StartReceipt = { + clientToken: taskId, + requestHash, + createdAt: new Date(now).toISOString(), + expiresAt: now + MICROVM_START_REPLAY_WINDOW_MS, + }; + try { + await ddb.send(new UpdateCommand({ + TableName: TABLE_NAME, + Key: { task_id: taskId }, + UpdateExpression: 'SET microvm_start = :receipt', + ConditionExpression: '#status = :hydrating AND user_id = :user AND attribute_not_exists(microvm_start)', + ExpressionAttributeNames: { '#status': 'status' }, + ExpressionAttributeValues: { + ':receipt': receiptToSave, + ':hydrating': TaskStatus.HYDRATING, + ':user': userId, + }, + })); + return { clientToken: taskId, closed: false }; + } catch (err) { + if ((err as { name?: string }).name !== 'ConditionalCheckFailedException') throw err; + // Re-read the winner, or observe cancellation, before any side effect. + } + } + throw new Error('MICROVM_START_STATE_INVALID: start receipt changed while being claimed'); +} + +/** Retain a known ID even if cancellation won while RunMicrovm was in flight. */ +export async function saveMicrovmStartHandle( + taskId: string, + clientToken: string, + handle: MicrovmHandle, +): Promise { + await ddb.send(new UpdateCommand({ + TableName: TABLE_NAME, + Key: { task_id: taskId }, + UpdateExpression: 'SET microvm_start.#handle = :handle, session_id = :id, ' + + 'compute_type = :type, compute_metadata = :metadata', + ConditionExpression: 'microvm_start.clientToken = :token AND ' + + '(attribute_not_exists(microvm_start.#handle) OR microvm_start.#handle.microvmId = :id) AND ' + + '(attribute_not_exists(session_id) OR session_id = :id)', + ExpressionAttributeNames: { '#handle': 'handle' }, + ExpressionAttributeValues: { + ':token': clientToken, + ':id': handle.microvmId, + ':handle': handle, + ':type': 'lambda-microvm', + ':metadata': { microvmId: handle.microvmId, endpoint: handle.endpoint }, + }, + })); +} diff --git a/cdk/src/handlers/shared/orchestrator.ts b/cdk/src/handlers/shared/orchestrator.ts index f6233da33..197941ba2 100644 --- a/cdk/src/handlers/shared/orchestrator.ts +++ b/cdk/src/handlers/shared/orchestrator.ts @@ -175,10 +175,11 @@ function substrateNoun(computeType: ComputeType | undefined): string { * @returns the task record. * @throws Error if the task is not found. */ -export async function loadTask(taskId: string): Promise { +export async function loadTask(taskId: string, consistentRead = false): Promise { const result = await ddb.send(new GetCommand({ TableName: TABLE_NAME, Key: { task_id: taskId }, + ...(consistentRead && { ConsistentRead: true }), })); if (!result.Item) { throw new Error(`Task ${taskId} not found`); @@ -1153,7 +1154,9 @@ export async function finalizeTask( pollState: PollState, userId: string, ): Promise { - const task = await loadTask(taskId); + // Finalization can immediately follow a committed start failure/cancellation. + // A stale active state would emit the wrong terminal event. + const task = await loadTask(taskId, true); const currentStatus = task.status; // Correlation envelope on this function's own log lines too, not just the // events it emits — admission→terminal logs must join by {user_id, repo}. diff --git a/cdk/src/handlers/shared/session-start-retry.ts b/cdk/src/handlers/shared/session-start-retry.ts index 2e264d97b..4b72fe802 100644 --- a/cdk/src/handlers/shared/session-start-retry.ts +++ b/cdk/src/handlers/shared/session-start-retry.ts @@ -20,22 +20,23 @@ /** * Session-start transient auto-retry (once), extracted from the durable * ``orchestrate-task`` handler so the four retry branches are unit-testable in - * isolation (the handler's inline ``start-session`` step is never invoked by - * the test suite). See #599 review B1/B2. + * isolation, with MicroVM receipt/replay behavior also covered through the + * strategy and durable-handler tests. See #599 review B1/B2. * * This retries the start API, before the caller has a session handle. That does * not guarantee idempotency: a lost success response can leave a live session - * behind, so each backend needs an idempotency/reconciliation policy. A transient hiccup here (an + * behind. MicroVM uses its saved start receipt and token; each other backend + * still needs its own idempotency policy. A transient hiccup here (an * ECS deploy-race "TaskDefinition is inactive", ENI/capacity delay, a * Bedrock/agentcore throttle) usually clears on a second attempt, so the first * transient failure is swallowed and retried once. A NON-transient failure (bad * config, missing ECS substrate) is re-thrown immediately — retrying it just - * wastes ~a minute. Mid-run crashes are NOT handled here (that's a later step; + * wastes another request. Mid-run crashes are NOT handled here (that's a later step; * the agent may have pushed commits). */ import type { ComputeStrategy, SessionHandle } from './compute-strategy'; -import { classifyError, isTransientError } from './error-classifier'; +import { classifyError, isTransientError, MicrovmStartUncertainError } from './error-classifier'; /** Emit a ``session_start_retry`` telemetry event. Matches the shape of * ``emitTaskEvent`` bound at the call site (best-effort — see below). */ @@ -114,17 +115,20 @@ export async function startSessionWithRetry( // classifier's transient patterns — e.g. ThrottlingException — is a separate // classifier-completeness concern, not this retry gate.) const classification = classifyError(String(firstErr)); - if (!isTransientError(classification)) { + if (!isTransientError(classification) && !(firstErr instanceof MicrovmStartUncertainError)) { throw firstErr; // service/user error — a retry won't help; surface now. } - deps.logger.warn('Session start hit a transient error — auto-retrying once', { + const uncertain = firstErr instanceof MicrovmStartUncertainError; + deps.logger.warn(uncertain + ? 'MicroVM start response is unknown — recovering the same request once' + : 'Session start hit a transient error — auto-retrying once', { task_id: deps.taskId, error: firstErr instanceof Error ? firstErr.message : String(firstErr), }); // Best-effort telemetry — a PutItem fault here must never abort the retry // or mis-report the outcome (B1). try { - await deps.emitRetryEvent(classification?.title ?? 'transient'); + await deps.emitRetryEvent(uncertain ? 'MicroVM start response unknown' : classification?.title ?? 'transient'); } catch (emitErr) { deps.logger.warn('session_start_retry event emit failed (non-fatal)', { task_id: deps.taskId, @@ -135,14 +139,19 @@ export async function startSessionWithRetry( const handle = await strategy.startSession(input); return { handle, autoRetried: true }; } catch (retryErr) { + // Even a definite rejection on the second call cannot prove that the + // first, unanswered call created nothing. Preserve that uncertainty. + const finalError = firstErr instanceof MicrovmStartUncertainError || retryErr instanceof MicrovmStartUncertainError + ? new MicrovmStartUncertainError(String(retryErr), { cause: retryErr }) + : retryErr; // Branch 4: the retry ALSO failed. Pin the retry fact onto the thrown error // so the caller can stamp ``[auto-retried]`` (N1) — otherwise a double- // transient failure is indistinguishable from a first-attempt failure and // the user is wrongly told "reply to retry" instead of "I already retried". - if (typeof retryErr === 'object' && retryErr !== null) { - (retryErr as Record)[AUTO_RETRIED] = true; + if (typeof finalError === 'object' && finalError !== null) { + (finalError as Record)[AUTO_RETRIED] = true; } - throw retryErr; + throw finalError; } } } diff --git a/cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts b/cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts index dc2dfb4d5..d278e3f37 100644 --- a/cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts +++ b/cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts @@ -30,7 +30,9 @@ import { DeleteObjectCommand, PutObjectCommand, S3Client } from '@aws-sdk/client // `tsc` fails on a renamed field — see `contracts/constants.md`. import sharedConstants from '../../../../../contracts/constants.json'; import type { ComputeStrategy, SessionHandle, SessionStatus } from '../compute-strategy'; +import { MicrovmStartUncertainError } from '../error-classifier'; import { logger } from '../logger'; +import { claimMicrovmStart, microvmStartRequestHash, saveMicrovmStartHandle } from '../microvm-start'; import type { BlueprintConfig } from '../repo-config'; import { makeClient } from '../ua'; @@ -80,6 +82,7 @@ const MICROVM_EGRESS_CONNECTOR_ARNS = process.env.MICROVM_EGRESS_CONNECTOR_ARNS; */ const MICROVM_INGRESS_CONNECTOR_ARNS = process.env.MICROVM_INGRESS_CONNECTOR_ARNS; const MICROVM_PAYLOAD_BUCKET = process.env.MICROVM_PAYLOAD_BUCKET; +const HTTP_REQUEST_TIMEOUT = 408; /** * Session wall-clock ceiling passed on EVERY ``RunMicrovm`` call, pinned to the @@ -516,9 +519,7 @@ export class LambdaMicrovmComputeStrategy implements ComputeStrategy { async startSession(input: { taskId: string; - /** Accepted to satisfy the ComputeStrategy interface. MicroVMs have no - * workload-token-injecting runtime (they inherit the ECS env-var identity - * posture until #249 / ADR-016 redesign the seam), so this is unused. */ + /** Checked against the stored task owner before claiming a start receipt. */ userId: string; payload: Record; blueprintConfig: BlueprintConfig; @@ -577,6 +578,7 @@ export class LambdaMicrovmComputeStrategy implements ComputeStrategy { let runHookPayload: string; let payloadS3Uri: string | undefined; + let uploadPayload: (() => Promise) | undefined; // EXACT boundary: `<= limit` inlines, `> limit` uploads. The service accepts // 4 096 bytes and rejects 4 097 (measured), so 4 096 must still go inline. if (inlineBytes <= RUN_HOOK_PAYLOAD_LIMIT_BYTES) { @@ -609,27 +611,28 @@ export class LambdaMicrovmComputeStrategy implements ComputeStrategy { // `platform_config` key today, and if one ever appeared the platform's // value is the authoritative one. const payloadJson = JSON.stringify({ ...payload, platform_config: platformConfig }); - try { - await getS3Client().send(new PutObjectCommand({ - Bucket: MICROVM_PAYLOAD_BUCKET, - Key: key, - Body: payloadJson, - ContentType: 'application/json', - })); - } catch (err) { - // Marked so the classifier attributes an upload failure to this backend - // rather than letting a bare S3 exception name fall through to UNKNOWN. - throw wrapMicrovmError('payload upload', err); - } + // Defer all writes until the persisted receipt accepts this exact input. + uploadPayload = async () => { + try { + await getS3Client().send(new PutObjectCommand({ + Bucket: MICROVM_PAYLOAD_BUCKET, + Key: key, + Body: payloadJson, + ContentType: 'application/json', + })); + } catch (err) { + throw wrapMicrovmError('payload upload', err); + } + logger.info('Wrote MicroVM run-hook payload to S3', { + task_id: taskId, + bytes: Buffer.byteLength(payloadJson, 'utf8'), + inline_bytes: inlineBytes, + inline_limit_bytes: RUN_HOOK_PAYLOAD_LIMIT_BYTES, + uri, + }); + }; payloadS3Uri = uri; runHookPayload = pointerEnvelope; - logger.info('Wrote MicroVM run-hook payload to S3', { - task_id: taskId, - bytes: Buffer.byteLength(payloadJson, 'utf8'), - inline_bytes: inlineBytes, - inline_limit_bytes: RUN_HOOK_PAYLOAD_LIMIT_BYTES, - uri: payloadS3Uri, - }); } // Explicit ingress control (F7, live 2026-07-31): `RunMicrovm` does NOT @@ -647,7 +650,7 @@ export class LambdaMicrovmComputeStrategy implements ComputeStrategy { ? configuredIngress : [noIngressConnectorArn()]; - const command = new RunMicrovmCommand({ + const request = { imageIdentifier: MICROVM_IMAGE_IDENTIFIER, ...(MICROVM_IMAGE_VERSION && { imageVersion: MICROVM_IMAGE_VERSION }), executionRoleArn: MICROVM_EXECUTION_ROLE_ARN, @@ -669,12 +672,28 @@ export class LambdaMicrovmComputeStrategy implements ComputeStrategy { // is present, so omission is the unambiguous disabled state. Suspension is // orchestrator-owned (P3) — do NOT reintroduce this field. // - // No application-stable `clientToken` is supplied. The SDK may generate a - // token for one command, but `startSessionWithRetry` constructs another - // command on its next attempt. A failed response does not prove the first - // VM was never created. Lost-response reconciliation and attempt-scoped - // tokens need coverage before this can claim idempotent session starts. - }); + // The receipt below supplies a task-stable token across new SDK commands. + }; + + const requestHash = microvmStartRequestHash(request, { ...payload, platform_config: platformConfig }); + const claim = await claimMicrovmStart(taskId, input.userId, requestHash); + if (claim.closed) { + if (claim.handle) await this.stopSession(claim.handle); + throw new Error('MICROVM_START_TASK_CLOSED: task became terminal before session start'); + } + if (claim.handle) return claim.handle; + if (uploadPayload) { + await uploadPayload(); + } + // Uploads can take time. Observe cancellation/another saved handle again + // immediately before the service call, using the same immutable request. + const latest = await claimMicrovmStart(taskId, input.userId, requestHash); + if (latest.closed) { + if (latest.handle) await this.stopSession(latest.handle); + throw new Error('MICROVM_START_TASK_CLOSED: task became terminal before session start'); + } + if (latest.handle) return latest.handle; + const command = new RunMicrovmCommand({ ...request, clientToken: latest.clientToken }); let result; try { @@ -684,7 +703,21 @@ export class LambdaMicrovmComputeStrategy implements ComputeStrategy { // / `ResourceNotFoundException` from THIS backend classify as MicroVM // faults, while identically-named AgentCore/ECS errors keep their existing // classification. See MICROVM_ERROR_MARKER. - throw wrapMicrovmError('RunMicrovm', err); + const wrapped = wrapMicrovmError('RunMicrovm', err); + const serviceError = err as { name?: string; $metadata?: { httpStatusCode?: number } }; + const httpStatus = serviceError?.$metadata?.httpStatusCode; + // A service timeout can carry a 4xx status without proving that creation + // never happened. Preserve uncertainty across any subsequent rejection. + const timedOut = httpStatus === HTTP_REQUEST_TIMEOUT + || ['TimeoutError', 'RequestTimeout', 'RequestTimeoutException'].includes(serviceError?.name ?? ''); + const knownRejection = !timedOut && (httpStatus !== undefined + ? httpStatus >= 400 && httpStatus < 500 + : ['AccessDeniedException', 'UnauthorizedException', 'ValidationException', + 'InvalidParameterValueException', 'ResourceNotFoundException', 'ThrottlingException', + 'TooManyRequestsException', 'ServiceQuotaExceededException', 'ConflictException'] + .includes(serviceError?.name ?? '')); + if (!knownRejection) throw new MicrovmStartUncertainError(wrapped.message, { cause: err }); + throw wrapped; } const { microvmId, endpoint } = result; @@ -702,13 +735,33 @@ export class LambdaMicrovmComputeStrategy implements ComputeStrategy { // failed` bucket with "Check AgentCore Runtime or ECS cluster health" — // advice that names the wrong substrate entirely. `RunMicrovm` is the // operation because that is the call whose response is malformed. - throw wrapMicrovmError( + const incomplete = wrapMicrovmError( 'RunMicrovm', new Error( `RunMicrovm returned an incomplete response (microvmId=${microvmId ?? 'missing'}, ` + `endpoint=${endpoint ? 'present' : 'missing'}, state=${result.state ?? 'unknown'})`, ), ); + if (!microvmId) throw new MicrovmStartUncertainError(incomplete.message, { cause: incomplete }); + throw incomplete; + } + + const handle: Extract = { + sessionId: microvmId, strategyType: 'lambda-microvm', microvmId, endpoint, + }; + try { + await saveMicrovmStartHandle(taskId, latest.clientToken, handle); + } catch (err) { + // The write may have committed before its response was lost. Recover that + // receipt before destroying a computer whose handle is already durable. + try { + const saved = await claimMicrovmStart(taskId, input.userId, requestHash); + if (!saved.closed && saved.handle?.microvmId === microvmId) return saved.handle; + } catch (readErr) { + logger.warn('Could not reconcile the MicroVM start receipt', { task_id: taskId, error: String(readErr) }); + } + await this.terminateBestEffort(microvmId, 'start receipt could not save handle'); + throw new Error(`MICROVM_START_RECEIPT_SAVE_FAILED: ${String(err)}`, { cause: err }); } // Image ARN/version is logged, NOT carried in the handle (ADR-021 @@ -731,20 +784,8 @@ export class LambdaMicrovmComputeStrategy implements ComputeStrategy { platform_config_keys: Object.keys(platformConfig), }); - return { - // sessionId = microvmId, mirroring the ECS variant's "sessionId = the - // substrate identifier" precedent (ECS uses the task ARN) rather than - // AgentCore's fresh UUID — AgentCore only needs a UUID because - // `runtimeSessionId` is a caller-minted value that must be ≥ 33 chars. - // Here the substrate mints the id, every lifecycle API keys on it, and it - // is what an operator needs to correlate `TaskRecord.session_id` with the - // MicroVM in logs/console. A second synthetic UUID would add a - // non-actionable identifier and leave `session_id` un-joinable. - sessionId: microvmId, - strategyType: 'lambda-microvm', - microvmId, - endpoint, - }; + // Use AWS's identifier for both lifecycle calls and TaskRecord.session_id. + return handle; } /** diff --git a/cdk/test/handlers/orchestrate-task-microvm.test.ts b/cdk/test/handlers/orchestrate-task-microvm.test.ts index 1023238fd..7c55f89d1 100644 --- a/cdk/test/handlers/orchestrate-task-microvm.test.ts +++ b/cdk/test/handlers/orchestrate-task-microvm.test.ts @@ -74,6 +74,14 @@ const mockFinalizeTask = jest.fn(); const mockPollTaskStatus = jest.fn(); const mockReconcile = jest.fn(); const mockFailTask = jest.fn(); +const mockLoadTask = jest.fn(); +const mockClaimStart = jest.fn(); +const mockSaveHandle = jest.fn(); +jest.mock('../../src/handlers/shared/microvm-start', () => ({ + ...jest.requireActual('../../src/handlers/shared/microvm-start'), + claimMicrovmStart: (...args: unknown[]) => mockClaimStart(...args), + saveMicrovmStartHandle: (...args: unknown[]) => mockSaveHandle(...args), +})); jest.mock('../../src/handlers/shared/orchestrator', () => ({ admissionControl: jest.fn().mockResolvedValue(true), emitTaskEvent: (...a: unknown[]) => mockEmitTaskEvent(...a), @@ -85,9 +93,7 @@ jest.mock('../../src/handlers/shared/orchestrator', () => ({ finalizeTask: (...a: unknown[]) => mockFinalizeTask(...a), hydrateAndTransition: jest.fn().mockResolvedValue({ repo_url: 'org/repo', task_id: 'TASK001' }), loadBlueprintConfig: jest.fn().mockResolvedValue({ compute_type: 'lambda-microvm', runtime_arn: '' }), - loadTask: jest.fn().mockResolvedValue({ - task_id: 'TASK001', user_id: 'user-1', status: 'SUBMITTED', repo: 'org/repo', - }), + loadTask: (...args: unknown[]) => mockLoadTask(...args), pollTaskStatus: (...a: unknown[]) => mockPollTaskStatus(...a), reconcileMicrovmSubstrateState: (...a: unknown[]) => mockReconcile(...a), transitionTask: (...a: unknown[]) => mockTransitionTask(...a), @@ -134,11 +140,14 @@ import { LambdaMicrovmComputeStrategy } from '../../src/handlers/shared/strategi */ function fakeContext(opts: { pollOnce?: boolean } = {}) { const steps: string[] = []; + const stepConfigs: Record = {}; return { steps, + stepConfigs, ctx: { - step: async (name: string, fn: () => Promise) => { + step: async (name: string, fn: () => Promise, config?: unknown) => { steps.push(name); + stepConfigs[name] = config; return fn(); }, waitForCondition: async ( @@ -216,8 +225,24 @@ function s3CommandsOfType(type: string) { return mockS3Send.mock.calls.map(c => c[0]).filter(c => c._type === type); } +function failedTransition() { + return mockTransitionTask.mock.calls.find(([, , to]) => to === TaskStatus.FAILED)!; +} + beforeEach(() => { jest.clearAllMocks(); + mockTransitionTask.mockReset().mockResolvedValue(undefined); + mockEmitTaskEvent.mockReset().mockResolvedValue(undefined); + mockFinalizeTask.mockReset().mockResolvedValue(undefined); + mockFailTask.mockReset().mockResolvedValue(undefined); + mockLoadTask.mockReset().mockImplementation(async (_id: string, consistentRead?: boolean) => ({ + task_id: 'TASK001', + user_id: 'user-1', + status: consistentRead ? 'HYDRATING' : 'SUBMITTED', + repo: 'org/repo', + })); + mockClaimStart.mockReset().mockImplementation(async (taskId: string) => ({ clientToken: taskId, closed: false })); + mockSaveHandle.mockReset().mockResolvedValue(undefined); mockS3Send.mockReset(); mockS3Send.mockResolvedValue({}); mockMicrovmSend.mockReset(); @@ -226,6 +251,151 @@ beforeEach(() => { }); describe('orchestrate-task for a lambda-microvm task', () => { + test('replaying a persisted start failure reaches finalization once without an earlier slot release', async () => { + let failed = false; + mockMicrovmSend.mockRejectedValue(Object.assign(new Error('denied'), { name: 'AccessDeniedException' })); + mockClaimStart.mockImplementation(async () => ({ clientToken: 'TASK001', closed: failed })); + mockTransitionTask.mockImplementation(async (_id, _from, to) => { failed = to === TaskStatus.FAILED; }); + mockLoadTask.mockImplementation(async (_id, consistentRead) => ({ + task_id: 'TASK001', + user_id: 'user-1', + repo: 'org/repo', + status: consistentRead ? (failed ? TaskStatus.FAILED : TaskStatus.HYDRATING) : TaskStatus.SUBMITTED, + })); + const { ctx } = fakeContext(); + await handler({ task_id: 'TASK001' }, { + ...ctx, + step: async (name: string, fn: () => Promise, config?: unknown) => { + if (name === 'start-session') await fn(); + return ctx.step(name, fn, config); + }, + } as never); + expect(commandsOfType('RunMicrovm')).toHaveLength(1); + expect(mockTransitionTask).toHaveBeenCalledTimes(1); + expect(mockFailTask).not.toHaveBeenCalled(); + expect(mockFinalizeTask).toHaveBeenCalledTimes(1); + }); + + test('durable replay after registration recovers one computer without a second transition', async () => { + runMicrovmOk(); + let savedHandle: unknown; + let registered = false; + mockClaimStart.mockImplementation(async () => ({ + clientToken: 'TASK001', closed: false, ...(savedHandle ? { handle: savedHandle } : {}), + })); + mockSaveHandle.mockImplementation(async (_taskId, _token, handle) => { savedHandle = handle; }); + mockTransitionTask.mockImplementationOnce(async () => { registered = true; }); + mockLoadTask.mockImplementation(async (_id, consistentRead) => ({ + task_id: 'TASK001', + user_id: 'user-1', + repo: 'org/repo', + status: consistentRead ? (registered ? TaskStatus.RUNNING : TaskStatus.HYDRATING) : TaskStatus.SUBMITTED, + ...(registered && { session_id: MICROVM_ID }), + })); + const { ctx, stepConfigs } = fakeContext(); + const replayContext = { + ...ctx, + step: async (name: string, fn: () => Promise, config?: unknown) => { + if (name === 'start-session') await fn(); // response/checkpoint lost + return ctx.step(name, fn, config); + }, + }; + await handler({ task_id: 'TASK001' }, replayContext as never); + expect(commandsOfType('RunMicrovm')).toHaveLength(1); + expect(mockTransitionTask).toHaveBeenCalledTimes(1); + expect(stepConfigs['start-session'].retryStrategy(new Error('failed'), 1)).toEqual({ shouldRetry: false }); + expect(mockFailTask).not.toHaveBeenCalled(); + }); + + test('a lost task-transition response is reconciled before stopping a registered computer', async () => { + runMicrovmOk(); + let registered = false; + mockTransitionTask.mockImplementationOnce(async () => { + registered = true; + throw new Error('DynamoDB response lost'); + }); + mockLoadTask.mockImplementation(async (_id, consistentRead) => ({ + task_id: 'TASK001', + user_id: 'user-1', + repo: 'org/repo', + status: consistentRead ? (registered ? TaskStatus.RUNNING : TaskStatus.HYDRATING) : TaskStatus.SUBMITTED, + ...(registered && { session_id: MICROVM_ID }), + })); + await handler({ task_id: 'TASK001' }, fakeContext().ctx as never); + expect(mockFailTask).not.toHaveBeenCalled(); + expect(mockFinalizeTask).toHaveBeenCalledTimes(1); + expect(commandsOfType('RunMicrovm')).toHaveLength(1); + }); + + test('cancellation after creation terminates and finalizes without restoring RUNNING', async () => { + runMicrovmOk(); + mockLoadTask.mockImplementation(async (_id, consistentRead) => ({ + task_id: 'TASK001', + user_id: 'user-1', + repo: 'org/repo', + status: consistentRead ? TaskStatus.CANCELLED : TaskStatus.SUBMITTED, + })); + const { ctx, steps } = fakeContext(); + await handler({ task_id: 'TASK001' }, ctx as never); + expect(mockTransitionTask).not.toHaveBeenCalled(); + expect(mockFailTask).not.toHaveBeenCalled(); + expect(commandsOfType('TerminateMicrovm')).toHaveLength(1); + expect(mockFinalizeTask).toHaveBeenCalledWith('TASK001', { attempts: 0 }, 'user-1'); + expect(steps).toContain('finalize-before-session'); + expect(mockPollTaskStatus).not.toHaveBeenCalled(); + }); + + test('a session_started audit-event failure does not destroy a registered session', async () => { + runMicrovmOk(); + mockEmitTaskEvent.mockImplementationOnce(async (_id, eventType) => { + if (eventType === 'session_started') throw new Error('event write failed'); + }); + await handler({ task_id: 'TASK001' }, fakeContext().ctx as never); + expect(mockFailTask).not.toHaveBeenCalled(); + const terminateIndex = mockMicrovmSend.mock.calls.findIndex(([command]) => command._type === 'TerminateMicrovm'); + expect(mockMicrovmSend.mock.invocationCallOrder[terminateIndex]) + .toBeGreaterThan(mockFinalizeTask.mock.invocationCallOrder[0]); + }); + + test('two unanswered starts persist an unknown outcome instead of inviting another task', async () => { + const timeout = Object.assign(new Error('response lost'), { name: 'TimeoutError' }); + mockMicrovmSend.mockRejectedValue(timeout); + await handler({ task_id: 'TASK001' }, fakeContext().ctx as never); + expect(commandsOfType('RunMicrovm')).toHaveLength(2); + expect(failedTransition()[3].error_message).toContain('MICROVM_START_OUTCOME_UNKNOWN:'); + }); + + test('an unanswered start remains visible if the agent already changed the task to RUNNING', async () => { + mockMicrovmSend.mockRejectedValue(Object.assign(new Error('response lost'), { name: 'TimeoutError' })); + mockLoadTask.mockImplementation(async (_id, consistentRead) => ({ + task_id: 'TASK001', + user_id: 'user-1', + repo: 'org/repo', + status: consistentRead ? TaskStatus.RUNNING : TaskStatus.SUBMITTED, + })); + await handler({ task_id: 'TASK001' }, fakeContext().ctx as never); + expect(failedTransition()[1]).toBe(TaskStatus.RUNNING); + expect(failedTransition()[3].error_message).toContain('MICROVM_START_OUTCOME_UNKNOWN:'); + }); + + test('cancellation after a lost response records the unknown computer without starting another', async () => { + mockClaimStart.mockResolvedValueOnce({ clientToken: 'TASK001', closed: false }) + .mockResolvedValueOnce({ clientToken: 'TASK001', closed: false }) + .mockResolvedValue({ clientToken: 'TASK001', closed: true }); + mockMicrovmSend.mockRejectedValue(Object.assign(new Error('response lost'), { name: 'TimeoutError' })); + mockLoadTask.mockImplementation(async (_id, consistentRead) => ({ + task_id: 'TASK001', + user_id: 'user-1', + repo: 'org/repo', + status: consistentRead ? TaskStatus.CANCELLED : TaskStatus.SUBMITTED, + })); + await handler({ task_id: 'TASK001' }, fakeContext().ctx as never); + expect(commandsOfType('RunMicrovm')).toHaveLength(1); + expect(mockEmitTaskEvent).toHaveBeenCalledWith('TASK001', 'microvm_start_outcome_unknown', + { client_token: 'TASK001', task_status: TaskStatus.CANCELLED }, expect.any(Object)); + expect(mockFailTask).not.toHaveBeenCalled(); + }); + test('persists microvmId and endpoint in compute_metadata on the RUNNING transition', async () => { runMicrovmOk(); const { ctx } = fakeContext(); @@ -472,16 +642,14 @@ describe('orchestrate-task for a lambda-microvm task', () => { }); describe('orphan reap when session start fails AFTER RunMicrovm succeeded', () => { - test('terminates the MicroVM from the in-memory handle when the persist write fails', async () => { - // The MicroVM is already RUNNING and billing, and its id exists ONLY in this - // Lambda's memory — no poll or finalize step will ever see it. Nothing - // self-terminates on this substrate, so without the reap it bills for the - // full 8 h cap while holding admission-gating memory quota. + test('terminates the known MicroVM when registration fails before committing', async () => { + // The service created the computer, but registration did not commit. + // Its start receipt retains the ID; normal task polling has not started. runMicrovmOk(); mockTransitionTask.mockRejectedValueOnce(new Error('ConditionalCheckFailedException')); - await expect(handler({ task_id: 'TASK001' }, fakeContext().ctx as never)) - .rejects.toThrow('ConditionalCheckFailedException'); + await handler({ task_id: 'TASK001' }, fakeContext().ctx as never); + expect(failedTransition()[3].error_message).toContain('ConditionalCheckFailedException'); const terminates = commandsOfType('TerminateMicrovm'); expect(terminates).toHaveLength(1); @@ -492,11 +660,13 @@ describe('orchestrate-task for a lambda-microvm task', () => { runMicrovmOk(); mockTransitionTask.mockRejectedValueOnce(new Error('persist exploded')); - await expect(handler({ task_id: 'TASK001' }, fakeContext().ctx as never)) - .rejects.toThrow('persist exploded'); + await handler({ task_id: 'TASK001' }, fakeContext().ctx as never); + expect(failedTransition()[3].error_message).toContain('persist exploded'); - expect(mockFailTask).toHaveBeenCalledTimes(1); - const [, fromStatus, reason] = mockFailTask.mock.calls[0]; + expect(mockFailTask).not.toHaveBeenCalled(); + expect(mockFinalizeTask).toHaveBeenCalledTimes(1); + const [, fromStatus, , attrs] = failedTransition(); + const reason = attrs.error_message; expect(fromStatus).toBe(TaskStatus.HYDRATING); expect(reason).toContain('Session start failed'); expect(reason).toContain('persist exploded'); @@ -512,8 +682,8 @@ describe('orchestrate-task for a lambda-microvm task', () => { // stopSession is internally best-effort (it logs AccessDenied at error level // and returns), so the reap is a no-op here — the user must still see why // session start actually failed. - await expect(handler({ task_id: 'TASK001' }, fakeContext().ctx as never)) - .rejects.toThrow('persist exploded'); + await handler({ task_id: 'TASK001' }, fakeContext().ctx as never); + expect(failedTransition()[3].error_message).toContain('persist exploded'); }); test('even a stopSession that BREAKS its no-throw contract cannot mask the original error', async () => { @@ -527,8 +697,8 @@ describe('orchestrate-task for a lambda-microvm task', () => { .mockRejectedValueOnce(new Error('stopSession itself threw')); try { - await expect(handler({ task_id: 'TASK001' }, fakeContext().ctx as never)) - .rejects.toThrow('persist exploded'); + await handler({ task_id: 'TASK001' }, fakeContext().ctx as never); + expect(failedTransition()[3].error_message).toContain('persist exploded'); } finally { stopSpy.mockRestore(); } @@ -539,7 +709,7 @@ describe('orchestrate-task for a lambda-microvm task', () => { err.name = 'ThrottlingException'; mockMicrovmSend.mockRejectedValueOnce(err); - await expect(handler({ task_id: 'TASK001' }, fakeContext().ctx as never)).rejects.toThrow(); + await handler({ task_id: 'TASK001' }, fakeContext().ctx as never); expect(commandsOfType('TerminateMicrovm')).toHaveLength(0); }); diff --git a/cdk/test/handlers/orchestrate-task.test.ts b/cdk/test/handlers/orchestrate-task.test.ts index 297833ff4..808bc423e 100644 --- a/cdk/test/handlers/orchestrate-task.test.ts +++ b/cdk/test/handlers/orchestrate-task.test.ts @@ -1223,6 +1223,38 @@ describe('hydrateAndTransition — registry asset resolution (#246)', () => { }); describe('finalizeTask', () => { + test.each([ + ['FAILED', 'HYDRATING'], + ['CANCELLED', 'RUNNING'], + ])('honors committed %s instead of stale %s during immediate finalization', async (committed, stale) => { + mockDdbSend.mockImplementation(async (command) => { + if (command._type === 'Get') { + return { + Item: { + ...baseTask, + status: command.input.ConsistentRead ? committed : stale, + memory_written: true, + }, + }; + } + // A stale transition loses to the terminal status already in DynamoDB. + if (command.input.ExpressionAttributeValues?.[':fromStatus']) { + throw Object.assign(new Error('terminal state already committed'), { name: 'ConditionalCheckFailedException' }); + } + return {}; + }); + await finalizeTask('TASK001', { attempts: 0 }, 'user-123'); + const events = mockDdbSend.mock.calls + .filter(([command]) => command._type === 'Put') + .map(([command]) => command.input.Item); + expect(events).toEqual([expect.objectContaining({ + event_type: `task_${committed.toLowerCase()}`, + metadata: expect.objectContaining({ final_status: committed }), + })]); + expect(mockDdbSend.mock.calls.some(([command]) => + command.input.ExpressionAttributeValues?.[':fromStatus'])).toBe(false); + }); + test('handles already-terminal task', async () => { mockDdbSend .mockResolvedValueOnce({ Item: { ...baseTask, status: 'COMPLETED' } }) // loadTask diff --git a/cdk/test/handlers/shared/error-classifier.test.ts b/cdk/test/handlers/shared/error-classifier.test.ts index 83822d6d2..d9c385147 100644 --- a/cdk/test/handlers/shared/error-classifier.test.ts +++ b/cdk/test/handlers/shared/error-classifier.test.ts @@ -480,6 +480,16 @@ describe('classifyError', () => { // --- Lambda MicroVMs (ADR-021) --- describe('Lambda MicroVMs errors', () => { + test('an unknown start requires investigation even when its detail describes a transient failure', () => { + const result = classifyError( + 'Session start failed: MICROVM_START_OUTCOME_UNKNOWN: MicroVM RunMicrovm failed: TimeoutError: response lost [auto-retried]', + )!; + expect(result.retryable).toBe(false); + expect(result.errorClass).toBe(ErrorClass.SERVICE); + expect(retryGuidance(result, true)).toMatch(/needs your ABCA admin/); + expect(result.remedy).toMatch(/before submitting another task/); + }); + test('classifies regional unavailability as a non-retryable CONFIG fault with the supported-Region list', () => { // ADR-021: "If startSession fails because the MicroVM service is unavailable // in the stack region, then the orchestrator shall classify the failure with diff --git a/cdk/test/handlers/shared/microvm-start-recovery.test.ts b/cdk/test/handlers/shared/microvm-start-recovery.test.ts new file mode 100644 index 000000000..717f2044f --- /dev/null +++ b/cdk/test/handlers/shared/microvm-start-recovery.test.ts @@ -0,0 +1,305 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import { TaskStatus } from '../../../src/constructs/task-status'; + +const mockDdbSend = jest.fn(); +const mockMicrovmSend = jest.fn(); +const mockS3Send = jest.fn(); +jest.mock('@aws-sdk/client-dynamodb', () => ({ DynamoDBClient: jest.fn(() => ({})) })); +jest.mock('@aws-sdk/lib-dynamodb', () => ({ + DynamoDBDocumentClient: { from: jest.fn(() => ({ send: mockDdbSend })) }, + GetCommand: jest.fn((input: unknown) => ({ kind: 'get', input })), + UpdateCommand: jest.fn((input: unknown) => ({ kind: 'update', input })), +})); +jest.mock('@aws-sdk/client-lambda-microvms', () => ({ + LambdaMicrovmsClient: jest.fn(() => ({ send: mockMicrovmSend })), + RunMicrovmCommand: jest.fn((input: unknown) => ({ kind: 'run', input })), + TerminateMicrovmCommand: jest.fn((input: unknown) => ({ kind: 'terminate', input })), + MicrovmState: {}, +})); +jest.mock('@aws-sdk/client-s3', () => ({ + S3Client: jest.fn(() => ({ send: mockS3Send })), + PutObjectCommand: jest.fn((input: unknown) => ({ kind: 'put', input })), +})); +jest.mock('../../../src/handlers/shared/logger', () => ({ + logger: { info: jest.fn(), warn: jest.fn(), error: jest.fn() }, +})); + +Object.assign(process.env, { + TASK_TABLE_NAME: 'tasks', + TASK_EVENTS_TABLE_NAME: 'events', + MICROVM_IMAGE_IDENTIFIER: 'arn:aws:lambda:us-east-1:123456789012:microvm-image:test', + MICROVM_IMAGE_VERSION: '1', + MICROVM_EXECUTION_ROLE_ARN: 'arn:aws:iam::123456789012:role/execution', + MICROVM_EGRESS_CONNECTOR_ARNS: 'arn:aws:lambda:us-east-1:123456789012:network-connector:egress', + MICROVM_INGRESS_CONNECTOR_ARNS: 'arn:aws:lambda:us-east-1:aws:network-connector:aws-network-connector:NO_INGRESS', + MICROVM_PAYLOAD_BUCKET: 'payloads', + GITHUB_TOKEN_SECRET_ARN: 'arn:aws:secretsmanager:us-east-1:123456789012:secret:token', + AGENT_SESSION_ROLE_ARN: 'arn:aws:iam::123456789012:role/session', +}); + +import { MicrovmStartUncertainError } from '../../../src/handlers/shared/error-classifier'; +import { claimMicrovmStart, MICROVM_START_REPLAY_WINDOW_MS, microvmStartRequestHash } from '../../../src/handlers/shared/microvm-start'; +import { startSessionWithRetry } from '../../../src/handlers/shared/session-start-retry'; +import { LambdaMicrovmComputeStrategy } from '../../../src/handlers/shared/strategies/lambda-microvm-strategy'; + +const TASK_ID = '01K4YKCNV8P7WZBCSFDV2RNH49'; +const input = { + taskId: TASK_ID, + userId: 'user', + payload: { task_id: TASK_ID, prompt: 'x'.repeat(5_000) }, + blueprintConfig: { compute_type: 'lambda-microvm' as const, runtime_arn: '' }, +}; +const handle = { + strategyType: 'lambda-microvm' as const, + sessionId: 'mvm-one', + microvmId: 'mvm-one', + endpoint: 'https://example.invalid', +}; +let record: Record; +let lostResponses: number; +let created: Map>; +let now: number; + +beforeEach(() => { + jest.clearAllMocks(); + record = { user_id: 'user', status: TaskStatus.HYDRATING }; + lostResponses = 0; + created = new Map(); + now = Date.parse('2026-09-13T15:00:00Z'); + jest.spyOn(Date, 'now').mockImplementation(() => now); + mockS3Send.mockReset().mockResolvedValue({}); + mockDdbSend.mockReset().mockImplementation(async ({ kind, input: command }) => { + if (kind === 'get') return { Item: structuredClone(record) }; + const values = command.ExpressionAttributeValues; + if (values[':receipt']) { + if (record.microvm_start || record.status !== TaskStatus.HYDRATING) { + throw Object.assign(new Error('condition failed'), { name: 'ConditionalCheckFailedException' }); + } + record.microvm_start = structuredClone(values[':receipt']); + } else { + record.microvm_start.handle = values[':handle']; + record.session_id = values[':id']; + record.compute_type = values[':type']; + record.compute_metadata = values[':metadata']; + } + return {}; + }); + // Service emulator: same-token/same-request replay returns the original ID. + // This tests our use of the API contract; AWS retention still needs live proof. + mockMicrovmSend.mockReset().mockImplementation(async ({ kind, input: command }) => { + if (kind === 'terminate') return {}; + const token = command.clientToken ?? `sdk-generated-${created.size}`; + if (!created.has(token)) created.set(token, structuredClone(command)); + expect(created.get(token)).toEqual(command); + if (lostResponses-- > 0) { + throw Object.assign(new Error('response lost after creation'), { name: 'TimeoutError' }); + } + return { microvmId: handle.microvmId, endpoint: handle.endpoint, state: 'RUNNING' }; + }); +}); + +afterEach(() => jest.restoreAllMocks()); + +function runCalls() { + return mockMicrovmSend.mock.calls.filter(([command]) => command.kind === 'run'); +} + +test('a lost successful response retries one logical start and records the recovered handle', async () => { + lostResponses = 1; + const result = await startSessionWithRetry(new LambdaMicrovmComputeStrategy(), input, { + taskId: TASK_ID, emitRetryEvent: jest.fn(), logger: { warn: jest.fn() }, + }); + expect(result).toEqual({ handle, autoRetried: true }); + expect(runCalls()).toHaveLength(2); + expect(created.size).toBe(1); + expect(record.microvm_start.clientToken).toBe(TASK_ID); + expect(record.microvm_start.handle).toEqual(handle); + expect(record.compute_metadata).toEqual({ microvmId: handle.microvmId, endpoint: handle.endpoint }); + expect(mockDdbSend.mock.calls.filter(([command]) => command.kind === 'get') + .every(([command]) => command.input.ConsistentRead)).toBe(true); +}); + +test('a fresh strategy after a process restart reuses the persisted token', async () => { + lostResponses = 1; + await expect(new LambdaMicrovmComputeStrategy().startSession(input)).rejects.toThrow('response lost'); + const second = await new LambdaMicrovmComputeStrategy().startSession(input); + expect(second).toEqual(handle); + expect(created.size).toBe(1); + expect(runCalls()).toHaveLength(2); +}); + +test('a definite rejection after a lost response does not erase the first unknown outcome', async () => { + mockMicrovmSend + .mockRejectedValueOnce(Object.assign(new Error('response lost'), { name: 'TimeoutError' })) + .mockRejectedValueOnce(Object.assign(new Error('denied'), { name: 'AccessDeniedException' })); + await expect(startSessionWithRetry(new LambdaMicrovmComputeStrategy(), input, { + taskId: TASK_ID, emitRetryEvent: jest.fn(), logger: { warn: jest.fn() }, + })).rejects.toBeInstanceOf(MicrovmStartUncertainError); +}); + +test.each([ + { name: 'RequestTimeout', $metadata: { httpStatusCode: 400 } }, + { name: 'RequestTimeoutException', $metadata: { httpStatusCode: 400 } }, + { name: 'ServiceError', $metadata: { httpStatusCode: 408 } }, +])('a service timeout stays uncertain despite its HTTP status ($name)', async (timeout) => { + mockMicrovmSend + .mockRejectedValueOnce(Object.assign(new Error('start response timed out'), timeout)) + .mockRejectedValueOnce(Object.assign(new Error('denied'), { name: 'AccessDeniedException' })); + await expect(startSessionWithRetry(new LambdaMicrovmComputeStrategy(), input, { + taskId: TASK_ID, emitRetryEvent: jest.fn(), logger: { warn: jest.fn() }, + })).rejects.toBeInstanceOf(MicrovmStartUncertainError); + expect(runCalls().map(([command]) => command.input.clientToken)).toEqual([TASK_ID, TASK_ID]); +}); + +test('a confirmed rejection may retry the same token without creating an extra computer', async () => { + mockMicrovmSend.mockRejectedValueOnce(Object.assign(new Error('quota'), { name: 'ServiceQuotaExceededException' })); + const result = await startSessionWithRetry(new LambdaMicrovmComputeStrategy(), input, { + taskId: TASK_ID, emitRetryEvent: jest.fn(), logger: { warn: jest.fn() }, + }); + expect(result.handle).toEqual(handle); + expect(created.size).toBe(1); + expect(runCalls().map(([command]) => command.input.clientToken)).toEqual([TASK_ID, TASK_ID]); +}); + +test('different tasks receive different persisted tokens', async () => { + expect((await claimMicrovmStart(TASK_ID, 'user', 'hash')).clientToken).toBe(TASK_ID); + record = { user_id: 'user', status: TaskStatus.HYDRATING }; + expect((await claimMicrovmStart('another-task', 'user', 'hash')).clientToken).toBe('another-task'); +}); + +test('a task owner mismatch prevents all start side effects', async () => { + record.user_id = 'different-user'; + await expect(new LambdaMicrovmComputeStrategy().startSession(input)).rejects.toThrow('owner'); + expect(mockS3Send).not.toHaveBeenCalled(); + expect(mockMicrovmSend).not.toHaveBeenCalled(); +}); + +test('replay after saving a handle makes no further RunMicrovm or payload write', async () => { + await new LambdaMicrovmComputeStrategy().startSession(input); + now += MICROVM_START_REPLAY_WINDOW_MS * 10; + expect(await new LambdaMicrovmComputeStrategy().startSession(input)).toEqual(handle); + expect(runCalls()).toHaveLength(1); + expect(mockS3Send).toHaveBeenCalledTimes(1); +}); + +test('changed input is refused before overwriting the first task payload', async () => { + lostResponses = 1; + await expect(new LambdaMicrovmComputeStrategy().startSession(input)).rejects.toThrow(); + await expect(new LambdaMicrovmComputeStrategy().startSession({ + ...input, payload: { ...input.payload, prompt: 'changed'.repeat(1_000) }, + })).rejects.toThrow('MICROVM_START_INPUT_CHANGED'); + expect(mockS3Send).toHaveBeenCalledTimes(1); + expect(runCalls()).toHaveLength(1); +}); + +test('an expired unknown start never receives a new token or another RunMicrovm call', async () => { + lostResponses = 1; + await expect(new LambdaMicrovmComputeStrategy().startSession(input)).rejects.toThrow(); + now += MICROVM_START_REPLAY_WINDOW_MS; + await expect(new LambdaMicrovmComputeStrategy().startSession(input)) + .rejects.toThrow('MICROVM_START_OUTCOME_UNKNOWN'); + expect(runCalls()).toHaveLength(1); + expect(mockS3Send).toHaveBeenCalledTimes(1); +}); + +test.each([TaskStatus.CANCELLED, TaskStatus.COMPLETED, TaskStatus.FAILED, TaskStatus.TIMED_OUT])( + '%s before starting never creates a MicroVM or writes a payload', async (status) => { + record.status = status; + await expect(new LambdaMicrovmComputeStrategy().startSession(input)).rejects.toThrow('MICROVM_START_TASK_CLOSED'); + expect(mockMicrovmSend).not.toHaveBeenCalled(); + expect(mockS3Send).not.toHaveBeenCalled(); + }, +); + +test('cancellation during payload upload prevents RunMicrovm', async () => { + mockS3Send.mockImplementationOnce(async () => { record.status = TaskStatus.CANCELLED; }); + await expect(new LambdaMicrovmComputeStrategy().startSession(input)).rejects.toThrow('MICROVM_START_TASK_CLOSED'); + expect(runCalls()).toHaveLength(0); +}); + +test('a cancelled task with a saved handle reaps that computer instead of starting another', async () => { + await new LambdaMicrovmComputeStrategy().startSession(input); + record.status = TaskStatus.CANCELLED; + await expect(new LambdaMicrovmComputeStrategy().startSession(input)).rejects.toThrow('MICROVM_START_TASK_CLOSED'); + expect(runCalls()).toHaveLength(1); + expect(mockMicrovmSend).toHaveBeenLastCalledWith({ + kind: 'terminate', input: { microvmIdentifier: handle.microvmId }, + }); +}); + +test('failure to save a returned handle terminates the known computer', async () => { + const normal = mockDdbSend.getMockImplementation()!; + mockDdbSend.mockImplementation(async (command) => { + if (command.input.ExpressionAttributeValues?.[':handle']) throw new Error('DynamoDB unavailable'); + return normal(command); + }); + await expect(new LambdaMicrovmComputeStrategy().startSession(input)) + .rejects.toThrow('MICROVM_START_RECEIPT_SAVE_FAILED'); + expect(mockMicrovmSend).toHaveBeenLastCalledWith({ + kind: 'terminate', input: { microvmIdentifier: handle.microvmId }, + }); +}); + +test('a lost receipt-write response recovers the committed handle without termination', async () => { + const normal = mockDdbSend.getMockImplementation()!; + mockDdbSend.mockImplementation(async (command) => { + const result = await normal(command); + if (command.input.ExpressionAttributeValues?.[':handle']) throw new Error('write response lost'); + return result; + }); + expect(await new LambdaMicrovmComputeStrategy().startSession(input)).toEqual(handle); + expect(mockMicrovmSend).toHaveBeenCalledTimes(1); + expect(record.microvm_start.handle).toEqual(handle); +}); + +test.each(['get', 'update'])( + 'a failed initial receipt %s prevents both payload upload and service creation', async (failedOperation) => { + const normal = mockDdbSend.getMockImplementation()!; + mockDdbSend.mockImplementation(async (command) => { + if (command.kind === failedOperation) throw new Error('DynamoDB unavailable'); + return normal(command); + }); + await expect(new LambdaMicrovmComputeStrategy().startSession(input)).rejects.toThrow('DynamoDB unavailable'); + expect(mockS3Send).not.toHaveBeenCalled(); + expect(mockMicrovmSend).not.toHaveBeenCalled(); + }, +); + +test('a competing receipt claim is re-read rather than overwritten', async () => { + const normal = mockDdbSend.getMockImplementation()!; + let raced = false; + mockDdbSend.mockImplementation(async (command) => { + if (command.input.ExpressionAttributeValues?.[':receipt'] && !raced) { + raced = true; + record.microvm_start = command.input.ExpressionAttributeValues[':receipt']; + } + return normal(command); + }); + expect(await claimMicrovmStart(TASK_ID, 'user', 'hash')).toEqual({ clientToken: TASK_ID, closed: false }); + expect(mockDdbSend.mock.calls.filter(([command]) => command.kind === 'update')).toHaveLength(1); +}); + +test('request fingerprints cover platform settings and full S3 content', () => { + const first = microvmStartRequestHash({ image: 'a', pointer: 's3://b/k' }, { prompt: 'one', repo: 'r' }); + expect(microvmStartRequestHash({ pointer: 's3://b/k', image: 'a' }, { repo: 'r', prompt: 'one' })).toBe(first); + expect(microvmStartRequestHash({ image: 'b', pointer: 's3://b/k' }, { prompt: 'one', repo: 'r' })).not.toBe(first); + expect(microvmStartRequestHash({ image: 'a', pointer: 's3://b/k' }, { prompt: 'two', repo: 'r' })).not.toBe(first); +}); diff --git a/cdk/test/handlers/shared/strategies/lambda-microvm-strategy.test.ts b/cdk/test/handlers/shared/strategies/lambda-microvm-strategy.test.ts index 43a062041..ac213407d 100644 --- a/cdk/test/handlers/shared/strategies/lambda-microvm-strategy.test.ts +++ b/cdk/test/handlers/shared/strategies/lambda-microvm-strategy.test.ts @@ -77,6 +77,13 @@ for (const optional of [ } const mockSend = jest.fn(); +const mockClaimStart = jest.fn(); +const mockSaveHandle = jest.fn(); +jest.mock('../../../../src/handlers/shared/microvm-start', () => ({ + ...jest.requireActual('../../../../src/handlers/shared/microvm-start'), + claimMicrovmStart: (...args: unknown[]) => mockClaimStart(...args), + saveMicrovmStartHandle: (...args: unknown[]) => mockSaveHandle(...args), +})); jest.mock('@aws-sdk/client-lambda-microvms', () => ({ LambdaMicrovmsClient: jest.fn(() => ({ send: mockSend })), RunMicrovmCommand: jest.fn((input: unknown) => ({ _type: 'RunMicrovm', input })), @@ -266,6 +273,8 @@ async function withEnvAsync(env: Record, body: () => Promise { jest.clearAllMocks(); + mockClaimStart.mockReset().mockImplementation(async (taskId: string) => ({ clientToken: taskId, closed: false })); + mockSaveHandle.mockReset().mockResolvedValue(undefined); }); describe('LambdaMicrovmComputeStrategy', () => { diff --git a/cdk/test/handlers/start-session-composition.test.ts b/cdk/test/handlers/start-session-composition.test.ts index 1d2e816b9..3f90a3d8a 100644 --- a/cdk/test/handlers/start-session-composition.test.ts +++ b/cdk/test/handlers/start-session-composition.test.ts @@ -20,8 +20,8 @@ /** * Integration-style tests for the start-session step composition: * resolveComputeStrategy → strategy.startSession → transitionTask → emitTaskEvent - * These verify that the orchestrate-task handler's step 4 logic correctly - * wires the strategy, state transitions, and event emission together. + * These compose the real helpers. The actual durable handler, including + * MicroVM recovery/finalization, is exercised in orchestrate-task-microvm.test.ts. */ const mockDdbSend = jest.fn(); @@ -199,6 +199,18 @@ describe('start-session step composition — lambda-microvm (ADR-021)', () => { const blueprintConfig: BlueprintConfig = { compute_type: 'lambda-microvm', runtime_arn: '' }; const payload = { repo_url: 'org/repo', task_id: taskId }; + beforeEach(() => { + let receipt: unknown; + mockDdbSend.mockReset().mockImplementation(async (command) => { + if (command._type === 'Get') { + return { Item: { user_id: 'cognito-test', status: TaskStatus.HYDRATING, microvm_start: receipt } }; + } + const saved = command.input.ExpressionAttributeValues?.[':receipt']; + if (saved) receipt = saved; + return {}; + }); + }); + test('startSession → buildComputeMetadata → transitionTask persists microvmId and endpoint', async () => { mockMicrovmSend.mockResolvedValueOnce({ microvmId: MICROVM_ID, @@ -207,8 +219,6 @@ describe('start-session step composition — lambda-microvm (ADR-021)', () => { imageArn: 'arn:image', imageVersion: '7', }); - mockDdbSend.mockResolvedValue({}); - const strategy = resolveComputeStrategy(blueprintConfig); const handle = await strategy.startSession({ taskId, userId: 'cognito-test', payload, blueprintConfig }); @@ -225,7 +235,8 @@ describe('start-session step composition — lambda-microvm (ADR-021)', () => { // The persisted attributes are what cancel-task (and P3's approve/deny // resume) read back, so assert them on the real UpdateCommand input. - const update = mockDdbSend.mock.calls.find(c => c[0]._type === 'Update')![0]; + const update = mockDdbSend.mock.calls.find(c => + c[0]._type === 'Update' && c[0].input.ExpressionAttributeValues[':toStatus'] === TaskStatus.RUNNING)![0]; const values = update.input.ExpressionAttributeValues as Record; expect(values[':attr_compute_type']).toBe('lambda-microvm'); expect(values[':attr_compute_metadata']).toEqual({ microvmId: MICROVM_ID, endpoint: ENDPOINT }); @@ -251,19 +262,15 @@ describe('start-session step composition — lambda-microvm (ADR-021)', () => { expect(JSON.stringify(metadata)).not.toContain('microvm-image'); }); - test('error path: a marked RunMicrovm failure flows into failTask', async () => { + test('a rejected RunMicrovm preserves its marked AWS exception for the caller', async () => { const err = new Error('Rate exceeded'); err.name = 'ThrottlingException'; mockMicrovmSend.mockRejectedValueOnce(err); - mockDdbSend.mockResolvedValue({}); const strategy = resolveComputeStrategy(blueprintConfig); await expect( strategy.startSession({ taskId, userId: 'cognito-test', payload, blueprintConfig }), ).rejects.toThrow('MicroVM RunMicrovm failed: ThrottlingException: Rate exceeded'); - - await failTask(taskId, TaskStatus.HYDRATING, 'Session start failed: boom', 'user-123', true); - expect(mockDdbSend).toHaveBeenCalled(); }); }); diff --git a/docs/design/ORCHESTRATOR.md b/docs/design/ORCHESTRATOR.md index 8e02ccd73..08e4f8647 100644 --- a/docs/design/ORCHESTRATOR.md +++ b/docs/design/ORCHESTRATOR.md @@ -205,7 +205,13 @@ The orchestrator resolves the repository's `ComputeStrategy` and calls `startSes AgentCore's session ID is pre-generated and reused on retry. ECS and Lambda MicroVMs use their substrate identifiers as session IDs. -If `RunMicrovm` succeeds but persisting the session handle or emitting the start event fails, the start step terminates the MicroVM best-effort using its in-memory handle before propagating the original error. This orphan reap is required because no later poll or finalization step can recover an unpersisted handle. +MicroVM starts first save an internal `microvm_start` receipt on the task: its stable client token (the task ID), a fingerprint of the request and full payload, creation time, and a local replay deadline. This happens before payload upload or `RunMicrovm`. Retries must match the saved fingerprint; a changed request cannot overwrite the earlier task's input. A returned handle is saved in the receipt and the normal task metadata before registration finishes. A replay can recover that handle without another start call. Registration uses strongly consistent reads to observe cancellation and already-committed writes. + +If a handle-save or registration response is lost, the code checks the committed task before terminating a known computer. If registration failed, cleanup remains best-effort and the receipt retains any saved handle for diagnosis. Failure to emit `session_started` alone does not terminate a registered MicroVM. MicroVM start failures persist the task outcome and reach `finalize-before-session`; they do not also release concurrency in the start step. Finalization begins with a strongly consistent read so a recently saved failure or cancellation is not reported using an older active state. + +An unanswered service request may already have created a MicroVM. Recovery reuses the same token and request within a **120-second local window**; after the window, it refuses another `RunMicrovm` call. This window is an application guard, not a verified AWS token-retention promise. An unrecovered request is reported as `MICROVM_START_OUTCOME_UNKNOWN` and requires inspection before submitting another task. Cancellation after an unanswered request records a `microvm_start_outcome_unknown` event. Without an ID, immediate termination cannot be guaranteed; the eight-hour service lifetime bound still applies. + +The MicroVM `start-session` step disables automatic durable **error** retries after its own recovery attempt. Crash replay is still possible and uses the saved receipt. Live AWS token-retention, changed-request and concurrent-conflict behavior remain verification gates. ### Step 5: Await completion @@ -246,9 +252,9 @@ After the session ends, the orchestrator determines the outcome from multiple si ### Step execution contract -Every step in the pipeline satisfies these properties: +Configured workflow steps target the following contract. The top-level durable orchestrator still needs explicit guards around external effects, as described under recovery below. -- **Idempotent** - Safe to retry after crashes. Context hydration produces the same prompt for the same inputs; session-start retry semantics are implemented by each backend strategy. +- **Replay-aware** - A retry must preserve task intent and avoid repeating external effects. Session-start recovery is implemented by each backend strategy; a checkpoint alone is not an idempotency guarantee. - **Timeout-bounded** - Each step has a configurable timeout to prevent blocking the pipeline. - **Failure-aware** - Returns `success` or `failed`. Infrastructure failures (throttle, transient errors) trigger exponential backoff retries (default: 2 retries, base 1s, max 10s). Explicit failures transition to `FAILED` without retry. - **Least-privilege input** - Each step receives only the `blueprintConfig` fields it needs. Custom Lambda steps get credential ARNs stripped. @@ -325,7 +331,7 @@ Long-running distributed systems fail. The orchestrator is designed so that ever ### Recovery mechanisms 1. **Durable execution** - Lambda Durable Functions checkpoints at each state transition and replays after crashes. -2. **Idempotent operations** - All steps are safe to retry. +2. **Replay guards** - Operations need their own idempotency controls; checkpointing alone does not make external effects exactly-once. MicroVM starts use saved receipts. The shared finalizer's direct concurrency decrement still needs a per-task atomic release guard and a crash-replay test. 3. **Stuck-task scanner** - Periodic Lambda detects tasks stuck beyond expected durations and either resumes or fails them. 4. **Counter reconciliation** - Lambda runs every 15 minutes, compares counters to actual running task counts, corrects drift. Emits `counter_drift_corrected` CloudWatch metric. 5. **Dead-letter queue** - Tasks that exhaust retries go to DLQ for investigation. diff --git a/docs/src/content/docs/architecture/Orchestrator.md b/docs/src/content/docs/architecture/Orchestrator.md index 2e6e6288d..163c79905 100644 --- a/docs/src/content/docs/architecture/Orchestrator.md +++ b/docs/src/content/docs/architecture/Orchestrator.md @@ -209,7 +209,13 @@ The orchestrator resolves the repository's `ComputeStrategy` and calls `startSes AgentCore's session ID is pre-generated and reused on retry. ECS and Lambda MicroVMs use their substrate identifiers as session IDs. -If `RunMicrovm` succeeds but persisting the session handle or emitting the start event fails, the start step terminates the MicroVM best-effort using its in-memory handle before propagating the original error. This orphan reap is required because no later poll or finalization step can recover an unpersisted handle. +MicroVM starts first save an internal `microvm_start` receipt on the task: its stable client token (the task ID), a fingerprint of the request and full payload, creation time, and a local replay deadline. This happens before payload upload or `RunMicrovm`. Retries must match the saved fingerprint; a changed request cannot overwrite the earlier task's input. A returned handle is saved in the receipt and the normal task metadata before registration finishes. A replay can recover that handle without another start call. Registration uses strongly consistent reads to observe cancellation and already-committed writes. + +If a handle-save or registration response is lost, the code checks the committed task before terminating a known computer. If registration failed, cleanup remains best-effort and the receipt retains any saved handle for diagnosis. Failure to emit `session_started` alone does not terminate a registered MicroVM. MicroVM start failures persist the task outcome and reach `finalize-before-session`; they do not also release concurrency in the start step. Finalization begins with a strongly consistent read so a recently saved failure or cancellation is not reported using an older active state. + +An unanswered service request may already have created a MicroVM. Recovery reuses the same token and request within a **120-second local window**; after the window, it refuses another `RunMicrovm` call. This window is an application guard, not a verified AWS token-retention promise. An unrecovered request is reported as `MICROVM_START_OUTCOME_UNKNOWN` and requires inspection before submitting another task. Cancellation after an unanswered request records a `microvm_start_outcome_unknown` event. Without an ID, immediate termination cannot be guaranteed; the eight-hour service lifetime bound still applies. + +The MicroVM `start-session` step disables automatic durable **error** retries after its own recovery attempt. Crash replay is still possible and uses the saved receipt. Live AWS token-retention, changed-request and concurrent-conflict behavior remain verification gates. ### Step 5: Await completion @@ -250,9 +256,9 @@ After the session ends, the orchestrator determines the outcome from multiple si ### Step execution contract -Every step in the pipeline satisfies these properties: +Configured workflow steps target the following contract. The top-level durable orchestrator still needs explicit guards around external effects, as described under recovery below. -- **Idempotent** - Safe to retry after crashes. Context hydration produces the same prompt for the same inputs; session-start retry semantics are implemented by each backend strategy. +- **Replay-aware** - A retry must preserve task intent and avoid repeating external effects. Session-start recovery is implemented by each backend strategy; a checkpoint alone is not an idempotency guarantee. - **Timeout-bounded** - Each step has a configurable timeout to prevent blocking the pipeline. - **Failure-aware** - Returns `success` or `failed`. Infrastructure failures (throttle, transient errors) trigger exponential backoff retries (default: 2 retries, base 1s, max 10s). Explicit failures transition to `FAILED` without retry. - **Least-privilege input** - Each step receives only the `blueprintConfig` fields it needs. Custom Lambda steps get credential ARNs stripped. @@ -329,7 +335,7 @@ Long-running distributed systems fail. The orchestrator is designed so that ever ### Recovery mechanisms 1. **Durable execution** - Lambda Durable Functions checkpoints at each state transition and replays after crashes. -2. **Idempotent operations** - All steps are safe to retry. +2. **Replay guards** - Operations need their own idempotency controls; checkpointing alone does not make external effects exactly-once. MicroVM starts use saved receipts. The shared finalizer's direct concurrency decrement still needs a per-task atomic release guard and a crash-replay test. 3. **Stuck-task scanner** - Periodic Lambda detects tasks stuck beyond expected durations and either resumes or fails them. 4. **Counter reconciliation** - Lambda runs every 15 minutes, compares counters to actual running task counts, corrects drift. Emits `counter_drift_corrected` CloudWatch metric. 5. **Dead-letter queue** - Tasks that exhaust retries go to DLQ for investigation. From 0f4545c77f01c5d905d9e32b8b375d9910e18324 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Sun, 13 Sep 2026 15:52:32 -0400 Subject: [PATCH 011/149] docs(645): record MicroVM start recovery and remaining gates --- .../645-p3-implementation-plan.md | 30 +++++++++++++++++-- docs/verification/645-p3-readiness-review.md | 2 ++ 2 files changed, 29 insertions(+), 3 deletions(-) diff --git a/docs/verification/645-p3-implementation-plan.md b/docs/verification/645-p3-implementation-plan.md index 2ee6ad89c..2d978281f 100644 --- a/docs/verification/645-p3-implementation-plan.md +++ b/docs/verification/645-p3-implementation-plan.md @@ -14,7 +14,9 @@ Prerequisite work is tracked here on `fix/645-microvm-readiness`. “Completed - [x] Exercise real S3 bad-byte paths and fix closed-stream error classification (#817). - [x] Require new ARN fields to participate in validation; pin contract fields and anchor (#817). - [ ] Bind configuration to trusted deployment identity and restrict payload reads per task (#817 / #700). -- [ ] Resolve uncertain session-start retries without duplicate or orphan VMs. +- [x] Implement saved MicroVM start receipts, stable tokens, input fingerprints and handle recovery. +- [ ] Verify AWS token retention/conflicts and unknown-start cleanup on a live deployment. +- [ ] Make the shared finalizer's concurrency release atomic per task across crash replay. - [ ] Finish logging-failure observability (#810) and registry-overflow coverage (#818). - [ ] Implement production nesting if included, then verify a clean P2 deployment. - [ ] Implement and verify the P3 sleep/wake lifecycle described below. @@ -39,7 +41,17 @@ Second prerequisite batch completed locally on 2026-09-13: | `2e20d54a` | #817: stable terminal failure codes, precise region/auth/config guidance and consistent task/reply classification | New regressions reproduced the prior misclassification. Tests assert saved message → task API → channel/panel guidance, legacy records, hook 4xx/5xx and start-retry decisions. | | `cfe95c5e` | #817: exact contract assertions and reverse ARN-validation guards | Negative mutations failed in both Python and the real constants-checker subprocess before the fix. New ARN fields pass only when added to the validation set. | -Final checks for the second batch: `mise run quality` passed **1,797 Python tests**, lint, formatting and type checks (**83.87%** coverage). Eight relevant CDK suites passed **476 tests**; CDK ESLint, TypeScript compilation, constants-sync and Markdown link checks passed. These counts describe the selected suites for each batch, not additional disjoint tests. No cloud resources were changed. Deploy the orchestrator update and an updated agent image to use these fixes; trusted configuration provenance, task-scoped payload reads, uncertain-start retries and P3 remain open. +Final checks for the second batch: `mise run quality` passed **1,797 Python tests**, lint, formatting and type checks (**83.87%** coverage). Eight relevant CDK suites passed **476 tests**; CDK ESLint, TypeScript compilation, constants-sync and Markdown link checks passed. These counts describe the selected suites for each batch, not additional disjoint tests. No cloud resources were changed. Deploy the orchestrator update and an updated agent image to use these fixes. At the close of that batch, trusted configuration provenance, task-scoped payload reads, uncertain-start retries and P3 remained open. + +Third prerequisite batch completed locally on 2026-09-13: + +| Commit | Completed work | Proof | +|---|---|---| +| `e9944475` | Saved MicroVM start receipts, stable request tokens, immutable payload fingerprints, bounded replay and handle recovery; cancellation/registration reconciliation; one start-failure finalization path; consistent initial finalizer reads | Fault injection covers a lost successful response, process restart, saved-handle reuse, changed/expired requests, cancellation and failed/lost database writes. Three HTTP-timeout cases and two stale-finalization cases reproduced incorrect outcomes before their fixes. | + +Final checks for the third batch: CDK ESLint and TypeScript compilation passed. `mise run testf -- test/handlers/ --detectOpenHandles` passed **151 suites / 3,557 tests** and exited successfully. An earlier ordinary broad run reported a delayed-exit warning; the diagnostic run produced no open-handle trace, so its cause remains unidentified. The changed start/recovery integration suites also exited cleanly in isolation. Documentation sync, the **77-page** Astro build and Markdown link checks passed. No Python source changed in this batch. + +Deploy the orchestrator update to activate these changes. This batch needs no additional IAM/bootstrap change or agent-image update. The receipt is internal task-table data, not a new public task field. The service emulator proves our retry behavior; AWS token retention/conflicts and unknown-ID cleanup still need live evidence. The shared finalizer's atomic capacity-slot release and the remaining P2/P3 work stay unchecked above. ## The result we want @@ -126,12 +138,24 @@ Regression coverage includes a task that waits over 240 seconds and resumes befo ### 1D. Make session-start retries honest -Fault-inject a successful service-side creation followed by a lost response. Observe a second application start attempt, not just retries inside one SDK command. Define a persisted logical start-attempt identity and stable client token for retries of an uncertain attempt. Mint a new attempt only when the old session is known terminal or the service's idempotency semantics require it. Test same-request replay, confirmed failed start, durable Lambda replay, cancellation and orphan cleanup. Check AWS token retention/conflict semantics before finalizing the policy; ECS's existing `clientToken: taskId` is useful precedent, not proof that MicroVM has identical semantics. +**Implemented locally:** `microvm-start.ts` conditionally records one start per task, using the task ID as its stable token. The internal `microvm_start` attribute stores the request fingerprint, creation time, a 120-second local replay deadline and any recovered handle. The fingerprint includes the full S3 payload, so a changed retry cannot overwrite the first task's instructions. It is checked before uploads and again before `RunMicrovm`. A new attempt requires a new task ID. + +The receipt works like an order number: when the reply gets lost, the next call asks about the same order. + +Fault tests exercise a successful simulated service creation followed by a lost response, a second application call with the same token, a fresh strategy instance, saved-handle replay, changed input, expired recovery, confirmed rejection, cancellation before/during/after creation, and lost DynamoDB responses. The handler recovers committed registration, treats start-audit failures as non-fatal, and routes start failures through one finalization path. Finalization reads the latest committed task, avoiding a stale cancellation/failure report. An unknown first outcome stays unknown even when the second call gets a definite rejection; HTTP 408 and named service timeouts remain uncertain even with a 4xx status. + +**Still required:** the installed SDK documents `clientToken` idempotency but gives no retention period. Public AWS API documentation URLs did not provide a usable RunMicrovm reference during this review. The local 120-second limit is a conservative application cutoff, not evidence of AWS's retention window. Verify same-token replay, changed parameters, simultaneous requests/conflicts, token expiry and returned handles after termination against AWS before accepting this prerequisite. The service emulator proves client behavior only. + +For an unknown outcome with no returned ID, the task error or cancellation event identifies the saved token for investigation. Do not automatically submit a replacement task. Verify how operators find and terminate that VM in the deployed service; if they cannot recover an ID, the eight-hour lifetime cap is the remaining bound. Keep this limitation explicit in live evidence. ### 1E. Make verification observable Resolve #810 by exposing a useful structured failure signal for CloudWatch writers or removing the dead counter and using another observable signal. Test failure of the logging system itself. For #818, document the shared runtime 443-only rule and test registry payload overflow; do not expand network ports just to satisfy an incorrect issue premise. +### 1F. Make concurrency release safe across replay + +`finalizeTask` currently decrements the user's counter directly after terminal events, without atomically recording a per-task release. A crash after the decrement but before the durable checkpoint can repeat it. Reproduce that exact failure locally, then make the release marker and counter change one conditional transaction. Cover normal completion, start failure, cancellation, timeout, competing finalizers, retries and an already-zero counter. Preserve admission-queue behavior. The new MicroVM start path avoids an additional early release; it does not close this shared finalizer window. + ## 2. Nest infrastructure if adopting the split 1. Introduce `LambdaMicrovmStack` as a `NestedStack` wrapper and an explicit way to supply the MicroVM execution role from the parent. Keep shared session-role trust and runtime-role ownership in the parent. Preserve exact grant behavior. diff --git a/docs/verification/645-p3-readiness-review.md b/docs/verification/645-p3-readiness-review.md index 24454696e..267e8eabf 100644 --- a/docs/verification/645-p3-readiness-review.md +++ b/docs/verification/645-p3-readiness-review.md @@ -6,6 +6,8 @@ This review covers the existing MicroVM implementation, related P2 follow-ups, c **Implementation update (2026-09-13):** subsequent prerequisite work fixes #841 thread isolation, #817 coordinator payload-deletion permission, terminal-failure classification, closed S3-stream handling, and the approval-resume heartbeat race locally. Real bad-byte route tests and ARN contract guards are also added. Trusted deployment identity, task-scoped payload reads and live verification remain open. The findings below preserve the reviewed baseline; see [implementation progress](./645-p3-implementation-plan.md#implementation-progress) for commits, checks and remaining work. +Further local work adds durable MicroVM start receipts, stable request tokens and bounded handle recovery. AWS token-retention/conflict behavior and unknown-ID cleanup remain live verification gates. Reviewing crash replay also exposed a shared finalizer risk: its counter decrement lacks an atomic per-task release marker. That source-level finding is now a prerequisite in the implementation plan; it has not yet been reproduced in AWS or fixed. + ## Start here: the pieces in plain language A **MicroVM** is a small, isolated computer rented from AWS. **Firecracker** is the technology that keeps these small computers separate. A **backend** is the kind of rented computer ABCA chooses to run a coding task. From 0bb9524313d3ca1e96d60607252a07c4e3ecfafb Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Sun, 13 Sep 2026 16:24:35 -0400 Subject: [PATCH 012/149] fix(orchestrator): reserve and release task capacity atomically (#645) --- cdk/src/constructs/concurrency-reconciler.ts | 7 +- .../constructs/stranded-task-reconciler.ts | 4 +- cdk/src/constructs/task-api.ts | 4 +- cdk/src/constructs/task-status.ts | 2 +- cdk/src/constructs/user-concurrency-table.ts | 3 +- cdk/src/handlers/confirm-uploads.ts | 87 +---- cdk/src/handlers/orchestrate-task.ts | 7 +- cdk/src/handlers/reconcile-admission-queue.ts | 4 +- cdk/src/handlers/reconcile-concurrency.ts | 195 ++++++----- cdk/src/handlers/reconcile-stranded-tasks.ts | 32 +- cdk/src/handlers/shared/orchestrator.ts | 112 +++--- cdk/src/handlers/shared/task-concurrency.ts | 192 +++++++++++ .../constructs/concurrency-reconciler.test.ts | 17 +- cdk/test/constructs/task-api.test.ts | 23 +- cdk/test/handlers/confirm-uploads.test.ts | 8 +- .../handlers/orchestrate-task-microvm.test.ts | 23 +- cdk/test/handlers/orchestrate-task.test.ts | 125 +++---- .../reconcile-admission-queue.test.ts | 2 +- .../handlers/reconcile-concurrency.test.ts | 295 ++++++---------- .../handlers/reconcile-stranded-tasks.test.ts | 24 +- .../shared/task-concurrency-local.test.ts | 320 ++++++++++++++++++ .../handlers/shared/task-concurrency.test.ts | 153 +++++++++ docs/design/ATTACHMENTS.md | 14 +- docs/design/ORCHESTRATOR.md | 42 ++- .../content/docs/architecture/Attachments.md | 14 +- .../content/docs/architecture/Orchestrator.md | 42 ++- .../verification/645-capacity-reservations.md | 48 +++ 27 files changed, 1208 insertions(+), 591 deletions(-) create mode 100644 cdk/src/handlers/shared/task-concurrency.ts create mode 100644 cdk/test/handlers/shared/task-concurrency-local.test.ts create mode 100644 cdk/test/handlers/shared/task-concurrency.test.ts create mode 100644 docs/verification/645-capacity-reservations.md diff --git a/cdk/src/constructs/concurrency-reconciler.ts b/cdk/src/constructs/concurrency-reconciler.ts index c0fed93dd..b28a9e1b6 100644 --- a/cdk/src/constructs/concurrency-reconciler.ts +++ b/cdk/src/constructs/concurrency-reconciler.ts @@ -41,7 +41,7 @@ const RECONCILER_MEMORY_MB = 256; */ export interface ConcurrencyReconcilerProps { /** - * The DynamoDB task table (has UserStatusIndex GSI). + * The DynamoDB task table (reservations are scanned from the base table). */ readonly taskTable: dynamodb.ITable; @@ -59,8 +59,8 @@ export interface ConcurrencyReconcilerProps { /** * Scheduled Lambda that reconciles user concurrency counters by comparing - * the active_count in the concurrency table against actual active tasks - * in the task table. Corrects drift caused by orchestrator crashes. + * active_count against saved task reservations, including approval waits. + * Releases terminal reservations left behind by interrupted cleanup. */ export class ConcurrencyReconciler extends Construct { public readonly fn: lambda.NodejsFunction; @@ -89,6 +89,7 @@ export class ConcurrencyReconciler extends Construct { }); props.taskTable.grantReadData(this.fn); + props.taskTable.grant(this.fn, 'dynamodb:UpdateItem'); props.userConcurrencyTable.grantReadWriteData(this.fn); const schedule = props.schedule ?? Duration.minutes(DEFAULT_SCHEDULE_MINUTES); diff --git a/cdk/src/constructs/stranded-task-reconciler.ts b/cdk/src/constructs/stranded-task-reconciler.ts index cb3dbdbc6..ac6bbc70b 100644 --- a/cdk/src/constructs/stranded-task-reconciler.ts +++ b/cdk/src/constructs/stranded-task-reconciler.ts @@ -55,7 +55,7 @@ export interface StrandedTaskReconcilerProps { /** TaskEventsTable (handler writes task_stranded + task_failed events). */ readonly taskEventsTable: dynamodb.ITable; - /** UserConcurrencyTable (handler decrements active_count on fail). */ + /** UserConcurrencyTable (handler atomically releases held task reservations). */ readonly userConcurrencyTable: dynamodb.ITable; /** @@ -142,7 +142,7 @@ export class StrandedTaskReconciler extends Construct { props.taskTable.grantReadWriteData(this.fn); // TaskEvents: write task_stranded + task_failed events. props.taskEventsTable.grantWriteData(this.fn); - // Concurrency: decrement active_count on fail. + // Concurrency: read/release a task-owned reservation transactionally. props.userConcurrencyTable.grantReadWriteData(this.fn); const schedule = props.schedule ?? Duration.minutes(DEFAULT_SCHEDULE_MINUTES); diff --git a/cdk/src/constructs/task-api.ts b/cdk/src/constructs/task-api.ts index be735cad2..e52cd1b0a 100644 --- a/cdk/src/constructs/task-api.ts +++ b/cdk/src/constructs/task-api.ts @@ -228,7 +228,7 @@ export interface TaskApiProps { readonly attachmentsBucket?: s3.IBucket; /** - * User concurrency table for admission control during confirm-uploads. + * User concurrency table for the advisory confirm-uploads pre-check. * Required when attachmentsBucket is provided. */ readonly userConcurrencyTable?: dynamodb.ITable; @@ -836,7 +836,7 @@ export class TaskApi extends Construct { props.taskEventsTable.grantReadWriteData(confirmUploadsFn); props.attachmentsBucket.grantReadWrite(confirmUploadsFn); props.attachmentsBucket.grantDelete(confirmUploadsFn); - props.userConcurrencyTable.grantReadWriteData(confirmUploadsFn); + props.userConcurrencyTable.grantReadData(confirmUploadsFn); if (props.orchestratorFunctionArn) { confirmUploadsFn.addToRolePolicy(new iam.PolicyStatement({ diff --git a/cdk/src/constructs/task-status.ts b/cdk/src/constructs/task-status.ts index 3451155c1..cda9ab789 100644 --- a/cdk/src/constructs/task-status.ts +++ b/cdk/src/constructs/task-status.ts @@ -112,7 +112,7 @@ export const VALID_TRANSITIONS: Readonly { /** * Non-mutating read to check if the user is at their concurrency limit. - * Used as a fast pre-check before expensive screening; the actual atomic - * increment happens in checkConcurrency() during transitionToSubmitted. + * Used as an advisory pre-check before expensive screening. The orchestrator + * reserves capacity atomically after submission, queuing if capacity fills. */ async function preCheckConcurrency(userId: string): Promise { try { @@ -745,7 +735,7 @@ async function preCheckConcurrency(userId: string): Promise { return activeCount < MAX_CONCURRENT; } catch (err: any) { // Only swallow DDB throttling errors — these are transient and the atomic - // check in transitionToSubmitted is the authoritative gate. + // orchestrator admission transaction is the authoritative gate. const throttleErrors = ['ProvisionedThroughputExceededException', 'RequestLimitExceeded', 'ThrottlingException']; if (throttleErrors.includes(err?.name)) { logger.warn('Pre-check concurrency throttled — allowing request to proceed', { @@ -765,66 +755,3 @@ async function preCheckConcurrency(userId: string): Promise { throw err; } } - -async function checkConcurrency(userId: string): Promise { - try { - await ddb.send(new UpdateCommand({ - TableName: CONCURRENCY_TABLE_NAME, - Key: { user_id: userId }, - UpdateExpression: 'SET active_count = if_not_exists(active_count, :zero) + :one, updated_at = :now', - ConditionExpression: 'attribute_not_exists(active_count) OR active_count < :max', - ExpressionAttributeValues: { - ':zero': 0, - ':one': 1, - ':max': MAX_CONCURRENT, - ':now': new Date().toISOString(), - }, - })); - return true; - } catch (err: any) { - if (err.name === 'ConditionalCheckFailedException') { - return false; - } - throw err; - } -} - -async function decrementConcurrency(userId: string): Promise { - const maxAttempts = 3; - for (let attempt = 0; attempt < maxAttempts; attempt++) { - try { - await ddb.send(new UpdateCommand({ - TableName: CONCURRENCY_TABLE_NAME, - Key: { user_id: userId }, - UpdateExpression: 'SET active_count = active_count - :one, updated_at = :now', - ConditionExpression: 'attribute_exists(active_count) AND active_count > :zero', - ExpressionAttributeValues: { - ':one': 1, - ':zero': 0, - ':now': new Date().toISOString(), - }, - })); - return; - } catch (err: any) { - if (err.name === 'ConditionalCheckFailedException') { - // Counter already at 0 or doesn't exist — nothing to roll back - return; - } - if (attempt < maxAttempts - 1) { - // Retry transient DDB errors (throttling, network) with backoff - await new Promise(resolve => setTimeout(resolve, 100 * Math.pow(2, attempt))); - continue; - } - logger.error('Failed to decrement concurrency counter after retries (leak possible)', { - user_id: userId, - attempts: maxAttempts, - error: err instanceof Error ? err.message : String(err), - metric_type: 'concurrency_counter_leak', - }); - throw new Error( - `Concurrency counter decrement failed for user ${userId} after ${maxAttempts} attempts. ` + - 'Manual intervention may be required to reset the counter.', - ); - } - } -} diff --git a/cdk/src/handlers/orchestrate-task.ts b/cdk/src/handlers/orchestrate-task.ts index 9ebd50285..a408ac0c6 100644 --- a/cdk/src/handlers/orchestrate-task.ts +++ b/cdk/src/handlers/orchestrate-task.ts @@ -44,6 +44,7 @@ import { runPreflightChecks } from './shared/preflight'; import { isAutoRetried, startSessionWithRetry } from './shared/session-start-retry'; import { deleteEcsPayload } from './shared/strategies/ecs-strategy'; import { deleteMicrovmPayload } from './shared/strategies/lambda-microvm-strategy'; +import { releaseTaskSlot } from './shared/task-concurrency'; import type { TaskRecord } from './shared/types'; import { workflowIsReadOnly, workflowRequiresRepo } from './shared/workflows'; @@ -96,7 +97,7 @@ const durableHandler: DurableExecutionHandler = asyn // up, flips QUEUED -> SUBMITTED, and re-invokes this orchestrator. const admitted = await context.step('admission-control', async () => { // Re-read status to detect external cancellation between steps - const current = await loadTask(taskId); + const current = await loadTask(taskId, true); if (TERMINAL_STATUSES.includes(current.status)) { return false; } @@ -132,13 +133,14 @@ const durableHandler: DurableExecutionHandler = asyn }); if (!admitted) { + await context.step('release-before-admission', () => releaseTaskSlot(taskId, task.user_id)); return; } // Step 2b: Pre-flight checks — verify external dependencies before consuming AgentCore runtime const preflightPassed = await context.step('pre-flight', async () => { try { - const current = await loadTask(taskId); + const current = await loadTask(taskId, true); if (TERMINAL_STATUSES.includes(current.status)) { return false; } @@ -167,6 +169,7 @@ const durableHandler: DurableExecutionHandler = asyn }); if (!preflightPassed) { + await context.step('release-before-work', () => releaseTaskSlot(taskId, task.user_id)); return; } diff --git a/cdk/src/handlers/reconcile-admission-queue.ts b/cdk/src/handlers/reconcile-admission-queue.ts index 0b08c48bf..849ca9785 100644 --- a/cdk/src/handlers/reconcile-admission-queue.ts +++ b/cdk/src/handlers/reconcile-admission-queue.ts @@ -213,7 +213,9 @@ async function requeueAfterInvokeFailure(taskId: string): Promise { TableName: TASK_TABLE, Key: { task_id: { S: taskId } }, UpdateExpression: 'SET #s = :queued, updated_at = :now, status_created_at = :sca', - ConditionExpression: '#s = :submitted', + // A lost invoke response may hide a successful admission. Never put a + // task that now holds capacity back into the no-reservation queue. + ConditionExpression: '#s = :submitted AND attribute_not_exists(concurrency_slot)', ExpressionAttributeNames: { '#s': 'status' }, ExpressionAttributeValues: { ':queued': { S: 'QUEUED' }, diff --git a/cdk/src/handlers/reconcile-concurrency.ts b/cdk/src/handlers/reconcile-concurrency.ts index 008f0e8a2..4e0828aee 100644 --- a/cdk/src/handlers/reconcile-concurrency.ts +++ b/cdk/src/handlers/reconcile-concurrency.ts @@ -17,108 +17,139 @@ * SOFTWARE. */ -import { DynamoDBClient, ScanCommand, QueryCommand, UpdateItemCommand } from '@aws-sdk/client-dynamodb'; +import { randomUUID } from 'node:crypto'; +import { ScanCommand, type ScanCommandOutput, UpdateCommand } from '@aws-sdk/lib-dynamodb'; +import { ACTIVE_STATUSES, TERMINAL_STATUSES } from '../constructs/task-status'; import { logger } from './shared/logger'; -import { makeClient } from './shared/ua'; +import { releaseTaskSlot, type ReservationTask } from './shared/task-concurrency'; +import { makeDocClient } from './shared/ua'; -const ddb = makeClient(DynamoDBClient); +const ddb = makeDocClient(); const TASK_TABLE = process.env.TASK_TABLE_NAME!; -const CONCURRENCY_TABLE = process.env.USER_CONCURRENCY_TABLE_NAME!; +const COUNTER_TABLE = process.env.USER_CONCURRENCY_TABLE_NAME!; -/** - * Count actual active tasks for a user by querying the UserStatusIndex GSI. - */ -async function countActiveTasks(userId: string): Promise { - let count = 0; - let lastKey: Record | undefined; - - do { - const resp = await ddb.send(new QueryCommand({ - TableName: TASK_TABLE, - IndexName: 'UserStatusIndex', - KeyConditionExpression: 'user_id = :uid', - FilterExpression: '#s IN (:s1, :s2, :s3, :s4)', - ExpressionAttributeNames: { '#s': 'status' }, - ExpressionAttributeValues: { - ':uid': { S: userId }, - ':s1': { S: 'SUBMITTED' }, - ':s2': { S: 'HYDRATING' }, - ':s3': { S: 'RUNNING' }, - ':s4': { S: 'FINALIZING' }, - }, - Select: 'COUNT', - ExclusiveStartKey: lastKey, - })); - count += resp.Count ?? 0; - lastKey = resp.LastEvaluatedKey; - } while (lastKey); +interface CounterSnapshot { + readonly user_id: string; + readonly active_count?: number; + readonly reservation_version?: string; +} - return count; +interface Reservations { + held: number; + ambiguous: boolean; + terminal: string[]; } /** - * Scheduled handler: scan the concurrency table and reconcile each user's - * active_count against actual active tasks in the task table. + * Read counters BEFORE scanning reservations. Every reservation mutation also + * changes its counter revision; a conditional repair then detects changes + * anywhere during the scan, including an increment/decrement with no net change. + * Strong base-table reads avoid the GSI lag that can hide a newly held slot. */ export async function handler(): Promise { logger.info('Concurrency reconciler started'); - let corrected = 0; - let scanned = 0; - let errors = 0; + const counters = new Map(); let lastKey: Record | undefined; - do { - const scanResp = await ddb.send(new ScanCommand({ - TableName: CONCURRENCY_TABLE, + const page: ScanCommandOutput = await ddb.send(new ScanCommand({ + TableName: COUNTER_TABLE, + ConsistentRead: true, + ProjectionExpression: 'user_id, active_count, reservation_version', ExclusiveStartKey: lastKey, })); + for (const row of page.Items ?? []) { + if (typeof row.user_id === 'string') counters.set(row.user_id, row as CounterSnapshot); + } + lastKey = page.LastEvaluatedKey; + } while (lastKey); - for (const rawItem of scanResp.Items ?? []) { - const userId = rawItem.user_id?.S; - const storedCount = Number(rawItem.active_count?.N ?? '0'); - if (!userId) continue; - scanned++; - - try { - const actualCount = await countActiveTasks(userId); + const reservations = new Map(); + do { + const page: ScanCommandOutput = await ddb.send(new ScanCommand({ + TableName: TASK_TABLE, + ConsistentRead: true, + ProjectionExpression: 'task_id, user_id, #status, concurrency_slot', + ExpressionAttributeNames: { '#status': 'status' }, + ExclusiveStartKey: lastKey, + })); + for (const row of page.Items ?? []) { + const task = row as ReservationTask; + if (!task.task_id || !task.user_id) continue; + const active = ACTIVE_STATUSES.some(status => status === task.status); + if (!task.concurrency_slot && !active) continue; + const owned = reservations.get(task.user_id) ?? { held: 0, ambiguous: false, terminal: [] }; + if (task.concurrency_slot?.state === 'held') { + // Terminal tasks still own their seat until release commits. Repair + // that total first, then release terminal reservations through the + // same transaction used by the orchestrator and stranded-task cleaner. + owned.held++; + if (TERMINAL_STATUSES.some(status => status === task.status)) owned.terminal.push(task.task_id); + } else if (active || (task.concurrency_slot && task.concurrency_slot.state !== 'released')) { + // Older active tasks and not-yet-admitted SUBMITTED tasks cannot be + // distinguished by status. Wait for them to settle; never guess a count. + owned.ambiguous = true; + } + reservations.set(task.user_id, owned); + } + lastKey = page.LastEvaluatedKey; + } while (lastKey); - if (storedCount !== actualCount) { - logger.info('Drift detected', { userId, storedCount, actualCount }); - try { - await ddb.send(new UpdateItemCommand({ - TableName: CONCURRENCY_TABLE, - Key: { user_id: { S: userId } }, - UpdateExpression: 'SET active_count = :count, updated_at = :now', - ConditionExpression: 'active_count = :stored', - ExpressionAttributeValues: { - ':count': { N: String(actualCount) }, - ':now': { S: new Date().toISOString() }, - ':stored': { N: String(storedCount) }, - }, - })); - corrected++; - } catch (updateErr: unknown) { - if (updateErr && typeof updateErr === 'object' && 'name' in updateErr && updateErr.name === 'ConditionalCheckFailedException') { - logger.info('Concurrent update detected, skipping', { userId }); - } else { - throw updateErr; - } + let corrected = 0; + let errors = 0; + const users = new Set([...counters.keys(), ...reservations.keys()]); + for (const userId of users) { + const snapshot = counters.get(userId); + const owned = reservations.get(userId) ?? { held: 0, ambiguous: false, terminal: [] }; + try { + const stored = snapshot?.active_count ?? 0; + if (!Number.isSafeInteger(stored)) throw new Error('Concurrency counter is not an integer'); + if (owned.ambiguous) { + logger.warn('Skipping capacity repair while task reservation ownership is ambiguous', { + user_id: userId, error_id: 'CONCURRENCY_RESERVATION_UNKNOWN', + }); + } else if (stored !== owned.held) { + const conditions: string[] = []; + const values: Record = { + ':count': owned.held, ':now': new Date().toISOString(), ':version': randomUUID(), + }; + if (!snapshot) { + conditions.push('attribute_not_exists(user_id)'); + } else { + conditions.push('attribute_exists(user_id)'); + if (snapshot.active_count === undefined) {conditions.push('attribute_not_exists(active_count)');} else { + conditions.push('active_count = :stored'); + values[':stored'] = stored; + } + if (snapshot.reservation_version === undefined) {conditions.push('attribute_not_exists(reservation_version)');} else { + conditions.push('reservation_version = :observed'); + values[':observed'] = snapshot.reservation_version; } } - } catch (err: unknown) { - errors++; - logger.warn('Per-user reconciliation failed, continuing', { - userId, - error: err instanceof Error ? err.message : String(err), - }); + try { + await ddb.send(new UpdateCommand({ + TableName: COUNTER_TABLE, + Key: { user_id: userId }, + UpdateExpression: 'SET active_count = :count, updated_at = :now, reservation_version = :version', + ConditionExpression: conditions.join(' AND '), + ExpressionAttributeValues: values, + })); + corrected++; + logger.info('Corrected capacity counter from saved reservations', { + user_id: userId, stored_count: stored, reservation_count: owned.held, + }); + } catch (error) { + if ((error as { name?: string })?.name !== 'ConditionalCheckFailedException') throw error; + logger.info('Capacity changed during reconciliation; skipped stale repair', { user_id: userId }); + } } + for (const taskId of owned.terminal) await releaseTaskSlot(taskId, userId); + } catch (error) { + errors++; + logger.warn('Per-user capacity reconciliation failed, continuing', { user_id: userId, error: String(error) }); } - - lastKey = scanResp.LastEvaluatedKey; - } while (lastKey); - - if (errors === scanned && scanned > 0) { - logger.error('All users failed reconciliation — possible systemic issue', { scanned, errors }); } - logger.info('Concurrency reconciler finished', { scanned, corrected, errors }); + if (errors === users.size && users.size > 0) { + logger.error('All users failed reconciliation — possible systemic issue', { scanned: users.size, errors }); + } + logger.info('Concurrency reconciler finished', { scanned: users.size, corrected, errors }); } diff --git a/cdk/src/handlers/reconcile-stranded-tasks.ts b/cdk/src/handlers/reconcile-stranded-tasks.ts index 4d5fa988f..d3ebc8c5b 100644 --- a/cdk/src/handlers/reconcile-stranded-tasks.ts +++ b/cdk/src/handlers/reconcile-stranded-tasks.ts @@ -48,12 +48,12 @@ import { } from '@aws-sdk/client-dynamodb'; import { ulid } from 'ulid'; import { logger } from './shared/logger'; +import { releaseTaskSlot } from './shared/task-concurrency'; import { makeClient } from './shared/ua'; const ddb = makeClient(DynamoDBClient); const TASK_TABLE = process.env.TASK_TABLE_NAME!; const EVENTS_TABLE = process.env.TASK_EVENTS_TABLE_NAME!; -const CONCURRENCY_TABLE = process.env.USER_CONCURRENCY_TABLE_NAME!; /** Stranded-task timeout. The orchestrator Lambda is async-invoked and * the agent runtime has a cold-start path; 1200 s covers Lambda retries @@ -286,32 +286,10 @@ async function failStrandedTask(task: StrandedCandidate): Promise { } } - // 3. Release the concurrency slot. Best-effort; drift is later corrected - // by the concurrency reconciler. - try { - await ddb.send(new UpdateItemCommand({ - TableName: CONCURRENCY_TABLE, - Key: { user_id: { S: task.user_id } }, - UpdateExpression: 'SET active_count = active_count - :one, updated_at = :now', - ConditionExpression: 'active_count > :zero', - ExpressionAttributeValues: { - ':one': { N: '1' }, - ':zero': { N: '0' }, - ':now': { S: now }, - }, - })); - } catch (decrErr: unknown) { - if (decrErr && typeof decrErr === 'object' && 'name' in decrErr - && decrErr.name !== 'ConditionalCheckFailedException') { - logger.warn('Failed to decrement concurrency for stranded task', { - task_id: task.task_id, - user_id: task.user_id, - error: decrErr instanceof Error ? decrErr.message : String(decrErr), - }); - } - // ConditionalCheckFailedException means the counter is already 0 — - // drift the concurrency reconciler will eventually catch. - } + // 3. Cooperate with normal finalization through the task-owned marker. + // If this fails after the terminal write, the capacity reconciler retries + // release for terminal held reservations on its next sweep. + await releaseTaskSlot(task.task_id, task.user_id); return true; } diff --git a/cdk/src/handlers/shared/orchestrator.ts b/cdk/src/handlers/shared/orchestrator.ts index 197941ba2..c70252771 100644 --- a/cdk/src/handlers/shared/orchestrator.ts +++ b/cdk/src/handlers/shared/orchestrator.ts @@ -32,6 +32,7 @@ import { parseRef } from './registry/ref'; import { RegistryResolutionError, type ResolvedAsset } from './registry/types'; import { loadRepoConfig, type BlueprintConfig, type ComputeType } from './repo-config'; import { resolveUrlAttachments } from './resolve-url-attachments'; +import { acquireTaskSlot, releaseTaskSlot } from './task-concurrency'; import { APPROVAL_GATE_CAP_MAX, APPROVAL_GATE_CAP_MIN, type AgentAttachmentPayload, type AttachmentRecord, type TaskRecord } from './types'; import { makeClient, makeDocClient } from './ua'; import { computeTtlEpoch, DEFAULT_MAX_TURNS } from './validation'; @@ -41,7 +42,6 @@ const ddb = makeDocClient(); const TABLE_NAME = process.env.TASK_TABLE_NAME!; const EVENTS_TABLE_NAME = process.env.TASK_EVENTS_TABLE_NAME!; -const CONCURRENCY_TABLE_NAME = process.env.USER_CONCURRENCY_TABLE_NAME!; const RUNTIME_ARN = process.env.RUNTIME_ARN!; const MAX_CONCURRENT = Number(process.env.MAX_CONCURRENT_TASKS_PER_USER ?? '3'); const TASK_RETENTION_DAYS = Number(process.env.TASK_RETENTION_DAYS ?? '90'); @@ -188,31 +188,12 @@ export async function loadTask(taskId: string, consistentRead = false): Promise< } /** - * Admission control: check user concurrency and increment counter. + * Admission control: acquire or recover this task's capacity reservation. * @param task - the task record. * @returns true if admitted, false if concurrency limit reached. */ export async function admissionControl(task: TaskRecord): Promise { - try { - await ddb.send(new UpdateCommand({ - TableName: CONCURRENCY_TABLE_NAME, - Key: { user_id: task.user_id }, - UpdateExpression: 'SET active_count = if_not_exists(active_count, :zero) + :one, updated_at = :now', - ConditionExpression: 'attribute_not_exists(active_count) OR active_count < :max', - ExpressionAttributeValues: { - ':zero': 0, - ':one': 1, - ':max': MAX_CONCURRENT, - ':now': new Date().toISOString(), - }, - })); - return true; - } catch (err: unknown) { - if (err && typeof err === 'object' && 'name' in err && err.name === 'ConditionalCheckFailedException') { - return false; - } - throw err; - } + return acquireTaskSlot(task.task_id, task.user_id, MAX_CONCURRENT); } /** @@ -267,7 +248,9 @@ export async function transitionTask( TableName: TABLE_NAME, Key: { task_id: taskId }, UpdateExpression: updateExpression, - ConditionExpression: '#status = :fromStatus', + ConditionExpression: fromStatus === TaskStatus.SUBMITTED && toStatus === TaskStatus.QUEUED + ? '#status = :fromStatus AND attribute_not_exists(concurrency_slot)' + : '#status = :fromStatus', ExpressionAttributeNames: expressionNames, ExpressionAttributeValues: expressionValues, })); @@ -1154,6 +1137,16 @@ export async function finalizeTask( pollState: PollState, userId: string, ): Promise { + try { + await finalizeTaskOutcome(taskId, pollState); + } finally { + // The marker makes this safe after a crash, an event failure, or a competing + // cleaner. A still-active task keeps its reservation. + await releaseTaskSlot(taskId, userId); + } +} + +async function finalizeTaskOutcome(taskId: string, pollState: PollState): Promise { // Finalization can immediately follow a committed start failure/cancellation. // A stale active state would emit the wrong terminal event. const task = await loadTask(taskId, true); @@ -1190,7 +1183,7 @@ export async function finalizeTask( transitioned = true; } catch (err) { // Task may have transitioned concurrently (e.g. agent wrote terminal status). - // Re-read to avoid double-decrement or contradictory events. + // Re-read to report the terminal status that actually won. log.warn('Finalization transition to FAILED (heartbeat) failed, task may have transitioned concurrently', { error: err instanceof Error ? err.message : String(err), }); @@ -1200,21 +1193,19 @@ export async function finalizeTask( reason: 'agent_heartbeat_stale', poll_attempts: pollState.attempts, }, correlation); - await decrementConcurrency(userId); } else { // Transition failed — re-read task to determine actual state. // If already terminal the block below will handle TTL + concurrency. - const reread = await loadTask(taskId); + const reread = await loadTask(taskId, true); if (TERMINAL_STATUSES.includes(reread.status)) { log.info('Heartbeat path: task already terminal after failed transition', { status: reread.status }); await emitTaskEvent(taskId, `task_${reread.status.toLowerCase()}`, { final_status: reread.status, poll_attempts: pollState.attempts, }, correlation); - await decrementConcurrency(userId); } else { - log.warn('Heartbeat path: task in unexpected state after failed transition, releasing concurrency', { status: reread.status }); - await decrementConcurrency(userId); + log.warn('Heartbeat path: task in unexpected state after failed transition, keeping its reservation until terminal', { status: reread.status }); + throw new Error(`Heartbeat finalization left task ${taskId} active in ${reread.status}`); } } return; @@ -1281,7 +1272,6 @@ export async function finalizeTask( final_status: currentStatus, poll_attempts: pollState.attempts, }, correlation); - await decrementConcurrency(userId); return; } @@ -1304,8 +1294,14 @@ export async function finalizeTask( : 'Orchestrator poll timeout exceeded', }); } catch (err) { - // Task may have transitioned concurrently — re-read and accept + // Accept a committed terminal winner; otherwise retry finalization. log.warn('Finalization transition failed, task may have transitioned concurrently', { error: err instanceof Error ? err.message : String(err) }); + const current = await loadTask(taskId, true); + if (!TERMINAL_STATUSES.includes(current.status)) throw err; + await emitTaskEvent(taskId, `task_${current.status.toLowerCase()}`, { + final_status: current.status, poll_attempts: pollState.attempts, + }, correlation); + return; } await emitTaskEvent(taskId, 'task_timed_out', { reason: currentStatus === TaskStatus.AWAITING_APPROVAL @@ -1313,7 +1309,6 @@ export async function finalizeTask( : 'poll_timeout', poll_attempts: pollState.attempts, }, correlation); - await decrementConcurrency(userId); return; } @@ -1327,18 +1322,23 @@ export async function finalizeTask( }); } catch (err) { log.warn('Finalization transition from HYDRATING failed, task may have transitioned concurrently', { error: err instanceof Error ? err.message : String(err) }); + const current = await loadTask(taskId, true); + if (!TERMINAL_STATUSES.includes(current.status)) throw err; + await emitTaskEvent(taskId, `task_${current.status.toLowerCase()}`, { + final_status: current.status, poll_attempts: pollState.attempts, + }, correlation); + return; } await emitTaskEvent(taskId, 'task_failed', { reason: 'session_never_started', poll_attempts: pollState.attempts, }, correlation); - await decrementConcurrency(userId); return; } - // Unexpected state — log and release concurrency + // Unexpected active state — retain its reservation until terminal. log.error('Unexpected task state during finalization', { status: currentStatus }); - await decrementConcurrency(userId); + throw new Error(`Cannot finalize task ${taskId} in ${currentStatus}`); } /** @@ -1394,7 +1394,7 @@ export async function queueTask(task: TaskRecord): Promise { * @param fromStatus - the current status. * @param errorMessage - the error reason. * @param userId - the user who owns the task. - * @param releaseConcurrency - whether to decrement the concurrency counter. + * @param releaseConcurrency - whether to release this task's held reservation. * @param repo - optional target repo (`owner/repo`) for the correlation * envelope; omit for repo-less workflows. */ @@ -1418,41 +1418,17 @@ export async function failTask( log.warn('Failed to transition task to FAILED', { error: err instanceof Error ? err.message : String(err), }); + const current = await loadTask(taskId, true); + if (!TERMINAL_STATUSES.includes(current.status)) throw err; } - // Only emit / release concurrency after a successful transition. Callers such as - // orchestrate-task rethrow after failTask; Durable Execution retries the step and - // would otherwise re-run emit + decrement while the task is already FAILED. - if (transitioned) { - await emitTaskEvent(taskId, 'task_failed', { error_message: errorMessage }, { user_id: userId, repo }); - if (releaseConcurrency) { - await decrementConcurrency(userId); - } - } -} - -/** - * Decrement the user's concurrency counter (best-effort). - * @param userId - the user ID. - */ -async function decrementConcurrency(userId: string): Promise { + // A replay may find the FAILED write already committed. Events remain + // transition-owned; reservation release is independently safe to repeat. try { - await ddb.send(new UpdateCommand({ - TableName: CONCURRENCY_TABLE_NAME, - Key: { user_id: userId }, - UpdateExpression: 'SET active_count = active_count - :one, updated_at = :now', - ConditionExpression: 'active_count > :zero', - ExpressionAttributeValues: { - ':one': 1, - ':zero': 0, - ':now': new Date().toISOString(), - }, - })); - } catch (err: unknown) { - if (err && typeof err === 'object' && 'name' in err && err.name === 'ConditionalCheckFailedException') { - logger.info('Concurrency counter already at zero, nothing to decrement', { user_id: userId }); - } else { - logger.warn('Failed to decrement concurrency counter', { user_id: userId, error: err instanceof Error ? err.message : String(err) }); + if (transitioned) { + await emitTaskEvent(taskId, 'task_failed', { error_message: errorMessage }, { user_id: userId, repo }); } + } finally { + if (releaseConcurrency) await releaseTaskSlot(taskId, userId); } } diff --git a/cdk/src/handlers/shared/task-concurrency.ts b/cdk/src/handlers/shared/task-concurrency.ts new file mode 100644 index 000000000..d6f979d20 --- /dev/null +++ b/cdk/src/handlers/shared/task-concurrency.ts @@ -0,0 +1,192 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import { randomUUID } from 'node:crypto'; +import { GetCommand, TransactWriteCommand } from '@aws-sdk/lib-dynamodb'; +import { logger } from './logger'; +import { makeDocClient } from './ua'; +import { ACTIVE_STATUSES, TaskStatus, TERMINAL_STATUSES } from '../../constructs/task-status'; + +/** Internal TaskTable data; a released task cannot acquire another reservation. */ +export interface ReservationTask { + readonly task_id: string; + readonly user_id: string; + readonly status: string; + readonly concurrency_slot?: { + readonly state: 'held' | 'released'; + readonly acquired_at: string; + readonly released_at?: string; + }; +} + +const ddb = makeDocClient(); +const TASK_TABLE = process.env.TASK_TABLE_NAME!; +const COUNTER_TABLE = process.env.USER_CONCURRENCY_TABLE_NAME!; + +function terminal(status: string): boolean { + return TERMINAL_STATUSES.some(value => value === status); +} + +async function readTask(taskId: string, userId: string): Promise { + const result = await ddb.send(new GetCommand({ + TableName: TASK_TABLE, Key: { task_id: taskId }, ConsistentRead: true, + })); + const task = result.Item as ReservationTask | undefined; + if (task && task.user_id !== userId) throw new Error('Concurrency reservation owner does not match task owner'); + return task; +} + +function conditionalFailure(error: unknown, index: number): boolean { + const failure = error as { name?: string; CancellationReasons?: { Code?: string }[] }; + return failure?.name === 'TransactionCanceledException' + && failure.CancellationReasons?.[index]?.Code === 'ConditionalCheckFailed'; +} + +/** Reserve once per task, including after a lost transaction acknowledgement. */ +export async function acquireTaskSlot(taskId: string, userId: string, limit: number): Promise { + const current = await readTask(taskId, userId); + if (!current) throw new Error(`Cannot reserve capacity for missing task ${taskId}`); + if (current.concurrency_slot?.state === 'held') { + return ACTIVE_STATUSES.some(status => status === current.status); + } + if (current.status !== TaskStatus.SUBMITTED || current.concurrency_slot) return false; + + const now = new Date().toISOString(); + const revision = randomUUID(); + try { + await ddb.send(new TransactWriteCommand({ + ClientRequestToken: revision, + TransactItems: [ + { + Update: { + TableName: TASK_TABLE, + Key: { task_id: taskId }, + UpdateExpression: 'SET concurrency_slot = :slot', + ConditionExpression: 'user_id = :user AND #status = :submitted AND attribute_not_exists(concurrency_slot)', + ExpressionAttributeNames: { '#status': 'status' }, + ExpressionAttributeValues: { + ':slot': { state: 'held', acquired_at: now }, + ':user': userId, + ':submitted': TaskStatus.SUBMITTED, + }, + }, + }, + { + Update: { + TableName: COUNTER_TABLE, + Key: { user_id: userId }, + UpdateExpression: 'SET active_count = if_not_exists(active_count, :zero) + :one, ' + + 'updated_at = :now, reservation_version = :version', + ConditionExpression: 'attribute_not_exists(active_count) OR active_count < :max', + ExpressionAttributeValues: { + ':zero': 0, ':one': 1, ':max': limit, ':now': now, ':version': revision, + }, + }, + }, + ], + })); + return true; + } catch (error) { + // A competing invocation or a lost successful response may have reserved it. + const latest = await readTask(taskId, userId); + if (latest?.concurrency_slot?.state === 'held') { + return ACTIVE_STATUSES.some(status => status === latest.status); + } + if (!latest || latest.status !== TaskStatus.SUBMITTED || latest.concurrency_slot) return false; + if (conditionalFailure(error, 1)) return false; // Capacity was full. + throw error; // Throttling/conflict/outage is not evidence that capacity is full. + } +} + +/** + * Return a terminal task's reservation atomically with its counter update. + * Never infer ownership from status alone: legacy/unadmitted tasks have no + * marker and cannot return another task's seat. Reconciliation handles drift. + */ +export async function releaseTaskSlot(taskId: string, userId: string): Promise { + const current = await readTask(taskId, userId); + if (!current || !terminal(current.status) || current.concurrency_slot?.state !== 'held') return false; + + let emptyCounter = false; + for (let attempt = 0; attempt < 2; attempt++) { + const now = new Date().toISOString(); + const revision = randomUUID(); + try { + await ddb.send(new TransactWriteCommand({ + ClientRequestToken: revision, + TransactItems: [ + { + Update: { + TableName: TASK_TABLE, + Key: { task_id: taskId }, + UpdateExpression: 'SET concurrency_slot.#state = :released, concurrency_slot.released_at = :now', + ConditionExpression: 'user_id = :user AND concurrency_slot.#state = :held ' + + 'AND #status IN (:completed, :failed, :cancelled, :timedOut)', + ExpressionAttributeNames: { '#state': 'state', '#status': 'status' }, + ExpressionAttributeValues: { + ':user': userId, + ':held': 'held', + ':released': 'released', + ':now': now, + ':completed': TaskStatus.COMPLETED, + ':failed': TaskStatus.FAILED, + ':cancelled': TaskStatus.CANCELLED, + ':timedOut': TaskStatus.TIMED_OUT, + }, + }, + }, + { + Update: { + TableName: COUNTER_TABLE, + Key: { user_id: userId }, + UpdateExpression: emptyCounter + ? 'SET active_count = if_not_exists(active_count, :zero), updated_at = :now, reservation_version = :version' + : 'SET active_count = active_count - :one, updated_at = :now, reservation_version = :version', + ConditionExpression: emptyCounter + ? 'attribute_not_exists(active_count) OR active_count >= :zero' + : 'active_count > :zero', + ExpressionAttributeValues: { + ':zero': 0, ...(!emptyCounter && { ':one': 1 }), ':now': now, ':version': revision, + }, + }, + }, + ], + })); + if (emptyCounter) { + logger.warn('Released task reservation whose counter was already empty', { + task_id: taskId, user_id: userId, error_id: 'CONCURRENCY_EMPTY_COUNTER', + }); + } + return true; + } catch (error) { + const latest = await readTask(taskId, userId); + if (latest?.concurrency_slot?.state === 'released') return false; + if (!latest || !terminal(latest.status) || latest.concurrency_slot?.state !== 'held') return false; + if (!emptyCounter && conditionalFailure(error, 1)) { + // This attempt observed no positive count. Close its marker without + // subtracting from seats reserved since then. A concurrent repair may + // leave an overcount, which a later revision-guarded sweep can correct. + emptyCounter = true; + continue; + } + throw error; + } + } + throw new Error(`Could not release capacity reservation for task ${taskId}`); +} diff --git a/cdk/test/constructs/concurrency-reconciler.test.ts b/cdk/test/constructs/concurrency-reconciler.test.ts index 34e1e88e4..b8e7403db 100644 --- a/cdk/test/constructs/concurrency-reconciler.test.ts +++ b/cdk/test/constructs/concurrency-reconciler.test.ts @@ -43,8 +43,10 @@ function createStack(): Template { } describe('ConcurrencyReconciler construct', () => { + let template: Template; + beforeAll(() => { template = createStack(); }); + test('creates a Lambda function', () => { - const template = createStack(); template.hasResourceProperties('AWS::Lambda::Function', { Runtime: 'nodejs24.x', Timeout: 300, @@ -52,14 +54,12 @@ describe('ConcurrencyReconciler construct', () => { }); test('creates an EventBridge rule with rate schedule', () => { - const template = createStack(); template.hasResourceProperties('AWS::Events::Rule', { ScheduleExpression: 'rate(15 minutes)', }); }); test('Lambda has correct environment variables', () => { - const template = createStack(); template.hasResourceProperties('AWS::Lambda::Function', { Environment: { Variables: Match.objectLike({ @@ -69,4 +69,15 @@ describe('ConcurrencyReconciler construct', () => { }, }); }); + + test('can read reservations and update their markers without deleting or creating tasks', () => { + const tableId = Object.keys(template.findResources('AWS::DynamoDB::Table')).find(id => id.startsWith('TaskTable'))!; + const statements = Object.values(template.findResources('AWS::IAM::Policy')) + .flatMap(policy => policy.Properties.PolicyDocument.Statement) + .filter(statement => JSON.stringify(statement.Resource).includes(tableId)); + const actions = statements.flatMap(statement => [statement.Action].flat()); + expect(actions).toEqual(expect.arrayContaining(['dynamodb:GetItem', 'dynamodb:Scan', 'dynamodb:UpdateItem'])); + expect(actions).not.toEqual(expect.arrayContaining(['dynamodb:PutItem'])); + expect(actions).not.toEqual(expect.arrayContaining(['dynamodb:DeleteItem'])); + }); }); diff --git a/cdk/test/constructs/task-api.test.ts b/cdk/test/constructs/task-api.test.ts index 4cb03b1ce..48737bd25 100644 --- a/cdk/test/constructs/task-api.test.ts +++ b/cdk/test/constructs/task-api.test.ts @@ -23,7 +23,7 @@ import * as dynamodb from 'aws-cdk-lib/aws-dynamodb'; import * as s3 from 'aws-cdk-lib/aws-s3'; import { TaskApi, type TaskApiProps } from '../../src/constructs/task-api'; -function createStack(overrides?: Partial): { stack: Stack; template: Template } { +function createStack(overrides?: Partial, withUploads = false): { stack: Stack; template: Template } { const app = new App(); const stack = new Stack(app, 'TestStack'); @@ -39,6 +39,12 @@ function createStack(overrides?: Partial): { stack: Stack; templat new TaskApi(stack, 'TaskApi', { taskTable, taskEventsTable, + ...(withUploads && { + attachmentsBucket: new s3.Bucket(stack, 'UploadsBucket'), + userConcurrencyTable: new dynamodb.Table(stack, 'CapacityTable', { + partitionKey: { name: 'user_id', type: dynamodb.AttributeType.STRING }, + }), + }), ...overrides, }); @@ -131,11 +137,26 @@ describe('TaskApi construct', () => { let baseTemplate: Template; let budgetTemplate: Template; let webhookTemplate: Template; + let uploadsTemplate: Template; beforeAll(() => { baseTemplate = createStack().template; budgetTemplate = createStackWithBudget().template; webhookTemplate = createStackWithWebhooks().template; + uploadsTemplate = createStack(undefined, true).template; + }); + + test('upload confirmation can inspect capacity but cannot reserve or return a seat', () => { + const policy = Object.entries(uploadsTemplate.findResources('AWS::IAM::Policy')) + .find(([id]) => id.includes('ConfirmUploadsFn'))![1]; + const tableId = Object.keys(uploadsTemplate.findResources('AWS::DynamoDB::Table')).find(id => id.startsWith('CapacityTable'))!; + const actions = policy.Properties.PolicyDocument.Statement + .filter((statement: any) => JSON.stringify(statement.Resource).includes(tableId)) + .flatMap((statement: any) => [statement.Action].flat()); + expect(actions).toContain('dynamodb:GetItem'); + expect(actions).not.toContain('dynamodb:UpdateItem'); + expect(actions).not.toContain('dynamodb:PutItem'); + expect(actions).not.toContain('dynamodb:DeleteItem'); }); test('creates a Cognito User Pool', () => { diff --git a/cdk/test/handlers/confirm-uploads.test.ts b/cdk/test/handlers/confirm-uploads.test.ts index dfa211fc6..83b37c9ef 100644 --- a/cdk/test/handlers/confirm-uploads.test.ts +++ b/cdk/test/handlers/confirm-uploads.test.ts @@ -226,9 +226,8 @@ describe('confirm-uploads handler', () => { switch (ddbCallCount) { case 1: return Promise.resolve({ Item: PENDING_TASK }); // GetCommand (task) case 2: return Promise.resolve({ Item: { active_count: 1 } }); // GetCommand (concurrency pre-check) - case 3: return Promise.resolve({}); // UpdateCommand (checkConcurrency) - case 4: return Promise.resolve({}); // UpdateCommand (status transition) - case 5: return Promise.resolve({}); // PutCommand (event) + case 3: return Promise.resolve({}); // UpdateCommand (status transition) + case 4: return Promise.resolve({}); // PutCommand (event) default: return Promise.resolve({}); } }); @@ -265,6 +264,9 @@ describe('confirm-uploads handler', () => { const body = JSON.parse(result.body); expect(body.data.status).toBe('SUBMITTED'); expect(lambdaSend).toHaveBeenCalled(); + const capacityCalls = ddbSend.mock.calls.filter(([command]) => command.input.TableName === 'Concurrency'); + expect(capacityCalls).toHaveLength(1); + expect(capacityCalls[0][0]._type).toBe('Get'); }); test('returns 429 when concurrency pre-check fails', async () => { diff --git a/cdk/test/handlers/orchestrate-task-microvm.test.ts b/cdk/test/handlers/orchestrate-task-microvm.test.ts index 7c55f89d1..21a5b6b38 100644 --- a/cdk/test/handlers/orchestrate-task-microvm.test.ts +++ b/cdk/test/handlers/orchestrate-task-microvm.test.ts @@ -64,6 +64,12 @@ jest.mock('@aws-sdk/client-s3', () => ({ DeleteObjectCommand: jest.fn((input: unknown) => ({ _type: 'DeleteObject', input })), })); +const mockReleaseTaskSlot = jest.fn().mockResolvedValue(false); +jest.mock('../../src/handlers/shared/task-concurrency', () => ({ + acquireTaskSlot: jest.fn().mockResolvedValue(true), + releaseTaskSlot: (...args: unknown[]) => mockReleaseTaskSlot(...args), +})); + // Real orchestrator helpers would talk to DynamoDB; stub them and assert on the // calls. `buildComputeMetadata` is kept REAL (re-exported from the actual module) // so the persisted metadata shape is genuinely exercised, not mirrored. @@ -251,6 +257,19 @@ beforeEach(() => { }); describe('orchestrate-task for a lambda-microvm task', () => { + test.each([2, 3])('cancellation on task read %s checks release before leaving the pipeline', async (cancelOnRead) => { + let reads = 0; + mockLoadTask.mockImplementation(async () => ({ + task_id: 'TASK001', + user_id: 'user-1', + repo: 'org/repo', + status: ++reads >= cancelOnRead ? TaskStatus.CANCELLED : TaskStatus.SUBMITTED, + })); + await handler({ task_id: 'TASK001' }, fakeContext().ctx as never); + expect(commandsOfType('RunMicrovm')).toHaveLength(0); + expect(mockReleaseTaskSlot).toHaveBeenCalledWith('TASK001', 'user-1'); + }); + test('replaying a persisted start failure reaches finalization once without an earlier slot release', async () => { let failed = false; mockMicrovmSend.mockRejectedValue(Object.assign(new Error('denied'), { name: 'AccessDeniedException' })); @@ -333,7 +352,7 @@ describe('orchestrate-task for a lambda-microvm task', () => { task_id: 'TASK001', user_id: 'user-1', repo: 'org/repo', - status: consistentRead ? TaskStatus.CANCELLED : TaskStatus.SUBMITTED, + status: consistentRead && commandsOfType('RunMicrovm').length > 0 ? TaskStatus.CANCELLED : TaskStatus.SUBMITTED, })); const { ctx, steps } = fakeContext(); await handler({ task_id: 'TASK001' }, ctx as never); @@ -387,7 +406,7 @@ describe('orchestrate-task for a lambda-microvm task', () => { task_id: 'TASK001', user_id: 'user-1', repo: 'org/repo', - status: consistentRead ? TaskStatus.CANCELLED : TaskStatus.SUBMITTED, + status: consistentRead && commandsOfType('RunMicrovm').length > 0 ? TaskStatus.CANCELLED : TaskStatus.SUBMITTED, })); await handler({ task_id: 'TASK001' }, fakeContext().ctx as never); expect(commandsOfType('RunMicrovm')).toHaveLength(1); diff --git a/cdk/test/handlers/orchestrate-task.test.ts b/cdk/test/handlers/orchestrate-task.test.ts index 808bc423e..e2f934f47 100644 --- a/cdk/test/handlers/orchestrate-task.test.ts +++ b/cdk/test/handlers/orchestrate-task.test.ts @@ -18,6 +18,12 @@ */ // --- Mocks --- +const mockAcquireTaskSlot = jest.fn(); +const mockReleaseTaskSlot = jest.fn(); +jest.mock('../../src/handlers/shared/task-concurrency', () => ({ + acquireTaskSlot: (...args: unknown[]) => mockAcquireTaskSlot(...args), + releaseTaskSlot: (...args: unknown[]) => mockReleaseTaskSlot(...args), +})); const mockDdbSend = jest.fn(); jest.mock('@aws-sdk/client-dynamodb', () => ({ DynamoDBClient: jest.fn(() => ({})) })); jest.mock('@aws-sdk/lib-dynamodb', () => ({ @@ -106,6 +112,9 @@ const baseTask = { beforeEach(() => { jest.clearAllMocks(); + mockDdbSend.mockReset(); + mockAcquireTaskSlot.mockReset().mockResolvedValue(true); + mockReleaseTaskSlot.mockReset().mockResolvedValue(true); ulidCounter = 0; mockLoadRepoConfig.mockResolvedValue(null); }); @@ -124,22 +133,16 @@ describe('loadTask', () => { }); describe('admissionControl', () => { - test('returns true when concurrency slot is available', async () => { - mockDdbSend.mockResolvedValueOnce({}); - const result = await admissionControl(baseTask as any); - expect(result).toBe(true); + test('reserves using the task identity and configured user limit', async () => { + expect(await admissionControl(baseTask as any)).toBe(true); + expect(mockAcquireTaskSlot).toHaveBeenCalledWith('TASK001', 'user-123', 3); }); - - test('returns false when concurrency limit reached', async () => { - const condErr = new Error('Conditional check failed'); - condErr.name = 'ConditionalCheckFailedException'; - mockDdbSend.mockRejectedValueOnce(condErr); - const result = await admissionControl(baseTask as any); - expect(result).toBe(false); + test('returns false when no reservation was acquired', async () => { + mockAcquireTaskSlot.mockResolvedValue(false); + expect(await admissionControl(baseTask as any)).toBe(false); }); - - test('throws on unexpected DDB errors', async () => { - mockDdbSend.mockRejectedValueOnce(new Error('DynamoDB error')); + test('propagates an outage so durable execution can retry', async () => { + mockAcquireTaskSlot.mockRejectedValue(new Error('DynamoDB error')); await expect(admissionControl(baseTask as any)).rejects.toThrow('DynamoDB error'); }); }); @@ -1258,7 +1261,7 @@ describe('finalizeTask', () => { test('handles already-terminal task', async () => { mockDdbSend .mockResolvedValueOnce({ Item: { ...baseTask, status: 'COMPLETED' } }) // loadTask - .mockResolvedValue({}); // emitTaskEvent + decrementConcurrency + .mockResolvedValue({}); // emitTaskEvent await finalizeTask('TASK001', { attempts: 10, lastStatus: 'COMPLETED' }, 'user-123'); // Verify emitTaskEvent was called (PutCommand) expect(mockDdbSend).toHaveBeenCalled(); @@ -1267,7 +1270,7 @@ describe('finalizeTask', () => { test('transitions RUNNING to FAILED when pollState.sessionUnhealthy', async () => { mockDdbSend .mockResolvedValueOnce({ Item: { ...baseTask, status: 'RUNNING' } }) // loadTask - .mockResolvedValue({}); // transitionTask + emitTaskEvent + decrementConcurrency + .mockResolvedValue({}); // transitionTask + emitTaskEvent await finalizeTask( 'TASK001', { attempts: 12, lastStatus: 'RUNNING', sessionUnhealthy: true }, @@ -1359,7 +1362,7 @@ describe('finalizeTask', () => { test('transitions RUNNING to TIMED_OUT on poll timeout', async () => { mockDdbSend .mockResolvedValueOnce({ Item: { ...baseTask, status: 'RUNNING' } }) // loadTask - .mockResolvedValue({}); // transitionTask + emitTaskEvent + decrementConcurrency + .mockResolvedValue({}); // transitionTask + emitTaskEvent await finalizeTask('TASK001', { attempts: 1020, lastStatus: 'RUNNING' }, 'user-123'); expect(mockDdbSend).toHaveBeenCalled(); }); @@ -1367,7 +1370,7 @@ describe('finalizeTask', () => { test('transitions HYDRATING to FAILED when session never started', async () => { mockDdbSend .mockResolvedValueOnce({ Item: { ...baseTask, status: 'HYDRATING' } }) // loadTask - .mockResolvedValue({}); // transitionTask + emitTaskEvent + decrementConcurrency + .mockResolvedValue({}); // transitionTask + emitTaskEvent await finalizeTask('TASK001', { attempts: 15, lastStatus: 'HYDRATING' }, 'user-123'); // First call: loadTask, second call: transitionTask (HYDRATING -> FAILED) const transitionCall = mockDdbSend.mock.calls[1][0]; @@ -1379,34 +1382,32 @@ describe('finalizeTask', () => { expect(eventCall.input.Item.metadata.reason).toBe('session_never_started'); }); - test('releases concurrency for unexpected task state', async () => { - mockDdbSend - .mockResolvedValueOnce({ Item: { ...baseTask, status: 'SUBMITTED' } }) // loadTask - .mockResolvedValue({}); // decrementConcurrency - await finalizeTask('TASK001', { attempts: 5, lastStatus: 'SUBMITTED' }, 'user-123'); - // Should still call decrementConcurrency (UpdateCommand for user concurrency) - const lastCall = mockDdbSend.mock.calls[mockDdbSend.mock.calls.length - 1][0]; - expect(lastCall.input.Key).toEqual({ user_id: 'user-123' }); + test('an unexpected active state makes finalization retry instead of reporting success', async () => { + mockDdbSend.mockResolvedValueOnce({ Item: { ...baseTask, status: 'SUBMITTED' } }); + mockReleaseTaskSlot.mockResolvedValue(false); + await expect(finalizeTask('TASK001', { attempts: 5 }, 'user-123')).rejects.toThrow('Cannot finalize'); + expect(mockDdbSend.mock.calls.some(([command]) => command._type === 'Put')).toBe(false); }); - test('resolves without throwing when decrementConcurrency hits ConditionalCheckFailedException', async () => { - const condErr = new Error('Conditional check failed'); - condErr.name = 'ConditionalCheckFailedException'; - mockDdbSend - .mockResolvedValueOnce({ Item: { ...baseTask, status: 'COMPLETED', memory_written: true } }) // loadTask - .mockResolvedValueOnce({}) // TTL stamp - .mockResolvedValueOnce({}) // emitTaskEvent - .mockRejectedValueOnce(condErr); // decrementConcurrency CCF - await expect(finalizeTask('TASK001', { attempts: 10, lastStatus: 'COMPLETED' }, 'user-123')).resolves.toBeUndefined(); + test('already-released reservations allow finalization to finish', async () => { + mockDdbSend.mockResolvedValueOnce({ Item: { ...baseTask, status: 'COMPLETED', memory_written: true } }) + .mockResolvedValue({}); + mockReleaseTaskSlot.mockResolvedValue(false); + await expect(finalizeTask('TASK001', { attempts: 10 }, 'user-123')).resolves.toBeUndefined(); }); - test('resolves without throwing when decrementConcurrency hits a non-CCF error', async () => { - mockDdbSend - .mockResolvedValueOnce({ Item: { ...baseTask, status: 'COMPLETED', memory_written: true } }) // loadTask - .mockResolvedValueOnce({}) // TTL stamp - .mockResolvedValueOnce({}) // emitTaskEvent - .mockRejectedValueOnce(new Error('DDB timeout')); // decrementConcurrency non-CCF - await expect(finalizeTask('TASK001', { attempts: 10, lastStatus: 'COMPLETED' }, 'user-123')).resolves.toBeUndefined(); + test('a reservation release outage propagates for retry', async () => { + mockDdbSend.mockResolvedValueOnce({ Item: { ...baseTask, status: 'COMPLETED', memory_written: true } }) + .mockResolvedValue({}); + mockReleaseTaskSlot.mockRejectedValue(new Error('DDB timeout')); + await expect(finalizeTask('TASK001', { attempts: 10 }, 'user-123')).rejects.toThrow('DDB timeout'); + }); + + test('a terminal-event failure still attempts reservation release', async () => { + mockDdbSend.mockResolvedValueOnce({ Item: { ...baseTask, status: 'COMPLETED', memory_written: true } }) + .mockResolvedValueOnce({}).mockRejectedValueOnce(new Error('event unavailable')); + await expect(finalizeTask('TASK001', { attempts: 10 }, 'user-123')).rejects.toThrow('event unavailable'); + expect(mockReleaseTaskSlot).toHaveBeenCalledWith('TASK001', 'user-123'); }); }); @@ -1420,8 +1421,7 @@ describe('failTask', () => { test('releases concurrency when requested', async () => { mockDdbSend.mockResolvedValue({}); await failTask('TASK001', 'HYDRATING', 'hydration error', 'user-123', true); - // transitionTask + emitTaskEvent + decrementConcurrency = 3 calls - expect(mockDdbSend).toHaveBeenCalledTimes(3); + expect(mockReleaseTaskSlot).toHaveBeenCalledWith('TASK001', 'user-123'); }); test('transitions from HYDRATING to FAILED when called with HYDRATING status', async () => { @@ -1433,24 +1433,29 @@ describe('failTask', () => { expect(transitionCall.input.ExpressionAttributeValues[':toStatus']).toBe('FAILED'); }); - test('handles transition failure gracefully without emitting when not transitioned', async () => { - mockDdbSend.mockRejectedValueOnce(new Error('Condition failed')); // transitionTask only + test('a committed terminal winner is accepted without another failure event', async () => { + mockDdbSend.mockRejectedValueOnce(new Error('Condition failed')) + .mockResolvedValueOnce({ Item: { ...baseTask, status: 'CANCELLED' } }); await expect(failTask('TASK001', 'SUBMITTED', 'error', 'user-123', false)).resolves.toBeUndefined(); - expect(mockDdbSend).toHaveBeenCalledTimes(1); + expect(mockDdbSend.mock.calls.some(([command]) => command._type === 'Put')).toBe(false); }); - test('second failTask does not re-emit or re-decrement when transition fails (idempotent under step retry)', async () => { + test('failure to persist a terminal state propagates without releasing an active task', async () => { + mockDdbSend.mockRejectedValueOnce(new Error('write unavailable')) + .mockResolvedValueOnce({ Item: { ...baseTask, status: 'HYDRATING' } }); + await expect(failTask('TASK001', 'HYDRATING', 'error', 'user-123', true)).rejects.toThrow('write unavailable'); + expect(mockReleaseTaskSlot).not.toHaveBeenCalled(); + }); + + test('failure replay rechecks release without duplicating the failure event', async () => { mockDdbSend.mockResolvedValue({}); await failTask('TASK001', 'HYDRATING', 'first failure', 'user-123', true); - expect(mockDdbSend).toHaveBeenCalledTimes(3); // transition + emit + decrement - mockDdbSend.mockClear(); - const condErr = new Error('The conditional request failed'); - condErr.name = 'ConditionalCheckFailedException'; - mockDdbSend.mockRejectedValueOnce(condErr); // already FAILED — transition no-ops - - await expect(failTask('TASK001', 'HYDRATING', 'durable replay', 'user-123', true)).resolves.toBeUndefined(); - expect(mockDdbSend).toHaveBeenCalledTimes(1); // transition attempt only; no Put, no concurrency Update + mockDdbSend.mockRejectedValueOnce(new Error('already FAILED')) + .mockResolvedValueOnce({ Item: { ...baseTask, status: 'FAILED' } }); + await expect(failTask('TASK001', 'HYDRATING', 'replay', 'user-123', true)).resolves.toBeUndefined(); + expect(mockDdbSend.mock.calls.some(([command]) => command._type === 'Put')).toBe(false); + expect(mockReleaseTaskSlot).toHaveBeenCalledTimes(2); }); }); @@ -1548,7 +1553,7 @@ describe('finalizeTask TTL stamping', () => { test('stamps TTL on task already in terminal state', async () => { mockDdbSend .mockResolvedValueOnce({ Item: { ...baseTask, status: 'COMPLETED' } }) // loadTask - .mockResolvedValue({}); // UpdateCommand (TTL stamp) + emitTaskEvent + decrementConcurrency + .mockResolvedValue({}); // UpdateCommand (TTL stamp) + emitTaskEvent await finalizeTask('TASK001', { attempts: 10, lastStatus: 'COMPLETED' }, 'user-123'); // Second call should be the TTL stamp UpdateCommand const ttlStampCall = mockDdbSend.mock.calls[1][0]; @@ -1561,7 +1566,7 @@ describe('finalizeTask TTL stamping', () => { mockDdbSend .mockResolvedValueOnce({ Item: { ...baseTask, status: 'COMPLETED' } }) // loadTask .mockRejectedValueOnce(new Error('DDB error')) // TTL stamp fails - .mockResolvedValue({}); // emitTaskEvent + decrementConcurrency + .mockResolvedValue({}); // emitTaskEvent // Should not throw await finalizeTask('TASK001', { attempts: 10, lastStatus: 'COMPLETED' }, 'user-123'); expect(mockDdbSend).toHaveBeenCalled(); @@ -1660,9 +1665,7 @@ describe('finalizeTask — memory fallback', () => { // Should not throw — the try-catch in finalizeTask prevents crash await finalizeTask('TASK001', { attempts: 10, lastStatus: 'COMPLETED' }, 'user-123'); expect(mockWriteMinimalEpisode).toHaveBeenCalled(); - // decrementConcurrency should still be called (last UpdateCommand for user concurrency) - const lastCall = mockDdbSend.mock.calls[mockDdbSend.mock.calls.length - 1][0]; - expect(lastCall.input.Key).toEqual({ user_id: 'user-123' }); + expect(mockReleaseTaskSlot).toHaveBeenCalledWith('TASK001', 'user-123'); }); test('converts string duration_s and cost_usd to numbers', async () => { diff --git a/cdk/test/handlers/reconcile-admission-queue.test.ts b/cdk/test/handlers/reconcile-admission-queue.test.ts index 6d3983fe3..38f4894de 100644 --- a/cdk/test/handlers/reconcile-admission-queue.test.ts +++ b/cdk/test/handlers/reconcile-admission-queue.test.ts @@ -279,7 +279,7 @@ describe('reconcile-admission-queue — races and failures', () => { expect(updates[0].input.ConditionExpression).toBe('#s = :queued'); expect(updates[0].input.ExpressionAttributeValues[':submitted']).toEqual({ S: 'SUBMITTED' }); // Restore: guarded on SUBMITTED, sets QUEUED with a fresh QUEUED# status key. - expect(updates[1].input.ConditionExpression).toBe('#s = :submitted'); + expect(updates[1].input.ConditionExpression).toBe('#s = :submitted AND attribute_not_exists(concurrency_slot)'); expect(updates[1].input.ExpressionAttributeValues[':queued']).toEqual({ S: 'QUEUED' }); expect(updates[1].input.ExpressionAttributeValues[':sca'].S).toMatch(/^QUEUED#/); }); diff --git a/cdk/test/handlers/reconcile-concurrency.test.ts b/cdk/test/handlers/reconcile-concurrency.test.ts index d93e1bf9d..ae58dc687 100644 --- a/cdk/test/handlers/reconcile-concurrency.test.ts +++ b/cdk/test/handlers/reconcile-concurrency.test.ts @@ -17,214 +17,127 @@ * SOFTWARE. */ -// --- Mocks --- -const mockDdbSend = jest.fn(); -jest.mock('@aws-sdk/client-dynamodb', () => ({ - DynamoDBClient: jest.fn(() => ({ send: mockDdbSend })), - ScanCommand: jest.fn((input: unknown) => ({ _type: 'Scan', input })), - QueryCommand: jest.fn((input: unknown) => ({ _type: 'Query', input })), - UpdateItemCommand: jest.fn((input: unknown) => ({ _type: 'UpdateItem', input })), +const mockSend = jest.fn(); +const mockRelease = jest.fn(); +jest.mock('@aws-sdk/lib-dynamodb', () => ({ + ScanCommand: jest.fn((input: unknown) => ({ kind: 'scan', input })), + UpdateCommand: jest.fn((input: unknown) => ({ kind: 'update', input })), +})); +jest.mock('../../src/handlers/shared/ua', () => ({ makeDocClient: () => ({ send: mockSend }) })); +jest.mock('../../src/handlers/shared/task-concurrency', () => ({ + releaseTaskSlot: (...args: unknown[]) => mockRelease(...args), })); - -// Set env vars before importing process.env.TASK_TABLE_NAME = 'Tasks'; -process.env.USER_CONCURRENCY_TABLE_NAME = 'UserConcurrency'; - +process.env.USER_CONCURRENCY_TABLE_NAME = 'Counters'; import { handler } from '../../src/handlers/reconcile-concurrency'; +function held(id: string, user = 'user', status = 'RUNNING') { + return { task_id: id, user_id: user, status, concurrency_slot: { state: 'held' } }; +} +function seed(counters: object[], tasks: object[]) { + mockSend.mockResolvedValueOnce({ Items: counters }).mockResolvedValueOnce({ Items: tasks }).mockResolvedValue({}); +} +function updates() { + return mockSend.mock.calls.filter(([command]) => command.kind === 'update').map(([command]) => command.input); +} beforeEach(() => { - jest.clearAllMocks(); + mockSend.mockReset(); + mockRelease.mockReset().mockResolvedValue(true); }); -describe('reconcile-concurrency handler', () => { - test('completes without errors when no users exist', async () => { - mockDdbSend.mockResolvedValueOnce({ Items: [], LastEvaluatedKey: undefined }); - await handler(); - expect(mockDdbSend).toHaveBeenCalledTimes(1); - }); - - test('no update when stored count matches actual count', async () => { - // Scan returns one user with active_count=2 - mockDdbSend - .mockResolvedValueOnce({ - Items: [{ user_id: { S: 'user-1' }, active_count: { N: '2' } }], - LastEvaluatedKey: undefined, - }) - // Query returns count=2 (matching) - .mockResolvedValueOnce({ Count: 2, LastEvaluatedKey: undefined }); - - await handler(); - - // 1 scan + 1 query = 2 calls, no UpdateItemCommand - expect(mockDdbSend).toHaveBeenCalledTimes(2); - const calls = mockDdbSend.mock.calls; - const updateCalls = calls.filter((c: any[]) => c[0]._type === 'UpdateItem'); - expect(updateCalls).toHaveLength(0); - }); - - test('fires UpdateItemCommand with ConditionExpression when count drifts', async () => { - // Scan returns one user with active_count=5 - mockDdbSend - .mockResolvedValueOnce({ - Items: [{ user_id: { S: 'user-1' }, active_count: { N: '5' } }], - LastEvaluatedKey: undefined, - }) - // Query returns actual count=2 (drift) - .mockResolvedValueOnce({ Count: 2, LastEvaluatedKey: undefined }) - // Update succeeds - .mockResolvedValueOnce({}); - - await handler(); - - // 1 scan + 1 query + 1 update = 3 calls - expect(mockDdbSend).toHaveBeenCalledTimes(3); - const updateCall = mockDdbSend.mock.calls[2][0]; - expect(updateCall._type).toBe('UpdateItem'); - // Verify ConditionExpression for TOCTOU protection - expect(updateCall.input.ConditionExpression).toBe('active_count = :stored'); - expect(updateCall.input.ExpressionAttributeValues[':stored']).toEqual({ N: '5' }); - expect(updateCall.input.ExpressionAttributeValues[':count']).toEqual({ N: '2' }); - }); - - test('continues to next user on ConditionalCheckFailedException', async () => { - const condErr = new Error('Conditional check failed'); - condErr.name = 'ConditionalCheckFailedException'; - - mockDdbSend - // Scan returns two users - .mockResolvedValueOnce({ - Items: [ - { user_id: { S: 'user-1' }, active_count: { N: '5' } }, - { user_id: { S: 'user-2' }, active_count: { N: '3' } }, - ], - LastEvaluatedKey: undefined, - }) - // User-1: query returns 2 (drift) - .mockResolvedValueOnce({ Count: 2, LastEvaluatedKey: undefined }) - // User-1: update fails with CCF - .mockRejectedValueOnce(condErr) - // User-2: query returns 1 (drift) - .mockResolvedValueOnce({ Count: 1, LastEvaluatedKey: undefined }) - // User-2: update succeeds - .mockResolvedValueOnce({}); - - await handler(); - - // 1 scan + 2 queries + 2 updates = 5 calls - expect(mockDdbSend).toHaveBeenCalledTimes(5); - }); - - test('continues to next user when query fails', async () => { - mockDdbSend - // Scan returns two users - .mockResolvedValueOnce({ - Items: [ - { user_id: { S: 'user-1' }, active_count: { N: '2' } }, - { user_id: { S: 'user-2' }, active_count: { N: '3' } }, - ], - LastEvaluatedKey: undefined, - }) - // User-1: query throws - .mockRejectedValueOnce(new Error('DynamoDB timeout')) - // User-2: query returns 3 (matches) - .mockResolvedValueOnce({ Count: 3, LastEvaluatedKey: undefined }); - - await handler(); - - // 1 scan + 2 queries (one failed) = 3 calls - expect(mockDdbSend).toHaveBeenCalledTimes(3); - }); +test('empty tables complete without writes', async () => { + seed([], []); + await handler(); + expect(updates()).toEqual([]); + expect(mockRelease).not.toHaveBeenCalled(); +}); - test('handles scan pagination', async () => { - mockDdbSend - // First scan page - .mockResolvedValueOnce({ - Items: [{ user_id: { S: 'user-1' }, active_count: { N: '1' } }], - LastEvaluatedKey: { user_id: { S: 'user-1' } }, - }) - // User-1 query - .mockResolvedValueOnce({ Count: 1, LastEvaluatedKey: undefined }) - // Second scan page - .mockResolvedValueOnce({ - Items: [{ user_id: { S: 'user-2' }, active_count: { N: '2' } }], - LastEvaluatedKey: undefined, - }) - // User-2 query - .mockResolvedValueOnce({ Count: 2, LastEvaluatedKey: undefined }); +test('approval waits count as held seats and a matching counter needs no repair', async () => { + seed([{ user_id: 'user', active_count: 2, reservation_version: 'v1' }], + [held('one'), held('two', 'user', 'AWAITING_APPROVAL')]); + await handler(); + expect(updates()).toEqual([]); + expect(mockRelease).not.toHaveBeenCalled(); +}); - await handler(); +test('repairs drift only while the observed count and reservation revision still match', async () => { + seed([{ user_id: 'user', active_count: 5, reservation_version: 'v1' }], [held('one'), held('two')]); + await handler(); + expect(updates()).toEqual([expect.objectContaining({ + Key: { user_id: 'user' }, + ConditionExpression: 'attribute_exists(user_id) AND active_count = :stored AND reservation_version = :observed', + ExpressionAttributeValues: expect.objectContaining({ ':count': 2, ':stored': 5, ':observed': 'v1' }), + })]); +}); - // 2 scans + 2 queries = 4 calls - expect(mockDdbSend).toHaveBeenCalledTimes(4); +test('missing counters are recreated only if no reservation writer has created them meanwhile', async () => { + seed([], [held('one')]); + await handler(); + expect(updates()[0]).toMatchObject({ + ConditionExpression: 'attribute_not_exists(user_id)', + ExpressionAttributeValues: { ':count': 1 }, }); +}); - test('handles query pagination for countActiveTasks', async () => { - mockDdbSend - // Scan - .mockResolvedValueOnce({ - Items: [{ user_id: { S: 'user-1' }, active_count: { N: '0' } }], - LastEvaluatedKey: undefined, - }) - // First query page: count=3, has more - .mockResolvedValueOnce({ - Count: 3, - LastEvaluatedKey: { user_id: { S: 'user-1' }, task_id: { S: 'T3' } }, - }) - // Second query page: count=2, done - .mockResolvedValueOnce({ Count: 2, LastEvaluatedKey: undefined }) - // Update (drift: stored=0 vs actual=5) - .mockResolvedValueOnce({}); - - await handler(); +test('legacy counter repair checks that a version has not been installed concurrently', async () => { + seed([{ user_id: 'user', active_count: 4 }], [held('one')]); + await handler(); + expect(updates()[0].ConditionExpression).toContain('attribute_not_exists(reservation_version)'); +}); - // 1 scan + 2 queries + 1 update = 4 calls - expect(mockDdbSend).toHaveBeenCalledTimes(4); - const updateCall = mockDdbSend.mock.calls[3][0]; - expect(updateCall._type).toBe('UpdateItem'); - expect(updateCall.input.ExpressionAttributeValues[':count']).toEqual({ N: '5' }); - }); +test('ambiguous legacy active tasks prevent guessing a count', async () => { + seed([{ user_id: 'user', active_count: 5 }], [{ task_id: 'legacy', user_id: 'user', status: 'AWAITING_APPROVAL' }]); + await handler(); + expect(updates()).toEqual([]); +}); - test('skips items without user_id', async () => { - mockDdbSend.mockResolvedValueOnce({ - Items: [{ active_count: { N: '1' } }], // no user_id - LastEvaluatedKey: undefined, - }); +test('queued and unadmitted terminal tasks do not inflate the count', async () => { + seed([{ user_id: 'user', active_count: 3 }], [ + { task_id: 'queued', user_id: 'user', status: 'QUEUED' }, + { task_id: 'failed', user_id: 'user', status: 'FAILED' }, + ]); + await handler(); + expect(updates()[0].ExpressionAttributeValues[':count']).toBe(0); +}); - await handler(); +test('counts terminal held markers before asking shared cleanup to release them', async () => { + seed([{ user_id: 'user', active_count: 0, reservation_version: 'v1' }], [held('done', 'user', 'FAILED'), held('live')]); + await handler(); + expect(updates()[0].ExpressionAttributeValues[':count']).toBe(2); + expect(mockRelease).toHaveBeenCalledWith('done', 'user'); + expect(mockSend.mock.invocationCallOrder.at(-1)!).toBeLessThan(mockRelease.mock.invocationCallOrder[0]); +}); - // Only the scan call, no query or update - expect(mockDdbSend).toHaveBeenCalledTimes(1); - }); +test('a changed revision skips stale repair and still tries terminal cleanup', async () => { + seed([{ user_id: 'user', active_count: 3, reservation_version: 'v1' }], [held('done', 'user', 'FAILED')]); + mockSend.mockRejectedValueOnce(Object.assign(new Error('changed'), { name: 'ConditionalCheckFailedException' })); + await handler(); + expect(updates()).toHaveLength(1); + expect(mockRelease).toHaveBeenCalledWith('done', 'user'); +}); - test('continues to next user when UpdateItemCommand fails with non-CCF error', async () => { - mockDdbSend - // Scan returns two users with drift - .mockResolvedValueOnce({ - Items: [ - { user_id: { S: 'user-1' }, active_count: { N: '5' } }, - { user_id: { S: 'user-2' }, active_count: { N: '4' } }, - ], - LastEvaluatedKey: undefined, - }) - // User-1: query returns 2 (drift) - .mockResolvedValueOnce({ Count: 2, LastEvaluatedKey: undefined }) - // User-1: update fails with non-CCF error - .mockRejectedValueOnce(new Error('InternalServerError')) - // User-2: query returns 1 (drift) - .mockResolvedValueOnce({ Count: 1, LastEvaluatedKey: undefined }) - // User-2: update succeeds - .mockResolvedValueOnce({}); +test('one user repair failure does not stop the next user', async () => { + seed([{ user_id: 'one', active_count: 3 }, { user_id: 'two', active_count: 3 }], []); + mockSend.mockRejectedValueOnce(new Error('unavailable')); + await handler(); + expect(updates().map(update => update.Key.user_id)).toEqual(['one', 'two']); +}); - await handler(); +test('scans every counter page before strongly scanning every task page', async () => { + mockSend.mockResolvedValueOnce({ Items: [{ user_id: 'user', active_count: 2 }], LastEvaluatedKey: { user_id: 'user' } }) + .mockResolvedValueOnce({ Items: [] }) + .mockResolvedValueOnce({ Items: [held('one')], LastEvaluatedKey: { task_id: 'one' } }) + .mockResolvedValueOnce({ Items: [held('two')] }); + await handler(); + expect(mockSend.mock.calls.map(([command]) => [command.input.TableName, command.input.ConsistentRead])) + .toEqual([['Counters', true], ['Counters', true], ['Tasks', true], ['Tasks', true]]); + expect(mockSend.mock.calls[3][0].input.ExclusiveStartKey).toEqual({ task_id: 'one' }); +}); - // 1 scan + 2 queries + 2 updates = 5 calls - expect(mockDdbSend).toHaveBeenCalledTimes(5); - // Verify user-2's update was still attempted and succeeded - const updateCalls = mockDdbSend.mock.calls.filter((c: any[]) => c[0]._type === 'UpdateItem'); - expect(updateCalls).toHaveLength(2); - // User-2's update should have the correct values - const user2Update = updateCalls[1][0]; - expect(user2Update.input.Key).toEqual({ user_id: { S: 'user-2' } }); - expect(user2Update.input.ExpressionAttributeValues[':count']).toEqual({ N: '1' }); - }); +test('an incomplete task scan aborts before any partial count is installed', async () => { + mockSend.mockResolvedValueOnce({ Items: [{ user_id: 'user', active_count: 2 }] }) + .mockResolvedValueOnce({ Items: [held('one')], LastEvaluatedKey: { task_id: 'one' } }) + .mockRejectedValueOnce(new Error('scan unavailable')); + await expect(handler()).rejects.toThrow('scan unavailable'); + expect(updates()).toEqual([]); }); diff --git a/cdk/test/handlers/reconcile-stranded-tasks.test.ts b/cdk/test/handlers/reconcile-stranded-tasks.test.ts index 8c2346c5c..0e1c1f1b5 100644 --- a/cdk/test/handlers/reconcile-stranded-tasks.test.ts +++ b/cdk/test/handlers/reconcile-stranded-tasks.test.ts @@ -19,6 +19,11 @@ // --- Mocks --- const mockDdbSend = jest.fn(); +const mockRelease = jest.fn(); +jest.mock('../../src/handlers/shared/task-concurrency', () => ({ + releaseTaskSlot: (...args: unknown[]) => mockRelease(...args), +})); +beforeEach(() => mockRelease.mockReset().mockResolvedValue(true)); jest.mock('@aws-sdk/client-dynamodb', () => ({ DynamoDBClient: jest.fn(() => ({ send: mockDdbSend })), QueryCommand: jest.fn((input: unknown) => ({ _type: 'Query', input })), @@ -89,7 +94,7 @@ describe('reconcile-stranded-tasks', () => { expect(mockDdbSend).toHaveBeenCalledTimes(3); }); - test('task older than 1200s → fails + emits events + decrements concurrency', async () => { + test('task older than 1200s → fails + emits events + releases the task reservation', async () => { const ancient = new Date(Date.now() - 25 * 60 * 1000).toISOString(); // 25 min ago primeResponses([ // Query SUBMITTED returns one stranded candidate. @@ -103,7 +108,6 @@ describe('reconcile-stranded-tasks', () => { {}, // conditional UpdateItem → FAILED {}, // PutItem task_stranded event {}, // PutItem task_failed event - {}, // UpdateItem decrement concurrency { Items: [] }, // Query HYDRATING { Items: [] }, // Query AWAITING_APPROVAL ]); @@ -132,10 +136,7 @@ describe('reconcile-stranded-tasks', () => { }); expect(eventTypes).toEqual(expect.arrayContaining(['task_stranded', 'task_failed'])); - // Concurrency decrement. - const decrementCall = (mockDdbSend.mock.calls as [{ _type: string; input: Record }][]) - .find(([c]) => c._type === 'UpdateItem' && String(c.input.UpdateExpression).includes('active_count')); - expect(decrementCall).toBeDefined(); + expect(mockRelease).toHaveBeenCalledWith('t-stranded', 'u-1'); }); test('#441: task with old created_at but FRESH status_created_at is NOT failed (freshly picked up from the queue)', async () => { @@ -180,7 +181,6 @@ describe('reconcile-stranded-tasks', () => { {}, // conditional UpdateItem → FAILED {}, // PutItem task_stranded event {}, // PutItem task_failed event - {}, // UpdateItem decrement concurrency { Items: [] }, // HYDRATING { Items: [] }, // AWAITING_APPROVAL ]); @@ -215,7 +215,6 @@ describe('reconcile-stranded-tasks', () => { {}, // conditional UpdateItem → FAILED {}, // PutItem task_stranded event {}, // PutItem task_failed event - {}, // UpdateItem decrement concurrency { Items: [] }, // HYDRATING { Items: [] }, // AWAITING_APPROVAL ]); @@ -246,7 +245,7 @@ describe('reconcile-stranded-tasks', () => { { Items: [] }, // AWAITING_APPROVAL query ]); - // Must NOT throw; no events written, no concurrency decrement. + // Must NOT throw; no events written, no reservation release. await handler(); const writes = (mockDdbSend.mock.calls as [{ _type: string; input: Record }][]) @@ -274,7 +273,6 @@ describe('reconcile-stranded-tasks', () => { {}, // PutItem task_stranded event {}, // PutItem task_failed event {}, // PutItem approval_stranded milestone (Chunk 10 new) - {}, // UpdateItem decrement concurrency ]); await handler(); @@ -312,7 +310,6 @@ describe('reconcile-stranded-tasks', () => { {}, // conditional UpdateItem → FAILED {}, // PutItem task_stranded {}, // PutItem task_failed - {}, // UpdateItem concurrency { Items: [] }, // HYDRATING { Items: [] }, // AWAITING_APPROVAL ]); @@ -441,7 +438,6 @@ describe('reconcile-stranded-tasks', () => { {}, // UpdateItem t-ok (transition) → success {}, // PutItem task_stranded event {}, // PutItem task_failed event - {}, // UpdateItem decrement concurrency ddbErr, // UpdateItem t-fail (transition) → throws { Items: [] }, // HYDRATING query { Items: [] }, // AWAITING_APPROVAL query @@ -467,8 +463,8 @@ describe('reconcile-stranded-tasks', () => { mockTaskRow({ task_id: 't-2', user_id: 'u-b', created_at: ancient }), ], }, - {}, {}, {}, {}, // t-1: transition + 2 events + decrement - {}, {}, {}, {}, // t-2: transition + 2 events + decrement + {}, {}, {}, // t-1: transition + 2 events + {}, {}, {}, // t-2: transition + 2 events { Items: [] }, // HYDRATING { Items: [] }, // AWAITING_APPROVAL ]); diff --git a/cdk/test/handlers/shared/task-concurrency-local.test.ts b/cdk/test/handlers/shared/task-concurrency-local.test.ts new file mode 100644 index 000000000..e669fca56 --- /dev/null +++ b/cdk/test/handlers/shared/task-concurrency-local.test.ts @@ -0,0 +1,320 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +/** + * Opt-in real DynamoDB Local tests: + * ABCA_DDB_LOCAL_ENDPOINT=http://127.0.0.1: mise run testf -- task-concurrency-local + * The endpoint is restricted to loopback and every client uses dummy credentials. + */ +import { randomUUID } from 'node:crypto'; +import { CreateTableCommand, DeleteTableCommand, DynamoDBClient } from '@aws-sdk/client-dynamodb'; +import { DeleteCommand, DynamoDBDocumentClient, GetCommand, PutCommand, ScanCommand, UpdateCommand } from '@aws-sdk/lib-dynamodb'; + +const endpoint = process.env.ABCA_DDB_LOCAL_ENDPOINT; +if (endpoint && (new URL(endpoint).hostname !== '127.0.0.1' || new URL(endpoint).protocol !== 'http:')) { + throw new Error('Capacity integration tests require an http://127.0.0.1 DynamoDB Local endpoint'); +} +const mockBeforeSend = jest.fn(); +const mockAfterSend = jest.fn(); +const mockClients: DynamoDBDocumentClient[] = []; +jest.mock('../../../src/handlers/shared/ua', () => { + const actual = jest.requireActual('../../../src/handlers/shared/ua'); + return { + ...actual, + makeDocClient: () => { + const client = actual.makeDocClient({ + endpoint: process.env.ABCA_DDB_LOCAL_ENDPOINT ?? 'http://127.0.0.1:1', + region: 'us-east-1', + credentials: { accessKeyId: 'local', secretAccessKey: 'local' }, + }); + const send = client.send.bind(client); + client.send = async (command: unknown) => { + await mockBeforeSend(command); + const result = await send(command); + await mockAfterSend(command, result); + return result; + }; + mockClients.push(client); + return client; + }, + }; +}); + +const suffix = randomUUID(); +const tasks = `capacity-tasks-${suffix}`; +const counters = `capacity-counters-${suffix}`; +const events = `capacity-events-${suffix}`; +Object.assign(process.env, { + TASK_TABLE_NAME: tasks, + USER_CONCURRENCY_TABLE_NAME: counters, + TASK_EVENTS_TABLE_NAME: events, + MEMORY_ID: '', +}); + +import { handler as reconcile } from '../../../src/handlers/reconcile-concurrency'; +import { failTask, finalizeTask } from '../../../src/handlers/shared/orchestrator'; +import { acquireTaskSlot, releaseTaskSlot } from '../../../src/handlers/shared/task-concurrency'; + +const raw = new DynamoDBClient({ + endpoint: endpoint ?? 'http://127.0.0.1:1', + region: 'us-east-1', + credentials: { accessKeyId: 'local', secretAccessKey: 'local' }, +}); +const admin = DynamoDBDocumentClient.from(raw); +const local = endpoint ? describe : describe.skip; +jest.setTimeout(30_000); + +local('task capacity against DynamoDB Local', () => { + beforeAll(async () => { + for (const [name, key] of [[tasks, 'task_id'], [counters, 'user_id'], [events, 'task_id']]) { + await raw.send(new CreateTableCommand({ + TableName: name, + BillingMode: 'PAY_PER_REQUEST', + AttributeDefinitions: [ + { AttributeName: key, AttributeType: 'S' }, + ...(name === events ? [{ AttributeName: 'event_id', AttributeType: 'S' as const }] : []), + ], + KeySchema: [ + { AttributeName: key, KeyType: 'HASH' }, + ...(name === events ? [{ AttributeName: 'event_id', KeyType: 'RANGE' as const }] : []), + ], + })); + } + }); + beforeEach(async () => { + mockBeforeSend.mockReset(); + mockAfterSend.mockReset(); + for (const [name, key] of [[tasks, 'task_id'], [counters, 'user_id'], [events, 'task_id']]) { + const page = await admin.send(new ScanCommand({ TableName: name })); + for (const item of page.Items ?? []) { + await admin.send(new DeleteCommand({ + TableName: name, + Key: { [key]: item[key], ...(name === events && { event_id: item.event_id }) }, + })); + } + } + }); + afterAll(async () => { + for (const name of [tasks, counters, events]) await raw.send(new DeleteTableCommand({ TableName: name })); + for (const client of mockClients) client.destroy(); + raw.destroy(); + }); + + async function seed(id: string, initialStatus = 'SUBMITTED', user = 'user') { + await admin.send(new PutCommand({ + TableName: tasks, + Item: { + task_id: id, user_id: user, status: initialStatus, memory_written: true, + }, + })); + } + async function status(id: string, value: string) { + await admin.send(new UpdateCommand({ + TableName: tasks, + Key: { task_id: id }, + UpdateExpression: 'SET #status = :status', + ExpressionAttributeNames: { '#status': 'status' }, + ExpressionAttributeValues: { ':status': value }, + })); + } + async function task(id: string) { + return (await admin.send(new GetCommand({ + TableName: tasks, Key: { task_id: id }, ConsistentRead: true, + }))).Item!; + } + async function count() { + return (await admin.send(new GetCommand({ + TableName: counters, Key: { user_id: 'user' }, ConsistentRead: true, + }))).Item?.active_count ?? 0; + } + async function twoSeats() { + await seed('one'); + await seed('two'); + await acquireTaskSlot('one', 'user', 3); + await acquireTaskSlot('two', 'user', 3); + } + + test('crash replay of real finalization preserves the other task reservation', async () => { + await twoSeats(); + await status('one', 'COMPLETED'); + await finalizeTask('one', { attempts: 10 }, 'user'); + await finalizeTask('one', { attempts: 10 }, 'user'); + expect(await count()).toBe(1); + expect((await task('one')).concurrency_slot.state).toBe('released'); + expect((await task('two')).concurrency_slot.state).toBe('held'); + }); + + test('simultaneous admission of one task creates exactly one reservation', async () => { + await seed('one'); + const results = await Promise.allSettled(Array.from({ length: 5 }, () => acquireTaskSlot('one', 'user', 3))); + expect(results.some(result => result.status === 'fulfilled' && result.value)).toBe(true); + expect(await acquireTaskSlot('one', 'user', 3)).toBe(true); + expect(await count()).toBe(1); + }); + + test('different tasks cannot exceed the admission cap', async () => { + const ids = ['a', 'b', 'c', 'd']; + for (const id of ids) await seed(id); + await Promise.allSettled(ids.map(id => acquireTaskSlot(id, 'user', 2))); + // Retry any transaction conflicts serially, as durable execution does. + for (const id of ids) await acquireTaskSlot(id, 'user', 2); + expect(await count()).toBe(2); + // Four fixed fixture reads; no input-dependent fan-out. + // eslint-disable-next-line @cdklabs/promiseall-no-unbounded-parallelism + const rows = await Promise.all([task('a'), task('b'), task('c'), task('d')]); + expect(rows.filter(row => row.concurrency_slot?.state === 'held')).toHaveLength(2); + }); + + test.each(['acquire', 'release'])('a lost committed %s response is recovered from the marker', async (operation) => { + await seed('one'); + if (operation === 'release') { + await acquireTaskSlot('one', 'user', 3); + await status('one', 'FAILED'); + } + mockAfterSend.mockImplementationOnce((command) => { + if (command.constructor.name === 'GetCommand') { + mockAfterSend.mockImplementationOnce(() => { throw new Error('transaction response lost'); }); + } + }); + if (operation === 'acquire') await acquireTaskSlot('one', 'user', 3); + else await releaseTaskSlot('one', 'user'); + expect(await count()).toBe(operation === 'acquire' ? 1 : 0); + expect((await task('one')).concurrency_slot.state).toBe(operation === 'acquire' ? 'held' : 'released'); + }); + + test('competing finalizers and direct cleanup only release once', async () => { + await twoSeats(); + await status('one', 'CANCELLED'); + await Promise.allSettled([ + finalizeTask('one', { attempts: 0 }, 'user'), + finalizeTask('one', { attempts: 0 }, 'user'), + releaseTaskSlot('one', 'user'), + ]); + await releaseTaskSlot('one', 'user'); + expect(await count()).toBe(1); + }); + + test('start failure and replay release a held slot once', async () => { + await twoSeats(); + await status('one', 'HYDRATING'); + await failTask('one', 'HYDRATING', 'start failed', 'user', true); + await failTask('one', 'HYDRATING', 'start failed', 'user', true); + expect(await count()).toBe(1); + expect((await task('one')).status).toBe('FAILED'); + }); + + test('approval waits retain capacity through reconciliation', async () => { + await twoSeats(); + await status('one', 'AWAITING_APPROVAL'); + expect(await releaseTaskSlot('one', 'user')).toBe(false); + await reconcile(); + expect(await count()).toBe(2); + }); + + test('queued and unadmitted terminal tasks cannot release another reservation', async () => { + await seed('running'); + await acquireTaskSlot('running', 'user', 3); + await seed('queued', 'QUEUED'); + await seed('cancelled', 'CANCELLED'); + expect(await acquireTaskSlot('queued', 'user', 3)).toBe(false); + await releaseTaskSlot('queued', 'user'); + await releaseTaskSlot('cancelled', 'user'); + expect(await count()).toBe(1); + }); + + test('owner mismatch changes neither the reservation nor counter', async () => { + await seed('one'); + await expect(acquireTaskSlot('one', 'someone-else', 3)).rejects.toThrow('owner'); + expect(await count()).toBe(0); + expect((await task('one')).concurrency_slot).toBeUndefined(); + }); + + test('cancellation between the read and admission transaction wins', async () => { + await seed('one'); + mockBeforeSend.mockImplementation(async (command) => { + if (command.constructor.name === 'TransactWriteCommand') { + mockBeforeSend.mockReset(); + await status('one', 'CANCELLED'); + } + }); + expect(await acquireTaskSlot('one', 'user', 3)).toBe(false); + expect(await count()).toBe(0); + }); + + test('an empty counter release preserves an admission that races into the empty-counter branch', async () => { + await seed('one'); + await acquireTaskSlot('one', 'user', 3); + await status('one', 'FAILED'); + await admin.send(new DeleteCommand({ TableName: counters, Key: { user_id: 'user' } })); + await seed('two'); + mockBeforeSend.mockImplementation(async (command) => { + if (command.constructor.name === 'TransactWriteCommand' + && command.input.TransactItems[1].Update.UpdateExpression.includes('if_not_exists')) { + mockBeforeSend.mockReset(); + await acquireTaskSlot('two', 'user', 3); + } + }); + await releaseTaskSlot('one', 'user'); + expect(await count()).toBe(1); + expect((await task('one')).concurrency_slot.state).toBe('released'); + expect((await task('two')).concurrency_slot.state).toBe('held'); + }); + + test('periodic cleanup recovers a crash between terminal status and release', async () => { + await twoSeats(); + await status('one', 'TIMED_OUT'); + await reconcile(); + await finalizeTask('one', { attempts: 10 }, 'user'); + expect(await count()).toBe(1); + }); + + test('an ambiguous legacy active task prevents guessing a replacement count', async () => { + await twoSeats(); + await seed('legacy', 'AWAITING_APPROVAL'); + await reconcile(); + expect(await count()).toBe(2); + expect((await task('legacy')).concurrency_slot).toBeUndefined(); + }); + + test('a revision change blocks stale repair even when the counter returns to its original number', async () => { + await seed('one'); + await acquireTaskSlot('one', 'user', 3); + await admin.send(new UpdateCommand({ + TableName: counters, + Key: { user_id: 'user' }, + UpdateExpression: 'SET active_count = :zero', + ExpressionAttributeValues: { ':zero': 0 }, + })); + mockAfterSend.mockImplementation(async (command) => { + if (command.constructor.name === 'ScanCommand' && command.input.TableName === tasks) { + mockAfterSend.mockReset(); + await seed('two'); + await acquireTaskSlot('two', 'user', 3); + await status('one', 'COMPLETED'); + await releaseTaskSlot('one', 'user'); + } + }); + await reconcile(); + // Admission then release brought the count back to zero, but changed its + // revision. The old scan must not write its answer into this new state. + expect(await count()).toBe(0); + await reconcile(); + expect(await count()).toBe(1); + }); +}); diff --git a/cdk/test/handlers/shared/task-concurrency.test.ts b/cdk/test/handlers/shared/task-concurrency.test.ts new file mode 100644 index 000000000..2663db132 --- /dev/null +++ b/cdk/test/handlers/shared/task-concurrency.test.ts @@ -0,0 +1,153 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +const mockSend = jest.fn(); +jest.mock('@aws-sdk/lib-dynamodb', () => ({ + GetCommand: jest.fn((input: unknown) => ({ kind: 'get', input })), + TransactWriteCommand: jest.fn((input: unknown) => ({ kind: 'transaction', input })), +})); +jest.mock('../../../src/handlers/shared/ua', () => ({ + makeDocClient: () => ({ send: mockSend }), +})); +process.env.TASK_TABLE_NAME = 'Tasks'; +process.env.USER_CONCURRENCY_TABLE_NAME = 'Counters'; + +import { acquireTaskSlot, releaseTaskSlot } from '../../../src/handlers/shared/task-concurrency'; + +const base = { task_id: 'task', user_id: 'user', status: 'SUBMITTED' }; +const held = { state: 'held', acquired_at: '2026-09-13T00:00:00Z' }; +function cancelled(index: number) { + return Object.assign(new Error('transaction cancelled'), { + name: 'TransactionCanceledException', + CancellationReasons: [0, 1].map(i => ({ Code: i === index ? 'ConditionalCheckFailed' : 'None' })), + }); +} +beforeEach(() => mockSend.mockReset()); + +test('admission atomically ties a counter increment to an owned SUBMITTED task', async () => { + mockSend.mockResolvedValueOnce({ Item: base }).mockResolvedValueOnce({}); + expect(await acquireTaskSlot('task', 'user', 3)).toBe(true); + const transaction = mockSend.mock.calls[1][0].input; + expect(transaction.TransactItems).toHaveLength(2); + expect(transaction.TransactItems[0].Update).toMatchObject({ + TableName: 'Tasks', + Key: { task_id: 'task' }, + ConditionExpression: 'user_id = :user AND #status = :submitted AND attribute_not_exists(concurrency_slot)', + ExpressionAttributeValues: { ':slot': { state: 'held' }, ':user': 'user', ':submitted': 'SUBMITTED' }, + }); + expect(transaction.TransactItems[1].Update).toMatchObject({ + TableName: 'Counters', + Key: { user_id: 'user' }, + ConditionExpression: 'attribute_not_exists(active_count) OR active_count < :max', + ExpressionAttributeValues: { ':max': 3, ':version': transaction.ClientRequestToken }, + }); +}); + +test.each(['SUBMITTED', 'HYDRATING', 'RUNNING', 'AWAITING_APPROVAL', 'FINALIZING'])( + 'replaying admission in %s reuses the held reservation', async (status) => { + mockSend.mockResolvedValueOnce({ Item: { ...base, status, concurrency_slot: held } }); + expect(await acquireTaskSlot('task', 'user', 3)).toBe(true); + expect(mockSend).toHaveBeenCalledTimes(1); + }, +); + +test('a full counter declines admission without leaving a marker', async () => { + mockSend.mockResolvedValueOnce({ Item: base }).mockRejectedValueOnce(cancelled(1)) + .mockResolvedValueOnce({ Item: base }); + expect(await acquireTaskSlot('task', 'user', 3)).toBe(false); +}); + +test('a transaction outage is not misreported as a full counter', async () => { + mockSend.mockResolvedValueOnce({ Item: base }).mockRejectedValueOnce(new Error('throttled')) + .mockResolvedValueOnce({ Item: base }); + await expect(acquireTaskSlot('task', 'user', 3)).rejects.toThrow('throttled'); +}); + +test('a lost admission response recovers the committed held marker', async () => { + mockSend.mockResolvedValueOnce({ Item: base }).mockRejectedValueOnce(new Error('response lost')) + .mockResolvedValueOnce({ Item: { ...base, concurrency_slot: held } }); + expect(await acquireTaskSlot('task', 'user', 3)).toBe(true); +}); + +test('a cancelled task cannot acquire capacity', async () => { + mockSend.mockResolvedValueOnce({ Item: { ...base, status: 'CANCELLED' } }); + expect(await acquireTaskSlot('task', 'user', 3)).toBe(false); +}); + +test('a released task cannot reacquire even if its status is changed', async () => { + mockSend.mockResolvedValueOnce({ Item: { ...base, concurrency_slot: { ...held, state: 'released' } } }); + expect(await acquireTaskSlot('task', 'user', 3)).toBe(false); +}); + +test('owner mismatch never writes', async () => { + mockSend.mockResolvedValueOnce({ Item: base }); + await expect(acquireTaskSlot('task', 'other', 3)).rejects.toThrow('owner'); + expect(mockSend).toHaveBeenCalledTimes(1); +}); + +test.each(['COMPLETED', 'FAILED', 'CANCELLED', 'TIMED_OUT'])( + '%s releases the marker and count in one guarded transaction', async (status) => { + mockSend.mockResolvedValueOnce({ Item: { ...base, status, concurrency_slot: held } }) + .mockResolvedValueOnce({}); + expect(await releaseTaskSlot('task', 'user')).toBe(true); + const transaction = mockSend.mock.calls[1][0].input; + expect(transaction.TransactItems[0].Update).toMatchObject({ + ConditionExpression: 'user_id = :user AND concurrency_slot.#state = :held AND #status IN (:completed, :failed, :cancelled, :timedOut)', + ExpressionAttributeValues: { ':released': 'released', ':held': 'held' }, + }); + expect(transaction.TransactItems[1].Update).toMatchObject({ + UpdateExpression: 'SET active_count = active_count - :one, updated_at = :now, reservation_version = :version', + ConditionExpression: 'active_count > :zero', + }); + }, +); + +test('an active task retains its slot', async () => { + mockSend.mockResolvedValueOnce({ Item: { ...base, status: 'AWAITING_APPROVAL', concurrency_slot: held } }); + expect(await releaseTaskSlot('task', 'user')).toBe(false); + expect(mockSend).toHaveBeenCalledTimes(1); +}); + +test('unadmitted or legacy terminal tasks do not return another task seat', async () => { + mockSend.mockResolvedValueOnce({ Item: { ...base, status: 'FAILED' } }); + expect(await releaseTaskSlot('task', 'user')).toBe(false); +}); + +test('release recovers a competing commit or lost successful response', async () => { + mockSend.mockResolvedValueOnce({ Item: { ...base, status: 'FAILED', concurrency_slot: held } }) + .mockRejectedValueOnce(new Error('response lost')) + .mockResolvedValueOnce({ Item: { ...base, status: 'FAILED', concurrency_slot: { ...held, state: 'released' } } }); + expect(await releaseTaskSlot('task', 'user')).toBe(false); +}); + +test('an empty counter closes the marker without subtracting from later admissions', async () => { + const row = { Item: { ...base, status: 'FAILED', concurrency_slot: held } }; + mockSend.mockResolvedValueOnce(row).mockRejectedValueOnce(cancelled(1)) + .mockResolvedValueOnce(row).mockResolvedValueOnce({}); + expect(await releaseTaskSlot('task', 'user')).toBe(true); + const update = mockSend.mock.calls[3][0].input.TransactItems[1].Update; + expect(update.UpdateExpression).toBe('SET active_count = if_not_exists(active_count, :zero), updated_at = :now, reservation_version = :version'); + expect(update.ExpressionAttributeValues).not.toHaveProperty(':one'); +}); + +test('a release outage propagates with the held marker intact', async () => { + const row = { Item: { ...base, status: 'FAILED', concurrency_slot: held } }; + mockSend.mockResolvedValueOnce(row).mockRejectedValueOnce(new Error('unavailable')).mockResolvedValueOnce(row); + await expect(releaseTaskSlot('task', 'user')).rejects.toThrow('unavailable'); +}); diff --git a/docs/design/ATTACHMENTS.md b/docs/design/ATTACHMENTS.md index 7f5932100..da5963e9c 100644 --- a/docs/design/ATTACHMENTS.md +++ b/docs/design/ATTACHMENTS.md @@ -1290,13 +1290,13 @@ The task record is preserved with status `CANCELLED` — `bgagent status 0`. +Counter changes belong to the reservation transactions described above. Do not add a standalone increment or decrement: it bypasses per-task replay protection. diff --git a/docs/src/content/docs/architecture/Attachments.md b/docs/src/content/docs/architecture/Attachments.md index e92b64771..730ec1d12 100644 --- a/docs/src/content/docs/architecture/Attachments.md +++ b/docs/src/content/docs/architecture/Attachments.md @@ -1294,13 +1294,13 @@ The task record is preserved with status `CANCELLED` — `bgagent status 0`. +Counter changes belong to the reservation transactions described above. Do not add a standalone increment or decrement: it bypasses per-task replay protection. diff --git a/docs/verification/645-capacity-reservations.md b/docs/verification/645-capacity-reservations.md new file mode 100644 index 000000000..4fe4e8384 --- /dev/null +++ b/docs/verification/645-capacity-reservations.md @@ -0,0 +1,48 @@ +# Capacity reservation verification for #645 + +The replay regression starts with two occupied seats, finalizes one task, then repeats that finalizer as if its checkpoint were lost. The old implementation reduced the counter to zero. The new implementation keeps the other task's reservation and a count of one. + +## Local transaction tests + +`cdk/test/handlers/shared/task-concurrency-local.test.ts` uses real DynamoDB Local transactions and conditional expressions. It replaces client construction only to select an explicit loopback endpoint and dummy credentials. Fault hooks can discard a response after the database commits. + +The suite covers repeated finalization, concurrent admission/cap enforcement, lost acquisition/release responses, competing cleaners, failure replay, cancellation before admission, approval occupancy, unadmitted/queued tasks, wrong owners, empty counters with a racing admission, periodic terminal cleanup, ambiguous legacy rows and stale repair after a revision change. + +Run an isolated in-memory instance: + +```sh +docker run --rm -d --name abca645-capacity-ddb \ + --memory 512m --cpus 1 -p 127.0.0.1::8000 \ + amazon/dynamodb-local@sha256:ff89bd48ff32cd8d9be5fee8873b65b8854dc408f1afe881be6eb00247bc0dab \ + -jar DynamoDBLocal.jar -inMemory -sharedDb +docker port abca645-capacity-ddb 8000/tcp +``` + +From `cdk/`, use the printed port: + +```sh +ABCA_DDB_LOCAL_ENDPOINT=http://127.0.0.1: mise run testf -- task-concurrency-local +``` + +The test rejects non-loopback endpoints. It creates uniquely named temporary tables, deletes them after the suite and closes its clients. Without the environment variable, this optional integration suite is skipped; ordinary helper/handler tests still run. Stop the temporary database after verification: + +```sh +docker stop abca645-capacity-ddb +``` + +## Upgrade and live verification + +1. Pause new submissions and allow old coordinator executions to finish, or terminate them through normal task cancellation/cleanup. Drain pending uploads/queue pickups as appropriate to prevent old code from starting work during the update. +2. Deploy all changed writers together: orchestrator, upload confirmation, stranded cleaner, counter reconciler and queue pickup. The counter reconciler now needs task-table `UpdateItem`; upload confirmation only needs counter reads. These changes use existing tables and roles. +3. Let older unmarked active tasks settle. The new helper does not guess their reservation ownership, and repair skips a user with ambiguous active rows. Inspect `CONCURRENCY_RESERVATION_UNKNOWN`; verify actual task/compute state before manual corrections. +4. Run reconciliation with admissions paused, check task reservations against counters, then reopen admission. For rollback, drain tasks using the new protocol before returning to old counter writers. +5. Verify one normal completion, start failure, cancellation, stranded task, approval wait, queued pickup and interrupted finalization against the deployed policies. Check that one task's cleanup preserves other tasks' seats. +6. Measure the two strongly consistent base-table scans at realistic retention volume. The function has a five-minute timeout; interrupted scans must not install partial counts. Monitor scan duration/read capacity and failures. + +`CONCURRENCY_EMPTY_COUNTER` means a held reservation was found without a positive counter. The fallback closes its marker without subtracting from later admissions. Concurrent repair can leave an overcount for the next sweep to correct. + +## Limits of this evidence + +Local tests prove the application requests and DynamoDB Local's transaction behavior. They do not establish deployed IAM, AWS scaling, successful rollout or MicroVM sleep/wake behavior. Terminal events may repeat or be lost independently of the atomic seat update. + +The reservation/start markers share an agent-writable task row. Their public-API omission does not protect them against a compromised agent. The [P3 plan](./645-p3-implementation-plan.md#1g-protect-coordinator-owned-metadata) tracks that separate security boundary. From 102b9ef12db261817d5d3cf3379f702293935da5 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Sun, 13 Sep 2026 16:25:33 -0400 Subject: [PATCH 013/149] docs(645): record capacity replay proof and metadata security gap --- .../645-p3-implementation-plan.md | 30 +++++++++++++++++-- docs/verification/645-p3-readiness-review.md | 4 ++- 2 files changed, 30 insertions(+), 4 deletions(-) diff --git a/docs/verification/645-p3-implementation-plan.md b/docs/verification/645-p3-implementation-plan.md index 2d978281f..d73459666 100644 --- a/docs/verification/645-p3-implementation-plan.md +++ b/docs/verification/645-p3-implementation-plan.md @@ -16,7 +16,9 @@ Prerequisite work is tracked here on `fix/645-microvm-readiness`. “Completed - [ ] Bind configuration to trusted deployment identity and restrict payload reads per task (#817 / #700). - [x] Implement saved MicroVM start receipts, stable tokens, input fingerprints and handle recovery. - [ ] Verify AWS token retention/conflicts and unknown-start cleanup on a live deployment. -- [ ] Make the shared finalizer's concurrency release atomic per task across crash replay. +- [x] Make capacity acquisition/release atomic per task across crash replay; unify counter writers and repair. +- [ ] Verify the capacity protocol's upgrade/drain procedure, deployed IAM and scan scale in AWS. +- [ ] Protect coordinator-owned task metadata from agent writes before claiming hostile-task isolation. - [ ] Finish logging-failure observability (#810) and registry-overflow coverage (#818). - [ ] Implement production nesting if included, then verify a clean P2 deployment. - [ ] Implement and verify the P3 sleep/wake lifecycle described below. @@ -51,7 +53,17 @@ Third prerequisite batch completed locally on 2026-09-13: Final checks for the third batch: CDK ESLint and TypeScript compilation passed. `mise run testf -- test/handlers/ --detectOpenHandles` passed **151 suites / 3,557 tests** and exited successfully. An earlier ordinary broad run reported a delayed-exit warning; the diagnostic run produced no open-handle trace, so its cause remains unidentified. The changed start/recovery integration suites also exited cleanly in isolation. Documentation sync, the **77-page** Astro build and Markdown link checks passed. No Python source changed in this batch. -Deploy the orchestrator update to activate these changes. This batch needs no additional IAM/bootstrap change or agent-image update. The receipt is internal task-table data, not a new public task field. The service emulator proves our retry behavior; AWS token retention/conflicts and unknown-ID cleanup still need live evidence. The shared finalizer's atomic capacity-slot release and the remaining P2/P3 work stay unchecked above. +Deploy the orchestrator update to activate these changes. This batch needs no additional IAM/bootstrap change or agent-image update. The receipt is internal task-table data, not a new public task field. The service emulator proves our retry behavior; AWS token retention/conflicts and unknown-ID cleanup still need live evidence. At the close of the third batch, atomic capacity-slot release and the remaining P2/P3 work were still open. + +Fourth prerequisite batch completed locally on 2026-09-13: + +| Commit | Completed work | Proof | +|---|---|---| +| `0bb95243` | Task-owned atomic capacity reservations; one admission owner; shared finalizer/stranded release; guarded queue restoration; revision-checked repair including approval waits; scoped IAM and stale counter-doc cleanup | The old replay regression reduced two seats to zero. Fifteen tests against DynamoDB Local now verify real transaction conditions, competing writers, lost committed replies, empty-counter races and stale-revision rejection. Unit and construct tests cover handler wiring and permission scope. | + +Final checks for the fourth batch: CDK ESLint and compilation passed. **153 handler suites / 3,600 tests** passed with the local integration suite enabled; **15** of those tests used DynamoDB Local. The focused run passed **323 tests**, including the two relevant IAM construct suites; those counts overlap. Documentation sync, the **77-page** Astro build and Markdown link checks passed. The broad handler run exited successfully after the previously observed delayed-exit warning; the focused run exited normally. The temporary database container was stopped and removed. + +No AWS resources changed. Activating this batch requires the coordinated deployment/drain procedure in [capacity verification](./645-capacity-reservations.md), including the reconciler's task-update permission and upload confirmation's narrowed counter access. No agent source/image or bootstrap bundle changed. Coordinator metadata protection is newly tracked in 1G; internal reservation fields currently share an agent-writable row. ## The result we want @@ -154,7 +166,19 @@ Resolve #810 by exposing a useful structured failure signal for CloudWatch write ### 1F. Make concurrency release safe across replay -`finalizeTask` currently decrements the user's counter directly after terminal events, without atomically recording a per-task release. A crash after the decrement but before the durable checkpoint can repeat it. Reproduce that exact failure locally, then make the release marker and counter change one conditional transaction. Cover normal completion, start failure, cancellation, timeout, competing finalizers, retries and an already-zero counter. Preserve admission-queue behavior. The new MicroVM start path avoids an additional early release; it does not close this shared finalizer window. +**Implemented locally:** `task-concurrency.ts` saves a per-task reservation together with the user counter change in one transaction. Repeated acquisition reuses the held reservation; release requires a terminal task and changes `held → released` atomically with the decrement. Missing/released markers never decrement another task's count. This covers cooperating platform writers and crash replay. + +The regression reproduced two finalizer executions reducing two occupied seats to zero, instead of leaving the other task's seat occupied. DynamoDB Local tests now exercise real transaction conditions for that replay, concurrent admissions/finalizers, lost committed responses, early failure, cancellation, approval waits, empty counters and a racing new admission. The normal finalizer, early failure path and stranded cleaner share release. Upload confirmation only submits and reads capacity; the orchestrator reserves once. Queue restoration cannot requeue a task that acquired a reservation after an uncertain invoke. + +Every counter change carries a fresh revision. Scheduled repair strongly scans the base counter/task tables, compares the saved revision before replacing a count, and completes abandoned terminal releases. It includes approval waits. An increment/decrement with no net count change still invalidates an old scan. Incomplete scans install no partial result. Ambiguous older active tasks prevent guessing a replacement count. + +**Deployment gate:** pause admissions and drain old executions, update all counter writers and the reconciler's scoped task-update permission, reconcile after legacy tasks settle, then reopen admissions. Rollback also requires draining. Verify scan duration/read capacity at deployment scale and the mixed-version/drain procedure in AWS. The local simulator does not prove deployed IAM or cloud-scale behavior. See [capacity verification](./645-capacity-reservations.md). + +### 1G. Protect coordinator-owned metadata + +The agent's task-scoped role restricts **which row** it can change, but currently permits replacement/deletion and unrestricted attribute updates within its own row. `microvm_start` and `concurrency_slot` are internal task-table fields; omitting them from API responses does not protect their storage. A compromised agent must not be able to rewrite the coordinator's start identity or revive a released reservation. + +Choose coordinator-only storage or carefully constrained agent write permissions, inventory every task-row writer, and test forged/replaced/deleted rows. Preserve legitimate agent progress/status updates and all backend identity tags. Treat this as a separate security prerequisite; the replay tests in 1D/1F do not establish a hostile-agent guarantee. ## 2. Nest infrastructure if adopting the split diff --git a/docs/verification/645-p3-readiness-review.md b/docs/verification/645-p3-readiness-review.md index 267e8eabf..6b45cc55a 100644 --- a/docs/verification/645-p3-readiness-review.md +++ b/docs/verification/645-p3-readiness-review.md @@ -6,7 +6,9 @@ This review covers the existing MicroVM implementation, related P2 follow-ups, c **Implementation update (2026-09-13):** subsequent prerequisite work fixes #841 thread isolation, #817 coordinator payload-deletion permission, terminal-failure classification, closed S3-stream handling, and the approval-resume heartbeat race locally. Real bad-byte route tests and ARN contract guards are also added. Trusted deployment identity, task-scoped payload reads and live verification remain open. The findings below preserve the reviewed baseline; see [implementation progress](./645-p3-implementation-plan.md#implementation-progress) for commits, checks and remaining work. -Further local work adds durable MicroVM start receipts, stable request tokens and bounded handle recovery. AWS token-retention/conflict behavior and unknown-ID cleanup remain live verification gates. Reviewing crash replay also exposed a shared finalizer risk: its counter decrement lacks an atomic per-task release marker. That source-level finding is now a prerequisite in the implementation plan; it has not yet been reproduced in AWS or fixed. +Further local work adds durable MicroVM start receipts, stable request tokens and bounded handle recovery. AWS token-retention/conflict behavior and unknown-ID cleanup remain live verification gates. The shared finalizer's repeated-decrement risk was subsequently reproduced and fixed locally with task-owned reservation transactions, unified counter writers and revision-guarded reconciliation, including approval waits. DynamoDB Local exercises the real transaction conditions; deployed IAM, scan scale and upgrade/drain behavior still need AWS verification. + +The reservation review also found a separate trust boundary: the agent role can write/replace/delete its own task row, including coordinator metadata. Internal API fields are not protected from that writer by their naming or TypeScript types. The plan tracks coordinator-only storage or constrained agent writes as a security prerequisite. The replay fix assumes cooperating writers; it is not a hostile-agent isolation claim. ## Start here: the pieces in plain language From 5ea6727046c916560a175f03127effe5da73814f Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Sun, 13 Sep 2026 16:42:10 -0400 Subject: [PATCH 014/149] fix(security): protect coordinator task metadata from agents (#645) --- agent/src/aws_session.py | 7 +- agent/src/task_state.py | 96 +------- agent/tests/test_task_state.py | 212 +++++++++++------- cdk/src/constructs/agent-session-role.ts | 86 ++++++- .../agent-task-write-attributes.json | 28 +++ cdk/src/constructs/ecs-agent-cluster.ts | 10 +- cdk/src/stacks/agent.ts | 11 +- .../constructs/agent-session-role.test.ts | 68 +++++- cdk/test/constructs/ecs-agent-cluster.test.ts | 64 +++--- .../constructs/lambda-microvm-compute.test.ts | 9 +- cdk/test/stacks/agent.test.ts | 16 ++ docs/design/SECURITY.md | 4 +- .../src/content/docs/architecture/Security.md | 4 +- .../verification/645-capacity-reservations.md | 2 +- docs/verification/645-coordinator-metadata.md | 69 ++++++ .../645-p3-implementation-plan.md | 18 +- docs/verification/645-p3-readiness-review.md | 2 +- 17 files changed, 462 insertions(+), 244 deletions(-) create mode 100644 cdk/src/constructs/agent-task-write-attributes.json create mode 100644 docs/verification/645-coordinator-metadata.md diff --git a/agent/src/aws_session.py b/agent/src/aws_session.py index 0f2e8f079..89e926546 100644 --- a/agent/src/aws_session.py +++ b/agent/src/aws_session.py @@ -8,8 +8,11 @@ uses the resulting short-lived, tag-scoped credentials. The SessionRole's IAM policy self-constrains via ``aws:PrincipalTag/*`` conditions (``dynamodb:LeadingKeys`` on ``task_id`` for the task tables, an S3 prefix -condition on ``user_id`` for the trace bucket), so a compromised session can -only reach its own task's data — not other tenants'. +condition on ``user_id`` for the trace bucket). Existing scoped credentials +can only reach their tagged task. The compute role chooses those tags: a +compromised worker with ambient credentials is a separate trust boundary. +Task updates are additionally restricted to reporting/approval attributes; +whole-row replacement and coordinator metadata writes are not granted. Two properties matter for correctness: diff --git a/agent/src/task_state.py b/agent/src/task_state.py index 980ff7da1..a9ac64bd9 100644 --- a/agent/src/task_state.py +++ b/agent/src/task_state.py @@ -1,8 +1,9 @@ """Best-effort task state persistence to DynamoDB. -All writes are wrapped in try/except so a DynamoDB outage never breaks the -agent pipeline. When the TASK_TABLE_NAME environment variable is unset, all -operations are no-ops. +Progress/status writes are best-effort; approval transactions fail closed. +The coordinator creates task records and owns compute identity and capacity +reservations. This module only reads tasks and updates reporting/approval fields; +its allowed attributes are pinned by the CDK agent-task-write-attributes contract. """ import os @@ -79,30 +80,6 @@ def _build_logs_url(task_id: str) -> str | None: ) -def write_submitted( - task_id: str, repo_url: str = "", issue_number: str = "", task_description: str = "" -) -> None: - """Record a task as SUBMITTED (called from the invoke script or server).""" - try: - table = _get_table() - if table is None: - return - item = { - "task_id": task_id, - "status": "SUBMITTED", - "created_at": _now_iso(), - } - if repo_url: - item["repo_url"] = repo_url - if issue_number: - item["issue_number"] = issue_number - if task_description: - item["task_description"] = task_description - table.put_item(Item=item) - except Exception as e: - log("WARN", f"[task_state] write_submitted failed (best-effort): {e}") - - def write_heartbeat(task_id: str) -> None: """Update ``agent_heartbeat_at`` while the task is RUNNING (orchestrator crash detection).""" try: @@ -127,63 +104,6 @@ def write_heartbeat(task_id: str) -> None: log("WARN", f"[task_state] write_heartbeat failed (best-effort): {type(e).__name__}: {e}") -def write_session_info(task_id: str, session_id: str, agent_runtime_arn: str) -> None: - """Record session_id + agent_runtime_arn on a pre-RUNNING task. - - The orchestrator Lambda writes these fields on the HYDRATING → RUNNING - transition so ``cancel-task`` can ``StopRuntimeSession`` on the right - runtime and operators can correlate a stuck task to a specific AgentCore - session. Currently only the orchestrator calls this; the agent-side - invocation path inherits the fields from the orchestrator's payload. - - Idempotent + best-effort. Skips silently if the task is already - past SUBMITTED/HYDRATING (concurrent transition winning is fine). - """ - if not task_id or (not session_id and not agent_runtime_arn): - return - try: - table = _get_table() - if table is None: - return - set_parts: list[str] = [] - expr_values: dict = { - ":submitted": "SUBMITTED", - ":hydrating": "HYDRATING", - } - if session_id: - set_parts.append("session_id = :sid") - expr_values[":sid"] = session_id - if agent_runtime_arn: - set_parts.append("agent_runtime_arn = :arn") - set_parts.append("compute_type = :ct") - set_parts.append("compute_metadata = :cm") - expr_values[":arn"] = agent_runtime_arn - expr_values[":ct"] = "agentcore" - expr_values[":cm"] = {"runtimeArn": agent_runtime_arn} - if not set_parts: - return - table.update_item( - Key={"task_id": task_id}, - UpdateExpression="SET " + ", ".join(set_parts), - ConditionExpression="#s IN (:submitted, :hydrating)", - ExpressionAttributeNames={"#s": "status"}, - ExpressionAttributeValues=expr_values, - ) - except Exception as e: - from botocore.exceptions import ClientError - - if ( - isinstance(e, ClientError) - and e.response.get("Error", {}).get("Code") == "ConditionalCheckFailedException" - ): - # Task already advanced — concurrent legitimate transition wins. - return - log( - "WARN", - f"[task_state] write_session_info failed (best-effort): {type(e).__name__}: {e}", - ) - - def write_running(task_id: str) -> None: """Transition a task to RUNNING (called at agent start). @@ -504,11 +424,9 @@ def get_task(task_id: str) -> dict | None: # --------------------------------------------------------------------------- # # ``TaskApprovalsTable`` and the AWAITING_APPROVAL status transitions are -# provisioned by the CDK stack. The agent-side helpers below are written to -# that contract and exposed so the ``pre_tool_use_hook`` can be implemented + -# unit-tested (via mocked boto3 clients); once the stack sets -# ``TASK_APPROVALS_TABLE_NAME`` + grants IAM, the same helpers start making -# real DDB calls with no further code change on the agent side. +# provisioned by the CDK stack. The ``pre_tool_use_hook`` uses these helpers +# with task-scoped credentials. Transactions authorize each item separately: +# Put on the approvals table and attribute-restricted Update on TaskTable. # # Primitives exposed: # - ``transact_write_approval_request`` — atomic Put(TaskApprovals) + diff --git a/agent/tests/test_task_state.py b/agent/tests/test_task_state.py index 04bc8aa25..8ce542240 100644 --- a/agent/tests/test_task_state.py +++ b/agent/tests/test_task_state.py @@ -1,5 +1,9 @@ -"""Unit tests for pure functions in task_state.py.""" +"""Task persistence behavior and the agent's IAM write contract.""" +import ast +import json +import re +from pathlib import Path from unittest.mock import MagicMock import pytest @@ -8,6 +12,136 @@ from task_state import TaskFetchError, _build_logs_url, _now_iso +class TestAgentWriteContract: + def test_current_task_writers_fit_the_deployed_attribute_allowlist(self, monkeypatch): + """Exercise real writers; detect a new field before IAM rejects it live. + + This checks request/contract compatibility, not AWS IAM enforcement. + """ + table = MagicMock() + client = MagicMock() + monkeypatch.setattr(task_state, "_get_table", lambda: table) + monkeypatch.setenv("TASK_TABLE_NAME", "Tasks") + monkeypatch.setenv("TASK_APPROVALS_TABLE_NAME", "Approvals") + monkeypatch.setenv("AWS_REGION", "us-east-1") + monkeypatch.setenv("LOG_GROUP_NAME", "/test") + + task_state.write_running("t1") + task_state.write_heartbeat("t1") + task_state.write_terminal( + "t1", + "COMPLETED", + { + "pr_url": "https://example.com/pr/1", + "error": "example", + "cost_usd": 1, + "duration_s": 10, + "turns": 3, + "turns_attempted": 3, + "turns_completed": 2, + "prompt_version": "v1", + "memory_written": True, + "build_passed": True, + "lint_passed": True, + "code_changed": True, + "head_sha": "abc", + "answer_text": "done", + "otel_trace_id": "trace", + "trace_s3_uri": "s3://b/trace", + "artifact_uri": "s3://b/artifact", + }, + ) + assert task_state.write_trace_uri_conditional("t1", "s3://b/trace") + task_state.transact_write_approval_request( + "t1", + "r1", + { + "task_id": "t1", + "request_id": "r1", + "status": "PENDING", + "tool_name": "Bash", + "tool_input_preview": "example", + "tool_input_sha256": "a" * 64, + "reason": "approval required", + "severity": "high", + "matching_rule_ids": ["rule1"], + "created_at": "2026-09-13T00:00:00Z", + "timeout_s": 300, + "ttl": 1800000000, + "user_id": "u1", + "repo": "owner/repo", + }, + client=client, + ) + task_state.transact_resume_from_approval("t1", "r1", client=client) + assert task_state.increment_approval_gate_count_in_ddb("t1", client=client) + + requests = [c.kwargs for c in table.update_item.call_args_list] + requests += [c.kwargs for c in client.update_item.call_args_list] + for call in client.transact_write_items.call_args_list: + for item in call.kwargs["TransactItems"]: + action, request = next(iter(item.items())) + if request["TableName"] == "Tasks": + assert action == "Update" # Never Put/Delete the task row. + requests.append(request) + assert len(requests) == 7 + table.put_item.assert_not_called() + table.delete_item.assert_not_called() + + contract = Path(__file__).resolve().parents[2] / ( + "cdk/src/constructs/agent-task-write-attributes.json" + ) + allowed = set(json.loads(contract.read_text())) + seen: set[str] = set() + for request in requests: + # Current writers use flat attributes. Resolve aliases and ignore + # value placeholders/operators/functions in their DDB expressions. + expression = request["UpdateExpression"] + " " + request.get("ConditionExpression", "") + expression = re.sub(r":[A-Za-z0-9_]+", "", expression) + for alias, name in request.get("ExpressionAttributeNames", {}).items(): + expression = expression.replace(alias, name) + names = set(re.findall(r"[A-Za-z_][A-Za-z0-9_]*", expression)) + names -= {"SET", "REMOVE", "ADD", "IN", "AND", "attribute_not_exists"} + names |= set(request["Key"]) + assert names <= allowed, ( + f"Task writer needs a reviewed IAM contract update: {names - allowed}" + ) + seen |= names + # No stale writable attribute may linger after its writer is removed. + assert seen == allowed + + def test_write_inventory_requires_review_when_a_new_writer_is_added(self): + tree = ast.parse(Path(task_state.__file__).read_text()) + writers = { + node.name + for node in tree.body + if isinstance(node, ast.FunctionDef) + and any( + isinstance(child, ast.Call) + and isinstance(child.func, ast.Attribute) + and child.func.attr + in { + "update_item", + "put_item", + "delete_item", + "transact_write_items", + "batch_writer", + } + for child in ast.walk(node) + ) + } + assert writers == { + "write_running", + "write_heartbeat", + "write_terminal", + "write_trace_uri_conditional", + "transact_write_approval_request", + "transact_resume_from_approval", + "increment_approval_gate_count_in_ddb", + "best_effort_update_approval_status", # Writes only the supporting approvals table. + } + + class TestNowIso: def test_format(self): result = _now_iso() @@ -97,82 +231,6 @@ def get_item(self, Key): assert "ProvisionedThroughputExceededException" in str(exc_info.value) -class TestWriteSessionInfo: - """Rev-5 OBS-4: interactive path writes session_id + agent_runtime_arn.""" - - def test_writes_session_id_and_arn(self, monkeypatch): - calls: list[dict] = [] - - class _FakeTable: - def update_item(self, **kwargs): - calls.append(kwargs) - - monkeypatch.setattr(task_state, "_get_table", lambda: _FakeTable()) - - task_state.write_session_info( - "t-interactive", - "sess-abc123", - "arn:aws:bedrock-agentcore:us-east-1:123:runtime/jwt-xyz", - ) - - assert len(calls) == 1 - call = calls[0] - assert call["Key"] == {"task_id": "t-interactive"} - assert "session_id = :sid" in call["UpdateExpression"] - assert "agent_runtime_arn = :arn" in call["UpdateExpression"] - assert "compute_type = :ct" in call["UpdateExpression"] - assert "compute_metadata = :cm" in call["UpdateExpression"] - values = call["ExpressionAttributeValues"] - assert values[":sid"] == "sess-abc123" - assert values[":arn"] == "arn:aws:bedrock-agentcore:us-east-1:123:runtime/jwt-xyz" - assert values[":ct"] == "agentcore" - assert values[":cm"] == { - "runtimeArn": "arn:aws:bedrock-agentcore:us-east-1:123:runtime/jwt-xyz" - } - - def test_noop_when_both_empty(self, monkeypatch): - calls: list[dict] = [] - - class _FakeTable: - def update_item(self, **kwargs): - calls.append(kwargs) - - monkeypatch.setattr(task_state, "_get_table", lambda: _FakeTable()) - - task_state.write_session_info("t-empty", "", "") - assert calls == [] - - def test_skips_silently_when_task_already_advanced(self, monkeypatch): - from botocore.exceptions import ClientError - - class _FakeTable: - def update_item(self, **kwargs): - raise ClientError( - {"Error": {"Code": "ConditionalCheckFailedException"}}, - "UpdateItem", - ) - - monkeypatch.setattr(task_state, "_get_table", lambda: _FakeTable()) - - # Must NOT raise — the conditional failure is expected when the - # task has already transitioned past SUBMITTED/HYDRATING. - task_state.write_session_info("t-raced", "sess-x", "arn:x") - - def test_writes_only_session_when_arn_missing(self, monkeypatch): - calls: list[dict] = [] - - class _FakeTable: - def update_item(self, **kwargs): - calls.append(kwargs) - - monkeypatch.setattr(task_state, "_get_table", lambda: _FakeTable()) - - task_state.write_session_info("t-partial", "sess-only", "") - assert len(calls) == 1 - assert "session_id = :sid" in calls[0]["UpdateExpression"] - assert "agent_runtime_arn" not in calls[0]["UpdateExpression"] - - class TestWriteRunningMaintainsStatusCreatedAt: """Regression guard: ``write_running`` must rewrite ``status_created_at`` so the ``UserStatusIndex`` GSI sort key reflects the current status. diff --git a/cdk/src/constructs/agent-session-role.ts b/cdk/src/constructs/agent-session-role.ts index 1602b734d..2d4d56cc6 100644 --- a/cdk/src/constructs/agent-session-role.ts +++ b/cdk/src/constructs/agent-session-role.ts @@ -24,6 +24,53 @@ import * as iam from 'aws-cdk-lib/aws-iam'; import * as s3 from 'aws-cdk-lib/aws-s3'; import { NagSuppressions } from 'cdk-nag'; import { Construct } from 'constructs'; +import agentTaskWriteAttributes from './agent-task-write-attributes.json'; + +/** + * Task reporting may update only the attributes written by task_state.py. + * Keep whole-row replacement/deletion and coordinator metadata out of this + * grant. DynamoDB evaluates each transaction item using its item action, so + * approval UpdateItem operations receive the same restriction. + * + * This protects writes, not reads: agents may read their complete task record. + * The JSON list is also checked against actual Python writer requests in tests. + * Used for both scoped sessions and the legacy ECS direct-grant fallback. + */ +export function grantAgentTaskTableAccess( + table: dynamodb.ITable, + grantee: iam.IGrantable, + taskScoped: boolean, +): void { + const leadingKeys = taskScoped + ? { 'dynamodb:LeadingKeys': ['${aws:PrincipalTag/task_id}'] } + : {}; + grantee.grantPrincipal.addToPrincipalPolicy(new iam.PolicyStatement({ + actions: ['dynamodb:GetItem', 'dynamodb:BatchGetItem', 'dynamodb:Query', 'dynamodb:ConditionCheckItem'], + resources: [table.tableArn], + ...(taskScoped ? { + conditions: { + 'ForAllValues:StringEquals': leadingKeys, + 'Null': { 'dynamodb:LeadingKeys': 'false' }, + }, + } : {}), + })); + grantee.grantPrincipal.addToPrincipalPolicy(new iam.PolicyStatement({ + actions: ['dynamodb:UpdateItem'], + resources: [table.tableArn], + conditions: { + 'ForAllValues:StringEquals': { + ...leadingKeys, + 'dynamodb:Attributes': agentTaskWriteAttributes, + }, + // ForAllValues alone also matches an absent context key. Require the + // attribute list (and session key when scoped) to be present. + 'Null': { + 'dynamodb:Attributes': 'false', + ...(taskScoped ? { 'dynamodb:LeadingKeys': 'false' } : {}), + }, + }, + })); +} /** S3 key prefixes the agent writes/reads, scoped per tenant. */ const TRACE_KEY_PREFIX = 'traces'; @@ -36,17 +83,24 @@ const ARTIFACT_KEY_PREFIX = 'artifacts'; */ export interface AgentSessionRoleProps { /** - * Compute roles (AgentCore Runtime ExecutionRole and/or ECS Fargate task - * role) permitted to assume this SessionRole and pass session tags. These - * are the only principals trusted to mint scoped credentials, so they bound - * the trust surface. Both run the same trusted agent code, which sources the + * Compute roles (AgentCore Runtime, ECS Fargate or Lambda MicroVM) permitted + * to assume this SessionRole and pass session tags. These principals mint + * scoped credentials. The agent code sources the * `{user_id, repo, task_id}` tag values from the resolved TaskConfig. */ readonly assumingRoles: iam.IRole[]; /** - * The four task-scoped DynamoDB tables, all partitioned by `task_id`. The - * SessionRole receives item-level access constrained by a + * The main task table: own-task reads and attribute-scoped reporting updates. + * Task creation/deletion, owner identity, compute handles, start receipts and + * capacity reservations belong to the coordinator. + */ + readonly taskTable: dynamodb.ITable; + + /** + * Supporting task-scoped tables (events, approvals, nudges), all partitioned + * by `task_id`. Do not include taskTable here: that would bypass its write + * restriction. The SessionRole receives item-level access constrained by a * `dynamodb:LeadingKeys` condition on `aws:PrincipalTag/task_id`, so a * session can only touch its own task's rows. Order is irrelevant. */ @@ -98,12 +152,14 @@ export interface AgentSessionRoleProps { * - S3 trace writes and attachment reads are scoped to the * `/${aws:PrincipalTag/user_id}/` object prefix. * - * The result: a compromised agent session can reach only its own task's data, - * not other tenants' — enforced at the IAM layer rather than in application - * code. Backend-agnostic: the same role serves agents booted under either the - * AgentCore Runtime execution role or the ECS Fargate task role. + * Existing session credentials are limited to their tagged task. Compute roles + * choose these tags when assuming this role; the trust policy does not bind + * those choices to a particular task. This is not an isolation claim for a + * compromised worker that can obtain ambient compute credentials. + * TaskTable writes additionally exclude coordinator-owned attributes and + * whole-row replacement/deletion. All three compute backends share this role. * - * CloudWatch Logs remains on the compute role (shared, non-tenant access). The + * CloudWatch Logs remains on the compute role (shared access). The * compute role *also* keeps `InvokeModel`; this role adds a parallel, session- * tagged Bedrock grant (#215) used by the Claude Code subprocess for cost * attribution. Long-task safety on the 1-hour-capped chained session is handled @@ -136,6 +192,9 @@ export class AgentSessionRole extends Construct { 'AgentSessionRole requires at least one assuming role (the compute role[s] that mint scoped credentials)', ); } + if (props.taskScopedTables.some((table) => table.tableArn === props.taskTable.tableArn)) { + throw new Error('taskTable must not appear in taskScopedTables; it requires restricted writes'); + } const [firstAssumingRole] = props.assumingRoles; @@ -153,7 +212,9 @@ export class AgentSessionRole extends Construct { maxSessionDuration: Duration.hours(1), }); - // --- DynamoDB: item access gated by task_id leading-key --- + grantAgentTaskTableAccess(props.taskTable, this.role, true); + + // --- Supporting tables: item access gated by task_id leading-key --- // One statement per table keeps the resource ARNs explicit. The condition // requires the request's partition key (task_id) to equal the session's // task_id tag. ForAllValues is required by DynamoDB for LeadingKeys. @@ -166,6 +227,7 @@ export class AgentSessionRole extends Construct { 'ForAllValues:StringEquals': { 'dynamodb:LeadingKeys': ['${aws:PrincipalTag/task_id}'], }, + 'Null': { 'dynamodb:LeadingKeys': 'false' }, }, }), ); diff --git a/cdk/src/constructs/agent-task-write-attributes.json b/cdk/src/constructs/agent-task-write-attributes.json new file mode 100644 index 000000000..ace3330da --- /dev/null +++ b/cdk/src/constructs/agent-task-write-attributes.json @@ -0,0 +1,28 @@ +[ + "task_id", + "status", + "status_created_at", + "started_at", + "completed_at", + "logs_url", + "agent_heartbeat_at", + "awaiting_approval_request_id", + "approval_gate_count", + "pr_url", + "error_message", + "cost_usd", + "duration_s", + "turns", + "turns_attempted", + "turns_completed", + "prompt_version", + "memory_written", + "build_passed", + "lint_passed", + "code_changed", + "head_sha", + "answer_text", + "otel_trace_id", + "trace_s3_uri", + "artifact_uri" +] diff --git a/cdk/src/constructs/ecs-agent-cluster.ts b/cdk/src/constructs/ecs-agent-cluster.ts index 450d5e272..48d55105b 100644 --- a/cdk/src/constructs/ecs-agent-cluster.ts +++ b/cdk/src/constructs/ecs-agent-cluster.ts @@ -29,7 +29,7 @@ import * as secretsmanager from 'aws-cdk-lib/aws-secretsmanager'; import { NagSuppressions } from 'cdk-nag'; import { Construct, type Node } from 'constructs'; import { AgentMemory } from './agent-memory'; -import { AgentSessionRole } from './agent-session-role'; +import { AgentSessionRole, grantAgentTaskTableAccess } from './agent-session-role'; import { resolveBedrockGeoRegion, resolveBedrockModelIds } from './bedrock-models'; import { LinearIdentityVault } from './linear-identity-vault'; import { buildAppId } from './solution-ua-aspect'; @@ -525,12 +525,10 @@ export class EcsAgentCluster extends Construct { if (props.agentSessionRole) { props.agentSessionRole.admitComputeRole(taskRole); } else { - props.taskTable.grantReadWriteData(taskRole); + grantAgentTaskTableAccess(props.taskTable, taskRole, false); props.taskEventsTable.grantReadWriteData(taskRole); } - // UserConcurrencyTable is user-scoped (not task_id leading-key-able) and is - // touched by the reconciler/orchestrator path; keep it on the task role. - props.userConcurrencyTable.grantReadWriteData(taskRole); + // Capacity counters are coordinator-owned. The agent never accesses them. // Secrets Manager read for GitHub token (read once at startup, before the // agent assumes the SessionRole — stays on the task role). @@ -681,7 +679,7 @@ export class EcsAgentCluster extends Construct { NagSuppressions.addResourceSuppressions(taskRole, [ { id: 'AwsSolutions-IAM5', - reason: 'DynamoDB index/* wildcards from CDK grantReadWriteData (UserConcurrencyTable, and task tables only when no SessionRole is wired); Secrets Manager wildcards from CDK grantRead (GitHub token) and the bgagent-linear-oauth-*/bgagent-jira-oauth-* prefix grant (ABCA-488 — per-workspace channel OAuth tokens are created by the CLI at setup, name unknown at synth, GetSecretValue only); CloudWatch Logs wildcards from CDK grantWrite; S3 object/* wildcard from CDK grantRead on the ECS payload bucket (read-only, scoped to that bucket — #502). Bedrock InvokeModel is scoped to explicit model/inference-profile ARNs (no wildcard resource). ec2:DescribeAvailabilityZones requires Resource:* (EC2 describe actions have no resource-level scoping) — read-only, no mutation/data access; needed so a CDK target repo\'s `cdk synth` build gate can resolve AZ context on a fresh clone (ECS-parity, no cdk.context.json cache in the container).', + reason: 'DynamoDB index/* wildcards from the legacy TaskEventsTable grant when no SessionRole is wired (TaskTable allows only reporting updates; the worker has no UserConcurrency access); Secrets Manager wildcards from CDK grantRead (GitHub token) and the bgagent-linear-oauth-*/bgagent-jira-oauth-* prefix grant (ABCA-488 — per-workspace channel OAuth tokens are created by the CLI at setup, name unknown at synth, GetSecretValue only); CloudWatch Logs wildcards from CDK grantWrite; S3 object/* wildcard from CDK grantRead on the ECS payload bucket (read-only, scoped to that bucket — #502). Bedrock InvokeModel is scoped to explicit model/inference-profile ARNs (no wildcard resource). ec2:DescribeAvailabilityZones requires Resource:* (EC2 describe actions have no resource-level scoping) — read-only, no mutation/data access; needed so a CDK target repo\'s `cdk synth` build gate can resolve AZ context on a fresh clone (ECS-parity, no cdk.context.json cache in the container).', }, { id: 'AwsSolutions-ECS2', diff --git a/cdk/src/stacks/agent.ts b/cdk/src/stacks/agent.ts index 20cb1cd04..67bf1d004 100644 --- a/cdk/src/stacks/agent.ts +++ b/cdk/src/stacks/agent.ts @@ -738,12 +738,9 @@ export class AgentStack extends Stack { // {user_id, repo, task_id}, and that role carries the tenant-data grants // constrained by aws:PrincipalTag conditions. The runtime role keeps only // non-tenant / shared access: - // - UserConcurrencyTable: user-scoped counter (agent path does not write - // it today; left here for the reconciler/orchestrator parity). // - GitHub PAT secret: read once at startup, before the agent assumes the // SessionRole. // - CloudWatch Logs + AgentCore Memory: shared/non-tenant. - userConcurrencyTable.table.grantReadWriteData(runtime); githubTokenSecret.grantRead(runtime); applicationLogGroup.grantWrite(runtime); agentMemory.grantReadWrite(runtime); @@ -781,15 +778,17 @@ export class AgentStack extends Stack { // --- Per-task SessionRole --- // Holds the tenant-data grants (the four task_id-partitioned tables, plus // per-user-prefixed trace writes and attachment reads), each constrained - // by aws:PrincipalTag conditions so a compromised session reaches only its - // own task's data. The agent assumes this with refreshable credentials + // by aws:PrincipalTag conditions for the existing session credentials. + // The compute role chooses the tags; that choice is not independently + // authenticated by this trust policy. Main-task writes are restricted to + // reporting attributes. The agent assumes this with refreshable credentials // (1h role-chaining cap, tasks run to 8h). Trust admits the runtime // ExecutionRole as the assuming principal; the ECS task role is added in // the ECS block below when that backend is enabled. const agentSessionRole = new AgentSessionRole(this, 'AgentSessionRole', { assumingRoles: [runtime.role], + taskTable: taskTable.table, taskScopedTables: [ - taskTable.table, taskEventsTable.table, taskApprovalsTable.table, taskNudgesTable.table, diff --git a/cdk/test/constructs/agent-session-role.test.ts b/cdk/test/constructs/agent-session-role.test.ts index 676b34f65..f1b2c818a 100644 --- a/cdk/test/constructs/agent-session-role.test.ts +++ b/cdk/test/constructs/agent-session-role.test.ts @@ -24,6 +24,7 @@ import * as dynamodb from 'aws-cdk-lib/aws-dynamodb'; import * as iam from 'aws-cdk-lib/aws-iam'; import * as s3 from 'aws-cdk-lib/aws-s3'; import { AgentSessionRole } from '../../src/constructs/agent-session-role'; +import taskWriteAttributes from '../../src/constructs/agent-task-write-attributes.json'; function createStack() { const app = new App(); @@ -51,8 +52,8 @@ function createStack() { const sessionRole = new AgentSessionRole(stack, 'AgentSessionRole', { assumingRoles: [computeRole], + taskTable, taskScopedTables: [ - taskTable, taskEventsTable, taskApprovalsTable, taskNudgesTable, @@ -107,14 +108,12 @@ describe('AgentSessionRole construct', () => { const actions = Array.isArray(s.Action) ? s.Action : [s.Action]; return actions.some((a: string) => a.startsWith('dynamodb:')); }); - // Four task-scoped tables → four conditioned statements. - expect(ddbStatements).toHaveLength(4); + // Main task read/update grants plus three supporting tables. + expect(ddbStatements).toHaveLength(5); for (const s of ddbStatements) { - expect(s.Condition).toEqual({ - 'ForAllValues:StringEquals': { - 'dynamodb:LeadingKeys': ['${aws:PrincipalTag/task_id}'], - }, - }); + expect(s.Condition['ForAllValues:StringEquals']['dynamodb:LeadingKeys']) + .toEqual(['${aws:PrincipalTag/task_id}']); + expect(s.Condition.Null['dynamodb:LeadingKeys']).toBe('false'); const actions = Array.isArray(s.Action) ? s.Action : [s.Action]; // Scan must NOT be granted — it ignores leading-keys. expect(actions).not.toContain('dynamodb:Scan'); @@ -140,6 +139,50 @@ describe('AgentSessionRole construct', () => { expect(tracePut).toBeDefined(); }); + test('task records cannot be replaced, deleted or updated without an attribute allowlist', () => { + const policy = Object.entries(template.findResources('AWS::IAM::Policy')) + .find(([id]) => id.includes('AgentSessionRole'))![1]; + const statements = policy.Properties.PolicyDocument.Statement.filter( + (s: { Resource: unknown }) => JSON.stringify(s.Resource).includes('TaskTable'), + ); + expect(statements).toHaveLength(2); + for (const s of statements) { + const actions: string[] = Array.isArray(s.Action) ? s.Action : [s.Action]; + expect(actions.every((action) => [ + 'dynamodb:GetItem', 'dynamodb:BatchGetItem', 'dynamodb:Query', + 'dynamodb:ConditionCheckItem', 'dynamodb:UpdateItem', + ].includes(action))).toBe(true); + if (actions.includes('dynamodb:UpdateItem')) { + const attrs = s.Condition?.['ForAllValues:StringEquals']?.['dynamodb:Attributes']; + expect(attrs).toEqual(taskWriteAttributes); + expect(s.Condition.Null['dynamodb:Attributes']).toBe('false'); + for (const protectedAttribute of [ + 'microvm_start', 'concurrency_slot', 'user_id', 'created_at', + 'session_id', 'compute_type', 'compute_metadata', 'agent_runtime_arn', + ]) { + expect(attrs).not.toContain(protectedAttribute); + } + } + } + }); + + test('rejects granting unrestricted supporting-table access to the main task table', () => { + const stack = new Stack(new App(), 'DuplicateTable'); + const table = new dynamodb.Table(stack, 'Tasks', { + partitionKey: { name: 'task_id', type: dynamodb.AttributeType.STRING }, + }); + const computeRole = new iam.Role(stack, 'Compute', { + assumedBy: new iam.ServicePrincipal('ecs-tasks.amazonaws.com'), + }); + expect(() => new AgentSessionRole(stack, 'Session', { + assumingRoles: [computeRole], + taskTable: table, + taskScopedTables: [table], + traceArtifactsBucket: new s3.Bucket(stack, 'Traces'), + attachmentsBucket: new s3.Bucket(stack, 'Attachments'), + })).toThrow('taskTable must not appear in taskScopedTables'); + }); + test('S3 artifact writes are scoped to the per-task_id prefix (#248 Phase 3)', () => { const policies = template.findResources('AWS::IAM::Policy'); const sessionPolicy = Object.entries(policies).find(([id]) => @@ -208,7 +251,8 @@ describe('AgentSessionRole construct', () => { }); new AgentSessionRole(stack, 'SR', { assumingRoles: [computeRole], - taskScopedTables: [table], + taskTable: table, + taskScopedTables: [], traceArtifactsBucket: new s3.Bucket(stack, 'TB'), attachmentsBucket: new s3.Bucket(stack, 'AB'), invokableModels: [model], @@ -231,8 +275,7 @@ describe('AgentSessionRole construct', () => { }); test('omitting invokableModels grants no bedrock action (isolated tests)', () => { - const { template: t } = createStack(); - const policies = t.findResources('AWS::IAM::Policy'); + const policies = template.findResources('AWS::IAM::Policy'); const sessionPolicy = Object.entries(policies).find(([id]) => id.includes('AgentSessionRole'), )![1]; @@ -254,7 +297,8 @@ describe('AgentSessionRole construct', () => { }); const sessionRole = new AgentSessionRole(stack, 'SR', { assumingRoles: [agentcoreRole], - taskScopedTables: [table], + taskTable: table, + taskScopedTables: [], traceArtifactsBucket: new s3.Bucket(stack, 'TB'), attachmentsBucket: new s3.Bucket(stack, 'AB'), }); diff --git a/cdk/test/constructs/ecs-agent-cluster.test.ts b/cdk/test/constructs/ecs-agent-cluster.test.ts index 30929ff58..4af23466c 100644 --- a/cdk/test/constructs/ecs-agent-cluster.test.ts +++ b/cdk/test/constructs/ecs-agent-cluster.test.ts @@ -571,6 +571,32 @@ describe('EcsAgentCluster construct', () => { }); }); + test('legacy direct access cannot replace task records, edit coordinator fields or access capacity', () => { + const policies = Object.entries(baseTemplate.findResources('AWS::IAM::Policy')) + .filter(([id]) => id.includes('TaskRole')); + expect(policies).toHaveLength(1); + const statements = policies[0][1].Properties.PolicyDocument.Statement; + expect(JSON.stringify(statements)).not.toContain('UserConcurrencyTable'); + const taskStatements = statements.filter( + (s: { Resource: unknown }) => JSON.stringify(s.Resource).includes('TaskTable'), + ); + expect(taskStatements).toHaveLength(2); + for (const s of taskStatements) { + const actions = Array.isArray(s.Action) ? s.Action : [s.Action]; + expect(actions.every((a: string) => [ + 'dynamodb:GetItem', 'dynamodb:BatchGetItem', 'dynamodb:Query', + 'dynamodb:ConditionCheckItem', 'dynamodb:UpdateItem', + ].includes(a))).toBe(true); + if (actions.includes('dynamodb:UpdateItem')) { + const attrs = s.Condition['ForAllValues:StringEquals']['dynamodb:Attributes']; + expect(attrs).toContain('agent_heartbeat_at'); + expect(attrs).not.toContain('microvm_start'); + expect(attrs).not.toContain('concurrency_slot'); + expect(s.Condition.Null['dynamodb:Attributes']).toBe('false'); + } + } + }); + test('build def caps build parallelism to prevent OOM (K14 / ABCA-691)', () => { // The build task def serializes the mise DAG (MISE_JOBS=1) and pins the jest // fleet (JEST_MAX_WORKERS=4) so the cross-package build storm can't OOM the @@ -695,7 +721,8 @@ describe('EcsAgentCluster construct', () => { assumedBy: new iam.ServicePrincipal('bedrock-agentcore.amazonaws.com'), }), ], - taskScopedTables: [taskTable, taskEventsTable], + taskTable, + taskScopedTables: [taskEventsTable], traceArtifactsBucket: new s3.Bucket(stack, 'TraceBucket'), attachmentsBucket: new s3.Bucket(stack, 'AttachmentsBucket'), }); @@ -712,8 +739,11 @@ describe('EcsAgentCluster construct', () => { return Template.fromStack(stack); } + let sessionTemplate: Template; + beforeAll(() => { sessionTemplate = createWithSessionRole(); }); + test('injects AGENT_SESSION_ROLE_ARN into the container', () => { - createWithSessionRole().hasResourceProperties('AWS::ECS::TaskDefinition', { + sessionTemplate.hasResourceProperties('AWS::ECS::TaskDefinition', { ContainerDefinitions: Match.arrayWith([ Match.objectLike({ Environment: Match.arrayWith([ @@ -725,7 +755,7 @@ describe('EcsAgentCluster construct', () => { }); test('task role gets sts:AssumeRole on the SessionRole, not direct task-table DDB grants', () => { - const template = createWithSessionRole(); + const template = sessionTemplate; const policies = template.findResources('AWS::IAM::Policy'); // Identify the task role's own inline policy: it is the one carrying the @@ -748,28 +778,10 @@ describe('EcsAgentCluster construct', () => { expect(taskRolePolicies).toHaveLength(1); const taskRoleStatements = taskRolePolicies[0][1].Properties.PolicyDocument.Statement; - // No unconditioned dynamodb item grant on the task role (the only DDB the - // task role may touch directly is UserConcurrencyTable — assert that any - // DDB statement present is NOT a leading-key-less task-table grant by - // checking none grant dynamodb write actions without a condition beyond - // the concurrency table). Simplest robust check: the task role carries no - // dynamodb:GetItem/Query/BatchWriteItem statement at all for the task - // tables — grantReadWriteData on a removed table would have produced one. - const ddbItemStatements = taskRoleStatements.filter((s: { Action: string | string[] }) => { - const actions = Array.isArray(s.Action) ? s.Action : [s.Action]; - return actions.some((a: string) => - ['dynamodb:GetItem', 'dynamodb:Query', 'dynamodb:BatchWriteItem'].includes(a), - ); - }); - // The only permitted DDB item access on the task role is the - // UserConcurrencyTable grant. The two task-scoped tables (TaskTable, - // TaskEventsTable) must NOT appear — assert no statement references them. - const serialized = JSON.stringify(ddbItemStatements); - expect(serialized).not.toContain('TaskTable'); - expect(serialized).not.toContain('TaskEventsTable'); - - // The conditioned (SessionRole) DDB statements still exist — exactly two - // task-scoped tables, each leading-key gated. + // All tenant data stays on SessionRole; the agent has no counter access. + expect(JSON.stringify(taskRoleStatements)).not.toContain('dynamodb:'); + + // Main-task read/update statements plus the events-table grant. let conditioned = 0; for (const policy of Object.values(policies)) { for (const s of policy.Properties.PolicyDocument.Statement) { @@ -778,7 +790,7 @@ describe('EcsAgentCluster construct', () => { } } } - expect(conditioned).toBe(2); + expect(conditioned).toBe(3); }); }); }); diff --git a/cdk/test/constructs/lambda-microvm-compute.test.ts b/cdk/test/constructs/lambda-microvm-compute.test.ts index d56f55336..1a1bd62f3 100644 --- a/cdk/test/constructs/lambda-microvm-compute.test.ts +++ b/cdk/test/constructs/lambda-microvm-compute.test.ts @@ -113,11 +113,10 @@ function instantiate(options: BuildOptions = {}): Omit { }); agentSessionRole = new AgentSessionRole(stack, 'AgentSessionRole', { assumingRoles: [runtimeRole], - taskScopedTables: [ - new dynamodb.Table(stack, 'TaskTable', { - partitionKey: { name: 'task_id', type: dynamodb.AttributeType.STRING }, - }), - ], + taskTable: new dynamodb.Table(stack, 'TaskTable', { + partitionKey: { name: 'task_id', type: dynamodb.AttributeType.STRING }, + }), + taskScopedTables: [], traceArtifactsBucket: new s3.Bucket(stack, 'TraceBucket'), attachmentsBucket: new s3.Bucket(stack, 'AttachmentsBucket'), }); diff --git a/cdk/test/stacks/agent.test.ts b/cdk/test/stacks/agent.test.ts index 0049b1c5b..d48ccc02b 100644 --- a/cdk/test/stacks/agent.test.ts +++ b/cdk/test/stacks/agent.test.ts @@ -45,6 +45,22 @@ describe('AgentStack', () => { expect(template).toBeDefined(); }); + test('AgentCore runtime has no direct DynamoDB grant, including capacity counters', () => { + const roles = Object.entries(template.findResources('AWS::IAM::Role')); + const runtimeRoleIds = roles.filter(([, role]) => + JSON.stringify(role.Properties.AssumeRolePolicyDocument).includes('bedrock-agentcore.amazonaws.com'), + ).map(([id]) => id); + expect(runtimeRoleIds.length).toBeGreaterThan(0); + const policies = Object.values(template.findResources('AWS::IAM::Policy')).filter( + (policy) => policy.Properties.Roles.some((role: { Ref?: string }) => + role.Ref && runtimeRoleIds.includes(role.Ref), + ), + ); + expect(policies.length).toBeGreaterThan(0); + expect(JSON.stringify(policies)).not.toContain('dynamodb:'); + expect(JSON.stringify(roles.filter(([id]) => runtimeRoleIds.includes(id)))).not.toContain('dynamodb:'); + }); + test('creates exactly 22 DynamoDB tables', () => { // task, task-events, repo, user-concurrency, budget, webhook, task-nudges, // task-approvals (Cedar HITL V2), diff --git a/docs/design/SECURITY.md b/docs/design/SECURITY.md index 3935d020e..5e0feeed3 100644 --- a/docs/design/SECURITY.md +++ b/docs/design/SECURITY.md @@ -41,10 +41,10 @@ Three authentication mechanisms protect the platform, matching its input channel **Per-session IAM scoping** - The agent does not use its long-lived compute role (the AgentCore Runtime `ExecutionRole`, ECS Fargate task role, or Lambda MicroVMs execution role) for tenant data. Instead, at task startup it assumes a per-task **SessionRole** via `sts:AssumeRole` with session tags `{user_id, repo, task_id}`, and uses the resulting short-lived credentials for all DynamoDB and S3 tenant-data access. The SessionRole's policies self-constrain on those tags: -- **DynamoDB**: item access on the four `task_id`-partitioned tables (task, events, approvals, nudges) is gated by a `dynamodb:LeadingKeys` condition equal to `${aws:PrincipalTag/task_id}`, so a session can read or write only its own task's rows. `Scan` is not granted (it ignores leading-keys). `task_id` is the isolation boundary because it is the base-table partition key — `LeadingKeys` cannot bind to a GSI partition key such as `user_id`. +- **DynamoDB**: item access on the four `task_id`-partitioned tables (task, events, approvals, nudges) is gated by a `dynamodb:LeadingKeys` condition equal to `${aws:PrincipalTag/task_id}` and requires that context key to be present. `Scan` is not granted. The main task table permits reads and only attribute-scoped `UpdateItem` writes: the agent can report status, heartbeat, results and approval details, but cannot replace/delete the row or edit coordinator-owned start receipts, capacity reservations, owner identity or compute handles. The reviewed write list is `cdk/src/constructs/agent-task-write-attributes.json`; Python tests exercise the current writers against it. Supporting tables retain task-scoped item writes. `LeadingKeys` binds the base-table partition key, not a GSI key such as `user_id`. - **S3**: trace writes and attachment reads are scoped to the `/${aws:PrincipalTag/user_id}/` object prefix. -The compute role retains only non-tenant access (Bedrock model invocation — already ARN-scoped; CloudWatch Logs; the GitHub PAT secret, read once before the SessionRole is assumed; AgentCore Memory) plus `sts:AssumeRole`/`sts:TagSession` on the SessionRole. Because the agent runs under credentials that are themselves an assumed role, its `AssumeRole` is *role chaining* — capped at one hour regardless of the role's max session duration — so the agent uses a **refreshable** credential provider that re-assumes before expiry (tasks can run up to the 8-hour `maxLifetime`). The design is backend-agnostic: the same SessionRole and agent code serve all three backends (AgentCore Runtime, ECS Fargate, Lambda MicroVMs). A compromised agent session is therefore confined to its own task's data, enforced at the IAM layer rather than by application-code conventions. The policy structure (the `dynamodb:LeadingKeys` condition on `${aws:PrincipalTag/task_id}`, per-user S3 prefixes, and `Scan` exclusion) is asserted by CDK template tests, and the refreshable-credential and session-tag flow by agent unit tests; the matching-tag → allow / mismatched-or-absent-tag → deny behaviour was additionally confirmed once via the IAM policy simulator during development. +The compute role retains only non-tenant access (Bedrock model invocation — already ARN-scoped; CloudWatch Logs; the GitHub PAT secret, read once before the SessionRole is assumed; AgentCore Memory) plus `sts:AssumeRole`/`sts:TagSession` on the SessionRole. Because the agent runs under credentials that are themselves an assumed role, its `AssumeRole` is *role chaining* — capped at one hour regardless of the role's max session duration — so the agent uses a **refreshable** credential provider that re-assumes before expiry (tasks can run up to the 8-hour `maxLifetime`). The design is backend-agnostic: the same SessionRole and agent code serve all three backends (AgentCore Runtime, ECS Fargate, Lambda MicroVMs). Existing scoped credentials are constrained to their tags, but the compute role chooses those tags when assuming the role. The trust policy does not independently bind a worker to one task; this is not complete isolation against a compromised worker with ambient credentials. No worker needs direct access to the shared capacity counter. ECS's legacy configuration without a session role retains unscoped task reads/reporting updates, with the same attribute restriction. The policy structure and Python writer compatibility are tested locally. A historical IAM simulator check covered the original tag conditions; it does not verify the new attribute restriction or deployed transaction authorization. The repository runbook `docs/verification/645-coordinator-metadata.md` contains the required AWS authorization checks. **Lambda MicroVMs compute-role delta** - On this backend the compute role additionally holds prefix-scoped `secretsmanager:GetSecretValue` on the per-workspace channel-OAuth secrets (`bgagent-linear-oauth-*`, `bgagent-jira-oauth-*`), read access to the `/run` payload bucket (the task payload arrives as an S3 object, not as environment), and `ec2:DescribeAvailabilityZones` (`Resource: *`, read-only — EC2 describe actions have no resource-level scoping) for a CDK repo's own synth gate. It is also **the only compute role in the platform whose trust policy carries no confused-deputy condition**: the Lambda MicroVMs service populates no `aws:SourceAccount` or `aws:SourceArn` when it assumes the role, so a trust policy carrying one is unassumable — verified live, not assumed (connector creation failed deterministically, and `RunMicrovm` surfaced the same root cause as a misleading caller-side `iam:PassRole` denial). The compensating controls are that each of the three MicroVM roles can be passed to `lambda.amazonaws.com` only by a named principal — the orchestrator, via an `iam:PassRole` scoped to the execution role's exact ARN, and the CloudFormation deployment role for the build and connector-operator roles — that every other resource they reach is account-scoped by ARN apart from two justified `Resource: *` read/create-time statements, and that none of them holds `iam:*`, cross-account trust, or any `sts:AssumeRole` beyond the execution role's scoped hop to the per-task SessionRole. Full evidence, the two-arm PassRole experiment, and the alternatives considered are in [ADR-021 §4](../decisions/ADR-021-lambda-microvms-compute-backend.md#4-infra-and-iam-conditional-resources-behind-bootstrap-computetypes). diff --git a/docs/src/content/docs/architecture/Security.md b/docs/src/content/docs/architecture/Security.md index d7a6d04e1..dd7b95df0 100644 --- a/docs/src/content/docs/architecture/Security.md +++ b/docs/src/content/docs/architecture/Security.md @@ -45,10 +45,10 @@ Three authentication mechanisms protect the platform, matching its input channel **Per-session IAM scoping** - The agent does not use its long-lived compute role (the AgentCore Runtime `ExecutionRole`, ECS Fargate task role, or Lambda MicroVMs execution role) for tenant data. Instead, at task startup it assumes a per-task **SessionRole** via `sts:AssumeRole` with session tags `{user_id, repo, task_id}`, and uses the resulting short-lived credentials for all DynamoDB and S3 tenant-data access. The SessionRole's policies self-constrain on those tags: -- **DynamoDB**: item access on the four `task_id`-partitioned tables (task, events, approvals, nudges) is gated by a `dynamodb:LeadingKeys` condition equal to `${aws:PrincipalTag/task_id}`, so a session can read or write only its own task's rows. `Scan` is not granted (it ignores leading-keys). `task_id` is the isolation boundary because it is the base-table partition key — `LeadingKeys` cannot bind to a GSI partition key such as `user_id`. +- **DynamoDB**: item access on the four `task_id`-partitioned tables (task, events, approvals, nudges) is gated by a `dynamodb:LeadingKeys` condition equal to `${aws:PrincipalTag/task_id}` and requires that context key to be present. `Scan` is not granted. The main task table permits reads and only attribute-scoped `UpdateItem` writes: the agent can report status, heartbeat, results and approval details, but cannot replace/delete the row or edit coordinator-owned start receipts, capacity reservations, owner identity or compute handles. The reviewed write list is `cdk/src/constructs/agent-task-write-attributes.json`; Python tests exercise the current writers against it. Supporting tables retain task-scoped item writes. `LeadingKeys` binds the base-table partition key, not a GSI key such as `user_id`. - **S3**: trace writes and attachment reads are scoped to the `/${aws:PrincipalTag/user_id}/` object prefix. -The compute role retains only non-tenant access (Bedrock model invocation — already ARN-scoped; CloudWatch Logs; the GitHub PAT secret, read once before the SessionRole is assumed; AgentCore Memory) plus `sts:AssumeRole`/`sts:TagSession` on the SessionRole. Because the agent runs under credentials that are themselves an assumed role, its `AssumeRole` is *role chaining* — capped at one hour regardless of the role's max session duration — so the agent uses a **refreshable** credential provider that re-assumes before expiry (tasks can run up to the 8-hour `maxLifetime`). The design is backend-agnostic: the same SessionRole and agent code serve all three backends (AgentCore Runtime, ECS Fargate, Lambda MicroVMs). A compromised agent session is therefore confined to its own task's data, enforced at the IAM layer rather than by application-code conventions. The policy structure (the `dynamodb:LeadingKeys` condition on `${aws:PrincipalTag/task_id}`, per-user S3 prefixes, and `Scan` exclusion) is asserted by CDK template tests, and the refreshable-credential and session-tag flow by agent unit tests; the matching-tag → allow / mismatched-or-absent-tag → deny behaviour was additionally confirmed once via the IAM policy simulator during development. +The compute role retains only non-tenant access (Bedrock model invocation — already ARN-scoped; CloudWatch Logs; the GitHub PAT secret, read once before the SessionRole is assumed; AgentCore Memory) plus `sts:AssumeRole`/`sts:TagSession` on the SessionRole. Because the agent runs under credentials that are themselves an assumed role, its `AssumeRole` is *role chaining* — capped at one hour regardless of the role's max session duration — so the agent uses a **refreshable** credential provider that re-assumes before expiry (tasks can run up to the 8-hour `maxLifetime`). The design is backend-agnostic: the same SessionRole and agent code serve all three backends (AgentCore Runtime, ECS Fargate, Lambda MicroVMs). Existing scoped credentials are constrained to their tags, but the compute role chooses those tags when assuming the role. The trust policy does not independently bind a worker to one task; this is not complete isolation against a compromised worker with ambient credentials. No worker needs direct access to the shared capacity counter. ECS's legacy configuration without a session role retains unscoped task reads/reporting updates, with the same attribute restriction. The policy structure and Python writer compatibility are tested locally. A historical IAM simulator check covered the original tag conditions; it does not verify the new attribute restriction or deployed transaction authorization. The repository runbook `docs/verification/645-coordinator-metadata.md` contains the required AWS authorization checks. **Lambda MicroVMs compute-role delta** - On this backend the compute role additionally holds prefix-scoped `secretsmanager:GetSecretValue` on the per-workspace channel-OAuth secrets (`bgagent-linear-oauth-*`, `bgagent-jira-oauth-*`), read access to the `/run` payload bucket (the task payload arrives as an S3 object, not as environment), and `ec2:DescribeAvailabilityZones` (`Resource: *`, read-only — EC2 describe actions have no resource-level scoping) for a CDK repo's own synth gate. It is also **the only compute role in the platform whose trust policy carries no confused-deputy condition**: the Lambda MicroVMs service populates no `aws:SourceAccount` or `aws:SourceArn` when it assumes the role, so a trust policy carrying one is unassumable — verified live, not assumed (connector creation failed deterministically, and `RunMicrovm` surfaced the same root cause as a misleading caller-side `iam:PassRole` denial). The compensating controls are that each of the three MicroVM roles can be passed to `lambda.amazonaws.com` only by a named principal — the orchestrator, via an `iam:PassRole` scoped to the execution role's exact ARN, and the CloudFormation deployment role for the build and connector-operator roles — that every other resource they reach is account-scoped by ARN apart from two justified `Resource: *` read/create-time statements, and that none of them holds `iam:*`, cross-account trust, or any `sts:AssumeRole` beyond the execution role's scoped hop to the per-task SessionRole. Full evidence, the two-arm PassRole experiment, and the alternatives considered are in [ADR-021 §4](/sample-autonomous-cloud-coding-agents/architecture/adr-021-lambda-microvms-compute-backend#4-infra-and-iam-conditional-resources-behind-bootstrap-computetypes). diff --git a/docs/verification/645-capacity-reservations.md b/docs/verification/645-capacity-reservations.md index 4fe4e8384..2222e283e 100644 --- a/docs/verification/645-capacity-reservations.md +++ b/docs/verification/645-capacity-reservations.md @@ -45,4 +45,4 @@ docker stop abca645-capacity-ddb Local tests prove the application requests and DynamoDB Local's transaction behavior. They do not establish deployed IAM, AWS scaling, successful rollout or MicroVM sleep/wake behavior. Terminal events may repeat or be lost independently of the atomic seat update. -The reservation/start markers share an agent-writable task row. Their public-API omission does not protect them against a compromised agent. The [P3 plan](./645-p3-implementation-plan.md#1g-protect-coordinator-owned-metadata) tracks that separate security boundary. +The reservation/start markers share the task row. Subsequent prerequisite work restricts agent updates to reporting/approval attributes and removes whole-row replacement/deletion plus direct worker access to the counter. Public-API omission alone was not protection. See [coordinator metadata verification](./645-coordinator-metadata.md) for the writer inventory, actual policy boundary, remaining status/tag trust limits and required AWS authorization checks. These local transaction tests do not prove that security boundary. diff --git a/docs/verification/645-coordinator-metadata.md b/docs/verification/645-coordinator-metadata.md new file mode 100644 index 000000000..93f1e62bb --- /dev/null +++ b/docs/verification/645-coordinator-metadata.md @@ -0,0 +1,69 @@ +# Coordinator metadata write protection (#645) + +The coordinator is the platform code that assigns capacity and starts/stops a task's computer. The agent reports what happened inside that computer. Both use the task record, but the agent must not rewrite the coordinator's saved machine identity or capacity reservation. + +## Implemented boundary + +`AgentSessionRole` now treats the main task table separately from events, approvals and nudges: + +- Main task reads remain scoped to the session's `task_id`. These fields are not secrets. +- Main task writes allow only `UpdateItem`, with `dynamodb:Attributes` restricted to the reviewed list in `cdk/src/constructs/agent-task-write-attributes.json`. Both the attribute context and the scoped leading key must be present; `ForAllValues` alone accepts an absent context key. +- No main-table `PutItem`, `DeleteItem`, `BatchWriteItem` or PartiQL write permission is granted. A replacement could erase a protected field without mentioning its name, so attribute filtering alone would not make replacement safe. +- Supporting table access remains scoped to the tagged task. Approval transactions still use `PutItem` on the approval table and restricted `UpdateItem` on the main task table. +- The main table cannot also be supplied as a supporting table: the construct rejects that bypass at synthesis. +- AgentCore and configured ECS/MicroVM workers have no direct DynamoDB access; they assume the session role. ECS's legacy configuration without a session role uses the same attribute restriction, but its reads/updates are not task-scoped. None of the compute roles has the old shared-capacity-counter grant. + +This is an allowlist: a new coordinator field is protected without adding its name to a denylist. New agent reporting fields require a deliberate contract update. + +AWS documents [`dynamodb:Attributes`](https://docs.aws.amazon.com/amazondynamodb/latest/developerguide/specifying-conditions.html) as the top-level attributes referenced in a request, including the parent of a nested path. For example, `SET #receipt.#handle = :value` with aliases resolving to `microvm_start.handle` references `microvm_start` and is outside the allowed set. Read responses remain unrestricted within the tagged task; this change does not claim attribute confidentiality. + +## Writer inventory + +| Agent writer | Main-task fields | +| --- | --- | +| `write_running` | `status`, `status_created_at`, `started_at`, optional `logs_url` | +| `write_heartbeat` | `agent_heartbeat_at`; condition reads `status` | +| `write_terminal` | Status/completion time, cost/turns, verification results, PR/answer and trace/artifact references | +| `write_trace_uri_conditional` | `trace_s3_uri`; condition reads terminal `status` | +| `transact_write_approval_request` | `status`, `awaiting_approval_request_id`; also creates the supporting approval row | +| `transact_resume_from_approval` | `status`, refreshed heartbeat, removes approval request ID | +| `increment_approval_gate_count_in_ddb` | `approval_gate_count` | + +`progress_writer.py` writes only the supporting events table. `nudge_reader.py` updates only the supporting nudges table. Approval decisions/timeouts use the supporting approvals table. + +The unused `write_submitted` and `write_session_info` Python helpers were removed. Neither had a production caller; their comments incorrectly attributed task creation/session registration to the agent. The TypeScript coordinator already owns those operations. + +## Local checks and their limits + +The regression first failed against the old policy because the main-table grant included `PutItem`. Construct tests now check the restricted main-task statements, missing-context guards, duplicate-table rejection, unchanged session tags and absence of direct worker counter access. Stack tests check actual AgentCore and MicroVM wiring; ECS tests cover both configured and legacy direct access. + +Python contract tests run the real task writers against recording clients, including all terminal-result fields and both approval transactions. They compare the requested attributes against the same JSON list used by CDK and require review when a new writer is added. + +These checks inspect generated policies and request compatibility. They do **not** execute AWS authorization. DynamoDB Local, used for the capacity tests, also does not implement IAM. Keep the following live gate open. + +## AWS acceptance and rollout gate + +Use an isolated deployment and disposable task rows with known initial `microvm_start`, `concurrency_slot`, owner and compute metadata. Record the deployed commit, image version and effective role policies. Use task-scoped credentials for two different tasks and separately check the ambient compute roles. + +| Request | Required result | +| --- | --- | +| Own-task heartbeat, full terminal result and trace repair | Allowed; protected fields unchanged | +| Approval request and resume transactions | Allowed; both records consistent and heartbeat refreshed | +| Update of another task or a session with no task tag | Denied | +| Put/replacement, Delete, BatchWrite put/delete on own task | Denied; original row unchanged | +| SET/REMOVE/ADD/DELETE involving `concurrency_slot` or `microvm_start`, including aliases/nested paths | Denied | +| Change `user_id`, TTL, session ID, compute type or compute metadata | Denied | +| Transaction containing a forbidden main-task update/put/delete plus an otherwise valid approval write | Entire transaction denied; neither record changes | +| PartiQL update/delete/insert against the task table | Denied | +| Direct task-table or capacity-counter operations using any configured worker's ambient credentials | Denied | +| Coordinator start, cancellation, finalization and counter repair | Allowed; replay retains exactly-once reservation accounting | + +Drain old executions according to the [capacity rollout procedure](./645-capacity-reservations.md#upgrade-and-live-verification). Review effective policies for additional grants or resource policies that could bypass this allowlist. Deploy the code/policies together and publish the matching agent image. Existing agent writers already fit the list, but an external or old custom caller of the removed helpers must be migrated. Wait for IAM propagation and verify newly assumed and existing sessions before reopening admissions. No bootstrap policy change is required by this patch; it changes application-role permissions. + +Do not roll back to unrestricted worker writes while relying on protected reservation/start metadata. Drain first and explicitly review the security consequence of a rollback. + +## Remaining trust limits + +The agent can still report status and results. This patch does not authenticate whether its reported success/failure is truthful or constrain status values/transitions at IAM level. Supporting approval/event/nudge rows retain their prior permissions. + +The compute role chooses `{user_id, repo, task_id}` session tags. Current trust policies do not independently prove that the chosen task belongs to that worker. A compromised whole worker with ambient credentials can therefore try to assume a differently tagged session. The attribute restriction applies to that session too, but this is not complete tenant isolation. Trusted task/deployment identity and task-scoped payload reads remain separate prerequisites in the [P3 plan](./645-p3-implementation-plan.md). diff --git a/docs/verification/645-p3-implementation-plan.md b/docs/verification/645-p3-implementation-plan.md index d73459666..b4cb9b74d 100644 --- a/docs/verification/645-p3-implementation-plan.md +++ b/docs/verification/645-p3-implementation-plan.md @@ -18,7 +18,8 @@ Prerequisite work is tracked here on `fix/645-microvm-readiness`. “Completed - [ ] Verify AWS token retention/conflicts and unknown-start cleanup on a live deployment. - [x] Make capacity acquisition/release atomic per task across crash replay; unify counter writers and repair. - [ ] Verify the capacity protocol's upgrade/drain procedure, deployed IAM and scan scale in AWS. -- [ ] Protect coordinator-owned task metadata from agent writes before claiming hostile-task isolation. +- [x] Restrict agent task updates to reporting fields; remove replacement/deletion and worker counter grants. +- [ ] Verify metadata restrictions with real AWS sessions/transactions; retain status/tag trust limits. - [ ] Finish logging-failure observability (#810) and registry-overflow coverage (#818). - [ ] Implement production nesting if included, then verify a clean P2 deployment. - [ ] Implement and verify the P3 sleep/wake lifecycle described below. @@ -65,6 +66,13 @@ Final checks for the fourth batch: CDK ESLint and compilation passed. **153 hand No AWS resources changed. Activating this batch requires the coordinated deployment/drain procedure in [capacity verification](./645-capacity-reservations.md), including the reconciler's task-update permission and upload confirmation's narrowed counter access. No agent source/image or bootstrap bundle changed. Coordinator metadata protection is newly tracked in 1G; internal reservation fields currently share an agent-writable row. +Fifth prerequisite batch completed locally on 2026-09-13: + +- Main-task agent writes now use a reviewed attribute allowlist; whole-row replacement/deletion and coordinator field updates are excluded. Missing session/attribute context fails closed. ECS's legacy direct path shares the restriction. +- Removed unused AgentCore/ECS capacity-table grants and the uncalled Python submission/session-registration helpers. Corrected comments that overstated tenant isolation or described approval writes as best-effort. +- The old-policy regression failed on `PutItem`. **1,795 Python tests** pass with **84.47%** coverage, including actual writer-request/permission-contract checks; **268 CDK tests** pass across session-role, ECS, MicroVM and full-stack suites. Python quality, CDK lint/compilation and documentation checks pass. These counts overlap earlier batches; four obsolete helper tests were removed and two contract tests added. +- [Metadata verification](./645-coordinator-metadata.md) records the effective source-policy boundary, writer inventory, rollback constraints and pending real-AWS allowed/denied transaction matrix. No table migration or bootstrap-policy update is required. Application-role deployment and a matching agent image remain necessary; nothing was deployed. + ## The result we want When the coding agent asks a human for permission, its computer may go to sleep. The human can approve or deny while it sleeps. The computer wakes, reads the saved answer, and continues or refuses the action. If nobody answers, it wakes before the deadline and applies the existing timeout-as-denial rule. Its files, task identity, permissions, progress and deadline remain correct. @@ -176,9 +184,13 @@ Every counter change carries a fresh revision. Scheduled repair strongly scans t ### 1G. Protect coordinator-owned metadata -The agent's task-scoped role restricts **which row** it can change, but currently permits replacement/deletion and unrestricted attribute updates within its own row. `microvm_start` and `concurrency_slot` are internal task-table fields; omitting them from API responses does not protect their storage. A compromised agent must not be able to rewrite the coordinator's start identity or revive a released reservation. +**Implemented locally:** the main task table now permits own-task reads and only `UpdateItem` on an explicit reporting/approval attribute list. Replacement/deletion, start receipts, capacity markers, owner identity and compute handles are excluded. Supporting tables retain their task-scoped access. Missing IAM context keys fail closed. ECS's legacy direct-grant path uses the same attribute restriction; AgentCore/ECS no longer receive the unused shared-counter grant, and MicroVM never had it. + +The writer inventory found no production callers of Python `write_submitted` or `write_session_info`; those unused helpers and stale comments are removed. Contract tests exercise every current main-task writer, including complete terminal results and approval transactions, against the JSON attribute list used by CDK. Construct/stack tests inspect the actual grants across all three backends and prevent the main table from being supplied as an unrestricted supporting table. + +**Deployment gate:** run the allowed/denied request matrix in [metadata verification](./645-coordinator-metadata.md) using real scoped and ambient credentials, including aliased/nested updates, replacement/deletion and transactions that mix forbidden task writes with valid approval writes. Inspect effective policies and preserve coordinator start/finalize/cancel behavior. This requires no new table or bootstrap policy change, but it does require application-role deployment and matching image verification. -Choose coordinator-only storage or carefully constrained agent write permissions, inventory every task-row writer, and test forged/replaced/deleted rows. Preserve legitimate agent progress/status updates and all backend identity tags. Treat this as a separate security prerequisite; the replay tests in 1D/1F do not establish a hostile-agent guarantee. +**Remaining trust limits:** agent status/results remain reports from the agent. Compute roles choose their session tags; existing trust does not independently bind those choices to a task. This patch protects coordinator attributes from the resulting session permissions; it does not establish complete hostile-worker tenant isolation. The replay tests in 1D/1F also do not prove AWS authorization. ## 2. Nest infrastructure if adopting the split diff --git a/docs/verification/645-p3-readiness-review.md b/docs/verification/645-p3-readiness-review.md index 6b45cc55a..51e0aed64 100644 --- a/docs/verification/645-p3-readiness-review.md +++ b/docs/verification/645-p3-readiness-review.md @@ -8,7 +8,7 @@ This review covers the existing MicroVM implementation, related P2 follow-ups, c Further local work adds durable MicroVM start receipts, stable request tokens and bounded handle recovery. AWS token-retention/conflict behavior and unknown-ID cleanup remain live verification gates. The shared finalizer's repeated-decrement risk was subsequently reproduced and fixed locally with task-owned reservation transactions, unified counter writers and revision-guarded reconciliation, including approval waits. DynamoDB Local exercises the real transaction conditions; deployed IAM, scan scale and upgrade/drain behavior still need AWS verification. -The reservation review also found a separate trust boundary: the agent role can write/replace/delete its own task row, including coordinator metadata. Internal API fields are not protected from that writer by their naming or TypeScript types. The plan tracks coordinator-only storage or constrained agent writes as a security prerequisite. The replay fix assumes cooperating writers; it is not a hostile-agent isolation claim. +The reservation review found a separate trust boundary: the old agent role could write/replace/delete its task row, including coordinator metadata. A subsequent local fix restricts main-task writes to reporting/approval attributes, removes replacement/deletion permissions and removes unused worker counter grants. It also removes unused Python submission/session-info helpers and corrects overstated tenant-isolation comments. [Metadata verification](./645-coordinator-metadata.md) records the tests and pending AWS gate. Status reports still come from the agent, and the compute role chooses session tags; this is not complete hostile-worker isolation. ## Start here: the pieces in plain language From 82a9df80267f8664c41109044657a13765acc260 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Sun, 13 Sep 2026 17:20:11 -0400 Subject: [PATCH 015/149] fix(security): authenticate task payload bootstrap for ECS and MicroVM (#645) --- agent/README.md | 26 +- agent/src/payload_bootstrap.py | 194 ++++ agent/src/server.py | 339 +------ agent/tests/test_payload_bootstrap.py | 301 ++++++ agent/tests/test_server.py | 872 ++---------------- cdk/src/constructs/ecs-agent-cluster.ts | 32 +- cdk/src/constructs/ecs-payload-bucket.ts | 25 +- cdk/src/constructs/lambda-microvm-compute.ts | 37 +- .../payload-bootstrap-permissions.ts | 67 ++ cdk/src/constructs/task-orchestrator.ts | 41 +- cdk/src/handlers/shared/payload-bootstrap.ts | 183 ++++ .../shared/strategies/ecs-strategy.ts | 142 +-- .../strategies/lambda-microvm-strategy.ts | 214 +---- cdk/src/stacks/agent.ts | 10 +- cdk/test/constructs/ecs-agent-cluster.test.ts | 10 +- .../constructs/lambda-microvm-compute.test.ts | 18 +- .../payload-bootstrap-permissions.test.ts | 67 ++ cdk/test/constructs/task-orchestrator.test.ts | 16 +- .../handlers/orchestrate-task-microvm.test.ts | 32 +- .../shared/microvm-start-recovery.test.ts | 24 +- .../handlers/shared/payload-bootstrap.test.ts | 160 ++++ .../shared/strategies/ecs-strategy.test.ts | 196 ++-- .../lambda-microvm-strategy.test.ts | 366 ++------ .../start-session-composition.test.ts | 15 +- cdk/test/scripts/check-constants-sync.test.ts | 26 + cdk/test/stacks/agent.test.ts | 10 +- contracts/constants.json | 9 + contracts/constants.md | 35 +- ...ADR-021-lambda-microvms-compute-backend.md | 37 +- docs/design/COMPUTE.md | 2 +- docs/design/SECURITY.md | 5 +- docs/src/content/docs/architecture/Compute.md | 2 +- .../src/content/docs/architecture/Security.md | 5 +- ...Adr-021-lambda-microvms-compute-backend.md | 37 +- docs/verification/645-coordinator-metadata.md | 4 + .../645-p3-implementation-plan.md | 23 +- docs/verification/645-p3-readiness-review.md | 6 +- docs/verification/645-payload-bootstrap.md | 127 +++ scripts/check-constants-sync.ts | 34 +- 39 files changed, 1737 insertions(+), 2012 deletions(-) create mode 100644 agent/src/payload_bootstrap.py create mode 100644 agent/tests/test_payload_bootstrap.py create mode 100644 cdk/src/constructs/payload-bootstrap-permissions.ts create mode 100644 cdk/src/handlers/shared/payload-bootstrap.ts create mode 100644 cdk/test/constructs/payload-bootstrap-permissions.test.ts create mode 100644 cdk/test/handlers/shared/payload-bootstrap.test.ts create mode 100644 docs/verification/645-payload-bootstrap.md diff --git a/agent/README.md b/agent/README.md index 58902300c..ba8c9fc9f 100644 --- a/agent/README.md +++ b/agent/README.md @@ -244,23 +244,25 @@ Baked secrets are **reported, not enforced**: `warnings` lists the names (never `microvmId` is parsed defensively and **arrives empty in practice**: the service sends `""` here, unlike `/run` where it is populated (live-verified, ADR-021 P2-F8). So an empty id is expected-normal, not a degraded read — and this hook therefore **cannot** join the guest's record to the control-plane one. `/run`'s `hook accepted task_id=… microvm_id=…` line carries that correlation; `/terminate`'s value is the pipeline-state snapshot it reports. -**`POST /aws/lambda-microvms/runtime/v1/run`** — Payload delivery. Validates the body, installs `platform_config` (below), starts the pipeline in a background thread (the same `_extract_invocation_params` → `_spawn_background` path `/invocations` uses), and returns 200 inside the 1–60 s hook budget. Body: +**`POST /aws/lambda-microvms/runtime/v1/run`** — Authenticate and download a task, install `platform_config` (below), start the pipeline in a background thread, and return 200 inside the hook budget. Protocol v2 is implemented locally; real AWS permission/network/expiry verification remains pending (repository runbook: `docs/verification/645-payload-bootstrap.md`). + +`runHookPayload` is a JSON **string** passed through by `RunMicrovm`, containing: ```json { - "microvmId": "microvm-b44b69d9-…", - "runHookPayload": "{\"agent_payload_s3_uri\": \"s3://bucket//payload.json\", \"platform_config\": {…}}" + "version": 2, + "task_id": "TASK001", + "bootstrap_s3_uri": "s3://deployment-bucket/bootstrap/.json", + "payload_url": "", + "expires_at": 1789312500000 } ``` -`runHookPayload` is an opaque **string** the service passes through from `RunMicrovm`. ABCA's contract for it is one of two shapes, mirroring the ECS container env contract (`AGENT_PAYLOAD` / `AGENT_PAYLOAD_S3_URI`): +The worker reads the deployment manifest with its ambient AWS role, which explicitly denies object reads outside that bucket's `bootstrap/*` prefix. It downloads the task using the coordinator's short-lived signed URL, verifies the task identity, and requires the task's configuration to equal the manifest. The digest in the manifest filename checks its bytes; IAM authenticates its origin. The entire reference must fit **4,096 bytes**. Manifests and task payloads are capped at **16 KiB** and **8 MiB**. Redirects, environment proxies and hosts/paths other than the task's regional S3 object are rejected. -| Envelope | When | -|---|---| -| `{"agent_payload": {…}, "platform_config": {…}}` | the whole orchestrator payload inline — only when it fits | -| `{"agent_payload_s3_uri": "s3://bucket/key", "platform_config": {…}}` | pointer to the payload in the platform payload bucket | +ECS uses the same reference in `AGENT_PAYLOAD_REF`, with an empty manifest config because deployment settings already come from its task definition/overrides. `load_ecs_payload()` removes the capability environment variable before importing the pipeline. Neither backend accepts the old unsigned `AGENT_PAYLOAD`, `AGENT_PAYLOAD_S3_URI`, inline hook or S3-pointer formats. Roll out the matching coordinator, worker images and policies with admissions paused and old tasks drained; the runbook records upgrade and rollback steps. -The service caps `runHookPayload` at **4 096 bytes**, so the **pointer form is the normal one** — a hydrated payload is essentially always larger. Fetching it needs no new env var: the MicroVM execution role holds read-only access to that bucket and the URI carries bucket + key. +A presigned URL is a temporary download permission: **never log it**. Its requested lifetime is at most 900 seconds, shortened by known signer credential expiry; initial creation requires at least 300 seconds. The coordinator privately saves the exact URL for retries and deletes the payload and saved launch reference at finalization. An expired saved reference fails rather than being silently re-signed. #### `platform_config` — the agent's env, delivered per task (P2) @@ -270,7 +272,7 @@ On AgentCore and ECS the agent's non-secret platform env arrives as runtime env { "platform_config": { "task_table_name": "…", "github_token_secret_arn": "arn:…" } } ``` -Each snake_case key installs into its UPPER_SNAKE env var, and a payload value **wins** over any image/pre-existing value (the payload describes the live deployment; the snapshot describes a past one). Installation happens **before** any credential or pipeline initialisation — the very next step resolves the GitHub token from `GITHUB_TOKEN_SECRET_ARN`. Everything the hook logs before that point goes to stdout only (`[server/run-pre-config]`), for the same reason the build hooks do: the CloudWatch writer resolves AWS credentials and can cache credential state in `boto3.DEFAULT_SESSION` (environment-derived region is re-read for new clients), and until the install has run the only environment available is whatever the snapshot baked. The single AWS call allowed before the install is the S3 payload fetch, because the config is inside the object being fetched. The allowlist lives in `contracts/constants.json` → `microvm_platform_config` (produced by the orchestrator, consumed here; shape enforced by `mise run check:constants-sync`): +Each snake_case key installs into its UPPER_SNAKE env var, and a payload value **wins** over any image/pre-existing value (the payload describes the live deployment; the snapshot describes a past one). Installation happens **before** task credential, secret or pipeline initialisation — the very next step resolves the GitHub token from `GITHUB_TOKEN_SECRET_ARN`. Everything the hook logs before that point goes to stdout only (`[server/run-pre-config]`), for the same reason the build hooks do: the CloudWatch writer resolves AWS credentials and can cache credential state in `boto3.DEFAULT_SESSION` (environment-derived region is re-read for new clients), and until the install has run the only environment available is whatever the snapshot baked. Before installation, bootstrap reads the deployment manifest through the attributed S3 client and downloads the single task object over signed HTTPS. Other pre-install diagnostics remain stdout-only. The allowlist lives in `contracts/constants.json` → `microvm_platform_config` (produced by the orchestrator, consumed here; shape enforced by `mise run check:constants-sync`): | Key | Env var | Required | |---|---|---| @@ -288,9 +290,9 @@ Each snake_case key installs into its UPPER_SNAKE env var, and a payload value * | `aws_sdk_ua_app_id` | `AWS_SDK_UA_APP_ID` | | | `anthropic_default_haiku_model` | `ANTHROPIC_DEFAULT_HAIKU_MODEL` | | -Values are **non-secret identifiers only** — secrets are still fetched at `/run` time from Secrets Manager using the ARNs delivered here, so task secrets need not be baked into the snapshot. The build-hook warning is not proof that a hand-built image contains no secrets. The allowlist **fails closed**: these values land in `os.environ` of the process that spawns the agent's tool subprocesses, so an unrecognised key is an env-injection attempt (`LD_PRELOAD`, `AWS_ENDPOINT_URL`, …) and the whole run is rejected with nothing installed. Blank/`null` values for optional keys are skipped rather than clobbering an image value; blank required keys are rejected. An envelope with no `platform_config` is accepted with a warning **only if the effective environment already contains all required identifiers**. Otherwise `/run` rejects it as incomplete. Control characters and inconsistent ARN account/partition fields are also rejected; the ARN check does not establish that an identifier belongs to this deployment ([#817](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/817)). +Values are **non-secret identifiers only** — secrets are still fetched at `/run` time from Secrets Manager using the ARNs delivered here, so task secrets need not be baked into the snapshot. The build-hook warning is not proof that a hand-built image contains no secrets. The allowlist **fails closed**: these values land in `os.environ` of the process that spawns the agent's tool subprocesses, so an unrecognised key is an env-injection attempt (`LD_PRELOAD`, `AWS_ENDPOINT_URL`, …) and the whole run is rejected with nothing installed. Blank/`null` values for optional keys are skipped rather than clobbering an image value; blank required keys are rejected. A verified MicroVM payload must contain `platform_config`; there is no baked-environment fallback. Control characters and inconsistent ARN account/partition fields are also rejected. ARN consistency alone is not deployment authentication: v2 supplies that through the IAM-read manifest and exact configuration comparison, including same-account workspace identifiers ([#817](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/817)). -Rejections are structured so they are readable in the MicroVM log group: `400 MICROVM_RUN_PAYLOAD_INVALID` (unusable envelope — retrying the same body cannot help), `500 MICROVM_RUN_PAYLOAD_UNREADABLE` (the S3 fetch failed), `400 MICROVM_RUN_PLATFORM_CONFIG_INVALID` (key off the allowlist, non-object block, or non-string value — fix the producer), `400 MICROVM_RUN_PLATFORM_CONFIG_INCOMPLETE` (a required key missing or blank — fix the deployment wiring), `400 TASK_RECORD_INCOMPLETE` (same validator and vocabulary as `/invocations`). +Rejections are structured so they are readable in the MicroVM log group: `400 MICROVM_RUN_PAYLOAD_INVALID` (unusable envelope — retrying the same body cannot help), `500 MICROVM_RUN_PAYLOAD_UNREADABLE` (manifest/payload read or stored bytes failed), `400 MICROVM_RUN_PLATFORM_CONFIG_INVALID` (key off the allowlist, non-object block, or non-string value — fix the producer), `400 MICROVM_RUN_PLATFORM_CONFIG_INCOMPLETE` (a required key missing or blank — fix the deployment wiring), `400 TASK_RECORD_INCOMPLETE` (same validator and vocabulary as `/invocations`). `/suspend` and `/resume` are deliberately **not** served — declaring a hook nothing answers fails the corresponding lifecycle transition, so the CDK construct declares exactly the hooks the agent serves. They land in P3 with the ComputeStrategy interface widening. diff --git a/agent/src/payload_bootstrap.py b/agent/src/payload_bootstrap.py new file mode 100644 index 000000000..b5133731b --- /dev/null +++ b/agent/src/payload_bootstrap.py @@ -0,0 +1,194 @@ +"""Authenticate deployment settings before reading one task's payload. + +The ambient worker role can GetObject only under its deployment's bootstrap/ +prefix, with an explicit deny outside that prefix (including public buckets). +Only the coordinator can write those manifests. The payload itself is fetched +without worker credentials, using a single-object capability. Never log it. +""" + +from __future__ import annotations + +import hashlib +import json +import os +import re +from typing import Any +from urllib.error import HTTPError +from urllib.parse import parse_qs, urlsplit +from urllib.request import HTTPRedirectHandler, ProxyHandler, build_opener + +from shared_constants import SHARED_CONSTANTS + +CONTRACT = SHARED_CONSTANTS["payload_bootstrap"] + + +class PayloadFetchError(RuntimeError): + """Authenticated bootstrap or payload bytes could not be read; contains no URL.""" + + +class _NoRedirect(HTTPRedirectHandler): + def redirect_request(self, req, fp, code, msg, headers, newurl): + return None + + +def _object(raw: bytes, label: str) -> dict: + try: + value = json.loads(raw) + except (UnicodeError, ValueError): + raise PayloadFetchError(f"{label} is not valid JSON") from None + if not isinstance(value, dict): + raise PayloadFetchError(f"{label} must contain an object") + return value + + +def _manifest(uri: str, backend: str) -> tuple[str, dict]: + parsed = urlsplit(uri) + if ( + parsed.scheme != "s3" + or not re.fullmatch(r"[a-z0-9][a-z0-9.-]{1,61}[a-z0-9]", parsed.netloc) + or parsed.query + or parsed.fragment + ): + raise ValueError("bootstrap_s3_uri must identify a deployment manifest") + key = parsed.path.removeprefix("/") + expected = re.fullmatch(re.escape(CONTRACT["manifest_prefix"]) + r"([a-f0-9]{64})\.json", key) + if not expected: + raise ValueError("bootstrap_s3_uri has an invalid manifest key") + + # Do not use a caller-supplied endpoint or default cached boto3 session. + from botocore.config import Config + + from aws_session import platform_client + + try: + client = platform_client( + "s3", + config=Config(connect_timeout=5, read_timeout=10, retries={"max_attempts": 2}), + ) + response = client.get_object(Bucket=parsed.netloc, Key=key) + body = response["Body"] + try: + length = response.get("ContentLength") + if not isinstance(length, int) or not 0 < length <= CONTRACT["max_manifest_bytes"]: + raise PayloadFetchError("deployment manifest has an invalid length") + raw = body.read(CONTRACT["max_manifest_bytes"] + 1) + if len(raw) != length: + raise PayloadFetchError("deployment manifest body is incomplete") + finally: + body.close() + except PayloadFetchError: + raise + except Exception as exc: + # SDK/HTTP exception text can contain URLs. Expose only the class. + raise PayloadFetchError(f"deployment manifest read failed ({type(exc).__name__})") from None + if hashlib.sha256(raw).hexdigest() != expected[1]: + raise PayloadFetchError("deployment manifest digest does not match its key") + manifest = _object(raw, "deployment manifest") + if manifest.get("version") != CONTRACT["version"] or manifest.get("backend") != backend: + raise ValueError("deployment manifest version/backend does not match this worker") + if not isinstance(manifest.get("platform_config"), dict): + raise ValueError("deployment manifest has no platform configuration") + return parsed.netloc, manifest["platform_config"] + + +def _payload_url(url: str, bucket: str, task_id: str) -> None: + """Permit only the exact object's regional S3 HTTPS endpoint, without redirects.""" + try: + parsed = urlsplit(url) + query = parse_qs(parsed.query, keep_blank_values=True, strict_parsing=True) + if any(len(values) != 1 for values in query.values()): + raise ValueError + _access_key, _date, region, service, terminator = query["X-Amz-Credential"][0].split("/") + if ( + service != "s3" + or terminator != "aws4_request" + or not re.fullmatch(r"[a-z]{2}(?:-[a-z]+)+-\d", region) + or query["X-Amz-Algorithm"] != ["AWS4-HMAC-SHA256"] + or query["X-Amz-SignedHeaders"] != ["host"] + or not re.fullmatch(r"[a-f0-9]{64}", query["X-Amz-Signature"][0]) + or not 0 < int(query["X-Amz-Expires"][0]) <= CONTRACT["url_ttl_seconds"] + or parsed.scheme != "https" + or parsed.username is not None + or parsed.password is not None + or parsed.port is not None + or parsed.fragment + ): + raise ValueError + suffix = "amazonaws.com.cn" if region.startswith("cn-") else "amazonaws.com" + host = f"s3.{region}.{suffix}" + valid = ( + parsed.netloc == f"{bucket}.{host}" and parsed.path == f"/{task_id}/payload.json" + ) or (parsed.netloc == host and parsed.path == f"/{bucket}/{task_id}/payload.json") + if not valid: + raise ValueError + except (KeyError, IndexError, TypeError, ValueError): + raise ValueError("payload reference must sign this task's exact S3 object") from None + + +def _download(url: str) -> dict: + # Disable environment proxies and redirects. No AWS credential provider is + # involved; S3 authorizes the coordinator's signature on this one object. + opener = build_opener(ProxyHandler({}), _NoRedirect()) + try: + with opener.open(url, timeout=10) as response: + length = int(response.headers.get("Content-Length", "0")) + if not 0 < length <= CONTRACT["max_payload_bytes"]: + raise PayloadFetchError("task payload has an invalid length") + raw = response.read(CONTRACT["max_payload_bytes"] + 1) + if len(raw) != length: + raise PayloadFetchError("task payload body is incomplete") + return _object(raw, "task payload") + except PayloadFetchError: + raise + except HTTPError as exc: + raise PayloadFetchError(f"task payload download returned HTTP {exc.code}") from None + except Exception as exc: + raise PayloadFetchError(f"task payload download failed ({type(exc).__name__})") from None + + +def resolve_payload_reference(reference: Any, backend: str) -> tuple[dict, dict]: + if not isinstance(reference, dict) or reference.get("version") != CONTRACT["version"]: + raise ValueError( + "payload bootstrap v2 is required; deploy a matching coordinator and image" + ) + task_id = reference.get("task_id") + uri = reference.get("bootstrap_s3_uri") + url = reference.get("payload_url") + if ( + not isinstance(task_id, str) + or not re.fullmatch(r"[A-Za-z0-9_-]{1,128}", task_id) + or not isinstance(uri, str) + or not isinstance(url, str) + ): + raise ValueError("payload reference is missing task identity or download coordinates") + bucket, config = _manifest(uri, backend) + _payload_url(url, bucket, task_id) + document = _download(url) + payload = document.get("agent_payload") + if ( + document.get("version") != CONTRACT["version"] + or document.get("task_id") != task_id + or not isinstance(payload, dict) + or payload.get("task_id") != task_id + ): + raise ValueError("downloaded payload does not belong to the referenced task") + if document.get("platform_config") != config: + raise ValueError( + "payload configuration does not match the authenticated deployment manifest" + ) + return payload, config + + +def load_ecs_payload() -> dict: + """Consume the capability before task subprocesses can inherit the environment.""" + raw = os.environ.pop("AGENT_PAYLOAD_REF", "") + if not raw: + raise ValueError("AGENT_PAYLOAD_REF is required; deploy a matching coordinator and image") + try: + reference = json.loads(raw) + except (UnicodeError, ValueError) as exc: + raise ValueError("AGENT_PAYLOAD_REF is not valid JSON") from exc + if not isinstance(reference, dict) or reference.get("task_id") != os.environ.get("TASK_ID"): + raise ValueError("payload reference does not match the ECS task identity") + payload, _ = resolve_payload_reference(reference, "ecs") + return payload diff --git a/agent/src/server.py b/agent/src/server.py index 71a0dcc75..8dd724ef2 100644 --- a/agent/src/server.py +++ b/agent/src/server.py @@ -855,7 +855,6 @@ async def invoke_agent(request: Request, body: InvocationRequest): MICROVM_HOOK_PREFIX = "/aws/lambda-microvms/runtime/v1" #: ``s3://`` scheme prefix for the out-of-band payload pointer. -_S3_URI_SCHEME = "s3://" # --- platform_config allowlist (ADR-021 P2) -------------------------------- # WHY the agent's platform env arrives in the ``/run`` payload at all, instead of @@ -907,70 +906,15 @@ async def invoke_agent(request: Request, body: InvocationRequest): #: WHAT THIS BUYS, STATED PRECISELY — because the honest answer is narrower than #: "stops secret exfiltration", and overstating it would hide the residual gap. #: -#: The key allowlist above stops a payload setting ``LD_PRELOAD``; it does not stop -#: a payload pointing an *allowlisted* key at a different resource. The value that -#: matters most is ``github_token_secret_arn``: ``config.resolve_github_token`` -#: fetches whatever ARN it names using the UNSCOPED execution role and caches the -#: raw ``SecretString`` into ``os.environ["GITHUB_TOKEN"]``, from which ``shell.py`` -#: hands the environment to every repo subprocess — i.e. into the model's tool -#: surface. So the value is worth validating. -#: -#: This check is DEFENCE IN DEPTH AND FAIL-FAST, not the primary control: -#: -#: * What actually stops a cross-account read today is IAM. Every grant on the -#: execution role is account-scoped by construction — ``grantRead`` on the GitHub -#: PAT secret, and the ``bgagent-linear-oauth-*`` / ``bgagent-jira-oauth-*`` -#: prefix grants built with ``stack.formatArn`` (``lambda-microvm-compute.ts``). -#: A foreign-account ARN therefore AccessDenies with or without this check. What -#: this adds is a structured 400 at the door instead of an opaque -#: ``AccessDeniedException`` mid-startup, and a guard that still holds if a -#: future grant is ever widened. -#: * What this check does NOT stop is an IN-ACCOUNT redirect. The channel-OAuth -#: grants are prefix grants (unavoidable: the CLI mints ``bgagent-*-oauth-`` -#: at setup, so the names are unknown at synth), so a block whose anchor and -#: whose ``github_token_secret_arn`` name the SAME account but a DIFFERENT -#: workspace's OAuth secret is accepted here. Partition + account is the only -#: boundary this check enforces; it is not a per-workspace authorization check. -#: * That residual gap is currently unreachable from the guest, which is why it is -#: left open. ``platform_config`` is produced by the orchestrator Lambda and the -#: MicroVM execution role holds ``grantRead`` ONLY on the payload bucket -#: (``payloadBucket.grantRead``), so a running MicroVM can read another task's -#: payload but cannot write one. Reaching the in-account case requires already -#: controlling the orchestrator's environment or the bucket's write path. -#: -#: ESCALATION: if the payload path ever becomes less trusted — a third-party -#: producer, an operator-editable envelope, or any grant that lets the guest write -#: the payload bucket — this must grow into a name-shape check that ties -#: ``github_token_secret_arn`` to the task's own channel/workspace, because -#: partition+account pinning provably does not cover that case. -#: -#: The prefix grant itself is at ECS parity. The ASYMMETRY that makes value -#: validation worth doing here at all is new to this backend: on ECS these ARNs -#: arrive as deploy-time container env; here they arrive in a network payload. +#: The allowlist restricts environment-variable names; ARN shape/account checks +#: below are defense in depth. Startup settings come from an IAM-authenticated +#: bootstrap manifest read before the task payload. That independent manifest +#: pins the exact GitHub secret and other deployment identifiers, including valid +#: cross-region secrets. A same-payload account anchor alone cannot do that. MICROVM_PLATFORM_CONFIG_ARN_KEYS: frozenset[str] = frozenset(_PLATFORM_CONFIG_CONTRACT["arn_keys"]) -#: The key whose ARN supplies the partition/account every other ARN is checked -#: against. -#: -#: Deliberately a payload key rather than ``os.environ`` or an STS call. -#: ``os.environ`` is empty here by construction (nothing is baked into the -#: snapshot — see ``imageEnvironmentVariables``), so anchoring on the environment -#: would silently degrade this whole check to shape-only in the intended -#: deployment. An ``sts:GetCallerIdentity`` is not available either: this runs -#: BEFORE ``platform_config`` is installed, on the path that must make zero AWS -#: calls beyond the S3 payload fetch. -#: -#: CONSEQUENCE, stated plainly: because the anchor travels in the same block as the -#: values it validates, this check enforces INTERNAL CONSISTENCY of the block, not -#: agreement with the account the guest is actually running in. A block that names -#: one foreign account throughout is self-consistent and passes here — it then -#: fails at IAM, which is the control that really holds (see -#: :data:`MICROVM_PLATFORM_CONFIG_ARN_KEYS`). ``agent_session_role_arn`` is still -#: the best available anchor: it is REQUIRED (so always present when this check -#: runs, which is what stops disarm-by-omission) and it is the one value whose -#: misdirection costs the attacker the run rather than gaining them anything — a -#: foreign session role fails closed at ``sts:AssumeRole`` -#: (``SessionScopingError``). +#: Consistency anchor after manifest authentication, not deployment identity. +#: Required membership is enforced by the shared contract. MICROVM_PLATFORM_CONFIG_ACCOUNT_ANCHOR_KEY: str = _PLATFORM_CONFIG_CONTRACT["account_anchor_key"] _PLATFORM_CONFIG_KEY_RE = re.compile(r"^[a-z][a-z0-9_]*$") @@ -1138,25 +1082,6 @@ def __init__(self, code: str, message: str) -> None: self.code = code -def _absent_required_platform_env() -> list[str]: - """Required ``platform_config`` env vars that are unset in the LIVE environment. - - The no-``platform_config`` path's audit. Checks the ENV VAR names rather than - the contract keys, because on that path the only possible source is whatever - the image snapshot baked — so the effective environment is the thing to - interrogate, and a value that arrived by any route counts. - - Blank/whitespace-only counts as absent, matching - :func:`_install_platform_config`'s own rule: CloudFormation renders an - unresolved value as ``""``, and an empty table name is not a table name. - """ - return sorted( - MICROVM_PLATFORM_CONFIG_ENV_BY_KEY[key] - for key in MICROVM_PLATFORM_CONFIG_REQUIRED_KEYS - if not os.environ.get(MICROVM_PLATFORM_CONFIG_ENV_BY_KEY[key], "").strip() - ) - - def _reject_foreign_arns(resolved: dict[str, str]) -> None: """Require every ARN-shaped value to agree with the anchor's partition + account. @@ -1459,173 +1384,27 @@ def _parse_terminate_microvm_id(raw: bytes) -> str: class MicrovmRunHookRequest(BaseModel): - """Body the MicroVM service POSTs to the ``/run`` hook. - - ``runHookPayload`` is the opaque STRING the orchestrator passed to - ``RunMicrovm`` — the service does not parse it. ABCA's contract for that - string (``lambda-microvm-strategy.ts``) is one of two shapes, mirroring the - ECS container env contract (``AGENT_PAYLOAD`` / ``AGENT_PAYLOAD_S3_URI``): - - * ``{"agent_payload": {...}, "platform_config": {...}}`` — inline. - * ``{"agent_payload_s3_uri": "s3://bucket/key", "platform_config": {...}}`` — - a pointer; the object at the URI carries the task payload (and a - ``platform_config`` copy, so either end of the fetch yields it). - - The pointer form is the DOMINANT one: the service caps ``runHookPayload`` at - 4 096 bytes and a hydrated payload is essentially always larger. - - ``platform_config`` (ADR-021 P2) is a SIBLING of ``agent_payload``, not a - field inside it: it configures the agent's *process*, whereas - ``agent_payload`` describes the *task* (``memory_id`` and friends stay - inside ``agent_payload``, unchanged). See ``_install_platform_config``. - - Both fields default to empty so a malformed call produces this module's - structured 400 rather than FastAPI's 422 — the service surfaces a 4xx as a - generic "client error" hook failure either way, and our own body is what ends - up in the MicroVM log group. + """Service request containing a serialized v2 payload reference. + + The reference identifies an IAM-authenticated deployment manifest and one + task's signed download URL. Unsigned inline and legacy S3 envelopes are + refused: this protocol requires a matching coordinator and agent image. + Missing fields reach our structured 400 instead of FastAPI's 422. """ microvmId: str = "" # service field name; camelCase on the wire runHookPayload: str = "" # service field name; camelCase on the wire -class _PayloadFetchError(Exception): - """A fetched ``/run`` payload that could not be read or decoded. - - Exists purely to be *not* a ``ValueError``, because the ``/run`` handler - discriminates its 400 from its 500 on exactly that type and the two answers - make opposite promises to the operator: - - * 400 ``MICROVM_RUN_PAYLOAD_INVALID`` — "the orchestrator built a bad - envelope; retrying an identical body cannot help." - * 500 ``MICROVM_RUN_PAYLOAD_UNREADABLE`` — "the payload could not be read; - retrying CAN help." - - Corrupt/truncated JSON or an interrupted body read is the SECOND kind, but - ``json.JSONDecodeError`` and a closed stream's error are ``ValueError`` - subclasses. Without this wrapper, the handler mistakes those fetch/decode - failures for malformed hook envelopes. Only the pre-fetch URI-shape check - should reach the handler's ``ValueError`` branch. - S3 publishes object writes atomically: a malformed stored object needs - replacement, and retrying the same bytes will not repair them. - """ +def _resolve_microvm_run_payload(run_hook_payload: str) -> tuple[dict, dict]: + """Authenticate deployment settings and resolve the v2 task reference.""" + from payload_bootstrap import resolve_payload_reference - -def _fetch_microvm_payload_from_s3(uri: str) -> dict: - """Read and parse the out-of-band ``/run`` payload from S3. - - Same fetch the ECS boot command performs for ``AGENT_PAYLOAD_S3_URI``; the - MicroVM **execution role** holds the read grant, scoped to the platform - payload bucket. Errors propagate to the caller, which turns them into a - structured 400/500 — silently starting a pipeline with no payload would - produce a task that runs with an empty prompt. - - The URI-SHAPE check raises ``ValueError`` (the orchestrator's envelope is - wrong → 400). Everything AFTER the fetch raises :class:`_PayloadFetchError` - (the object is wrong → 500, retryable). See that class. - - Built through ``aws_session.platform_client`` so the call carries the ABCA - ``md/`` solution-attribution segment (#319). Platform, not tenant: the bucket - is platform-owned and — decisively — this is the ONE call that must happen - BEFORE ``platform_config`` is installed (the config is inside the object - being fetched), so ``AGENT_SESSION_ROLE_ARN`` may not be set yet and a - tenant-scoped client could not be built. ``platform_client`` does not touch - the cached session, so this call also cannot pin an unscoped session for the - rest of the task. The ``app/`` UA segment (native, from ``AWS_SDK_UA_APP_ID``) - is the one attribution field this single call can miss for the same - chicken-and-egg reason. - """ - remainder = uri[len(_S3_URI_SCHEME) :] - bucket, _, key = remainder.partition("/") - if not bucket or not key: - raise ValueError(f"agent_payload_s3_uri is not a bucket/key URI: {uri!r}") - - from aws_session import platform_client - - region = os.environ.get("AWS_REGION") or os.environ.get("AWS_DEFAULT_REGION") - client = platform_client("s3", region_name=region) try: - body = client.get_object(Bucket=bucket, Key=key)["Body"].read() - payload = json.loads(body) - except ValueError as exc: - # Decode failures and a closed response stream can both raise ValueError. - # Keep them out of the handler's malformed-envelope (400) branch. - raise _PayloadFetchError( - f"S3 payload at {uri!r} could not be read as JSON ({exc})" - ) from exc - if not isinstance(payload, dict): - raise _PayloadFetchError( - f"S3 payload at {uri!r} is {type(payload).__name__}, expected an object" - ) - return payload - - -def _resolve_microvm_run_payload(run_hook_payload: str) -> tuple[dict, Any]: - """Split the ``runHookPayload`` string into (agent payload, platform config). - - The second element is returned RAW (unvalidated) — ``_install_platform_config`` - owns its allowlist checks so the two failure classes get distinct wire codes. - ``None`` means the envelope carried no ``platform_config`` at all. - - Raises ``ValueError`` for every ENVELOPE shape the agent cannot act on — the - caller maps that onto its 400 ("the orchestrator built this; a retry cannot - help"). Problems with the CONTENT of a fetched S3 object raise - :class:`_PayloadFetchError` instead, which the caller maps onto its retryable - 500: the orchestrator's envelope was fine and the object was not. - """ - if not run_hook_payload.strip(): - raise ValueError("runHookPayload is empty") - - try: - envelope = json.loads(run_hook_payload) - except json.JSONDecodeError as exc: - raise ValueError(f"runHookPayload is not valid JSON: {exc}") from exc - - if not isinstance(envelope, dict): - raise ValueError(f"runHookPayload must be a JSON object, got {type(envelope).__name__}") - - inline = envelope.get("agent_payload") - if inline is not None: - if not isinstance(inline, dict): - raise ValueError(f"agent_payload must be an object, got {type(inline).__name__}") - return inline, envelope.get("platform_config") - - uri = envelope.get("agent_payload_s3_uri") - if isinstance(uri, str) and uri.startswith(_S3_URI_SCHEME): - fetched = _fetch_microvm_payload_from_s3(uri) - # ``platform_config`` may sit beside the pointer (outer envelope) or - # inside the fetched object — the producer writes it in BOTH places on - # this path deliberately, so the agent gets it whichever end it reads. - # Inner first, outer as the fallback. - platform_config = fetched.get("platform_config") - if platform_config is None: - platform_config = envelope.get("platform_config") - # The fetched object is EITHER the same envelope shape as the inline form - # ({"agent_payload": …}) or the task payload itself with ``platform_config`` - # merged in at the top level (what the strategy writes today, and what P1 - # wrote without the config). Both are accepted because the image snapshot - # and the orchestrator Lambda deploy on independent cadences — a new image - # must not require a same-instant orchestrator. Discriminating on the - # ``agent_payload`` key is unambiguous: no orchestrator task payload has a - # field by that name. A stray ``platform_config`` key left in the bare - # form is inert — ``_extract_invocation_params`` reads named fields only. - nested = fetched.get("agent_payload") - if nested is None: - return fetched, platform_config - if not isinstance(nested, dict): - # Content of the FETCHED OBJECT, not of the envelope — so this is the - # retryable class, same as a truncated body. See ``_PayloadFetchError``. - raise _PayloadFetchError( - f"agent_payload in the S3 payload must be an object, got {type(nested).__name__}" - ) - return nested, platform_config - if uri is not None: - raise ValueError(f"agent_payload_s3_uri must be an s3:// URI, got {uri!r}") - - raise ValueError( - "runHookPayload envelope has neither agent_payload nor agent_payload_s3_uri " - f"(keys: {sorted(envelope)})" - ) + reference = json.loads(run_hook_payload) + except (UnicodeError, ValueError) as exc: + raise ValueError("runHookPayload is not valid JSON") from exc + return resolve_payload_reference(reference, "lambda-microvm") #: The ONE executable whose warm-up gates the snapshot, exec'd FIRST. @@ -2076,7 +1855,7 @@ def microvm_run(request: Request, body: MicrovmRunHookRequest): mechanism ``/invocations`` uses — ``_extract_invocation_params`` → ``_validate_required_params`` → ``_spawn_background`` — rather than a second, drifting payload mapper. The orchestrator payload is byte-identical across - substrates (AgentCore receives it as ``input``, ECS as ``AGENT_PAYLOAD``, + substrates (AgentCore receives it as ``input``, ECS through the authenticated payload reference, MicroVMs inside this envelope), which is what makes that reuse correct. ``platform_config`` (P2) is installed into ``os.environ`` FIRST — before @@ -2093,15 +1872,15 @@ def microvm_run(request: Request, body: MicrovmRunHookRequest): has, per ADR-021 sub-decision 3's identity delta. Sync ``def`` for the same reason as ``/ready``, and additionally because the - S3 payload fetch is a blocking boto3 call: in a threadpool it cannot stall + manifest read and signed download are blocking calls: in a threadpool it cannot stall the event loop. **Every log line before the install goes through ``_pre_config_log``** (stdout only). Until ``platform_config`` is in the environment, a ``_debug_cw`` here would resolve AWS credentials and pin ``boto3.DEFAULT_SESSION`` off whatever the snapshot happens to carry — the same defect the build hooks avoid, one - phase later. The single AWS call this phase is allowed to make is the S3 - payload fetch, because the config is inside the object being fetched. + phase later. Pre-install operations are the IAM-authenticated manifest read + and the single-object HTTPS download. Build hooks remain AWS-silent. """ _pre_config_log( f"/run hook received: microvm_id={body.microvmId!r} bytes={len(body.runHookPayload)}" @@ -2122,16 +1901,14 @@ def microvm_run(request: Request, body: MicrovmRunHookRequest): ) except Exception as exc: # Payload could not be READ: S3 AccessDenied / NoSuchKey / transient, or a - # `_PayloadFetchError` for an object that fetched but was truncated, + # `PayloadFetchError` for an object that fetched but was truncated, # non-JSON, or not an object. 500 so the failure is distinguishable from a - # malformed ENVELOPE (the 400 above) and is correctly reported as - # retryable, and loud enough to find in the MicroVM log group — via the - # response body, since the CloudWatch writer is off-limits until the config - # is installed. - _pre_config_log( - f"/run hook payload fetch FAILED [{type(exc).__name__}: {exc}]\n" - f"{traceback.format_exc()}" - ) + # malformed reference (the 400 above). The response body preserves the + # distinction while CloudWatch is off-limits before config installation. + # Corrupt stored bytes require repair, not an assumption that retry helps. + # A chained HTTP error may contain the bearer URL. Log only the + # bootstrap reader's sanitized message, never its exception chain. + _pre_config_log(f"/run hook payload fetch FAILED [{type(exc).__name__}: {exc}]") return JSONResponse( status_code=500, content={ @@ -2143,6 +1920,11 @@ def microvm_run(request: Request, body: MicrovmRunHookRequest): task_id_log = str(payload.get("task_id", "")) try: + if platform_config is None: + raise _PlatformConfigError( + "MICROVM_RUN_PLATFORM_CONFIG_INVALID", + "Authenticated platform configuration is required", + ) installed_env = _install_platform_config(platform_config) except _PlatformConfigError as exc: _pre_config_log(f"/run hook rejected: {exc}") @@ -2160,55 +1942,6 @@ def microvm_run(request: Request, body: MicrovmRunHookRequest): f"/run hook installed platform_config env: {installed_env}", task_id=task_id_log or None, ) - else: - # No `platform_config` — the legacy P1 envelope. This branch must NOT simply - # shrug: the required keys exist because without them the agent cannot write - # status/progress, resolve the GitHub token, or (decisively) - # tenant-scope its credentials — `aws_session.get_session` falls back to the - # ambient compute role with scoping silently OFF when - # `AGENT_SESSION_ROLE_ARN` is unset. So the check is re-run against the - # EFFECTIVE environment: a legacy or hand-built image that bakes those - # values still runs (that is the compatibility this branch is for), while - # version skew — a pre-Stage-B orchestrator launching a P2 image, which - # bakes nothing — is REJECTED instead of running unscoped. - # - # STILL pre-install, so the line is stdout only. A `_warn_cw` here would - # spawn the CloudWatch writer thread and pin `boto3.DEFAULT_SESSION` off the - # snapshot's baked env, which is the very defect this branch is reporting. - # Nothing is lost: on the intended deployment (no baked `LOG_GROUP_NAME`) - # `_warn_cw` would have degraded to this same stdout line, on a legacy image - # the log group would be the wrong one anyway, and the rejection reason also - # travels in the structured response body the service surfaces. - absent = _absent_required_platform_env() - if absent: - _pre_config_log( - f"/run hook REJECTED: no platform_config and the image snapshot does " - f"not supply required value(s) either: {absent}" - + (f" task_id={task_id_log!r}" if task_id_log else "") - ) - return JSONResponse( - status_code=400, - content={ - "code": "MICROVM_RUN_PLATFORM_CONFIG_INCOMPLETE", - "message": ( - "The /run envelope carried no platform_config and the image " - "snapshot does not carry the required values either, so this " - f"MicroVM cannot run a task: {absent} are unset. This is a " - "version skew — an orchestrator predating ADR-021 P2 launching " - "a P2 image, which bakes no environment by design. Refusing " - "rather than running with tenant scoping disabled. Redeploy the " - "orchestrator so it sends platform_config." - ), - "missing_env": absent, - }, - ) - _pre_config_log( - "/run hook received no platform_config; running on the image snapshot's " - "own environment, which is frozen at build time and which DOES supply " - "every required value. Expected only from an orchestrator that predates " - "ADR-021 P2 paired with an image that bakes its own configuration." - + (f" task_id={task_id_log!r}" if task_id_log else "") - ) try: params = _extract_invocation_params(payload, request) diff --git a/agent/tests/test_payload_bootstrap.py b/agent/tests/test_payload_bootstrap.py new file mode 100644 index 000000000..04af816b0 --- /dev/null +++ b/agent/tests/test_payload_bootstrap.py @@ -0,0 +1,301 @@ +"""Real bootstrap parsing/stream handling with AWS and HTTPS service boundaries stubbed.""" + +import hashlib +import json +import traceback +from email.message import Message +from io import BytesIO +from unittest.mock import MagicMock +from urllib.error import HTTPError +from urllib.parse import urlencode + +import pytest +from botocore.response import StreamingBody +from fastapi.testclient import TestClient + +import aws_session +import payload_bootstrap as bootstrap +import server + + +class HttpBody(BytesIO): + def __init__(self, raw: bytes, length: int): + super().__init__(raw) + self.headers = {"Content-Length": str(length)} + + +def signed_url(task_id="task-1", bucket="payload-bucket", region="us-east-1"): + suffix = "amazonaws.com.cn" if region.startswith("cn-") else "amazonaws.com" + query = urlencode( + { + "X-Amz-Algorithm": "AWS4-HMAC-SHA256", + "X-Amz-Credential": f"EXAMPLE/20260913/{region}/s3/aws4_request", + "X-Amz-Date": "20260913T120000Z", + "X-Amz-Expires": "900", + "X-Amz-SignedHeaders": "host", + "X-Amz-Signature": "a" * 64, + "X-Amz-Security-Token": "BEARER-SECRET", + } + ) + return f"https://{bucket}.s3.{region}.{suffix}/{task_id}/payload.json?{query}" + + +@pytest.fixture +def transport(monkeypatch): + config = { + "task_table_name": "Tasks", + "task_events_table_name": "Events", + "agent_session_role_arn": "arn:aws:iam::123456789012:role/AgentSession", + "github_token_secret_arn": ( + "arn:aws:secretsmanager:us-west-2:123456789012:secret:github-token-123456" + ), + } + manifest = {"version": 2, "backend": "lambda-microvm", "platform_config": config} + raw_manifest = json.dumps(manifest).encode() + key = "bootstrap/" + hashlib.sha256(raw_manifest).hexdigest() + ".json" + reference = { + "version": 2, + "task_id": "task-1", + "bootstrap_s3_uri": f"s3://payload-bucket/{key}", + "payload_url": signed_url(), + "expires_at": 1800000000000, + } + payload = { + "task_id": "task-1", + "repo_url": "org/repo", + "prompt": "do it", + "github_token": "fake", + } + document = { + "version": 2, + "task_id": "task-1", + "agent_payload": payload, + "platform_config": config, + } + streams = [] + s3 = MagicMock() + + def get_object(**kwargs): + assert kwargs == {"Bucket": "payload-bucket", "Key": key} + body = StreamingBody(BytesIO(raw_manifest), len(raw_manifest)) + streams.append(body) + return {"Body": body, "ContentLength": len(raw_manifest)} + + s3.get_object.side_effect = get_object + platform = MagicMock(return_value=s3) + monkeypatch.setattr(aws_session, "platform_client", platform) + opener = MagicMock() + + def response(_url, **kwargs): + assert kwargs == {"timeout": 10} + raw = json.dumps(document).encode() + return HttpBody(raw, len(raw)) + + opener.open.side_effect = response + build = MagicMock(return_value=opener) + monkeypatch.setattr(bootstrap, "build_opener", build) + return { + "reference": reference, + "config": config, + "payload": payload, + "document": document, + "manifest": manifest, + "s3": s3, + "platform": platform, + "opener": opener, + "build": build, + "streams": streams, + } + + +def test_authenticates_manifest_before_downloading_and_preserves_cross_region_secret(transport): + payload, config = bootstrap.resolve_payload_reference(transport["reference"], "lambda-microvm") + assert payload == transport["payload"] + assert config == transport["config"] + assert ":us-west-2:" in config["github_token_secret_arn"] + transport["platform"].assert_called_once() + assert transport["platform"].call_args.args == ("s3",) + assert len(transport["streams"]) == 1 + assert transport["streams"][0]._raw_stream.closed + assert transport["opener"].open.call_args.args == (transport["reference"]["payload_url"],) + assert isinstance(transport["build"].call_args.args[1], bootstrap._NoRedirect) + assert transport["build"].call_args.args[0].proxies == {} + + +@pytest.mark.parametrize("reference", [{}, {"agent_payload": {}}, {"version": 1}, [], "raw"]) +def test_legacy_and_malformed_envelopes_never_read_or_start(reference, transport): + with pytest.raises(ValueError, match="v2 is required"): + bootstrap.resolve_payload_reference(reference, "lambda-microvm") + transport["platform"].assert_not_called() + + +@pytest.mark.parametrize( + "uri", + [ + "http://169.254.169.254/metadata", + "s3://payload-bucket/task-2/payload.json", + "s3://payload-bucket/bootstrap/../task-2/payload.json", + "s3://payload-bucket/bootstrap/not-a-digest.json", + "s3://payload-bucket/bootstrap/" + "a" * 64 + ".json?versionId=evil", + ], +) +def test_manifest_shape_rejects_other_objects_before_aws(transport, uri): + transport["reference"]["bootstrap_s3_uri"] = uri + with pytest.raises(ValueError): + bootstrap.resolve_payload_reference(transport["reference"], "lambda-microvm") + transport["platform"].assert_not_called() + + +def test_foreign_manifest_denial_prevents_download_and_config_install( + transport, monkeypatch, capfd +): + transport["s3"].get_object.side_effect = PermissionError("AccessDenied") + install = MagicMock() + spawn = MagicMock() + monkeypatch.setattr(server, "_install_platform_config", install) + monkeypatch.setattr(server, "_spawn_background", spawn) + with TestClient(server.app) as client: + response = client.post( + server.MICROVM_HOOK_PREFIX + "/run", + json={ + "microvmId": "mvm", + "runHookPayload": json.dumps(transport["reference"]), + }, + ) + assert response.status_code == 500 + assert response.json()["code"] == "MICROVM_RUN_PAYLOAD_UNREADABLE" + install.assert_not_called() + spawn.assert_not_called() + transport["opener"].open.assert_not_called() + assert "BEARER-SECRET" not in capfd.readouterr().out + + +def test_run_hook_never_logs_the_chained_http_error_url(transport, monkeypatch, capfd): + url = transport["reference"]["payload_url"] + transport["opener"].open.side_effect = HTTPError(url, 403, url, Message(), None) + install = MagicMock() + monkeypatch.setattr(server, "_install_platform_config", install) + with TestClient(server.app) as client: + response = client.post( + server.MICROVM_HOOK_PREFIX + "/run", + json={ + "microvmId": "mvm", + "runHookPayload": json.dumps(transport["reference"]), + }, + ) + assert response.status_code == 500 + install.assert_not_called() + assert "BEARER-SECRET" not in response.text + assert "BEARER-SECRET" not in capfd.readouterr().out + + +@pytest.mark.parametrize( + "mutate", + [ + lambda d: d["platform_config"].update( + github_token_secret_arn="arn:aws:secretsmanager:us-west-2:123456789012:secret:bgagent-linear-oauth-victim" + ), + lambda d: d.update(task_id="task-2"), + lambda d: d["agent_payload"].update(task_id="task-2"), + lambda d: d.update(platform_config=None), + ], +) +def test_same_account_workspace_redirect_or_task_substitution_is_rejected(transport, mutate): + # Detach the document from the fixture's authenticated manifest values. + transport["document"]["platform_config"] = dict(transport["config"]) + mutate(transport["document"]) + with pytest.raises(ValueError): + bootstrap.resolve_payload_reference(transport["reference"], "lambda-microvm") + + +@pytest.mark.parametrize( + "url", + [ + "http://169.254.169.254/latest/meta-data", + signed_url(task_id="task-2"), + signed_url(bucket="another-bucket"), + signed_url().replace("amazonaws.com/", "amazonaws.com.evil.example/"), + signed_url().replace("https://", "https://user@"), + signed_url().replace("amazonaws.com/", "amazonaws.com:443/"), + signed_url() + "#fragment", + signed_url() + "&X-Amz-Signature=" + "b" * 64, + signed_url().replace("X-Amz-Expires=900", "X-Amz-Expires=999999"), + ], +) +def test_download_reference_rejects_wrong_task_hosts_redirect_coordinates_and_ambiguity( + transport, url +): + transport["reference"]["payload_url"] = url + with pytest.raises(ValueError, match="exact S3 object"): + bootstrap.resolve_payload_reference(transport["reference"], "lambda-microvm") + transport["opener"].open.assert_not_called() + + +@pytest.mark.parametrize( + "raw,length", + [ + (b'{"truncated":', 13), + (b"\xff", 1), + (b"[]", 2), + (b"{}", 40), + (b"{}", 0), + (b"{}", bootstrap.CONTRACT["max_payload_bytes"] + 1), + ], +) +def test_bad_payload_bytes_are_unreadable_not_bad_envelopes(transport, raw, length): + body = HttpBody(raw, length) + transport["opener"].open.side_effect = None + transport["opener"].open.return_value = body + with pytest.raises(bootstrap.PayloadFetchError): + bootstrap.resolve_payload_reference(transport["reference"], "lambda-microvm") + assert body.closed + + +@pytest.mark.parametrize("mode", ["digest", "short", "closed", "too_large", "array"]) +def test_bad_manifest_streams_fail_before_download_and_are_closed(transport, mode): + raw = b"[]" if mode == "array" else b"{}" + length = 100 if mode == "short" else len(raw) + if mode == "too_large": + length = bootstrap.CONTRACT["max_manifest_bytes"] + 1 + body = StreamingBody(BytesIO(raw), length) + if mode == "closed": + body.close() + transport["s3"].get_object.side_effect = None + transport["s3"].get_object.return_value = {"Body": body, "ContentLength": length} + with pytest.raises(bootstrap.PayloadFetchError): + bootstrap.resolve_payload_reference(transport["reference"], "lambda-microvm") + assert body._raw_stream.closed + transport["opener"].open.assert_not_called() + + +@pytest.mark.parametrize("code", [301, 307, 403, 500]) +def test_redirect_expiry_and_service_errors_do_not_echo_capabilities(transport, code): + url = transport["reference"]["payload_url"] + transport["opener"].open.side_effect = HTTPError(url, code, url, Message(), None) + with pytest.raises(bootstrap.PayloadFetchError, match=f"HTTP {code}") as error: + bootstrap.resolve_payload_reference(transport["reference"], "lambda-microvm") + assert "BEARER-SECRET" not in str(error.value) + # ECS's Python boot command can print an uncaught exception. Its formatted + # traceback must be safe too, not just the MicroVM route's JSON response. + assert "BEARER-SECRET" not in "".join(traceback.format_exception(error.value)) + assert bootstrap._NoRedirect().redirect_request(None, None, code, "", {}, url) is None + + +def test_ecs_consumes_reference_before_running_code(transport, monkeypatch): + monkeypatch.setenv("TASK_ID", "task-1") + monkeypatch.setenv("AGENT_PAYLOAD_REF", json.dumps(transport["reference"])) + resolve = MagicMock(return_value=(transport["payload"], {})) + monkeypatch.setattr(bootstrap, "resolve_payload_reference", resolve) + assert bootstrap.load_ecs_payload() == transport["payload"] + import os + + assert "AGENT_PAYLOAD_REF" not in os.environ + assert resolve.call_args.args[1] == "ecs" + + +def test_ecs_rejects_another_tasks_reference(transport, monkeypatch): + monkeypatch.setenv("TASK_ID", "task-2") + monkeypatch.setenv("AGENT_PAYLOAD_REF", json.dumps(transport["reference"])) + with pytest.raises(ValueError, match="ECS task identity"): + bootstrap.load_ecs_payload() + transport["platform"].assert_not_called() diff --git a/agent/tests/test_server.py b/agent/tests/test_server.py index af6b912fa..691bb9a84 100644 --- a/agent/tests/test_server.py +++ b/agent/tests/test_server.py @@ -8,9 +8,7 @@ import sys import threading import time -from io import BytesIO from pathlib import Path -from types import SimpleNamespace from typing import Any from unittest.mock import MagicMock @@ -940,13 +938,45 @@ def _platform_config(**overrides) -> dict: return config -def _run_hook_body(envelope: dict, microvm_id: str = "microvm-abc") -> dict: - """Wrap an ABCA payload envelope in the service's ``/run`` request body. +# Route tests isolate the authenticated transport boundary. The real manifest +# IAM/stream/URL consumer is exercised in test_payload_bootstrap.py. +_run_documents: dict[str, tuple[dict, Any]] = {} - The service passes ``runHookPayload`` through as an opaque STRING (it never - parses it), so the double encoding here is the real wire shape, not a test - artifact. - """ + +@pytest.fixture(autouse=True) +def _authenticated_test_transport(monkeypatch): + import payload_bootstrap + + _run_documents.clear() + monkeypatch.setattr( + payload_bootstrap, + "_manifest", + lambda uri, backend: ("payload-bucket", _run_documents[uri][1]), + ) + monkeypatch.setattr(payload_bootstrap, "_download", lambda url: _run_documents[url][0]) + yield + _run_documents.clear() + + +def _run_hook_body(envelope: dict, microvm_id: str = "microvm-abc") -> dict: + """Wrap pipeline/config fixtures in the current v2 service envelope.""" + from tests.test_payload_bootstrap import signed_url + + if "agent_payload" in envelope and isinstance(envelope["agent_payload"], dict): + payload = envelope["agent_payload"] + task_id = payload.get("task_id", "invalid") + config = envelope.get("platform_config", _platform_config()) + uri = f"s3://payload-bucket/bootstrap/{'a' * 64}.json" + url = signed_url(task_id=task_id) + document = { + "version": 2, + "task_id": task_id, + "agent_payload": payload, + "platform_config": config, + } + _run_documents[uri] = (document, config) + _run_documents[url] = (document, config) + envelope = {"version": 2, "task_id": task_id, "bootstrap_s3_uri": uri, "payload_url": url} return {"microvmId": microvm_id, "runHookPayload": json.dumps(envelope)} @@ -1334,21 +1364,16 @@ def fake_run(argv, **kwargs): assert "skipping best-effort warm-up of 'opt'" in capfd.readouterr().out -class TestMicrovmRunHookInlinePayload: - """Inline envelope: ``{"agent_payload": {...}}``. +class TestMicrovmRunHookVerifiedPayload: + """Map an authenticated v2 payload to the asynchronous task pipeline. - The exception rather than the rule — the service caps ``runHookPayload`` at - 4 096 bytes and a hydrated payload is larger — but it is the branch that - proves the payload→pipeline mapping without any S3 involvement. + The module fixture supplies manifest/download bytes; the shared consumer's + authentication and transport checks have their own regression suite. """ def test_accepts_the_payload_and_starts_the_pipeline_asynchronously( - self, client, monkeypatch, baked_platform_env + self, client, monkeypatch, cached_github_token ): - # `baked_platform_env`: this envelope carries no `platform_config`, which - # since review N2 requires the effective env to supply the required values - # (a legacy image that bakes its own). This class is about the - # payload->pipeline mapping, not about config delivery. started = threading.Event() seen: dict = {} @@ -1371,7 +1396,7 @@ def fake_run_task(**kwargs): "aws_region": "us-east-1", } }, - microvm_id="microvm-inline", + microvm_id="microvm-verified", ), ) @@ -1380,7 +1405,7 @@ def fake_run_task(**kwargs): assert body["status"] == "accepted" assert body["task_id"] == "t-microvm-1" # Echoed so a MicroVM log line can be joined to the control-plane id. - assert body["microvm_id"] == "microvm-inline" + assert body["microvm_id"] == "microvm-verified" assert started.wait(timeout=5.0), "pipeline thread did not start" # Same mapping the /invocations path performs: prompt→task_description, @@ -1389,11 +1414,7 @@ def fake_run_task(**kwargs): assert seen["repo_url"] == "org/repo" assert seen["task_description"] == "Fix the bug" - def test_returns_before_the_pipeline_finishes(self, client, monkeypatch, baked_platform_env): - # `baked_platform_env`: this envelope carries no `platform_config`, which - # since review N2 requires the effective env to supply the required values - # (a legacy image that bakes its own). This class is about the - # payload->pipeline mapping, not about config delivery. + def test_returns_before_the_pipeline_finishes(self, client, monkeypatch, cached_github_token): release = threading.Event() entered = threading.Event() @@ -1421,12 +1442,8 @@ def slow_run_task(**_kwargs): release.set() def test_uses_the_same_model_id_and_prompt_aliases_as_invocations( - self, client, monkeypatch, baked_platform_env + self, client, monkeypatch, cached_github_token ): - # `baked_platform_env`: this envelope carries no `platform_config`, which - # since review N2 requires the effective env to supply the required values - # (a legacy image that bakes its own). This class is about the - # payload->pipeline mapping, not about config delivery. seen: dict = {} started = threading.Event() @@ -1458,171 +1475,6 @@ def fake_run_task(**kwargs): assert seen["channel_source"] == "linear" -class TestMicrovmRunHookS3Payload: - """S3-pointer envelope: ``{"agent_payload_s3_uri": "s3://bucket/key"}``. - - The DOMINANT path on this backend: with a 4 096-byte ``runHookPayload`` cap, - any hydrated payload is offloaded to the platform payload bucket and only the - pointer travels in the hook body. - """ - - @pytest.mark.parametrize( - "body_bytes,stream_fault", - [ - (b'{"platform_config":{"task_table_name":"must-not-install"},"task_id":', None), - (b'{"task_id":"bad-encoding-\xff"}', None), - (b'{"task_id":"short-stream"}', "incomplete"), - (b'{"task_id":"closed-stream"}', "closed"), - (b"[1,2,3]", None), - ], - ids=["truncated-json", "invalid-encoding", "incomplete-stream", "closed-stream", "array"], - ) - def test_bad_s3_bytes_use_the_real_fetch_path_and_start_nothing( - self, client, monkeypatch, body_bytes, stream_fault - ): - from botocore.response import StreamingBody - - import aws_session - - raw = BytesIO(body_bytes) - if stream_fault == "closed": - raw.close() - body = StreamingBody(raw, len(body_bytes) + (stream_fault == "incomplete")) - s3 = MagicMock() - s3.get_object.return_value = {"Body": body} - factory = MagicMock(return_value=s3) - run_task = MagicMock() - install_config = MagicMock(wraps=server._install_platform_config) - monkeypatch.setattr(aws_session, "platform_client", factory) - monkeypatch.setattr(server, "run_task", run_task) - monkeypatch.setattr(server, "_install_platform_config", install_config) - before_env = dict(os.environ) - - response = client.post( - RUN_HOOK, - json=_run_hook_body( - { - "agent_payload_s3_uri": "s3://payload-bucket/t-bad/payload.json", - "platform_config": {"task_table_name": "outer-must-not-install"}, - } - ), - ) - - assert response.status_code == 500 - assert response.json()["code"] == "MICROVM_RUN_PAYLOAD_UNREADABLE" - assert response.json()["message"] - s3.get_object.assert_called_once_with(Bucket="payload-bucket", Key="t-bad/payload.json") - assert factory.call_args.args == ("s3",) - install_config.assert_not_called() - run_task.assert_not_called() - assert dict(os.environ) == before_env - with server._threads_lock: - assert server._active_threads == [] - - def test_fetches_the_payload_from_s3_and_starts_the_pipeline( - self, client, monkeypatch, baked_platform_env - ): - # `baked_platform_env`: this envelope carries no `platform_config`, which - # since review N2 requires the effective env to supply the required values - # (a legacy image that bakes its own). This class is about the - # payload->pipeline mapping, not about config delivery. - seen: dict = {} - started = threading.Event() - - def fake_run_task(**kwargs): - seen.update(kwargs) - started.set() - - fetched: dict = {} - - def fake_fetch(uri): - fetched["uri"] = uri - return {"task_id": "t-s3", "repo_url": "org/repo", "prompt": "from s3"} - - monkeypatch.setattr(server, "_fetch_microvm_payload_from_s3", fake_fetch) - monkeypatch.setattr(server, "run_task", fake_run_task) - monkeypatch.setattr(server.task_state, "write_terminal", MagicMock()) - - r = client.post( - RUN_HOOK, - json=_run_hook_body({"agent_payload_s3_uri": "s3://payload-bucket/t-s3/payload.json"}), - ) - - assert r.status_code == 200 - assert r.json()["task_id"] == "t-s3" - assert fetched["uri"] == "s3://payload-bucket/t-s3/payload.json" - assert started.wait(timeout=5.0) - assert seen["task_description"] == "from s3" - - def test_parses_bucket_and_key_out_of_the_uri(self, monkeypatch): - captured: dict = {} - - class _Body: - @staticmethod - def read(): - return b'{"task_id": "t-1", "repo_url": "o/r"}' - - class _S3: - @staticmethod - def get_object(**kwargs): - captured.update(kwargs) - return {"Body": _Body} - - import boto3 - - monkeypatch.setattr(boto3, "client", lambda *_a, **_k: _S3) - - payload = server._fetch_microvm_payload_from_s3("s3://my-bucket/prefix/t-1/payload.json") - - # Key keeps every slash after the bucket — a naive split would truncate it. - assert captured == {"Bucket": "my-bucket", "Key": "prefix/t-1/payload.json"} - assert payload == {"task_id": "t-1", "repo_url": "o/r"} - - def test_rejects_a_uri_with_no_key(self, monkeypatch): - with pytest.raises(ValueError, match="not a bucket/key URI"): - server._fetch_microvm_payload_from_s3("s3://bucket-only") - - def test_rejects_a_non_object_s3_body(self, monkeypatch): - class _Body: - @staticmethod - def read(): - return b"[1, 2, 3]" - - class _S3: - @staticmethod - def get_object(**_kwargs): - return {"Body": _Body} - - import boto3 - - monkeypatch.setattr(boto3, "client", lambda *_a, **_k: _S3) - - # `_PayloadFetchError`, NOT `ValueError` (review N1): the object is bad, not - # the orchestrator's envelope, so the handler must route it to its RETRYABLE - # 500 rather than the "retrying cannot help" 400. Asserting the type is the - # point — `ValueError` here would silently restore the misclassification. - with pytest.raises(server._PayloadFetchError, match="expected an object"): - server._fetch_microvm_payload_from_s3("s3://b/k") - assert not issubclass(server._PayloadFetchError, ValueError) - - def test_s3_failure_returns_500_and_starts_nothing(self, client, monkeypatch): - def boom(_uri): - raise RuntimeError("AccessDenied") - - monkeypatch.setattr(server, "_fetch_microvm_payload_from_s3", boom) - monkeypatch.setattr(server, "run_task", MagicMock()) - - r = client.post(RUN_HOOK, json=_run_hook_body({"agent_payload_s3_uri": "s3://bucket/key"})) - - # 500, not 400: the body was well-formed, the fetch was not. Retrying an - # identical body CAN help here, unlike a malformed envelope. - assert r.status_code == 500 - assert r.json()["code"] == "MICROVM_RUN_PAYLOAD_UNREADABLE" - assert "AccessDenied" in r.json()["message"] - with server._threads_lock: - assert server._active_threads == [] - - class TestMicrovmRunHookRejections: """Every shape the agent cannot act on must fail LOUDLY, before spawning. @@ -1633,15 +1485,15 @@ class TestMicrovmRunHookRejections: @pytest.mark.parametrize( "run_hook_payload,expected_fragment", [ - ("", "runHookPayload is empty"), - (" ", "runHookPayload is empty"), + ("", "not valid JSON"), + (" ", "not valid JSON"), ("not json at all", "not valid JSON"), - ('"a string"', "must be a JSON object"), - ("[1,2,3]", "must be a JSON object"), - ('{"agent_payload": "not-an-object"}', "agent_payload must be an object"), - ('{"agent_payload_s3_uri": "https://example.com/x"}', "must be an s3:// URI"), - ('{"agent_payload_s3_uri": 42}', "must be an s3:// URI"), - ('{"something_else": 1}', "neither agent_payload nor agent_payload_s3_uri"), + ('"a string"', "v2 is required"), + ("[1,2,3]", "v2 is required"), + ('{"agent_payload": "not-an-object"}', "v2 is required"), + ('{"agent_payload_s3_uri": "https://example.com/x"}', "v2 is required"), + ('{"agent_payload_s3_uri": 42}', "v2 is required"), + ('{"something_else": 1}', "v2 is required"), ], ) def test_returns_400_with_a_named_code( @@ -1670,9 +1522,9 @@ def test_a_missing_body_field_is_a_400_not_a_422(self, client, monkeypatch): assert r.json()["code"] == "MICROVM_RUN_PAYLOAD_INVALID" def test_incomplete_task_record_reuses_the_invocations_rejection_shape( - self, client, monkeypatch, baked_platform_env + self, client, monkeypatch, cached_github_token ): - # `baked_platform_env` so the run gets PAST config delivery and reaches the + # Cache the GitHub token so this test reaches the # task-record check this test is actually about. monkeypatch.setattr(server, "run_task", MagicMock()) @@ -1700,12 +1552,8 @@ class TestMicrovmRunHookHeaderPosture: """ def test_session_id_and_workload_token_resolve_empty( - self, client, monkeypatch, baked_platform_env + self, client, monkeypatch, cached_github_token ): - # `baked_platform_env`: this envelope carries no `platform_config`, which - # since review N2 requires the effective env to supply the required values - # (a legacy image that bakes its own). This class is about the - # payload->pipeline mapping, not about config delivery. seen: dict = {} started = threading.Event() @@ -1811,26 +1659,9 @@ def test_merge_branches_non_string_entries_filtered(self): @pytest.fixture -def baked_platform_env(env_guard): - """Simulate a legacy/hand-built image that BAKES its own required config. - - Needed by every ``/run`` test whose envelope carries no ``platform_config``. - Since review N2 the no-config branch re-runs the required-key check against the - EFFECTIVE environment and rejects when it is unsatisfied — because - ``aws_session`` silently drops tenant scoping when ``AGENT_SESSION_ROLE_ARN`` - is unset, and running a task unscoped is worse than refusing it. That check is - what this fixture satisfies, and satisfying it is exactly what a real legacy - image does: the compatibility path is "the snapshot supplies the values", not - "nobody supplies them". - - Depends on ``env_guard`` so the writes are reverted with everything else. - """ - for key in server.MICROVM_PLATFORM_CONFIG_REQUIRED_KEYS: - os.environ[server.MICROVM_PLATFORM_CONFIG_ENV_BY_KEY[key]] = _platform_config_value(key) - # A baked GITHUB_TOKEN_SECRET_ARN makes `resolve_github_token` reach for real - # Secrets Manager; pre-seeding its cache keeps these tests offline. Realistic - # for the image this fixture models, and it changes nothing they assert. - os.environ["GITHUB_TOKEN"] = "ghp_baked_for_tests" # noqa: S105 -- test placeholder, not a secret +def cached_github_token(env_guard): + """Keep pipeline mapping tests offline after authenticated config installation.""" + os.environ["GITHUB_TOKEN"] = "ghp_cached_for_tests" # noqa: S105 -- test placeholder yield @@ -2325,20 +2156,10 @@ def test_an_arn_in_a_foreign_account_is_rejected(self, env_guard): assert "GITHUB_TOKEN_SECRET_ARN" not in os.environ def test_an_in_account_redirect_is_NOT_rejected(self, env_guard): - # The KNOWN LIMITATION, pinned as a test so it cannot be quietly mistaken for - # coverage. This check compares partition + account only, so a block naming - # another workspace's channel-OAuth secret in the SAME account is accepted — - # and the `bgagent-*-oauth-*` grants are prefix grants, so IAM would allow - # that read too. - # - # It is left open because it is currently unreachable from the guest: - # `platform_config` is produced by the orchestrator Lambda, and the MicroVM - # execution role holds `grantRead` ONLY on the payload bucket, so a running - # MicroVM can read another task's payload but cannot write one. - # - # If this test ever needs to flip to `pytest.raises`, the escalation is a - # name-shape check tying `github_token_secret_arn` to the task's own - # channel/workspace — see `MICROVM_PLATFORM_CONFIG_ARN_KEYS`. + # Unit scope: the installer checks internal ARN agreement, not origin. + # The v2 resolver authenticates the manifest and compares the downloaded + # config BEFORE calling this installer. Its own route/transport tests + # reject same-account workspace substitutions. installed = server._install_platform_config( _platform_config( github_token_secret_arn=( @@ -2350,10 +2171,8 @@ def test_an_in_account_redirect_is_NOT_rejected(self, env_guard): assert "GITHUB_TOKEN_SECRET_ARN" in installed def test_a_wholesale_partition_swap_is_NOT_rejected(self, env_guard): - # The other half of the same limitation, and the reason the docstrings say - # "internally consistent" rather than "pinned to this deployment": the anchor - # travels in the block it validates, so a payload that moves EVERY ARN to - # another partition agrees with itself and passes. IAM is what refuses it. + # A self-consistent block passes this internal-consistency helper. + # Deployment provenance is enforced earlier by the v2 bootstrap resolver. installed = server._install_platform_config( _platform_config( agent_session_role_arn=f"arn:aws-cn:iam::{_TEST_ACCOUNT}:role/r", @@ -2523,7 +2342,7 @@ def _payload(self, **extra) -> dict: **extra, } - def test_inline_envelope_installs_the_config_and_accepts_the_task( + def test_verified_payload_installs_the_config_and_accepts_the_task( self, client, monkeypatch, env_guard ): monkeypatch.setattr(server, "run_task", MagicMock()) @@ -2634,195 +2453,6 @@ def test_a_non_object_block_returns_400(self, client, monkeypatch, env_guard): assert r.status_code == 400 assert r.json()["code"] == "MICROVM_RUN_PLATFORM_CONFIG_INVALID" - def test_an_envelope_without_platform_config_is_accepted_when_the_image_bakes_it( - self, client, monkeypatch, baked_platform_env, capfd - ): - # P1 compatibility, PRECISELY scoped (review N2): image snapshot and - # orchestrator Lambda deploy on independent cadences, so a new image must not - # require a Stage-B orchestrator — PROVIDED the values come from somewhere. - # Here the image bakes them, which is what the compatibility path is for. - monkeypatch.setattr(server, "run_task", MagicMock()) - monkeypatch.setattr(server.task_state, "write_terminal", MagicMock()) - - r = client.post(RUN_HOOK, json=_run_hook_body({"agent_payload": self._payload()})) - - assert r.status_code == 200 - # Still pre-install (nothing was installed), so the warning is stdout-only: - # `_warn_cw` here would spawn the CloudWatch thread off the snapshot's own - # baked env — the very thing it is warning about. See `TestMicrovmRunHook - # PreInstallAwsSilence`. - assert "[server/run-pre-config] /run hook received no platform_config" in ( - capfd.readouterr().out - ) - - def test_no_platform_config_and_no_baked_env_is_refused_not_run_unscoped( - self, client, monkeypatch, env_guard, capfd - ): - # Review N2, the version-skew case: a pre-Stage-B orchestrator launching a P2 - # image. The image bakes NOTHING by design (`imageEnvironmentVariables` - # defaults to `{}`), so nothing supplies `AGENT_SESSION_ROLE_ARN` — and - # `aws_session.get_session` silently falls back to the ambient compute role - # with tenant scoping OFF when it is unset. Refusing is the only correct - # answer; the previous behaviour was a 200 and a stdout breadcrumb. - run_task = MagicMock() - monkeypatch.setattr(server, "run_task", run_task) - monkeypatch.setattr(server.task_state, "write_terminal", MagicMock()) - for key in server.MICROVM_PLATFORM_CONFIG_REQUIRED_KEYS: - os.environ.pop(server.MICROVM_PLATFORM_CONFIG_ENV_BY_KEY[key], None) - - r = client.post(RUN_HOOK, json=_run_hook_body({"agent_payload": self._payload()})) - - assert r.status_code == 400 - body = r.json() - assert body["code"] == "MICROVM_RUN_PLATFORM_CONFIG_INCOMPLETE" - # The response NAMES the unset variables — the whole point is that a skewed - # deployment is diagnosable rather than mysterious. - assert body["missing_env"] == sorted( - server.MICROVM_PLATFORM_CONFIG_ENV_BY_KEY[key] - for key in server.MICROVM_PLATFORM_CONFIG_REQUIRED_KEYS - ) - assert "AGENT_SESSION_ROLE_ARN" in body["missing_env"] - # Nothing was started. - run_task.assert_not_called() - # And the rejection is attributable in the log, not just in the response. - assert "/run hook REJECTED: no platform_config" in capfd.readouterr().out - - def test_a_partially_baked_env_names_only_what_is_actually_missing( - self, client, monkeypatch, baked_platform_env - ): - # The realistic skew: an image that bakes SOME config. The rejection must - # name only the genuinely-unset variables, or an operator chases the wrong - # one. - monkeypatch.setattr(server, "run_task", MagicMock()) - monkeypatch.setattr(server.task_state, "write_terminal", MagicMock()) - os.environ.pop("AGENT_SESSION_ROLE_ARN", None) - - r = client.post(RUN_HOOK, json=_run_hook_body({"agent_payload": self._payload()})) - - assert r.status_code == 400 - assert r.json()["missing_env"] == ["AGENT_SESSION_ROLE_ARN"] - - def test_s3_pointer_takes_the_config_from_the_outer_envelope( - self, client, monkeypatch, env_guard - ): - # The producer's pointer form: the bare task payload lands in S3 and the - # config rides beside the pointer, inside the 4 KB hook body. - monkeypatch.setattr( - server, - "_fetch_microvm_payload_from_s3", - lambda _uri: self._payload(task_id="t-outer"), - ) - monkeypatch.setattr(server, "run_task", MagicMock()) - monkeypatch.setattr(server.task_state, "write_terminal", MagicMock()) - - r = client.post( - RUN_HOOK, - json=_run_hook_body( - { - "agent_payload_s3_uri": "s3://bucket/t-outer/payload.json", - "platform_config": _platform_config(task_table_name="outer-table"), - } - ), - ) - - assert r.status_code == 200 - assert r.json()["task_id"] == "t-outer" - assert os.environ["TASK_TABLE_NAME"] == "outer-table" - - def test_s3_pointer_takes_the_config_merged_into_the_fetched_object( - self, client, monkeypatch, env_guard - ): - # The producer ALSO merges the config into the S3 object, so the agent - # gets it whichever end of the fetch it reads. A stray platform_config key - # left in the bare payload is inert — the extractor reads named fields. - fetched = self._payload(task_id="t-inner") - fetched["platform_config"] = _platform_config(task_table_name="inner-table") - monkeypatch.setattr(server, "_fetch_microvm_payload_from_s3", lambda _uri: fetched) - monkeypatch.setattr(server, "run_task", MagicMock()) - monkeypatch.setattr(server.task_state, "write_terminal", MagicMock()) - - r = client.post( - RUN_HOOK, - json=_run_hook_body({"agent_payload_s3_uri": "s3://bucket/t-inner/payload.json"}), - ) - - assert r.status_code == 200 - assert r.json()["task_id"] == "t-inner" - assert os.environ["TASK_TABLE_NAME"] == "inner-table" - - def test_s3_object_may_itself_be_the_full_envelope(self, client, monkeypatch, env_guard): - monkeypatch.setattr( - server, - "_fetch_microvm_payload_from_s3", - lambda _uri: { - "agent_payload": self._payload(task_id="t-nested"), - "platform_config": _platform_config(task_table_name="nested-table"), - }, - ) - monkeypatch.setattr(server, "run_task", MagicMock()) - monkeypatch.setattr(server.task_state, "write_terminal", MagicMock()) - - r = client.post( - RUN_HOOK, - json=_run_hook_body({"agent_payload_s3_uri": "s3://bucket/t-nested/payload.json"}), - ) - - assert r.status_code == 200 - assert r.json()["task_id"] == "t-nested" - assert os.environ["TASK_TABLE_NAME"] == "nested-table" - - def test_the_fetched_object_wins_over_the_outer_envelope(self, client, monkeypatch, env_guard): - fetched = self._payload(task_id="t-prec") - fetched["platform_config"] = _platform_config(task_table_name="inner-wins") - monkeypatch.setattr(server, "_fetch_microvm_payload_from_s3", lambda _uri: fetched) - monkeypatch.setattr(server, "run_task", MagicMock()) - monkeypatch.setattr(server.task_state, "write_terminal", MagicMock()) - - r = client.post( - RUN_HOOK, - json=_run_hook_body( - { - "agent_payload_s3_uri": "s3://bucket/t-prec/payload.json", - "platform_config": _platform_config(task_table_name="outer-loses"), - } - ), - ) - - assert r.status_code == 200 - assert os.environ["TASK_TABLE_NAME"] == "inner-wins" - - def test_a_nested_agent_payload_of_the_wrong_type_is_a_retryable_500(self, client, monkeypatch): - # Review N1: this is a problem with the FETCHED OBJECT, not with the envelope - # the orchestrator built, so it belongs on the retryable 500 branch. The old - # 400 told the operator "the orchestrator built a bad envelope; retrying - # cannot help". A fetched-object failure is kept distinct from that - # producer-envelope error; invalid stored bytes still need replacement. - monkeypatch.setattr( - server, - "_fetch_microvm_payload_from_s3", - lambda _uri: {"agent_payload": "not-an-object"}, - ) - monkeypatch.setattr(server, "run_task", MagicMock()) - - r = client.post(RUN_HOOK, json=_run_hook_body({"agent_payload_s3_uri": "s3://b/k"})) - - assert r.status_code == 500 - assert r.json()["code"] == "MICROVM_RUN_PAYLOAD_UNREADABLE" - assert "agent_payload in the S3 payload must be an object" in r.json()["message"] - - def test_resolve_returns_the_config_alongside_the_payload(self): - payload, config = server._resolve_microvm_run_payload( - json.dumps({"agent_payload": {"task_id": "t"}, "platform_config": {"a": "b"}}) - ) - assert payload == {"task_id": "t"} - assert config == {"a": "b"} - - def test_resolve_returns_none_for_an_envelope_without_a_config(self): - _payload, config = server._resolve_microvm_run_payload( - json.dumps({"agent_payload": {"task_id": "t"}}) - ) - assert config is None - # -------------------------------------------------------------------------- # /validate + /terminate (ADR-021 P2) @@ -3041,12 +2671,8 @@ def test_never_writes_terminal_task_status(self, client, monkeypatch): write_heartbeat.assert_not_called() def test_returns_200_without_joining_a_running_pipeline( - self, client, monkeypatch, baked_platform_env + self, client, monkeypatch, cached_github_token ): - # `baked_platform_env`: this envelope carries no `platform_config`, which - # since review N2 requires the effective env to supply the required values - # (a legacy image that bakes its own). This class is about the - # payload->pipeline mapping, not about config delivery. # A drain can take minutes (that is lifespan's job on graceful shutdown); # the hook budget is 1-60 s, so /terminate must observe and return. release = threading.Event() @@ -3124,56 +2750,6 @@ def __iter__(self): assert "best-effort step failed" in capfd.readouterr().out -class TestMicrovmPayloadFetchAttribution: - """Every outbound AWS call carries ABCA's solution attribution (#319).""" - - def test_the_s3_payload_fetch_goes_through_the_attributed_factory(self, monkeypatch): - captured: dict = {} - - class _Body: - @staticmethod - def read(): - return b'{"task_id": "t-1"}' - - def fake_platform_client(service_name, **kwargs): - captured["service"] = service_name - captured["kwargs"] = kwargs - return SimpleNamespace(get_object=lambda **_kw: {"Body": _Body}) - - import aws_session - - monkeypatch.setattr(aws_session, "platform_client", fake_platform_client) - - assert server._fetch_microvm_payload_from_s3("s3://b/k") == {"task_id": "t-1"} - assert captured["service"] == "s3" - - def test_the_fetch_client_carries_the_md_user_agent_segment(self, monkeypatch): - # A naked boto3.client('s3') would silently drop the md/ segment. Assert on - # the OUTCOME (the UA on the config) rather than on which helper was used. - captured: dict = {} - - class _Body: - @staticmethod - def read(): - return b'{"task_id": "t-1"}' - - def fake_boto3_client(service_name, **kwargs): - captured["service"] = service_name - captured["config"] = kwargs.get("config") - return SimpleNamespace(get_object=lambda **_kw: {"Body": _Body}) - - import boto3 - - monkeypatch.setattr(boto3, "client", fake_boto3_client) - - server._fetch_microvm_payload_from_s3("s3://b/k") - - import ua - - assert captured["service"] == "s3" - assert ua.static_user_agent_extra() in captured["config"].user_agent_extra - - class TestSnapshotCredentialHygiene: """Nothing on the server's import or /ready path may cache an SDK session. @@ -3259,316 +2835,6 @@ def test_import_and_build_hooks_create_no_boto3_session(self): } -class TestMicrovmRunHookPreInstallAwsSilence: - """Before ``platform_config`` is installed, ``/run`` may touch exactly ONE AWS seam. - - Same defect class the build hooks avoid, one phase later: until the install has - run, ``LOG_GROUP_NAME`` is whatever the snapshot happens to carry, so a - ``_debug_cw`` on this path would resolve credentials and pin - ``boto3.DEFAULT_SESSION`` *before* the orchestrator's own region / - ``AWS_SDK_UA_APP_ID`` / session role are in the environment. The sole permitted - pre-install call is the S3 payload fetch, because the config is inside the - object being fetched. - - Every test here runs with a **baked ``LOG_GROUP_NAME``** — the hostile case the - fix exists for. Without it, ``_debug_cw`` degrades to stdout on its own and the - assertions would pass vacuously. - """ - - def _payload(self, **extra) -> dict: - return { - "task_id": "t-silent", - "repo_url": "org/repo", - "prompt": "do it", - "github_token": "ghp_x", - **extra, - } - - @pytest.fixture(autouse=True) - def _disable_background_pipeline(self, monkeypatch): - """Keep handler-only assertions isolated from asynchronous pipeline work. - - Mocking ``run_task`` is insufficient because ``_spawn_background`` returns - before its thread necessarily dereferences that global. Pytest can restore - the mock between parametrized cases while the prior thread is still - starting, letting it run the real pipeline under the next case's AWS seam - guard. Stub the spawn boundary instead: these tests specify only the - pre-install handler phase and none need a pipeline thread. - """ - monkeypatch.setattr(server, "_spawn_background", MagicMock()) - - @pytest.fixture - def seam_guard(self, monkeypatch): - """Arm every AWS/credential seam to raise until the pre-install phase is OVER. - - Two things end that phase, and only two: - - * ``_install_platform_config`` returning a **non-empty** env list — a real - install. Flipping on *any* return would be a hole big enough to drive B2 - through: the ``raw is None`` early return installs nothing and returns - ``[]``, so treating it as "installed" disarms the guard for the entire - legacy no-``platform_config`` path — which is exactly where a ``_warn_cw`` - was spawning the CloudWatch thread off the snapshot's baked env. - * ``_extract_invocation_params`` being entered. Past that point the legacy - path is *allowed* to talk to AWS: running on the snapshot's own env is the - documented P1-compatibility behaviour, so the accepted-line ``_debug_cw`` - and the pipeline below it are legitimate. Everything the handler does - *before* it — including the "no platform_config" warning — is not. - - A rejection path reaches neither, so the seams stay armed for the whole - request: a rejected run installed nothing and has no more right to an AWS - call than it had before. - """ - state: dict[str, Any] = { - "install_phase_done": False, - "installed_env": None, - "violations": [], - } - real_install = server._install_platform_config - real_extract = server._extract_invocation_params - - def spy_install(raw): - result = real_install(raw) - state["installed_env"] = result - if result: - state["install_phase_done"] = True - return result - - def spy_extract(*args, **kwargs): - state["install_phase_done"] = True - return real_extract(*args, **kwargs) - - monkeypatch.setattr(server, "_install_platform_config", spy_install) - monkeypatch.setattr(server, "_extract_invocation_params", spy_extract) - - def guard(name): - def _seam(*_args, **_kwargs): - if not state["install_phase_done"]: - state["violations"].append(name) - raise AssertionError(f"{name} touched before platform_config was installed") - return MagicMock() - - return _seam - - import boto3 - - import aws_session - - # Kept so a test can re-enable exactly the ONE permitted pre-install seam - # (the S3 payload fetch) and assert on it positively. - state["real_platform_client"] = aws_session.platform_client - - for module, attr in ( - (boto3, "client"), - (boto3, "Session"), - (aws_session, "platform_client"), - (aws_session, "tenant_client"), - (aws_session, "tenant_resource"), - (aws_session, "get_session"), - (server, "_debug_cw"), - (server, "_warn_cw"), - (server, "_debug_cw_exc"), - ): - monkeypatch.setattr(module, attr, guard(f"{module.__name__}.{attr}")) - - monkeypatch.setenv("LOG_GROUP_NAME", "/abca/agent") - return state - - @pytest.mark.parametrize("with_config", [True, False], ids=["with-config", "no-config"]) - def test_no_cloudwatch_or_credential_seam_is_touched_before_the_install( - self, client, monkeypatch, baked_platform_env, seam_guard, capfd, with_config - ): - # The ``no-config`` arm is the legacy P1 envelope, and it is the harder case: - # nothing is ever installed, so EVERY line up to param extraction — including - # the "running on the snapshot's frozen env" warning itself — is still - # pre-install. A ``_warn_cw`` there would spawn the CloudWatch writer thread - # and pin ``boto3.DEFAULT_SESSION`` off the baked ``LOG_GROUP_NAME`` this - # fixture sets, which is precisely the defect the warning is reporting. - # - # ``baked_platform_env`` (rather than ``env_guard``) so that arm reaches the - # ACCEPT path: since review N2 a no-config run with an unsatisfied effective - # env is refused, and a rejected run installs nothing and so proves nothing - # about the seams staying silent all the way to param extraction. Baking the - # env is also the only shape in which the no-config path is legitimate. - monkeypatch.setattr(server, "run_task", MagicMock()) - monkeypatch.setattr(server.task_state, "write_terminal", MagicMock()) - - envelope: dict[str, Any] = {"agent_payload": self._payload()} - if with_config: - envelope["platform_config"] = _platform_config() - - r = client.post(RUN_HOOK, json=_run_hook_body(envelope)) - - assert r.status_code == 200 - assert seam_guard["violations"] == [] - assert seam_guard["install_phase_done"] is True - if with_config: - assert seam_guard["installed_env"] - else: - # Vacuously "installed": the early return the flag must NOT trust. - assert seam_guard["installed_env"] == [] - # The warning still reaches an operator — stdout, via the pre-install sink. - assert ( - "[server/run-pre-config] /run hook received no platform_config" - in capfd.readouterr().out - ) - - def test_the_no_config_REJECTION_also_touches_no_seam( - self, client, monkeypatch, env_guard, seam_guard, capfd - ): - # Review N2's rejection is itself a pre-install path, and it emits a NEW log - # line — so it needs the same guarantee as every other rejection here: the - # refusal must not be the thing that pins `boto3.DEFAULT_SESSION` off the - # snapshot's baked `LOG_GROUP_NAME` (which `seam_guard` sets). - monkeypatch.setattr(server, "run_task", MagicMock()) - for key in server.MICROVM_PLATFORM_CONFIG_REQUIRED_KEYS: - os.environ.pop(server.MICROVM_PLATFORM_CONFIG_ENV_BY_KEY[key], None) - - r = client.post(RUN_HOOK, json=_run_hook_body({"agent_payload": self._payload()})) - - assert r.status_code == 400 - assert r.json()["code"] == "MICROVM_RUN_PLATFORM_CONFIG_INCOMPLETE" - assert seam_guard["violations"] == [] - assert "/run hook REJECTED: no platform_config" in capfd.readouterr().out - - def test_the_received_line_is_stdout_only( - self, client, monkeypatch, env_guard, seam_guard, capfd - ): - monkeypatch.setattr(server, "run_task", MagicMock()) - monkeypatch.setattr(server.task_state, "write_terminal", MagicMock()) - - client.post( - RUN_HOOK, - json=_run_hook_body( - {"agent_payload": self._payload(), "platform_config": _platform_config()}, - microvm_id="microvm-quiet", - ), - ) - - out = capfd.readouterr().out - assert "[server/run-pre-config] /run hook received:" in out - assert "microvm-quiet" in out - - def test_the_s3_payload_fetch_is_the_only_pre_install_aws_call( - self, client, monkeypatch, env_guard, seam_guard - ): - # The permitted exception, asserted positively: exactly one client, for s3, - # while the CloudWatch/credential seams stay armed. - services: list[str] = [] - - class _Body: - @staticmethod - def read(): - return json.dumps(self._payload(task_id="t-from-s3")).encode() - - def recording_client(service_name, **_kwargs): - services.append(service_name) - return SimpleNamespace(get_object=lambda **_kw: {"Body": _Body}) - - import boto3 - - import aws_session - - # Re-enable the one permitted seam, and only it: the fetch must still go - # through the attributed factory (#319), which delegates to boto3.client. - monkeypatch.setattr(aws_session, "platform_client", seam_guard["real_platform_client"]) - monkeypatch.setattr(boto3, "client", recording_client) - monkeypatch.setattr(server, "run_task", MagicMock()) - monkeypatch.setattr(server.task_state, "write_terminal", MagicMock()) - - r = client.post( - RUN_HOOK, - json=_run_hook_body( - { - "agent_payload_s3_uri": "s3://payload-bucket/t-from-s3/payload.json", - "platform_config": _platform_config(), - } - ), - ) - - assert r.status_code == 200 - assert r.json()["task_id"] == "t-from-s3" - assert services == ["s3"] - assert seam_guard["violations"] == [] - - def test_a_malformed_envelope_is_rejected_without_touching_a_seam( - self, client, monkeypatch, seam_guard, capfd - ): - monkeypatch.setattr(server, "run_task", MagicMock()) - - r = client.post(RUN_HOOK, json={"microvmId": "m", "runHookPayload": "not json"}) - - assert r.status_code == 400 - assert r.json()["code"] == "MICROVM_RUN_PAYLOAD_INVALID" - assert seam_guard["violations"] == [] - assert seam_guard["install_phase_done"] is False - # The reason still reaches an operator: stdout here, and the response body - # (which the MicroVM service surfaces) in every case. - assert "[server/run-pre-config] /run hook rejected:" in capfd.readouterr().out - - def test_a_failed_payload_fetch_is_reported_without_touching_a_seam( - self, client, monkeypatch, seam_guard, capfd - ): - def boom(_uri): - raise RuntimeError("AccessDenied") - - monkeypatch.setattr(server, "_fetch_microvm_payload_from_s3", boom) - monkeypatch.setattr(server, "run_task", MagicMock()) - - r = client.post(RUN_HOOK, json=_run_hook_body({"agent_payload_s3_uri": "s3://b/k"})) - - assert r.status_code == 500 - assert r.json()["code"] == "MICROVM_RUN_PAYLOAD_UNREADABLE" - assert seam_guard["violations"] == [] - out = capfd.readouterr().out - assert "[server/run-pre-config] /run hook payload fetch FAILED" in out - # The traceback is preserved on the stdout line (it is the only diagnostic - # the response body does not carry). - assert "Traceback" in out - - @pytest.mark.parametrize( - "config,expected_code", - [ - ({"ld_preload": "/tmp/evil.so"}, "MICROVM_RUN_PLATFORM_CONFIG_INVALID"), - ({"log_group_name": "lg"}, "MICROVM_RUN_PLATFORM_CONFIG_INCOMPLETE"), - ], - ) - def test_a_rejected_platform_config_touches_no_seam( - self, client, monkeypatch, env_guard, seam_guard, config, expected_code - ): - monkeypatch.setattr(server, "run_task", MagicMock()) - - r = client.post( - RUN_HOOK, - json=_run_hook_body({"agent_payload": self._payload(), "platform_config": config}), - ) - - assert r.status_code == 400 - assert r.json()["code"] == expected_code - # Nothing was installed, so nothing earned the right to an AWS call. - assert seam_guard["violations"] == [] - assert seam_guard["install_phase_done"] is False - - def test_the_accepted_line_correlates_task_and_microvm_ids( - self, client, monkeypatch, env_guard, capfd - ): - # The pre-install "received" line is stdout-only now, so the first line that - # reaches the task's log group has to join both ids by itself. - monkeypatch.setattr(server, "run_task", MagicMock()) - monkeypatch.setattr(server.task_state, "write_terminal", MagicMock()) - - client.post( - RUN_HOOK, - json=_run_hook_body( - {"agent_payload": self._payload(), "platform_config": _platform_config()}, - microvm_id="microvm-joined", - ), - ) - - out = capfd.readouterr().out - assert "/run hook accepted task_id='t-silent' microvm_id='microvm-joined'" in out - - class TestTerminateHookBodyTolerance: """``/terminate`` must answer 200 for ANY body — that is why it takes the raw request. diff --git a/cdk/src/constructs/ecs-agent-cluster.ts b/cdk/src/constructs/ecs-agent-cluster.ts index 48d55105b..28f01661a 100644 --- a/cdk/src/constructs/ecs-agent-cluster.ts +++ b/cdk/src/constructs/ecs-agent-cluster.ts @@ -32,6 +32,7 @@ import { AgentMemory } from './agent-memory'; import { AgentSessionRole, grantAgentTaskTableAccess } from './agent-session-role'; import { resolveBedrockGeoRegion, resolveBedrockModelIds } from './bedrock-models'; import { LinearIdentityVault } from './linear-identity-vault'; +import { grantWorkerBootstrap } from './payload-bootstrap-permissions'; import { buildAppId } from './solution-ua-aspect'; import { ToolGateway } from './tool-gateway'; @@ -52,14 +53,12 @@ export interface EcsAgentClusterProps { readonly taskSizing?: EcsTaskSizing; /** - * S3 bucket holding per-task ECS payloads. The orchestrator writes the - * payload (incl. the large hydrated_context, which can't fit in the 8 KB - * RunTask containerOverrides limit) here and passes only an - * `AGENT_PAYLOAD_S3_URI` pointer; the container fetches it on boot. The task - * role gets **read-only** on this bucket — the container runs untrusted repo - * code, so it must not be able to delete payloads (the trusted orchestrator - * owns write + delete). When omitted (isolated construct tests / deployments - * that still pass the payload inline), no grant or env var is added. + * S3 storage for deployment manifests, task payloads and private launch + * references. The v2 coordinator sends AGENT_PAYLOAD_REF, containing a + * single-object signed download URL. The task role can read only bootstrap/*; + * object reads elsewhere and bucket listing are explicitly denied. The + * coordinator owns writes and cleanup. Optional for isolated construct tests; + * production ECS launches require this bucket and a matching v2 image. */ readonly payloadBucket?: s3.IBucket; @@ -387,9 +386,8 @@ export class EcsAgentCluster extends Construct { LOG_GROUP_NAME: logGroup.logGroupName, GITHUB_TOKEN_SECRET_ARN: props.githubTokenSecret.secretArn, ...(props.memoryId && { MEMORY_ID: props.memoryId }), - // The payload bucket name so the orchestrator-issued AGENT_PAYLOAD_S3_URI - // can be fetched. (The orchestrator sets the URI per-task via container - // override; this is set here for parity with the runtime env.) + // Deployment metadata; the per-task AGENT_PAYLOAD_REF supplies the + // manifest URI and signed payload URL. IAM authenticates the manifest. ...(props.payloadBucket && { ECS_PAYLOAD_BUCKET: props.payloadBucket.bucketName }), // Artifact workflows (planning/analysis) deliver their document to // this bucket. The AgentCore runtime has ARTIFACTS_BUCKET_NAME; the ECS task @@ -534,13 +532,11 @@ export class EcsAgentCluster extends Construct { // agent assumes the SessionRole — stays on the task role). props.githubTokenSecret.grantRead(taskRole); - // Read-only on the ECS payload bucket so the container can fetch its payload - // (AGENT_PAYLOAD_S3_URI) at boot. READ only — the container runs untrusted - // repo code, so it must not be able to write or delete payloads (the trusted - // orchestrator owns write + delete). Stays on the task role (read once at - // startup, before the agent assumes any SessionRole). + // Only deployment manifests use worker credentials. The boot helper reads + // its exact task object through a signed URL and removes that capability + // from the environment before starting repository code. if (props.payloadBucket) { - props.payloadBucket.grantRead(taskRole); + grantWorkerBootstrap(props.payloadBucket, taskRole); } // Artifact workflows (planning/analysis) deliver their document to the @@ -679,7 +675,7 @@ export class EcsAgentCluster extends Construct { NagSuppressions.addResourceSuppressions(taskRole, [ { id: 'AwsSolutions-IAM5', - reason: 'DynamoDB index/* wildcards from the legacy TaskEventsTable grant when no SessionRole is wired (TaskTable allows only reporting updates; the worker has no UserConcurrency access); Secrets Manager wildcards from CDK grantRead (GitHub token) and the bgagent-linear-oauth-*/bgagent-jira-oauth-* prefix grant (ABCA-488 — per-workspace channel OAuth tokens are created by the CLI at setup, name unknown at synth, GetSecretValue only); CloudWatch Logs wildcards from CDK grantWrite; S3 object/* wildcard from CDK grantRead on the ECS payload bucket (read-only, scoped to that bucket — #502). Bedrock InvokeModel is scoped to explicit model/inference-profile ARNs (no wildcard resource). ec2:DescribeAvailabilityZones requires Resource:* (EC2 describe actions have no resource-level scoping) — read-only, no mutation/data access; needed so a CDK target repo\'s `cdk synth` build gate can resolve AZ context on a fresh clone (ECS-parity, no cdk.context.json cache in the container).', + reason: 'DynamoDB index/* wildcards from the legacy TaskEventsTable grant when no SessionRole is wired (TaskTable allows only reporting updates; the worker has no UserConcurrency access); Secrets Manager wildcards from CDK grantRead (GitHub token) and the bgagent-linear-oauth-*/bgagent-jira-oauth-* prefix grant (ABCA-488 — per-workspace channel OAuth tokens are created by the CLI at setup, name unknown at synth, GetSecretValue only); CloudWatch Logs wildcards from CDK grantWrite; Worker S3 GetObject is restricted to bootstrap/*; other object reads and payload-bucket listing are explicitly denied (#700). Bedrock InvokeModel is scoped to explicit model/inference-profile ARNs (no wildcard resource). ec2:DescribeAvailabilityZones requires Resource:* (EC2 describe actions have no resource-level scoping) — read-only, no mutation/data access; needed so a CDK target repo\'s `cdk synth` build gate can resolve AZ context on a fresh clone (ECS-parity, no cdk.context.json cache in the container).', }, { id: 'AwsSolutions-ECS2', diff --git a/cdk/src/constructs/ecs-payload-bucket.ts b/cdk/src/constructs/ecs-payload-bucket.ts index 87dc47553..4c1b07402 100644 --- a/cdk/src/constructs/ecs-payload-bucket.ts +++ b/cdk/src/constructs/ecs-payload-bucket.ts @@ -55,24 +55,17 @@ export interface EcsPayloadBucketProps { } /** - * S3 bucket for ECS task payloads (#502). + * Storage for ECS v2 bootstrap manifests, task payloads and private references. * - * The ECS compute strategy cannot pass the orchestrator payload (repo URL, - * prompt, and the large ``hydrated_context``) inline: a Fargate ``RunTask`` - * caps the entire ``containerOverrides`` blob at 8192 bytes, and the hydrated - * context routinely exceeds that, so the call is rejected with - * ``InvalidParameterException``. (AgentCore is unaffected — it passes the - * payload in the ``InvokeAgentRuntime`` request body, which has no comparable - * limit.) Instead, the orchestrator writes the payload to - * ``s3:////payload.json`` and passes only a small - * ``AGENT_PAYLOAD_S3_URI`` pointer in the override; the container fetches and - * parses it on boot. + * RunTask caps the entire overrides object at 8192 bytes. The coordinator + * writes task instructions to /payload.json and passes AGENT_PAYLOAD_REF, + * containing an authenticated-manifest URI and a signed one-object URL. + * EcsAgentCluster grants the worker only bootstrap/* reads and explicitly denies + * other object reads/listing. The coordinator owns writes, signing and cleanup. * - * Dedicated (not co-tenant with attachments/traces) so the boundary is - * structural: the ECS task role gets S3 **read** here and nowhere else, the - * attachments feature can never collide with payload keys, and the tight - * 1-day TTL is whole-bucket rather than a prefix-scoped rule grafted onto a - * shared bucket. + * A dedicated bucket separates boot data from attachments/traces and gives it + * a one-day lifecycle backstop. Finalize deletes payload.json and launch.json; + * shared deployment manifests expire through the lifecycle rule. * * Security / hygiene (parity with TraceArtifactsBucket): * - ``blockPublicAccess: BLOCK_ALL`` + ``enforceSSL: true`` — no public read, diff --git a/cdk/src/constructs/lambda-microvm-compute.ts b/cdk/src/constructs/lambda-microvm-compute.ts index ab9e31d14..6fb0c882b 100644 --- a/cdk/src/constructs/lambda-microvm-compute.ts +++ b/cdk/src/constructs/lambda-microvm-compute.ts @@ -41,15 +41,16 @@ import { Construct } from 'constructs'; import { AgentMemory } from './agent-memory'; import { AgentSessionRole } from './agent-session-role'; import { resolveBedrockModelIds } from './bedrock-models'; +import { grantWorkerBootstrap } from './payload-bootstrap-permissions'; import sharedConstants from '../../../contracts/constants.json'; import { LAMBDA_MICROVM_SUPPORTED_REGIONS, isLambdaMicrovmRegionSupported } from '../handlers/shared/microvm-regions'; /** * Lifecycle expiry for MicroVM `/run` hook payloads, in days. * - * Mirrors {@link ECS_PAYLOAD_TTL_DAYS}. Finalization deletes the object - * identified by `microvmPayloadKey`; lifecycle expiry is the fallback if that - * step fails. S3 processes expiry asynchronously, not exactly 24 hours after + * Mirrors {@link ECS_PAYLOAD_TTL_DAYS}. Finalization deletes payload.json and + * the private launch.json; shared manifests expire through lifecycle cleanup. + * Lifecycle expiry also removes task objects when finalization fails. S3 processes expiry asynchronously, not exactly 24 hours after * upload. Payloads carry hydrated prompt context and are read once at `/run`. */ export const MICROVM_PAYLOAD_TTL_DAYS = 1; @@ -658,15 +659,15 @@ export interface LambdaMicrovmComputeProps extends LambdaMicrovmImageInputs { * `apt-get`, which is plain HTTP; a 443-only build path fails every * snapshot build (see {@link HTTP_PORT}). Runtime egress stays 443-only. * 2. **Artifact bucket** for the zip + Dockerfile the service builds the - * snapshot from, and a **payload bucket** for `/run` payloads that exceed - * the 4 KB `runHookPayload` cap (which is nearly all of them). + * snapshot from, and a **payload bucket** for deployment manifests, task instructions and private + * launch references. Every task uses the authenticated v2 transport. * 3. **Build role** — assumed by Lambda during image creation: `s3:GetObject` * on the artifact object and CloudWatch Logs writes. Without it Lambda * cannot emit build logs, which makes a failed snapshot build undebuggable. * 4. **Execution role** — assumed by the running MicroVM: CloudWatch Logs (both * the service's own `/aws/lambda-microvms/*` namespace and the platform - * APPLICATION_LOGS group whose name `platform_config` delivers), read-only on - * the payload bucket, the P2 runtime-parity grants (GitHub PAT + + * APPLICATION_LOGS group whose name `platform_config` delivers), read access only to + * the payload bucket's bootstrap manifests, the P2 runtime-parity grants (GitHub PAT + * channel-OAuth secret reads, scoped Bedrock invocation, AgentCore Memory, * `ec2:DescribeAvailabilityZones` for a CDK repo's synth gate), and — when a * SessionRole is wired — admission to the per-task SessionRole, which is the @@ -740,7 +741,7 @@ export class LambdaMicrovmCompute extends Construct { /** Key of the artifact object inside {@link artifactBucket}. */ public readonly artifactObjectKey: string; - /** S3 bucket holding oversized `/run` payloads (S3-pointer delivery). */ + /** S3 bucket holding bootstrap manifests, task payloads and private launch references. */ public readonly payloadBucket: s3.Bucket; /** Role Lambda assumes while building the snapshot image. */ @@ -1097,7 +1098,7 @@ export class LambdaMicrovmCompute extends Construct { assumedBy: microvmAssumedBy, description: 'ABCA Lambda MicroVMs execution role: assumed by the running MicroVM and its runtime ' - + 'lifecycle hooks; writes logs and reads out-of-band /run payloads.', + + 'lifecycle hooks; writes logs and reads deployment bootstrap manifests.', }); grantTagSession(this.executionRole, microvmAssumedBy); // Execution role: NO `logs:CreateLogGroup`. It runs untrusted repo code and @@ -1116,12 +1117,10 @@ export class LambdaMicrovmCompute extends Construct { // See `applicationLogGroup` for the denial this fixes. props.applicationLogGroup?.grantWrite(this.executionRole); - // READ-ONLY on the payload bucket (ADR-021: "The MicroVM execution role - // shall hold read-only access to the payload bucket, scoped to that - // bucket"). Read-only is not a nicety: the MicroVM runs untrusted repo - // code, so it must not be able to clobber another task's payload. Write + - // lifecycle stay with the trusted orchestrator. - this.payloadBucket.grantRead(this.executionRole); + // Authenticate only this deployment's manifests. Payload and launch reads + // using worker credentials are explicitly denied; one-object signed URLs + // carry the coordinator's authorization instead. + grantWorkerBootstrap(this.payloadBucket, this.executionRole); // Tenant-data access is delegated to the per-task SessionRole, exactly as // EcsAgentCluster does for the Fargate task role. NOTE the asymmetry with @@ -1404,8 +1403,8 @@ export class LambdaMicrovmCompute extends Construct { id: 'AwsSolutions-S1', reason: 'Artifact bucket holds a single build input (the agent zip+Dockerfile) read only by ' + 'the Lambda MicroVMs build role; the payload bucket holds ephemeral per-task /run payloads ' - + `with a ${MICROVM_PAYLOAD_TTL_DAYS}-day TTL, written only by the orchestrator (grantPut) and ` - + 'read only by the MicroVM execution role, both scoped to the bucket. Object-level access ' + + `with a ${MICROVM_PAYLOAD_TTL_DAYS}-day TTL, written only by the orchestrator and ` + + 'read through single-object signed URLs; the worker reads only bootstrap manifests. Object-level access ' + 'logging (a second log bucket + CloudTrail data events) is not justified for a single ' + 'build input or for transient boot payloads.', }, @@ -1416,8 +1415,8 @@ export class LambdaMicrovmCompute extends Construct { id: 'AwsSolutions-IAM5', reason: 'CloudWatch Logs wildcard is the service-owned ' + `${MICROVM_LOG_GROUP_PREFIX}/* namespace (log stream names are minted per MicroVM, so no ` - + 'synth-time ARN exists); S3 object/* wildcard comes from CDK grantRead on the dedicated ' - + 'payload bucket (read-only, scoped to that bucket — ADR-021 sub-decision 3). The build ' + + 'synth-time ARN exists); worker S3 GetObject is limited to bootstrap/* in its payload bucket, ' + + 'with explicit denial outside that prefix and for bucket listing. The build ' + 'role\'s s3:GetObject is scoped to a single object key, not a wildcard. On the execution ' + 'role (ADR-021 P2 runtime parity, mirroring the ECS task role): the second Logs grant is ' + 'CDK grantWrite (CreateLogStream + PutLogEvents only) on the SINGLE platform ' diff --git a/cdk/src/constructs/payload-bootstrap-permissions.ts b/cdk/src/constructs/payload-bootstrap-permissions.ts new file mode 100644 index 000000000..c6f215c0e --- /dev/null +++ b/cdk/src/constructs/payload-bootstrap-permissions.ts @@ -0,0 +1,67 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import * as iam from 'aws-cdk-lib/aws-iam'; +import * as s3 from 'aws-cdk-lib/aws-s3'; +import constants from '../../../contracts/constants.json'; + +/** Authenticate deployment settings with IAM; task payloads require a one-object capability. */ +export function grantWorkerBootstrap(bucket: s3.IBucket, worker: iam.IGrantable): void { + const manifests = bucket.arnForObjects(`${constants.payload_bootstrap.manifest_prefix}*`); + worker.grantPrincipal.addToPrincipalPolicy(new iam.PolicyStatement({ + actions: ['s3:GetObject'], + resources: [manifests], + })); + // An Allow alone is not an authentication boundary: another bucket could + // publicly grant access to an attacker's fake deployment manifest. + worker.grantPrincipal.addToPrincipalPolicy(new iam.PolicyStatement({ + effect: iam.Effect.DENY, + actions: ['s3:GetObject*'], + notResources: [manifests], + })); + worker.grantPrincipal.addToPrincipalPolicy(new iam.PolicyStatement({ + effect: iam.Effect.DENY, + actions: ['s3:List*'], + resources: [bucket.bucketArn], + })); +} + +export function grantCoordinatorPayloads(bucket: s3.IBucket, coordinator: iam.IGrantable): void { + // S3 returns AccessDenied rather than NoSuchKey for an absent launch record + // without ListBucket. Only the trusted coordinator needs this permission. + coordinator.grantPrincipal.addToPrincipalPolicy(new iam.PolicyStatement({ + actions: ['s3:ListBucket'], + resources: [bucket.bucketArn], + })); + const taskObjects = [ + bucket.arnForObjects('*/payload.json'), + bucket.arnForObjects(`*/${constants.payload_bootstrap.launch_filename}`), + ]; + coordinator.grantPrincipal.addToPrincipalPolicy(new iam.PolicyStatement({ + actions: ['s3:PutObject'], + resources: [ + ...taskObjects, + bucket.arnForObjects(`${constants.payload_bootstrap.manifest_prefix}*`), + ], + })); + coordinator.grantPrincipal.addToPrincipalPolicy(new iam.PolicyStatement({ + actions: ['s3:GetObject', 's3:DeleteObject'], + resources: taskObjects, + })); +} diff --git a/cdk/src/constructs/task-orchestrator.ts b/cdk/src/constructs/task-orchestrator.ts index 3243f4ddb..a505533b2 100644 --- a/cdk/src/constructs/task-orchestrator.ts +++ b/cdk/src/constructs/task-orchestrator.ts @@ -51,6 +51,8 @@ const ORCHESTRATOR_TIMEOUT_SECONDS = 60; /** Orchestrator Lambda memory (MB). */ const ORCHESTRATOR_MEMORY_MB = 1024; +import { grantCoordinatorPayloads } from './payload-bootstrap-permissions'; + /** * Properties for TaskOrchestrator construct. */ @@ -175,12 +177,11 @@ export interface TaskOrchestratorProps { }; /** - * S3 bucket for per-task ECS payloads. When provided (alongside - * ``ecsConfig``), the orchestrator writes the payload here and passes only an - * ``AGENT_PAYLOAD_S3_URI`` pointer in the RunTask override (the full payload - * exceeds the 8 KB containerOverrides limit), then deletes the object in the - * finalize step. The orchestrator gets write + delete; the ECS task role gets - * read-only (granted on the bucket by ``EcsAgentCluster``). + * S3 storage for ECS v2 bootstrap manifests, task payloads and private launch + * references. The coordinator publishes/signs/replays the reference delivered + * in AGENT_PAYLOAD_REF, then deletes both task objects at finalization. + * EcsAgentCluster separately grants worker bootstrap-only reads and denies + * other object reads and payload-bucket listing. */ readonly ecsPayloadBucket?: s3.IBucket; @@ -407,7 +408,6 @@ export class TaskOrchestrator extends Construct { '@aws-sdk/client-ecs', '@aws-sdk/client-lambda', '@aws-sdk/client-bedrock-runtime', - '@aws-sdk/client-s3', '@aws-sdk/client-secrets-manager', '@aws-sdk/lib-dynamodb', '@aws-sdk/util-dynamodb', @@ -514,27 +514,10 @@ export class TaskOrchestrator extends Construct { props.attachmentsBucket.grantReadWrite(this.fn); } - // ECS payload bucket — the orchestrator writes the payload before - // RunTask and deletes it at finalize. Write + delete only (it never reads - // its own payload back; the ECS container is the reader, with its own - // read-only grant from EcsAgentCluster). - if (props.ecsPayloadBucket) { - props.ecsPayloadBucket.grantPut(this.fn); - props.ecsPayloadBucket.grantDelete(this.fn); - } - - // ADR-021: upload oversized /run payloads; the execution role is the reader. - // Match deleteMicrovmPayload's /payload.json key shape (#817). - // grantDelete would also grant DeleteObjectVersion; this unversioned bucket - // needs only DeleteObject, and no delete permission belongs on the worker. - if (props.microvmConfig) { - props.microvmConfig.payloadBucket.grantPut(this.fn); - this.fn.addToRolePolicy(new iam.PolicyStatement({ - sid: 'MicrovmDeletePayload', - actions: ['s3:DeleteObject'], - resources: [props.microvmConfig.payloadBucket.arnForObjects('*/payload.json')], - })); - } + // Publish manifests/payloads, sign one-object reads and persist private + // launch references for replay. Workers cannot read the launch records. + if (props.ecsPayloadBucket) grantCoordinatorPayloads(props.ecsPayloadBucket, this.fn); + if (props.microvmConfig) grantCoordinatorPayloads(props.microvmConfig.payloadBucket, this.fn); // Durable execution managed policy this.fn.role!.addManagedPolicy( @@ -803,7 +786,7 @@ export class TaskOrchestrator extends Construct { }, { id: 'AwsSolutions-IAM5', - reason: 'DynamoDB index/* wildcards generated by CDK grantReadWriteData; AgentCore runtime/* required for sub-resource invocation; Secrets Manager wildcards generated by CDK grantRead; AgentCore Memory wildcards generated by CDK grantRead/grantWrite; ECS RunTask/DescribeTasks/StopTask conditioned on cluster ARN; iam:PassRole scoped to ECS task/execution roles and conditioned on ecs-tasks.amazonaws.com; S3 object/* wildcard from CDK grantPut on the dedicated MicroVM payload bucket; MicroVM cleanup DeleteObject restricted to */payload.json in that bucket; MicroVM lifecycle actions (RunMicrovm/GetMicrovm/TerminateMicrovm) are scoped to the single platform MicroVM image ARN plus a :* version-suffix sibling (every one of them authorizes against the image resource, not the per-session instance; no account-wide wildcard is used); lambda:PassNetworkConnector requires Resource:* because the action supports no resource-level permissions and the AWS-managed connectors live outside this account; iam:PassRole is scoped to the exact MicroVM execution role without iam:PassedToService (ADR-021 P2r2-F10); Agent Registry read scoped to the wired registry ARN, with a record/* suffix wildcard because record ids are server-assigned and unknown at synth (#246)', + reason: 'DynamoDB index/* wildcards generated by CDK grantReadWriteData; AgentCore runtime/* required for sub-resource invocation; Secrets Manager wildcards generated by CDK grantRead; AgentCore Memory wildcards generated by CDK grantRead/grantWrite; ECS RunTask/DescribeTasks/StopTask conditioned on cluster ARN; iam:PassRole scoped to ECS task/execution roles and conditioned on ecs-tasks.amazonaws.com; S3 writes restricted to bootstrap manifests and task payload/launch objects; GetObject and DeleteObject restricted to */payload.json and */launch.json for signing, replay and cleanup; ListBucket is scoped to each payload bucket so absent launch records return NoSuchKey; MicroVM lifecycle actions (RunMicrovm/GetMicrovm/TerminateMicrovm) are scoped to the single platform MicroVM image ARN plus a :* version-suffix sibling (every one of them authorizes against the image resource, not the per-session instance; no account-wide wildcard is used); lambda:PassNetworkConnector requires Resource:* because the action supports no resource-level permissions and the AWS-managed connectors live outside this account; iam:PassRole is scoped to the exact MicroVM execution role without iam:PassedToService (ADR-021 P2r2-F10); Agent Registry read scoped to the wired registry ARN, with a record/* suffix wildcard because record ids are server-assigned and unknown at synth (#246)', }, ], true); } diff --git a/cdk/src/handlers/shared/payload-bootstrap.ts b/cdk/src/handlers/shared/payload-bootstrap.ts new file mode 100644 index 000000000..8134d3835 --- /dev/null +++ b/cdk/src/handlers/shared/payload-bootstrap.ts @@ -0,0 +1,183 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import { createHash } from 'node:crypto'; +import { DeleteObjectCommand, GetObjectCommand, PutObjectCommand, S3Client } from '@aws-sdk/client-s3'; +import { getSignedUrl } from '@aws-sdk/s3-request-presigner'; +import { logger } from './logger'; +import { makeClient } from './ua'; +import constants from '../../../../contracts/constants.json'; + +export const PAYLOAD_BOOTSTRAP = constants.payload_bootstrap; +type Backend = 'ecs' | 'lambda-microvm'; + +export interface PayloadReference { + version: number; + task_id: string; + bootstrap_s3_uri: string; + payload_url: string; + expires_at: number; +} + +interface LaunchRecord { + fingerprint: string; + reference: PayloadReference; +} + +let client: S3Client | undefined; +function s3(): S3Client { + return client ??= makeClient(S3Client); +} + +function canonical(value: unknown): string { + return JSON.stringify(value, (_key, item: unknown) => + item && typeof item === 'object' && !Array.isArray(item) + ? Object.fromEntries(Object.entries(item).sort(([a], [b]) => a.localeCompare(b))) + : item); +} + +function sha256(value: string): string { + return createHash('sha256').update(value).digest('hex'); +} + +/** Never let SDK errors echo bearer URLs into task records or logs. */ +export function redactPayloadUrls(message: string): string { + return message.replace(/https?:\/\/[^\s"'<>\\]+/gi, url => + /X-Amz-/i.test(url) ? '[redacted payload URL]' : url); +} + +async function readObject(bucket: string, key: string): Promise { + try { + const response = await s3().send(new GetObjectCommand({ Bucket: bucket, Key: key })); + if (!response.Body) throw new Error('PAYLOAD_BOOTSTRAP_UNREADABLE: empty object response'); + return await response.Body.transformToString(); + } catch (error) { + if ((error as { name?: string }).name === 'NoSuchKey') return undefined; + throw error; + } +} + +/** Conditional creation, including recovery after a committed write loses its reply. */ +async function createOnce(bucket: string, key: string, body: string): Promise { + try { + await s3().send(new PutObjectCommand({ + Bucket: bucket, Key: key, Body: body, ContentType: 'application/json', IfNoneMatch: '*', + })); + return body; + } catch (error) { + const saved = await readObject(bucket, key); + if (saved !== undefined) return saved; + throw error; + } +} + +/** + * Save a single-object capability outside the agent-readable task table. + * Replays read the same launch.json; re-signing would change Run's request + * while reusing its clientToken. The worker can read only bootstrap/* using + * its own credentials, never payload.json or launch.json. + */ +export async function preparePayloadReference(input: { + bucket: string; + taskId: string; + backend: Backend; + payload: Record; + platformConfig?: Record; +}): Promise { + const { bucket, taskId, backend, payload } = input; + if (!/^[A-Za-z0-9_-]{1,128}$/.test(taskId) || payload.task_id !== taskId) { + throw new Error('PAYLOAD_BOOTSTRAP_INVALID: task identity does not match the payload'); + } + const manifest = canonical({ + version: PAYLOAD_BOOTSTRAP.version, backend, platform_config: input.platformConfig ?? {}, + }); + const manifestKey = `${PAYLOAD_BOOTSTRAP.manifest_prefix}${sha256(manifest)}.json`; + const payloadBody = canonical({ + version: PAYLOAD_BOOTSTRAP.version, + task_id: taskId, + agent_payload: payload, + platform_config: input.platformConfig ?? {}, + }); + if (Buffer.byteLength(manifest) > PAYLOAD_BOOTSTRAP.max_manifest_bytes + || Buffer.byteLength(payloadBody) > PAYLOAD_BOOTSTRAP.max_payload_bytes) { + throw new Error('PAYLOAD_BOOTSTRAP_TOO_LARGE: bootstrap manifest or task payload exceeds its byte limit'); + } + const fingerprint = sha256(canonical({ bucket, backend, manifest, payloadBody })); + const launchKey = `${taskId}/${PAYLOAD_BOOTSTRAP.launch_filename}`; + const accept = (saved: string): PayloadReference => { + const record = JSON.parse(saved) as LaunchRecord; + if (record.fingerprint !== fingerprint || record.reference?.task_id !== taskId) { + throw new Error('PAYLOAD_BOOTSTRAP_CONFLICT: task already has different launch instructions'); + } + if (record.reference.expires_at <= Date.now()) { + throw new Error('PAYLOAD_BOOTSTRAP_EXPIRED: inspect the existing launch; do not start a replacement task'); + } + return record.reference; + }; + + // Refresh only identical, public deployment settings so bucket lifecycle + // expiry cannot reap an old manifest just as a new task starts using it. + await s3().send(new PutObjectCommand({ + Bucket: bucket, Key: manifestKey, Body: manifest, ContentType: 'application/json', + })); + const existing = await readObject(bucket, launchKey); + if (existing !== undefined) return accept(existing); + + const payloadKey = `${taskId}/payload.json`; + const savedPayload = await createOnce(bucket, payloadKey, payloadBody); + if (savedPayload !== payloadBody) { + throw new Error('PAYLOAD_BOOTSTRAP_CONFLICT: task already has different stored instructions'); + } + const now = Date.now(); + const credentials = await s3().config.credentials(); + const lifetime = Math.min( + PAYLOAD_BOOTSTRAP.url_ttl_seconds, + credentials.expiration + ? Math.floor((credentials.expiration.getTime() - now) / 1000) + : PAYLOAD_BOOTSTRAP.url_ttl_seconds, + ); + if (lifetime < PAYLOAD_BOOTSTRAP.minimum_url_lifetime_seconds) { + throw new Error('PAYLOAD_BOOTSTRAP_CREDENTIALS_EXPIRING: refresh coordinator credentials before launch'); + } + const url = await getSignedUrl(s3(), new GetObjectCommand({ + Bucket: bucket, Key: payloadKey, + }), { expiresIn: lifetime }); + const reference: PayloadReference = { + version: PAYLOAD_BOOTSTRAP.version, + task_id: taskId, + bootstrap_s3_uri: `s3://${bucket}/${manifestKey}`, + payload_url: url, + expires_at: now + lifetime * 1000, + }; + const record = canonical({ fingerprint, reference }); + return accept(await createOnce(bucket, launchKey, record)); +} + +/** Delete both task instructions and their saved capability; shared manifests expire by lifecycle. */ +export async function deletePayloadReference(bucket: string, taskId: string): Promise { + for (const filename of ['payload.json', PAYLOAD_BOOTSTRAP.launch_filename]) { + try { + await s3().send(new DeleteObjectCommand({ Bucket: bucket, Key: `${taskId}/${filename}` })); + } catch (error) { + logger.warn('Payload bootstrap cleanup failed (non-fatal)', { + task_id: taskId, filename, error: (error as { name?: string }).name ?? 'UnknownError', + }); + } + } +} diff --git a/cdk/src/handlers/shared/strategies/ecs-strategy.ts b/cdk/src/handlers/shared/strategies/ecs-strategy.ts index 75853ee39..0718f8b34 100644 --- a/cdk/src/handlers/shared/strategies/ecs-strategy.ts +++ b/cdk/src/handlers/shared/strategies/ecs-strategy.ts @@ -18,9 +18,9 @@ */ import { ECSClient, RunTaskCommand, DescribeTasksCommand, StopTaskCommand } from '@aws-sdk/client-ecs'; -import { S3Client, PutObjectCommand, DeleteObjectCommand } from '@aws-sdk/client-s3'; import type { ComputeStrategy, SessionHandle, SessionStatus } from '../compute-strategy'; import { logger } from '../logger'; +import { deletePayloadReference, preparePayloadReference, redactPayloadUrls } from '../payload-bootstrap'; import type { BlueprintConfig } from '../repo-config'; import { makeClient } from '../ua'; import { DEFAULT_MAX_TURNS } from '../validation'; @@ -33,14 +33,6 @@ function getClient(): ECSClient { return sharedClient; } -let sharedS3Client: S3Client | undefined; -function getS3Client(): S3Client { - if (!sharedS3Client) { - sharedS3Client = makeClient(S3Client); - } - return sharedS3Client; -} - const ECS_CLUSTER_ARN = process.env.ECS_CLUSTER_ARN; const ECS_TASK_DEFINITION_ARN = process.env.ECS_TASK_DEFINITION_ARN; /** @@ -79,45 +71,23 @@ export function toTaskDefinitionFamily(ref: string): string { const ECS_SECURITY_GROUP = process.env.ECS_SECURITY_GROUP; const ECS_CONTAINER_NAME = process.env.ECS_CONTAINER_NAME ?? 'AgentContainer'; const ECS_PAYLOAD_BUCKET = process.env.ECS_PAYLOAD_BUCKET; +const ECS_OVERRIDES_LIMIT_BYTES = 8192; -/** - * Inline-payload size (bytes) above which we warn that RunTask will likely - * reject the call when no payload bucket is configured. ECS caps the TOTAL - * containerOverrides blob at 8192 bytes; the other env vars + command consume - * some of that, so 6 KB of payload is the practical danger line. - */ -const INLINE_PAYLOAD_WARN_BYTES = 6144; - -/** - * S3 object key for a task's ECS payload. One object per task under its own - * task-id prefix; deleted by the orchestrator at finalize (see - * ``deleteEcsPayload``), with the bucket's 1-day lifecycle rule as a backstop. - */ -export function ecsPayloadKey(taskId: string): string { - return `${taskId}/payload.json`; +function safeLaunchError(error: unknown): Error { + const safe = new Error(redactPayloadUrls(error instanceof Error ? error.message : String(error))); + safe.name = error instanceof Error ? error.name : 'Error'; + return safe; } /** - * Delete a task's ECS payload object. Best-effort: a failed delete must never + * Delete a task's ECS payload and private launch reference. A failed delete must never * fail the task — the bucket's 1-day lifecycle rule reaps it regardless. Called * from the orchestrator's ``finalize`` step once the task is terminal. No-ops * when the payload bucket isn't configured (AgentCore-only deployments). */ export async function deleteEcsPayload(taskId: string): Promise { if (!ECS_PAYLOAD_BUCKET) return; - try { - await getS3Client().send(new DeleteObjectCommand({ - Bucket: ECS_PAYLOAD_BUCKET, - Key: ecsPayloadKey(taskId), - })); - logger.info('Deleted ECS payload object', { task_id: taskId }); - } catch (err) { - // Non-fatal — the lifecycle rule is the backstop. - logger.warn('Failed to delete ECS payload object (non-fatal)', { - task_id: taskId, - error: err instanceof Error ? err.message : String(err), - }); - } + await deletePayloadReference(ECS_PAYLOAD_BUCKET, taskId); } export class EcsComputeStrategy implements ComputeStrategy { @@ -170,43 +140,18 @@ export class EcsComputeStrategy implements ComputeStrategy { // that request. We override the container command to invoke run_task() // directly with the full orchestrator payload (including hydrated_context). // This avoids the server entirely and runs the agent in batch mode. - const payloadJson = JSON.stringify(payload); - - // The payload (especially hydrated_context) routinely exceeds the 8192-byte - // cap that ECS RunTask enforces on the TOTAL containerOverrides blob, which - // rejected the call with InvalidParameterException. Write the payload to S3 - // and pass only a small pointer (AGENT_PAYLOAD_S3_URI); the container fetches - // it on boot. The inline AGENT_PAYLOAD remains as a fallback for small - // payloads / deployments without a payload bucket configured. - let payloadS3Uri: string | undefined; - if (ECS_PAYLOAD_BUCKET) { - const key = ecsPayloadKey(taskId); - await getS3Client().send(new PutObjectCommand({ - Bucket: ECS_PAYLOAD_BUCKET, - Key: key, - Body: payloadJson, - ContentType: 'application/json', - })); - payloadS3Uri = `s3://${ECS_PAYLOAD_BUCKET}/${key}`; - logger.info('Wrote ECS payload to S3', { - task_id: taskId, - bytes: payloadJson.length, - uri: payloadS3Uri, - }); - } else if (payloadJson.length > INLINE_PAYLOAD_WARN_BYTES) { - // No bucket configured AND the payload is large enough that the inline - // path will almost certainly blow the 8192-byte overrides cap. Surface a - // clear cause rather than a raw InvalidParameterException from RunTask. - logger.warn('ECS payload is large but ECS_PAYLOAD_BUCKET is not set — RunTask may reject it (see #502)', { - task_id: taskId, - bytes: payloadJson.length, - }); + if (!ECS_PAYLOAD_BUCKET) { + throw new Error('PAYLOAD_BOOTSTRAP_INVALID: ECS_PAYLOAD_BUCKET is required; deploy matching infrastructure'); } + const reference = await preparePayloadReference({ + bucket: ECS_PAYLOAD_BUCKET, taskId, backend: 'ecs', payload, + }).catch((error: unknown) => { + throw safeLaunchError(error); + }); const containerEnv = [ { name: 'TASK_ID', value: taskId }, { name: 'REPO_URL', value: String(payload.repo_url ?? '') }, - ...(payload.prompt ? [{ name: 'TASK_DESCRIPTION', value: String(payload.prompt) }] : []), ...(payload.issue_number ? [{ name: 'ISSUE_NUMBER', value: String(payload.issue_number) }] : []), // Single source of truth with the hydrate path in `orchestrator.ts`, which // resolves the same default via `DEFAULT_MAX_TURNS`. A literal here would @@ -217,47 +162,34 @@ export class EcsComputeStrategy implements ComputeStrategy { ...(blueprintConfig.model_id ? [{ name: 'ANTHROPIC_MODEL', value: blueprintConfig.model_id }] : []), ...(blueprintConfig.system_prompt_overrides ? [{ name: 'SYSTEM_PROMPT_OVERRIDES', value: blueprintConfig.system_prompt_overrides }] : []), { name: 'CLAUDE_CODE_USE_BEDROCK', value: '1' }, - // Prefer the S3 pointer; fall back to the inline payload when no bucket is - // configured (keeps small-payload / AgentCore-only deployments working with - // no behavior change). - ...(payloadS3Uri - ? [{ name: 'AGENT_PAYLOAD_S3_URI', value: payloadS3Uri }] - : [{ name: 'AGENT_PAYLOAD', value: payloadJson }]), + { name: 'AGENT_PAYLOAD_REF', value: JSON.stringify(reference) }, ...(payload.github_token_secret_arn ? [{ name: 'GITHUB_TOKEN_SECRET_ARN', value: String(payload.github_token_secret_arn) }] : []), ...(payload.memory_id ? [{ name: 'MEMORY_ID', value: String(payload.memory_id) }] : []), ]; - // Override the container command to run a Python one-liner that: - // 1. Loads the payload — from S3 (AGENT_PAYLOAD_S3_URI) when set, else the - // inline AGENT_PAYLOAD env var (fallback). - // 2. Calls entrypoint.run_task_from_payload(p), which maps the WHOLE payload - // dict to run_task's signature (rename prompt→task_description / - // model_id→anthropic_model, filter to accepted params, coerce str/int). - // This replaces an older hand-listed kwarg subset that silently dropped - // fields such as channel_source/channel_metadata (which meant no - // Linear/Jira reactions or channel MCP on ECS), build_command, - // cedar_policies, base_branch/merge_branches, attachments, trace, user_id, - // etc. Single source of truth in the agent, unit-tested (see - // test_run_task_from_payload). - // 3. Exits with code 0 on success, 1 on failure. - // This bypasses the uvicorn server entirely — no HTTP, no OTEL noise. + // Consume the one-object capability before importing/running the pipeline. + // The helper removes it from os.environ so repo subprocesses do not inherit it. const bootCommand = [ 'python', '-c', - 'import json, os, sys; ' - + 'sys.path.insert(0, "/app/src"); ' + 'import sys; sys.path.insert(0, "/app/src"); ' + + 'from payload_bootstrap import load_ecs_payload; p = load_ecs_payload(); ' + 'from entrypoint import run_task_from_payload; ' - + '_uri = os.environ.get("AGENT_PAYLOAD_S3_URI"); ' - + 'p = (' - + 'json.loads(__import__("boto3").client("s3").get_object(' - + 'Bucket=_uri.split("/",3)[2], Key=_uri.split("/",3)[3])["Body"].read()) ' - + 'if _uri else json.loads(os.environ["AGENT_PAYLOAD"])' - + '); ' + 'r = run_task_from_payload(p); ' + 'sys.exit(0 if r.get("status")=="success" else 1)', ]; + const overrides = { + containerOverrides: [{ + name: ECS_CONTAINER_NAME, + environment: containerEnv, + command: bootCommand, + }], + }; + if (Buffer.byteLength(JSON.stringify(overrides), 'utf8') > ECS_OVERRIDES_LIMIT_BYTES) { + throw new Error('PAYLOAD_BOOTSTRAP_TOO_LARGE: ECS container overrides exceed 8192 bytes'); + } const command = new RunTaskCommand({ cluster: ECS_CLUSTER_ARN, taskDefinition, @@ -278,21 +210,17 @@ export class EcsComputeStrategy implements ComputeStrategy { assignPublicIp: 'DISABLED', }, }, - overrides: { - containerOverrides: [{ - name: ECS_CONTAINER_NAME, - environment: containerEnv, - command: bootCommand, - }], - }, + overrides, }); - const result = await getClient().send(command); + const result = await getClient().send(command).catch((error: unknown) => { + throw safeLaunchError(error); + }); const ecsTask = result.tasks?.[0]; if (!ecsTask?.taskArn) { const failures = result.failures?.map(f => `${f.arn}: ${f.reason}`).join('; ') ?? 'unknown'; - throw new Error(`ECS RunTask returned no task: ${failures}`); + throw new Error(`ECS RunTask returned no task: ${redactPayloadUrls(failures)}`); } logger.info('ECS Fargate task started', { diff --git a/cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts b/cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts index d278e3f37..be6845707 100644 --- a/cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts +++ b/cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts @@ -24,7 +24,6 @@ import { RunMicrovmCommand, TerminateMicrovmCommand, } from '@aws-sdk/client-lambda-microvms'; -import { DeleteObjectCommand, PutObjectCommand, S3Client } from '@aws-sdk/client-s3'; // Cross-language contract (S9): `microvm_platform_config` is read by BOTH this // producer and `agent/src/server.py`'s `/run` consumer. Imported (not copied) so // `tsc` fails on a renamed field — see `contracts/constants.md`. @@ -33,6 +32,7 @@ import type { ComputeStrategy, SessionHandle, SessionStatus } from '../compute-s import { MicrovmStartUncertainError } from '../error-classifier'; import { logger } from '../logger'; import { claimMicrovmStart, microvmStartRequestHash, saveMicrovmStartHandle } from '../microvm-start'; +import { deletePayloadReference, preparePayloadReference, redactPayloadUrls } from '../payload-bootstrap'; import type { BlueprintConfig } from '../repo-config'; import { makeClient } from '../ua'; @@ -44,14 +44,6 @@ function getClient(): LambdaMicrovmsClient { return sharedClient; } -let sharedS3Client: S3Client | undefined; -function getS3Client(): S3Client { - if (!sharedS3Client) { - sharedS3Client = makeClient(S3Client); - } - return sharedS3Client; -} - /** * Fully-qualified MicroVM image **ARN** passed as `imageIdentifier` on every * `RunMicrovm`. @@ -115,18 +107,8 @@ export const MICROVM_MAX_DURATION_SECONDS = 28_800; * old 16 384 threshold would have inlined every envelope between 4 097 and * 16 384 bytes and had the service reject all of them. * - * This is the EXACT branch point for the inline/S3-pointer decision, with no - * safety margin — deliberately unlike ``ecs-strategy``, which keeps its inline - * warn line at 6 144 of ECS's 8 192-byte cap. That margin exists because ECS - * counts the *whole* ``containerOverrides`` blob (env vars, command, and payload - * share one budget), so the strategy cannot know how much of the 8 192 the - * payload actually gets. ``runHookPayload`` is a single standalone string, so - * the counted size is exactly what we measure and the boundary is computable. - * - * Consequence worth stating plainly: at 4 KB the **S3-pointer path is the - * dominant one**. A hydrated task payload (prompt + issue thread + repo context) - * essentially always exceeds 4 KB, so the inline branch is the exception (tiny - * repo-less prompts), not the common case. + * V2 always sends a signed payload reference. Enforce this byte limit on the + * final serialized reference; payload/config bytes live in S3, not in the hook. */ const RUN_HOOK_PAYLOAD_LIMIT_BYTES = 4_096; @@ -201,7 +183,7 @@ const MICROVM_BENIGN_STATE_REASON = 'Success.'; * * The contract's declaration order is the emission order (`JSON.stringify` * preserves insertion order for string keys), which keeps the serialized - * envelope — and therefore the 4 KB inline/S3 branch decision — deterministic + * manifest serialization deterministic * for a given environment. */ const PLATFORM_CONFIG_CONTRACT = sharedConstants.microvm_platform_config; @@ -354,87 +336,36 @@ export function buildMicrovmPlatformConfig( export const MICROVM_ERROR_MARKER = 'MicroVM'; /** - * Wrap an error escaping a MicroVM control-plane (or payload-upload) call so it + * Wrap an error escaping a MicroVM control-plane or payload-bootstrap call so it * carries {@link MICROVM_ERROR_MARKER} plus the originating operation. * * The AWS exception NAME is spliced into the message explicitly because * ``err.message`` alone omits it (``String(err)`` would include it, but the * classifier is handed the *wrapped* error) and the classifier keys on that - * name. ``cause`` retains the original for anyone who needs ``err.name``. + * name. ``cause`` retains a sanitized name/message copy: SDK errors and their + * nested causes can contain signed URLs or request metadata. * * The wrapper's own ``name`` is intentionally left as ``Error`` so * ``String(wrapped)`` reads ``Error: MicroVM failed: : `` — * marker first, which is the order the classifier patterns document. */ function wrapMicrovmError(operation: string, err: unknown): Error { - const name = err instanceof Error ? err.name : undefined; - const message = err instanceof Error ? err.message : String(err); + const name = err instanceof Error ? redactPayloadUrls(err.name) : undefined; + const message = redactPayloadUrls(err instanceof Error ? err.message : String(err)); const detail = name && name !== 'Error' && !message.includes(name) ? `${name}: ${message}` : message; - return new Error(`${MICROVM_ERROR_MARKER} ${operation} failed: ${detail}`, { cause: err }); -} - -/** - * S3 object key for a task's MicroVM ``/run`` payload. Same key shape as the - * ECS payload bucket (``ecsPayloadKey``): one object per task under its own - * task-id prefix, deleted by the orchestrator at finalize (see - * {@link deleteMicrovmPayload}), with the payload bucket's lifecycle-expiry rule - * (ADR-021 sub-decision 3, ``MICROVM_PAYLOAD_TTL_DAYS``) as the backstop. - */ -export function microvmPayloadKey(taskId: string): string { - return `${taskId}/payload.json`; + const safeCause = new Error(message); + safeCause.name = name ?? 'Error'; + return new Error(`${MICROVM_ERROR_MARKER} ${operation} failed: ${detail}`, { cause: safeCause }); } -/** - * Delete a task's MicroVM ``/run`` payload object. Best-effort: a failed delete - * must never fail the task; the bucket's 1-day lifecycle rule is the fallback. - * Called from the orchestrator's ``finalize`` step once the task is terminal. - * No-ops when the payload bucket isn't configured. - * - * The execution role currently reads the whole payload bucket before task-scoped - * credentials are established. Untrusted task code can therefore read other - * tasks' payloads where it knows their keys. Deleting completed payloads shortens - * that exposure; it does not isolate active tasks. #700 tracks task-scoped - * transport. S3 processes lifecycle expiry asynchronously, so it is not an exact - * 24-hour bound on retention if this deletion fails. - * - * ISSUED UNCONDITIONALLY, including for a task whose payload went INLINE (under - * the {@link RUN_HOOK_PAYLOAD_LIMIT_BYTES} cap, so no object was ever written). - * That is on purpose: ``DeleteObject`` on a missing key succeeds, so the call is - * harmless and idempotent, whereas *deciding* to skip it would mean trusting a - * per-task record of the delivery mode — and if that record were ever wrong or - * absent, the skip would leave a real payload behind for the full TTL. Attempting - * always is the fail-safe direction. - * - * The consequence is that a successful call proves a delete was ISSUED, never that - * an object existed — S3 returns nothing that distinguishes the two on an - * unversioned bucket. The log line below says exactly that and no more; an earlier - * "Deleted MicroVM payload object" asserted a deletion that never happened on - * every inline task. +/** Remove task instructions and their saved download capability after finalization. + * Best-effort; bucket lifecycle reaps leftovers. Deployment manifests are shared. */ export async function deleteMicrovmPayload(taskId: string): Promise { if (!MICROVM_PAYLOAD_BUCKET) return; - const key = microvmPayloadKey(taskId); - try { - await getS3Client().send(new DeleteObjectCommand({ - Bucket: MICROVM_PAYLOAD_BUCKET, - Key: key, - })); - // "issued", not "deleted": see the docstring. An inline-delivered task has no - // object here and the call still succeeds. - logger.info('MicroVM payload delete issued', { - task_id: taskId, - bucket: MICROVM_PAYLOAD_BUCKET, - key, - }); - } catch (err) { - // Non-fatal — the lifecycle rule is the backstop. - logger.warn('Failed to delete MicroVM payload object (non-fatal)', { - task_id: taskId, - error: err instanceof Error ? err.message : String(err), - }); - } + await deletePayloadReference(MICROVM_PAYLOAD_BUCKET, taskId); } /** Split a comma-separated env-var list into trimmed, non-empty entries. */ @@ -547,93 +478,9 @@ export class LambdaMicrovmComputeStrategy implements ComputeStrategy { // payload upload so a misconfiguration never leaves an orphan S3 object. assertImageArn(MICROVM_IMAGE_IDENTIFIER); - // Payload delivery (ADR-021 sub-decision 3): the `/run` lifecycle hook - // receives `runHookPayload` as its request body, capped at 4 KB by the - // service. The hydrated_context essentially always blows that, so the - // S3-pointer path (mirroring ECS #502) is the DOMINANT one here and the - // inline branch is the exception. The MicroVM EXECUTION role holds the read - // grant, exactly as the ECS task role does today. - // - // Three keys, deliberately mirroring the ECS container env contract - // (AGENT_PAYLOAD / AGENT_PAYLOAD_S3_URI) so the agent's `/run` hook has one - // self-describing shape to branch on: - // { "agent_payload": {...}, "platform_config": {...} } — inline - // { "agent_payload_s3_uri": "…", "platform_config": {...} } — pointer - // - // `platform_config` (see MICROVM_PLATFORM_CONFIG_KEYS) rides in BOTH forms, - // and is ALSO merged into the S3 object on the pointer path: - // s3://…//payload.json = { ...agent_payload, "platform_config": {…} } - // The duplication is deliberate and cheap (a few hundred bytes). It is the - // agent's env-block substitute — nothing else delivers it, because the - // snapshot must not bake it in — so it must be reachable whether the agent - // reads it off the hook body before fetching S3 or out of the fetched object. + // The manifest authenticates deployment settings through the worker's IAM + // grant. Payload access uses a single-object URL, saved outside TaskTable. const platformConfig = buildMicrovmPlatformConfig(); - const inlineEnvelope = JSON.stringify({ agent_payload: payload, platform_config: platformConfig }); - // Measure the SERIALIZED envelope, not the bare payload: the envelope is - // what the service counts against the 4 KB cap, and `platform_config` is part - // of it — which is precisely why nearly everything lands on the S3 path. - // Byte length (not String.length) because a multi-byte prompt/diff makes - // chars an undercount. - const inlineBytes = Buffer.byteLength(inlineEnvelope, 'utf8'); - - let runHookPayload: string; - let payloadS3Uri: string | undefined; - let uploadPayload: (() => Promise) | undefined; - // EXACT boundary: `<= limit` inlines, `> limit` uploads. The service accepts - // 4 096 bytes and rejects 4 097 (measured), so 4 096 must still go inline. - if (inlineBytes <= RUN_HOOK_PAYLOAD_LIMIT_BYTES) { - runHookPayload = inlineEnvelope; - } else { - const key = microvmPayloadKey(taskId); - const uri = `s3://${MICROVM_PAYLOAD_BUCKET}/${key}`; - const pointerEnvelope = JSON.stringify({ - agent_payload_s3_uri: uri, - platform_config: platformConfig, - }); - // The pointer envelope is the LAST RESORT — there is no smaller shape to - // fall back to — so check it BEFORE the upload (an upload followed by a - // throw would leave an orphan object for the lifecycle rule to reap) and - // name the one thing an operator can actually act on. Unreachable in - // practice: the pointer plus all thirteen identifiers is well under 4 KB. - const pointerBytes = Buffer.byteLength(pointerEnvelope, 'utf8'); - if (pointerBytes > RUN_HOOK_PAYLOAD_LIMIT_BYTES) { - throw new Error( - `The MicroVM /run pointer envelope is ${pointerBytes} bytes, over the service's ` - + `${RUN_HOOK_PAYLOAD_LIMIT_BYTES}-byte runHookPayload cap, with the payload already moved ` - + 'to S3. The remaining size is the S3 URI plus the platform_config identifiers, so a ' - + 'pathologically long table/bucket/ARN name is the only possible cause — shorten the ' - + 'stack name (physical resource names derive from it) and redeploy.', - ); - } - // The S3 object carries the payload with `platform_config` merged in at the - // top level, so an agent that fetches the object gets the config with it. - // Platform config wins on a key collision — the payload has no - // `platform_config` key today, and if one ever appeared the platform's - // value is the authoritative one. - const payloadJson = JSON.stringify({ ...payload, platform_config: platformConfig }); - // Defer all writes until the persisted receipt accepts this exact input. - uploadPayload = async () => { - try { - await getS3Client().send(new PutObjectCommand({ - Bucket: MICROVM_PAYLOAD_BUCKET, - Key: key, - Body: payloadJson, - ContentType: 'application/json', - })); - } catch (err) { - throw wrapMicrovmError('payload upload', err); - } - logger.info('Wrote MicroVM run-hook payload to S3', { - task_id: taskId, - bytes: Buffer.byteLength(payloadJson, 'utf8'), - inline_bytes: inlineBytes, - inline_limit_bytes: RUN_HOOK_PAYLOAD_LIMIT_BYTES, - uri, - }); - }; - payloadS3Uri = uri; - runHookPayload = pointerEnvelope; - } // Explicit ingress control (F7, live 2026-07-31): `RunMicrovm` does NOT // default to "no ingress" — omitting the field attaches the AWS-managed @@ -661,7 +508,7 @@ export class LambdaMicrovmComputeStrategy implements ComputeStrategy { // Never omitted — see the comment above. `NO_INGRESS` is the suppression // mechanism, not an empty list. ingressNetworkConnectors, - runHookPayload, + runHookPayload: 'payload-bootstrap-v2', maximumDurationInSeconds: MICROVM_MAX_DURATION_SECONDS, // `idlePolicy` is OMITTED — never passed, in any phase (ADR-021 // sub-decision 1, asserted by an invariant unit test). MicroVM idle @@ -675,15 +522,21 @@ export class LambdaMicrovmComputeStrategy implements ComputeStrategy { // The receipt below supplies a task-stable token across new SDK commands. }; - const requestHash = microvmStartRequestHash(request, { ...payload, platform_config: platformConfig }); + const requestHash = microvmStartRequestHash( + { ...request, payloadBucket: MICROVM_PAYLOAD_BUCKET }, { ...payload, platform_config: platformConfig }, + ); const claim = await claimMicrovmStart(taskId, input.userId, requestHash); if (claim.closed) { if (claim.handle) await this.stopSession(claim.handle); throw new Error('MICROVM_START_TASK_CLOSED: task became terminal before session start'); } if (claim.handle) return claim.handle; - if (uploadPayload) { - await uploadPayload(); + const reference = await preparePayloadReference({ + bucket: MICROVM_PAYLOAD_BUCKET, taskId, backend: 'lambda-microvm', payload, platformConfig, + }).catch((error: unknown) => { throw wrapMicrovmError('payload bootstrap', error); }); + const runHookPayload = JSON.stringify(reference); + if (Buffer.byteLength(runHookPayload, 'utf8') > RUN_HOOK_PAYLOAD_LIMIT_BYTES) { + throw new Error('PAYLOAD_BOOTSTRAP_TOO_LARGE: launch reference exceeds the MicroVM hook limit'); } // Uploads can take time. Observe cancellation/another saved handle again // immediately before the service call, using the same immutable request. @@ -693,7 +546,7 @@ export class LambdaMicrovmComputeStrategy implements ComputeStrategy { throw new Error('MICROVM_START_TASK_CLOSED: task became terminal before session start'); } if (latest.handle) return latest.handle; - const command = new RunMicrovmCommand({ ...request, clientToken: latest.clientToken }); + const command = new RunMicrovmCommand({ ...request, runHookPayload, clientToken: latest.clientToken }); let result; try { @@ -716,7 +569,7 @@ export class LambdaMicrovmComputeStrategy implements ComputeStrategy { 'InvalidParameterValueException', 'ResourceNotFoundException', 'ThrottlingException', 'TooManyRequestsException', 'ServiceQuotaExceededException', 'ConflictException'] .includes(serviceError?.name ?? '')); - if (!knownRejection) throw new MicrovmStartUncertainError(wrapped.message, { cause: err }); + if (!knownRejection) throw new MicrovmStartUncertainError(wrapped.message, { cause: wrapped }); throw wrapped; } @@ -775,8 +628,7 @@ export class LambdaMicrovmComputeStrategy implements ComputeStrategy { image_arn: result.imageArn, image_version: result.imageVersion ?? MICROVM_IMAGE_VERSION, maximum_duration_seconds: MICROVM_MAX_DURATION_SECONDS, - payload_delivery: payloadS3Uri ? 's3_pointer' : 'inline', - ...(payloadS3Uri && { payload_s3_uri: payloadS3Uri }), + payload_delivery: 'signed_reference', // KEY NAMES only, never values: this is the one operator-visible record of // which optional platform identifiers a given session actually received, and // "the agent said ARTIFACTS_BUCKET_NAME is not configured" is otherwise a @@ -970,13 +822,13 @@ export class LambdaMicrovmComputeStrategy implements ComputeStrategy { microvm_id: microvmId, reason, error_type: errName, - error: err instanceof Error ? err.message : String(err), + error: redactPayloadUrls(err instanceof Error ? err.message : String(err)), }); } else { logger.warn('Failed to terminate MicroVM (best-effort)', { microvm_id: microvmId, reason, - error: err instanceof Error ? err.message : String(err), + error: redactPayloadUrls(err instanceof Error ? err.message : String(err)), }); } } @@ -986,7 +838,7 @@ export class LambdaMicrovmComputeStrategy implements ComputeStrategy { /** * Re-exported so tests and future callers can assert the documented cap without * duplicating the literal. This is BOTH the service's limit and our exact - * inline/S3-pointer branch point — there is no separate threshold. + * maximum serialized v2 launch-reference size. */ export const MICROVM_RUN_HOOK_PAYLOAD_LIMIT_BYTES = RUN_HOOK_PAYLOAD_LIMIT_BYTES; diff --git a/cdk/src/stacks/agent.ts b/cdk/src/stacks/agent.ts index 67bf1d004..77cb28fe9 100644 --- a/cdk/src/stacks/agent.ts +++ b/cdk/src/stacks/agent.ts @@ -955,10 +955,10 @@ export class AgentStack extends Stack { // agentcore-only, matching how other optional constructs are context-gated. // (``computeType`` is read near the top of the constructor — TaskApi needs it // for the conditional MicroVM cancel grant.) - // Ephemeral bucket for ECS task payloads — the orchestrator writes the - // payload here (it exceeds the 8 KB RunTask containerOverrides limit) and - // passes only an S3 URI pointer; the container fetches it on boot, the - // orchestrator deletes it at finalize. Only synthesized under the ecs gate. + // ECS v2 bootstrap storage: deployment manifests, task instructions and + // private launch references. Workers read manifests with IAM and download + // their task through a signed one-object URL. Finalize deletes task objects; + // the one-day lifecycle reaps leftovers. Synthesized only under the ECS gate. const ecsPayloadBucket = computeType === 'ecs' ? new EcsPayloadBucket(this, 'EcsPayloadBucket') : undefined; @@ -966,7 +966,7 @@ export class AgentStack extends Stack { NagSuppressions.addResourceSuppressions(ecsPayloadBucket.bucket, [ { id: 'AwsSolutions-S1', - reason: 'Ephemeral per-task payloads with a 1-day TTL; writes confined to the orchestrator IAM role by grantPut, reads to the ECS task role by grantRead, both scoped to this bucket. Object deleted at finalize. Object-level audit intentionally omitted — CloudTrail data events / a log bucket are not justified for transient boot payloads.', + reason: 'Ephemeral bootstrap storage with a 1-day TTL. The coordinator writes manifests and task objects, signs task reads and deletes task objects at finalize. Workers read only bootstrap/* with IAM and use single-object signed URLs for payloads; other object reads and bucket listing are explicitly denied. Object-level audit intentionally omitted — CloudTrail data events / a log bucket are not justified for transient boot payloads.', }, ]); } diff --git a/cdk/test/constructs/ecs-agent-cluster.test.ts b/cdk/test/constructs/ecs-agent-cluster.test.ts index 4af23466c..887001439 100644 --- a/cdk/test/constructs/ecs-agent-cluster.test.ts +++ b/cdk/test/constructs/ecs-agent-cluster.test.ts @@ -840,7 +840,7 @@ describe('EcsAgentCluster payload bucket (#502)', () => { }); }); - test('grants the task role READ on the payload bucket, never write/delete', () => { + test('allows only bootstrap reads and explicitly denies other objects and listing', () => { const template = createWithPayloadBucket(); const policies = template.findResources('AWS::IAM::Policy'); const s3Actions = new Set(); @@ -852,7 +852,13 @@ describe('EcsAgentCluster payload bucket (#502)', () => { } } } - // Read actions present... + const statements = Object.values(policies).flatMap(p => p.Properties.PolicyDocument.Statement); + const allow = statements.find(s => s.Effect === 'Allow' && s.Action === 's3:GetObject'); + expect(JSON.stringify(allow.Resource)).toContain('/bootstrap/*'); + expect(statements).toEqual(expect.arrayContaining([ + expect.objectContaining({ Effect: 'Deny', Action: 's3:GetObject*', NotResource: allow.Resource }), + expect.objectContaining({ Effect: 'Deny', Action: 's3:List*' }), + ])); expect([...s3Actions].some(a => a === 's3:GetObject' || a === 's3:GetObject*')).toBe(true); // ...and NO write/delete on the payload bucket from the task role. expect(s3Actions.has('s3:PutObject')).toBe(false); diff --git a/cdk/test/constructs/lambda-microvm-compute.test.ts b/cdk/test/constructs/lambda-microvm-compute.test.ts index 1a1bd62f3..e263a35c2 100644 --- a/cdk/test/constructs/lambda-microvm-compute.test.ts +++ b/cdk/test/constructs/lambda-microvm-compute.test.ts @@ -680,19 +680,23 @@ describe('LambdaMicrovmCompute — image provisioned from a managed base image', expect(JSON.stringify(s3Statement.Resource)).toContain(MICROVM_ARTIFACT_OBJECT_KEY); }); - test('execution role gets READ-ONLY on the payload bucket and no write/delete', () => { + test('execution role gets only bootstrap reads, explicit payload/list denies and no writes', () => { const policies = Object.entries(template.findResources('AWS::IAM::Policy')) .filter(([id]) => id.includes('LambdaMicrovmComputeExecutionRole')); const statements = policies.flatMap(([, p]) => p.Properties.PolicyDocument.Statement); const actions: string[] = statements.flatMap((s: { Action: string | string[] }) => Array.isArray(s.Action) ? s.Action : [s.Action]); - // CDK's grantRead renders the Get*/List* read set. - expect(actions).toContain('s3:GetObject*'); + const allowed = statements.filter((s: { Effect: string }) => s.Effect === 'Allow' && s.Action === 's3:GetObject'); + expect(allowed).toHaveLength(1); + expect(JSON.stringify(allowed[0].Resource)).toContain('/bootstrap/*'); + const denied = statements.find((s: { NotResource?: unknown }) => s.NotResource); + expect(denied.Effect).toBe('Deny'); + expect(denied.NotResource).toEqual(allowed[0].Resource); // The MicroVM runs untrusted repo code — it must not be able to clobber // another task's payload, so nothing mutating may appear. const s3Actions = actions.filter(a => a.startsWith('s3:')); - expect(s3Actions).toEqual(['s3:GetObject*', 's3:GetBucket*', 's3:List*']); + expect(s3Actions).toEqual(['s3:GetObject', 's3:GetObject*', 's3:List*']); for (const action of s3Actions) { expect(action).not.toMatch(/Put|Delete|Abort|Write|^s3:\*$/); } @@ -744,7 +748,7 @@ describe('LambdaMicrovmCompute — image provisioned from a managed base image', // ecs-agent-cluster). Without it a Linear/Jira task's 👀→✅ reaction and the // channel MCP silently no-op. const prefixStatement = executionRoleStatements().find( - statement => JSON.stringify(statement.Resource).includes('bgagent-linear-oauth-*'), + statement => JSON.stringify(statement.Resource ?? '').includes('bgagent-linear-oauth-*'), )!; expect(prefixStatement).toBeDefined(); expect(prefixStatement.Action).toBe('secretsmanager:GetSecretValue'); @@ -799,7 +803,7 @@ describe('LambdaMicrovmCompute — image provisioned from a managed base image', // why it survived P2: the substrate looked fine while the platform's canonical // observability streams were empty. const statements = executionRoleStatements().filter((statement) => { - const resource = JSON.stringify(statement.Resource); + const resource = JSON.stringify(statement.Resource ?? ''); return resource.includes('ApplicationLogGroup'); }); expect(statements).toHaveLength(1); @@ -892,7 +896,7 @@ describe('LambdaMicrovmCompute — image provisioned from a managed base image', const s3Resources = executionRoleStatements() .flatMap(st => (Array.isArray(st.Action) ? st.Action : [st.Action])) .filter(action => action.startsWith('s3:')); - expect(s3Resources).toEqual(['s3:GetObject*', 's3:GetBucket*', 's3:List*']); + expect(s3Resources).toEqual(['s3:GetObject', 's3:GetObject*', 's3:List*']); const rendered = JSON.stringify( executionRoleStatements().filter((st) => { const actions = Array.isArray(st.Action) ? st.Action : [st.Action]; diff --git a/cdk/test/constructs/payload-bootstrap-permissions.test.ts b/cdk/test/constructs/payload-bootstrap-permissions.test.ts new file mode 100644 index 000000000..20bfddfba --- /dev/null +++ b/cdk/test/constructs/payload-bootstrap-permissions.test.ts @@ -0,0 +1,67 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import { App, Stack } from 'aws-cdk-lib'; +import { Template } from 'aws-cdk-lib/assertions'; +import * as iam from 'aws-cdk-lib/aws-iam'; +import * as s3 from 'aws-cdk-lib/aws-s3'; +import { grantCoordinatorPayloads, grantWorkerBootstrap } from '../../src/constructs/payload-bootstrap-permissions'; + +describe('payload bootstrap permission boundary', () => { + let template: Template; + beforeAll(() => { + const stack = new Stack(new App(), 'PayloadPermissions'); + const bucket = new s3.Bucket(stack, 'Payloads'); + const worker = new iam.Role(stack, 'Worker', { assumedBy: new iam.ServicePrincipal('lambda.amazonaws.com') }); + const coordinator = new iam.Role(stack, 'Coordinator', { assumedBy: new iam.ServicePrincipal('lambda.amazonaws.com') }); + grantWorkerBootstrap(bucket, worker); + grantCoordinatorPayloads(bucket, coordinator); + template = Template.fromStack(stack); + }); + + test('worker can read only deployment manifests, including against another public bucket', () => { + const policy = Object.entries(template.findResources('AWS::IAM::Policy')) + .find(([id]) => id.startsWith('Worker'))![1].Properties.PolicyDocument.Statement; + expect(policy).toHaveLength(3); + const allow = policy.find((s: { Effect: string }) => s.Effect === 'Allow'); + expect(allow.Action).toBe('s3:GetObject'); + expect(JSON.stringify(allow.Resource)).toContain('/bootstrap/*'); + const deny = policy.find((s: { NotResource?: unknown }) => s.NotResource); + expect(deny.Effect).toBe('Deny'); + expect(deny.Action).toBe('s3:GetObject*'); + expect(deny.NotResource).toEqual(allow.Resource); + expect(policy.find((s: { Action: string }) => s.Action === 's3:List*').Effect).toBe('Deny'); + expect(JSON.stringify(policy)).not.toContain('s3:Put'); + expect(JSON.stringify(policy)).not.toContain('s3:Delete'); + }); + + test('coordinator publishes manifests and privately persists, signs and deletes task references', () => { + const policy = Object.entries(template.findResources('AWS::IAM::Policy')) + .find(([id]) => id.startsWith('Coordinator'))![1].Properties.PolicyDocument.Statement; + const writes = policy.find((s: { Action: string }) => s.Action === 's3:PutObject'); + expect(JSON.stringify(writes.Resource)).toContain('/bootstrap/*'); + const reads = policy.find((s: { Action: string[] }) => Array.isArray(s.Action) && s.Action.includes('s3:GetObject')); + expect(reads.Action).toEqual(['s3:GetObject', 's3:DeleteObject']); + expect(JSON.stringify(reads.Resource)).toContain('*/payload.json'); + expect(JSON.stringify(reads.Resource)).toContain('*/launch.json'); + expect(JSON.stringify(reads.Resource)).not.toContain('/bootstrap/*'); + const list = policy.find((s: { Action: string }) => s.Action === 's3:ListBucket'); + expect(list.Resource).toEqual({ 'Fn::GetAtt': [expect.stringMatching(/^Payloads/), 'Arn'] }); + }); +}); diff --git a/cdk/test/constructs/task-orchestrator.test.ts b/cdk/test/constructs/task-orchestrator.test.ts index 7cb64500f..214cda176 100644 --- a/cdk/test/constructs/task-orchestrator.test.ts +++ b/cdk/test/constructs/task-orchestrator.test.ts @@ -814,17 +814,17 @@ describe('TaskOrchestrator with the Lambda MicroVMs backend (ADR-021)', () => { const actions = payloadStatements.flatMap(s => Array.isArray(s.Action) ? s.Action : [s.Action]); expect(actions).toContain('s3:PutObject'); expect(actions).toContain('s3:DeleteObject'); - expect(actions.some(action => action.startsWith('s3:Get') || action.startsWith('s3:List'))).toBe(false); + expect(actions).toContain('s3:GetObject'); + expect(actions.filter(action => action.startsWith('s3:List'))).toEqual(['s3:ListBucket']); + const list = payloadStatements.find(s => s.Action === 's3:ListBucket'); + expect(list!.Resource).toEqual({ 'Fn::GetAtt': [expect.stringMatching(/^MicrovmPayloadBucket/), 'Arn'] }); const deletes = payloadStatements.filter(s => (Array.isArray(s.Action) ? s.Action : [s.Action]).some(action => action.startsWith('s3:Delete'))); expect(deletes).toHaveLength(1); - expect(deletes[0]!.Action).toBe('s3:DeleteObject'); - expect(deletes[0]!.Resource).toEqual({ - 'Fn::Join': ['', [ - { 'Fn::GetAtt': [expect.stringMatching(/^MicrovmPayloadBucket/), 'Arn'] }, - '/*/payload.json', - ]], - }); + expect(deletes[0]!.Action).toEqual(['s3:GetObject', 's3:DeleteObject']); + expect(JSON.stringify(deletes[0]!.Resource)).toContain('/*/payload.json'); + expect(JSON.stringify(deletes[0]!.Resource)).toContain('/*/launch.json'); + expect(JSON.stringify(deletes[0]!.Resource)).not.toContain('/bootstrap/*'); }); test('adds no MicroVM statements when microvmConfig is omitted', () => { diff --git a/cdk/test/handlers/orchestrate-task-microvm.test.ts b/cdk/test/handlers/orchestrate-task-microvm.test.ts index 21a5b6b38..86d4dd8aa 100644 --- a/cdk/test/handlers/orchestrate-task-microvm.test.ts +++ b/cdk/test/handlers/orchestrate-task-microvm.test.ts @@ -58,8 +58,11 @@ jest.mock('@aws-sdk/client-lambda-microvms', () => ({ // PUT and the finalize DELETE are both assertions this file needs to make, and a // per-instance mock silently discards them. const mockS3Send = jest.fn().mockResolvedValue({}); +jest.mock('@aws-sdk/s3-request-presigner', () => ({ getSignedUrl: async () => 'https://payloads.s3.us-east-1.amazonaws.com/task/payload.json?X-Amz-Signature=' + Date.now() })); +const mockObjects = new Map(); jest.mock('@aws-sdk/client-s3', () => ({ - S3Client: jest.fn(() => ({ send: mockS3Send })), + GetObjectCommand: jest.fn((input: unknown) => ({ _type: 'GetObject', input })), + S3Client: jest.fn(() => ({ send: mockS3Send, config: { credentials: async () => ({ accessKeyId: 'EXAMPLE', secretAccessKey: 'unused' }) } })), PutObjectCommand: jest.fn((input: unknown) => ({ _type: 'PutObject', input })), DeleteObjectCommand: jest.fn((input: unknown) => ({ _type: 'DeleteObject', input })), })); @@ -249,8 +252,17 @@ beforeEach(() => { })); mockClaimStart.mockReset().mockImplementation(async (taskId: string) => ({ clientToken: taskId, closed: false })); mockSaveHandle.mockReset().mockResolvedValue(undefined); - mockS3Send.mockReset(); - mockS3Send.mockResolvedValue({}); + mockObjects.clear(); + mockS3Send.mockReset().mockImplementation(async ({ _type: type, input: command }) => { + if (type === 'GetObject') { + if (!mockObjects.has(command.Key)) throw Object.assign(new Error('missing'), { name: 'NoSuchKey' }); + return { Body: { transformToString: async () => mockObjects.get(command.Key) } }; + } + if (type === 'DeleteObject') { mockObjects.delete(command.Key); return {}; } + if (command.IfNoneMatch === '*' && mockObjects.has(command.Key)) throw Object.assign(new Error('exists'), { name: 'PreconditionFailed' }); + mockObjects.set(command.Key, command.Body); + return {}; + }); mockMicrovmSend.mockReset(); mockPollTaskStatus.mockResolvedValue({ attempts: 1, lastStatus: TaskStatus.COMPLETED }); mockReconcile.mockResolvedValue({ taskFailed: false }); @@ -488,21 +500,23 @@ describe('orchestrate-task for a lambda-microvm task', () => { // --- finalize-time payload delete (review NB3) --- - test('deletes its OWN payload object on finalize, closing the cross-task read window', async () => { - // The execution role's payload-bucket grant is bucket-wide `grantRead`, so a - // TTL-only reaper left every finished task's hydrated prompt readable by any - // running MicroVM for ~24 h. ECS already deleted at finalize; this is the parity - // that matters. + test('deletes its own payload and saved capability on finalize', async () => { + // Revoke the signed payload link and remove its private replay record. + // The one-day lifecycle remains a cleanup backstop. runMicrovmOk(); await handler({ task_id: 'TASK001' }, fakeContext().ctx as never); const deletes = s3CommandsOfType('DeleteObject'); - expect(deletes).toHaveLength(1); + expect(deletes).toHaveLength(2); expect(deletes[0].input).toEqual({ Bucket: 'test-microvm-payload-bucket', Key: 'TASK001/payload.json', }); + expect(deletes[1].input).toEqual({ + Bucket: 'test-microvm-payload-bucket', + Key: 'TASK001/launch.json', + }); // ...and it did not reach for the ECS deleter. expect(mockDeleteEcsPayload).not.toHaveBeenCalled(); }); diff --git a/cdk/test/handlers/shared/microvm-start-recovery.test.ts b/cdk/test/handlers/shared/microvm-start-recovery.test.ts index 717f2044f..882bcae71 100644 --- a/cdk/test/handlers/shared/microvm-start-recovery.test.ts +++ b/cdk/test/handlers/shared/microvm-start-recovery.test.ts @@ -34,8 +34,12 @@ jest.mock('@aws-sdk/client-lambda-microvms', () => ({ TerminateMicrovmCommand: jest.fn((input: unknown) => ({ kind: 'terminate', input })), MicrovmState: {}, })); +jest.mock('@aws-sdk/s3-request-presigner', () => ({ getSignedUrl: async () => 'https://payloads.s3.us-east-1.amazonaws.com/task/payload.json?X-Amz-Signature=' + Date.now() })); +const mockObjects = new Map(); jest.mock('@aws-sdk/client-s3', () => ({ - S3Client: jest.fn(() => ({ send: mockS3Send })), + DeleteObjectCommand: jest.fn((input: unknown) => ({ kind: 'delete', input })), + GetObjectCommand: jest.fn((input: unknown) => ({ kind: 'get', input })), + S3Client: jest.fn(() => ({ send: mockS3Send, config: { credentials: async () => ({ accessKeyId: 'EXAMPLE', secretAccessKey: 'unused' }) } })), PutObjectCommand: jest.fn((input: unknown) => ({ kind: 'put', input })), })); jest.mock('../../../src/handlers/shared/logger', () => ({ @@ -85,7 +89,17 @@ beforeEach(() => { created = new Map(); now = Date.parse('2026-09-13T15:00:00Z'); jest.spyOn(Date, 'now').mockImplementation(() => now); - mockS3Send.mockReset().mockResolvedValue({}); + mockObjects.clear(); + mockS3Send.mockReset().mockImplementation(async ({ kind: type, input: command }) => { + if (type === 'get') { + if (!mockObjects.has(command.Key)) throw Object.assign(new Error('missing'), { name: 'NoSuchKey' }); + return { Body: { transformToString: async () => mockObjects.get(command.Key) } }; + } + if (type === 'delete') { mockObjects.delete(command.Key); return {}; } + if (command.IfNoneMatch === '*' && mockObjects.has(command.Key)) throw Object.assign(new Error('exists'), { name: 'PreconditionFailed' }); + mockObjects.set(command.Key, command.Body); + return {}; + }); mockDdbSend.mockReset().mockImplementation(async ({ kind, input: command }) => { if (kind === 'get') return { Item: structuredClone(record) }; const values = command.ExpressionAttributeValues; @@ -197,7 +211,7 @@ test('replay after saving a handle makes no further RunMicrovm or payload write' now += MICROVM_START_REPLAY_WINDOW_MS * 10; expect(await new LambdaMicrovmComputeStrategy().startSession(input)).toEqual(handle); expect(runCalls()).toHaveLength(1); - expect(mockS3Send).toHaveBeenCalledTimes(1); + expect(mockS3Send.mock.calls.filter(([c]) => c.kind === 'put' && c.input.Key.endsWith('/payload.json'))).toHaveLength(1); }); test('changed input is refused before overwriting the first task payload', async () => { @@ -206,7 +220,7 @@ test('changed input is refused before overwriting the first task payload', async await expect(new LambdaMicrovmComputeStrategy().startSession({ ...input, payload: { ...input.payload, prompt: 'changed'.repeat(1_000) }, })).rejects.toThrow('MICROVM_START_INPUT_CHANGED'); - expect(mockS3Send).toHaveBeenCalledTimes(1); + expect(mockS3Send.mock.calls.filter(([c]) => c.kind === 'put' && c.input.Key.endsWith('/payload.json'))).toHaveLength(1); expect(runCalls()).toHaveLength(1); }); @@ -217,7 +231,7 @@ test('an expired unknown start never receives a new token or another RunMicrovm await expect(new LambdaMicrovmComputeStrategy().startSession(input)) .rejects.toThrow('MICROVM_START_OUTCOME_UNKNOWN'); expect(runCalls()).toHaveLength(1); - expect(mockS3Send).toHaveBeenCalledTimes(1); + expect(mockS3Send.mock.calls.filter(([c]) => c.kind === 'put' && c.input.Key.endsWith('/payload.json'))).toHaveLength(1); }); test.each([TaskStatus.CANCELLED, TaskStatus.COMPLETED, TaskStatus.FAILED, TaskStatus.TIMED_OUT])( diff --git a/cdk/test/handlers/shared/payload-bootstrap.test.ts b/cdk/test/handlers/shared/payload-bootstrap.test.ts new file mode 100644 index 000000000..091c58f07 --- /dev/null +++ b/cdk/test/handlers/shared/payload-bootstrap.test.ts @@ -0,0 +1,160 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import { createHash } from 'node:crypto'; + +const mockSend = jest.fn(); +const mockSign = jest.fn(); +const mockCredentials = jest.fn(); +jest.mock('@aws-sdk/client-s3', () => ({ + S3Client: jest.fn(() => ({ send: mockSend, config: { credentials: mockCredentials } })), + PutObjectCommand: jest.fn(input => ({ kind: 'put', input })), + GetObjectCommand: jest.fn(input => ({ kind: 'get', input })), + DeleteObjectCommand: jest.fn(input => ({ kind: 'delete', input })), +})); +jest.mock('@aws-sdk/s3-request-presigner', () => ({ + getSignedUrl: (...args: unknown[]) => mockSign(...args), +})); + +import { deletePayloadReference, PAYLOAD_BOOTSTRAP, preparePayloadReference, redactPayloadUrls } from '../../../src/handlers/shared/payload-bootstrap'; + +const objects = new Map(); +const input = { + bucket: 'payload-bucket', + taskId: 'task-1', + backend: 'lambda-microvm' as const, + payload: { task_id: 'task-1', prompt: 'hello' }, + platformConfig: { github_token_secret_arn: 'arn:aws:secretsmanager:us-west-2:123456789012:secret:github' }, +}; +let loseReplyFor: string | undefined; + +beforeEach(() => { + jest.clearAllMocks(); + objects.clear(); + loseReplyFor = undefined; + mockCredentials.mockResolvedValue({ accessKeyId: 'EXAMPLE', secretAccessKey: 'unused' }); + mockSign.mockImplementation(async () => `https://payload-bucket.s3.us-east-1.amazonaws.com/task-1/payload.json?X-Amz-Security-Token=secret&X-Amz-Signature=${mockSign.mock.calls.length}`); + mockSend.mockImplementation(async ({ kind, input: request }) => { + if (kind === 'get') { + if (!objects.has(request.Key)) throw Object.assign(new Error('missing'), { name: 'NoSuchKey' }); + return { Body: { transformToString: async () => objects.get(request.Key)! } }; + } + if (kind === 'delete') { objects.delete(request.Key); return {}; } + if (request.IfNoneMatch === '*' && objects.has(request.Key)) { + throw Object.assign(new Error('exists'), { name: 'PreconditionFailed' }); + } + objects.set(request.Key, request.Body); + if (loseReplyFor === request.Key) { + loseReplyFor = undefined; + throw new Error('lost committed reply'); + } + return {}; + }); +}); + +test('separates public manifest from private payload and saved capability', async () => { + const ref = await preparePayloadReference(input); + const key = ref.bootstrap_s3_uri.split('/').slice(3).join('/'); + const manifest = objects.get(key)!; + expect(key).toBe(`bootstrap/${createHash('sha256').update(manifest).digest('hex')}.json`); + expect(JSON.parse(manifest)).toEqual({ + version: 2, backend: 'lambda-microvm', platform_config: input.platformConfig, + }); + expect(manifest).not.toContain('hello'); + expect(manifest).not.toContain('X-Amz'); + expect(JSON.parse(objects.get('task-1/payload.json')!).agent_payload).toEqual(input.payload); + expect(JSON.parse(objects.get('task-1/launch.json')!).reference).toEqual(ref); + expect(mockSign).toHaveBeenCalledWith(expect.anything(), expect.objectContaining({ + input: { Bucket: input.bucket, Key: 'task-1/payload.json' }, + }), { expiresIn: 900 }); +}); + +test('replay and concurrent preparation return identical launch bytes without resigning on replay', async () => { + const first = await preparePayloadReference(input); + const replay = await preparePayloadReference({ ...input, payload: { prompt: 'hello', task_id: 'task-1' } }); + expect(JSON.stringify(replay)).toBe(JSON.stringify(first)); + expect(mockSign).toHaveBeenCalledTimes(1); + // Fixed pair, covering conditional object publication by competing callers. + objects.clear(); + // eslint-disable-next-line @cdklabs/promiseall-no-unbounded-parallelism + const [a, b] = await Promise.all([preparePayloadReference(input), preparePayloadReference(input)]); + expect(JSON.stringify(a)).toBe(JSON.stringify(b)); +}); + +test.each(['task-1/payload.json', 'task-1/launch.json'])('recovers a lost committed response for %s', async key => { + loseReplyFor = key; + const first = await preparePayloadReference(input); + expect(await preparePayloadReference(input)).toEqual(first); +}); + +test('changed instructions cannot overwrite an existing payload or launch reference', async () => { + await preparePayloadReference(input); + const before = objects.get('task-1/payload.json'); + await expect(preparePayloadReference({ ...input, payload: { ...input.payload, prompt: 'changed' } })) + .rejects.toThrow('PAYLOAD_BOOTSTRAP_CONFLICT'); + expect(objects.get('task-1/payload.json')).toBe(before); +}); + +test('orphaned payload after a crash cannot be overwritten with changed instructions', async () => { + await preparePayloadReference(input); + objects.delete('task-1/launch.json'); + await expect(preparePayloadReference({ ...input, payload: { ...input.payload, prompt: 'changed' } })) + .rejects.toThrow('PAYLOAD_BOOTSTRAP_CONFLICT'); + expect(JSON.parse(objects.get('task-1/payload.json')!).agent_payload.prompt).toBe('hello'); +}); + +test('expired references fail closed without signing a replacement', async () => { + await preparePayloadReference(input); + const record = JSON.parse(objects.get('task-1/launch.json')!); + record.reference.expires_at = Date.now() - 1; + objects.set('task-1/launch.json', JSON.stringify(record)); + await expect(preparePayloadReference(input)).rejects.toThrow('PAYLOAD_BOOTSTRAP_EXPIRED'); + expect(mockSign).toHaveBeenCalledTimes(1); +}); + +test('credential lifetime bounds the link and refuses credentials expiring too soon', async () => { + mockCredentials.mockResolvedValue({ expiration: new Date(Date.now() + 20_000) }); + await expect(preparePayloadReference(input)).rejects.toThrow('CREDENTIALS_EXPIRING'); + expect(mockSign).not.toHaveBeenCalled(); +}); + +test.each(['../bootstrap', 'task/other', ''])('invalid task key %s performs no S3 writes', async taskId => { + await expect(preparePayloadReference({ ...input, taskId })).rejects.toThrow('identity'); + expect(mockSend).not.toHaveBeenCalled(); +}); + +test('oversized payload fails before uploading or signing', async () => { + await expect(preparePayloadReference({ + ...input, payload: { ...input.payload, prompt: 'x'.repeat(PAYLOAD_BOOTSTRAP.max_payload_bytes) }, + })).rejects.toThrow('TOO_LARGE'); + expect(mockSend).not.toHaveBeenCalled(); +}); + +test('cleanup removes the capability and payload, preserving shared configuration', async () => { + const ref = await preparePayloadReference(input); + await deletePayloadReference(input.bucket, input.taskId); + expect(objects.has('task-1/payload.json')).toBe(false); + expect(objects.has('task-1/launch.json')).toBe(false); + expect(objects.has(ref.bootstrap_s3_uri.split('/').slice(3).join('/'))).toBe(true); +}); + +test('error redaction removes the whole signed URL including the session token', () => { + expect(redactPayloadUrls('failed https://b.s3.us-east-1.amazonaws.com/key?X-Amz-Security-Token=secret&X-Amz-Signature=abc end')) + .toBe('failed [redacted payload URL] end'); +}); diff --git a/cdk/test/handlers/shared/strategies/ecs-strategy.test.ts b/cdk/test/handlers/shared/strategies/ecs-strategy.test.ts index db474ddd3..904e133cd 100644 --- a/cdk/test/handlers/shared/strategies/ecs-strategy.test.ts +++ b/cdk/test/handlers/shared/strategies/ecs-strategy.test.ts @@ -27,13 +27,9 @@ process.env.ECS_TASK_DEFINITION_ARN = TASK_DEF_ARN; process.env.ECS_SUBNETS = 'subnet-aaa,subnet-bbb'; process.env.ECS_SECURITY_GROUP = 'sg-12345'; process.env.ECS_CONTAINER_NAME = 'AgentContainer'; -// The top-of-file import's inline-fallback / no-op tests assume these OPTIONAL -// vars are ABSENT at load time. They are unset in a dev shell but the real ECS -// agent container HAS ECS_PAYLOAD_BUCKET set (#502) — so leaving this to ambient -// env made the build pass locally yet FAIL on ECS ("works local, dies on ECS"). -// The #502 / #299 describe blocks below set these via isolateModules; delete them -// here so the top-of-file import is hermetic regardless of the runner's env. -delete process.env.ECS_PAYLOAD_BUCKET; +// Pin the payload bucket and optional planning definition at import time so the +// test runner's own deployment environment cannot change these expectations. +process.env.ECS_PAYLOAD_BUCKET = 'payload-bucket'; delete process.env.ECS_PLANNING_TASK_DEFINITION_ARN; const mockSend = jest.fn(); @@ -44,17 +40,20 @@ jest.mock('@aws-sdk/client-ecs', () => ({ StopTaskCommand: jest.fn((input: unknown) => ({ _type: 'StopTask', input })), })); -const mockS3Send = jest.fn(); -jest.mock('@aws-sdk/client-s3', () => ({ - S3Client: jest.fn(() => ({ send: mockS3Send })), - PutObjectCommand: jest.fn((input: unknown) => ({ _type: 'PutObject', input })), - DeleteObjectCommand: jest.fn((input: unknown) => ({ _type: 'DeleteObject', input })), +const mockPrepare = jest.fn(); +const mockDelete = jest.fn(); +jest.mock('../../../../src/handlers/shared/payload-bootstrap', () => ({ + ...jest.requireActual('../../../../src/handlers/shared/payload-bootstrap'), + preparePayloadReference: (...args: unknown[]) => mockPrepare(...args), + deletePayloadReference: (...args: unknown[]) => mockDelete(...args), })); import { EcsComputeStrategy } from '../../../../src/handlers/shared/strategies/ecs-strategy'; beforeEach(() => { jest.clearAllMocks(); + mockPrepare.mockReset().mockImplementation(async ({ taskId }: { taskId: string }) => ({ version: 2, task_id: taskId, bootstrap_s3_uri: 's3://b/bootstrap/example.json', payload_url: 'https://signed.example/task', expires_at: Date.now()+900000 })); + mockDelete.mockResolvedValue(undefined); }); describe('EcsComputeStrategy', () => { @@ -103,21 +102,14 @@ describe('EcsComputeStrategy', () => { expect(envVars).toEqual(expect.arrayContaining([ { name: 'TASK_ID', value: 'TASK001' }, { name: 'REPO_URL', value: 'org/repo' }, - { name: 'TASK_DESCRIPTION', value: 'Fix the bug' }, { name: 'ISSUE_NUMBER', value: '42' }, { name: 'MAX_TURNS', value: '50' }, { name: 'CLAUDE_CODE_USE_BEDROCK', value: '1' }, ])); - // No ECS_PAYLOAD_BUCKET in this module's env → inline fallback (#502): the - // full payload rides in AGENT_PAYLOAD and nothing is written to S3. - const agentPayload = envVars.find((e: { name: string }) => e.name === 'AGENT_PAYLOAD'); - expect(agentPayload).toBeDefined(); - const parsed = JSON.parse(agentPayload.value); - expect(parsed.repo_url).toBe('org/repo'); - expect(parsed.prompt).toBe('Fix the bug'); - expect(envVars.find((e: { name: string }) => e.name === 'AGENT_PAYLOAD_S3_URI')).toBeUndefined(); - expect(mockS3Send).not.toHaveBeenCalled(); + const reference = envVars.find((e: { name: string }) => e.name === 'AGENT_PAYLOAD_REF'); + expect(JSON.parse(reference.value).task_id).toBe('TASK001'); + expect(mockPrepare).toHaveBeenCalledWith(expect.objectContaining({ payload: expect.objectContaining({ repo_url: 'org/repo', prompt: 'Fix the bug' }) })); // Container command override — runs Python directly instead of uvicorn expect(override.command).toBeDefined(); @@ -379,113 +371,47 @@ describe('EcsComputeStrategy', () => { }); }); -// #502: the S3-pointer path requires ECS_PAYLOAD_BUCKET to be set BEFORE the -// module is imported (it's a module-level constant). Re-import under -// jest.isolateModules with the env var set so these tests don't perturb the -// inline-fallback tests above. -describe('EcsComputeStrategy with ECS_PAYLOAD_BUCKET (S3-pointer path, #502)', () => { - const PAYLOAD_BUCKET = 'test-ecs-payload-bucket'; - - function loadStrategyWithBucket(): { - EcsComputeStrategy: typeof import('../../../../src/handlers/shared/strategies/ecs-strategy').EcsComputeStrategy; - deleteEcsPayload: typeof import('../../../../src/handlers/shared/strategies/ecs-strategy').deleteEcsPayload; - ecsPayloadKey: typeof import('../../../../src/handlers/shared/strategies/ecs-strategy').ecsPayloadKey; - } { - let mod!: ReturnType; - jest.isolateModules(() => { - process.env.ECS_PAYLOAD_BUCKET = PAYLOAD_BUCKET; - process.env.ECS_CLUSTER_ARN = CLUSTER_ARN; - process.env.ECS_TASK_DEFINITION_ARN = TASK_DEF_ARN; - process.env.ECS_SUBNETS = 'subnet-aaa,subnet-bbb'; - process.env.ECS_SECURITY_GROUP = 'sg-12345'; - process.env.ECS_CONTAINER_NAME = 'AgentContainer'; - // eslint-disable-next-line @typescript-eslint/no-require-imports - mod = require('../../../../src/handlers/shared/strategies/ecs-strategy'); - }); - return mod; - } - - afterEach(() => { - delete process.env.ECS_PAYLOAD_BUCKET; - }); - - test('writes payload to S3 and passes AGENT_PAYLOAD_S3_URI, not the inline blob', async () => { - mockS3Send.mockResolvedValueOnce({}); +describe('ECS v2 bootstrap integration', () => { + test('passes the prepared capability and consumes it before entering the pipeline', async () => { mockSend.mockResolvedValueOnce({ tasks: [{ taskArn: TASK_ARN }] }); + await new EcsComputeStrategy().startSession({ taskId: 'TASK001', userId: 'u1', payload: { task_id: 'TASK001', prompt: 'x'.repeat(10000) }, blueprintConfig: { compute_type: 'ecs', runtime_arn: '' } }); + expect(mockPrepare).toHaveBeenCalledWith(expect.objectContaining({ backend: 'ecs', taskId: 'TASK001', bucket: 'payload-bucket' })); + const override=mockSend.mock.calls[0][0].input.overrides.containerOverrides[0]; + expect(override.environment.some((e: { name: string }) => ['AGENT_PAYLOAD', 'AGENT_PAYLOAD_S3_URI', 'TASK_DESCRIPTION'].includes(e.name))).toBe(false); + const source=override.command[2]; + expect(source).toContain('load_ecs_payload()'); + expect(source.indexOf('load_ecs_payload()')).toBeLessThan(source.indexOf('from entrypoint')); + }); + test('delegates deletion of the payload and private launch capability', async () => { + const { deleteEcsPayload }=await import('../../../../src/handlers/shared/strategies/ecs-strategy'); + await deleteEcsPayload('TASK001'); + expect(mockDelete).toHaveBeenCalledWith('payload-bucket', 'TASK001'); + }); - const { EcsComputeStrategy: Strategy } = loadStrategyWithBucket(); - const strategy = new Strategy(); - await strategy.startSession({ + test('rejects oversized UTF-8 overrides before RunTask', async () => { + await expect(new EcsComputeStrategy().startSession({ taskId: 'TASK001', - userId: 'cognito-test', - payload: { repo_url: 'org/repo', prompt: 'Fix the bug', hydrated_context: { big: 'x'.repeat(10000) } }, - blueprintConfig: { compute_type: 'ecs', runtime_arn: '' }, - }); - - // PutObject to the payload bucket at /payload.json - expect(mockS3Send).toHaveBeenCalledTimes(1); - const put = mockS3Send.mock.calls[0][0]; - expect(put._type).toBe('PutObject'); - expect(put.input.Bucket).toBe(PAYLOAD_BUCKET); - expect(put.input.Key).toBe('TASK001/payload.json'); - expect(JSON.parse(put.input.Body).repo_url).toBe('org/repo'); - - // Override carries the URI pointer, NOT the inline payload - const envVars = mockSend.mock.calls[0][0].input.overrides.containerOverrides[0].environment; - const uri = envVars.find((e: { name: string }) => e.name === 'AGENT_PAYLOAD_S3_URI'); - expect(uri.value).toBe(`s3://${PAYLOAD_BUCKET}/TASK001/payload.json`); - expect(envVars.find((e: { name: string }) => e.name === 'AGENT_PAYLOAD')).toBeUndefined(); + userId: 'u1', + payload: { task_id: 'TASK001' }, + blueprintConfig: { compute_type: 'ecs', runtime_arn: '', system_prompt_overrides: '🌍'.repeat(2100) }, + })).rejects.toThrow('ECS container overrides exceed 8192 bytes'); + expect(mockSend).not.toHaveBeenCalled(); }); - test('boot command loads payload from S3 when the URI is set, else inline', async () => { - mockS3Send.mockResolvedValueOnce({}); - mockSend.mockResolvedValueOnce({ tasks: [{ taskArn: TASK_ARN }] }); - - const { EcsComputeStrategy: Strategy } = loadStrategyWithBucket(); - await new Strategy().startSession({ + test('redacts a capability even when preparation fails before RunTask', async () => { + mockPrepare.mockRejectedValueOnce(new Error('failed https://bucket.example/key?X-Amz-Signature=BEARER-SECRET')); + await expect(new EcsComputeStrategy().startSession({ taskId: 'TASK001', - userId: 'cognito-test', - payload: { repo_url: 'org/repo' }, + userId: 'u1', + payload: { task_id: 'TASK001' }, blueprintConfig: { compute_type: 'ecs', runtime_arn: '' }, - }); - - const cmd = mockSend.mock.calls[0][0].input.overrides.containerOverrides[0].command; - const src = cmd[2]; - // Reads the URI, fetches via boto3 S3 when set, falls back to inline env. - expect(src).toContain('AGENT_PAYLOAD_S3_URI'); - expect(src).toContain('get_object'); - expect(src).toContain('AGENT_PAYLOAD'); - // ABCA-487: the boot command maps the WHOLE payload via - // run_task_from_payload (not a hand-listed kwarg subset that dropped - // channel_source/channel_metadata → no Linear reactions on ECS). Assert we - // call the mapper and no longer hand-pick the old prompt/model_id kwargs. - expect(src).toContain('run_task_from_payload(p)'); - expect(src).not.toContain('task_description=p.get'); - expect(src).not.toContain('channel_source'); // never hand-listed; the mapper forwards it - }); - - test('deleteEcsPayload deletes the task payload object', async () => { - mockS3Send.mockResolvedValueOnce({}); - const { deleteEcsPayload, ecsPayloadKey } = loadStrategyWithBucket(); - await deleteEcsPayload('TASK001'); - expect(mockS3Send).toHaveBeenCalledTimes(1); - const del = mockS3Send.mock.calls[0][0]; - expect(del._type).toBe('DeleteObject'); - expect(del.input.Bucket).toBe(PAYLOAD_BUCKET); - expect(del.input.Key).toBe(ecsPayloadKey('TASK001')); - expect(ecsPayloadKey('TASK001')).toBe('TASK001/payload.json'); - }); - - test('deleteEcsPayload swallows S3 errors (best-effort — lifecycle is the backstop)', async () => { - mockS3Send.mockRejectedValueOnce(new Error('AccessDenied')); - const { deleteEcsPayload } = loadStrategyWithBucket(); - await expect(deleteEcsPayload('TASK001')).resolves.toBeUndefined(); + })).rejects.toThrow('failed [redacted payload URL]'); + expect(mockSend).not.toHaveBeenCalled(); }); }); // #299 ECS_RIGHTSIZED_PLANNING: the planning task def ARN is a module-level -// constant, so set it BEFORE import via isolateModules (mirrors the #502 bucket -// pattern above) — this keeps it out of the inline tests at the top. +// constant, so set it BEFORE import via isolateModules. describe('EcsComputeStrategy read-only planning-def selection (#299 ECS_RIGHTSIZED_PLANNING)', () => { const PLANNING_DEF_ARN = 'arn:aws:ecs:us-east-1:123456789012:task-definition/agent-planning:1'; @@ -548,12 +474,34 @@ describe('EcsComputeStrategy read-only planning-def selection (#299 ECS_RIGHTSIZ }); }); -describe('deleteEcsPayload without ECS_PAYLOAD_BUCKET', () => { +describe('ECS without ECS_PAYLOAD_BUCKET', () => { + let isolated: typeof import('../../../../src/handlers/shared/strategies/ecs-strategy'); + beforeEach(() => { + const bucket = process.env.ECS_PAYLOAD_BUCKET; + delete process.env.ECS_PAYLOAD_BUCKET; + try { + jest.isolateModules(() => { + // eslint-disable-next-line @typescript-eslint/no-require-imports + isolated = require('../../../../src/handlers/shared/strategies/ecs-strategy'); + }); + } finally { + process.env.ECS_PAYLOAD_BUCKET = bucket; + } + }); + test('no-ops when no payload bucket is configured', async () => { - // The top-of-file import has no ECS_PAYLOAD_BUCKET set. - // eslint-disable-next-line @typescript-eslint/no-require-imports - const { deleteEcsPayload } = require('../../../../src/handlers/shared/strategies/ecs-strategy'); - await expect(deleteEcsPayload('TASK001')).resolves.toBeUndefined(); - expect(mockS3Send).not.toHaveBeenCalled(); + await expect(isolated.deleteEcsPayload('TASK001')).resolves.toBeUndefined(); + expect(mockDelete).not.toHaveBeenCalled(); + }); + + test('rejects a launch before preparing or running anything', async () => { + await expect(new isolated.EcsComputeStrategy().startSession({ + taskId: 'TASK001', + userId: 'u1', + payload: { task_id: 'TASK001' }, + blueprintConfig: { compute_type: 'ecs', runtime_arn: '' }, + })).rejects.toThrow('ECS_PAYLOAD_BUCKET is required'); + expect(mockPrepare).not.toHaveBeenCalled(); + expect(mockSend).not.toHaveBeenCalled(); }); }); diff --git a/cdk/test/handlers/shared/strategies/lambda-microvm-strategy.test.ts b/cdk/test/handlers/shared/strategies/lambda-microvm-strategy.test.ts index ac213407d..2d85a6676 100644 --- a/cdk/test/handlers/shared/strategies/lambda-microvm-strategy.test.ts +++ b/cdk/test/handlers/shared/strategies/lambda-microvm-strategy.test.ts @@ -101,11 +101,12 @@ jest.mock('@aws-sdk/client-lambda-microvms', () => ({ }, })); -const mockS3Send = jest.fn(); -jest.mock('@aws-sdk/client-s3', () => ({ - S3Client: jest.fn(() => ({ send: mockS3Send })), - PutObjectCommand: jest.fn((input: unknown) => ({ _type: 'PutObject', input })), - DeleteObjectCommand: jest.fn((input: unknown) => ({ _type: 'DeleteObject', input })), +const mockPrepare = jest.fn(); +const mockDelete = jest.fn(); +jest.mock('../../../../src/handlers/shared/payload-bootstrap', () => ({ + ...jest.requireActual('../../../../src/handlers/shared/payload-bootstrap'), + preparePayloadReference: (...args: unknown[]) => mockPrepare(...args), + deletePayloadReference: (...args: unknown[]) => mockDelete(...args), })); // The real logger writes JSON to process.stdout/stderr, so level assertions need @@ -129,7 +130,6 @@ import { buildMicrovmPlatformConfig, deleteMicrovmPayload, microvmNoIngressConnectorArnForRegion, - microvmPayloadKey, } from '../../../../src/handlers/shared/strategies/lambda-microvm-strategy'; const BLUEPRINT: BlueprintConfig = { compute_type: 'lambda-microvm', runtime_arn: '' }; @@ -145,31 +145,6 @@ const EXPECTED_PLATFORM_CONFIG = { agent_session_role_arn: AGENT_SESSION_ROLE_ARN, }; -/** - * Build a payload whose serialized `{"agent_payload": …, "platform_config": …}` - * envelope is EXACTLY `targetBytes` long, so the 4 KB boundary can be probed on - * both sides. - * - * `platform_config` is part of the counted envelope (ADR-021 P2), so its bytes are - * subtracted from the payload's budget here. Asserts its own arithmetic — if the - * envelope shape or the platform block ever changes, this fails loudly rather than - * silently testing the wrong boundary. - */ -function payloadWithEnvelopeBytes(targetBytes: number): Record { - const overhead = Buffer.byteLength( - JSON.stringify({ agent_payload: { p: '' }, platform_config: EXPECTED_PLATFORM_CONFIG }), - 'utf8', - ); - const payload = { p: 'x'.repeat(targetBytes - overhead) }; - expect( - Buffer.byteLength( - JSON.stringify({ agent_payload: payload, platform_config: EXPECTED_PLATFORM_CONFIG }), - 'utf8', - ), - ).toBe(targetBytes); - return payload; -} - function runMicrovmOk() { mockSend.mockResolvedValueOnce({ microvmId: MICROVM_ID, @@ -257,22 +232,10 @@ async function withoutEnvAsync(keys: string[], body: () => Promise): Promi } } -/** {@link withoutEnvAsync}'s inverse: run `body` with extra env vars SET. */ -async function withEnvAsync(env: Record, body: () => Promise): Promise { - const saved = Object.fromEntries(Object.keys(env).map(key => [key, process.env[key]])); - try { - Object.assign(process.env, env); - await body(); - } finally { - for (const [key, value] of Object.entries(saved)) { - if (value === undefined) delete process.env[key]; - else process.env[key] = value; - } - } -} - beforeEach(() => { jest.clearAllMocks(); + mockPrepare.mockReset().mockImplementation(async ({ taskId }: { taskId: string }) => ({ version: 2, task_id: taskId, bootstrap_s3_uri: 's3://b/bootstrap/example.json', payload_url: 'https://signed.example/task', expires_at: Date.now()+900000 })); + mockDelete.mockResolvedValue(undefined); mockClaimStart.mockReset().mockImplementation(async (taskId: string) => ({ clientToken: taskId, closed: false })); mockSaveHandle.mockReset().mockResolvedValue(undefined); }); @@ -422,142 +385,17 @@ describe('LambdaMicrovmComputeStrategy', () => { ]); }); - test('inlines a small payload in runHookPayload and never touches S3', async () => { - runMicrovmOk(); - - await new LambdaMicrovmComputeStrategy().startSession({ - taskId: 'TASK001', - userId: 'cognito-test', - payload: { repo_url: 'org/repo', prompt: 'Fix the bug', max_turns: 50 }, - blueprintConfig: BLUEPRINT, - }); - - expect(mockS3Send).not.toHaveBeenCalled(); - const envelope = JSON.parse(mockSend.mock.calls[0][0].input.runHookPayload); - expect(envelope.agent_payload).toEqual({ repo_url: 'org/repo', prompt: 'Fix the bug', max_turns: 50 }); - expect(envelope.agent_payload_s3_uri).toBeUndefined(); - // The MicroVM's substitute for the env block the other two backends get at - // deploy time — a snapshot must not bake it in (ADR-021 sub-decision 3), so - // it rides the envelope alongside the payload. - expect(envelope.platform_config).toEqual(EXPECTED_PLATFORM_CONFIG); - }); - - test('uploads an oversized payload to S3 and inlines only the pointer', async () => { - mockS3Send.mockResolvedValueOnce({}); - runMicrovmOk(); - - const big = { repo_url: 'org/repo', hydrated_context: { blob: 'x'.repeat(20_000) } }; - await new LambdaMicrovmComputeStrategy().startSession({ - taskId: 'TASK001', - userId: 'cognito-test', - payload: big, - blueprintConfig: BLUEPRINT, - }); - - // Same key shape as the ECS payload bucket: /payload.json - expect(mockS3Send).toHaveBeenCalledTimes(1); - const put = mockS3Send.mock.calls[0][0]; - expect(put._type).toBe('PutObject'); - expect(put.input.Bucket).toBe(PAYLOAD_BUCKET); - expect(put.input.Key).toBe('TASK001/payload.json'); - expect(put.input.ContentType).toBe('application/json'); - // The S3 object carries the payload with platform_config merged in at the top - // level, so an agent that resolves the pointer gets the config with it. - expect(JSON.parse(put.input.Body)).toEqual({ ...big, platform_config: EXPECTED_PLATFORM_CONFIG }); - - const runHookPayload = mockSend.mock.calls[0][0].input.runHookPayload; - const envelope = JSON.parse(runHookPayload); - expect(envelope.agent_payload_s3_uri).toBe(`s3://${PAYLOAD_BUCKET}/TASK001/payload.json`); - expect(envelope.agent_payload).toBeUndefined(); - // ...and ALSO inline on the pointer envelope, deliberately duplicated: the - // agent must be able to read its platform configuration whether it takes it - // off the hook body before fetching S3 or out of the fetched object. - expect(envelope.platform_config).toEqual(EXPECTED_PLATFORM_CONFIG); - // The whole point: the hook body must sit far under the 4 KB cap. - expect(Buffer.byteLength(runHookPayload, 'utf8')).toBeLessThan(MICROVM_RUN_HOOK_PAYLOAD_LIMIT_BYTES); - }); - - test('the inline/S3 branch point IS the 4096-byte service cap, with no headroom', () => { - // Live-measured, NOT read off the SDK docs (which say 16,384): the service - // rejects 4097 with "Member must have length less than or equal to 4096". - // Unlike ECS (whose 8192-byte cap is shared with env vars + command, so it - // needs a margin), runHookPayload is the entire counted string. - expect(MICROVM_RUN_HOOK_PAYLOAD_LIMIT_BYTES).toBe(4_096); - }); - - test('a payload whose envelope is EXACTLY 4096 bytes stays inline', async () => { - runMicrovmOk(); - - await new LambdaMicrovmComputeStrategy().startSession({ - taskId: 'TASK001', - userId: 'cognito-test', - payload: payloadWithEnvelopeBytes(4_096), - blueprintConfig: BLUEPRINT, - }); - - // Boundary is `<=`: 4096 passed the service's length validation live. - expect(mockS3Send).not.toHaveBeenCalled(); - const runHookPayload = mockSend.mock.calls[0][0].input.runHookPayload; - expect(Buffer.byteLength(runHookPayload, 'utf8')).toBe(4_096); - expect(JSON.parse(runHookPayload).agent_payload).toBeDefined(); - }); - - test('a payload whose envelope is 4097 bytes — one over — goes to S3', async () => { - mockS3Send.mockResolvedValueOnce({}); - runMicrovmOk(); - - await new LambdaMicrovmComputeStrategy().startSession({ - taskId: 'TASK001', - userId: 'cognito-test', - payload: payloadWithEnvelopeBytes(4_097), - blueprintConfig: BLUEPRINT, - }); - - expect(mockS3Send).toHaveBeenCalledTimes(1); - const envelope = JSON.parse(mockSend.mock.calls[0][0].input.runHookPayload); - expect(envelope.agent_payload_s3_uri).toBe(`s3://${PAYLOAD_BUCKET}/TASK001/payload.json`); - expect(envelope.agent_payload).toBeUndefined(); - }); - - test('a mid-sized envelope the SDK-documented 16KB cap would have inlined goes to S3', async () => { - mockS3Send.mockResolvedValueOnce({}); - runMicrovmOk(); - - // Regression guard for the live-verification fix: anything from 4,097 to - // 16,384 bytes used to be inlined and would be REJECTED by the service. - await new LambdaMicrovmComputeStrategy().startSession({ - taskId: 'TASK001', - userId: 'cognito-test', - payload: payloadWithEnvelopeBytes(13_000), - blueprintConfig: BLUEPRINT, - }); - - expect(mockS3Send).toHaveBeenCalledTimes(1); - const envelope = JSON.parse(mockSend.mock.calls[0][0].input.runHookPayload); - expect(envelope.agent_payload_s3_uri).toBeDefined(); + test.each(['small', 'x'.repeat(20000)])('delivers a persisted v2 reference for every payload size', async prompt => { + mockSend.mockResolvedValueOnce({ microvmId: MICROVM_ID, endpoint: ENDPOINT }); + await new LambdaMicrovmComputeStrategy().startSession({ taskId: 'TASK001', userId: 'u1', payload: { prompt }, blueprintConfig: BLUEPRINT }); + expect(mockPrepare).toHaveBeenCalledWith(expect.objectContaining({ bucket: PAYLOAD_BUCKET, backend: 'lambda-microvm', platformConfig: EXPECTED_PLATFORM_CONFIG, payload: { prompt } })); + const envelope=JSON.parse(mockSend.mock.calls[0][0].input.runHookPayload); + expect(envelope.version).toBe(2); + expect(envelope.task_id).toBe('TASK001'); + expect(envelope.platform_config).toBeUndefined(); expect(envelope.agent_payload).toBeUndefined(); }); - test('measures the serialized envelope in BYTES, so a multi-byte payload still goes to S3', async () => { - mockS3Send.mockResolvedValueOnce({}); - runMicrovmOk(); - - // 3-byte UTF-8 characters: 2000 chars is ~6 KB of bytes but only 2 KB of - // chars, so measuring String.length would have wrongly inlined this. - const payload = { prompt: '\u4f60'.repeat(2_000) }; - expect(JSON.stringify({ agent_payload: payload, platform_config: EXPECTED_PLATFORM_CONFIG }).length) - .toBeLessThan(MICROVM_RUN_HOOK_PAYLOAD_LIMIT_BYTES); - await new LambdaMicrovmComputeStrategy().startSession({ - taskId: 'TASK001', - userId: 'cognito-test', - payload, - blueprintConfig: BLUEPRINT, - }); - - expect(mockS3Send).toHaveBeenCalledTimes(1); - expect(JSON.parse(mockSend.mock.calls[0][0].input.runHookPayload).agent_payload_s3_uri).toBeDefined(); - }); - test('throws when RunMicrovm returns no microvmId', async () => { mockSend.mockResolvedValueOnce({ endpoint: ENDPOINT, state: 'PENDING' }); @@ -654,34 +492,6 @@ describe('LambdaMicrovmComputeStrategy', () => { expect(mockSend.mock.calls.filter(c => c[0]._type === 'TerminateMicrovm')).toHaveLength(0); }); - test('platform_config COUNTS toward the 4 KB boundary — an otherwise-inlineable payload goes to S3', async () => { - // The regression this locks: measuring `{agent_payload}` alone and then - // sending `{agent_payload, platform_config}` would inline an envelope the - // service rejects outright. So the branch decision has to be made on the - // FULL envelope. This payload is exactly 4 096 bytes WITHOUT the platform - // block — i.e. the old code would have inlined it — and must now upload. - mockS3Send.mockResolvedValueOnce({}); - runMicrovmOk(); - - const overheadWithoutConfig = Buffer.byteLength(JSON.stringify({ agent_payload: { p: '' } }), 'utf8'); - const payload = { p: 'x'.repeat(MICROVM_RUN_HOOK_PAYLOAD_LIMIT_BYTES - overheadWithoutConfig) }; - expect(Buffer.byteLength(JSON.stringify({ agent_payload: payload }), 'utf8')) - .toBe(MICROVM_RUN_HOOK_PAYLOAD_LIMIT_BYTES); - - await new LambdaMicrovmComputeStrategy().startSession({ - taskId: 'TASK001', - userId: 'cognito-test', - payload, - blueprintConfig: BLUEPRINT, - }); - - expect(mockS3Send).toHaveBeenCalledTimes(1); - const runHookPayload = mockSend.mock.calls[0][0].input.runHookPayload; - expect(JSON.parse(runHookPayload).agent_payload_s3_uri).toBeDefined(); - expect(Buffer.byteLength(runHookPayload, 'utf8')) - .toBeLessThanOrEqual(MICROVM_RUN_HOOK_PAYLOAD_LIMIT_BYTES); - }); - test.each(MICROVM_PLATFORM_CONFIG_REQUIRED_KEYS.map(key => [key]))( 'refuses to start — before any AWS call — when required platform config %s is missing', async (key) => { @@ -693,14 +503,14 @@ describe('LambdaMicrovmComputeStrategy', () => { userId: 'cognito-test', // Oversized on purpose: the guard must fire before the payload upload, // or a misconfiguration leaves orphan objects in the payload bucket. - payload: payloadWithEnvelopeBytes(20_000), + payload: { prompt: 'x'.repeat(20_000) }, blueprintConfig: BLUEPRINT, }); await expect(start).rejects.toThrow(new RegExp(`${key} <- ${envVar}`)); await expect(start).rejects.toThrow(/redeploy the stack/); expect(mockSend).not.toHaveBeenCalled(); - expect(mockS3Send).not.toHaveBeenCalled(); + expect(mockPrepare).not.toHaveBeenCalled(); }); }, ); @@ -723,72 +533,48 @@ describe('LambdaMicrovmComputeStrategy', () => { expect(JSON.stringify(started[1])).not.toContain(GITHUB_TOKEN_SECRET_ARN); }); - test('fails BEFORE the upload when even the POINTER envelope cannot fit (no orphan object)', async () => { - // The one shape with no smaller fallback: the payload has already been moved - // to S3, so if `{pointer + platform_config}` still exceeds 4 096 bytes there - // is nothing left to shed. Only a pathological identifier length can cause it - // — hence the check, and hence its placement BEFORE the PutObject so a - // misconfiguration cannot leave objects behind for the lifecycle rule to reap. - await withEnvAsync({ LOG_GROUP_NAME: `/aws/${'x'.repeat(5_000)}` }, async () => { - const start = new LambdaMicrovmComputeStrategy().startSession({ - taskId: 'TASK001', - userId: 'cognito-test', - payload: payloadWithEnvelopeBytes(20_000), - blueprintConfig: BLUEPRINT, - }); + test('rejects a saved reference exceeding the service cap before RunMicrovm', async () => { + mockPrepare.mockResolvedValueOnce({ version: 2, task_id: 'TASK001', payload_url: 'x'.repeat(MICROVM_RUN_HOOK_PAYLOAD_LIMIT_BYTES + 1) }); + await expect(new LambdaMicrovmComputeStrategy().startSession({ taskId: 'TASK001', userId: 'u1', payload: {}, blueprintConfig: BLUEPRINT })).rejects.toThrow('hook limit'); + expect(mockSend).not.toHaveBeenCalled(); + }); - await expect(start).rejects.toThrow(/pointer envelope is \d+ bytes/); - await expect(start).rejects.toThrow(/shorten the stack name/); - expect(mockS3Send).not.toHaveBeenCalled(); - expect(mockSend).not.toHaveBeenCalled(); + test.each(['bootstrap', 'run'])('redacts signed URLs throughout the %s error chain', async stage => { + const { inspect } = await import('node:util'); + const original = new Error('request failed https://bucket.example/key?X-Amz-Signature=BEARER-SECRET', { + cause: new Error('nested BEARER-SECRET'), }); + if (stage === 'bootstrap') mockPrepare.mockRejectedValueOnce(original); + else mockSend.mockRejectedValueOnce(original); + await new LambdaMicrovmComputeStrategy().startSession({ + taskId: 'TASK001', userId: 'u1', payload: {}, blueprintConfig: BLUEPRINT, + }).catch(error => { + expect(inspect(error, { depth: null })).not.toContain('BEARER-SECRET'); + }); + expect.assertions(1); }); - test('microvmPayloadKey matches the ECS payload key shape', () => { - expect(microvmPayloadKey('TASK001')).toBe('TASK001/payload.json'); + test('keeps the capability out of ordinary logs and start-receipt arguments', async () => { + mockPrepare.mockResolvedValueOnce({ version: 2, task_id: 'TASK001', payload_url: 'BEARER-SECRET' }); + runMicrovmOk(); + await new LambdaMicrovmComputeStrategy().startSession({ + taskId: 'TASK001', userId: 'u1', payload: {}, blueprintConfig: BLUEPRINT, + }); + expect(JSON.stringify([ + mockLogger.info.mock.calls, mockLogger.warn.mock.calls, mockLogger.error.mock.calls, + mockClaimStart.mock.calls, mockSaveHandle.mock.calls, + ])).not.toContain('BEARER-SECRET'); }); }); // --- finalize-time payload delete (review NB3 / ayushtr nit 2) --- // - // Not cosmetic parity with ECS. The execution role's payload-bucket grant is - // `grantRead` on the WHOLE bucket (the guest must read its object before any - // tenant identity exists) and keys are `/payload.json`, so a TTL-only - // reaper left every finished task's HYDRATED PROMPT readable by any concurrently - // running MicroVM — which runs untrusted repo code — for up to ~24 h. - describe('deleteMicrovmPayload — closes the cross-task payload-read window', () => { - test('deletes the task\'s own object from the payload bucket', async () => { - await deleteMicrovmPayload('TASK001'); - - expect(mockS3Send).toHaveBeenCalledTimes(1); - const call = mockS3Send.mock.calls[0][0]; - expect(call._type).toBe('DeleteObject'); - expect(call.input).toEqual({ - Bucket: PAYLOAD_BUCKET, - Key: microvmPayloadKey('TASK001'), - }); - }); - - test('is best-effort — a failed delete never throws', async () => { - mockS3Send.mockRejectedValueOnce(new Error('AccessDenied')); - - await expect(deleteMicrovmPayload('TASK001')).resolves.toBeUndefined(); - // Not silent: the lifecycle rule is the backstop, but an operator must be - // able to see the delete is failing. - expect(mockLogger.warn).toHaveBeenCalledWith( - 'Failed to delete MicroVM payload object (non-fatal)', - expect.objectContaining({ task_id: 'TASK001', error: 'AccessDenied' }), - ); - }); - - test('deletes ONLY the given task\'s key — never a prefix or the bucket', async () => { - // The delete must not become a cleanup that can reach another task's object. + // Finalize revokes the single-object link and removes its private replay + // record; worker credentials already deny direct task-object reads. + describe('deleteMicrovmPayload', () => { + test('delegates cleanup of payload and private launch record', async () => { await deleteMicrovmPayload('TASK001'); - - const { input } = mockS3Send.mock.calls[0][0]; - expect(input.Key).toBe('TASK001/payload.json'); - expect(input.Key).not.toContain('*'); - expect(input).not.toHaveProperty('Prefix'); + expect(mockDelete).toHaveBeenCalledWith(PAYLOAD_BUCKET, 'TASK001'); }); }); @@ -1003,7 +789,7 @@ describe('LambdaMicrovmComputeStrategy', () => { await expect(start).rejects.toThrow('MicroVM RunMicrovm failed: ThrottlingException: Rate exceeded'); }); - test('preserves the original error as `cause` so err.name stays inspectable', async () => { + test('preserves the AWS error name in a sanitized cause', async () => { const err = new Error('quota'); err.name = 'ServiceQuotaExceededException'; mockSend.mockRejectedValueOnce(err); @@ -1011,7 +797,7 @@ describe('LambdaMicrovmComputeStrategy', () => { await new LambdaMicrovmComputeStrategy() .startSession({ taskId: 'TASK001', userId: 'u', payload: {}, blueprintConfig: BLUEPRINT }) .catch((thrown: Error) => { - expect(thrown.cause).toBe(err); + expect(thrown.cause).not.toBe(err); expect((thrown.cause as Error).name).toBe('ServiceQuotaExceededException'); }); expect.assertions(2); @@ -1029,16 +815,16 @@ describe('LambdaMicrovmComputeStrategy', () => { test('marks a payload-upload failure so an S3 fault is attributed to this backend', async () => { const err = new Error('Access Denied'); err.name = 'AccessDenied'; - mockS3Send.mockRejectedValueOnce(err); + mockPrepare.mockRejectedValueOnce(err); const start = new LambdaMicrovmComputeStrategy().startSession({ taskId: 'TASK001', userId: 'cognito-test', - payload: payloadWithEnvelopeBytes(20_000), + payload: { prompt: 'x'.repeat(20_000) }, blueprintConfig: BLUEPRINT, }); - await expect(start).rejects.toThrow('MicroVM payload upload failed: AccessDenied: Access Denied'); + await expect(start).rejects.toThrow('MicroVM payload bootstrap failed: AccessDenied: Access Denied'); // Never reaches RunMicrovm — no half-started MicroVM on an upload fault. expect(mockSend).not.toHaveBeenCalled(); }); @@ -1106,7 +892,7 @@ describe('LambdaMicrovmComputeStrategy without the MicroVM substrate deployed', await expect(start).rejects.toThrow(new RegExp(envVar)); // Fails BEFORE any AWS call — no half-started MicroVM, no orphan S3 object. expect(mockSend).not.toHaveBeenCalled(); - expect(mockS3Send).not.toHaveBeenCalled(); + expect(mockPrepare).not.toHaveBeenCalled(); }); test('MICROVM_IMAGE_VERSION is optional — the field is omitted so the service picks the default', async () => { @@ -1200,7 +986,7 @@ describe('LambdaMicrovmComputeStrategy image-identifier validation', () => { userId: 'cognito-test', // Oversized on purpose: the guard must fire before the payload upload, or a // misconfiguration leaves orphan objects in the payload bucket. - payload: payloadWithEnvelopeBytes(20_000), + payload: { prompt: 'x'.repeat(20_000) }, blueprintConfig: BLUEPRINT, }); @@ -1210,7 +996,7 @@ describe('LambdaMicrovmComputeStrategy image-identifier validation', () => { // ...and the remedy. await expect(start).rejects.toThrow(/--context compute_type=lambda-microvm/); expect(mockSend).not.toHaveBeenCalled(); - expect(mockS3Send).not.toHaveBeenCalled(); + expect(mockPrepare).not.toHaveBeenCalled(); }); test('accepts a full image ARN', async () => { @@ -1269,8 +1055,8 @@ describe('buildMicrovmPlatformConfig — the MicroVM substitute for a deploy-tim ]); expect(sharedConstants.microvm_platform_config.account_anchor_key).toBe('agent_session_role_arn'); - // Order is part of the contract: it is the serialization order, which the 4 KB - // inline/S3 branch decision is computed against. + // Keep the reviewed key inventory explicit. Payload/bootstrap hashing uses + // canonical JSON, so insertion order does not alter retry identity. expect([...MICROVM_PLATFORM_CONFIG_KEYS]).toEqual([ 'task_table_name', 'task_events_table_name', @@ -1462,18 +1248,6 @@ describe('buildMicrovmPlatformConfig — the MicroVM substitute for a deploy-tim const config = buildMicrovmPlatformConfig(); expect(config).toEqual(EXPECTED_PLATFORM_CONFIG); }); - - test('the full thirteen-key block fits the pointer envelope inside the 4 KB cap', () => { - // The one shape with no smaller fallback: if the pointer envelope itself - // exceeded 4 096 bytes there would be nothing left to move to S3. This asserts - // the design has real headroom rather than relying on the guard. - const pointerEnvelope = JSON.stringify({ - agent_payload_s3_uri: `s3://${PAYLOAD_BUCKET}/TASK001/payload.json`, - platform_config: buildMicrovmPlatformConfig(FULL_ENV), - }); - expect(Buffer.byteLength(pointerEnvelope, 'utf8')) - .toBeLessThan(MICROVM_RUN_HOOK_PAYLOAD_LIMIT_BYTES / 2); - }); }); describe('LambdaMicrovmComputeStrategy with the FULL platform_config environment', () => { @@ -1497,21 +1271,9 @@ describe('LambdaMicrovmComputeStrategy with the FULL platform_config environment for (const key of Object.keys(OPTIONAL_ENV)) delete process.env[key]; }); - test('delivers every configured identifier on the wire, in both envelope halves', async () => { - mockS3Send.mockResolvedValueOnce({}); - runMicrovmOk(); - - await new LambdaMicrovmComputeStrategy().startSession({ - taskId: 'TASK001', - userId: 'cognito-test', - payload: { repo_url: 'org/repo', hydrated_context: { blob: 'x'.repeat(10_000) } }, - blueprintConfig: BLUEPRINT, - }); - - const expected = { ...EXPECTED_PLATFORM_CONFIG, ...buildMicrovmPlatformConfig() }; - const envelope = JSON.parse(mockSend.mock.calls[0][0].input.runHookPayload); - expect(envelope.platform_config).toEqual(expected); - expect(Object.keys(envelope.platform_config)).toHaveLength(13); - expect(JSON.parse(mockS3Send.mock.calls[0][0].input.Body).platform_config).toEqual(expected); + test('passes every configured identifier to the authenticated manifest producer', async () => { + mockSend.mockResolvedValueOnce({ microvmId: MICROVM_ID, endpoint: ENDPOINT }); + await new LambdaMicrovmComputeStrategy().startSession({ taskId: 'TASK001', userId: 'u1', payload: {}, blueprintConfig: BLUEPRINT }); + expect(mockPrepare.mock.calls[0][0].platformConfig).toEqual(buildMicrovmPlatformConfig()); }); }); diff --git a/cdk/test/handlers/start-session-composition.test.ts b/cdk/test/handlers/start-session-composition.test.ts index 3f90a3d8a..f1ea4ba40 100644 --- a/cdk/test/handlers/start-session-composition.test.ts +++ b/cdk/test/handlers/start-session-composition.test.ts @@ -56,10 +56,17 @@ jest.mock('@aws-sdk/client-lambda-microvms', () => ({ }, })); -jest.mock('@aws-sdk/client-s3', () => ({ - S3Client: jest.fn(() => ({ send: jest.fn().mockResolvedValue({}) })), - PutObjectCommand: jest.fn((input: unknown) => ({ _type: 'PutObject', input })), - DeleteObjectCommand: jest.fn((input: unknown) => ({ _type: 'DeleteObject', input })), +// This suite checks strategy/metadata composition. Real bootstrap replay is +// covered by microvm-start-recovery and orchestrate-task-microvm. +jest.mock('../../src/handlers/shared/payload-bootstrap', () => ({ + ...jest.requireActual('../../src/handlers/shared/payload-bootstrap'), + preparePayloadReference: jest.fn(async ({ taskId }: { taskId: string }) => ({ + version: 2, + task_id: taskId, + bootstrap_s3_uri: 's3://bucket/bootstrap/example.json', + payload_url: 'https://signed.example/task', + expires_at: Date.now() + 900000, + })), })); jest.mock('../../src/handlers/shared/repo-config', () => ({ diff --git a/cdk/test/scripts/check-constants-sync.test.ts b/cdk/test/scripts/check-constants-sync.test.ts index 73f791ddd..4f2f88e17 100644 --- a/cdk/test/scripts/check-constants-sync.test.ts +++ b/cdk/test/scripts/check-constants-sync.test.ts @@ -62,6 +62,8 @@ const FIXTURE_FILES = [ 'agent/src/jira_reactions.py', 'agent/src/server.py', 'agent/src/config.py', + 'agent/src/payload_bootstrap.py', + 'cdk/src/handlers/shared/payload-bootstrap.ts', 'cdk/src/constructs/lambda-microvm-compute.ts', ]; @@ -126,6 +128,30 @@ describe('check-constants-sync', () => { // subprocess spawns, so give it room on a cold cache. jest.setTimeout(60_000); + describe('payload bootstrap contract', () => { + test.each([ + ['max_payload_bytes', 0], + ['minimum_url_lifetime_seconds', 901], + ['manifest_prefix', '../'], + ['launch_filename', 'payload.json'], + ])('rejects unsafe %s', (key, value) => { + const result = runInMutatedRepo(root => patchContract(root, json => { + json.payload_bootstrap[key] = value; + })); + expect(result.status).toBe(1); + expect(result.stderr).toContain('payload_bootstrap'); + }); + + test.each([ + ['agent/src/payload_bootstrap.py', 'CONTRACT = {}'], + ['cdk/src/handlers/shared/payload-bootstrap.ts', 'export const PAYLOAD_BOOTSTRAP = {};'], + ])('rejects a consumer with its own copy: %s', (file, source) => { + const result = runInMutatedRepo(root => write(root, file, source)); + expect(result.status).toBe(1); + expect(result.stderr).toContain('consumers must read the shared contract'); + }); + }); + describe('the real repository', () => { test('passes, and says what it actually checked', () => { const result = runInMutatedRepo(() => {}); diff --git a/cdk/test/stacks/agent.test.ts b/cdk/test/stacks/agent.test.ts index d48ccc02b..8156d64cb 100644 --- a/cdk/test/stacks/agent.test.ts +++ b/cdk/test/stacks/agent.test.ts @@ -1219,13 +1219,15 @@ describe('AgentStack with the Lambda MicroVMs substrate gate (--context compute_ const actions = payloadStatements.flatMap(s => Array.isArray(s.Action) ? s.Action : [s.Action]); expect(actions).toContain('s3:PutObject'); expect(actions).toContain('s3:DeleteObject'); - const deletion = payloadStatements.find(s => s.Action === 's3:DeleteObject'); - expect(deletion!.Resource).toEqual({ + expect(actions).toContain('s3:GetObject'); + expect(actions).toContain('s3:ListBucket'); + const deletion = payloadStatements.find(s => Array.isArray(s.Action) && s.Action.includes('s3:DeleteObject')); + expect(deletion!.Resource).toEqual(['payload.json', 'launch.json'].map(filename => ({ 'Fn::Join': ['', [ { 'Fn::GetAtt': [expect.stringMatching(/^LambdaMicrovmComputePayloadBucket/), 'Arn'] }, - '/*/payload.json', + `/*/${filename}`, ]], - }); + }))); }); test('cancel Lambda may terminate a MicroVM (and only terminate), image-scoped', () => { diff --git a/contracts/constants.json b/contracts/constants.json index ae677cb8e..e8e9df1ad 100644 --- a/contracts/constants.json +++ b/contracts/constants.json @@ -61,6 +61,15 @@ ], "account_anchor_key": "agent_session_role_arn" }, + "payload_bootstrap": { + "version": 2, + "manifest_prefix": "bootstrap/", + "launch_filename": "launch.json", + "max_manifest_bytes": 16384, + "max_payload_bytes": 8388608, + "url_ttl_seconds": 900, + "minimum_url_lifetime_seconds": 300 + }, "microvm_hook_budgets": { "ready_hook_timeout_seconds": 300, "warmup_total_budget_seconds": 240, diff --git a/contracts/constants.md b/contracts/constants.md index c9224ce3d..3bf857283 100644 --- a/contracts/constants.md +++ b/contracts/constants.md @@ -21,6 +21,8 @@ the contract. This is the neutral location both runtimes read. |---|---|---| | `agent/src/shared_constants.py` | `/app/contracts/constants.json` | import-time | | `agent/src/policy.py`, `agent/src/jira_reactions.py` | `SHARED_CONSTANTS` | import-time | +| `agent/src/payload_bootstrap.py` | `SHARED_CONSTANTS["payload_bootstrap"]` | import-time | +| `cdk/src/handlers/shared/payload-bootstrap.ts`, `cdk/src/constructs/payload-bootstrap-permissions.ts` | `payload_bootstrap` | import-time | | `agent/src/server.py` | `SHARED_CONSTANTS["microvm_platform_config"]`, `SHARED_CONSTANTS["microvm_hook_budgets"]` | import-time | | `cdk/src/handlers/shared/types.ts`, `jira-app-actor.ts` | `../../../../contracts/constants.json` | synth-time `import` | | `cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts` | `microvm_platform_config` | synth-time `import`, read per session start | @@ -77,6 +79,15 @@ JSON at TypeScript compile time via `resolveJsonModule`. "jira_oauth_secret_arn", "agent_session_role_arn"], "account_anchor_key": "agent_session_role_arn" }, + "payload_bootstrap": { + "version": 2, + "manifest_prefix": "bootstrap/", + "launch_filename": "launch.json", + "max_manifest_bytes": 16384, + "max_payload_bytes": 8388608, + "url_ttl_seconds": 900, + "minimum_url_lifetime_seconds": 300 + }, "microvm_hook_budgets": { "ready_hook_timeout_seconds": 300, "warmup_total_budget_seconds": 240, @@ -121,7 +132,7 @@ JSON at TypeScript compile time via `resolveJsonModule`. variable the agent installs it as (UPPER_SNAKE). This block is unlike the others — it is a **security allowlist**, not a tuning bound. The MicroVM image is a snapshot whose env is frozen at build time, so the agent's non-secret - platform env arrives in the `/run` hook payload instead; the values land in + platform env arrives through an authenticated v2 manifest and task payload instead; the values land in `os.environ`, which makes an unrecognised key an env-injection attempt. The consumer (`agent/src/server.py`) therefore **rejects** any `platform_config` carrying a key that is not in this map. Values are non-secret identifiers @@ -138,15 +149,15 @@ JSON at TypeScript compile time via `resolveJsonModule`. that disagrees with the anchor below on partition or account, is rejected (HTTP 400 `MICROVM_RUN_PLATFORM_CONFIG_INVALID`). This is **fail-fast plus defence in depth, not an ownership proof** — the account-scoped IAM grants are - what actually deny a foreign read, and an in-account redirect is deliberately - *not* covered (see `MICROVM_PLATFORM_CONFIG_ARN_KEYS` in - `agent/src/server.py` for the full statement of what this does and does not buy). + what actually deny a foreign read, and this ARN check alone does not cover an in-account redirect. The v2 bootstrap + authenticates a deployment manifest with IAM and requires the downloaded config + to match it; that separate check rejects same-account workspace substitutions. - **`microvm_platform_config.account_anchor_key`** — which `arn_keys` entry supplies the expected partition + account. Must be `agent_session_role_arn`-shaped: a payload key rather than `os.environ` or an `sts:GetCallerIdentity`, because the environment is empty by construction on this backend (the snapshot bakes nothing) - and this check runs on the path that must make zero AWS calls beyond the S3 - payload fetch. Two invariants follow and both are enforced: the anchor must be in + and the ARN consistency check itself makes no AWS calls. The v2 bootstrap + manifest read and signed task download precede it. Two invariants follow and both are enforced: the anchor must be in `arn_keys`, **and** it must be in `required` — an optional anchor would let a payload disarm the whole check by simply omitting it. @@ -159,6 +170,18 @@ duplication is deliberate: a malformed entry here would silently **widen** what agent accepts from a network payload, so neither side is trusted to be the only gate. +- **`payload_bootstrap`** — the shared ECS/MicroVM v2 transport. `version` is + required on references, manifests and task documents. `manifest_prefix` and + `launch_filename` define the authenticated settings directory and private retry + record. Byte limits bound manifest and task downloads; URL lifetime is at most + `url_ttl_seconds`, shortened by known signer credential expiry, with initial + creation requiring `minimum_url_lifetime_seconds`. The constants checker + rejects nonpositive/noninteger bounds, unsafe/colliding paths, a minimum above + its maximum, and consumers that stop reading the shared block. Changing the + protocol requires matching coordinator, worker images and policies; old unsigned + transports are rejected. See repository runbook + `docs/verification/645-payload-bootstrap.md` for rollout and live checks. + - **`microvm_hook_budgets.ready_hook_timeout_seconds`** — the `/ready` build-hook budget the CDK construct declares to `CreateMicrovmImage` (`READY_HOOK_TIMEOUT_SECONDS` in `cdk/src/constructs/lambda-microvm-compute.ts`). diff --git a/docs/decisions/ADR-021-lambda-microvms-compute-backend.md b/docs/decisions/ADR-021-lambda-microvms-compute-backend.md index 764a02ef6..b3f531124 100644 --- a/docs/decisions/ADR-021-lambda-microvms-compute-backend.md +++ b/docs/decisions/ADR-021-lambda-microvms-compute-backend.md @@ -197,25 +197,28 @@ The binary was fine — in the identical image, locally, `claude --version` answ Both halves of the fix are kept, because they answer different questions. `/ready` now **exec's the heavyweight binaries before returning 200** (`claude` required, `git`/`node` best-effort), which is the only mechanism that can make the shipped snapshot warm — and its own budget rises to 300 s, well inside the 3600 s build-hook window, because it now does work whose duration is a cold `exec`. Two structural rules keep that honest, because **per-command timeouts do not compose**: the required command runs FIRST with its own budget so no best-effort warm-up can starve the one that decides whether the snapshot is usable, and the best-effort ones then SHARE the remainder of a total warm-up ceiling that sits inside the hook budget with margin (240 s against 300 s). Without them, three commands at 120 s each would be 360 s — a fix for a runtime failure that produces a build failure instead — and a single hung optional command could hold up a 200 that the required warm-up had already earned. Separately, the version probe's timeout goes from 10 s to 60 s: a probe that exists to print a version string into a log line gains nothing from a tight bound and loses the whole task when it trips. The general rule this generalises to, and the reason it belongs in the ADR rather than only in a comment: **on this backend, a first-touch cost that other substrates pay during container start is deferred to the first task instead**, so anything large and lazily-loaded is a turn-0 hazard unless it is touched in `/ready`. -**Payload delivery** reuses the ECS strategy's S3-pointer pattern, adapted to `runHookPayload` (**≤ 4 KB** — measured, see below): payloads that fit ride inline; the rest are uploaded by the strategy to a platform payload bucket (the ECS payload bucket pattern in `ecs-agent-cluster.ts`: orchestrator write access, compute-role read-only scoped to the bucket, lifecycle expiry on objects) with the S3 URI in `runHookPayload` in place of the payload itself — the MicroVM **execution role** holds the read grant, exactly as the ECS task role does today. Since P2 the hook body also carries `platform_config` in both branches, so `runHookPayload` is never *only* the URI (see the canonical shapes below). +**Payload delivery — v2 amendment (2026-09-13, #817/#700).** The local implementation now shares an authenticated bootstrap protocol between ECS and MicroVM. This replaces the original inline/S3-pointer protocol and broad worker payload-bucket read grant. The historical P1/P2 runs below predate this amendment; AWS authorization, networking, expiry and coordinated deployment validation remain pending. The repository runbook is `docs/verification/645-payload-bootstrap.md`. -The cap is **4 096 bytes**, not the 16 384 the SDK documents. Measured exactly: 4 096 passes, 4 097 is rejected with *"Value at 'runHookPayload' failed to satisfy constraint: Member must have length less than or equal to 4096"*. Two consequences follow. First, the original threshold would have inlined every envelope between 4 097 and 16 384 bytes and had the service reject all of them. Second, and more structurally: **the S3-pointer path is now the dominant one, and inline is the exception.** A hydrated task payload (prompt + issue thread + repo context) essentially always exceeds 4 KB, so "small payloads ride inline" describes tiny repo-less prompts rather than the common case. The payload bucket is therefore not a rarely-exercised overflow valve but a required part of every normal task, which raises its lifecycle rule (`MICROVM_PAYLOAD_TTL_DAYS`) and the execution role's read grant from edge-case plumbing to load-bearing. +Every task uses S3. The coordinator publishes a non-secret deployment manifest at `bootstrap/.json`, conditionally creates `/payload.json`, and sends a short-lived signed URL for that one task object. The worker reads the manifest with its own AWS role, whose policy explicitly denies object reads outside its deployment's `bootstrap/*` and denies payload-bucket listing. The explicit deny also prevents a foreign public bucket from authenticating a forged manifest. The task document must name the reference's task ID and contain the exact manifest configuration before the MicroVM installs any of it. -**Canonical wire shapes.** Three, and the producer (`lambda-microvm-strategy.ts`) emits exactly these: +The `runHookPayload` cap remains **4,096 bytes**, measured against the service despite the SDK's documented 16,384. It now applies to the serialized reference rather than choosing an inline branch. Payloads are capped at 8 MiB and manifests at 16 KiB; the bounds and version are shared in `contracts/constants.json`. -| Where | Exact shape | +| Location | v2 shape | |---|---| -| `runHookPayload`, inline branch | `{"agent_payload": {…}, "platform_config": {…}}` | -| `runHookPayload`, pointer branch | `{"agent_payload_s3_uri": "s3://…", "platform_config": {…}}` | -| the object at that S3 URI | `{…agent_payload fields…, "platform_config": {…}}` — the payload's own fields at the TOP level, with the config merged in beside them | +| `runHookPayload` string / ECS `AGENT_PAYLOAD_REF` | `{version:2, task_id, bootstrap_s3_uri, payload_url, expires_at}` | +| `bootstrap/.json` | `{version:2, backend, platform_config}` | +| `/payload.json` | `{version:2, task_id, agent_payload, platform_config}` | +| Private `/launch.json` | `{fingerprint, reference}`; coordinator-readable only, never on the agent-readable task row | -Two asymmetries are deliberate and must not be "tidied" without changing both sides. First, **the S3 object is not the envelope**: the payload's fields sit at the top level (that is what P1 uploaded, before `platform_config` existed) rather than nested under an `agent_payload` key. Second, **`platform_config` is duplicated** on the pointer path — once beside the pointer, once inside the uploaded object. It costs a few hundred bytes and buys the property that the config is reachable whichever end of the fetch a reader looks at, which matters because it is the agent's only substitute for an env block. +The coordinator saves the exact reference for replay: re-signing would change a Run request with the same client token. Task objects use conditional creation and read-back recovery after a lost committed write. URL lifetime is at most 900 seconds, bounded by known credential expiry, and initial creation requires at least 300 seconds. Expired saved references fail without re-signing. The coordinator needs bucket-scoped `ListBucket` so an absent launch record returns `NoSuchKey` rather than `AccessDenied`. It deletes both task objects at finalization; one-day lifecycle expiry is the backstop. Manifests are refreshed with identical bytes at preparation and may coexist across configuration revisions until lifecycle cleanup. -The agent's reader is deliberately more permissive than this contract: it also accepts an S3 object shaped like the envelope (`agent_payload` nested), and `platform_config` present in only one of the two places (the fetched object wins, the hook body is the fallback). Those are **defensive compatibility** for the independent deploy cadences of a snapshot image and the orchestrator Lambda — a tolerant reader, not an alternative contract. A producer must emit the three shapes above. +The image accepts only v2. The old inline/pointer forms and baked-environment fallback are rejected. A coordinated drain and switch of coordinator, images and application IAM is required; there is no new KMS/SSM resource or bootstrap bundle version. Signed URLs are bearer capabilities and must never appear in task rows, ordinary logs or repository subprocess environments. ECS consumes/removes its reference before importing the task pipeline. The URL consumer restricts requests to the exact task object on regional S3 HTTPS, disables redirects/proxies and bounds response bytes. AWS verifies signatures and actual expiration. -**Platform configuration delivery (P2): payload-sourced, allowlisted, fail-closed.** The other two backends hand the agent its non-secret platform env at launch — AgentCore Runtime env vars, ECS container overrides — and there is no equivalent on this substrate: a MicroVM starts from a **snapshot**, so its process environment is whatever was frozen at *image build* time and is then replayed by every MicroVM launched from that image version. Baking the deployment's identifiers into the snapshot would make them **version-frozen**: a redeploy that renames a table, adds a bucket or rotates the session role would leave every existing image version describing a deployment that no longer exists, and the drift would surface as a task-time `ResourceNotFound` rather than a deploy-time error. So the values travel with the task instead: `platform_config` is a SIBLING of the payload — beside `agent_payload` in the inline branch, beside `agent_payload_s3_uri` in the pointer branch, and merged in beside the payload's own fields inside the S3 object (the canonical shapes above give each one exactly) — whose snake_case keys the agent installs into `os.environ` as their UPPER_SNAKE equivalents. A payload value therefore **wins** over any pre-existing/image value — the orchestrator is describing the live deployment, the snapshot is describing a past one. `platform_config` carries **non-secret identifiers only** (table and bucket names, secret ARNs, the session-role ARN); secrets are still fetched at `/run` time from Secrets Manager using those ARNs, so the snapshot-must-stay-secret-free requirement above is untouched. Per-task fields — `memory_id` and friends — stay inside `agent_payload`: `platform_config` configures the *process*, `agent_payload` describes the *task*. +**Platform configuration delivery (P2, strengthened by v2).** A MicroVM restores a snapshot, including its process environment, so current deployment identifiers must arrive at task launch. The authenticated manifest and downloaded document both contain `platform_config`: non-secret table/bucket names, secret ARNs and session-role ARN. Per-task fields such as `memory_id` remain in `agent_payload`. ECS's manifest config is empty because ECS already receives deployment settings through its task definition and coordinator overrides. -Two rules make it safe. First, **the allowlist fails closed**: the agent installs a fixed set of keys and *rejects the entire run* (HTTP 400, nothing spawned, not one key installed) if the block carries anything else. These values become environment variables of the process that spawns the agent's tool subprocesses, so an unrecognised key is an attempt to set an arbitrary variable in the agent (`AWS_ENDPOINT_URL`, `LD_PRELOAD`, `PATH`, …) — an injection attempt, not a forward-compatibility gap, which is why unknown keys are refused rather than filtered out. Second, **installation happens before any credential or pipeline initialisation** on the hook path: the very next step reads `GITHUB_TOKEN_SECRET_ARN` to resolve the GitHub token and `AGENT_SESSION_ROLE_ARN` to scope the task's credentials, so installing later would silently resolve the whole task against the snapshot's frozen env. The one call that must precede installation is the S3 payload fetch (the config is *inside* the fetched object), which therefore runs on the ambient compute role via the attributed platform client — and it is the ONLY one: the same rule covers **logging**, so every `/run` log line before the install is stdout-only. The CloudWatch writer would otherwise resolve credentials and cache credentials in a boto3 default session (environment-derived region is re-read for each new client) off whatever a snapshot happened to bake, which is the build-hook defect one phase later. Nothing is lost — in the intended deployment there is no baked `LOG_GROUP_NAME`, so those lines would have gone to stdout anyway, and the reason for every pre-install rejection also travels in the structured 4xx/5xx body the service surfaces. A **required subset** (task table, task-events table, GitHub token secret ARN, session-role ARN) is rejected as `…_INCOMPLETE` when missing or blank — a distinct wire code from the `…_INVALID` allowlist rejection, because the remedies differ (deployment wiring vs. producer bug). A `/run` envelope with *no* `platform_config` is accepted with a warning only when the effective environment already provides every required identifier; otherwise it is rejected as incomplete. The image and orchestrator can deploy on independent cadences without accepting an unusable environment. The key set is a cross-package contract in `contracts/constants.json` (`microvm_platform_config`), consumed by the agent's `/run` hook and produced by the orchestrator, with shape and required-subset invariants enforced by `scripts/check-constants-sync.ts`. +MicroVM installs only allowlisted keys and rejects unknown keys before installing any. Required identifiers must be nonblank; malformed/control-character values and inconsistent ARN partitions/accounts are rejected. Legitimate cross-region secrets remain allowed when present in the trusted manifest. Account agreement alone cannot authenticate a sibling ARN; v2's manifest/config comparison supplies the provenance check. No-config envelopes are rejected regardless of the image's existing environment. + +Manifest read and signed payload download precede configuration installation. Secrets, task-scoped credential setup, CloudWatch initialization and the pipeline follow installation; pre-install diagnostics use stdout without unsafe exception chains. Build hooks remain AWS-silent. The settings allowlist and bootstrap bounds are cross-language contracts with drift checks. These source guarantees do not establish complete hostile-worker isolation: roles still retain other platform grants and choose session tags, and a stolen signed URL remains usable until expiry or revocation. **No orchestrator→agent HTTP path exists in P1–P3**: payload arrives through the `/run` hook, all agent work is outbound, and therefore **no JWE auth tokens are minted at all** — token minting (and its ≤ 60 min TTL refresh problem) is deferred until a real consumer exists (e.g. operator shell access, [#391](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/391)). The `endpoint` stays in the `SessionHandle` because it is genuinely per-session state that becomes load-bearing the day such a consumer appears. But note the service does not agree by default: omitting `ingressNetworkConnectors` on `RunMicrovm` attaches a **public** `HTTP_INGRESS` connector, so the strategy passes the Lambda-managed `NO_INGRESS` connector explicitly on every launch (see sub-decision 4's security table). @@ -224,12 +227,12 @@ Two rules make it safe. First, **the allowlist fails closed**: the agent install Normative requirements (EARS). **Each requirement's own `(Pn)` tag is authoritative**; there is no blanket phase for the list. The tags are per-requirement because the original list *was* split P1/P2 on the assumption that a hook could be declared in one phase and served in a later one — which the service does not permit (see the phasing table above), so the hook-serving requirements collapsed into P1 while the P2 items below arrived with the P2 hooks and `platform_config`: - (P1) The image build shall not embed secrets, tokens, or per-task identity in the snapshot. -- (P1) If the task payload exceeds the 4 KB `runHookPayload` limit, then the strategy shall upload the payload to the platform payload bucket and pass its S3 URI in `runHookPayload` in place of the payload (from P2, alongside `platform_config`). -- (P1) The MicroVM execution role shall hold read-only access to the payload bucket, scoped to that bucket. +- (P1, amended by v2) Every task shall use the authenticated manifest and single-object signed payload reference; the serialized hook reference shall not exceed 4,096 bytes. +- (P1, amended by v2) The worker role shall read only deployment bootstrap manifests with ambient credentials; other object reads and payload-bucket listing shall be explicitly denied. The coordinator shall own task-object writes, signing, replay and deletion. - (P1) Where no ingress is configured for a deployment, the strategy shall pass the Lambda-managed `NO_INGRESS` network connector on every `RunMicrovm` call (the field shall not be omitted). - (P1) Where the image enables any MicroVM lifecycle hook, the image shall also enable the `/ready` hook and the agent shall serve it. - (P1) When the `/run` hook receives the task payload, the agent shall validate it, start the pipeline asynchronously, and return HTTP 200 within the hook budget. -- (P1) If the `/run` hook payload cannot be resolved to a task payload, then the agent shall reject the hook with a client error and shall not start a pipeline. +- (P1, clarified) Invalid hook references shall return a structured client error; unreadable manifest/payload bytes shall return a structured server error. Neither path shall install configuration or start a pipeline. - (P1) The agent shall not execute the clone→verify→PR pipeline on the hook path. - (P1) The agent shall resolve credentials at `/run` time. - (P1) Where a deployment configures a MicroVM image before smoke parity is verified, the platform shall warn that the backend has no smoke-parity guarantee. @@ -394,8 +397,8 @@ Lambda MicroVMs launched in 5 regions (us-east-1/2, us-west-2, eu-west-1, ap-nor **P1 (start/poll/stop — no suspend):** -- Unit tests for the strategy: start/poll/stop mapping (including `SessionStatus` `'suspended'` reported mechanically, without task-state interpretation), payload-size branching (inline vs S3 pointer, at the exact 4 096/4 097-byte boundary), the image-identifier-must-be-an-ARN guard, the explicit `NO_INGRESS` argument (including the blank-env-var fallback, which must never omit the field), error classification (`ServiceQuotaExceededException`, `ThrottlingException`, `ResourceNotFoundException`, regional-unavailability), the omit-`idlePolicy` invariant, and `maximumDurationInSeconds` fixed at 28 800. -- Agent tests: `/ready` returns 200 after required local warm-up succeeds, without starting a task or calling AWS; `/run` accepts both envelope shapes (inline and S3 pointer), starts the pipeline asynchronously through the same mapper `/invocations` uses, returns before the pipeline finishes, and rejects every unusable envelope with a named code before spawning; `/suspend` and `/resume` are NOT served (`/validate` and `/terminate` joined the served set in P2). +- Unit tests for the strategy: start/poll/stop mapping (including `SessionStatus` `'suspended'` reported mechanically, without task-state interpretation), v2 reference-size bounds (4,096 bytes), immutable signed-reference replay and scoped payload transport, the image-identifier-must-be-an-ARN guard, the explicit `NO_INGRESS` argument (including the blank-env-var fallback, which must never omit the field), error classification (`ServiceQuotaExceededException`, `ThrottlingException`, `ResourceNotFoundException`, regional-unavailability), the omit-`idlePolicy` invariant, and `maximumDurationInSeconds` fixed at 28 800. +- Agent tests: `/ready` returns 200 after required local warm-up succeeds, without starting a task or calling AWS; `/run` accepts authenticated v2 references and rejects old unsigned envelopes, starts the pipeline asynchronously through the same mapper `/invocations` uses, returns before the pipeline finishes, and rejects every unusable envelope with a named code before spawning; `/suspend` and `/resume` are NOT served (`/validate` and `/terminate` joined the served set in P2). - Orchestrator tests: substrate-terminal + non-terminal task status → failed classification; `suspended` + non-`AWAITING_APPROVAL` status → anomaly event, no fail-fast; `compute_metadata` persisted with `microvmId`/`endpoint` after `startSession`. - CDK assertions: MicroVM resources present only when `ComputeTypes` includes the backend; synth failure for unsupported regions (plus the context-flag escape hatch); memory size validated against the accepted list at synth; the connector operator role and its trust; two connectors with the build-only one carrying port 80 and the runtime one not; the `/ready` + `/run` hook declaration and the absence of the others; IAM actions scoped as specified (orchestrator lifecycle set; no `CreateMicrovmAuthToken` anywhere); payload-bucket grants (execution role read-only); backend cost-allocation tags; types-sync check covers the widened `ComputeType`. - CLI tests: onboarding rejection with remedy when the availability probe fails; doctor check present when a blueprint selects the backend. @@ -404,7 +407,7 @@ Lambda MicroVMs launched in 5 regions (us-east-1/2, us-west-2, eu-west-1, ap-nor **P2 (smoke parity):** - Agent tests: `/validate` returns 200 with its individual check results, 503 while initialising, reports a missing hook route / unsupported interpreter, starts nothing, and makes **zero AWS calls even with `LOG_GROUP_NAME` set** (asserted by poisoning the boto3 and CloudWatch-writer seams — the same assertion covers `/ready`); `/terminate` returns 200 with no body at all, with a malformed / non-object / wrong-content-type / whitespace-only body, when the body read itself fails, with a pipeline still running (without joining it), and when its own best-effort step raises — and never calls `task_state.write_terminal`; a structural assertion that the route carries no typed body param keeps the 422 from being reintroduced. -- `platform_config` tests: the allowlist and required subset are read from `contracts/constants.json` (the wire key set is additionally asserted literally, as the agent-side tripwire on a contract edit); an unknown key rejects the whole block with nothing installed; a non-object block and a non-string value are rejected; a blank/`null` optional value is skipped without clobbering an image value while a blank required value is rejected; a payload value beats a pre-existing env value; installation is observed to happen before the GitHub-token resolver runs and before any pipeline thread exists; the config is picked up from the inline envelope, from beside the S3 pointer, and from inside the fetched object (inner wins); an envelope with no `platform_config` is accepted with a warning only if the effective environment already supplies every required identifier; otherwise it is rejected. +- `platform_config` tests: the allowlist and required subset are read from `contracts/constants.json` (the wire key set is additionally asserted literally, as the agent-side tripwire on a contract edit); an unknown key rejects the whole block with nothing installed; a non-object block and a non-string value are rejected; a blank/`null` optional value is skipped without clobbering an image value while a blank required value is rejected; a payload value beats a pre-existing env value; installation is observed to happen before the GitHub-token resolver runs and before any pipeline thread exists; the downloaded config must equal the IAM-authenticated manifest, including same-account workspace identifiers; missing config and legacy envelopes are rejected regardless of baked environment values. Transport tests cover signed URLs, exact task identity, bad bytes and rejection before installation. - Snapshot credential hygiene: a subprocess probe asserts that importing `server` and serving `/ready` + `/validate` imports neither `boto3` nor `botocore`, caches no `aws_session` session, and spawns no CloudWatch writer thread — the property that keeps a build-role credential chain and the build-time region out of the snapshot. - `/run` pre-install silence: with a **baked `LOG_GROUP_NAME`** (the hostile case — without it the assertions pass vacuously) every AWS/credential seam (`boto3.client`/`Session`, the `aws_session` factories, `_debug_cw`/`_warn_cw`) is armed to raise until the install succeeds. Asserted on the accepted path, on all three rejection paths (bad envelope, `platform_config` invalid, `platform_config` incomplete) and on the failed-fetch 500 — where the seams stay armed for the whole request, because a rejected run installed nothing and so earns no AWS call. The permitted exception is asserted POSITIVELY: exactly one client is built pre-install, for `s3`, through the attributed factory. - `/ready` warm-up tests (P2-F5): the hook exec's each configured binary exactly once with a generous timeout; `claude` is the only REQUIRED entry; a timeout, a missing binary, a non-zero exit and an unexpected `OSError` each produce **503 with the reason logged to stdout** rather than a 200 or a 500; a best-effort failure still reports ready; the warm-up makes zero AWS calls with `LOG_GROUP_NAME` baked. Plus the backstop half: the `claude --version` probe's bound is asserted to be ≥ 60 s and to be applied to the *exec* rather than to the PATH lookup, and a missing CLI warns instead of raising. diff --git a/docs/design/COMPUTE.md b/docs/design/COMPUTE.md index 0ffc9921c..45296944a 100644 --- a/docs/design/COMPUTE.md +++ b/docs/design/COMPUTE.md @@ -79,7 +79,7 @@ See [ORCHESTRATOR.md](./ORCHESTRATOR.md) for how the orchestrator handles these Lambda MicroVMs are an opt-in third backend, selected per repository with `compute_type: lambda-microvm`; AgentCore remains the default. Image configuration has three states: a managed base-image ARN and version creates the snapshot image in CDK; an external image identifier uses a snapshot built out of band; and supplying neither provisions only the roles, buckets, and connectors needed for the bootstrap deploy. `cdk/scripts/package-microvm-artifact.sh` packages the agent as zip + Dockerfile, uploads it to the artifact bucket, and can create the external image. Lambda MicroVMs are available in five launch regions (us-east-1, us-east-2, us-west-2, eu-west-1, ap-northeast-1) and will expand; the platform enforces regional availability in layers via a synth-time constant, onboarding live probes, and orchestration-time classification. -Because a snapshot freezes its build-time environment, deployment-specific, non-secret identifiers travel in the `/run` hook's `platform_config` block instead. The strategy sends the canonical inline envelope or, when that envelope exceeds the verified 4,096-byte `runHookPayload` limit, an S3-pointer envelope with the configuration also merged into the uploaded payload. The agent accepts only allowlisted keys and installs them before pipeline initialization; [ADR-021 §3](../decisions/ADR-021-lambda-microvms-compute-backend.md#3-packaging-same-agent-image-source-new-build-path) defines the exact wire shapes and validation rules. +Because a snapshot freezes its build-time environment, current deployment identifiers arrive through the v2 payload bootstrap. The coordinator publishes a non-secret deployment manifest and sends a single-object signed download URL for the task. The worker reads only its deployment's `bootstrap/*` with ambient credentials; other object reads and payload-bucket listing are explicitly denied. The downloaded task identity and configuration must match the authenticated manifest before configuration installation. The serialized reference fits the verified 4,096-byte hook limit; all payload sizes use S3. ECS shares this transport through `AGENT_PAYLOAD_REF`. [ADR-021 §3](../decisions/ADR-021-lambda-microvms-compute-backend.md#3-packaging-same-agent-image-source-new-build-path) defines the wire format, compatibility change and pending live gates. Networking separates image build from execution: the build-only connector permits TCP 80 and 443 because the Dockerfile uses `apt-get`, while running MicroVMs retain 443-only egress through the platform VPC. Every launch explicitly passes the Lambda-managed `NO_INGRESS` connector; omission would select the service's public-ingress default. The P2 image declares and serves `/ready` and `/validate` at build time and `/run` and `/terminate` at runtime. `/suspend` and `/resume` remain disabled until their P3 implementation. diff --git a/docs/design/SECURITY.md b/docs/design/SECURITY.md index 5e0feeed3..091e34fac 100644 --- a/docs/design/SECURITY.md +++ b/docs/design/SECURITY.md @@ -46,7 +46,10 @@ Three authentication mechanisms protect the platform, matching its input channel The compute role retains only non-tenant access (Bedrock model invocation — already ARN-scoped; CloudWatch Logs; the GitHub PAT secret, read once before the SessionRole is assumed; AgentCore Memory) plus `sts:AssumeRole`/`sts:TagSession` on the SessionRole. Because the agent runs under credentials that are themselves an assumed role, its `AssumeRole` is *role chaining* — capped at one hour regardless of the role's max session duration — so the agent uses a **refreshable** credential provider that re-assumes before expiry (tasks can run up to the 8-hour `maxLifetime`). The design is backend-agnostic: the same SessionRole and agent code serve all three backends (AgentCore Runtime, ECS Fargate, Lambda MicroVMs). Existing scoped credentials are constrained to their tags, but the compute role chooses those tags when assuming the role. The trust policy does not independently bind a worker to one task; this is not complete isolation against a compromised worker with ambient credentials. No worker needs direct access to the shared capacity counter. ECS's legacy configuration without a session role retains unscoped task reads/reporting updates, with the same attribute restriction. The policy structure and Python writer compatibility are tested locally. A historical IAM simulator check covered the original tag conditions; it does not verify the new attribute restriction or deployed transaction authorization. The repository runbook `docs/verification/645-coordinator-metadata.md` contains the required AWS authorization checks. -**Lambda MicroVMs compute-role delta** - On this backend the compute role additionally holds prefix-scoped `secretsmanager:GetSecretValue` on the per-workspace channel-OAuth secrets (`bgagent-linear-oauth-*`, `bgagent-jira-oauth-*`), read access to the `/run` payload bucket (the task payload arrives as an S3 object, not as environment), and `ec2:DescribeAvailabilityZones` (`Resource: *`, read-only — EC2 describe actions have no resource-level scoping) for a CDK repo's own synth gate. It is also **the only compute role in the platform whose trust policy carries no confused-deputy condition**: the Lambda MicroVMs service populates no `aws:SourceAccount` or `aws:SourceArn` when it assumes the role, so a trust policy carrying one is unassumable — verified live, not assumed (connector creation failed deterministically, and `RunMicrovm` surfaced the same root cause as a misleading caller-side `iam:PassRole` denial). The compensating controls are that each of the three MicroVM roles can be passed to `lambda.amazonaws.com` only by a named principal — the orchestrator, via an `iam:PassRole` scoped to the execution role's exact ARN, and the CloudFormation deployment role for the build and connector-operator roles — that every other resource they reach is account-scoped by ARN apart from two justified `Resource: *` read/create-time statements, and that none of them holds `iam:*`, cross-account trust, or any `sts:AssumeRole` beyond the execution role's scoped hop to the per-task SessionRole. Full evidence, the two-arm PassRole experiment, and the alternatives considered are in [ADR-021 §4](../decisions/ADR-021-lambda-microvms-compute-backend.md#4-infra-and-iam-conditional-resources-behind-bootstrap-computetypes). +**Lambda MicroVMs compute-role delta** - On this backend the compute role additionally holds prefix-scoped `secretsmanager:GetSecretValue` on the per-workspace channel-OAuth secrets (`bgagent-linear-oauth-*`, `bgagent-jira-oauth-*`), read access only to its payload bucket's `bootstrap/*` manifests (other object reads and payload-bucket listing are explicitly denied), and `ec2:DescribeAvailabilityZones` (`Resource: *`, read-only — EC2 describe actions have no resource-level scoping) for a CDK repo's own synth gate. It is also **the only compute role in the platform whose trust policy carries no confused-deputy condition**: the Lambda MicroVMs service populates no `aws:SourceAccount` or `aws:SourceArn` when it assumes the role, so a trust policy carrying one is unassumable — verified live, not assumed (connector creation failed deterministically, and `RunMicrovm` surfaced the same root cause as a misleading caller-side `iam:PassRole` denial). The compensating controls are that each of the three MicroVM roles can be passed to `lambda.amazonaws.com` only by a named principal — the orchestrator, via an `iam:PassRole` scoped to the execution role's exact ARN, and the CloudFormation deployment role for the build and connector-operator roles — that every other resource they reach is account-scoped by ARN apart from two justified `Resource: *` read/create-time statements, and that none of them holds `iam:*`, cross-account trust, or any `sts:AssumeRole` beyond the execution role's scoped hop to the per-task SessionRole. Full evidence, the two-arm PassRole experiment, and the alternatives considered are in [ADR-021 §4](../decisions/ADR-021-lambda-microvms-compute-backend.md#4-infra-and-iam-conditional-resources-behind-bootstrap-computetypes). + +**ECS/MicroVM task bootstrap (v2)** — The coordinator publishes non-secret deployment manifests and supplies a short-lived signed URL for exactly one task's payload. The worker reads the manifest with its ambient role, whose explicit deny outside its own `bootstrap/*` prevents a foreign public bucket from authorizing fake configuration. It verifies downloaded task identity and exact manifest/config agreement before installing MicroVM settings. The coordinator privately persists the URL outside TaskTable for retry and deletes both task objects at finalization. Worker roles cannot read/list task objects using their ambient credentials. A signed URL is a bearer capability: redact it in errors and keep it out of task rows and repository subprocess environments. Existing session-tag trust and other platform grants remain separate limitations. This protocol is implemented locally; real IAM/network/expiry gates and the coordinated image/coordinator/policy rollout are documented in the repository's `docs/verification/645-payload-bootstrap.md`. + > Out of scope for this control and tracked separately as GitHub issues: replacing the shared GitHub PAT (GitHub App / Token Vault), binding credentials to the MicroVM via attestation, and scoping AgentCore Memory (namespace isolation by `actorId`/`sessionId` remains its boundary). diff --git a/docs/src/content/docs/architecture/Compute.md b/docs/src/content/docs/architecture/Compute.md index a7ae76083..00a81d9c4 100644 --- a/docs/src/content/docs/architecture/Compute.md +++ b/docs/src/content/docs/architecture/Compute.md @@ -83,7 +83,7 @@ See [ORCHESTRATOR.md](/sample-autonomous-cloud-coding-agents/architecture/orches Lambda MicroVMs are an opt-in third backend, selected per repository with `compute_type: lambda-microvm`; AgentCore remains the default. Image configuration has three states: a managed base-image ARN and version creates the snapshot image in CDK; an external image identifier uses a snapshot built out of band; and supplying neither provisions only the roles, buckets, and connectors needed for the bootstrap deploy. `cdk/scripts/package-microvm-artifact.sh` packages the agent as zip + Dockerfile, uploads it to the artifact bucket, and can create the external image. Lambda MicroVMs are available in five launch regions (us-east-1, us-east-2, us-west-2, eu-west-1, ap-northeast-1) and will expand; the platform enforces regional availability in layers via a synth-time constant, onboarding live probes, and orchestration-time classification. -Because a snapshot freezes its build-time environment, deployment-specific, non-secret identifiers travel in the `/run` hook's `platform_config` block instead. The strategy sends the canonical inline envelope or, when that envelope exceeds the verified 4,096-byte `runHookPayload` limit, an S3-pointer envelope with the configuration also merged into the uploaded payload. The agent accepts only allowlisted keys and installs them before pipeline initialization; [ADR-021 §3](/sample-autonomous-cloud-coding-agents/architecture/adr-021-lambda-microvms-compute-backend#3-packaging-same-agent-image-source-new-build-path) defines the exact wire shapes and validation rules. +Because a snapshot freezes its build-time environment, current deployment identifiers arrive through the v2 payload bootstrap. The coordinator publishes a non-secret deployment manifest and sends a single-object signed download URL for the task. The worker reads only its deployment's `bootstrap/*` with ambient credentials; other object reads and payload-bucket listing are explicitly denied. The downloaded task identity and configuration must match the authenticated manifest before configuration installation. The serialized reference fits the verified 4,096-byte hook limit; all payload sizes use S3. ECS shares this transport through `AGENT_PAYLOAD_REF`. [ADR-021 §3](/sample-autonomous-cloud-coding-agents/architecture/adr-021-lambda-microvms-compute-backend#3-packaging-same-agent-image-source-new-build-path) defines the wire format, compatibility change and pending live gates. Networking separates image build from execution: the build-only connector permits TCP 80 and 443 because the Dockerfile uses `apt-get`, while running MicroVMs retain 443-only egress through the platform VPC. Every launch explicitly passes the Lambda-managed `NO_INGRESS` connector; omission would select the service's public-ingress default. The P2 image declares and serves `/ready` and `/validate` at build time and `/run` and `/terminate` at runtime. `/suspend` and `/resume` remain disabled until their P3 implementation. diff --git a/docs/src/content/docs/architecture/Security.md b/docs/src/content/docs/architecture/Security.md index dd7b95df0..a392f8b8c 100644 --- a/docs/src/content/docs/architecture/Security.md +++ b/docs/src/content/docs/architecture/Security.md @@ -50,7 +50,10 @@ Three authentication mechanisms protect the platform, matching its input channel The compute role retains only non-tenant access (Bedrock model invocation — already ARN-scoped; CloudWatch Logs; the GitHub PAT secret, read once before the SessionRole is assumed; AgentCore Memory) plus `sts:AssumeRole`/`sts:TagSession` on the SessionRole. Because the agent runs under credentials that are themselves an assumed role, its `AssumeRole` is *role chaining* — capped at one hour regardless of the role's max session duration — so the agent uses a **refreshable** credential provider that re-assumes before expiry (tasks can run up to the 8-hour `maxLifetime`). The design is backend-agnostic: the same SessionRole and agent code serve all three backends (AgentCore Runtime, ECS Fargate, Lambda MicroVMs). Existing scoped credentials are constrained to their tags, but the compute role chooses those tags when assuming the role. The trust policy does not independently bind a worker to one task; this is not complete isolation against a compromised worker with ambient credentials. No worker needs direct access to the shared capacity counter. ECS's legacy configuration without a session role retains unscoped task reads/reporting updates, with the same attribute restriction. The policy structure and Python writer compatibility are tested locally. A historical IAM simulator check covered the original tag conditions; it does not verify the new attribute restriction or deployed transaction authorization. The repository runbook `docs/verification/645-coordinator-metadata.md` contains the required AWS authorization checks. -**Lambda MicroVMs compute-role delta** - On this backend the compute role additionally holds prefix-scoped `secretsmanager:GetSecretValue` on the per-workspace channel-OAuth secrets (`bgagent-linear-oauth-*`, `bgagent-jira-oauth-*`), read access to the `/run` payload bucket (the task payload arrives as an S3 object, not as environment), and `ec2:DescribeAvailabilityZones` (`Resource: *`, read-only — EC2 describe actions have no resource-level scoping) for a CDK repo's own synth gate. It is also **the only compute role in the platform whose trust policy carries no confused-deputy condition**: the Lambda MicroVMs service populates no `aws:SourceAccount` or `aws:SourceArn` when it assumes the role, so a trust policy carrying one is unassumable — verified live, not assumed (connector creation failed deterministically, and `RunMicrovm` surfaced the same root cause as a misleading caller-side `iam:PassRole` denial). The compensating controls are that each of the three MicroVM roles can be passed to `lambda.amazonaws.com` only by a named principal — the orchestrator, via an `iam:PassRole` scoped to the execution role's exact ARN, and the CloudFormation deployment role for the build and connector-operator roles — that every other resource they reach is account-scoped by ARN apart from two justified `Resource: *` read/create-time statements, and that none of them holds `iam:*`, cross-account trust, or any `sts:AssumeRole` beyond the execution role's scoped hop to the per-task SessionRole. Full evidence, the two-arm PassRole experiment, and the alternatives considered are in [ADR-021 §4](/sample-autonomous-cloud-coding-agents/architecture/adr-021-lambda-microvms-compute-backend#4-infra-and-iam-conditional-resources-behind-bootstrap-computetypes). +**Lambda MicroVMs compute-role delta** - On this backend the compute role additionally holds prefix-scoped `secretsmanager:GetSecretValue` on the per-workspace channel-OAuth secrets (`bgagent-linear-oauth-*`, `bgagent-jira-oauth-*`), read access only to its payload bucket's `bootstrap/*` manifests (other object reads and payload-bucket listing are explicitly denied), and `ec2:DescribeAvailabilityZones` (`Resource: *`, read-only — EC2 describe actions have no resource-level scoping) for a CDK repo's own synth gate. It is also **the only compute role in the platform whose trust policy carries no confused-deputy condition**: the Lambda MicroVMs service populates no `aws:SourceAccount` or `aws:SourceArn` when it assumes the role, so a trust policy carrying one is unassumable — verified live, not assumed (connector creation failed deterministically, and `RunMicrovm` surfaced the same root cause as a misleading caller-side `iam:PassRole` denial). The compensating controls are that each of the three MicroVM roles can be passed to `lambda.amazonaws.com` only by a named principal — the orchestrator, via an `iam:PassRole` scoped to the execution role's exact ARN, and the CloudFormation deployment role for the build and connector-operator roles — that every other resource they reach is account-scoped by ARN apart from two justified `Resource: *` read/create-time statements, and that none of them holds `iam:*`, cross-account trust, or any `sts:AssumeRole` beyond the execution role's scoped hop to the per-task SessionRole. Full evidence, the two-arm PassRole experiment, and the alternatives considered are in [ADR-021 §4](/sample-autonomous-cloud-coding-agents/architecture/adr-021-lambda-microvms-compute-backend#4-infra-and-iam-conditional-resources-behind-bootstrap-computetypes). + +**ECS/MicroVM task bootstrap (v2)** — The coordinator publishes non-secret deployment manifests and supplies a short-lived signed URL for exactly one task's payload. The worker reads the manifest with its ambient role, whose explicit deny outside its own `bootstrap/*` prevents a foreign public bucket from authorizing fake configuration. It verifies downloaded task identity and exact manifest/config agreement before installing MicroVM settings. The coordinator privately persists the URL outside TaskTable for retry and deletes both task objects at finalization. Worker roles cannot read/list task objects using their ambient credentials. A signed URL is a bearer capability: redact it in errors and keep it out of task rows and repository subprocess environments. Existing session-tag trust and other platform grants remain separate limitations. This protocol is implemented locally; real IAM/network/expiry gates and the coordinated image/coordinator/policy rollout are documented in the repository's `docs/verification/645-payload-bootstrap.md`. + > Out of scope for this control and tracked separately as GitHub issues: replacing the shared GitHub PAT (GitHub App / Token Vault), binding credentials to the MicroVM via attestation, and scoping AgentCore Memory (namespace isolation by `actorId`/`sessionId` remains its boundary). diff --git a/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md b/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md index aed019a84..db2c6cf3c 100644 --- a/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md +++ b/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md @@ -201,25 +201,28 @@ The binary was fine — in the identical image, locally, `claude --version` answ Both halves of the fix are kept, because they answer different questions. `/ready` now **exec's the heavyweight binaries before returning 200** (`claude` required, `git`/`node` best-effort), which is the only mechanism that can make the shipped snapshot warm — and its own budget rises to 300 s, well inside the 3600 s build-hook window, because it now does work whose duration is a cold `exec`. Two structural rules keep that honest, because **per-command timeouts do not compose**: the required command runs FIRST with its own budget so no best-effort warm-up can starve the one that decides whether the snapshot is usable, and the best-effort ones then SHARE the remainder of a total warm-up ceiling that sits inside the hook budget with margin (240 s against 300 s). Without them, three commands at 120 s each would be 360 s — a fix for a runtime failure that produces a build failure instead — and a single hung optional command could hold up a 200 that the required warm-up had already earned. Separately, the version probe's timeout goes from 10 s to 60 s: a probe that exists to print a version string into a log line gains nothing from a tight bound and loses the whole task when it trips. The general rule this generalises to, and the reason it belongs in the ADR rather than only in a comment: **on this backend, a first-touch cost that other substrates pay during container start is deferred to the first task instead**, so anything large and lazily-loaded is a turn-0 hazard unless it is touched in `/ready`. -**Payload delivery** reuses the ECS strategy's S3-pointer pattern, adapted to `runHookPayload` (**≤ 4 KB** — measured, see below): payloads that fit ride inline; the rest are uploaded by the strategy to a platform payload bucket (the ECS payload bucket pattern in `ecs-agent-cluster.ts`: orchestrator write access, compute-role read-only scoped to the bucket, lifecycle expiry on objects) with the S3 URI in `runHookPayload` in place of the payload itself — the MicroVM **execution role** holds the read grant, exactly as the ECS task role does today. Since P2 the hook body also carries `platform_config` in both branches, so `runHookPayload` is never *only* the URI (see the canonical shapes below). +**Payload delivery — v2 amendment (2026-09-13, #817/#700).** The local implementation now shares an authenticated bootstrap protocol between ECS and MicroVM. This replaces the original inline/S3-pointer protocol and broad worker payload-bucket read grant. The historical P1/P2 runs below predate this amendment; AWS authorization, networking, expiry and coordinated deployment validation remain pending. The repository runbook is `docs/verification/645-payload-bootstrap.md`. -The cap is **4 096 bytes**, not the 16 384 the SDK documents. Measured exactly: 4 096 passes, 4 097 is rejected with *"Value at 'runHookPayload' failed to satisfy constraint: Member must have length less than or equal to 4096"*. Two consequences follow. First, the original threshold would have inlined every envelope between 4 097 and 16 384 bytes and had the service reject all of them. Second, and more structurally: **the S3-pointer path is now the dominant one, and inline is the exception.** A hydrated task payload (prompt + issue thread + repo context) essentially always exceeds 4 KB, so "small payloads ride inline" describes tiny repo-less prompts rather than the common case. The payload bucket is therefore not a rarely-exercised overflow valve but a required part of every normal task, which raises its lifecycle rule (`MICROVM_PAYLOAD_TTL_DAYS`) and the execution role's read grant from edge-case plumbing to load-bearing. +Every task uses S3. The coordinator publishes a non-secret deployment manifest at `bootstrap/.json`, conditionally creates `/payload.json`, and sends a short-lived signed URL for that one task object. The worker reads the manifest with its own AWS role, whose policy explicitly denies object reads outside its deployment's `bootstrap/*` and denies payload-bucket listing. The explicit deny also prevents a foreign public bucket from authenticating a forged manifest. The task document must name the reference's task ID and contain the exact manifest configuration before the MicroVM installs any of it. -**Canonical wire shapes.** Three, and the producer (`lambda-microvm-strategy.ts`) emits exactly these: +The `runHookPayload` cap remains **4,096 bytes**, measured against the service despite the SDK's documented 16,384. It now applies to the serialized reference rather than choosing an inline branch. Payloads are capped at 8 MiB and manifests at 16 KiB; the bounds and version are shared in `contracts/constants.json`. -| Where | Exact shape | +| Location | v2 shape | |---|---| -| `runHookPayload`, inline branch | `{"agent_payload": {…}, "platform_config": {…}}` | -| `runHookPayload`, pointer branch | `{"agent_payload_s3_uri": "s3://…", "platform_config": {…}}` | -| the object at that S3 URI | `{…agent_payload fields…, "platform_config": {…}}` — the payload's own fields at the TOP level, with the config merged in beside them | +| `runHookPayload` string / ECS `AGENT_PAYLOAD_REF` | `{version:2, task_id, bootstrap_s3_uri, payload_url, expires_at}` | +| `bootstrap/.json` | `{version:2, backend, platform_config}` | +| `/payload.json` | `{version:2, task_id, agent_payload, platform_config}` | +| Private `/launch.json` | `{fingerprint, reference}`; coordinator-readable only, never on the agent-readable task row | -Two asymmetries are deliberate and must not be "tidied" without changing both sides. First, **the S3 object is not the envelope**: the payload's fields sit at the top level (that is what P1 uploaded, before `platform_config` existed) rather than nested under an `agent_payload` key. Second, **`platform_config` is duplicated** on the pointer path — once beside the pointer, once inside the uploaded object. It costs a few hundred bytes and buys the property that the config is reachable whichever end of the fetch a reader looks at, which matters because it is the agent's only substitute for an env block. +The coordinator saves the exact reference for replay: re-signing would change a Run request with the same client token. Task objects use conditional creation and read-back recovery after a lost committed write. URL lifetime is at most 900 seconds, bounded by known credential expiry, and initial creation requires at least 300 seconds. Expired saved references fail without re-signing. The coordinator needs bucket-scoped `ListBucket` so an absent launch record returns `NoSuchKey` rather than `AccessDenied`. It deletes both task objects at finalization; one-day lifecycle expiry is the backstop. Manifests are refreshed with identical bytes at preparation and may coexist across configuration revisions until lifecycle cleanup. -The agent's reader is deliberately more permissive than this contract: it also accepts an S3 object shaped like the envelope (`agent_payload` nested), and `platform_config` present in only one of the two places (the fetched object wins, the hook body is the fallback). Those are **defensive compatibility** for the independent deploy cadences of a snapshot image and the orchestrator Lambda — a tolerant reader, not an alternative contract. A producer must emit the three shapes above. +The image accepts only v2. The old inline/pointer forms and baked-environment fallback are rejected. A coordinated drain and switch of coordinator, images and application IAM is required; there is no new KMS/SSM resource or bootstrap bundle version. Signed URLs are bearer capabilities and must never appear in task rows, ordinary logs or repository subprocess environments. ECS consumes/removes its reference before importing the task pipeline. The URL consumer restricts requests to the exact task object on regional S3 HTTPS, disables redirects/proxies and bounds response bytes. AWS verifies signatures and actual expiration. -**Platform configuration delivery (P2): payload-sourced, allowlisted, fail-closed.** The other two backends hand the agent its non-secret platform env at launch — AgentCore Runtime env vars, ECS container overrides — and there is no equivalent on this substrate: a MicroVM starts from a **snapshot**, so its process environment is whatever was frozen at *image build* time and is then replayed by every MicroVM launched from that image version. Baking the deployment's identifiers into the snapshot would make them **version-frozen**: a redeploy that renames a table, adds a bucket or rotates the session role would leave every existing image version describing a deployment that no longer exists, and the drift would surface as a task-time `ResourceNotFound` rather than a deploy-time error. So the values travel with the task instead: `platform_config` is a SIBLING of the payload — beside `agent_payload` in the inline branch, beside `agent_payload_s3_uri` in the pointer branch, and merged in beside the payload's own fields inside the S3 object (the canonical shapes above give each one exactly) — whose snake_case keys the agent installs into `os.environ` as their UPPER_SNAKE equivalents. A payload value therefore **wins** over any pre-existing/image value — the orchestrator is describing the live deployment, the snapshot is describing a past one. `platform_config` carries **non-secret identifiers only** (table and bucket names, secret ARNs, the session-role ARN); secrets are still fetched at `/run` time from Secrets Manager using those ARNs, so the snapshot-must-stay-secret-free requirement above is untouched. Per-task fields — `memory_id` and friends — stay inside `agent_payload`: `platform_config` configures the *process*, `agent_payload` describes the *task*. +**Platform configuration delivery (P2, strengthened by v2).** A MicroVM restores a snapshot, including its process environment, so current deployment identifiers must arrive at task launch. The authenticated manifest and downloaded document both contain `platform_config`: non-secret table/bucket names, secret ARNs and session-role ARN. Per-task fields such as `memory_id` remain in `agent_payload`. ECS's manifest config is empty because ECS already receives deployment settings through its task definition and coordinator overrides. -Two rules make it safe. First, **the allowlist fails closed**: the agent installs a fixed set of keys and *rejects the entire run* (HTTP 400, nothing spawned, not one key installed) if the block carries anything else. These values become environment variables of the process that spawns the agent's tool subprocesses, so an unrecognised key is an attempt to set an arbitrary variable in the agent (`AWS_ENDPOINT_URL`, `LD_PRELOAD`, `PATH`, …) — an injection attempt, not a forward-compatibility gap, which is why unknown keys are refused rather than filtered out. Second, **installation happens before any credential or pipeline initialisation** on the hook path: the very next step reads `GITHUB_TOKEN_SECRET_ARN` to resolve the GitHub token and `AGENT_SESSION_ROLE_ARN` to scope the task's credentials, so installing later would silently resolve the whole task against the snapshot's frozen env. The one call that must precede installation is the S3 payload fetch (the config is *inside* the fetched object), which therefore runs on the ambient compute role via the attributed platform client — and it is the ONLY one: the same rule covers **logging**, so every `/run` log line before the install is stdout-only. The CloudWatch writer would otherwise resolve credentials and cache credentials in a boto3 default session (environment-derived region is re-read for each new client) off whatever a snapshot happened to bake, which is the build-hook defect one phase later. Nothing is lost — in the intended deployment there is no baked `LOG_GROUP_NAME`, so those lines would have gone to stdout anyway, and the reason for every pre-install rejection also travels in the structured 4xx/5xx body the service surfaces. A **required subset** (task table, task-events table, GitHub token secret ARN, session-role ARN) is rejected as `…_INCOMPLETE` when missing or blank — a distinct wire code from the `…_INVALID` allowlist rejection, because the remedies differ (deployment wiring vs. producer bug). A `/run` envelope with *no* `platform_config` is accepted with a warning only when the effective environment already provides every required identifier; otherwise it is rejected as incomplete. The image and orchestrator can deploy on independent cadences without accepting an unusable environment. The key set is a cross-package contract in `contracts/constants.json` (`microvm_platform_config`), consumed by the agent's `/run` hook and produced by the orchestrator, with shape and required-subset invariants enforced by `scripts/check-constants-sync.ts`. +MicroVM installs only allowlisted keys and rejects unknown keys before installing any. Required identifiers must be nonblank; malformed/control-character values and inconsistent ARN partitions/accounts are rejected. Legitimate cross-region secrets remain allowed when present in the trusted manifest. Account agreement alone cannot authenticate a sibling ARN; v2's manifest/config comparison supplies the provenance check. No-config envelopes are rejected regardless of the image's existing environment. + +Manifest read and signed payload download precede configuration installation. Secrets, task-scoped credential setup, CloudWatch initialization and the pipeline follow installation; pre-install diagnostics use stdout without unsafe exception chains. Build hooks remain AWS-silent. The settings allowlist and bootstrap bounds are cross-language contracts with drift checks. These source guarantees do not establish complete hostile-worker isolation: roles still retain other platform grants and choose session tags, and a stolen signed URL remains usable until expiry or revocation. **No orchestrator→agent HTTP path exists in P1–P3**: payload arrives through the `/run` hook, all agent work is outbound, and therefore **no JWE auth tokens are minted at all** — token minting (and its ≤ 60 min TTL refresh problem) is deferred until a real consumer exists (e.g. operator shell access, [#391](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/391)). The `endpoint` stays in the `SessionHandle` because it is genuinely per-session state that becomes load-bearing the day such a consumer appears. But note the service does not agree by default: omitting `ingressNetworkConnectors` on `RunMicrovm` attaches a **public** `HTTP_INGRESS` connector, so the strategy passes the Lambda-managed `NO_INGRESS` connector explicitly on every launch (see sub-decision 4's security table). @@ -228,12 +231,12 @@ Two rules make it safe. First, **the allowlist fails closed**: the agent install Normative requirements (EARS). **Each requirement's own `(Pn)` tag is authoritative**; there is no blanket phase for the list. The tags are per-requirement because the original list *was* split P1/P2 on the assumption that a hook could be declared in one phase and served in a later one — which the service does not permit (see the phasing table above), so the hook-serving requirements collapsed into P1 while the P2 items below arrived with the P2 hooks and `platform_config`: - (P1) The image build shall not embed secrets, tokens, or per-task identity in the snapshot. -- (P1) If the task payload exceeds the 4 KB `runHookPayload` limit, then the strategy shall upload the payload to the platform payload bucket and pass its S3 URI in `runHookPayload` in place of the payload (from P2, alongside `platform_config`). -- (P1) The MicroVM execution role shall hold read-only access to the payload bucket, scoped to that bucket. +- (P1, amended by v2) Every task shall use the authenticated manifest and single-object signed payload reference; the serialized hook reference shall not exceed 4,096 bytes. +- (P1, amended by v2) The worker role shall read only deployment bootstrap manifests with ambient credentials; other object reads and payload-bucket listing shall be explicitly denied. The coordinator shall own task-object writes, signing, replay and deletion. - (P1) Where no ingress is configured for a deployment, the strategy shall pass the Lambda-managed `NO_INGRESS` network connector on every `RunMicrovm` call (the field shall not be omitted). - (P1) Where the image enables any MicroVM lifecycle hook, the image shall also enable the `/ready` hook and the agent shall serve it. - (P1) When the `/run` hook receives the task payload, the agent shall validate it, start the pipeline asynchronously, and return HTTP 200 within the hook budget. -- (P1) If the `/run` hook payload cannot be resolved to a task payload, then the agent shall reject the hook with a client error and shall not start a pipeline. +- (P1, clarified) Invalid hook references shall return a structured client error; unreadable manifest/payload bytes shall return a structured server error. Neither path shall install configuration or start a pipeline. - (P1) The agent shall not execute the clone→verify→PR pipeline on the hook path. - (P1) The agent shall resolve credentials at `/run` time. - (P1) Where a deployment configures a MicroVM image before smoke parity is verified, the platform shall warn that the backend has no smoke-parity guarantee. @@ -398,8 +401,8 @@ Lambda MicroVMs launched in 5 regions (us-east-1/2, us-west-2, eu-west-1, ap-nor **P1 (start/poll/stop — no suspend):** -- Unit tests for the strategy: start/poll/stop mapping (including `SessionStatus` `'suspended'` reported mechanically, without task-state interpretation), payload-size branching (inline vs S3 pointer, at the exact 4 096/4 097-byte boundary), the image-identifier-must-be-an-ARN guard, the explicit `NO_INGRESS` argument (including the blank-env-var fallback, which must never omit the field), error classification (`ServiceQuotaExceededException`, `ThrottlingException`, `ResourceNotFoundException`, regional-unavailability), the omit-`idlePolicy` invariant, and `maximumDurationInSeconds` fixed at 28 800. -- Agent tests: `/ready` returns 200 after required local warm-up succeeds, without starting a task or calling AWS; `/run` accepts both envelope shapes (inline and S3 pointer), starts the pipeline asynchronously through the same mapper `/invocations` uses, returns before the pipeline finishes, and rejects every unusable envelope with a named code before spawning; `/suspend` and `/resume` are NOT served (`/validate` and `/terminate` joined the served set in P2). +- Unit tests for the strategy: start/poll/stop mapping (including `SessionStatus` `'suspended'` reported mechanically, without task-state interpretation), v2 reference-size bounds (4,096 bytes), immutable signed-reference replay and scoped payload transport, the image-identifier-must-be-an-ARN guard, the explicit `NO_INGRESS` argument (including the blank-env-var fallback, which must never omit the field), error classification (`ServiceQuotaExceededException`, `ThrottlingException`, `ResourceNotFoundException`, regional-unavailability), the omit-`idlePolicy` invariant, and `maximumDurationInSeconds` fixed at 28 800. +- Agent tests: `/ready` returns 200 after required local warm-up succeeds, without starting a task or calling AWS; `/run` accepts authenticated v2 references and rejects old unsigned envelopes, starts the pipeline asynchronously through the same mapper `/invocations` uses, returns before the pipeline finishes, and rejects every unusable envelope with a named code before spawning; `/suspend` and `/resume` are NOT served (`/validate` and `/terminate` joined the served set in P2). - Orchestrator tests: substrate-terminal + non-terminal task status → failed classification; `suspended` + non-`AWAITING_APPROVAL` status → anomaly event, no fail-fast; `compute_metadata` persisted with `microvmId`/`endpoint` after `startSession`. - CDK assertions: MicroVM resources present only when `ComputeTypes` includes the backend; synth failure for unsupported regions (plus the context-flag escape hatch); memory size validated against the accepted list at synth; the connector operator role and its trust; two connectors with the build-only one carrying port 80 and the runtime one not; the `/ready` + `/run` hook declaration and the absence of the others; IAM actions scoped as specified (orchestrator lifecycle set; no `CreateMicrovmAuthToken` anywhere); payload-bucket grants (execution role read-only); backend cost-allocation tags; types-sync check covers the widened `ComputeType`. - CLI tests: onboarding rejection with remedy when the availability probe fails; doctor check present when a blueprint selects the backend. @@ -408,7 +411,7 @@ Lambda MicroVMs launched in 5 regions (us-east-1/2, us-west-2, eu-west-1, ap-nor **P2 (smoke parity):** - Agent tests: `/validate` returns 200 with its individual check results, 503 while initialising, reports a missing hook route / unsupported interpreter, starts nothing, and makes **zero AWS calls even with `LOG_GROUP_NAME` set** (asserted by poisoning the boto3 and CloudWatch-writer seams — the same assertion covers `/ready`); `/terminate` returns 200 with no body at all, with a malformed / non-object / wrong-content-type / whitespace-only body, when the body read itself fails, with a pipeline still running (without joining it), and when its own best-effort step raises — and never calls `task_state.write_terminal`; a structural assertion that the route carries no typed body param keeps the 422 from being reintroduced. -- `platform_config` tests: the allowlist and required subset are read from `contracts/constants.json` (the wire key set is additionally asserted literally, as the agent-side tripwire on a contract edit); an unknown key rejects the whole block with nothing installed; a non-object block and a non-string value are rejected; a blank/`null` optional value is skipped without clobbering an image value while a blank required value is rejected; a payload value beats a pre-existing env value; installation is observed to happen before the GitHub-token resolver runs and before any pipeline thread exists; the config is picked up from the inline envelope, from beside the S3 pointer, and from inside the fetched object (inner wins); an envelope with no `platform_config` is accepted with a warning only if the effective environment already supplies every required identifier; otherwise it is rejected. +- `platform_config` tests: the allowlist and required subset are read from `contracts/constants.json` (the wire key set is additionally asserted literally, as the agent-side tripwire on a contract edit); an unknown key rejects the whole block with nothing installed; a non-object block and a non-string value are rejected; a blank/`null` optional value is skipped without clobbering an image value while a blank required value is rejected; a payload value beats a pre-existing env value; installation is observed to happen before the GitHub-token resolver runs and before any pipeline thread exists; the downloaded config must equal the IAM-authenticated manifest, including same-account workspace identifiers; missing config and legacy envelopes are rejected regardless of baked environment values. Transport tests cover signed URLs, exact task identity, bad bytes and rejection before installation. - Snapshot credential hygiene: a subprocess probe asserts that importing `server` and serving `/ready` + `/validate` imports neither `boto3` nor `botocore`, caches no `aws_session` session, and spawns no CloudWatch writer thread — the property that keeps a build-role credential chain and the build-time region out of the snapshot. - `/run` pre-install silence: with a **baked `LOG_GROUP_NAME`** (the hostile case — without it the assertions pass vacuously) every AWS/credential seam (`boto3.client`/`Session`, the `aws_session` factories, `_debug_cw`/`_warn_cw`) is armed to raise until the install succeeds. Asserted on the accepted path, on all three rejection paths (bad envelope, `platform_config` invalid, `platform_config` incomplete) and on the failed-fetch 500 — where the seams stay armed for the whole request, because a rejected run installed nothing and so earns no AWS call. The permitted exception is asserted POSITIVELY: exactly one client is built pre-install, for `s3`, through the attributed factory. - `/ready` warm-up tests (P2-F5): the hook exec's each configured binary exactly once with a generous timeout; `claude` is the only REQUIRED entry; a timeout, a missing binary, a non-zero exit and an unexpected `OSError` each produce **503 with the reason logged to stdout** rather than a 200 or a 500; a best-effort failure still reports ready; the warm-up makes zero AWS calls with `LOG_GROUP_NAME` baked. Plus the backstop half: the `claude --version` probe's bound is asserted to be ≥ 60 s and to be applied to the *exec* rather than to the PATH lookup, and a missing CLI warns instead of raising. diff --git a/docs/verification/645-coordinator-metadata.md b/docs/verification/645-coordinator-metadata.md index 93f1e62bb..1784ec45b 100644 --- a/docs/verification/645-coordinator-metadata.md +++ b/docs/verification/645-coordinator-metadata.md @@ -67,3 +67,7 @@ Do not roll back to unrestricted worker writes while relying on protected reserv The agent can still report status and results. This patch does not authenticate whether its reported success/failure is truthful or constrain status values/transitions at IAM level. Supporting approval/event/nudge rows retain their prior permissions. The compute role chooses `{user_id, repo, task_id}` session tags. Current trust policies do not independently prove that the chosen task belongs to that worker. A compromised whole worker with ambient credentials can therefore try to assume a differently tagged session. The attribute restriction applies to that session too, but this is not complete tenant isolation. Trusted task/deployment identity and task-scoped payload reads remain separate prerequisites in the [P3 plan](./645-p3-implementation-plan.md). + +## Payload capability storage + +The subsequent [v2 payload bootstrap](./645-payload-bootstrap.md) stores signed download URLs only in coordinator-owned S3 launch records. They are not added to `microvm_start` or other TaskTable attributes: own-task reads still include internal metadata, so API omission and attribute write restrictions cannot provide confidentiality for a bearer URL. diff --git a/docs/verification/645-p3-implementation-plan.md b/docs/verification/645-p3-implementation-plan.md index b4cb9b74d..2695fe29f 100644 --- a/docs/verification/645-p3-implementation-plan.md +++ b/docs/verification/645-p3-implementation-plan.md @@ -13,7 +13,8 @@ Prerequisite work is tracked here on `fix/645-microvm-readiness`. “Completed - [x] Stabilize terminal failure classification and user retry guidance (#817). - [x] Exercise real S3 bad-byte paths and fix closed-stream error classification (#817). - [x] Require new ARN fields to participate in validation; pin contract fields and anchor (#817). -- [ ] Bind configuration to trusted deployment identity and restrict payload reads per task (#817 / #700). +- [x] Bind configuration to IAM-authenticated deployment manifests and use single-object payload links for ECS/MicroVM (#817 / #700). +- [ ] Verify v2 bootstrap policies, S3 conditional writes, expiry, networking and coordinated rollout in AWS. - [x] Implement saved MicroVM start receipts, stable tokens, input fingerprints and handle recovery. - [ ] Verify AWS token retention/conflicts and unknown-start cleanup on a live deployment. - [x] Make capacity acquisition/release atomic per task across crash replay; unify counter writers and repair. @@ -73,6 +74,13 @@ Fifth prerequisite batch completed locally on 2026-09-13: - The old-policy regression failed on `PutItem`. **1,795 Python tests** pass with **84.47%** coverage, including actual writer-request/permission-contract checks; **268 CDK tests** pass across session-role, ECS, MicroVM and full-stack suites. Python quality, CDK lint/compilation and documentation checks pass. These counts overlap earlier batches; four obsolete helper tests were removed and two contract tests added. - [Metadata verification](./645-coordinator-metadata.md) records the effective source-policy boundary, writer inventory, rollback constraints and pending real-AWS allowed/denied transaction matrix. No table migration or bootstrap-policy update is required. Application-role deployment and a matching agent image remain necessary; nothing was deployed. +Sixth prerequisite batch completed locally (2026-09-13): + +- v2 payload bootstrap for #817/#700 now covers both ECS and MicroVM: IAM-authenticated deployment manifests, one-object signed downloads, immutable private retry references, bounded bytes/expiry and coordinator cleanup of both task objects. Worker reads outside the manifest prefix and payload-bucket listing are explicitly denied. +- Regressions reproduced and fixed bearer-URL leaks through Python and JavaScript exception chains. Removed the old unsigned transports, stale permission/compatibility comments and unused per-backend key helpers. +- Python quality passed **1,806 tests / 84.51% coverage**. The broad CDK run passed **159 suites / 3,935 tests** with `--detectOpenHandles` and exited normally; **15 existing DynamoDB Local tests were skipped** because this batch did not start that service or change its transaction protocol. The final five transport/strategy suites passed **187 tests** after the last cleanup (overlapping the broad run); CDK lint/compilation, Python lint/type checks, constants-sync, the **77-page** docs build and link checks also pass. +- The [bootstrap runbook](./645-payload-bootstrap.md) records the design, AWS documentation evidence, remaining trust boundaries, live allowed/denied matrix and coordinated deployment/rollback procedure. Nothing was deployed; effective IAM, conditional S3 writes, expiry, DNS/HTTPS, ingress negatives and clean launches remain gates. + ## The result we want When the coding agent asks a human for permission, its computer may go to sleep. The human can approve or deny while it sleeps. The computer wakes, reads the saved answer, and continues or refuses the action. If nobody answers, it wakes before the deadline and applies the existing timeout-as-denial rule. Its files, task identity, permissions, progress and deadline remain correct. @@ -103,7 +111,8 @@ A few implementation words used below: - Keep the eight-hour `maximumDurationInSeconds = 28800`, including suspended time. - Keep `idlePolicy` absent. Traffic-based idleness would mistake outbound-only coding work for inactivity. - Keep explicit `NO_INGRESS` and no `CreateMicrovmAuthToken` grant. No public agent-control endpoint or JWE token refresh system is needed for P3. -- Keep the 4,096-byte hook-payload boundary and S3 fallback, including registry-resolved assets. +- Keep the 4,096-byte serialized hook-reference boundary and v2 S3 transport for every task, including registry-resolved assets. +- Resume the restored task context; do not repeat bootstrap or reuse its short-lived launch URL after sleep. - Only the orchestrator initiates suspension. Approval handlers may request resume after committing a decision; the orchestrator repairs missed resumes. - The agent remains the authority that consumes a decision or times out a gate. Approval HTTP handlers must not simply mark the coding task RUNNING. - Preserve tenant-scoped credentials, task/user/repository tags and fail-closed behavior. “Fail closed” means refusing an action when safe authorization cannot be established. @@ -137,18 +146,20 @@ Keep nesting in a separate change from lifecycle logic. Developing P3 locally ne ### 1A. Finish #817 -- [x] **Deletion IAM:** the coordinator has exact `s3:DeleteObject` on `*/payload.json` in the dedicated bucket. Construct and stack tests pin that scope. The worker retains read-only permissions, and deletion failure does not hide the task outcome. +- [x] **Deletion IAM:** the coordinator has exact `s3:DeleteObject` on `*/payload.json` and `*/launch.json` in the dedicated bucket. Construct and stack tests pin that scope. Worker ambient reads are now restricted to bootstrap manifests; deletion failure does not hide the task outcome. - [x] **Classifier:** reconciliation persists `MICROVM_SUBSTRATE_TERMINATED` or `MICROVM_RUN_HOOK_REJECTED` separately from the descriptive AWS reason. The known leading run-hook 4xx response selects the latter; arbitrary appended text cannot override the code. Legacy task records remain readable. Tests cover task API classification, channel/panel retry guidance, host/capacity failures, regional faults, authorization/configuration, concurrency words, hook 400/500 and other backends. -- [ ] **Trusted configuration:** decide which data is trusted at `/run`. A same-payload account anchor cannot authenticate its siblings. Bind accepted deployment identifiers to trusted deployment configuration or an authenticated payload reference; preserve legitimate cross-region secrets. Reject another workspace's secret even when it has the same account number. +- [x] **Trusted configuration (local):** v2 reads a deployment manifest with ambient IAM credentials restricted by an explicit deny outside that deployment's bootstrap prefix, then verifies downloaded task/config equality. Another workspace's secret in the same account is rejected; an exact trusted cross-region secret is preserved. Both languages share the version/caps. Live effective-IAM, public-bucket and ingress negatives remain open. - [x] **Contract guards:** pin exact contract fields, ARN fields and the account anchor in both languages. Python import-time validation and the constants checker reject newly added `*_arn` / `*_ARN` fields omitted from `arn_keys`. This prevents accidental validation gaps; it does not establish deployment identity. - [x] **S3 bad bytes:** real `StreamingBody` tests cover truncated JSON, invalid encoding, short and closed streams, and non-object JSON through fetch/decode/envelope/route. They assert a structured unreadable response, no configuration installation/environment mutation, and no pipeline thread. A closed-stream `ValueError` now reaches the unreadable-payload 500 branch. Malformed hook envelopes retain the separate 400 response. S3 writes are atomic; invalid stored bytes need replacement, not an assumption that another identical read repairs them. - **Documentation:** verify all remaining contract/status changes update source docs and their generated copies through the sync script. ### 1B. Narrow payload reads (#700) -Choose task-scoped transport before describing the backend as suitable for untrusted multi-tenant tasks. The current role reads the payload **before** it establishes task-scoped identity, so merely moving an existing S3 grant onto the session role is not enough. +**Completed locally for ECS and MicroVM:** every task uses an IAM-authenticated deployment manifest plus a short-lived, single-object signed URL. Worker ambient roles explicitly deny other object reads and payload-bucket listing. The coordinator stores the exact URL privately in S3 for replay, conditionally creates task objects, rejects changed/expired launches, and deletes payload plus launch record at finalization. ECS consumes/removes its capability before the pipeline; errors and Python exception chains must not expose it. + +The [bootstrap runbook](./645-payload-bootstrap.md) records wire/storage shapes, caps, credential lifetime, the coordinator's `ListBucket` requirement for missing-object detection, coordinated drain/image/controller/policy upgrade and rollback, and the real AWS allow/deny matrix. Old unsigned envelopes are intentionally rejected; this is a coordinated contract change, not a rolling mixed-version deployment. -Evaluate a short-lived, single-object signed URL or a trusted bootstrap envelope that safely establishes task identity first. For a signed URL, check whether the guest's network/DNS policy permits its host and whether retry duration fits the URL lifetime; keep the bearer URL out of logs. For role-based fetching, prove how the role/session tags are trusted before the object is read. Test wrong task, guessed key, expired reference, retries and maximum payload size. Preserve the ECS contract or document a deliberate staged rollout. Finalize-time deletion complements this fix; it does not prevent reads of other active tasks. +**Still required:** effective-role cross-task/public-bucket negatives, actual S3 conditional writes and missing-object behavior, signer/URL expiry, runtime DNS/HTTPS and clean launches for both backends. The role/tag and other platform-grant limits in 1G remain; this boot-path fix does not establish complete hostile-worker isolation. ### 1C. Fix approval/heartbeat ordering diff --git a/docs/verification/645-p3-readiness-review.md b/docs/verification/645-p3-readiness-review.md index 51e0aed64..341d3adc2 100644 --- a/docs/verification/645-p3-readiness-review.md +++ b/docs/verification/645-p3-readiness-review.md @@ -4,7 +4,7 @@ Reviewed 2026-09-13 against `main` commit `5e10038c7e28179b302ac4de78b709795aeba This review covers the existing MicroVM implementation, related P2 follow-ups, comment accuracy and nested-stack feasibility. It includes local experiments, not an AWS deployment or a new live smoke run. The associated [implementation plan](./645-p3-implementation-plan.md) turns the findings into ordered work. -**Implementation update (2026-09-13):** subsequent prerequisite work fixes #841 thread isolation, #817 coordinator payload-deletion permission, terminal-failure classification, closed S3-stream handling, and the approval-resume heartbeat race locally. Real bad-byte route tests and ARN contract guards are also added. Trusted deployment identity, task-scoped payload reads and live verification remain open. The findings below preserve the reviewed baseline; see [implementation progress](./645-p3-implementation-plan.md#implementation-progress) for commits, checks and remaining work. +**Implementation update (2026-09-13):** subsequent local batches fix thread isolation, deletion/error/byte/contract bugs, approval heartbeat, stable MicroVM start recovery, atomic capacity reservations and coordinator metadata permissions. The latest batch implements v2 authenticated deployment manifests and single-object payload links for both ECS and MicroVM (#817/#700), with no old unsigned fallback. See the [bootstrap runbook](./645-payload-bootstrap.md) and [implementation progress](./645-p3-implementation-plan.md#implementation-progress). Effective AWS policies, expiry/networking, clean deployment, logging/registry follow-ups and P3 sleep/wake remain open. Findings below preserve the original reviewed baseline, rather than describing all of them as current defects. Further local work adds durable MicroVM start receipts, stable request tokens and bounded handle recovery. AWS token-retention/conflict behavior and unknown-ID cleanup remain live verification gates. The shared finalizer's repeated-decrement risk was subsequently reproduced and fixed locally with task-owned reservation transactions, unified counter writers and revision-guarded reconciliation, including approval waits. DynamoDB Local exercises the real transaction conditions; deployed IAM, scan scale and upgrade/drain behavior still need AWS verification. @@ -82,7 +82,7 @@ Nesting is useful preparation, especially for the vault combination, but it does ## Behavior findings that need follow-up -“Confirmed” below means visible in current source or reproduced locally. It does not mean reproduced on AWS during this review. +“Confirmed” below means visible in the baseline source or reproduced locally during the initial review. It does not mean reproduced on AWS during this review. | Finding | Evidence and consequence | Treatment | |---|---|---| @@ -111,7 +111,7 @@ The security findings are readiness work, not cosmetic cleanup. This review does - **[#702](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/702): teardown leaks.** Account for AgentCore ENIs (network attachments) and Memory deletion state when cleaning the test deployment. Separate platform operations work. - **[#736](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/736): future IAM conditions.** Revisit when the service supports suitable context keys. Do not restore the previously broken `iam:PassedToService` condition just to make policies look tighter. -## Cleanup applied in this branch +## Cleanup in the original review commit Changes are comments, Python docstrings, documentation, one cdk-nag explanation string and the vault guard’s error wording. The error no longer claims a current 505-resource overflow; the guard still rejects exactly the same combination. The cdk-nag change affects template metadata, not IAM permissions. diff --git a/docs/verification/645-payload-bootstrap.md b/docs/verification/645-payload-bootstrap.md new file mode 100644 index 000000000..34018e577 --- /dev/null +++ b/docs/verification/645-payload-bootstrap.md @@ -0,0 +1,127 @@ +# #645: trusted task delivery for ECS and MicroVM + +Implementation date: 2026-09-13. Tracks the local prerequisites in #817 and #700. **Implemented locally; AWS authorization, networking, expiry and deployment checks below are still pending.** This does not implement P3 sleep/wake. + +## What changed, in plain language + +Think of S3 as a locked filing cabinet. A **bucket** is one cabinet and an **object** is one file. The **coordinator** is the supervisor that starts tasks. A **worker** is the ECS container or MicroVM that runs the coding agent. + +Previously, each worker's permission badge could read other tasks' instruction files. Also, checking that two AWS addresses named the same account did not prove either address belonged to this deployment. + +The new protocol gives the worker: + +1. A small **manifest**: a list of the deployment's non-secret settings. The worker reads this with its own AWS permission badge. Only this deployment's bootstrap directory is allowed. +2. A **presigned URL**: a temporary download ticket, created by the coordinator, for exactly one task's instruction file. AWS checks its signature. Possessing the ticket permits the read, so the complete URL must be treated like a password. + +The downloaded task must name the expected task ID and contain exactly the settings in the authenticated manifest. Changing a secret's address to another workspace in the same AWS account now fails this comparison. + +This bootstrap means “load the information needed to start a task.” It is distinct from the infrastructure bootstrap that installs deployment permissions. + +## Storage and wire contract + +Both backends use `contracts/constants.json` → `payload_bootstrap`, version 2. All task payloads use this path, including small tasks. + +| Location | Content / owner | +|---|---| +| `bootstrap/.json` | `{version:2, backend, platform_config}`. Coordinator writes; worker reads. The filename contains a SHA-256 digest: a fingerprint of the exact JSON bytes. | +| `/payload.json` | `{version:2, task_id, agent_payload, platform_config}`. Coordinator writes; worker downloads using the signed URL. | +| `/launch.json` | `{fingerprint, reference}`. Private coordinator replay record containing the signed URL. No worker read. | +| MicroVM `runHookPayload` / ECS `AGENT_PAYLOAD_REF` | JSON string containing the reference below. Never log it. | + +```json +{ + "version": 2, + "task_id": "TASK001", + "bootstrap_s3_uri": "s3://deployment-bucket/bootstrap/.json", + "payload_url": "", + "expires_at": 1789312500000 +} +``` + +The example is descriptive; the manifest digest and signed URL must be produced by the coordinator. `expires_at` is milliseconds since the Unix epoch. + +MicroVM manifests contain the allowlisted environment identifiers built from the coordinator's deployment environment. ECS manifests contain `{}` for `platform_config`: its platform settings already arrive through the trusted task definition and coordinator overrides. Task-specific fields remain in `agent_payload`. + +Manifest and payload limits are **16 KiB** and **8 MiB**, respectively. The serialized MicroVM reference must fit **4,096 bytes**. The entire ECS overrides object must fit **8,192 UTF-8 bytes**; the prompt is no longer duplicated in `TASK_DESCRIPTION`. + +The image rejects the old unsigned inline/S3 envelopes. There is no compatibility branch that accepts settings merely because an image already contains the required environment variables. + +## Permission boundary + +| Principal | Allowed | Denied / not granted | +|---|---|---| +| Worker ambient role | Exact `s3:GetObject` on its bucket's `bootstrap/*` | Explicit `Deny s3:GetObject*` outside that prefix, including other/public buckets; explicit `Deny s3:List*` on its payload bucket. No payload write/delete grant. | +| Trusted coordinator | `PutObject` on manifests and `*/payload.json`, `*/launch.json`; `GetObject`/`DeleteObject` on the two task paths; `ListBucket` on its payload bucket | No task-object version deletion or manifest deletion grant added by this helper. | + +The explicit denial outside `bootstrap/*` is essential. An allow-only policy would not authenticate configuration: an attacker-controlled public bucket could grant the worker access to a fake manifest. A successful read with the deployed deny policy establishes its permitted origin; the hash only detects different bytes, not authorship. + +The coordinator needs `ListBucket` because S3 returns **403 AccessDenied**, rather than **404 NoSuchKey**, for a missing object when the caller cannot list the bucket. The first launch probes for its saved record. Workers do not get this permission. + +AWS documents this in [GetObject permissions](https://docs.aws.amazon.com/AmazonS3/latest/API/API_GetObject.html). Its [conditional-write guide](https://docs.aws.amazon.com/AmazonS3/latest/userguide/conditional-writes.html) specifies that the first completed conditional write wins and later writes receive 412; a concurrent delete can produce 409. The [presigned-URL guide](https://docs.aws.amazon.com/AmazonS3/latest/userguide/using-presigned-url.html) confirms that temporary credential expiry can end a URL's validity before its requested expiry. These documents were checked during this review; deployed behavior remains a separate gate. + +The worker reads the manifest through the attributed `platform_client("s3")` before installing configuration. It then downloads the payload over HTTPS without adding worker credentials. Only the exact bucket/task key on a regional S3 host is accepted; redirects, environment proxies, alternate hosts, credentials in URLs, custom ports, duplicate query parameters and non-HTTPS URLs are rejected. AWS validates the actual signature and expiry; local URL checks are not a cryptographic verifier. + +The runtime still needs HTTPS/443 and DNS access to the selected S3 region. The source checks do not establish that the deployed connector permits it. Build hooks remain AWS-silent; before configuration installation, runtime bootstrap uses manifest-read and payload-download operations, and diagnostics go to stdout. + +## Retry, expiry and cleanup + +- The coordinator serializes JSON consistently and conditionally creates task objects with `IfNoneMatch: '*'`. A second writer cannot overwrite existing instructions. Read-back handles a write that committed but lost its reply. +- The saved launch record returns the **same URL** after a retry or coordinator restart. Generating a new URL would change the MicroVM Run request while reusing its client token. The private record is deliberately outside TaskTable because the agent can read its own task row. +- A URL is requested for at most **900 seconds**. If the SDK exposes an earlier credential expiry, that shortens the requested lifetime. Initial creation refuses a lifetime under **300 seconds**. AWS can still reject credentials earlier; the provider's actual expiration behavior needs live evidence. +- An expired saved reference fails explicitly; it is not silently re-signed. Recover any existing compute handle before deciding what to do. A timeout or expired ticket does not prove no worker started. MicroVM's saved-start receipt still governs its 120-second application replay window. +- The coordinator refreshes identical manifest bytes at each preparation so the bucket's one-day lifecycle does not remove an old manifest just before a new task reads it. Historical authorized manifests can coexist until lifecycle deletion; this is provenance authentication, not a “latest configuration only” policy. +- Finalization best-effort deletes both `payload.json` and `launch.json`. Deleting the current payload revokes its unversioned download link. Lifecycle deletion remains a backstop and is asynchronous, not an exact one-day alarm. +- ECS removes `AGENT_PAYLOAD_REF` from its environment before importing the pipeline or starting repository subprocesses. Errors redact signed URLs; Python exception chains must not reintroduce them. Control-plane request metadata and the coordinator's private record still contain the capability and require trusted access. + +No new KMS key, SSM parameter, table or bootstrap bundle version is needed. Application IAM policies, coordinator code and worker images all change. + +For P3, `/resume` must continue the restored task, not re-run bootstrap or reuse its expired launch ticket. The task context/workspace and durable approval records supply the information needed after sleep; credential refresh remains a separate resume barrier. + +## Local verification + +Tests exercise real producer/consumer code with fake services. The Python tests use real streaming bodies for truncation/closure cases. The MicroVM recovery tests use the shared transport while simulating lost Run replies and coordinator restart. + +Coverage includes immutable/replayed/conflicting writes, competing launch preparation, committed writes with lost responses, expiry/short-lived credentials, wrong task/configuration, same-account workspace substitution, public-bucket denial before download, cross-region secret preservation, malformed/oversized bytes, redirect/host rejection, ECS environment removal and both cleanup keys. + +A regression reproduced a signed-URL leak through a chained Python traceback. Route logging and sanitized exceptions now prevent that leak. Separate tests cover coordinator-side error redaction and oversized ECS/MicroVM references. + +Permission tests inspect synthesized IAM policies, including the `NotResource` explicit deny and the coordinator's missing-object permission. The constants-checker subprocess rejects invalid bounds/paths and consumers that stop importing the contract. + +An offline compatibility probe generated a URL using the actual JavaScript AWS SDK with dummy credentials and passed it through the Python URL parser. That proves the observed SDK URL shape is supported; it makes no AWS authorization claim. + +See the [implementation checklist](./645-p3-implementation-plan.md#implementation-progress) for final suite counts. Mock authorization failures are not proof of effective AWS IAM. Conditional S3 writes and the missing-object behavior must also be checked live. + +## Coordinated deployment and rollback + +1. Record the source commit, existing coordinator version, ECS task definitions, MicroVM image/version and effective application policies. Retain the previous deployable artifacts and inspect the CloudFormation change set. Do not delete/recreate storage as an upgrade mechanism. +2. Pause new admissions and scheduled/queue submissions for both affected backends. Drain existing tasks, approvals and pending/retry launches; reconcile uncertain starts and terminate any live test compute. Do not upgrade beneath a running old coordinator. +3. Build the new ECS image and a MicroVM image from the same source/contract. Validate build hooks without task secrets. Retain the old image definitions for rollback. Building the new image before the switch is safe; starting production tasks against mixed versions is not. +4. With admissions paused, deploy the matching coordinator, image/task-definition references and IAM changes. Wait for all resources and permissions to settle. Old producer + new image and new producer + old image both fail the version contract; old workers with new policies cannot use the former broad S3 fetch. +5. Verify effective policies and image identifiers, then perform the live matrix below in the test deployment. Inspect sanitized logs, cleanup and retry records. Resume admissions only after these gates pass. +6. On failure, keep admissions paused. Drain/terminate new-version workers and reconcile unknown launches. Restore the previous code, images/task definitions and application policies together. Old permissions restore the old security exposure; rollback is an availability recovery, not a security fix. Do not reuse a task ID with a conflicting or expired launch record; inspect it and use a fresh task submission once the old compute is accounted for. + +The normal stack already creates distinct payload buckets for the selected backend. Check each backend in its own representative deployment. This runbook does not authorize or perform a deployment. + +## Required AWS evidence + +For **both ECS and MicroVM**, retain source/image identifiers, relevant policy snippets, sanitized error codes, AWS request IDs and timing. Never retain signed URLs, credentials or downloaded customer prompts in evidence. + +| Check | Expected result | +|---|---| +| New task with no stored launch | Coordinator gets `NoSuchKey`, creates immutable payload/reference, worker starts successfully. | +| Worker reads own deployment manifest | Allowed; hash/backend/config validation passes. | +| Worker lists its payload bucket | Explicitly denied. | +| Worker reads own or another task's payload/launch with ambient credentials | Explicitly denied, including guessed keys. | +| Worker reads a valid-looking manifest in another deployment/public bucket | Explicitly denied before configuration installation or payload download. | +| Worker puts/deletes a manifest, payload or launch record | Denied by effective policies; no alternate grant should permit mutation. | +| Correct signed URL versus modified task path/signature | Own URL succeeds; modified request fails. | +| Expired URL / expired signer credentials | S3 rejects; no configuration or pipeline starts; diagnostic has no bearer URL. | +| Same-account other-workspace secret substitution | Reject; cross-region secret explicitly present in the trusted manifest still works. | +| Two preparations / lost committed S3 reply | Same saved reference or explicit conflict, no overwrite of task instructions. | +| Lost MicroVM Run reply / coordinator restart | Same client token and exact request; existing VM recovered, no replacement. Record actual AWS retention/conflict behavior. | +| Typical/large registry-hydrated task | Manifest and exact S3 download work over runtime DNS/HTTPS; hook/override limits hold. #818 still needs its registry-resolution end-to-end case. | +| Success, failure, cancellation | Expected terminal status; compute terminated; both private task objects deleted or cleanup failure visibly recorded. | +| Attempted public MicroVM ingress | No usable public control path with explicit `NO_INGRESS`. Test while the VM is running. | +| Environment/log inspection | Repository subprocesses do not inherit `AGENT_PAYLOAD_REF`; normal/error logs contain no signed URL. | + +This boundary authenticates the boot path under the deployed policy. It does **not** prove complete isolation from a fully compromised worker: compute roles still select session tags and retain other platform permissions, and a stolen bearer link can be used until it expires or is revoked. See [coordinator metadata verification](./645-coordinator-metadata.md) for the remaining task-reporting and session-tag trust limits. diff --git a/scripts/check-constants-sync.ts b/scripts/check-constants-sync.ts index d4dc86afa..1fd1f3a3f 100644 --- a/scripts/check-constants-sync.ts +++ b/scripts/check-constants-sync.ts @@ -49,7 +49,9 @@ const POLICY_PY = path.join(REPO_ROOT, 'agent/src/policy.py'); const JIRA_REACTIONS_PY = path.join(REPO_ROOT, 'agent/src/jira_reactions.py'); const SERVER_PY = path.join(REPO_ROOT, 'agent/src/server.py'); const CONFIG_PY = path.join(REPO_ROOT, 'agent/src/config.py'); -const PYTHON_CONSUMERS = [POLICY_PY, JIRA_REACTIONS_PY, SERVER_PY, CONFIG_PY]; +const PAYLOAD_BOOTSTRAP_PY = path.join(REPO_ROOT, 'agent/src/payload_bootstrap.py'); +const PAYLOAD_BOOTSTRAP_TS = path.join(REPO_ROOT, 'cdk/src/handlers/shared/payload-bootstrap.ts'); +const PYTHON_CONSUMERS = [POLICY_PY, JIRA_REACTIONS_PY, SERVER_PY, CONFIG_PY, PAYLOAD_BOOTSTRAP_PY]; const MICROVM_COMPUTE_TS = path.join(REPO_ROOT, 'cdk/src/constructs/lambda-microvm-compute.ts'); const TS_CONSUMERS = [MICROVM_COMPUTE_TS]; @@ -185,6 +187,15 @@ function main(): number { warmup_total_budget_seconds: number; warmup_required_timeout_seconds: number; }; + payload_bootstrap?: { + version: number; + manifest_prefix: string; + launch_filename: string; + max_manifest_bytes: number; + max_payload_bytes: number; + url_ttl_seconds: number; + minimum_url_lifetime_seconds: number; + }; }; try { json = JSON.parse(fs.readFileSync(CONSTANTS_JSON, 'utf-8')); @@ -363,6 +374,27 @@ function main(): number { ); } + const bootstrap = json.payload_bootstrap; + const bootstrapNumbers = [ + 'version', 'max_manifest_bytes', 'max_payload_bytes', + 'url_ttl_seconds', 'minimum_url_lifetime_seconds', + ] as const; + if (!bootstrap || bootstrapNumbers.some(key => !Number.isInteger(bootstrap[key]) || bootstrap[key] <= 0) + || typeof bootstrap.manifest_prefix !== 'string' || !/^[a-z][a-z0-9_-]*\/$/.test(bootstrap.manifest_prefix) + || typeof bootstrap.launch_filename !== 'string' || !/^[a-z][a-z0-9_-]*\.json$/.test(bootstrap.launch_filename) + || bootstrap.launch_filename === 'payload.json') { + invariantErrors.push('payload_bootstrap must define positive integer bounds and distinct safe object paths'); + } else if (bootstrap.minimum_url_lifetime_seconds > bootstrap.url_ttl_seconds + || bootstrap.max_manifest_bytes > bootstrap.max_payload_bytes) { + invariantErrors.push('payload_bootstrap minimum lifetime/manifest size exceeds its corresponding maximum'); + } + // Check that both implementations still read the shared block. Literal copies + // would allow a security cap or protocol version to drift across languages. + if (!/CONTRACT\s*=\s*SHARED_CONSTANTS\["payload_bootstrap"\]/.test(fs.readFileSync(PAYLOAD_BOOTSTRAP_PY, 'utf8')) + || !/PAYLOAD_BOOTSTRAP\s*=\s*constants\.payload_bootstrap/.test(fs.readFileSync(PAYLOAD_BOOTSTRAP_TS, 'utf8'))) { + invariantErrors.push('payload_bootstrap consumers must read the shared contract'); + } + if (invariantErrors.length > 0) { console.error(`Semantic invariant violations in ${CONSTANTS_JSON}:\n`); for (const e of invariantErrors) console.error(` - ${e}`); From 6a8ab35fd5aeacc40ed3ca6f35d4447c27001abe Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Sun, 13 Sep 2026 17:28:17 -0400 Subject: [PATCH 016/149] fix(agent): report logging failures and verify registry delivery (#645) --- agent/README.md | 2 +- agent/src/server.py | 64 +++++------ agent/tests/test_payload_bootstrap.py | 45 ++++++++ agent/tests/test_server.py | 106 ++++++++---------- .../handlers/orchestrate-task-microvm.test.ts | 66 ++++++++++- docs/design/COMPUTE.md | 2 +- docs/design/REGISTRY.md | 6 + docs/guides/DEPLOYMENT_GUIDE.md | 4 +- docs/src/content/docs/architecture/Compute.md | 2 +- .../src/content/docs/architecture/Registry.md | 6 + .../docs/getting-started/Deployment-guide.md | 4 +- .../645-p3-implementation-plan.md | 15 ++- docs/verification/645-p3-readiness-review.md | 2 +- docs/verification/645-payload-bootstrap.md | 4 +- 14 files changed, 217 insertions(+), 111 deletions(-) diff --git a/agent/README.md b/agent/README.md index ba8c9fc9f..fffb8070c 100644 --- a/agent/README.md +++ b/agent/README.md @@ -236,7 +236,7 @@ The warm-up is the *primary* fix; the probe that failed is also now non-fatal. ` **`POST /aws/lambda-microvms/runtime/v1/validate`** — Build hook (P2). A **shallow self-check only**: server alive, every hook route registered, interpreter floor, `platform_config` contract loaded. Returns 200 with the individual check results, or 503 while still initialising (which fails the build if it never clears — the right outcome for a broken snapshot). That 503 branch is a **refactor tripwire**, not a state you can reach today: `_module_initialized` is set as the module's last statement and uvicorn accepts no request until the import completes, so it only becomes reachable once someone moves warm-up work behind the bind — and a hook with only a 200 path would then report a still-initialising snapshot as valid. -It runs under the **build role**, which has no Bedrock, Secrets Manager or DynamoDB runtime grants. Both build hooks make **zero AWS API calls**, including logging: `_build_hook_log` writes to stdout. The build role can write within its MicroVM log namespace, but an application log group may be outside that grant. More importantly, resolving credentials before taking the snapshot can preserve build-role credential state. Boto3 caches resolved credentials; environment-derived region settings are re-read for each new client. The `_debug_cw_failures` counter is currently unexported ([#810](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/810)), so it is not an alarm signal. Local binary warm-up belongs in `/ready`; runtime AWS access must be tested on a real task. +It runs under the **build role**, which has no Bedrock, Secrets Manager or DynamoDB runtime grants. Both build hooks make **zero AWS API calls**, including logging: `_build_hook_log` writes to stdout. The build role can write within its MicroVM log namespace, but an application log group may be outside that grant. More importantly, resolving credentials before taking the snapshot can preserve build-role credential state. Boto3 caches resolved credentials; environment-derived region settings are re-read for each new client. Runtime debug/warn writer failures emit a `cloudwatch_write_failed` stdout record containing `writer`, `task_id` and `error_type` ([#810](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/810)). The unused counter was removed. The fallback performs no AWS call and includes neither the failed log body nor exception message. It is a structured log, not a metric or configured alarm; its visibility depends on guest stdout collection, which AgentCore APPLICATION_LOGS does not automatically provide. Local binary warm-up belongs in `/ready`; runtime AWS access must be tested on a real task. Baked secrets are **reported, not enforced**: `warnings` lists the names (never values) of any credential-shaped env var present in the snapshot, because the build environment's own credentials may legitimately be in that env and failing here would fail every build. diff --git a/agent/src/server.py b/agent/src/server.py index 8dd724ef2..79da35a20 100644 --- a/agent/src/server.py +++ b/agent/src/server.py @@ -35,17 +35,6 @@ from pipeline import run_task from shared_constants import SHARED_CONSTANTS -# --- _debug_cw / _warn_cw failure counter ------------------------------- -# Shared counter for BOTH the debug and warn CloudWatch writers. It is not -# exported or read yet (#810), so it cannot currently reveal a broken writer -# to an operator. Defined BEFORE any function that -# references it (including ``_debug_cw`` / ``_warn_cw``) so the ordering is -# import-time safe: a daemon thread spawned from a write-blocking function -# can never race with module-level globals still being assigned. -_debug_cw_failures = 0 -_debug_cw_failures_lock = threading.Lock() -_DEBUG_CW_FAILURE_EMIT_EVERY = 5 - # Only redact secrets at least this long — replacing very short strings # would mangle unrelated text that happens to contain them. _MIN_REDACTABLE_SECRET_LEN = 12 @@ -84,6 +73,25 @@ def _emit_stdout_line(stamped: str) -> None: pass +def _report_cloudwatch_failure(writer: str, task_id: str | None, exc: Exception) -> None: + """Emit a stdout fallback without another AWS call or the failed message. + + This is a structured log, not a metric or configured alarm. Its availability + depends on the backend collecting guest stdout; AgentCore APPLICATION_LOGS + does not automatically forward it. + """ + _emit_stdout_line( + json.dumps( + { + "event": "cloudwatch_write_failed", + "writer": writer, + "task_id": task_id, + "error_type": type(exc).__name__, + } + ) + ) + + def _debug_cw(msg: str, *, task_id: str | None = None) -> None: """Write a debug line to a CloudWatch stream in a background thread. @@ -139,9 +147,9 @@ def _warn_cw(msg: str, *, task_id: str | None = None) -> None: The stdout emission is preserved so local ``docker-compose`` runs and the ``capfd``-based unit tests still observe the line. - CloudWatch delivery is fire-and-forget — failures bump the - shared ``_debug_cw_failures`` counter via ``_warn_cw_write_blocking`` - which is currently unexported (#810); it is not an observable metric yet. + CloudWatch delivery is fire-and-forget. Failures emit a structured + ``cloudwatch_write_failed`` stdout record without retrying through the + failed logging path. No metric or alarm is installed by this helper. """ # Redact cached credentials and emit via the same os.write path as # ``_debug_cw``: warn messages can embed payload fragments, so they @@ -170,9 +178,8 @@ def _warn_cw_write_blocking(log_group: str, task_id: str | None, stamped: str) - Mirrors ``_debug_cw_write_blocking`` but writes to the ``server_warn/`` stream so warn-level traffic is easy to - alarm on independently of debug breadcrumbs. Failures bump the - shared ``_debug_cw_failures`` counter. Nothing exports that counter - yet (#810), so it does not currently provide an alarm surface. + filter independently of debug breadcrumbs. Failures emit the shared + structured stdout fallback without another AWS call. """ try: from aws_session import platform_client @@ -189,14 +196,8 @@ def _warn_cw_write_blocking(log_group: str, task_id: str | None, stamped: str) - logStreamName=stream, logEvents=[{"timestamp": int(_time_for_debug.time() * 1000), "message": stamped}], ) - except Exception as _exc: - global _debug_cw_failures - with _debug_cw_failures_lock: - _debug_cw_failures += 1 - print( - f"[server/warn/self] CloudWatch write failed: {type(_exc).__name__}: {_exc}", - flush=True, - ) + except Exception as exc: + _report_cloudwatch_failure("warn", task_id, exc) def _debug_cw_write_blocking(log_group: str, task_id: str | None, stamped: str) -> None: @@ -216,16 +217,9 @@ def _debug_cw_write_blocking(log_group: str, task_id: str | None, stamped: str) logStreamName=stream, logEvents=[{"timestamp": int(_time_for_debug.time() * 1000), "message": stamped}], ) - except Exception as _exc: - # Never let debug logging break the request path. Bump the failure - # counter so operators can alarm on a blind debug path. - global _debug_cw_failures - with _debug_cw_failures_lock: - _debug_cw_failures += 1 - print( - f"[server/debug/self] CloudWatch write failed: {type(_exc).__name__}: {_exc}", - flush=True, - ) + except Exception as exc: + # Logging failures must not break the request or recursively log to AWS. + _report_cloudwatch_failure("debug", task_id, exc) # Log the active event loop policy at import time. diff --git a/agent/tests/test_payload_bootstrap.py b/agent/tests/test_payload_bootstrap.py index 04af816b0..f3b842eee 100644 --- a/agent/tests/test_payload_bootstrap.py +++ b/agent/tests/test_payload_bootstrap.py @@ -122,6 +122,51 @@ def test_authenticates_manifest_before_downloading_and_preserves_cross_region_se assert transport["build"].call_args.args[0].proxies == {} +def test_large_registry_bundle_reaches_run_mapper_and_mcp_loader(transport, monkeypatch, tmp_path): + """Resolve real v2 bytes and carry the bundle from /run into the real local loader.""" + from registry.loader import apply_resolved_assets + + runtime = { + "transport": "http", + "url": "https://mcp.example.com/tools", + "headers": {"X-Registry-Context": "x" * 6000}, + } + assets = [ + { + "kind": "mcp_server", + "namespace": "acme", + "name": "large", + "version": "1.0.0", + "runtime": runtime, + } + ] + transport["payload"]["resolved_assets"] = assets + # Isolate real config installation; no task thread or CloudWatch client runs. + monkeypatch.setattr(server.os, "environ", dict(server.os.environ)) + monkeypatch.setattr(server, "_debug_cw", lambda *args, **kwargs: None) + spawn = MagicMock() + monkeypatch.setattr(server, "_spawn_background", spawn) + with TestClient(server.app) as client: + response = client.post( + server.MICROVM_HOOK_PREFIX + "/run", + json={ + "microvmId": "mvm-registry", + "runHookPayload": json.dumps(transport["reference"]), + }, + ) + assert response.status_code == 200 + spawn.assert_called_once() + received = spawn.call_args.args[0]["resolved_assets"] + assert received == assets + assert apply_resolved_assets(str(tmp_path), received) == ["acme__large"] + saved = json.loads((tmp_path / ".mcp.json").read_text()) + assert saved["mcpServers"]["acme__large"] == { + "type": "http", + "url": runtime["url"], + "headers": runtime["headers"], + } + + @pytest.mark.parametrize("reference", [{}, {"agent_payload": {}}, {"version": 1}, [], "raw"]) def test_legacy_and_malformed_envelopes_never_read_or_start(reference, transport): with pytest.raises(ValueError, match="v2 is required"): diff --git a/agent/tests/test_server.py b/agent/tests/test_server.py index 691bb9a84..08d395896 100644 --- a/agent/tests/test_server.py +++ b/agent/tests/test_server.py @@ -492,36 +492,47 @@ def test_debug_cw_exc_appends_the_traceback(monkeypatch, capfd): assert "Traceback" in out -def test_debug_cw_write_blocking_bumps_failure_counter_on_boto_error(monkeypatch): - """On boto errors the failure counter increments so operators can alarm. - - AgentCore doesn't forward container stdout to APPLICATION_LOGS, so a - broken ``_debug_cw`` is invisible except for this counter. If the - counter ever stops bumping on error the blind-debug alarm breaks - silently. - """ - # Seed the counter to a known value so we can assert the delta without - # being sensitive to other tests. - with server._debug_cw_failures_lock: - server._debug_cw_failures = 0 - - # Stub ``boto3.client`` to raise so the except branch (which bumps - # the counter) runs. - class _BrokenBoto3: - @staticmethod - def client(*args, **kwargs): - raise RuntimeError("simulated boto failure") +@pytest.mark.parametrize("writer", ["debug", "warn"]) +@pytest.mark.parametrize("stage", ["client", "stream", "events"]) +def test_cloudwatch_failures_emit_structured_stdout_without_recursion( + writer, stage, monkeypatch, capfd +): + """The fallback survives a broken writer without exposing the failed log text.""" + import aws_session + + class StreamExists(Exception): + pass + + logs = MagicMock() + logs.exceptions.ResourceAlreadyExistsException = StreamExists + failure = RuntimeError("BEARER-SECRET in SDK error") + factory = MagicMock(return_value=logs) + if stage == "client": + factory.side_effect = failure + elif stage == "stream": + logs.create_log_stream.side_effect = failure + else: + logs.put_log_events.side_effect = failure + monkeypatch.setattr(aws_session, "platform_client", factory) - monkeypatch.setitem(__import__("sys").modules, "boto3", _BrokenBoto3) + def forbidden(*args, **kwargs): + pytest.fail("the fallback must not call a CloudWatch writer") - server._debug_cw_write_blocking( - log_group="/some/log-group", - task_id="t-1", - stamped="2026-01-01T00:00:00Z hello", + monkeypatch.setattr(server, "_debug_cw", forbidden) + monkeypatch.setattr(server, "_warn_cw", forbidden) + getattr(server, f"_{writer}_cw_write_blocking")( + log_group="/test/logs", task_id="task-log-failure", stamped="PRIVATE-TASK-PROMPT" ) - - with server._debug_cw_failures_lock: - assert server._debug_cw_failures == 1 + factory.assert_called_once() + output = capfd.readouterr().out + assert json.loads(output) == { + "event": "cloudwatch_write_failed", + "writer": writer, + "task_id": "task-log-failure", + "error_type": "RuntimeError", + } + assert "BEARER-SECRET" not in output + assert "PRIVATE-TASK-PROMPT" not in output # Chunk 7c — _warn_cw parallels _debug_cw so warn-level invocation-payload @@ -578,33 +589,6 @@ def start(self) -> None: ) -def test_warn_cw_write_blocking_bumps_failure_counter_on_boto_error(monkeypatch): - """Warn-path boto errors bump the same failure counter as debug. - - A single alarm surface is intentional (§server.py comment on - ``_debug_cw_failures``). If the counter ever stops bumping on a - warn write failure the blind-warn alarm breaks silently. - """ - with server._debug_cw_failures_lock: - server._debug_cw_failures = 0 - - class _BrokenBoto3: - @staticmethod - def client(*args, **kwargs): - raise RuntimeError("simulated boto failure") - - monkeypatch.setitem(__import__("sys").modules, "boto3", _BrokenBoto3) - - server._warn_cw_write_blocking( - log_group="/some/log-group", - task_id="t-1", - stamped="[server/warn] malformed payload", - ) - - with server._debug_cw_failures_lock: - assert server._debug_cw_failures == 1 - - def test_warn_cw_write_blocking_uses_server_warn_stream(monkeypatch): """Warn writes land in ``server_warn/``, not the debug stream. @@ -1163,10 +1147,9 @@ def exploding(argv, **kwargs): def test_the_warm_up_makes_no_aws_call_even_with_a_log_group_baked( self, client, monkeypatch, capfd, warm_ready ): - # /ready runs under the BUILD role: a Logs write can only fail (and each - # failure pollutes the shared _debug_cw_failures alarm), and any boto3 - # client built here freezes the build role's credential chain and the build - # region into the snapshot. Adding a subprocess must not have changed that. + # The BUILD role cannot write application logs outside its MicroVM log + # namespace. Initializing AWS clients here can preserve build credentials + # in the snapshot. Warm-up subprocesses must not change the stdout-only rule. monkeypatch.setenv("LOG_GROUP_NAME", "/abca/agent") def forbidden(*_args, **_kwargs): @@ -2514,10 +2497,9 @@ def forbidden(*_args, **_kwargs): def test_ready_is_also_aws_silent_with_a_log_group_configured( self, client, monkeypatch, capfd, warm_ready ): - # /ready runs under the same build role, so the same rule applies. It used - # to route through _debug_cw, whose write can only FAIL under a role with - # no Logs grant — and each failure bumps the shared _debug_cw_failures - # counter, poisoning the "debug path is blind" signal on every build. + # /ready shares the build role's namespace-scoped Logs permissions. + # Build diagnostics stay on stdout so no runtime logging client or + # build-role credential state is initialized before the snapshot. monkeypatch.setenv("LOG_GROUP_NAME", "/abca/agent") def forbidden(*_args, **_kwargs): diff --git a/cdk/test/handlers/orchestrate-task-microvm.test.ts b/cdk/test/handlers/orchestrate-task-microvm.test.ts index 86d4dd8aa..55dd88c8c 100644 --- a/cdk/test/handlers/orchestrate-task-microvm.test.ts +++ b/cdk/test/handlers/orchestrate-task-microvm.test.ts @@ -58,6 +58,21 @@ jest.mock('@aws-sdk/client-lambda-microvms', () => ({ // PUT and the finalize DELETE are both assertions this file needs to make, and a // per-instance mock silently discards them. const mockS3Send = jest.fn().mockResolvedValue({}); +const mockDdbSend = jest.fn().mockResolvedValue({}); +jest.mock('@aws-sdk/client-dynamodb', () => ({ DynamoDBClient: jest.fn(() => ({})) })); +jest.mock('@aws-sdk/lib-dynamodb', () => ({ + DynamoDBDocumentClient: { from: jest.fn(() => ({ send: mockDdbSend })) }, + GetCommand: jest.fn((input: unknown) => ({ _type: 'Get', input })), + PutCommand: jest.fn((input: unknown) => ({ _type: 'Put', input })), + UpdateCommand: jest.fn((input: unknown) => ({ _type: 'Update', input })), +})); +const mockResolveAsset = jest.fn(); +jest.mock('../../src/handlers/shared/registry/factory', () => ({ + makeRegistryClient: () => ({ resolve: mockResolveAsset }), +})); +jest.mock('../../src/handlers/shared/context-hydration', () => ({ + hydrateContext: jest.fn().mockResolvedValue({ sources: [], token_estimate: 0, truncated: false }), +})); jest.mock('@aws-sdk/s3-request-presigner', () => ({ getSignedUrl: async () => 'https://payloads.s3.us-east-1.amazonaws.com/task/payload.json?X-Amz-Signature=' + Date.now() })); const mockObjects = new Map(); jest.mock('@aws-sdk/client-s3', () => ({ @@ -86,6 +101,8 @@ const mockFailTask = jest.fn(); const mockLoadTask = jest.fn(); const mockClaimStart = jest.fn(); const mockSaveHandle = jest.fn(); +const mockHydrateAndTransition = jest.fn(); +const mockLoadBlueprint = jest.fn(); jest.mock('../../src/handlers/shared/microvm-start', () => ({ ...jest.requireActual('../../src/handlers/shared/microvm-start'), claimMicrovmStart: (...args: unknown[]) => mockClaimStart(...args), @@ -100,8 +117,8 @@ jest.mock('../../src/handlers/shared/orchestrator', () => ({ }), failTask: (...a: unknown[]) => mockFailTask(...a), finalizeTask: (...a: unknown[]) => mockFinalizeTask(...a), - hydrateAndTransition: jest.fn().mockResolvedValue({ repo_url: 'org/repo', task_id: 'TASK001' }), - loadBlueprintConfig: jest.fn().mockResolvedValue({ compute_type: 'lambda-microvm', runtime_arn: '' }), + hydrateAndTransition: (...a: unknown[]) => mockHydrateAndTransition(...a), + loadBlueprintConfig: (...a: unknown[]) => mockLoadBlueprint(...a), loadTask: (...args: unknown[]) => mockLoadTask(...args), pollTaskStatus: (...a: unknown[]) => mockPollTaskStatus(...a), reconcileMicrovmSubstrateState: (...a: unknown[]) => mockReconcile(...a), @@ -140,7 +157,7 @@ process.env.AGENT_SESSION_ROLE_ARN = 'arn:aws:iam::123456789012:role/AbcaAgentSe import { TaskStatus } from '../../src/constructs/task-status'; import { handler } from '../../src/handlers/orchestrate-task'; -import { LambdaMicrovmComputeStrategy } from '../../src/handlers/shared/strategies/lambda-microvm-strategy'; +import { LambdaMicrovmComputeStrategy, MICROVM_RUN_HOOK_PAYLOAD_LIMIT_BYTES } from '../../src/handlers/shared/strategies/lambda-microvm-strategy'; /** * Minimal stand-in for the durable-execution context: `step` runs its body @@ -240,6 +257,10 @@ function failedTransition() { beforeEach(() => { jest.clearAllMocks(); + mockDdbSend.mockReset().mockResolvedValue({}); + mockResolveAsset.mockReset(); + mockHydrateAndTransition.mockReset().mockResolvedValue({ repo_url: 'org/repo', task_id: 'TASK001' }); + mockLoadBlueprint.mockReset().mockResolvedValue({ compute_type: 'lambda-microvm', runtime_arn: '' }); mockTransitionTask.mockReset().mockResolvedValue(undefined); mockEmitTaskEvent.mockReset().mockResolvedValue(undefined); mockFinalizeTask.mockReset().mockResolvedValue(undefined); @@ -269,6 +290,45 @@ beforeEach(() => { }); describe('orchestrate-task for a lambda-microvm task', () => { + test('registry assets alone exceed the hook cap and survive real hydration and v2 S3 delivery (#818)', async () => { + const runtime = { + transport: 'http', + url: 'https://mcp.example.com/tools', + headers: { 'X-Registry-Context': 'x'.repeat(6000) }, + }; + const asset = { kind: 'mcp_server', namespace: 'acme', name: 'large', version: '1.0.0', runtime }; + mockResolveAsset.mockResolvedValue({ ...asset, warnings: [] }); + mockLoadBlueprint.mockResolvedValue({ + compute_type: 'lambda-microvm', + runtime_arn: '', + mcp_servers: ['registry://mcp_server/acme/large@1.0.0'], + }); + // Exercise the real registry resolution, payload assembly and audit writes; + // only external service boundaries are stubbed on this hydration path. + mockHydrateAndTransition.mockImplementationOnce(realOrchestrator.hydrateAndTransition); + runMicrovmOk(); + + await handler({ task_id: 'TASK001' }, fakeContext().ctx as never); + + expect(mockResolveAsset).toHaveBeenCalledTimes(1); + const upload = s3CommandsOfType('PutObject').find(c => c.input.Key === 'TASK001/payload.json'); + const document = JSON.parse(upload.input.Body); + expect(document.agent_payload.resolved_assets).toEqual([asset]); + const { resolved_assets: assets, ...basePayload } = document.agent_payload; + expect(Buffer.byteLength(JSON.stringify(basePayload))).toBeLessThan(MICROVM_RUN_HOOK_PAYLOAD_LIMIT_BYTES); + expect(Buffer.byteLength(JSON.stringify(assets))).toBeGreaterThan(MICROVM_RUN_HOOK_PAYLOAD_LIMIT_BYTES); + const run = commandsOfType('RunMicrovm'); + expect(run).toHaveLength(1); + const wire = run[0].input.runHookPayload; + expect(Buffer.byteLength(wire)).toBeLessThanOrEqual(MICROVM_RUN_HOOK_PAYLOAD_LIMIT_BYTES); + expect(JSON.parse(wire)).toMatchObject({ version: 2, task_id: 'TASK001', payload_url: expect.any(String) }); + expect(wire).not.toContain('X-Registry-Context'); + const audit = mockDdbSend.mock.calls.find(([c]) => c.input.ExpressionAttributeValues?.[':ra']); + expect(audit![0].input.ExpressionAttributeValues[':ra']).toEqual([ + { kind: 'mcp_server', id: 'acme/large', version: '1.0.0' }, + ]); + }); + test.each([2, 3])('cancellation on task read %s checks release before leaving the pipeline', async (cancelOnRead) => { let reads = 0; mockLoadTask.mockImplementation(async () => ({ diff --git a/docs/design/COMPUTE.md b/docs/design/COMPUTE.md index 45296944a..8e0ee760c 100644 --- a/docs/design/COMPUTE.md +++ b/docs/design/COMPUTE.md @@ -81,7 +81,7 @@ Lambda MicroVMs are an opt-in third backend, selected per repository with `compu Because a snapshot freezes its build-time environment, current deployment identifiers arrive through the v2 payload bootstrap. The coordinator publishes a non-secret deployment manifest and sends a single-object signed download URL for the task. The worker reads only its deployment's `bootstrap/*` with ambient credentials; other object reads and payload-bucket listing are explicitly denied. The downloaded task identity and configuration must match the authenticated manifest before configuration installation. The serialized reference fits the verified 4,096-byte hook limit; all payload sizes use S3. ECS shares this transport through `AGENT_PAYLOAD_REF`. [ADR-021 §3](../decisions/ADR-021-lambda-microvms-compute-backend.md#3-packaging-same-agent-image-source-new-build-path) defines the wire format, compatibility change and pending live gates. -Networking separates image build from execution: the build-only connector permits TCP 80 and 443 because the Dockerfile uses `apt-get`, while running MicroVMs retain 443-only egress through the platform VPC. Every launch explicitly passes the Lambda-managed `NO_INGRESS` connector; omission would select the service's public-ingress default. The P2 image declares and serves `/ready` and `/validate` at build time and `/run` and `/terminate` at runtime. `/suspend` and `/resume` remain disabled until their P3 implementation. +Networking separates image build from execution: the build-only connector permits TCP 80 and 443 because the Dockerfile uses `apt-get`, while running MicroVMs retain 443-only egress through the platform VPC. Every launch explicitly passes the Lambda-managed `NO_INGRESS` connector; omission would select the service's public-ingress default. The P2 image declares and serves `/ready` and `/validate` at build time and `/run` and `/terminate` at runtime. `/suspend` and `/resume` remain disabled until their P3 implementation. Registry HTTP/SSE tools therefore need reachable HTTPS/443 endpoints; remote non-443 tools are unsupported under the default policies of AgentCore and ECS as well. Local `stdio` tools can run, with their outbound traffic subject to the same restriction. Asset resolution/loading does not probe connectivity; see [registry network support](./REGISTRY.md#2-asset-kinds-for-mvp). ## ECS Fargate task sizing (build vs. planning) diff --git a/docs/design/REGISTRY.md b/docs/design/REGISTRY.md index a74c1e222..8136e424f 100644 --- a/docs/design/REGISTRY.md +++ b/docs/design/REGISTRY.md @@ -37,6 +37,12 @@ A **registry asset** is a versioned, immutable-per-version runtime artifact that **Cedar parity.** Registry Cedar text reaches the agent through the **same** `cedar_policies` payload field as inline blueprint policies, so it is byte-identical from the `PolicyEngine`'s view. **Skills** are prompt text only: a skill cannot invoke tools; its `tool_hints` are advisory prose referencing tools an MCP server separately provides (no transitive dependency — the operator attaches both). +**Runtime network support (#818).** An MCP server is a program the coding agent calls to use extra tools. Remote `http`/`sse` assets need an **HTTPS endpoint reachable on TCP port 443** under the shipped **AgentCore, ECS and Lambda MicroVM** network policies. Remote ports such as 80, 8080 or 8443 are unsupported on all three defaults; this is not a MicroVM-only limitation. A `stdio` asset starts a local program and uses its input/output pipes, but that program's outbound network requests still face the runtime policy. The MicroVM image builder's separate 80/443 permission does not apply to task execution. + +Registry resolution validates/pins the asset and the loader writes its configuration; neither probes network reachability or rejects an asset merely because its URL uses a different port. This is a documented support constraint, not a new synth/onboarding validator. Use a reachable HTTPS/443 service or proxy and verify DNS, routing, TLS and authentication on the deployed backend. Port 443 alone does not prove reachability. + +**Task delivery.** `resolved_assets` stays in the shared agent payload. ECS and MicroVM now deliver every payload through v2 authenticated bootstrap and a single-object signed URL. Registry content may exceed 4,096 bytes without expanding the MicroVM hook reference; the 8 MiB task-document cap still applies. Local integration coverage resolves a large MCP asset, runs the real payload assembly, verifies S3 bytes and the small hook reference, then separately exercises the Python hook mapper and local `.mcp.json` loader. That proves delivery/configuration, not a live connection to the remote tool. See [COMPUTE.md](./COMPUTE.md) for runtime behavior. + ## 3. Substrate mapping (the core design) Agent Registry answers *"what servers/skills exist, find me one"* (discovery metadata + semantic search). ABCA needs *"give me the exact runtime config to load this pinned asset."* These are two different objects, and the service validates the discovery object against the official schemas (an MCP record's body must be a valid MCP `server.json`, not our `.mcp.json`). So **every record carries BOTH a discovery descriptor AND ABCA's runtime payload.** diff --git a/docs/guides/DEPLOYMENT_GUIDE.md b/docs/guides/DEPLOYMENT_GUIDE.md index 4d6629650..fa6f6a7ec 100644 --- a/docs/guides/DEPLOYMENT_GUIDE.md +++ b/docs/guides/DEPLOYMENT_GUIDE.md @@ -55,7 +55,9 @@ Operational notes specific to this backend: - **Nothing self-terminates.** A MicroVM whose task finished, crashed, or hung stays `RUNNING` and billing until the 8-hour cap. The orchestrator calls `TerminateMicrovm` on finalize, and the heartbeat-staleness check detects loss of the in-guest heartbeat writer (a pipeline hang can leave that writer running) -- but a leaked handle is a cost incident. The one exception: the service reaps a VM whose `/run` hook returns 4xx (~12s). - **Logs** land in `/aws/lambda-microvms/`. Guest stdout goes there too, which is the fallback path when the agent cannot reach the application log group. -- **Deployment identifiers are not baked into the image.** The snapshot carries no configuration; table names, secret ARNs, and the per-task session-role ARN arrive in the `/run` payload as a `platform_config` block. A version-skewed orchestrator that does not send it is refused rather than run with tenant scoping disabled. +- **Deployment identifiers are not baked into the image.** Current table names, secret ARNs and session-role ARN arrive through the v2 IAM-authenticated manifest and signed task document. Old unsigned envelopes are refused. Deploy matching coordinator code, worker images and IAM with admissions paused and old tasks drained; the repository runbook `docs/verification/645-payload-bootstrap.md` records the procedure and pending live checks. +- **Registry tools share the runtime network restriction.** Remote HTTP/SSE MCP assets need reachable HTTPS/443 endpoints. AgentCore and ECS defaults also block remote non-443 ports. `stdio` programs run locally but their outbound calls remain restricted; resolution does not test connectivity. The image builder's 80/443 access does not widen runtime egress. See [REGISTRY.md](../design/REGISTRY.md). +- **Logging failures have a fallback record.** Debug/warn CloudWatch failures emit `cloudwatch_write_failed` to stdout with writer/task/error class, without another AWS call or sensitive log text. This is not a configured metric/alarm; verify collection in the guest log stream while the VM is running. ### Optional Agent Registry diff --git a/docs/src/content/docs/architecture/Compute.md b/docs/src/content/docs/architecture/Compute.md index 00a81d9c4..f8630678f 100644 --- a/docs/src/content/docs/architecture/Compute.md +++ b/docs/src/content/docs/architecture/Compute.md @@ -85,7 +85,7 @@ Lambda MicroVMs are an opt-in third backend, selected per repository with `compu Because a snapshot freezes its build-time environment, current deployment identifiers arrive through the v2 payload bootstrap. The coordinator publishes a non-secret deployment manifest and sends a single-object signed download URL for the task. The worker reads only its deployment's `bootstrap/*` with ambient credentials; other object reads and payload-bucket listing are explicitly denied. The downloaded task identity and configuration must match the authenticated manifest before configuration installation. The serialized reference fits the verified 4,096-byte hook limit; all payload sizes use S3. ECS shares this transport through `AGENT_PAYLOAD_REF`. [ADR-021 §3](/sample-autonomous-cloud-coding-agents/architecture/adr-021-lambda-microvms-compute-backend#3-packaging-same-agent-image-source-new-build-path) defines the wire format, compatibility change and pending live gates. -Networking separates image build from execution: the build-only connector permits TCP 80 and 443 because the Dockerfile uses `apt-get`, while running MicroVMs retain 443-only egress through the platform VPC. Every launch explicitly passes the Lambda-managed `NO_INGRESS` connector; omission would select the service's public-ingress default. The P2 image declares and serves `/ready` and `/validate` at build time and `/run` and `/terminate` at runtime. `/suspend` and `/resume` remain disabled until their P3 implementation. +Networking separates image build from execution: the build-only connector permits TCP 80 and 443 because the Dockerfile uses `apt-get`, while running MicroVMs retain 443-only egress through the platform VPC. Every launch explicitly passes the Lambda-managed `NO_INGRESS` connector; omission would select the service's public-ingress default. The P2 image declares and serves `/ready` and `/validate` at build time and `/run` and `/terminate` at runtime. `/suspend` and `/resume` remain disabled until their P3 implementation. Registry HTTP/SSE tools therefore need reachable HTTPS/443 endpoints; remote non-443 tools are unsupported under the default policies of AgentCore and ECS as well. Local `stdio` tools can run, with their outbound traffic subject to the same restriction. Asset resolution/loading does not probe connectivity; see [registry network support](/sample-autonomous-cloud-coding-agents/architecture/registry#2-asset-kinds-for-mvp). ## ECS Fargate task sizing (build vs. planning) diff --git a/docs/src/content/docs/architecture/Registry.md b/docs/src/content/docs/architecture/Registry.md index e6e2457ac..1d0ac94fd 100644 --- a/docs/src/content/docs/architecture/Registry.md +++ b/docs/src/content/docs/architecture/Registry.md @@ -41,6 +41,12 @@ A **registry asset** is a versioned, immutable-per-version runtime artifact that **Cedar parity.** Registry Cedar text reaches the agent through the **same** `cedar_policies` payload field as inline blueprint policies, so it is byte-identical from the `PolicyEngine`'s view. **Skills** are prompt text only: a skill cannot invoke tools; its `tool_hints` are advisory prose referencing tools an MCP server separately provides (no transitive dependency — the operator attaches both). +**Runtime network support (#818).** An MCP server is a program the coding agent calls to use extra tools. Remote `http`/`sse` assets need an **HTTPS endpoint reachable on TCP port 443** under the shipped **AgentCore, ECS and Lambda MicroVM** network policies. Remote ports such as 80, 8080 or 8443 are unsupported on all three defaults; this is not a MicroVM-only limitation. A `stdio` asset starts a local program and uses its input/output pipes, but that program's outbound network requests still face the runtime policy. The MicroVM image builder's separate 80/443 permission does not apply to task execution. + +Registry resolution validates/pins the asset and the loader writes its configuration; neither probes network reachability or rejects an asset merely because its URL uses a different port. This is a documented support constraint, not a new synth/onboarding validator. Use a reachable HTTPS/443 service or proxy and verify DNS, routing, TLS and authentication on the deployed backend. Port 443 alone does not prove reachability. + +**Task delivery.** `resolved_assets` stays in the shared agent payload. ECS and MicroVM now deliver every payload through v2 authenticated bootstrap and a single-object signed URL. Registry content may exceed 4,096 bytes without expanding the MicroVM hook reference; the 8 MiB task-document cap still applies. Local integration coverage resolves a large MCP asset, runs the real payload assembly, verifies S3 bytes and the small hook reference, then separately exercises the Python hook mapper and local `.mcp.json` loader. That proves delivery/configuration, not a live connection to the remote tool. See [COMPUTE.md](/sample-autonomous-cloud-coding-agents/architecture/compute) for runtime behavior. + ## 3. Substrate mapping (the core design) Agent Registry answers *"what servers/skills exist, find me one"* (discovery metadata + semantic search). ABCA needs *"give me the exact runtime config to load this pinned asset."* These are two different objects, and the service validates the discovery object against the official schemas (an MCP record's body must be a valid MCP `server.json`, not our `.mcp.json`). So **every record carries BOTH a discovery descriptor AND ABCA's runtime payload.** diff --git a/docs/src/content/docs/getting-started/Deployment-guide.md b/docs/src/content/docs/getting-started/Deployment-guide.md index 50826dd95..b7ef74e91 100644 --- a/docs/src/content/docs/getting-started/Deployment-guide.md +++ b/docs/src/content/docs/getting-started/Deployment-guide.md @@ -59,7 +59,9 @@ Operational notes specific to this backend: - **Nothing self-terminates.** A MicroVM whose task finished, crashed, or hung stays `RUNNING` and billing until the 8-hour cap. The orchestrator calls `TerminateMicrovm` on finalize, and the heartbeat-staleness check detects loss of the in-guest heartbeat writer (a pipeline hang can leave that writer running) -- but a leaked handle is a cost incident. The one exception: the service reaps a VM whose `/run` hook returns 4xx (~12s). - **Logs** land in `/aws/lambda-microvms/`. Guest stdout goes there too, which is the fallback path when the agent cannot reach the application log group. -- **Deployment identifiers are not baked into the image.** The snapshot carries no configuration; table names, secret ARNs, and the per-task session-role ARN arrive in the `/run` payload as a `platform_config` block. A version-skewed orchestrator that does not send it is refused rather than run with tenant scoping disabled. +- **Deployment identifiers are not baked into the image.** Current table names, secret ARNs and session-role ARN arrive through the v2 IAM-authenticated manifest and signed task document. Old unsigned envelopes are refused. Deploy matching coordinator code, worker images and IAM with admissions paused and old tasks drained; the repository runbook `docs/verification/645-payload-bootstrap.md` records the procedure and pending live checks. +- **Registry tools share the runtime network restriction.** Remote HTTP/SSE MCP assets need reachable HTTPS/443 endpoints. AgentCore and ECS defaults also block remote non-443 ports. `stdio` programs run locally but their outbound calls remain restricted; resolution does not test connectivity. The image builder's 80/443 access does not widen runtime egress. See [REGISTRY.md](/sample-autonomous-cloud-coding-agents/architecture/registry). +- **Logging failures have a fallback record.** Debug/warn CloudWatch failures emit `cloudwatch_write_failed` to stdout with writer/task/error class, without another AWS call or sensitive log text. This is not a configured metric/alarm; verify collection in the guest log stream while the VM is running. ### Optional Agent Registry diff --git a/docs/verification/645-p3-implementation-plan.md b/docs/verification/645-p3-implementation-plan.md index 2695fe29f..4afb136a8 100644 --- a/docs/verification/645-p3-implementation-plan.md +++ b/docs/verification/645-p3-implementation-plan.md @@ -21,7 +21,7 @@ Prerequisite work is tracked here on `fix/645-microvm-readiness`. “Completed - [ ] Verify the capacity protocol's upgrade/drain procedure, deployed IAM and scan scale in AWS. - [x] Restrict agent task updates to reporting fields; remove replacement/deletion and worker counter grants. - [ ] Verify metadata restrictions with real AWS sessions/transactions; retain status/tag trust limits. -- [ ] Finish logging-failure observability (#810) and registry-overflow coverage (#818). +- [x] Replace unused logging-failure bookkeeping with structured stdout diagnostics (#810); document shared runtime networking and verify large registry payload delivery (#818). - [ ] Implement production nesting if included, then verify a clean P2 deployment. - [ ] Implement and verify the P3 sleep/wake lifecycle described below. @@ -74,13 +74,20 @@ Fifth prerequisite batch completed locally on 2026-09-13: - The old-policy regression failed on `PutItem`. **1,795 Python tests** pass with **84.47%** coverage, including actual writer-request/permission-contract checks; **268 CDK tests** pass across session-role, ECS, MicroVM and full-stack suites. Python quality, CDK lint/compilation and documentation checks pass. These counts overlap earlier batches; four obsolete helper tests were removed and two contract tests added. - [Metadata verification](./645-coordinator-metadata.md) records the effective source-policy boundary, writer inventory, rollback constraints and pending real-AWS allowed/denied transaction matrix. No table migration or bootstrap-policy update is required. Application-role deployment and a matching agent image remain necessary; nothing was deployed. -Sixth prerequisite batch completed locally (2026-09-13): +Sixth prerequisite batch completed locally (2026-09-13), commit `82a9df80`: - v2 payload bootstrap for #817/#700 now covers both ECS and MicroVM: IAM-authenticated deployment manifests, one-object signed downloads, immutable private retry references, bounded bytes/expiry and coordinator cleanup of both task objects. Worker reads outside the manifest prefix and payload-bucket listing are explicitly denied. - Regressions reproduced and fixed bearer-URL leaks through Python and JavaScript exception chains. Removed the old unsigned transports, stale permission/compatibility comments and unused per-backend key helpers. - Python quality passed **1,806 tests / 84.51% coverage**. The broad CDK run passed **159 suites / 3,935 tests** with `--detectOpenHandles` and exited normally; **15 existing DynamoDB Local tests were skipped** because this batch did not start that service or change its transaction protocol. The final five transport/strategy suites passed **187 tests** after the last cleanup (overlapping the broad run); CDK lint/compilation, Python lint/type checks, constants-sync, the **77-page** docs build and link checks also pass. - The [bootstrap runbook](./645-payload-bootstrap.md) records the design, AWS documentation evidence, remaining trust boundaries, live allowed/denied matrix and coordinated deployment/rollback procedure. Nothing was deployed; effective IAM, conditional S3 writes, expiry, DNS/HTTPS, ingress negatives and clean launches remain gates. +Seventh prerequisite batch completed locally (2026-09-13): + +- #810: removed the unread CloudWatch failure counter, lock and unused threshold. Debug/warn failures emit structured stdout records with writer, task ID and exception class; no AWS retry or failed-message contents. Six client/stream/event failure cases reproduced the old unstructured output and now pass. There is no metric or configured alarm; stdout collection remains a live gate. +- #818: documented that remote non-443 endpoints are unsupported under all three shipped runtime policies, with no new connectivity validator or port grants. A real-hydration test resolves a large MCP asset, checks its durable audit record and v2 S3 bytes, and verifies that the Run reference stays within 4,096 bytes. Python separately verifies the real download, hook mapping and `.mcp.json` loader. Live remote-tool connectivity is still unproven. +- Full agent quality passed **1,811 tests / 84.51% coverage**. CDK lint/compilation and **38 tests** in the two relevant orchestrator/registry suites passed. These counts overlap previous runs; no CDK runtime code or IAM changed in this batch. +- Source/generated registry/compute/deployment documentation, the **77-page** build and link checks pass. An offline coordinator bundle check includes the S3 client, presigner and shared constants; it does not replace a full deployed packaging check. Nothing was deployed or posted to the issue trackers. Package 3's clean P2 rerun and package 7's live P3 matrix remain required. + ## The result we want When the coding agent asks a human for permission, its computer may go to sleep. The human can approve or deny while it sleeps. The computer wakes, reads the saved answer, and continues or refuses the action. If nobody answers, it wakes before the deadline and applies the existing timeout-as-denial rule. Its files, task identity, permissions, progress and deadline remain correct. @@ -181,7 +188,9 @@ For an unknown outcome with no returned ID, the task error or cancellation event ### 1E. Make verification observable -Resolve #810 by exposing a useful structured failure signal for CloudWatch writers or removing the dead counter and using another observable signal. Test failure of the logging system itself. For #818, document the shared runtime 443-only rule and test registry payload overflow; do not expand network ports just to satisfy an incorrect issue premise. +**Completed locally:** #810 uses the removal option. Both CloudWatch writers emit `cloudwatch_write_failed` to stdout with `writer`, `task_id` and `error_type`; the unused counter and threshold are gone. Six injected failures cover client creation, stream creation and event submission without recursive logging or sensitive error contents. This is a structured log, not an alarm or metric. Verify guest stdout ingestion while the VM is running; AgentCore APPLICATION_LOGS does not automatically collect it. + +For #818, documentation explicitly limits remote MCP endpoints to reachable HTTPS/443 on all shipped backends. No synth/onboarding connectivity probe is added, and no ports are widened. Registry-specific integration tests exercise real resolution/hydration, audit writes, >4,096-byte asset data in S3, the bounded v2 reference, Python hook mapping and the real local loader. The retired inline branch is no longer the acceptance target. Successful config delivery does not prove remote-tool connectivity; test DNS/routing/TLS/auth live. ### 1F. Make concurrency release safe across replay diff --git a/docs/verification/645-p3-readiness-review.md b/docs/verification/645-p3-readiness-review.md index 341d3adc2..5112c228a 100644 --- a/docs/verification/645-p3-readiness-review.md +++ b/docs/verification/645-p3-readiness-review.md @@ -4,7 +4,7 @@ Reviewed 2026-09-13 against `main` commit `5e10038c7e28179b302ac4de78b709795aeba This review covers the existing MicroVM implementation, related P2 follow-ups, comment accuracy and nested-stack feasibility. It includes local experiments, not an AWS deployment or a new live smoke run. The associated [implementation plan](./645-p3-implementation-plan.md) turns the findings into ordered work. -**Implementation update (2026-09-13):** subsequent local batches fix thread isolation, deletion/error/byte/contract bugs, approval heartbeat, stable MicroVM start recovery, atomic capacity reservations and coordinator metadata permissions. The latest batch implements v2 authenticated deployment manifests and single-object payload links for both ECS and MicroVM (#817/#700), with no old unsigned fallback. See the [bootstrap runbook](./645-payload-bootstrap.md) and [implementation progress](./645-p3-implementation-plan.md#implementation-progress). Effective AWS policies, expiry/networking, clean deployment, logging/registry follow-ups and P3 sleep/wake remain open. Findings below preserve the original reviewed baseline, rather than describing all of them as current defects. +**Implementation update (2026-09-13):** subsequent local batches fix thread isolation, deletion/error/byte/contract bugs, approval heartbeat, stable MicroVM start recovery, atomic capacity reservations and coordinator metadata permissions. The latest batch implements v2 authenticated deployment manifests and single-object payload links for both ECS and MicroVM (#817/#700), with no old unsigned fallback. See the [bootstrap runbook](./645-payload-bootstrap.md) and [implementation progress](./645-p3-implementation-plan.md#implementation-progress). A further local batch removes unused logging counters in favor of structured stdout failures and verifies large registry assets through v2 delivery and the local loader. Effective AWS policies, expiry/networking, stdout ingestion, remote-tool connectivity, clean deployment and P3 sleep/wake remain open. Findings below preserve the original reviewed baseline, rather than describing all of them as current defects. Further local work adds durable MicroVM start receipts, stable request tokens and bounded handle recovery. AWS token-retention/conflict behavior and unknown-ID cleanup remain live verification gates. The shared finalizer's repeated-decrement risk was subsequently reproduced and fixed locally with task-owned reservation transactions, unified counter writers and revision-guarded reconciliation, including approval waits. DynamoDB Local exercises the real transaction conditions; deployed IAM, scan scale and upgrade/drain behavior still need AWS verification. diff --git a/docs/verification/645-payload-bootstrap.md b/docs/verification/645-payload-bootstrap.md index 34018e577..e8e0cc4c2 100644 --- a/docs/verification/645-payload-bootstrap.md +++ b/docs/verification/645-payload-bootstrap.md @@ -87,7 +87,7 @@ A regression reproduced a signed-URL leak through a chained Python traceback. Ro Permission tests inspect synthesized IAM policies, including the `NotResource` explicit deny and the coordinator's missing-object permission. The constants-checker subprocess rejects invalid bounds/paths and consumers that stop importing the contract. -An offline compatibility probe generated a URL using the actual JavaScript AWS SDK with dummy credentials and passed it through the Python URL parser. That proves the observed SDK URL shape is supported; it makes no AWS authorization claim. +An offline compatibility probe generated a URL using the actual JavaScript AWS SDK with dummy credentials and passed it through the Python URL parser. That proves the observed SDK URL shape is supported; it makes no AWS authorization claim. An offline esbuild check using the coordinator's external-module settings also includes the matching S3 client, presigner and constants in the JavaScript bundle. It excludes the separately packaged `pdf-parse` dependency and does not replace full CDK packaging or deployment verification. See the [implementation checklist](./645-p3-implementation-plan.md#implementation-progress) for final suite counts. Mock authorization failures are not proof of effective AWS IAM. Conditional S3 writes and the missing-object behavior must also be checked live. @@ -119,7 +119,7 @@ For **both ECS and MicroVM**, retain source/image identifiers, relevant policy s | Same-account other-workspace secret substitution | Reject; cross-region secret explicitly present in the trusted manifest still works. | | Two preparations / lost committed S3 reply | Same saved reference or explicit conflict, no overwrite of task instructions. | | Lost MicroVM Run reply / coordinator restart | Same client token and exact request; existing VM recovered, no replacement. Record actual AWS retention/conflict behavior. | -| Typical/large registry-hydrated task | Manifest and exact S3 download work over runtime DNS/HTTPS; hook/override limits hold. #818 still needs its registry-resolution end-to-end case. | +| Typical/large registry-hydrated task | Manifest and exact S3 download work over runtime DNS/HTTPS; hook/override limits hold. #818 now has local resolution/hydration/storage and hook-to-loader cases; live tool connectivity remains required. | | Success, failure, cancellation | Expected terminal status; compute terminated; both private task objects deleted or cleanup failure visibly recorded. | | Attempted public MicroVM ingress | No usable public control path with explicit `NO_INGRESS`. Test while the VM is running. | | Environment/log inspection | Repository subprocesses do not inherit `AGENT_PAYLOAD_REF`; normal/error logs contain no signed URL. | From f3e684d462aa50e4f688eebd1dd3a5c5e969f4ac Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Sun, 13 Sep 2026 19:43:33 -0400 Subject: [PATCH 017/149] feat(compute): add MicroVM lifecycle foundations (#645) --- agent/src/hooks.py | 50 +++- agent/tests/test_hooks.py | 233 +++++++++++++++++- agent/tests/test_poll_for_decision.py | 20 +- cdk/src/handlers/shared/compute-strategy.ts | 11 + .../shared/strategies/agentcore-strategy.ts | 12 +- .../shared/strategies/ecs-strategy.ts | 12 +- .../strategies/lambda-microvm-strategy.ts | 51 +++- .../handlers/shared/session-lifecycle.test.ts | 135 ++++++++++ ...ADR-021-lambda-microvms-compute-backend.md | 6 +- docs/design/CEDAR_HITL_GATES.md | 40 ++- docs/design/COMPUTE.md | 2 + .../docs/architecture/Cedar-hitl-gates.md | 40 ++- docs/src/content/docs/architecture/Compute.md | 2 + ...Adr-021-lambda-microvms-compute-backend.md | 6 +- .../645-p3-implementation-plan.md | 24 +- docs/verification/645-p3-readiness-review.md | 2 + 16 files changed, 559 insertions(+), 87 deletions(-) create mode 100644 cdk/test/handlers/shared/session-lifecycle.test.ts diff --git a/agent/src/hooks.py b/agent/src/hooks.py index 99a1252cd..6db262f75 100644 --- a/agent/src/hooks.py +++ b/agent/src/hooks.py @@ -24,7 +24,8 @@ import re import time from collections.abc import Callable -from datetime import UTC +from dataclasses import dataclass +from datetime import UTC, datetime from typing import TYPE_CHECKING, Any import nudge_reader @@ -57,6 +58,36 @@ TOOL_INPUT_PREVIEW_MAX: int = 256 # cap for the strip-ANSI, truncated input preview ELLIPSIS_LEN: int = 3 # chars reserved for the "..." truncation marker + +@dataclass(frozen=True) +class _ApprovalDeadline: + """One gate's deadline, retained across polling and a possible VM sleep. + + UTC counts time while the guest's monotonic clock is frozen. The monotonic + cap prevents a backward UTC correction from extending the original window. + Create this with the approval row, before database writes or notifications; + never recreate it on resume. + """ + + wall_deadline: float + monotonic_deadline: float + + @classmethod + def from_recorded(cls, created_at: str, timeout_s: int) -> _ApprovalDeadline: + created_epoch = ( + datetime.strptime(created_at, "%Y-%m-%dT%H:%M:%SZ").replace(tzinfo=UTC).timestamp() + ) + wall_deadline = created_epoch + timeout_s + remaining = max(0.0, min(timeout_s, wall_deadline - time.time())) + return cls(wall_deadline, time.monotonic() + remaining) + + def remaining_s(self) -> float: + return max( + 0.0, + min(self.monotonic_deadline - time.monotonic(), self.wall_deadline - time.time()), + ) + + # ANSI CSI / OSC escape sequence stripper for ``tool_input_preview`` + # ``permissionDecisionReason`` fields, so persisted/logged reasons can't carry # terminal-escape injection. Kept local to avoid a cross-module dependency for @@ -627,6 +658,7 @@ async def _handle_require_approval( "user_id": user_id or "", "repo": engine.repo, } + deadline = _ApprovalDeadline.from_recorded(row["created_at"], effective_timeout) # Step 6 — bump counters BEFORE the write so cap/rate checks on # subsequent gates reflect the attempt even if the DDB write itself @@ -700,7 +732,7 @@ async def _handle_require_approval( outcome = await _poll_for_decision( task_id=task_id, request_id=request_id, - timeout_s=effective_timeout, + deadline=deadline, progress=progress, ts=ts, ) @@ -937,7 +969,7 @@ async def _poll_for_decision( *, task_id: str, request_id: str, - timeout_s: int, + deadline: _ApprovalDeadline, progress: Any, ts: Any, ) -> dict: @@ -949,16 +981,16 @@ async def _poll_for_decision( ``approval_poll_degraded``; at ``POLL_MAX_CONSECUTIVE_FAILS`` we fall through as TIMED_OUT with a distinct reason. - Returns an outcome dict mirroring the approval row's terminal fields. + Uses the deadline captured with the original approval row, including time + spent writing/notifying or suspended. Returns an outcome dict mirroring + the approval row's terminal fields. """ - deadline = time.monotonic() + timeout_s start = time.monotonic() consecutive_fails = 0 degraded_emitted = False while True: - now = time.monotonic() - if now >= deadline: + if deadline.remaining_s() <= 0: return {"status": "TIMED_OUT", "reason": None} try: @@ -1012,7 +1044,7 @@ async def _poll_for_decision( elapsed = time.monotonic() - start interval = POLL_FAST_INTERVAL_S if elapsed < POLL_FAST_DURATION_S else POLL_SLOW_INTERVAL_S # Clamp sleep against remaining deadline so we don't oversleep. - sleep_for = min(interval, max(0.0, deadline - time.monotonic())) + sleep_for = min(interval, deadline.remaining_s()) if sleep_for <= 0: return {"status": "TIMED_OUT", "reason": None} await asyncio.sleep(sleep_for) @@ -1096,8 +1128,6 @@ def _remaining_maxlifetime_s() -> int | None: if started_at.isdigit(): started_epoch = int(started_at) else: - from datetime import datetime - # The trailing Z means UTC; strptime returns a naive datetime whose # .timestamp() would otherwise be interpreted in the container's # local TZ, skewing remaining-lifetime math by the UTC offset. diff --git a/agent/tests/test_hooks.py b/agent/tests/test_hooks.py index f78f0588e..29627b327 100644 --- a/agent/tests/test_hooks.py +++ b/agent/tests/test_hooks.py @@ -1,6 +1,7 @@ """Unit tests for hooks.py — Cedar policy SDK hook callbacks.""" import asyncio +from datetime import UTC, datetime from unittest.mock import MagicMock, patch import pytest @@ -755,7 +756,6 @@ def test_matchers_with_trajectory(self): import hashlib import json as _json from collections import deque -from datetime import UTC from typing import Any import hooks @@ -971,10 +971,9 @@ def _fast_poll(monkeypatch): """Collapse poll intervals so tests run instantly. Swaps ``asyncio.sleep`` for a no-op AND advances ``hooks.time.monotonic`` - by the requested sleep duration each call so the poll's wall-clock - deadline actually trips. Without the monotonic advance the poll spins - forever when the script runs out of rows (deque empty → default - PENDING row → never terminal). + by the requested sleep duration each call so the monotonic deadline + trips without waiting for real UTC time to pass. When scripted rows + run out, the fake returns PENDING until that deadline. """ fake_clock = {"now": 0.0} @@ -988,6 +987,230 @@ async def _zero_sleep(seconds): monkeypatch.setattr(hooks.asyncio, "sleep", _zero_sleep) +@pytest.fixture() +def approval_clock(monkeypatch): + """Control elapsed and UTC time independently, including a frozen guest clock.""" + clock = {"wall": 1_800_000_000.0, "monotonic": 100.0} + monkeypatch.setattr(hooks.time, "time", lambda: clock["wall"]) + monkeypatch.setattr(hooks.time, "monotonic", lambda: clock["monotonic"]) + monkeypatch.setattr( + hooks, + "_iso_now", + lambda: datetime.fromtimestamp(clock["wall"], UTC).strftime("%Y-%m-%dT%H:%M:%SZ"), + ) + return clock + + +class TestApprovalDeadline: + def test_frozen_monotonic_clock_still_expires_after_sleep( + self, fake_task_state, progress, engine_with_soft_gate, monkeypatch, approval_clock + ): + engine_with_soft_gate._task_default_timeout_s = 30 + sleeps = [] + + async def frozen_sleep(seconds): + sleeps.append(seconds) + assert len(sleeps) == 1, "The expired gate must not restart polling after waking" + approval_clock["wall"] += 600 + + monkeypatch.setattr(hooks.asyncio, "sleep", frozen_sleep) + result = _run( + pre_tool_use_hook( + _hook_input(), + "tu-1", + {}, + engine=engine_with_soft_gate, + task_id="01KTASK", + progress=progress, + task_state_module=fake_task_state, + ) + ) + + assert result["hookSpecificOutput"]["permissionDecision"] == "deny" + assert fake_task_state.update_calls[-1][2] == "TIMED_OUT" + assert len(fake_task_state.get_calls) == 1 + + @pytest.mark.parametrize("wall_change", [12, -120]) + def test_database_write_time_counts_even_if_wall_clock_moves_back( + self, + fake_task_state, + progress, + engine_with_soft_gate, + monkeypatch, + approval_clock, + wall_change, + ): + engine_with_soft_gate._task_default_timeout_s = 30 + original_write = fake_task_state.transact_write_approval_request + sleeps = [] + + def delayed_write(*args, **kwargs): + original_write(*args, **kwargs) + approval_clock["wall"] += wall_change + approval_clock["monotonic"] += 12 + + async def advance(seconds): + sleeps.append(seconds) + approval_clock["wall"] += seconds + approval_clock["monotonic"] += seconds + + monkeypatch.setattr(fake_task_state, "transact_write_approval_request", delayed_write) + monkeypatch.setattr(hooks.asyncio, "sleep", advance) + result = _run( + pre_tool_use_hook( + _hook_input(), + "tu-1", + {}, + engine=engine_with_soft_gate, + task_id="01KTASK", + progress=progress, + task_state_module=fake_task_state, + ) + ) + + assert result["hookSpecificOutput"]["permissionDecision"] == "deny" + assert sum(sleeps) == 18 + assert approval_clock["monotonic"] == 130 + + @pytest.mark.parametrize("wall_jump,expected_wait", [(27, 3), (-3600, 30)]) + def test_clock_changes_clamp_next_sleep_and_never_extend_window( + self, + fake_task_state, + progress, + engine_with_soft_gate, + monkeypatch, + approval_clock, + wall_jump, + expected_wait, + ): + engine_with_soft_gate._task_default_timeout_s = 30 + sleeps = [] + + async def advance(seconds): + sleeps.append(seconds) + approval_clock["monotonic"] += seconds + approval_clock["wall"] += seconds + (wall_jump if len(sleeps) == 1 else 0) + + monkeypatch.setattr(hooks.asyncio, "sleep", advance) + result = _run( + pre_tool_use_hook( + _hook_input(), + "tu-1", + {}, + engine=engine_with_soft_gate, + task_id="01KTASK", + progress=progress, + task_state_module=fake_task_state, + ) + ) + + assert result["hookSpecificOutput"]["permissionDecision"] == "deny" + assert sum(sleeps) == expected_wait + if wall_jump > 0: + assert sleeps == [2, 1] + + @pytest.mark.parametrize( + "reread_row,cancelled,expected", + [ + ({"status": "APPROVED", "scope": "this_call"}, False, "allow"), + ({"status": "DENIED", "deny_reason": "no"}, False, "deny"), + ({"status": "PENDING"}, False, "deny"), + (None, False, "deny"), + ({"status": "APPROVED", "scope": "this_call"}, True, "deny"), + ], + ) + def test_waking_after_deadline_preserves_decision_race_and_cancellation( + self, + fake_task_state, + progress, + engine_with_soft_gate, + monkeypatch, + approval_clock, + reread_row, + cancelled, + expected, + ): + engine_with_soft_gate._task_default_timeout_s = 30 + fake_task_state.best_effort_return = False + fake_task_state.reread_row = reread_row + if reread_row is None: + # A missing/TTL-reaped row during the poll also stays fail-closed. + fake_task_state.get_row_script.append(None) + if cancelled: + fake_task_state.resume_raises = _FakeApprovalResumeError("cancelled") + + async def frozen_sleep(_seconds): + approval_clock["wall"] += 600 + + monkeypatch.setattr(hooks.asyncio, "sleep", frozen_sleep) + result = _run( + pre_tool_use_hook( + _hook_input(), + "tu-1", + {}, + engine=engine_with_soft_gate, + task_id="01KTASK", + progress=progress, + task_state_module=fake_task_state, + ) + ) + + assert result["hookSpecificOutput"]["permissionDecision"] == expected + assert fake_task_state.update_calls[-1][2] == "TIMED_OUT" + assert len(fake_task_state.get_calls) == 2 + assert all(consistent for _, _, consistent in fake_task_state.get_calls) + if cancelled: + assert "write_approval_granted" not in progress.milestones() + + def test_decision_committed_during_slow_read_is_honored( + self, fake_task_state, progress, engine_with_soft_gate, monkeypatch, approval_clock + ): + engine_with_soft_gate._task_default_timeout_s = 30 + + def slow_read(*_args, **kwargs): + assert kwargs["consistent_read"] + approval_clock["wall"] += 600 + return {"status": "APPROVED", "scope": "this_call"} + + monkeypatch.setattr(fake_task_state, "get_approval_row", slow_read) + result = _run( + pre_tool_use_hook( + _hook_input(), + "tu-1", + {}, + engine=engine_with_soft_gate, + task_id="01KTASK", + progress=progress, + task_state_module=fake_task_state, + ) + ) + + assert result["hookSpecificOutput"]["permissionDecision"] == "allow" + assert not fake_task_state.update_calls + + def test_coroutine_cancellation_does_not_become_a_timeout_or_allow( + self, fake_task_state, progress, engine_with_soft_gate, monkeypatch, approval_clock + ): + async def cancelled_sleep(_seconds): + raise asyncio.CancelledError + + monkeypatch.setattr(hooks.asyncio, "sleep", cancelled_sleep) + with pytest.raises(asyncio.CancelledError): + _run( + pre_tool_use_hook( + _hook_input(), + "tu-1", + {}, + engine=engine_with_soft_gate, + task_id="01KTASK", + progress=progress, + task_state_module=fake_task_state, + ) + ) + assert not fake_task_state.update_calls + assert not fake_task_state.resume_calls + + # --- Happy paths ---------------------------------------------------------- diff --git a/agent/tests/test_poll_for_decision.py b/agent/tests/test_poll_for_decision.py index 67739b708..d0c45a818 100644 --- a/agent/tests/test_poll_for_decision.py +++ b/agent/tests/test_poll_for_decision.py @@ -28,8 +28,7 @@ def test_n_consecutive_failures_returns_timed_out_with_reason(self, monkeypatch) without further polling. """ # Tiny intervals so the loop iterates fast but doesn't immediately - # bail via the ``sleep_for <= 0`` early-return on line 723 of - # hooks.py. + # bail via the ``sleep_for <= 0`` early return. monkeypatch.setattr(hooks, "POLL_FAST_INTERVAL_S", 0.001) monkeypatch.setattr(hooks, "POLL_FAST_DURATION_S", 0.001) monkeypatch.setattr(hooks, "POLL_SLOW_INTERVAL_S", 0.001) @@ -43,7 +42,7 @@ def test_n_consecutive_failures_returns_timed_out_with_reason(self, monkeypatch) hooks._poll_for_decision( task_id="01KTASK", request_id="01KREQ", - timeout_s=300, # large enough that the deadline doesn't fire first + deadline=hooks._ApprovalDeadline.from_recorded(hooks._iso_now(), 300), progress=progress, ts=ts, ) @@ -59,8 +58,7 @@ def test_degraded_emitted_once_at_threshold(self, monkeypatch): on every subsequent poll — IMPL-22 / §13.2. """ # Tiny intervals so the loop iterates fast but doesn't immediately - # bail via the ``sleep_for <= 0`` early-return on line 723 of - # hooks.py. + # bail via the ``sleep_for <= 0`` early return. monkeypatch.setattr(hooks, "POLL_FAST_INTERVAL_S", 0.001) monkeypatch.setattr(hooks, "POLL_FAST_DURATION_S", 0.001) monkeypatch.setattr(hooks, "POLL_SLOW_INTERVAL_S", 0.001) @@ -73,7 +71,7 @@ def test_degraded_emitted_once_at_threshold(self, monkeypatch): hooks._poll_for_decision( task_id="01KTASK", request_id="01KREQ", - timeout_s=300, + deadline=hooks._ApprovalDeadline.from_recorded(hooks._iso_now(), 300), progress=progress, ts=ts, ) @@ -100,8 +98,7 @@ def test_recovery_resets_failure_counter(self, monkeypatch): outage. """ # Tiny intervals so the loop iterates fast but doesn't immediately - # bail via the ``sleep_for <= 0`` early-return on line 723 of - # hooks.py. + # bail via the ``sleep_for <= 0`` early return. monkeypatch.setattr(hooks, "POLL_FAST_INTERVAL_S", 0.001) monkeypatch.setattr(hooks, "POLL_FAST_DURATION_S", 0.001) monkeypatch.setattr(hooks, "POLL_SLOW_INTERVAL_S", 0.001) @@ -126,7 +123,7 @@ def test_recovery_resets_failure_counter(self, monkeypatch): hooks._poll_for_decision( task_id="01KTASK", request_id="01KREQ", - timeout_s=300, + deadline=hooks._ApprovalDeadline.from_recorded(hooks._iso_now(), 300), progress=progress, ts=ts, ) @@ -143,8 +140,7 @@ def test_deadline_beats_failures_when_timeout_short(self, monkeypatch): IMPL-24 design. """ # Tiny intervals so the loop iterates fast but doesn't immediately - # bail via the ``sleep_for <= 0`` early-return on line 723 of - # hooks.py. + # bail via the ``sleep_for <= 0`` early return. monkeypatch.setattr(hooks, "POLL_FAST_INTERVAL_S", 0.001) monkeypatch.setattr(hooks, "POLL_FAST_DURATION_S", 0.001) monkeypatch.setattr(hooks, "POLL_SLOW_INTERVAL_S", 0.001) @@ -158,7 +154,7 @@ def test_deadline_beats_failures_when_timeout_short(self, monkeypatch): hooks._poll_for_decision( task_id="01KTASK", request_id="01KREQ", - timeout_s=0, + deadline=hooks._ApprovalDeadline.from_recorded(hooks._iso_now(), 0), progress=progress, ts=ts, ) diff --git a/cdk/src/handlers/shared/compute-strategy.ts b/cdk/src/handlers/shared/compute-strategy.ts index 9ac48c5e3..a2741c1ab 100644 --- a/cdk/src/handlers/shared/compute-strategy.ts +++ b/cdk/src/handlers/shared/compute-strategy.ts @@ -90,6 +90,15 @@ export type SessionStatus = | { readonly status: 'completed'; readonly reason?: string } | { readonly status: 'failed'; readonly error: string; readonly reason?: string }; +/** + * `supported: true` means the lifecycle command was acknowledged, not that the + * target state has been reached. Callers must observe/reconcile the session. + * Failures throw; they must never be disguised as an unsupported capability. + */ +export type SessionLifecycleResult = + | { readonly supported: false } + | { readonly supported: true }; + export interface ComputeStrategy { readonly type: ComputeType; startSession(input: { @@ -121,6 +130,8 @@ export interface ComputeStrategy { }): Promise; pollSession(handle: SessionHandle): Promise; stopSession(handle: SessionHandle): Promise; + suspendSession(handle: SessionHandle): Promise; + resumeSession(handle: SessionHandle): Promise; } export function resolveComputeStrategy(blueprintConfig: BlueprintConfig): ComputeStrategy { diff --git a/cdk/src/handlers/shared/strategies/agentcore-strategy.ts b/cdk/src/handlers/shared/strategies/agentcore-strategy.ts index 762387abe..3e814958e 100644 --- a/cdk/src/handlers/shared/strategies/agentcore-strategy.ts +++ b/cdk/src/handlers/shared/strategies/agentcore-strategy.ts @@ -19,7 +19,7 @@ import { randomUUID } from 'crypto'; import { BedrockAgentCoreClient, InvokeAgentRuntimeCommand, StopRuntimeSessionCommand } from '@aws-sdk/client-bedrock-agentcore'; -import type { ComputeStrategy, SessionHandle, SessionStatus } from '../compute-strategy'; +import type { ComputeStrategy, SessionHandle, SessionLifecycleResult, SessionStatus } from '../compute-strategy'; import { logger } from '../logger'; import type { BlueprintConfig } from '../repo-config'; import { makeClient } from '../ua'; @@ -83,6 +83,16 @@ export class AgentCoreComputeStrategy implements ComputeStrategy { return { status: 'running' }; } + async suspendSession(handle: SessionHandle): Promise { + if (handle.strategyType !== 'agentcore') throw new Error('suspendSession called with non-agentcore handle'); + return { supported: false }; + } + + async resumeSession(handle: SessionHandle): Promise { + if (handle.strategyType !== 'agentcore') throw new Error('resumeSession called with non-agentcore handle'); + return { supported: false }; + } + async stopSession(handle: SessionHandle): Promise { if (handle.strategyType !== 'agentcore') { throw new Error('stopSession called with non-agentcore handle'); diff --git a/cdk/src/handlers/shared/strategies/ecs-strategy.ts b/cdk/src/handlers/shared/strategies/ecs-strategy.ts index 0718f8b34..44689eb14 100644 --- a/cdk/src/handlers/shared/strategies/ecs-strategy.ts +++ b/cdk/src/handlers/shared/strategies/ecs-strategy.ts @@ -18,7 +18,7 @@ */ import { ECSClient, RunTaskCommand, DescribeTasksCommand, StopTaskCommand } from '@aws-sdk/client-ecs'; -import type { ComputeStrategy, SessionHandle, SessionStatus } from '../compute-strategy'; +import type { ComputeStrategy, SessionHandle, SessionLifecycleResult, SessionStatus } from '../compute-strategy'; import { logger } from '../logger'; import { deletePayloadReference, preparePayloadReference, redactPayloadUrls } from '../payload-bootstrap'; import type { BlueprintConfig } from '../repo-config'; @@ -93,6 +93,16 @@ export async function deleteEcsPayload(taskId: string): Promise { export class EcsComputeStrategy implements ComputeStrategy { readonly type = 'ecs'; + async suspendSession(handle: SessionHandle): Promise { + if (handle.strategyType !== 'ecs') throw new Error('suspendSession called with non-ecs handle'); + return { supported: false }; + } + + async resumeSession(handle: SessionHandle): Promise { + if (handle.strategyType !== 'ecs') throw new Error('resumeSession called with non-ecs handle'); + return { supported: false }; + } + async startSession(input: { taskId: string; /** Accepted to satisfy the ComputeStrategy interface; ECS doesn't diff --git a/cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts b/cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts index be6845707..b5e36803d 100644 --- a/cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts +++ b/cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts @@ -23,12 +23,14 @@ import { MicrovmState, RunMicrovmCommand, TerminateMicrovmCommand, + SuspendMicrovmCommand, + ResumeMicrovmCommand, } from '@aws-sdk/client-lambda-microvms'; // Cross-language contract (S9): `microvm_platform_config` is read by BOTH this // producer and `agent/src/server.py`'s `/run` consumer. Imported (not copied) so // `tsc` fails on a renamed field — see `contracts/constants.md`. import sharedConstants from '../../../../../contracts/constants.json'; -import type { ComputeStrategy, SessionHandle, SessionStatus } from '../compute-strategy'; +import type { ComputeStrategy, SessionHandle, SessionLifecycleResult, SessionStatus } from '../compute-strategy'; import { MicrovmStartUncertainError } from '../error-classifier'; import { logger } from '../logger'; import { claimMicrovmStart, microvmStartRequestHash, saveMicrovmStartHandle } from '../microvm-start'; @@ -44,6 +46,9 @@ function getClient(): LambdaMicrovmsClient { return sharedClient; } +/** Bound a control request, not the transition itself. A timeout needs reconciliation. */ +export const MICROVM_LIFECYCLE_REQUEST_TIMEOUT_MS = 10_000; + /** * Fully-qualified MicroVM image **ARN** passed as `imageIdentifier` on every * `RunMicrovm`. @@ -439,11 +444,10 @@ function assertImageArn(identifier: string): void { * control-plane state machine the orchestrator can observe through * {@link LambdaMicrovmComputeStrategy.pollSession}. * - * P1 scope is start / poll / stop only. ``suspendSession`` / ``resumeSession`` - * (the interface widening across all three strategies) land in P3 — do NOT add - * them here piecemeal, ADR-021 sub-decision 1 requires them in one commit so - * the exhaustive-``never`` switch culture forces every backend to make a - * compile-checked decision about its suspend semantics. + * P3 command primitives implement mandatory suspend/resume alongside the + * explicit unsupported results in the other two strategies. The supervisor + * must still supply gate policy, durable intent and state reconciliation + * before automatic suspension can be enabled with compatible agent hooks. */ export class LambdaMicrovmComputeStrategy implements ComputeStrategy { readonly type = 'lambda-microvm'; @@ -783,6 +787,41 @@ export class LambdaMicrovmComputeStrategy implements ComputeStrategy { await this.terminateBestEffort(handle.microvmId, 'session stop'); } + /** Submit a suspend request; the caller owns gate checks and state reconciliation. */ + async suspendSession(handle: SessionHandle): Promise { + return this.requestLifecycle('suspendSession', handle); + } + + /** Submit a resume request; acknowledgement alone does not establish RUNNING. */ + async resumeSession(handle: SessionHandle): Promise { + return this.requestLifecycle('resumeSession', handle); + } + + private async requestLifecycle( + operation: 'suspendSession' | 'resumeSession', + handle: SessionHandle, + ): Promise { + if (handle.strategyType !== 'lambda-microvm') { + throw new Error(`${operation} called with non-lambda-microvm handle`); + } + if (typeof handle.microvmId !== 'string' || !handle.microvmId.trim()) { + throw new Error(`${operation} requires a non-empty MicroVM identifier`); + } + const suspend = operation === 'suspendSession'; + const request = { microvmIdentifier: handle.microvmId }; + try { + await getClient().send( + suspend ? new SuspendMicrovmCommand(request) : new ResumeMicrovmCommand(request), + { abortSignal: AbortSignal.timeout(MICROVM_LIFECYCLE_REQUEST_TIMEOUT_MS) }, + ); + } catch (error) { + // Includes Conflict/NotFound: neither proves the desired state was reached. + // Even a timeout may have committed; the durable caller must observe again. + throw wrapMicrovmError(suspend ? 'SuspendMicrovm' : 'ResumeMicrovm', error); + } + return { supported: true }; + } + /** * `TerminateMicrovm` that never throws, with the log LEVEL carrying the * diagnosis. diff --git a/cdk/test/handlers/shared/session-lifecycle.test.ts b/cdk/test/handlers/shared/session-lifecycle.test.ts new file mode 100644 index 000000000..edda851fa --- /dev/null +++ b/cdk/test/handlers/shared/session-lifecycle.test.ts @@ -0,0 +1,135 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import { inspect } from 'node:util'; + +const mockMicrovmSend = jest.fn(); +const mockEcsSend = jest.fn(); +const mockAgentcoreSend = jest.fn(); +jest.mock('@aws-sdk/client-lambda-microvms', () => ({ + ...jest.requireActual('@aws-sdk/client-lambda-microvms'), + LambdaMicrovmsClient: jest.fn(() => ({ send: mockMicrovmSend })), +})); +jest.mock('@aws-sdk/client-ecs', () => ({ + ...jest.requireActual('@aws-sdk/client-ecs'), + ECSClient: jest.fn(() => ({ send: mockEcsSend })), +})); +jest.mock('@aws-sdk/client-bedrock-agentcore', () => ({ + ...jest.requireActual('@aws-sdk/client-bedrock-agentcore'), + BedrockAgentCoreClient: jest.fn(() => ({ send: mockAgentcoreSend })), +})); + +import { ResumeMicrovmCommand, SuspendMicrovmCommand } from '@aws-sdk/client-lambda-microvms'; +import { resolveComputeStrategy, type SessionHandle } from '../../../src/handlers/shared/compute-strategy'; +import { MICROVM_LIFECYCLE_REQUEST_TIMEOUT_MS } from '../../../src/handlers/shared/strategies/lambda-microvm-strategy'; + +const handles: SessionHandle[] = [ + { strategyType: 'agentcore', sessionId: 'session-agentcore', runtimeArn: 'arn:runtime' }, + { strategyType: 'ecs', sessionId: 'arn:task', taskArn: 'arn:task', clusterArn: 'arn:cluster' }, + { strategyType: 'lambda-microvm', sessionId: 'mvm-one', microvmId: 'mvm-one', endpoint: 'https://unused.example' }, +]; +const microvm = handles[2] as Extract; + +beforeEach(() => { + jest.clearAllMocks(); + mockMicrovmSend.mockReset().mockResolvedValue({}); +}); + +describe.each(['suspendSession', 'resumeSession'] as const)('%s contract', operation => { + test.each(handles.filter(h => h.strategyType !== 'lambda-microvm'))( + '$strategyType explicitly reports unsupported without calling AWS', async handle => { + const strategy = resolveComputeStrategy({ compute_type: handle.strategyType, runtime_arn: 'arn:runtime' }); + await expect(strategy[operation](handle)).resolves.toEqual({ supported: false }); + expect(mockMicrovmSend).not.toHaveBeenCalled(); + expect(mockEcsSend).not.toHaveBeenCalled(); + expect(mockAgentcoreSend).not.toHaveBeenCalled(); + }, + ); + + test.each(handles)('$strategyType rejects a handle from another backend', async handle => { + const strategy = resolveComputeStrategy({ compute_type: handle.strategyType, runtime_arn: 'arn:runtime' }); + for (const other of handles.filter(h => h.strategyType !== handle.strategyType)) { + await expect(strategy[operation](other)).rejects.toThrow(`${operation} called with non-${handle.strategyType} handle`); + } + expect(mockMicrovmSend).not.toHaveBeenCalled(); + expect(mockEcsSend).not.toHaveBeenCalled(); + expect(mockAgentcoreSend).not.toHaveBeenCalled(); + }); + + test('sends only the identifier and reports acknowledgement, without pretending to observe state', async () => { + const strategy = resolveComputeStrategy({ compute_type: 'lambda-microvm', runtime_arn: '' }); + await expect(strategy[operation](microvm)).resolves.toEqual({ supported: true }); + expect(mockMicrovmSend).toHaveBeenCalledTimes(1); + const [command, options] = mockMicrovmSend.mock.calls[0]; + expect(command).toBeInstanceOf(operation === 'suspendSession' ? SuspendMicrovmCommand : ResumeMicrovmCommand); + expect(command.input).toEqual({ microvmIdentifier: 'mvm-one' }); + expect(options.abortSignal).toBeInstanceOf(AbortSignal); + expect(options.abortSignal.aborted).toBe(false); + }); + + test.each(['', ' ', undefined])('rejects an empty runtime identifier (%s)', async microvmId => { + const strategy = resolveComputeStrategy({ compute_type: 'lambda-microvm', runtime_arn: '' }); + await expect(strategy[operation]({ ...microvm, microvmId } as SessionHandle)).rejects.toThrow('non-empty MicroVM identifier'); + expect(mockMicrovmSend).not.toHaveBeenCalled(); + }); + + test.each([ + 'AccessDeniedException', 'ConflictException', 'ResourceNotFoundException', + 'ThrottlingException', 'InternalServerException', 'ValidationException', + ])('preserves %s as an operational failure with a sanitized cause', async name => { + mockMicrovmSend.mockRejectedValueOnce(Object.assign(new Error( + 'failed https://example.test/task?X-Amz-Signature=BEARER-SECRET', + { cause: new Error('private request metadata: BEARER-SECRET') }, + ), { name })); + const strategy = resolveComputeStrategy({ compute_type: 'lambda-microvm', runtime_arn: '' }); + await strategy[operation](microvm).then( + () => { throw new Error('Expected lifecycle request to fail'); }, + (error: Error) => { + expect(error.message).toContain(name); + expect(error.cause).toMatchObject({ name }); + expect(inspect(error, { depth: null })).not.toContain('BEARER-SECRET'); + }, + ); + }); + + test('a repeated command cannot silently turn a simulated state conflict into success', async () => { + mockMicrovmSend.mockResolvedValueOnce({}).mockRejectedValueOnce(Object.assign(new Error('state changed'), { name: 'ConflictException' })); + const strategy = resolveComputeStrategy({ compute_type: 'lambda-microvm', runtime_arn: '' }); + await expect(strategy[operation](microvm)).resolves.toEqual({ supported: true }); + await expect(strategy[operation](microvm)).rejects.toThrow('ConflictException'); + expect(mockMicrovmSend.mock.calls[0][0].input).toEqual(mockMicrovmSend.mock.calls[1][0].input); + }); + + test('bounds the SDK request and surfaces abort for later reconciliation', async () => { + const controller = new AbortController(); + const timeout = jest.spyOn(AbortSignal, 'timeout').mockReturnValue(controller.signal); + try { + mockMicrovmSend.mockImplementationOnce((_command, options) => new Promise((_resolve, reject) => { + options.abortSignal.addEventListener('abort', () => reject(Object.assign(new Error('deadline'), { name: 'AbortError' }))); + })); + const strategy = resolveComputeStrategy({ compute_type: 'lambda-microvm', runtime_arn: '' }); + const result = expect(strategy[operation](microvm)).rejects.toThrow('AbortError'); + controller.abort(); + await result; + expect(timeout).toHaveBeenCalledWith(MICROVM_LIFECYCLE_REQUEST_TIMEOUT_MS); + } finally { + timeout.mockRestore(); + } + }); +}); diff --git a/docs/decisions/ADR-021-lambda-microvms-compute-backend.md b/docs/decisions/ADR-021-lambda-microvms-compute-backend.md index b3f531124..4b4ab50d8 100644 --- a/docs/decisions/ADR-021-lambda-microvms-compute-backend.md +++ b/docs/decisions/ADR-021-lambda-microvms-compute-backend.md @@ -1,6 +1,6 @@ # ADR-021: AWS Lambda MicroVMs as a third ComputeStrategy backend -> **Implementation status (2026-09-13):** P1 ([#689](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/689)) and P2 ([#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733)) are merged. P2 completed real coding tasks with an IAM workaround; a clean rerun after re-bootstrap remains outstanding. P3 suspend/resume integration is not implemented. This ADR defines P1–P3, not a P4. See the [current review](../verification/645-p3-readiness-review.md) and [P3 implementation plan](../verification/645-p3-implementation-plan.md). +> **Implementation status (2026-09-13):** P1 ([#689](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/689)) and P2 ([#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733)) are merged. P2 completed real coding tasks with an IAM workaround; a clean rerun after re-bootstrap remains outstanding. Local P3 foundations now include strategy suspend/resume commands and approval deadlines that survive a frozen clock. Automatic policy, agent lifecycle hooks and supervisor/approval-handler integration remain unimplemented. This ADR defines P1–P3, not a P4. See the [current review](../verification/645-p3-readiness-review.md) and [P3 implementation plan](../verification/645-p3-implementation-plan.md). **Status:** proposed **Date:** 2026-07-29 @@ -64,7 +64,9 @@ Adopt **AWS Lambda MicroVMs as a third, opt-in `ComputeStrategy` backend** named **A second, sharper naming seam: `imageIdentifier` must be an ARN.** The name suggests a bare image name is acceptable — `create-microvm-image --name` takes one, and this ADR originally assumed `run-microvm --image-identifier` would too. It does not: a bare name is rejected with `ValidationException: Malformed ARN - doesn't start with 'arn:'`, and so is `list-microvm-image-builds --image-identifier ` (`Invalid ARN format`). The construct therefore resolves an operator-supplied name to its exact `arn:${Partition}:lambda:${Region}:${Account}:microvm-image:${Name}` ARN **once** — the same value it scopes the lifecycle IAM grant to — and injects THAT as `MICROVM_IMAGE_IDENTIFIER`. One derivation, two consumers, so a request field and an IAM resource can never disagree. The strategy validates the invariant and fails fast with the remedy, because the service's own error names neither the env var nor the fix. -The `ComputeStrategy` interface gains **mandatory** `suspendSession(handle)` / `resumeSession(handle)` methods returning a typed result (`{ supported: false } | { supported: true }`-shaped, exact type at implementation time); the widening lands in P3 (see sub-decision 5), in one commit across all three strategies. Mandatory-with-explicit-stub is the codebase idiom, not optional methods: no behavioral interface in the codebase has an optional method, `AgentCoreComputeStrategy.pollSession` is already a mandatory explicit stub rather than an optional member, and the exhaustive-`never` switch culture means a fourth backend must make a compile-checked decision about its suspend semantics instead of silently falling through a `strategy.suspendSession?.()` feature-detection. The agentcore and ecs strategies return `unsupported` (not a silent success — a suspend that silently no-ops would let the orchestrator believe compute billing stopped when it did not); the orchestrator gates its suspend policy on the typed response, consistent with how `pollTaskStatus` already branches explicitly on `computeType`. +The `ComputeStrategy` interface now has **mandatory** `suspendSession(handle)` / `resumeSession(handle)` methods returning `SessionLifecycleResult`, defined as `{ supported: false } | { supported: true }`. This local P3 foundation changes all three strategies together: AgentCore and ECS explicitly return false without calling AWS; MicroVM submits a command with a 10-second request bound. A future backend must implement both methods. The planned supervisor policy will use this explicit capability result. + +`supported: true` means AWS acknowledged the command, not that the target state was reached. The SDK's suspend/resume response bodies are empty. All operational failures, including conflicts and not-found responses, propagate through sanitized errors; no generic conflict is normalized into success without observed-state evidence. A timed-out command may still have committed and requires reconciliation. The installed SDK has no `RESUMING` state, and the existing coarse `running` poll result also includes PENDING/unknown observations. P3 integration must obtain explicit service-state evidence before declaring wake complete. **Poll semantics — the strategy reports, the orchestrator interprets.** `pollSession(handle)` receives only the session handle and cannot see task state, so the health rules must live where the DynamoDB status lives. `SessionStatus` gains a `'suspended'` variant; the strategy maps `GetMicrovm` state mechanically and the **orchestrator** cross-references against the task row — the same division of labor `finalPollState` already uses for ECS (substrate stopped + non-terminal DynamoDB status → failed) and `pollTaskStatus` uses for agentcore heartbeats: substrate `suspended` + task `AWAITING_APPROVAL` is healthy (orchestrator-intended suspend); `suspended` with any other task status is an anomaly to surface, not fail-fast; substrate terminal + non-terminal task status → classify failed. diff --git a/docs/design/CEDAR_HITL_GATES.md b/docs/design/CEDAR_HITL_GATES.md index bafbed104..b41c329fe 100644 --- a/docs/design/CEDAR_HITL_GATES.md +++ b/docs/design/CEDAR_HITL_GATES.md @@ -230,25 +230,22 @@ Narrative walk-through of the happy path. Sequence diagrams in the round-trip Me timeout 300s ``` Severity colors the line (respecting `NO_COLOR` env var). -18. Hook enters poll loop with strongly-consistent reads: +18. Hook enters poll loop with strongly-consistent reads. The deadline is captured with the original approval row, before database writes/notifications: UTC expiry is `created_at + timeout_s`, capped by the original monotonic remaining duration. `deadline.remaining_s()` takes the smaller remainder and clamps at zero, so a frozen guest clock or backward UTC correction cannot restart the window. ```python - async def _poll_for_decision(task_id, request_id, timeout_s): + async def _poll_for_decision(task_id, request_id, deadline): start = time.monotonic() interval = 2 consecutive_failures = 0 while True: elapsed = time.monotonic() - start - if elapsed >= timeout_s: + if deadline.remaining_s() <= 0: return TimedOut() if elapsed > 30: interval = 5 # backoff try: row = await _ddb_get_approval(task_id, request_id, ConsistentRead=True) consecutive_failures = 0 - if row is None: - # Row disappeared between write and poll — treat as stranded - return TimedOut(reason="approval row missing; fail-closed") - if row["status"] != "PENDING": + if row is not None and row["status"] in ("APPROVED", "DENIED"): return Decided(row) except Exception as exc: consecutive_failures += 1 @@ -257,9 +254,13 @@ Narrative walk-through of the happy path. Sequence diagrams in the round-trip Me emit_milestone("approval_poll_degraded", {...}) if consecutive_failures >= 10: return TimedOut(reason="approval poll consecutive failures") - await asyncio.sleep(interval) + # Missing rows keep waiting within the original deadline; never allow. + remaining = deadline.remaining_s() + if remaining <= 0: + return TimedOut() + await asyncio.sleep(min(interval, remaining)) ``` -19. The approval CAP and local-timeout paths ALWAYS attempt to write the row to TIMED_OUT (best-effort conditional update `status = :pending`) before returning. This prevents orphan PENDING rows when the agent bails internally. +19. The local-timeout path attempts to write the row to TIMED_OUT (best-effort conditional update `status = :pending`) before returning. If that write loses or fails, the hook rereads consistently to honor an already-committed decision. The approval-cap check runs before row creation and has no row to update. ### User responds @@ -815,6 +816,7 @@ async def pre_tool_use_hook(hook_input, tool_use_id, ctx, *, "ttl": int(time.time()) + effective_timeout + CLEANUP_MARGIN_120S, "user_id": user_id, "repo": engine.repo, } + deadline = _ApprovalDeadline.from_recorded(row["created_at"], effective_timeout) # ATOMIC: put approval row + transition TaskTable status in one transaction. try: @@ -832,7 +834,7 @@ async def pre_tool_use_hook(hook_input, tool_use_id, ctx, *, "matching_rule_ids": list(decision.matching_rule_ids), }) - outcome = await _poll_for_decision(task_id, request_id, effective_timeout) + outcome = await _poll_for_decision(task_id, request_id, deadline) # On TIMED_OUT, attempt to write the row to TIMED_OUT so future reads see # a terminal state (not orphaned PENDING). The conditional write is guarded @@ -920,7 +922,7 @@ async def pre_tool_use_hook(hook_input, tool_use_id, ctx, *, `engine._queue_denial_injection` appends to a list consumed by `_denial_between_turns_hook` — registered **after** `_nudge_between_turns_hook` in the `between_turns_hooks` list (which itself runs after `_cancel_between_turns_hook`). At the next Stop hook fire, the denial is emitted as `…` XML (sanitized via `_xml_escape` from the shared utility introduced with Phase 2). If a `bgagent cancel` has landed between the deny and the next Stop seam, `_cancel_between_turns_hook` short-circuits the dispatcher and the denial text is NOT injected — in which case the guaranteed surface is `permissionDecisionReason` on the hook return. See finding #2 scenario in §4 for the cancel-vs-deny race reasoning. -**Scenario (§13.12 VM-throttle + late-approval race).** User Alice hits a soft-deny gate at t=0 with `timeout_s=300`. The AgentCore VM is evicted from its warm CPU share around t=285 due to noisy-neighbor pressure on the host; poll ticks stretch by ~400ms. Alice, seeing the approval prompt in Terminal A, types `bgagent approve 01KPW... 01KPR...` at t=294. The approve-transaction lands in DDB at t=294.7 (APPROVED). The agent's next poll-tick was due at t=290 but the VM throttle delayed it to t=295.1. The monotonic wall-clock already shows elapsed >300 (actual since-start ~300.3s), so `_poll_for_decision` returns `TimedOut()`. The hook runs `_best_effort_update_status("TIMED_OUT", ... WHERE status = :pending)` — the conditional fails because the row is APPROVED. **Without the re-read**, the hook would proceed with stale local `outcome.status = "TIMED_OUT"`, queue a denial injection, and return `{"permissionDecision": "deny"}` — Alice sees "I approved it" on Terminal B but the agent denies the tool call anyway. **With the re-read** (the `wrote_timeout` branch in the pseudocode above): the hook fetches the row with ConsistentRead, sees `status = APPROVED`, rebuilds `outcome` from the row (preserving `scope`, `decided_by`, `decided_at`), emits an `approval_late_win` milestone, runs the normal resume transaction + allow flow, and returns `{"permissionDecision": "allow"}`. Alice's tool runs. The cost is one extra strongly-consistent GetItem on the race path; the benefit is that user intent is authoritative. Without this fix, a timer design that is otherwise sound would produce a confounding and unrecoverable UX. See IMPL-24, §13.12, and §15.2 task #43 for the race test. +**Scenario (§13.12 VM-throttle + late-approval race).** Alice hits a gate with a 300-second window and commits APPROVED at t=294.7s. Scheduling delays prevent the next agent poll until t=300.2s. That poll sees the original deadline has passed and returns TIMED_OUT before reading the row. The conditional TIMED_OUT write then loses because APPROVED is already stored. The hook rereads with `ConsistentRead`, preserves Alice's scope and decision metadata, emits `approval_late_win`, and proceeds through the guarded resume transaction and allow flow. Without that reread it would deny an already-approved call. See IMPL-24, §13.12, and §15.2 task #43 for the race test. --- @@ -1810,7 +1812,7 @@ Addressed by the parity contract (decision #23, §15.6). Golden-file CI test run ### 13.12 VM-throttle + late-approval race -The agent's poll loop computes a local timeout wall-clock (`timeout_s` worth of elapsed monotonic time). If the VM is throttled by the hypervisor — either an AgentCore noisy-neighbor eviction window, or a CPU-throttle under memory pressure — poll ticks can stretch past their nominal cadence. In the worst case, the user's APPROVE transaction lands in DDB a few hundred milliseconds before the agent's local clock trips past `timeout_s` and the agent attempts to write `status = TIMED_OUT WHERE status = :pending`. The ConditionCheckFailed path fires (APPROVED already won), but without a re-read the agent's local state is stale: `outcome.status == "TIMED_OUT"` locally while DDB holds APPROVED. The agent would return DENY, the user sees "I approved it" and the agent still blocks — a confounding experience that also violates the design principle that user-observed state is authoritative. +The agent's poll loop retains the original approval row's UTC expiry and a monotonic cap, using whichever expires first. Slow writes/notifications count toward that same window. This also covers a MicroVM whose monotonic clock stops while suspended; the future `/resume` hook must reuse this deadline and wake the decision loop. If CPU throttling or suspension delays polling beyond expiry, a user's APPROVE transaction may already have committed. The agent then attempts `status = TIMED_OUT WHERE status = :pending`. The condition fails because APPROVED already won. Without a re-read, the agent's local state would remain TIMED_OUT while DDB holds APPROVED, and it would incorrectly deny the approved call. **Mitigation**: the §6.5 pseudocode re-reads the approval row with `ConsistentRead=True` whenever `_best_effort_update_status("TIMED_OUT", ...)` returns ConditionCheckFailed, and honors whatever terminal state the row carries: - If `status == "APPROVED"`: rebuild the local `outcome` to reflect APPROVED, preserving `scope`, `decided_by`, `decided_at`, and proceed through the normal allow flow (scope-propagation, `approval_granted` milestone, resume transaction, return `{"permissionDecision": "allow"}`). Emit a `approval_late_win` milestone so operator telemetry can count races. @@ -2025,22 +2027,18 @@ t=0.02s TransactWriteItems: approval row PENDING + TaskTable AWAITING_APPROVA t=0.03s agent_milestone: approval_requested → Terminal A stream t=0.04s _poll_for_decision begins; interval=2s for first 30s, then 5s -... (poll ticks every 5s from t=30 to t=295) ... +... (poll ticks every 5s from t=30 to t=285) ... t=285.0s host hypervisor evicts VM from warm CPU share (noisy neighbor). - Next scheduled poll was t=290.0s; actual scheduling delay ~5.1s. + Next scheduled poll was t=290.0s; actual scheduling delay ~10.2s. t=294.7s Alice's bgagent approve lands at API Gateway. ApproveTaskFn TransactWriteItems: ApprovalsTable: PENDING → APPROVED (user_id matches, status was PENDING) TaskTable: state guard holds (still AWAITING_APPROVAL, rid matches) → 202 returned to CLI; Alice sees "approved!" in Terminal B -t=295.1s agent's delayed poll tick fires. Elapsed wall-clock = 295.1s. - Monotonic elapsed is 295.1 > timeout_s=300? NO — but the poll - function computes `elapsed >= timeout_s` and on the NEXT tick - (t=300.2s) it will exceed. -t=300.2s next tick: elapsed=300.2 ≥ timeout_s=300 → TimedOut() returned. - (Alice's APPROVED write at t=294.7s was MISSED — the previous - poll was due at t=295.0 but the VM throttle stretched it past.) +t=300.2s delayed tick: original deadline has passed → TimedOut() returned + before reading the approval row. Alice's APPROVED write at + t=294.7s has not yet been observed by this agent. t=300.3s _best_effort_update_status("TIMED_OUT", ... WHERE status = :pending) → ConditionCheckFailed (row is APPROVED, not PENDING) → wrote_timeout = False diff --git a/docs/design/COMPUTE.md b/docs/design/COMPUTE.md index 8e0ee760c..89cc59efd 100644 --- a/docs/design/COMPUTE.md +++ b/docs/design/COMPUTE.md @@ -83,6 +83,8 @@ Because a snapshot freezes its build-time environment, current deployment identi Networking separates image build from execution: the build-only connector permits TCP 80 and 443 because the Dockerfile uses `apt-get`, while running MicroVMs retain 443-only egress through the platform VPC. Every launch explicitly passes the Lambda-managed `NO_INGRESS` connector; omission would select the service's public-ingress default. The P2 image declares and serves `/ready` and `/validate` at build time and `/run` and `/terminate` at runtime. `/suspend` and `/resume` remain disabled until their P3 implementation. Registry HTTP/SSE tools therefore need reachable HTTPS/443 endpoints; remote non-443 tools are unsupported under the default policies of AgentCore and ECS as well. Local `stdio` tools can run, with their outbound traffic subject to the same restriction. Asset resolution/loading does not probe connectivity; see [registry network support](./REGISTRY.md#2-asset-kinds-for-mvp). +The local P3 foundation adds bounded suspend/resume commands to the MicroVM strategy and explicit unsupported results to AgentCore/ECS. These methods are not wired into automatic policy or human-decision handlers yet. Approval polling now retains the original UTC/monotonic deadline through slow writes or a frozen guest clock; resume hooks must still reuse that deadline and refresh credentials before work continues. + ## ECS Fargate task sizing (build vs. planning) When a repo is `compute_type: ecs`, `EcsAgentCluster` provisions **two** Fargate task definitions, and the orchestrator picks between them per task by whether the resolved workflow is **read-only**: diff --git a/docs/src/content/docs/architecture/Cedar-hitl-gates.md b/docs/src/content/docs/architecture/Cedar-hitl-gates.md index b54af4b52..e47f209cf 100644 --- a/docs/src/content/docs/architecture/Cedar-hitl-gates.md +++ b/docs/src/content/docs/architecture/Cedar-hitl-gates.md @@ -234,25 +234,22 @@ Narrative walk-through of the happy path. Sequence diagrams in the round-trip Me timeout 300s ``` Severity colors the line (respecting `NO_COLOR` env var). -18. Hook enters poll loop with strongly-consistent reads: +18. Hook enters poll loop with strongly-consistent reads. The deadline is captured with the original approval row, before database writes/notifications: UTC expiry is `created_at + timeout_s`, capped by the original monotonic remaining duration. `deadline.remaining_s()` takes the smaller remainder and clamps at zero, so a frozen guest clock or backward UTC correction cannot restart the window. ```python - async def _poll_for_decision(task_id, request_id, timeout_s): + async def _poll_for_decision(task_id, request_id, deadline): start = time.monotonic() interval = 2 consecutive_failures = 0 while True: elapsed = time.monotonic() - start - if elapsed >= timeout_s: + if deadline.remaining_s() <= 0: return TimedOut() if elapsed > 30: interval = 5 # backoff try: row = await _ddb_get_approval(task_id, request_id, ConsistentRead=True) consecutive_failures = 0 - if row is None: - # Row disappeared between write and poll — treat as stranded - return TimedOut(reason="approval row missing; fail-closed") - if row["status"] != "PENDING": + if row is not None and row["status"] in ("APPROVED", "DENIED"): return Decided(row) except Exception as exc: consecutive_failures += 1 @@ -261,9 +258,13 @@ Narrative walk-through of the happy path. Sequence diagrams in the round-trip Me emit_milestone("approval_poll_degraded", {...}) if consecutive_failures >= 10: return TimedOut(reason="approval poll consecutive failures") - await asyncio.sleep(interval) + # Missing rows keep waiting within the original deadline; never allow. + remaining = deadline.remaining_s() + if remaining <= 0: + return TimedOut() + await asyncio.sleep(min(interval, remaining)) ``` -19. The approval CAP and local-timeout paths ALWAYS attempt to write the row to TIMED_OUT (best-effort conditional update `status = :pending`) before returning. This prevents orphan PENDING rows when the agent bails internally. +19. The local-timeout path attempts to write the row to TIMED_OUT (best-effort conditional update `status = :pending`) before returning. If that write loses or fails, the hook rereads consistently to honor an already-committed decision. The approval-cap check runs before row creation and has no row to update. ### User responds @@ -819,6 +820,7 @@ async def pre_tool_use_hook(hook_input, tool_use_id, ctx, *, "ttl": int(time.time()) + effective_timeout + CLEANUP_MARGIN_120S, "user_id": user_id, "repo": engine.repo, } + deadline = _ApprovalDeadline.from_recorded(row["created_at"], effective_timeout) # ATOMIC: put approval row + transition TaskTable status in one transaction. try: @@ -836,7 +838,7 @@ async def pre_tool_use_hook(hook_input, tool_use_id, ctx, *, "matching_rule_ids": list(decision.matching_rule_ids), }) - outcome = await _poll_for_decision(task_id, request_id, effective_timeout) + outcome = await _poll_for_decision(task_id, request_id, deadline) # On TIMED_OUT, attempt to write the row to TIMED_OUT so future reads see # a terminal state (not orphaned PENDING). The conditional write is guarded @@ -924,7 +926,7 @@ async def pre_tool_use_hook(hook_input, tool_use_id, ctx, *, `engine._queue_denial_injection` appends to a list consumed by `_denial_between_turns_hook` — registered **after** `_nudge_between_turns_hook` in the `between_turns_hooks` list (which itself runs after `_cancel_between_turns_hook`). At the next Stop hook fire, the denial is emitted as `…` XML (sanitized via `_xml_escape` from the shared utility introduced with Phase 2). If a `bgagent cancel` has landed between the deny and the next Stop seam, `_cancel_between_turns_hook` short-circuits the dispatcher and the denial text is NOT injected — in which case the guaranteed surface is `permissionDecisionReason` on the hook return. See finding #2 scenario in §4 for the cancel-vs-deny race reasoning. -**Scenario (§13.12 VM-throttle + late-approval race).** User Alice hits a soft-deny gate at t=0 with `timeout_s=300`. The AgentCore VM is evicted from its warm CPU share around t=285 due to noisy-neighbor pressure on the host; poll ticks stretch by ~400ms. Alice, seeing the approval prompt in Terminal A, types `bgagent approve 01KPW... 01KPR...` at t=294. The approve-transaction lands in DDB at t=294.7 (APPROVED). The agent's next poll-tick was due at t=290 but the VM throttle delayed it to t=295.1. The monotonic wall-clock already shows elapsed >300 (actual since-start ~300.3s), so `_poll_for_decision` returns `TimedOut()`. The hook runs `_best_effort_update_status("TIMED_OUT", ... WHERE status = :pending)` — the conditional fails because the row is APPROVED. **Without the re-read**, the hook would proceed with stale local `outcome.status = "TIMED_OUT"`, queue a denial injection, and return `{"permissionDecision": "deny"}` — Alice sees "I approved it" on Terminal B but the agent denies the tool call anyway. **With the re-read** (the `wrote_timeout` branch in the pseudocode above): the hook fetches the row with ConsistentRead, sees `status = APPROVED`, rebuilds `outcome` from the row (preserving `scope`, `decided_by`, `decided_at`), emits an `approval_late_win` milestone, runs the normal resume transaction + allow flow, and returns `{"permissionDecision": "allow"}`. Alice's tool runs. The cost is one extra strongly-consistent GetItem on the race path; the benefit is that user intent is authoritative. Without this fix, a timer design that is otherwise sound would produce a confounding and unrecoverable UX. See IMPL-24, §13.12, and §15.2 task #43 for the race test. +**Scenario (§13.12 VM-throttle + late-approval race).** Alice hits a gate with a 300-second window and commits APPROVED at t=294.7s. Scheduling delays prevent the next agent poll until t=300.2s. That poll sees the original deadline has passed and returns TIMED_OUT before reading the row. The conditional TIMED_OUT write then loses because APPROVED is already stored. The hook rereads with `ConsistentRead`, preserves Alice's scope and decision metadata, emits `approval_late_win`, and proceeds through the guarded resume transaction and allow flow. Without that reread it would deny an already-approved call. See IMPL-24, §13.12, and §15.2 task #43 for the race test. --- @@ -1814,7 +1816,7 @@ Addressed by the parity contract (decision #23, §15.6). Golden-file CI test run ### 13.12 VM-throttle + late-approval race -The agent's poll loop computes a local timeout wall-clock (`timeout_s` worth of elapsed monotonic time). If the VM is throttled by the hypervisor — either an AgentCore noisy-neighbor eviction window, or a CPU-throttle under memory pressure — poll ticks can stretch past their nominal cadence. In the worst case, the user's APPROVE transaction lands in DDB a few hundred milliseconds before the agent's local clock trips past `timeout_s` and the agent attempts to write `status = TIMED_OUT WHERE status = :pending`. The ConditionCheckFailed path fires (APPROVED already won), but without a re-read the agent's local state is stale: `outcome.status == "TIMED_OUT"` locally while DDB holds APPROVED. The agent would return DENY, the user sees "I approved it" and the agent still blocks — a confounding experience that also violates the design principle that user-observed state is authoritative. +The agent's poll loop retains the original approval row's UTC expiry and a monotonic cap, using whichever expires first. Slow writes/notifications count toward that same window. This also covers a MicroVM whose monotonic clock stops while suspended; the future `/resume` hook must reuse this deadline and wake the decision loop. If CPU throttling or suspension delays polling beyond expiry, a user's APPROVE transaction may already have committed. The agent then attempts `status = TIMED_OUT WHERE status = :pending`. The condition fails because APPROVED already won. Without a re-read, the agent's local state would remain TIMED_OUT while DDB holds APPROVED, and it would incorrectly deny the approved call. **Mitigation**: the §6.5 pseudocode re-reads the approval row with `ConsistentRead=True` whenever `_best_effort_update_status("TIMED_OUT", ...)` returns ConditionCheckFailed, and honors whatever terminal state the row carries: - If `status == "APPROVED"`: rebuild the local `outcome` to reflect APPROVED, preserving `scope`, `decided_by`, `decided_at`, and proceed through the normal allow flow (scope-propagation, `approval_granted` milestone, resume transaction, return `{"permissionDecision": "allow"}`). Emit a `approval_late_win` milestone so operator telemetry can count races. @@ -2029,22 +2031,18 @@ t=0.02s TransactWriteItems: approval row PENDING + TaskTable AWAITING_APPROVA t=0.03s agent_milestone: approval_requested → Terminal A stream t=0.04s _poll_for_decision begins; interval=2s for first 30s, then 5s -... (poll ticks every 5s from t=30 to t=295) ... +... (poll ticks every 5s from t=30 to t=285) ... t=285.0s host hypervisor evicts VM from warm CPU share (noisy neighbor). - Next scheduled poll was t=290.0s; actual scheduling delay ~5.1s. + Next scheduled poll was t=290.0s; actual scheduling delay ~10.2s. t=294.7s Alice's bgagent approve lands at API Gateway. ApproveTaskFn TransactWriteItems: ApprovalsTable: PENDING → APPROVED (user_id matches, status was PENDING) TaskTable: state guard holds (still AWAITING_APPROVAL, rid matches) → 202 returned to CLI; Alice sees "approved!" in Terminal B -t=295.1s agent's delayed poll tick fires. Elapsed wall-clock = 295.1s. - Monotonic elapsed is 295.1 > timeout_s=300? NO — but the poll - function computes `elapsed >= timeout_s` and on the NEXT tick - (t=300.2s) it will exceed. -t=300.2s next tick: elapsed=300.2 ≥ timeout_s=300 → TimedOut() returned. - (Alice's APPROVED write at t=294.7s was MISSED — the previous - poll was due at t=295.0 but the VM throttle stretched it past.) +t=300.2s delayed tick: original deadline has passed → TimedOut() returned + before reading the approval row. Alice's APPROVED write at + t=294.7s has not yet been observed by this agent. t=300.3s _best_effort_update_status("TIMED_OUT", ... WHERE status = :pending) → ConditionCheckFailed (row is APPROVED, not PENDING) → wrote_timeout = False diff --git a/docs/src/content/docs/architecture/Compute.md b/docs/src/content/docs/architecture/Compute.md index f8630678f..25bce1371 100644 --- a/docs/src/content/docs/architecture/Compute.md +++ b/docs/src/content/docs/architecture/Compute.md @@ -87,6 +87,8 @@ Because a snapshot freezes its build-time environment, current deployment identi Networking separates image build from execution: the build-only connector permits TCP 80 and 443 because the Dockerfile uses `apt-get`, while running MicroVMs retain 443-only egress through the platform VPC. Every launch explicitly passes the Lambda-managed `NO_INGRESS` connector; omission would select the service's public-ingress default. The P2 image declares and serves `/ready` and `/validate` at build time and `/run` and `/terminate` at runtime. `/suspend` and `/resume` remain disabled until their P3 implementation. Registry HTTP/SSE tools therefore need reachable HTTPS/443 endpoints; remote non-443 tools are unsupported under the default policies of AgentCore and ECS as well. Local `stdio` tools can run, with their outbound traffic subject to the same restriction. Asset resolution/loading does not probe connectivity; see [registry network support](/sample-autonomous-cloud-coding-agents/architecture/registry#2-asset-kinds-for-mvp). +The local P3 foundation adds bounded suspend/resume commands to the MicroVM strategy and explicit unsupported results to AgentCore/ECS. These methods are not wired into automatic policy or human-decision handlers yet. Approval polling now retains the original UTC/monotonic deadline through slow writes or a frozen guest clock; resume hooks must still reuse that deadline and refresh credentials before work continues. + ## ECS Fargate task sizing (build vs. planning) When a repo is `compute_type: ecs`, `EcsAgentCluster` provisions **two** Fargate task definitions, and the orchestrator picks between them per task by whether the resolved workflow is **read-only**: diff --git a/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md b/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md index db2c6cf3c..e2fb78c1f 100644 --- a/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md +++ b/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md @@ -4,7 +4,7 @@ title: Adr 021 lambda microvms compute backend # ADR-021: AWS Lambda MicroVMs as a third ComputeStrategy backend -> **Implementation status (2026-09-13):** P1 ([#689](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/689)) and P2 ([#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733)) are merged. P2 completed real coding tasks with an IAM workaround; a clean rerun after re-bootstrap remains outstanding. P3 suspend/resume integration is not implemented. This ADR defines P1–P3, not a P4. See the [current review](/sample-autonomous-cloud-coding-agents/architecture/645-p3-readiness-review) and [P3 implementation plan](/sample-autonomous-cloud-coding-agents/architecture/645-p3-implementation-plan). +> **Implementation status (2026-09-13):** P1 ([#689](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/689)) and P2 ([#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733)) are merged. P2 completed real coding tasks with an IAM workaround; a clean rerun after re-bootstrap remains outstanding. Local P3 foundations now include strategy suspend/resume commands and approval deadlines that survive a frozen clock. Automatic policy, agent lifecycle hooks and supervisor/approval-handler integration remain unimplemented. This ADR defines P1–P3, not a P4. See the [current review](/sample-autonomous-cloud-coding-agents/architecture/645-p3-readiness-review) and [P3 implementation plan](/sample-autonomous-cloud-coding-agents/architecture/645-p3-implementation-plan). **Status:** proposed **Date:** 2026-07-29 @@ -68,7 +68,9 @@ Adopt **AWS Lambda MicroVMs as a third, opt-in `ComputeStrategy` backend** named **A second, sharper naming seam: `imageIdentifier` must be an ARN.** The name suggests a bare image name is acceptable — `create-microvm-image --name` takes one, and this ADR originally assumed `run-microvm --image-identifier` would too. It does not: a bare name is rejected with `ValidationException: Malformed ARN - doesn't start with 'arn:'`, and so is `list-microvm-image-builds --image-identifier ` (`Invalid ARN format`). The construct therefore resolves an operator-supplied name to its exact `arn:${Partition}:lambda:${Region}:${Account}:microvm-image:${Name}` ARN **once** — the same value it scopes the lifecycle IAM grant to — and injects THAT as `MICROVM_IMAGE_IDENTIFIER`. One derivation, two consumers, so a request field and an IAM resource can never disagree. The strategy validates the invariant and fails fast with the remedy, because the service's own error names neither the env var nor the fix. -The `ComputeStrategy` interface gains **mandatory** `suspendSession(handle)` / `resumeSession(handle)` methods returning a typed result (`{ supported: false } | { supported: true }`-shaped, exact type at implementation time); the widening lands in P3 (see sub-decision 5), in one commit across all three strategies. Mandatory-with-explicit-stub is the codebase idiom, not optional methods: no behavioral interface in the codebase has an optional method, `AgentCoreComputeStrategy.pollSession` is already a mandatory explicit stub rather than an optional member, and the exhaustive-`never` switch culture means a fourth backend must make a compile-checked decision about its suspend semantics instead of silently falling through a `strategy.suspendSession?.()` feature-detection. The agentcore and ecs strategies return `unsupported` (not a silent success — a suspend that silently no-ops would let the orchestrator believe compute billing stopped when it did not); the orchestrator gates its suspend policy on the typed response, consistent with how `pollTaskStatus` already branches explicitly on `computeType`. +The `ComputeStrategy` interface now has **mandatory** `suspendSession(handle)` / `resumeSession(handle)` methods returning `SessionLifecycleResult`, defined as `{ supported: false } | { supported: true }`. This local P3 foundation changes all three strategies together: AgentCore and ECS explicitly return false without calling AWS; MicroVM submits a command with a 10-second request bound. A future backend must implement both methods. The planned supervisor policy will use this explicit capability result. + +`supported: true` means AWS acknowledged the command, not that the target state was reached. The SDK's suspend/resume response bodies are empty. All operational failures, including conflicts and not-found responses, propagate through sanitized errors; no generic conflict is normalized into success without observed-state evidence. A timed-out command may still have committed and requires reconciliation. The installed SDK has no `RESUMING` state, and the existing coarse `running` poll result also includes PENDING/unknown observations. P3 integration must obtain explicit service-state evidence before declaring wake complete. **Poll semantics — the strategy reports, the orchestrator interprets.** `pollSession(handle)` receives only the session handle and cannot see task state, so the health rules must live where the DynamoDB status lives. `SessionStatus` gains a `'suspended'` variant; the strategy maps `GetMicrovm` state mechanically and the **orchestrator** cross-references against the task row — the same division of labor `finalPollState` already uses for ECS (substrate stopped + non-terminal DynamoDB status → failed) and `pollTaskStatus` uses for agentcore heartbeats: substrate `suspended` + task `AWAITING_APPROVAL` is healthy (orchestrator-intended suspend); `suspended` with any other task status is an anomaly to surface, not fail-fast; substrate terminal + non-terminal task status → classify failed. diff --git a/docs/verification/645-p3-implementation-plan.md b/docs/verification/645-p3-implementation-plan.md index 4afb136a8..b9d812948 100644 --- a/docs/verification/645-p3-implementation-plan.md +++ b/docs/verification/645-p3-implementation-plan.md @@ -23,7 +23,10 @@ Prerequisite work is tracked here on `fix/645-microvm-readiness`. “Completed - [ ] Verify metadata restrictions with real AWS sessions/transactions; retain status/tag trust limits. - [x] Replace unused logging-failure bookkeeping with structured stdout diagnostics (#810); document shared runtime networking and verify large registry payload delivery (#818). - [ ] Implement production nesting if included, then verify a clean P2 deployment. -- [ ] Implement and verify the P3 sleep/wake lifecycle described below. +- [x] Add mandatory pause/wake command methods across all three compute strategies, with explicit unsupported results and bounded MicroVM requests. +- [x] Keep the original approval deadline through database writes and polling, including frozen/backward clocks; preserve decision races and cancellation. +- [ ] Add durable lifecycle intent/policy, compatible agent hooks and credential/durability barriers. +- [ ] Connect supervisor and approval handlers, then verify the complete P3 sleep/wake lifecycle in AWS. First prerequisite batch completed locally on 2026-09-13: @@ -88,6 +91,13 @@ Seventh prerequisite batch completed locally (2026-09-13): - Full agent quality passed **1,811 tests / 84.51% coverage**. CDK lint/compilation and **38 tests** in the two relevant orchestrator/registry suites passed. These counts overlap previous runs; no CDK runtime code or IAM changed in this batch. - Source/generated registry/compute/deployment documentation, the **77-page** build and link checks pass. An offline coordinator bundle check includes the S3 client, presigner and shared constants; it does not replace a full deployed packaging check. Nothing was deployed or posted to the issue trackers. Package 3's clean P2 rerun and package 7's live P3 matrix remain required. +First P3 foundation batch completed locally (2026-09-13): + +- The shared strategy contract now requires `suspendSession` and `resumeSession`. AgentCore/ECS return `{supported:false}` without calling AWS. MicroVM sends the exact identifier with a 10-second request bound, returns `{supported:true}` only on acknowledgement, and surfaces sanitized errors. Conflicts, missing VMs and uncertain timeouts still require state reconciliation; they are not treated as success. +- An immutable approval deadline is captured with the original row, before database writes/notifications. Polling uses the smaller UTC/monotonic remainder. Regressions reproduced a frozen-clock gate that waited after ten minutes and a 30-second window stretched to 42 seconds by a slow write. The fix also preserves late committed decisions, missing-row handling and cancellation. +- CDK lint/compilation and **5 suites / 169 tests** pass. Full agent quality passes **1,823 tests / 84.53% coverage**, including **12 new clock/race/cancellation cases**. Counts overlap earlier runs. +- No callers, IAM grants or image hooks enable automatic suspension yet. The same deadline must still be registered in the future lifecycle context and checked immediately by `/resume`. Durable intent, policy, credential refresh, acknowledged progress durability and live validation remain open. + ## The result we want When the coding agent asks a human for permission, its computer may go to sleep. The human can approve or deny while it sleeps. The computer wakes, reads the saved answer, and continues or refuses the action. If nobody answers, it wakes before the deadline and applies the existing timeout-as-denial rule. Its files, task identity, permissions, progress and deadline remain correct. @@ -245,9 +255,11 @@ Follow `cdk/scripts/package-microvm-artifact.sh` and the P1/P2 runbooks: ### Strategy methods -Add mandatory `suspendSession(handle)` and `resumeSession(handle)` to `ComputeStrategy` in the **same commit as all three implementations**. Use an explicit result such as `{ supported: false } | { supported: true }`; supported means the capability/request is supported, not that the VM is already in its final state. Operational failures must remain distinguishable from unsupported capability. +**Implemented locally:** mandatory `suspendSession(handle)` and `resumeSession(handle)` were added to `ComputeStrategy` together with all three implementations. `SessionLifecycleResult` is `{ supported: false } | { supported: true }`; true means the command was acknowledged, not that the VM reached its final state. Operational failures throw. + +AgentCore and ECS return explicit unsupported results without an AWS request. MicroVM issues `SuspendMicrovm`/`ResumeMicrovm` with `microvmIdentifier: handle.microvmId` and a 10-second abort bound. Local tests cover wrong/empty handles, repeated requests, simulated conflicts/not-found and retriable/permanent errors. **Still required:** verify actual AWS behavior for already-target-state/terminated VMs and races before normalizing any conflict into success. -AgentCore and ECS return explicit unsupported results. MicroVM issues `SuspendMicrovm`/`ResumeMicrovm` with `microvmIdentifier: handle.microvmId`. Test wrong handle variants, already-target-state requests, state-conflict races, missing/terminated VMs and retriable/permanent API errors. Verify actual AWS behavior before normalizing a conflict into success. Preserve the strategy's existing state mapping; keep `reason` diagnostic. +The installed SDK returns empty suspend/resume responses. Its observed states are `PENDING`, `RUNNING`, `SUSPENDING`, `SUSPENDED`, `TERMINATING` and `TERMINATED`; there is no `RESUMING` value. The current coarse poll mapping groups PENDING/unknown with running and SUSPENDING with suspended. Preserve existing consumers, but expose an explicit service-state observation for P3 reconciliation: the coarse `running` result alone cannot prove wake completion. Keep `reason` diagnostic rather than parsing it for policy. ### Durable intent and policy @@ -283,8 +295,8 @@ Files: `agent/src/server.py`, `hooks.py`, `task_state.py`, `aws_session.py`, `pr 3. `/resume` refreshes ambient/runtime credential providers as necessary, then ensures tenant-scoped assumed credentials are usable **with the same task/user/repo tags**. Inventory cached DynamoDB/S3/Memory/Logs clients and the Claude Bedrock credential helper; replacing one global session does not replace every already-created client or subprocess cache. Do not call the test-only `reset_session_cache()` and lose identity. Fail closed on refresh failure. 4. Keep the coding action blocked behind the resume barrier until refresh and gate reconciliation finish. Handle duplicate hook calls and concurrent lifecycle requests without deadlocks. Expired credentials or a slow AWS call must not hold the hook beyond its service budget. 5. Reseed the application PRNG from fresh OS entropy on **both `/run` and `/resume`**. Do not seed it with task IDs, timestamps or an image-fixed value. Continue using cryptographic randomness for secrets. Test resumed/sibling snapshot uniqueness where meaningful; do not claim that `random` becomes cryptographically safe. -6. Compute approval remaining time as `min(monotonic_deadline - monotonic_now, created_at + timeout_s - wall_now)`, clamped at zero. Pass the original recorded `created_at` into the gate context; do not create a fresh timeout on resume. Check it on every poll and at resume, then wake the existing agent-owned decision loop to apply its transaction rules. -7. Preserve the conditional TIMED_OUT write, strongly consistent reread when that write loses, and the late-approval winner behavior. A decision committed before timeout must not be overwritten because the VM woke late. Test forward/backward clock changes, frozen monotonic time, missing/TTL-reaped row, approval at the boundary and cancellation. TTL is asynchronous garbage collection, not a precise alarm clock. +6. **Polling implemented locally:** `_ApprovalDeadline` captures the original recorded UTC expiry and a monotonic cap before database writes. Remaining time is `min(monotonic_deadline - monotonic_now, created_at + timeout_s - wall_now)`, clamped at zero, including each sleep bound. **Still required:** register this same object in the lifecycle context, check it at resume and wake the existing agent-owned decision loop. Never create a fresh timeout on resume. +7. **Preserved/tested locally:** conditional TIMED_OUT write, strongly consistent reread when that write loses, and late-decision winner behavior. Forward/backward clocks, frozen monotonic time, slow writes/reads, missing rows and cancellation have regression coverage. **Still required:** exercise these through the actual resume barrier and live AWS lifecycle. TTL is asynchronous garbage collection, not a precise alarm clock. 8. Declare `/suspend` and `/resume` as enabled image hooks only when the same source version serves them. Add shared hook-budget constants and route/contract assertions. Keep `/ready` and `/validate` AWS-silent. An old image without the new hooks must not be eligible for automatic suspension; enable policy only after deploying a compatible pinned image, with explicit capability/version gating if mixed versions can coexist. ## 6. Wire the supervisor and human decisions @@ -295,7 +307,7 @@ Implement the state/action table in a small testable policy/reconciliation helpe An approval can arrive before a pending suspend finishes. Even if an inline resume sees “already running,” the orchestrator must later notice that the machine became suspended and wake it. Do not clear durable wake intent merely because one API call appeared successful. -When `ResumeMicrovm` succeeds but the VM is still RESUMING, keep reconciling; grant a bounded recovery interval rather than immediately applying a pre-suspend stale heartbeat. When the agent restores task RUNNING, use the fresh timestamp from prerequisite 1C. Never exempt genuine crashed RUNNING tasks indefinitely. +When `ResumeMicrovm` is acknowledged but a subsequent observation does not yet confirm `RUNNING`, keep reconciling within a bounded recovery interval. Do not invent a `RESUMING` service state or treat an unknown/coarse state as confirmation. Defer the pre-suspend stale-heartbeat check only within that bound. When the agent restores task RUNNING, use the fresh timestamp from prerequisite 1C. Never exempt genuine crashed RUNNING tasks indefinitely. Add consecutive MicroVM poll-error tracking. Reset on successful observations; classify permanent failures separately from transient ones. At the chosen threshold, perform a final consistent task read, record an explicit infrastructure failure and finalize/terminate through the existing single-owner path. Emit recovery/orphan diagnostics when termination itself fails; do not silently lose the handle. Keep suspend failures distinguishable from lost compute: failure to save money can leave a task safely awake, whereas failure to wake threatens correctness and needs bounded escalation. diff --git a/docs/verification/645-p3-readiness-review.md b/docs/verification/645-p3-readiness-review.md index 5112c228a..7f0c402b2 100644 --- a/docs/verification/645-p3-readiness-review.md +++ b/docs/verification/645-p3-readiness-review.md @@ -8,6 +8,8 @@ This review covers the existing MicroVM implementation, related P2 follow-ups, c Further local work adds durable MicroVM start receipts, stable request tokens and bounded handle recovery. AWS token-retention/conflict behavior and unknown-ID cleanup remain live verification gates. The shared finalizer's repeated-decrement risk was subsequently reproduced and fixed locally with task-owned reservation transactions, unified counter writers and revision-guarded reconciliation, including approval waits. DynamoDB Local exercises the real transaction conditions; deployed IAM, scan scale and upgrade/drain behavior still need AWS verification. +The first local P3 foundation adds the supervisor's pause/wake command methods and fixes approval timing. The timer now keeps both an elapsed-time stopwatch and the original clock deadline, using whichever runs out first. A sleeping VM therefore does not get a fresh approval window. Automatic sleeping, guest safety hooks and supervisor repair logic remain unfinished. The installed AWS SDK does not expose a `RESUMING` state; receiving a wake acknowledgement alone does not prove that the VM is awake. + The reservation review found a separate trust boundary: the old agent role could write/replace/delete its task row, including coordinator metadata. A subsequent local fix restricts main-task writes to reporting/approval attributes, removes replacement/deletion permissions and removes unused worker counter grants. It also removes unused Python submission/session-info helpers and corrects overstated tenant-isolation comments. [Metadata verification](./645-coordinator-metadata.md) records the tests and pending AWS gate. Status reports still come from the agent, and the compute role chooses session tags; this is not complete hostile-worker isolation. ## Start here: the pieces in plain language From eb7e2071a3affa09ca0fd2f0064028545537d21f Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Sun, 13 Sep 2026 20:16:30 -0400 Subject: [PATCH 018/149] feat(compute): persist MicroVM sleep and wake intent (#645) --- cdk/src/handlers/shared/compute-strategy.ts | 12 +- .../shared/microvm-lifecycle-policy.ts | 120 ++++++++ cdk/src/handlers/shared/microvm-lifecycle.ts | 268 ++++++++++++++++++ .../strategies/lambda-microvm-strategy.ts | 18 +- .../constructs/agent-session-role.test.ts | 2 +- .../handlers/orchestrate-task-microvm.test.ts | 2 +- .../shared/microvm-lifecycle-local.test.ts | 258 +++++++++++++++++ .../handlers/shared/microvm-lifecycle.test.ts | 258 +++++++++++++++++ .../lambda-microvm-strategy.test.ts | 16 +- ...ADR-021-lambda-microvms-compute-backend.md | 4 +- docs/design/COMPUTE.md | 2 +- docs/src/content/docs/architecture/Compute.md | 2 +- ...Adr-021-lambda-microvms-compute-backend.md | 4 +- docs/verification/645-lifecycle-intent.md | 85 ++++++ .../645-p3-implementation-plan.md | 28 +- docs/verification/645-p3-readiness-review.md | 2 + 16 files changed, 1047 insertions(+), 34 deletions(-) create mode 100644 cdk/src/handlers/shared/microvm-lifecycle-policy.ts create mode 100644 cdk/src/handlers/shared/microvm-lifecycle.ts create mode 100644 cdk/test/handlers/shared/microvm-lifecycle-local.test.ts create mode 100644 cdk/test/handlers/shared/microvm-lifecycle.test.ts create mode 100644 docs/verification/645-lifecycle-intent.md diff --git a/cdk/src/handlers/shared/compute-strategy.ts b/cdk/src/handlers/shared/compute-strategy.ts index a2741c1ab..5b96cc302 100644 --- a/cdk/src/handlers/shared/compute-strategy.ts +++ b/cdk/src/handlers/shared/compute-strategy.ts @@ -17,6 +17,7 @@ * SOFTWARE. */ +import type { MicrovmState } from '@aws-sdk/client-lambda-microvms'; import type { BlueprintConfig, ComputeType } from './repo-config'; import { AgentCoreComputeStrategy } from './strategies/agentcore-strategy'; import { EcsComputeStrategy } from './strategies/ecs-strategy'; @@ -84,11 +85,18 @@ export type SessionHandle = * task/approval state, not by parsing this service-provided text. Keeping the * field on every variant preserves the same diagnostic shape. */ -export type SessionStatus = +/** UNKNOWN/NOT_FOUND are local observations, not AWS service states. */ +export type MicrovmObservedState = MicrovmState | 'UNKNOWN' | 'NOT_FOUND'; + +export type SessionStatus = ( | { readonly status: 'running'; readonly reason?: string } | { readonly status: 'suspended'; readonly reason?: string } | { readonly status: 'completed'; readonly reason?: string } - | { readonly status: 'failed'; readonly error: string; readonly reason?: string }; + | { readonly status: 'failed'; readonly error: string; readonly reason?: string } +) & { + /** Explicit MicroVM observation; coarse `running` also covers pending/unknown. */ + readonly microvmState?: MicrovmObservedState; +}; /** * `supported: true` means the lifecycle command was acknowledged, not that the diff --git a/cdk/src/handlers/shared/microvm-lifecycle-policy.ts b/cdk/src/handlers/shared/microvm-lifecycle-policy.ts new file mode 100644 index 000000000..feac21fc6 --- /dev/null +++ b/cdk/src/handlers/shared/microvm-lifecycle-policy.ts @@ -0,0 +1,120 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import type { SessionStatus } from './compute-strategy'; +import { intentMatchesGate, type MicrovmLifecycleSnapshot } from './microvm-lifecycle'; +import { TaskStatus, TERMINAL_STATUSES } from '../../constructs/task-status'; + +// Initial policy values, not service limits. Live timings must validate them. +export const MICROVM_SUSPEND_GRACE_MS = 30_000; +export const MICROVM_WAKE_MARGIN_MS = 60_000; +export const MICROVM_MIN_USEFUL_SLEEP_MS = 30_000; +export const MICROVM_TRANSITION_POLL_MS = 5_000; + +export interface MicrovmLifecyclePolicyInput { + readonly snapshot: MicrovmLifecycleSnapshot; + readonly substrate: SessionStatus; + readonly nowMs: number; + readonly sessionDeadlineMs: number; + readonly pollIntervalMs: number; + /** Stops new suspends; already-sleeping VMs can still wake or terminate. */ + readonly suspendEnabled: boolean; + /** True only for the compatible pinned image once hooks/barriers are deployed. */ + readonly imageSupportsLifecycle: boolean; +} + +export type MicrovmLifecycleDecision = ( + | { readonly action: 'wait' | 'terminate' | 'reconcile-terminal' } + | { readonly action: 'suspend' | 'resume'; readonly requestReady: boolean } +) & { readonly reason: string; readonly nextPollInMs: number }; + +/** Pure policy: no API calls, status changes, or approval decisions. */ +export function decideMicrovmLifecycle(input: MicrovmLifecyclePolicyInput): MicrovmLifecycleDecision { + const { snapshot, substrate, nowMs, sessionDeadlineMs, pollIntervalMs } = input; + if (![nowMs, sessionDeadlineMs, pollIntervalMs].every(value => Number.isSafeInteger(value) && value >= 0) || pollIntervalMs === 0) { + throw new Error('MicroVM lifecycle policy requires valid millisecond times and a positive poll interval'); + } + const state = substrate.microvmState ?? 'UNKNOWN'; + const terminal = state === 'TERMINATING' || state === 'TERMINATED' || state === 'NOT_FOUND'; + const nextPollInMs = Math.max(1, Math.min(pollIntervalMs, sessionDeadlineMs - nowMs)); + const transitionPoll = Math.min(nextPollInMs, MICROVM_TRANSITION_POLL_MS); + if (TERMINAL_STATUSES.some(status => status === snapshot.status) || snapshot.status === TaskStatus.FINALIZING) { + return { action: terminal ? 'wait' : 'terminate', reason: 'task-closed', nextPollInMs }; + } + if (terminal) return { action: 'reconcile-terminal', reason: 'substrate-terminal', nextPollInMs }; + if (nowMs >= sessionDeadlineMs) return { action: 'terminate', reason: 'session-deadline', nextPollInMs }; + if (state !== 'RUNNING' && state !== 'SUSPENDING' && state !== 'SUSPENDED') { + return { action: 'wait', reason: state === 'PENDING' ? 'starting' : 'unconfirmed-state', nextPollInMs: transitionPoll }; + } + + const sleeping = state === 'SUSPENDING' || state === 'SUSPENDED'; + const wake = (reason: string): MicrovmLifecycleDecision => ({ + action: 'resume', requestReady: state === 'SUSPENDED', reason, nextPollInMs: transitionPoll, + }); + const sameGate = intentMatchesGate(snapshot); + if (snapshot.status !== TaskStatus.AWAITING_APPROVAL) { + if (snapshot.status !== TaskStatus.RUNNING && snapshot.status !== TaskStatus.HYDRATING) { + return { action: 'wait', reason: 'task-not-active', nextPollInMs }; + } + if (sleeping || (snapshot.intent?.action === 'suspend')) return wake('suspended-outside-gate'); + return { action: 'wait', reason: 'working', nextPollInMs }; + } + + const approval = snapshot.approval; + if (approval.kind !== 'present') { + // A failed/missing read is not permission to sleep. Preserve wake intent + // even while RUNNING if an earlier suspend might still be in flight. + if (sleeping || snapshot.intent) return wake(`approval-${approval.kind}`); + return { action: 'wait', reason: `approval-${approval.kind}`, nextPollInMs: transitionPoll }; + } + if (approval.status !== 'PENDING') return wake('approval-terminal'); + if (sameGate && snapshot.intent?.deadline_ms !== approval.deadlineMs) return wake('approval-deadline-changed'); + if (sameGate && snapshot.intent?.action === 'resume') { + return sleeping ? wake('wake-intent') : { action: 'wait', reason: 'wake-intent', nextPollInMs: transitionPoll }; + } + if (approval.createdAtMs > nowMs) return wake('approval-time-invalid'); + const wakeAt = Math.min(approval.deadlineMs, sessionDeadlineMs) - MICROVM_WAKE_MARGIN_MS; + if (nowMs >= wakeAt) return wake('wake-deadline'); + + if (sleeping) { + if (!sameGate || snapshot.intent?.action !== 'suspend' || snapshot.intent.deadline_ms !== approval.deadlineMs) { + return wake('unintended-suspension'); + } + return { action: 'wait', reason: 'intentionally-suspended', nextPollInMs: Math.min(nextPollInMs, wakeAt - nowMs) }; + } + if (!input.suspendEnabled || !input.imageSupportsLifecycle) { + return { action: 'wait', reason: 'suspend-disabled', nextPollInMs }; + } + // A prior gate's in-flight suspend must be resolved conservatively. Persist a + // wake for this gate rather than attributing that old sleep request to it. + if (snapshot.intent?.action === 'suspend' && !sameGate) return wake('previous-gate-suspend'); + const graceEndsAt = approval.createdAtMs + MICROVM_SUSPEND_GRACE_MS; + if (nowMs < graceEndsAt) { + return { action: 'wait', reason: 'suspend-grace', nextPollInMs: Math.min(nextPollInMs, graceEndsAt - nowMs, wakeAt - nowMs) }; + } + if (wakeAt - nowMs < MICROVM_MIN_USEFUL_SLEEP_MS) { + return { action: 'wait', reason: 'short-window', nextPollInMs: Math.min(nextPollInMs, wakeAt - nowMs) }; + } + return { + action: 'suspend', + requestReady: true, + reason: 'pending-long-gate', + nextPollInMs: Math.min(transitionPoll, wakeAt - nowMs), + }; +} diff --git a/cdk/src/handlers/shared/microvm-lifecycle.ts b/cdk/src/handlers/shared/microvm-lifecycle.ts new file mode 100644 index 000000000..124885869 --- /dev/null +++ b/cdk/src/handlers/shared/microvm-lifecycle.ts @@ -0,0 +1,268 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import { randomUUID } from 'node:crypto'; +import { GetCommand, TransactWriteCommand } from '@aws-sdk/lib-dynamodb'; +import type { SessionHandle } from './compute-strategy'; +import type { ApprovalStatus } from './types'; +import { makeDocClient } from './ua'; +import { TaskStatus, type TaskStatusType } from '../../constructs/task-status'; + +type MicrovmHandle = Extract; +export type LifecycleAction = 'suspend' | 'resume'; + +/** Internal coordinator data. Retain it until task expiry; never erase a wake intent. */ +export interface MicrovmLifecycleIntent { + readonly version: 1; + readonly generation: string; + readonly microvm_id: string; + readonly request_id: string | null; + readonly action: LifecycleAction; + readonly requested_at_ms: number; + readonly deadline_ms: number | null; +} + +export type LifecycleApproval = + | { + readonly kind: 'present'; + readonly status: ApprovalStatus; + readonly created_at: string; + readonly timeout_s: number; + readonly createdAtMs: number; + readonly deadlineMs: number; + } + | { readonly kind: 'none' | 'missing' | 'invalid' } + | { readonly kind: 'unavailable'; readonly errorType: string }; + +/** A read is only an observation; saving intent rechecks identity and generation atomically. */ +export interface MicrovmLifecycleSnapshot { + readonly taskId: string; + readonly userId: string; + readonly status: TaskStatusType; + readonly handle: MicrovmHandle; + readonly requestId: string | null; + readonly intent?: MicrovmLifecycleIntent; + readonly approval: LifecycleApproval; +} + +export type SaveLifecycleResult = + | { readonly status: 'saved'; readonly intent: MicrovmLifecycleIntent } + | { readonly status: 'stale' | 'ineligible' }; + +// Per request/read sequence. A write plus lost-reply recovery can take two such +// budgets; the caller must also bound its whole reconciliation cycle. +export const MICROVM_LIFECYCLE_STORE_TIMEOUT_MS = 5_000; +const ddb = makeDocClient(); +const TASK_TABLE = process.env.TASK_TABLE_NAME!; +const APPROVALS_TABLE = process.env.TASK_APPROVALS_TABLE_NAME!; +const APPROVAL_STATUSES: readonly ApprovalStatus[] = ['PENDING', 'APPROVED', 'DENIED', 'TIMED_OUT', 'STRANDED']; +const LIVE_TASK_STATUSES: readonly TaskStatusType[] = [TaskStatus.HYDRATING, TaskStatus.RUNNING, TaskStatus.AWAITING_APPROVAL]; + +function nonblank(value: unknown): value is string { + return typeof value === 'string' && value.trim().length > 0; +} +function timestamp(value: unknown): value is number { + return typeof value === 'number' && Number.isSafeInteger(value) && value >= 0; +} +function validIntent(value: unknown): value is MicrovmLifecycleIntent { + if (!value || typeof value !== 'object') return false; + const item = value as MicrovmLifecycleIntent; + return item.version === 1 && nonblank(item.generation) && nonblank(item.microvm_id) + && (item.request_id === null || nonblank(item.request_id)) + && (item.action === 'suspend' || item.action === 'resume') + && timestamp(item.requested_at_ms) && (item.deadline_ms === null || timestamp(item.deadline_ms)) + && (item.action !== 'suspend' || (item.request_id !== null && item.deadline_ms !== null)); +} + +function parseApproval(row: Record | undefined, taskId: string, userId: string, requestId: string): LifecycleApproval { + if (!row) return { kind: 'missing' }; + if (row.task_id !== taskId || row.user_id !== userId || row.request_id !== requestId + || !APPROVAL_STATUSES.includes(row.status as ApprovalStatus) + || typeof row.created_at !== 'string' || !/^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(\.\d{3})?Z$/.test(row.created_at) + || !timestamp(row.timeout_s) || row.timeout_s === 0) return { kind: 'invalid' }; + const createdAtMs = Date.parse(row.created_at); + const canonical = row.created_at.includes('.') ? row.created_at : row.created_at.replace('Z', '.000Z'); + const deadlineMs = createdAtMs + row.timeout_s * 1000; + if (!timestamp(createdAtMs) || new Date(createdAtMs).toISOString() !== canonical || !timestamp(deadlineMs)) return { kind: 'invalid' }; + return { + kind: 'present', + status: row.status as ApprovalStatus, + created_at: row.created_at, + timeout_s: row.timeout_s, + createdAtMs, + deadlineMs, + }; +} + +/** Missing/non-MicroVM tasks are inapplicable. Invalid identity/state fails visibly. */ +export async function readMicrovmLifecycleSnapshot(taskId: string, userId: string): Promise { + const abortSignal = AbortSignal.timeout(MICROVM_LIFECYCLE_STORE_TIMEOUT_MS); + const result = await ddb.send(new GetCommand({ TableName: TASK_TABLE, Key: { task_id: taskId }, ConsistentRead: true }), { abortSignal }); + const task = result.Item; + if (!task) return undefined; + if (task.user_id !== userId || task.task_id !== taskId) throw new Error('MicroVM lifecycle task identity mismatch'); + if (task.compute_type !== 'lambda-microvm') return undefined; + const metadata = task.compute_metadata; + if (!nonblank(task.session_id) || metadata?.microvmId !== task.session_id || !nonblank(metadata?.endpoint) + || !Object.values(TaskStatus).includes(task.status)) throw new Error('MicroVM lifecycle task has invalid handle or status'); + const requestId = task.awaiting_approval_request_id ?? null; + if ((requestId !== null && !nonblank(requestId)) + || (task.status === TaskStatus.AWAITING_APPROVAL && requestId === null) + || ((task.status === TaskStatus.RUNNING || task.status === TaskStatus.HYDRATING) && requestId !== null)) { + throw new Error('MicroVM lifecycle task has inconsistent approval identity'); + } + if (task.microvm_lifecycle !== undefined && !validIntent(task.microvm_lifecycle)) { + throw new Error('MicroVM lifecycle intent is malformed or has an unsupported version'); + } + let approval: LifecycleApproval = { kind: 'none' }; + // Cancellation/terminal writers can retain the old gate pointer. They need + // cleanup, not another approval read or a wake, so preserve that identity only. + if (task.status === TaskStatus.AWAITING_APPROVAL && requestId !== null) { + try { + const response = await ddb.send(new GetCommand({ + TableName: APPROVALS_TABLE, + Key: { task_id: taskId, request_id: requestId }, + ConsistentRead: true, + }), { abortSignal }); + approval = parseApproval(response.Item, taskId, userId, requestId); + } catch (error) { + // This explicit observation forbids suspend and permits conservative wake. + // The caller must count/report the failure; no raw SDK message is retained. + const name = (error as { name?: unknown })?.name; + approval = { kind: 'unavailable', errorType: typeof name === 'string' && /^[A-Za-z0-9_]{1,100}$/.test(name) ? name : 'Error' }; + } + } + return { + taskId, + userId, + status: task.status, + requestId, + approval, + intent: task.microvm_lifecycle, + handle: { strategyType: 'lambda-microvm', sessionId: task.session_id, microvmId: metadata.microvmId, endpoint: metadata.endpoint }, + }; +} + +export function intentMatchesGate(snapshot: MicrovmLifecycleSnapshot): boolean { + return snapshot.intent?.microvm_id === snapshot.handle.microvmId && snapshot.intent.request_id === snapshot.requestId; +} + +function eligible(snapshot: MicrovmLifecycleSnapshot, action: LifecycleAction, nowMs: number): boolean { + if (!LIVE_TASK_STATUSES.includes(snapshot.status)) return false; + if (action === 'resume') return true; + return snapshot.status === TaskStatus.AWAITING_APPROVAL && snapshot.requestId !== null + && snapshot.approval.kind === 'present' && snapshot.approval.status === 'PENDING' + && nowMs >= snapshot.approval.createdAtMs && nowMs < snapshot.approval.deadlineMs + && !(intentMatchesGate(snapshot) && (snapshot.intent?.action === 'resume' + || snapshot.intent?.deadline_ms !== snapshot.approval.deadlineMs)); +} + +/** + * Save before touching AWS compute. A wake is sticky for this gate, including + * while AWS still reports RUNNING: an older suspend may be in flight. + * Caller must recheck the gate/time before suspend and reconcile after commands. + */ +export async function saveMicrovmLifecycleIntent( + snapshot: MicrovmLifecycleSnapshot, action: LifecycleAction, nowMs = Date.now(), +): Promise { + if (!timestamp(nowMs)) throw new Error('MicroVM lifecycle time must be epoch milliseconds'); + if (!eligible(snapshot, action, nowMs)) return { status: 'ineligible' }; + const sameGate = intentMatchesGate(snapshot); + // Repeated requests retain their original age/generation, so polling cannot + // reset the eventual recovery budget. A new gate/action gets a new generation. + const intent: MicrovmLifecycleIntent = sameGate && snapshot.intent?.action === action ? snapshot.intent : { + version: 1, + generation: randomUUID(), + microvm_id: snapshot.handle.microvmId, + request_id: snapshot.requestId, + action, + requested_at_ms: nowMs, + deadline_ms: snapshot.approval.kind === 'present' ? snapshot.approval.deadlineMs : sameGate ? snapshot.intent!.deadline_ms : null, + }; + const names: Record = { '#status': 'status' }; + const values: Record = { + ':intent': intent, + ':user': snapshot.userId, + ':status': snapshot.status, + ':type': 'lambda-microvm', + ':id': snapshot.handle.microvmId, + ':endpoint': snapshot.handle.endpoint, + }; + let condition = 'user_id = :user AND #status = :status AND compute_type = :type AND session_id = :id ' + + 'AND compute_metadata.microvmId = :id AND compute_metadata.endpoint = :endpoint'; + if (snapshot.requestId === null) { + condition += ' AND attribute_not_exists(awaiting_approval_request_id)'; + } else { + condition += ' AND awaiting_approval_request_id = :request'; + values[':request'] = snapshot.requestId; + } + if (!snapshot.intent) { + condition += ' AND attribute_not_exists(microvm_lifecycle)'; + } else { + condition += ' AND microvm_lifecycle.generation = :generation'; + values[':generation'] = snapshot.intent.generation; + } + const command = new TransactWriteCommand({ + // Separate from the persistent generation: a repeated save has a different + // condition shape, so reusing that generation as an AWS token would conflict. + ClientRequestToken: randomUUID(), + TransactItems: [ + { + Update: { + TableName: TASK_TABLE, + Key: { task_id: snapshot.taskId }, + UpdateExpression: 'SET microvm_lifecycle = :intent', + ConditionExpression: condition, + ExpressionAttributeNames: names, + ExpressionAttributeValues: values, + }, + }, + ...(action === 'suspend' && snapshot.approval.kind === 'present' ? [{ + ConditionCheck: { + TableName: APPROVALS_TABLE, + Key: { task_id: snapshot.taskId, request_id: snapshot.requestId }, + ConditionExpression: '#status = :pending AND user_id = :user AND created_at = :created AND timeout_s = :timeout', + ExpressionAttributeNames: { '#status': 'status' }, + ExpressionAttributeValues: { + ':pending': 'PENDING', ':user': snapshot.userId, ':created': snapshot.approval.created_at, ':timeout': snapshot.approval.timeout_s, + }, + }, + }] : []), + ], + }); + try { + await ddb.send(command, { abortSignal: AbortSignal.timeout(MICROVM_LIFECYCLE_STORE_TIMEOUT_MS) }); + return { status: 'saved', intent }; + } catch (error) { + const failure = error as { name?: string; CancellationReasons?: { Code?: string }[] }; + if (failure?.name === 'TransactionCanceledException' + && failure.CancellationReasons?.some(reason => reason.Code === 'ConditionalCheckFailed')) return { status: 'stale' }; + // Lost committed reply: observe exactly our generation and unchanged task + // identity before reporting success. Unknown/unreadable outcomes stay errors. + const current = await readMicrovmLifecycleSnapshot(snapshot.taskId, snapshot.userId); + if (current?.intent?.generation === intent.generation) { + if (current.status === snapshot.status && current.requestId === snapshot.requestId + && current.handle.microvmId === snapshot.handle.microvmId && current.handle.endpoint === snapshot.handle.endpoint + && eligible(current, action, Date.now())) return { status: 'saved', intent: current.intent }; + // Our write committed, but cancellation/a decision moved the task on. + return { status: 'stale' }; + } + throw error; + } +} diff --git a/cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts b/cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts index b5e36803d..a924844a0 100644 --- a/cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts +++ b/cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts @@ -657,9 +657,9 @@ export class LambdaMicrovmComputeStrategy implements ComputeStrategy { * because the VM is already on its way to frozen; reporting ``running`` * would tell the orchestrator compute is still progressing when it is not. * Both map to a state the orchestrator treats as benign-or-anomalous - * depending on the task status, never as a failure. (``SUSPENDING`` was - * never observable live — suspend reaches ``SUSPENDED`` in under a second - * — so nothing may WAIT for it; it is mapped for completeness only.) + * depending on the task status, never as a failure. Earlier probes skipped + * this short transition, but P3 must handle it if observed: save wake intent + * and wait for SUSPENDED before issuing ResumeMicrovm. * - ``TERMINATING`` / ``TERMINATED`` → ``completed``. Both are terminal or * terminal-bound and carry no exit code, so "the substrate is gone" is all * the strategy can honestly say; whether that is success or failure is the @@ -669,6 +669,8 @@ export class LambdaMicrovmComputeStrategy implements ComputeStrategy { * that waited for NotFound would spin on a finished VM. * - anything else (an unrecognized future state) → ``running``, so a service * enum addition can never fail a healthy task. + * ``microvmState`` also reports the explicit observed state (or local UNKNOWN / + * NOT_FOUND). P3 uses it to distinguish readiness from the coarse status. * * ``stateReason`` is carried through on every mapped state as * ``SessionStatus.reason``, VERBATIM and uninterpreted. It is the substrate's @@ -729,7 +731,7 @@ export class LambdaMicrovmComputeStrategy implements ComputeStrategy { logger.info('MicroVM not found on poll — treating as terminal', { microvm_id: microvmId, }); - return { status: 'completed' }; + return { status: 'completed', microvmState: 'NOT_FOUND' }; } throw wrapMicrovmError('GetMicrovm', err); } @@ -737,10 +739,10 @@ export class LambdaMicrovmComputeStrategy implements ComputeStrategy { switch (state) { case MicrovmState.PENDING: case MicrovmState.RUNNING: - return { status: 'running', ...(stateReason && { reason: stateReason }) }; + return { status: 'running', microvmState: state, ...(stateReason && { reason: stateReason }) }; case MicrovmState.SUSPENDING: case MicrovmState.SUSPENDED: - return { status: 'suspended', ...(stateReason && { reason: stateReason }) }; + return { status: 'suspended', microvmState: state, ...(stateReason && { reason: stateReason }) }; case MicrovmState.TERMINATING: case MicrovmState.TERMINATED: if (stateReason) { @@ -754,14 +756,14 @@ export class LambdaMicrovmComputeStrategy implements ComputeStrategy { state_reason: stateReason, }); } - return { status: 'completed', ...(stateReason && { reason: stateReason }) }; + return { status: 'completed', microvmState: state, ...(stateReason && { reason: stateReason }) }; default: logger.warn('Unrecognized MicroVM state — reporting running', { microvm_id: microvmId, state, ...(stateReason && { state_reason: stateReason }), }); - return { status: 'running', ...(stateReason && { reason: stateReason }) }; + return { status: 'running', microvmState: 'UNKNOWN', ...(stateReason && { reason: stateReason }) }; } } diff --git a/cdk/test/constructs/agent-session-role.test.ts b/cdk/test/constructs/agent-session-role.test.ts index f1b2c818a..cfd9fd938 100644 --- a/cdk/test/constructs/agent-session-role.test.ts +++ b/cdk/test/constructs/agent-session-role.test.ts @@ -157,7 +157,7 @@ describe('AgentSessionRole construct', () => { expect(attrs).toEqual(taskWriteAttributes); expect(s.Condition.Null['dynamodb:Attributes']).toBe('false'); for (const protectedAttribute of [ - 'microvm_start', 'concurrency_slot', 'user_id', 'created_at', + 'microvm_start', 'microvm_lifecycle', 'concurrency_slot', 'user_id', 'created_at', 'session_id', 'compute_type', 'compute_metadata', 'agent_runtime_arn', ]) { expect(attrs).not.toContain(protectedAttribute); diff --git a/cdk/test/handlers/orchestrate-task-microvm.test.ts b/cdk/test/handlers/orchestrate-task-microvm.test.ts index 55dd88c8c..abb33908f 100644 --- a/cdk/test/handlers/orchestrate-task-microvm.test.ts +++ b/cdk/test/handlers/orchestrate-task-microvm.test.ts @@ -649,7 +649,7 @@ describe('orchestrate-task for a lambda-microvm task', () => { expect(args.microvmId).toBe(MICROVM_ID); expect(args.ddbStatus).toBe(TaskStatus.RUNNING); // The strategy's mechanical mapping is what the orchestrator interprets. - expect(args.substrate).toEqual({ status: 'suspended' }); + expect(args.substrate).toEqual({ status: 'suspended', microvmState: 'SUSPENDED' }); }); test('returns a failed poll state when reconciliation fails the task', async () => { diff --git a/cdk/test/handlers/shared/microvm-lifecycle-local.test.ts b/cdk/test/handlers/shared/microvm-lifecycle-local.test.ts new file mode 100644 index 000000000..9c446a9a9 --- /dev/null +++ b/cdk/test/handlers/shared/microvm-lifecycle-local.test.ts @@ -0,0 +1,258 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +/** Opt-in DynamoDB Local: loopback endpoint and dummy credentials only. */ +import { randomUUID } from 'node:crypto'; +import { CreateTableCommand, DeleteTableCommand, DynamoDBClient } from '@aws-sdk/client-dynamodb'; +import { DeleteCommand, DynamoDBDocumentClient, GetCommand, PutCommand, ScanCommand, TransactWriteCommand, UpdateCommand } from '@aws-sdk/lib-dynamodb'; + +const endpoint = process.env.ABCA_DDB_LOCAL_ENDPOINT; +if (endpoint && (new URL(endpoint).hostname !== '127.0.0.1' || new URL(endpoint).protocol !== 'http:')) { + throw new Error('Lifecycle integration tests require an http://127.0.0.1 DynamoDB Local endpoint'); +} +const mockBeforeSend = jest.fn(); +const mockAfterSend = jest.fn(); +const mockClients: DynamoDBDocumentClient[] = []; +jest.mock('../../../src/handlers/shared/ua', () => { + const actual = jest.requireActual('../../../src/handlers/shared/ua'); + return { + ...actual, + makeDocClient: () => { + const client = actual.makeDocClient({ + endpoint: process.env.ABCA_DDB_LOCAL_ENDPOINT ?? 'http://127.0.0.1:1', + region: 'us-east-1', + credentials: { accessKeyId: 'local', secretAccessKey: 'local' }, + }); + const send = client.send.bind(client); + client.send = async (command: unknown, options: unknown) => { + await mockBeforeSend(command); + const result = await send(command, options); + await mockAfterSend(command, result); + return result; + }; + mockClients.push(client); + return client; + }, + }; +}); +const suffix = randomUUID(); +const tasks = `lifecycle-tasks-${suffix}`; +const approvals = `lifecycle-approvals-${suffix}`; +Object.assign(process.env, { TASK_TABLE_NAME: tasks, TASK_APPROVALS_TABLE_NAME: approvals }); +import { readMicrovmLifecycleSnapshot, saveMicrovmLifecycleIntent } from '../../../src/handlers/shared/microvm-lifecycle'; + +const raw = new DynamoDBClient({ + endpoint: endpoint ?? 'http://127.0.0.1:1', + region: 'us-east-1', + credentials: { accessKeyId: 'local', secretAccessKey: 'local' }, +}); +const admin = DynamoDBDocumentClient.from(raw); +const local = endpoint ? describe : describe.skip; +jest.setTimeout(30_000); + +local('MicroVM lifecycle against DynamoDB Local', () => { + beforeAll(async () => { + for (const name of [tasks, approvals]) { + await raw.send(new CreateTableCommand({ + TableName: name, + BillingMode: 'PAY_PER_REQUEST', + AttributeDefinitions: [{ AttributeName: 'task_id', AttributeType: 'S' }, + ...(name === approvals ? [{ AttributeName: 'request_id', AttributeType: 'S' as const }] : [])], + KeySchema: [{ AttributeName: 'task_id', KeyType: 'HASH' }, + ...(name === approvals ? [{ AttributeName: 'request_id', KeyType: 'RANGE' as const }] : [])], + })); + } + }); + beforeEach(async () => { + mockBeforeSend.mockReset(); + mockAfterSend.mockReset(); + for (const name of [tasks, approvals]) { + const result = await admin.send(new ScanCommand({ TableName: name })); + for (const item of result.Items ?? []) { + await admin.send(new DeleteCommand({ + TableName: name, + Key: { + task_id: item.task_id, + ...(name === approvals ? { request_id: item.request_id } : {}), + }, + })); + } + } + await admin.send(new PutCommand({ + TableName: tasks, + Item: { + task_id: 'task', + user_id: 'user', + status: 'AWAITING_APPROVAL', + compute_type: 'lambda-microvm', + session_id: 'vm', + compute_metadata: { microvmId: 'vm', endpoint: 'https://vm.example' }, + awaiting_approval_request_id: 'gate', + }, + })); + await approval('gate'); + }); + afterAll(async () => { + try { for (const name of [tasks, approvals]) await raw.send(new DeleteTableCommand({ TableName: name })); } finally { for (const client of mockClients) client.destroy(); raw.destroy(); } + }); + async function approval(requestId: string) { + await admin.send(new PutCommand({ + TableName: approvals, + Item: { + task_id: 'task', + request_id: requestId, + user_id: 'user', + status: 'PENDING', + created_at: new Date(Date.now() - 45_000).toISOString(), + timeout_s: 600, + }, + })); + } + async function current() { + const value = await readMicrovmLifecycleSnapshot('task', 'user'); + if (!value) throw new Error('Expected a MicroVM task'); + return value; + } + async function taskRow() { + return (await admin.send(new GetCommand({ TableName: tasks, Key: { task_id: 'task' }, ConsistentRead: true }))).Item!; + } + async function set(table: string, field: string, value: unknown) { + await admin.send(new UpdateCommand({ + TableName: table, + Key: { task_id: 'task', ...(table === approvals && { request_id: 'gate' }) }, + UpdateExpression: 'SET #field = :value', + ExpressionAttributeNames: { '#field': field }, + ExpressionAttributeValues: { ':value': value }, + })); + } + + test('suspend atomically records only coordinator intent and original deadline', async () => { + const observed = await current(); + const result = await saveMicrovmLifecycleIntent(observed, 'suspend'); + expect(result.status).toBe('saved'); + const saved = await taskRow(); + expect(saved.status).toBe('AWAITING_APPROVAL'); + expect(saved.compute_metadata).toEqual({ microvmId: 'vm', endpoint: 'https://vm.example' }); + expect(saved.microvm_lifecycle).toMatchObject({ + action: 'suspend', + request_id: 'gate', + microvm_id: 'vm', + deadline_ms: observed.approval.kind === 'present' ? observed.approval.deadlineMs : NaN, + }); + }); + test('a wake blocks an older absent-record sleep and remains sticky on fresh reads', async () => { + const old = await current(); + await saveMicrovmLifecycleIntent(await current(), 'resume'); + expect(await saveMicrovmLifecycleIntent(old, 'suspend')).toEqual({ status: 'stale' }); + expect(await saveMicrovmLifecycleIntent(await current(), 'suspend')).toEqual({ status: 'ineligible' }); + expect((await taskRow()).microvm_lifecycle.action).toBe('resume'); + }); + test('a new wake generation fences a previously recorded suspend', async () => { + await saveMicrovmLifecycleIntent(await current(), 'suspend'); + const oldSleep = await current(); + await saveMicrovmLifecycleIntent(await current(), 'resume'); + expect(await saveMicrovmLifecycleIntent(oldSleep, 'suspend')).toEqual({ status: 'stale' }); + expect((await taskRow()).microvm_lifecycle.generation).not.toBe(oldSleep.intent?.generation); + }); + test('a later gate may sleep without allowing a late writer from the earlier gate', async () => { + await saveMicrovmLifecycleIntent(await current(), 'resume'); + const oldGate = await current(); + await approval('gate-two'); + await set(tasks, 'awaiting_approval_request_id', 'gate-two'); + expect(await saveMicrovmLifecycleIntent(oldGate, 'resume')).toEqual({ status: 'stale' }); + expect((await saveMicrovmLifecycleIntent(await current(), 'suspend')).status).toBe('saved'); + expect((await taskRow()).microvm_lifecycle.request_id).toBe('gate-two'); + }); + test.each([ + ['status', 'CANCELLED'], ['user_id', 'other-user'], ['session_id', 'other-vm'], + ['compute_type', 'ecs'], ['awaiting_approval_request_id', 'different-gate'], + ['compute_metadata', { microvmId: 'vm', endpoint: 'https://replacement.example' }], + ])('task %s changing after the read rejects intent', async (field, value) => { + const old = await current(); + await set(tasks, field as string, value); + expect(await saveMicrovmLifecycleIntent(old, 'suspend')).toEqual({ status: 'stale' }); + expect((await taskRow()).microvm_lifecycle).toBeUndefined(); + }); + test.each([ + ['status', 'APPROVED'], ['status', 'DENIED'], ['status', 'TIMED_OUT'], + ['created_at', '2020-01-01T00:00:00Z'], ['timeout_s', 10], ['user_id', 'other-user'], + ])('approval %s changing after the read rolls back the entire suspend write', async (field, value) => { + const old = await current(); + await set(approvals, field as string, value); + expect(await saveMicrovmLifecycleIntent(old, 'suspend')).toEqual({ status: 'stale' }); + expect((await taskRow()).microvm_lifecycle).toBeUndefined(); + }); + test('missing gate rejects sleep but permits conservative wake', async () => { + const old = await current(); + await admin.send(new DeleteCommand({ TableName: approvals, Key: { task_id: 'task', request_id: 'gate' } })); + expect(await saveMicrovmLifecycleIntent(old, 'suspend')).toEqual({ status: 'stale' }); + expect((await saveMicrovmLifecycleIntent(await current(), 'resume')).status).toBe('saved'); + }); + test('lost committed reply recovers the exact generation from the database', async () => { + const observed = await current(); + mockAfterSend.mockImplementationOnce(async command => { + expect(command).toBeInstanceOf(TransactWriteCommand); + throw new Error('lost committed response'); + }); + const result = await saveMicrovmLifecycleIntent(observed, 'suspend'); + expect(result).toEqual({ status: 'saved', intent: (await taskRow()).microvm_lifecycle }); + }); + test.each(['cancel', 'approve'])('%s after a committed write but before lost-reply recovery is stale', async change => { + const observed = await current(); + mockAfterSend.mockImplementationOnce(async command => { + expect(command).toBeInstanceOf(TransactWriteCommand); + await set(change === 'cancel' ? tasks : approvals, 'status', change === 'cancel' ? 'CANCELLED' : 'APPROVED'); + throw new Error('lost committed response'); + }); + expect(await saveMicrovmLifecycleIntent(observed, 'suspend')).toEqual({ status: 'stale' }); + }); + test('fresh module/client observes the saved wake without resetting age or generation', async () => { + await saveMicrovmLifecycleIntent(await current(), 'resume'); + const saved = (await taskRow()).microvm_lifecycle; + let restarted: typeof import('../../../src/handlers/shared/microvm-lifecycle'); + jest.isolateModules(() => { restarted = jest.requireActual('../../../src/handlers/shared/microvm-lifecycle'); }); + const observed = await restarted!.readMicrovmLifecycleSnapshot('task', 'user'); + expect(observed?.intent).toEqual(saved); + expect(await restarted!.saveMicrovmLifecycleIntent(observed!, 'resume')).toEqual({ status: 'saved', intent: saved }); + expect((await taskRow()).microvm_lifecycle).toEqual(saved); + }); + test('competing sleepers converge; a later wake fences both old snapshots', async () => { + const first = await current(); + const second = await current(); + const results = await Promise.allSettled([saveMicrovmLifecycleIntent(first, 'suspend'), saveMicrovmLifecycleIntent(second, 'suspend')]); + expect(results.filter(result => result.status === 'fulfilled' && result.value.status === 'saved')).toHaveLength(1); + for (const result of results) { + if (result.status === 'rejected') expect(result.reason.name).toBe('TransactionCanceledException'); + } + await saveMicrovmLifecycleIntent(await current(), 'resume'); + expect(await saveMicrovmLifecycleIntent(first, 'suspend')).toEqual({ status: 'stale' }); + expect(await saveMicrovmLifecycleIntent(second, 'suspend')).toEqual({ status: 'stale' }); + }); + test('approval-read failure is explicit and cannot prevent recording a wake', async () => { + mockBeforeSend.mockImplementation(async command => { + if (command instanceof GetCommand && command.input.TableName === approvals) { + throw Object.assign(new Error('simulated authorization failure'), { name: 'AccessDeniedException' }); + } + }); + const observed = await current(); + expect(observed.approval).toEqual({ kind: 'unavailable', errorType: 'AccessDeniedException' }); + expect(await saveMicrovmLifecycleIntent(observed, 'suspend')).toEqual({ status: 'ineligible' }); + expect((await saveMicrovmLifecycleIntent(observed, 'resume')).status).toBe('saved'); + }); +}); diff --git a/cdk/test/handlers/shared/microvm-lifecycle.test.ts b/cdk/test/handlers/shared/microvm-lifecycle.test.ts new file mode 100644 index 000000000..6436d6bf1 --- /dev/null +++ b/cdk/test/handlers/shared/microvm-lifecycle.test.ts @@ -0,0 +1,258 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +const mockSend = jest.fn(); +jest.mock('../../../src/handlers/shared/ua', () => ({ makeDocClient: () => ({ send: mockSend }) })); +process.env.TASK_TABLE_NAME = 'LifecycleTasks'; +process.env.TASK_APPROVALS_TABLE_NAME = 'LifecycleApprovals'; + +import { GetCommand, TransactWriteCommand } from '@aws-sdk/lib-dynamodb'; +import type { MicrovmObservedState } from '../../../src/handlers/shared/compute-strategy'; +import { + readMicrovmLifecycleSnapshot, saveMicrovmLifecycleIntent, type MicrovmLifecycleSnapshot, + type MicrovmLifecycleIntent, +} from '../../../src/handlers/shared/microvm-lifecycle'; +import { decideMicrovmLifecycle, type MicrovmLifecyclePolicyInput } from '../../../src/handlers/shared/microvm-lifecycle-policy'; + +const NOW = 1_800_000_000_000; +const CREATED = new Date(NOW - 45_000).toISOString(); +const DEADLINE = NOW + 555_000; +const task = { + task_id: 'task', + user_id: 'user', + status: 'AWAITING_APPROVAL', + compute_type: 'lambda-microvm', + session_id: 'vm', + compute_metadata: { microvmId: 'vm', endpoint: 'https://vm.example' }, + awaiting_approval_request_id: 'gate', +}; +const row = { task_id: 'task', user_id: 'user', request_id: 'gate', status: 'PENDING', created_at: CREATED, timeout_s: 600 }; +const intent = (action: 'suspend' | 'resume' = 'suspend'): MicrovmLifecycleIntent => ({ + version: 1, + generation: 'generation-one', + microvm_id: 'vm', + request_id: 'gate', + action, + requested_at_ms: NOW - 10_000, + deadline_ms: DEADLINE, +}); +const snapshot = (overrides: Partial = {}): MicrovmLifecycleSnapshot => ({ + taskId: 'task', + userId: 'user', + status: 'AWAITING_APPROVAL', + requestId: 'gate', + handle: { strategyType: 'lambda-microvm', sessionId: 'vm', microvmId: 'vm', endpoint: 'https://vm.example' }, + approval: { kind: 'present', status: 'PENDING', created_at: CREATED, timeout_s: 600, createdAtMs: NOW - 45_000, deadlineMs: DEADLINE }, + ...overrides, +}); +const policy = (state: MicrovmObservedState = 'RUNNING', change: Partial = {}) => decideMicrovmLifecycle({ + snapshot: snapshot(), + substrate: { status: 'running', microvmState: state }, + nowMs: NOW, + sessionDeadlineMs: NOW + 3_600_000, + pollIntervalMs: 30_000, + suspendEnabled: true, + imageSupportsLifecycle: true, + ...change, +}); + +beforeEach(() => { + mockSend.mockReset(); + jest.spyOn(Date, 'now').mockReturnValue(NOW); +}); +afterEach(() => jest.restoreAllMocks()); + +describe('MicroVM lifecycle policy', () => { + test('long pending gate may suspend only after grace and explicit RUNNING', () => { + expect(policy()).toEqual({ action: 'suspend', requestReady: true, reason: 'pending-long-gate', nextPollInMs: 5_000 }); + expect(mockSend).not.toHaveBeenCalled(); + }); + test.each(['PENDING', 'UNKNOWN'] as const)('%s is not proof of awake state', state => { + expect(policy(state)).toMatchObject({ action: 'wait', nextPollInMs: 5_000 }); + }); + test('legacy coarse running without a service observation cannot trigger suspend', () => { + expect(policy('RUNNING', { substrate: { status: 'running' } })).toMatchObject({ action: 'wait', reason: 'unconfirmed-state' }); + }); + test.each([{ suspendEnabled: false }, { imageSupportsLifecycle: false }])('requires both enable and compatible image: %j', change => { + expect(policy('RUNNING', change)).toMatchObject({ action: 'wait', reason: 'suspend-disabled' }); + }); + test('long poll setting is clamped to the end of grace', () => { + const gate = snapshot(); + expect(policy('RUNNING', { nowMs: NOW - 20_000, snapshot: gate, pollIntervalMs: 600_000 })) + .toMatchObject({ action: 'wait', reason: 'suspend-grace', nextPollInMs: 5_000 }); + }); + test('too little useful sleep keeps the VM awake', () => { + expect(policy('RUNNING', { nowMs: DEADLINE - 80_000 })) + .toMatchObject({ action: 'wait', reason: 'short-window', nextPollInMs: 20_000 }); + expect(policy('RUNNING', { nowMs: DEADLINE - 90_000 })).toMatchObject({ action: 'suspend' }); + }); + test('intended sleep waits until wake margin even with an oversized poll interval', () => { + expect(policy('SUSPENDED', { snapshot: snapshot({ intent: intent() }), pollIntervalMs: 900_000, suspendEnabled: false })) + .toEqual({ action: 'wait', reason: 'intentionally-suspended', nextPollInMs: DEADLINE - NOW - 60_000 }); + }); + test.each(['APPROVED', 'DENIED', 'TIMED_OUT', 'STRANDED'] as const)('%s always preserves wake intent', status => { + const pending = snapshot().approval; + if (pending.kind !== 'present') throw new Error('fixture'); + for (const state of ['RUNNING', 'SUSPENDING', 'SUSPENDED'] as const) { + expect(policy(state, { snapshot: snapshot({ approval: { ...pending, status }, intent: intent() }), suspendEnabled: false })) + .toMatchObject({ action: 'resume', requestReady: state === 'SUSPENDED', reason: 'approval-terminal' }); + } + }); + test.each([DEADLINE - 60_000, DEADLINE + 1000])('wakes at/past the original deadline margin (%s)', nowMs => { + expect(policy('SUSPENDED', { nowMs, snapshot: snapshot({ intent: intent() }) })) + .toMatchObject({ action: 'resume', requestReady: true, reason: 'wake-deadline' }); + }); + test('session lifetime also bounds the wake margin', () => { + expect(policy('SUSPENDED', { sessionDeadlineMs: NOW + 30_000, snapshot: snapshot({ intent: intent() }) })) + .toMatchObject({ action: 'resume', reason: 'wake-deadline' }); + expect(policy('SUSPENDED', { sessionDeadlineMs: NOW })).toMatchObject({ action: 'terminate', reason: 'session-deadline' }); + }); + test.each(['SUSPENDING', 'SUSPENDED'] as const)('unintended %s is repaired, but resume waits until SUSPENDED', state => { + expect(policy(state)).toMatchObject({ action: 'resume', requestReady: state === 'SUSPENDED', reason: 'unintended-suspension' }); + }); + test('acknowledged wake remains sticky even when a delayed suspend finishes later', () => { + const current = snapshot({ intent: intent('resume') }); + expect(policy('RUNNING', { snapshot: current })).toMatchObject({ action: 'wait', reason: 'wake-intent' }); + expect(policy('SUSPENDING', { snapshot: current })).toMatchObject({ action: 'resume', requestReady: false }); + expect(policy('SUSPENDED', { snapshot: current })).toMatchObject({ action: 'resume', requestReady: true }); + }); + test('another gate or changed deadline cannot inherit an old sleep intent', () => { + expect(policy('RUNNING', { snapshot: snapshot({ intent: { ...intent(), request_id: 'old-gate' } }) })) + .toMatchObject({ action: 'resume', reason: 'previous-gate-suspend' }); + expect(policy('SUSPENDED', { snapshot: snapshot({ intent: { ...intent(), deadline_ms: DEADLINE + 1 } }) })) + .toMatchObject({ action: 'resume', reason: 'approval-deadline-changed' }); + }); + test.each(['missing', 'invalid', 'unavailable'] as const)('%s approval forbids sleep and wakes a sleeping VM', kind => { + const current = snapshot({ approval: kind === 'unavailable' ? { kind, errorType: 'AccessDeniedException' } : { kind } }); + expect(policy('RUNNING', { snapshot: current })).toMatchObject({ action: 'wait' }); + expect(policy('SUSPENDED', { snapshot: current })).toMatchObject({ action: 'resume', requestReady: true }); + expect(policy('RUNNING', { snapshot: { ...current, intent: intent() } })).toMatchObject({ action: 'resume', requestReady: false }); + }); + test.each(['COMPLETED', 'FAILED', 'CANCELLED', 'TIMED_OUT', 'FINALIZING'] as const)('%s tasks are never revived', status => { + expect(policy('SUSPENDED', { snapshot: snapshot({ status }) })).toMatchObject({ action: 'terminate' }); + }); + test.each(['TERMINATING', 'TERMINATED', 'NOT_FOUND'] as const)('%s triggers task reconciliation', state => { + expect(policy(state)).toMatchObject({ action: 'reconcile-terminal' }); + expect(policy(state, { snapshot: snapshot({ status: 'COMPLETED' }) })).toMatchObject({ action: 'wait' }); + }); + test('a working task is never suspended for lack of traffic; unexpected sleep recovers', () => { + const current = snapshot({ status: 'RUNNING', requestId: null, approval: { kind: 'none' } }); + expect(policy('RUNNING', { snapshot: current })).toMatchObject({ action: 'wait', reason: 'working' }); + expect(policy('SUSPENDED', { snapshot: current })).toMatchObject({ action: 'resume', reason: 'suspended-outside-gate' }); + }); + test.each([0, -1, NaN, Infinity])('rejects invalid poll interval %s', pollIntervalMs => { + expect(() => policy('RUNNING', { pollIntervalMs })).toThrow('positive poll interval'); + }); +}); + +describe('MicroVM lifecycle store', () => { + test('a cancelled task may retain its gate pointer and needs no approval read', async () => { + mockSend.mockResolvedValueOnce({ Item: { ...task, status: 'CANCELLED' } }); + const closed = await readMicrovmLifecycleSnapshot('task', 'user'); + expect(closed).toMatchObject({ status: 'CANCELLED', requestId: 'gate', approval: { kind: 'none' } }); + expect(await saveMicrovmLifecycleIntent(closed!, 'resume', NOW)).toEqual({ status: 'ineligible' }); + expect(mockSend).toHaveBeenCalledTimes(1); + }); + test('reads current task and only its current gate consistently with one bounded read budget', async () => { + mockSend.mockResolvedValueOnce({ Item: task }).mockResolvedValueOnce({ Item: row }); + expect(await readMicrovmLifecycleSnapshot('task', 'user')).toEqual(snapshot()); + expect(mockSend.mock.calls.map(([command]) => command.input)).toEqual([ + { TableName: 'LifecycleTasks', Key: { task_id: 'task' }, ConsistentRead: true }, + { TableName: 'LifecycleApprovals', Key: { task_id: 'task', request_id: 'gate' }, ConsistentRead: true }, + ]); + expect(mockSend.mock.calls[0][1].abortSignal).toBe(mockSend.mock.calls[1][1].abortSignal); + }); + test.each([undefined, { ...task, compute_type: 'ecs' }])('missing/non-MicroVM task is inapplicable', async item => { + mockSend.mockResolvedValueOnce({ Item: item }); + await expect(readMicrovmLifecycleSnapshot('task', 'user')).resolves.toBeUndefined(); + }); + test.each([ + { user_id: 'other' }, { session_id: 'other-vm' }, { compute_metadata: {} }, + { awaiting_approval_request_id: null }, { status: 'RUNNING' }, + { microvm_lifecycle: { ...intent(), version: 99 } }, + ])('invalid task/intent fails visibly: %j', async change => { + mockSend.mockResolvedValueOnce({ Item: { ...task, ...change } }); + await expect(readMicrovmLifecycleSnapshot('task', 'user')).rejects.toThrow('MicroVM lifecycle'); + expect(mockSend).toHaveBeenCalledTimes(1); + }); + test.each([ + { user_id: 'other' }, { request_id: 'old-gate' }, { status: 'MADE_UP' }, + { created_at: '2026-02-30T00:00:00Z' }, { created_at: 'not-a-date' }, + { timeout_s: -1 }, { timeout_s: Number.MAX_SAFE_INTEGER }, + ])('bad approval data remains explicitly invalid: %j', async change => { + mockSend.mockResolvedValueOnce({ Item: task }).mockResolvedValueOnce({ Item: { ...row, ...change } }); + expect((await readMicrovmLifecycleSnapshot('task', 'user'))?.approval).toEqual({ kind: 'invalid' }); + }); + test('missing approval and failed approval reads remain distinguishable', async () => { + mockSend.mockResolvedValueOnce({ Item: task }).mockResolvedValueOnce({}); + expect((await readMicrovmLifecycleSnapshot('task', 'user'))?.approval).toEqual({ kind: 'missing' }); + mockSend.mockResolvedValueOnce({ Item: task }).mockRejectedValueOnce(Object.assign(new Error('private data'), { name: 'AccessDeniedException' })); + expect((await readMicrovmLifecycleSnapshot('task', 'user'))?.approval).toEqual({ kind: 'unavailable', errorType: 'AccessDeniedException' }); + }); + test('suspend records intent with an atomic exact-gate condition', async () => { + mockSend.mockResolvedValueOnce({}); + const saved = await saveMicrovmLifecycleIntent(snapshot(), 'suspend', NOW); + expect(saved).toMatchObject({ status: 'saved', intent: { action: 'suspend', request_id: 'gate', deadline_ms: DEADLINE } }); + const command = mockSend.mock.calls[0][0]; + expect(command).toBeInstanceOf(TransactWriteCommand); + expect(command.input.TransactItems).toHaveLength(2); + expect(command.input.TransactItems[1].ConditionCheck.ExpressionAttributeValues) + .toEqual({ ':pending': 'PENDING', ':user': 'user', ':created': CREATED, ':timeout': 600 }); + expect(command.input.TransactItems[0].Update.UpdateExpression).toBe('SET microvm_lifecycle = :intent'); + }); + test('resume preserves its generation and original recovery age when replayed', async () => { + mockSend.mockResolvedValue({}); + const previous = intent('resume'); + expect(await saveMicrovmLifecycleIntent(snapshot({ intent: previous }), 'resume', NOW)).toEqual({ status: 'saved', intent: previous }); + const command = mockSend.mock.calls[0][0]; + expect(command.input.TransactItems).toHaveLength(1); + expect(command.input.ClientRequestToken).not.toBe(previous.generation); + expect(command.input.TransactItems[0].Update.ExpressionAttributeValues[':generation']).toBe(previous.generation); + }); + test('a wake cannot become sleep again within the same gate', async () => { + expect(await saveMicrovmLifecycleIntent(snapshot({ intent: intent('resume') }), 'suspend', NOW)).toEqual({ status: 'ineligible' }); + expect(mockSend).not.toHaveBeenCalled(); + }); + test('expired gate and closed task cannot receive suspend intent', async () => { + expect(await saveMicrovmLifecycleIntent(snapshot(), 'suspend', DEADLINE)).toEqual({ status: 'ineligible' }); + expect(await saveMicrovmLifecycleIntent(snapshot({ status: 'CANCELLED' }), 'resume', NOW)).toEqual({ status: 'ineligible' }); + expect(mockSend).not.toHaveBeenCalled(); + }); + test('conditional conflict asks caller to re-observe rather than overwrite', async () => { + mockSend.mockRejectedValueOnce({ name: 'TransactionCanceledException', CancellationReasons: [{ Code: 'ConditionalCheckFailed' }] }); + expect(await saveMicrovmLifecycleIntent(snapshot(), 'suspend', NOW)).toEqual({ status: 'stale' }); + }); + test('lost committed reply is recovered only by reading the exact saved generation', async () => { + let saved: MicrovmLifecycleIntent; + mockSend.mockImplementation(async command => { + if (command instanceof TransactWriteCommand) { + saved = command.input.TransactItems![0].Update!.ExpressionAttributeValues![':intent'] as MicrovmLifecycleIntent; + throw new Error('lost response'); + } + return { Item: command.input.TableName === 'LifecycleTasks' ? { ...task, microvm_lifecycle: saved! } : row }; + }); + expect(await saveMicrovmLifecycleIntent(snapshot(), 'suspend', NOW)).toMatchObject({ status: 'saved', intent: { generation: expect.any(String) } }); + expect(mockSend.mock.calls.filter(([command]) => command instanceof TransactWriteCommand)).toHaveLength(1); + expect(mockSend.mock.calls.filter(([command]) => command instanceof GetCommand)).toHaveLength(2); + }); + test('definite permission failure with no committed intent remains an error', async () => { + mockSend.mockRejectedValueOnce(new Error('access denied')).mockResolvedValueOnce({ Item: task }).mockResolvedValueOnce({ Item: row }); + await expect(saveMicrovmLifecycleIntent(snapshot(), 'suspend', NOW)).rejects.toThrow('access denied'); + }); +}); diff --git a/cdk/test/handlers/shared/strategies/lambda-microvm-strategy.test.ts b/cdk/test/handlers/shared/strategies/lambda-microvm-strategy.test.ts index 2d85a6676..d5b69136c 100644 --- a/cdk/test/handlers/shared/strategies/lambda-microvm-strategy.test.ts +++ b/cdk/test/handlers/shared/strategies/lambda-microvm-strategy.test.ts @@ -590,7 +590,7 @@ describe('LambdaMicrovmComputeStrategy', () => { mockSend.mockResolvedValueOnce({ microvmId: MICROVM_ID, state }); const result = await new LambdaMicrovmComputeStrategy().pollSession(makeHandle()); - expect(result).toEqual({ status: expected }); + expect(result).toEqual({ status: expected, microvmState: state }); }); test('sends GetMicrovm keyed on microvmIdentifier', async () => { @@ -609,7 +609,7 @@ describe('LambdaMicrovmComputeStrategy', () => { // No task id, no DDB read — the strategy cannot see the task row at all, // which is exactly why the health rules live in the orchestrator. const result = await new LambdaMicrovmComputeStrategy().pollSession(makeHandle()); - expect(result).toEqual({ status: 'suspended' }); + expect(result).toEqual({ status: 'suspended', microvmState: 'SUSPENDED' }); }); test('treats ResourceNotFoundException as completed (a reaped MicroVM is gone, not broken)', async () => { @@ -618,7 +618,7 @@ describe('LambdaMicrovmComputeStrategy', () => { mockSend.mockRejectedValueOnce(err); const result = await new LambdaMicrovmComputeStrategy().pollSession(makeHandle()); - expect(result).toEqual({ status: 'completed' }); + expect(result).toEqual({ status: 'completed', microvmState: 'NOT_FOUND' }); }); test('rethrows non-NotFound errors so the caller can count poll failures', async () => { @@ -632,7 +632,7 @@ describe('LambdaMicrovmComputeStrategy', () => { mockSend.mockResolvedValueOnce({ microvmId: MICROVM_ID, state: 'HIBERNATING_SOMEDAY' }); const result = await new LambdaMicrovmComputeStrategy().pollSession(makeHandle()); - expect(result).toEqual({ status: 'running' }); + expect(result).toEqual({ status: 'running', microvmState: 'UNKNOWN' }); }); test('throws when the handle is not a lambda-microvm handle', async () => { @@ -659,7 +659,7 @@ describe('LambdaMicrovmComputeStrategy', () => { const result = await new LambdaMicrovmComputeStrategy().pollSession(makeHandle()); - expect(result).toEqual({ status: 'completed', reason }); + expect(result).toEqual({ status: 'completed', microvmState: 'TERMINATED', reason }); }); test('logs a WARNING when a terminal MicroVM carries a reason', async () => { @@ -683,7 +683,7 @@ describe('LambdaMicrovmComputeStrategy', () => { const result = await new LambdaMicrovmComputeStrategy().pollSession(makeHandle()); - expect(result).toEqual({ status: 'completed' }); + expect(result).toEqual({ status: 'completed', microvmState: 'TERMINATED' }); expect(mockLogger.warn).not.toHaveBeenCalled(); }); @@ -695,7 +695,7 @@ describe('LambdaMicrovmComputeStrategy', () => { const result = await new LambdaMicrovmComputeStrategy().pollSession(makeHandle()); - expect(result).toEqual({ status: 'running' }); + expect(result).toEqual({ status: 'running', microvmState: 'RUNNING' }); expect('reason' in result).toBe(false); }); @@ -708,7 +708,7 @@ describe('LambdaMicrovmComputeStrategy', () => { const result = await new LambdaMicrovmComputeStrategy().pollSession(makeHandle()); - expect(result).toEqual({ status: expected, reason: 'because' }); + expect(result).toEqual({ status: expected, microvmState: state === 'HIBERNATING_SOMEDAY' ? 'UNKNOWN' : state, reason: 'because' }); }); }); diff --git a/docs/decisions/ADR-021-lambda-microvms-compute-backend.md b/docs/decisions/ADR-021-lambda-microvms-compute-backend.md index 4b4ab50d8..475c74c24 100644 --- a/docs/decisions/ADR-021-lambda-microvms-compute-backend.md +++ b/docs/decisions/ADR-021-lambda-microvms-compute-backend.md @@ -1,6 +1,6 @@ # ADR-021: AWS Lambda MicroVMs as a third ComputeStrategy backend -> **Implementation status (2026-09-13):** P1 ([#689](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/689)) and P2 ([#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733)) are merged. P2 completed real coding tasks with an IAM workaround; a clean rerun after re-bootstrap remains outstanding. Local P3 foundations now include strategy suspend/resume commands and approval deadlines that survive a frozen clock. Automatic policy, agent lifecycle hooks and supervisor/approval-handler integration remain unimplemented. This ADR defines P1–P3, not a P4. See the [current review](../verification/645-p3-readiness-review.md) and [P3 implementation plan](../verification/645-p3-implementation-plan.md). +> **Implementation status (2026-09-13):** P1 ([#689](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/689)) and P2 ([#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733)) are merged. P2 completed real coding tasks with an IAM workaround; a clean rerun after re-bootstrap remains outstanding. Local P3 foundations include suspend/resume commands, explicit VM observations, guarded persistent intent, a policy helper and approval deadlines that survive a frozen clock. Agent lifecycle hooks and supervisor/approval-handler integration remain unimplemented; automatic suspension is not enabled. This ADR defines P1–P3, not a P4. See the [current review](../verification/645-p3-readiness-review.md) and [P3 implementation plan](../verification/645-p3-implementation-plan.md). **Status:** proposed **Date:** 2026-07-29 @@ -66,7 +66,7 @@ Adopt **AWS Lambda MicroVMs as a third, opt-in `ComputeStrategy` backend** named The `ComputeStrategy` interface now has **mandatory** `suspendSession(handle)` / `resumeSession(handle)` methods returning `SessionLifecycleResult`, defined as `{ supported: false } | { supported: true }`. This local P3 foundation changes all three strategies together: AgentCore and ECS explicitly return false without calling AWS; MicroVM submits a command with a 10-second request bound. A future backend must implement both methods. The planned supervisor policy will use this explicit capability result. -`supported: true` means AWS acknowledged the command, not that the target state was reached. The SDK's suspend/resume response bodies are empty. All operational failures, including conflicts and not-found responses, propagate through sanitized errors; no generic conflict is normalized into success without observed-state evidence. A timed-out command may still have committed and requires reconciliation. The installed SDK has no `RESUMING` state, and the existing coarse `running` poll result also includes PENDING/unknown observations. P3 integration must obtain explicit service-state evidence before declaring wake complete. +`supported: true` means AWS acknowledged the command, not that the target state was reached. The SDK's suspend/resume response bodies are empty. All operational failures, including conflicts and not-found responses, propagate through sanitized errors; no generic conflict is normalized into success without observed-state evidence. A timed-out command may still have committed and requires reconciliation. The installed SDK has no `RESUMING` state. `SessionStatus.microvmState` now supplies explicit observations because the coarse `running` result also includes PENDING/unknown. The local policy/store uses retained generations and sticky wake intent to handle a delayed suspend after approval; the supervisor still needs to connect that policy, bound recovery and confirm wake state. See `docs/verification/645-lifecycle-intent.md`. **Poll semantics — the strategy reports, the orchestrator interprets.** `pollSession(handle)` receives only the session handle and cannot see task state, so the health rules must live where the DynamoDB status lives. `SessionStatus` gains a `'suspended'` variant; the strategy maps `GetMicrovm` state mechanically and the **orchestrator** cross-references against the task row — the same division of labor `finalPollState` already uses for ECS (substrate stopped + non-terminal DynamoDB status → failed) and `pollTaskStatus` uses for agentcore heartbeats: substrate `suspended` + task `AWAITING_APPROVAL` is healthy (orchestrator-intended suspend); `suspended` with any other task status is an anomaly to surface, not fail-fast; substrate terminal + non-terminal task status → classify failed. diff --git a/docs/design/COMPUTE.md b/docs/design/COMPUTE.md index 89cc59efd..213c9ba1a 100644 --- a/docs/design/COMPUTE.md +++ b/docs/design/COMPUTE.md @@ -83,7 +83,7 @@ Because a snapshot freezes its build-time environment, current deployment identi Networking separates image build from execution: the build-only connector permits TCP 80 and 443 because the Dockerfile uses `apt-get`, while running MicroVMs retain 443-only egress through the platform VPC. Every launch explicitly passes the Lambda-managed `NO_INGRESS` connector; omission would select the service's public-ingress default. The P2 image declares and serves `/ready` and `/validate` at build time and `/run` and `/terminate` at runtime. `/suspend` and `/resume` remain disabled until their P3 implementation. Registry HTTP/SSE tools therefore need reachable HTTPS/443 endpoints; remote non-443 tools are unsupported under the default policies of AgentCore and ECS as well. Local `stdio` tools can run, with their outbound traffic subject to the same restriction. Asset resolution/loading does not probe connectivity; see [registry network support](./REGISTRY.md#2-asset-kinds-for-mvp). -The local P3 foundation adds bounded suspend/resume commands to the MicroVM strategy and explicit unsupported results to AgentCore/ECS. These methods are not wired into automatic policy or human-decision handlers yet. Approval polling now retains the original UTC/monotonic deadline through slow writes or a frozen guest clock; resume hooks must still reuse that deadline and refresh credentials before work continues. +Local P3 foundations include bounded suspend/resume commands, explicit MicroVM state observations, persistent intent protected against stale writers, and a policy helper. AgentCore/ECS return explicit unsupported results. The supervisor and human-decision handlers do not use the lifecycle policy/store yet, and automatic suspension remains disabled. Approval polling retains the original UTC/monotonic deadline through slow writes or a frozen guest clock; resume hooks must still reuse that deadline and refresh credentials before work continues. ## ECS Fargate task sizing (build vs. planning) diff --git a/docs/src/content/docs/architecture/Compute.md b/docs/src/content/docs/architecture/Compute.md index 25bce1371..ef65554e9 100644 --- a/docs/src/content/docs/architecture/Compute.md +++ b/docs/src/content/docs/architecture/Compute.md @@ -87,7 +87,7 @@ Because a snapshot freezes its build-time environment, current deployment identi Networking separates image build from execution: the build-only connector permits TCP 80 and 443 because the Dockerfile uses `apt-get`, while running MicroVMs retain 443-only egress through the platform VPC. Every launch explicitly passes the Lambda-managed `NO_INGRESS` connector; omission would select the service's public-ingress default. The P2 image declares and serves `/ready` and `/validate` at build time and `/run` and `/terminate` at runtime. `/suspend` and `/resume` remain disabled until their P3 implementation. Registry HTTP/SSE tools therefore need reachable HTTPS/443 endpoints; remote non-443 tools are unsupported under the default policies of AgentCore and ECS as well. Local `stdio` tools can run, with their outbound traffic subject to the same restriction. Asset resolution/loading does not probe connectivity; see [registry network support](/sample-autonomous-cloud-coding-agents/architecture/registry#2-asset-kinds-for-mvp). -The local P3 foundation adds bounded suspend/resume commands to the MicroVM strategy and explicit unsupported results to AgentCore/ECS. These methods are not wired into automatic policy or human-decision handlers yet. Approval polling now retains the original UTC/monotonic deadline through slow writes or a frozen guest clock; resume hooks must still reuse that deadline and refresh credentials before work continues. +Local P3 foundations include bounded suspend/resume commands, explicit MicroVM state observations, persistent intent protected against stale writers, and a policy helper. AgentCore/ECS return explicit unsupported results. The supervisor and human-decision handlers do not use the lifecycle policy/store yet, and automatic suspension remains disabled. Approval polling retains the original UTC/monotonic deadline through slow writes or a frozen guest clock; resume hooks must still reuse that deadline and refresh credentials before work continues. ## ECS Fargate task sizing (build vs. planning) diff --git a/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md b/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md index e2fb78c1f..31eb5f718 100644 --- a/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md +++ b/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md @@ -4,7 +4,7 @@ title: Adr 021 lambda microvms compute backend # ADR-021: AWS Lambda MicroVMs as a third ComputeStrategy backend -> **Implementation status (2026-09-13):** P1 ([#689](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/689)) and P2 ([#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733)) are merged. P2 completed real coding tasks with an IAM workaround; a clean rerun after re-bootstrap remains outstanding. Local P3 foundations now include strategy suspend/resume commands and approval deadlines that survive a frozen clock. Automatic policy, agent lifecycle hooks and supervisor/approval-handler integration remain unimplemented. This ADR defines P1–P3, not a P4. See the [current review](/sample-autonomous-cloud-coding-agents/architecture/645-p3-readiness-review) and [P3 implementation plan](/sample-autonomous-cloud-coding-agents/architecture/645-p3-implementation-plan). +> **Implementation status (2026-09-13):** P1 ([#689](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/689)) and P2 ([#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733)) are merged. P2 completed real coding tasks with an IAM workaround; a clean rerun after re-bootstrap remains outstanding. Local P3 foundations include suspend/resume commands, explicit VM observations, guarded persistent intent, a policy helper and approval deadlines that survive a frozen clock. Agent lifecycle hooks and supervisor/approval-handler integration remain unimplemented; automatic suspension is not enabled. This ADR defines P1–P3, not a P4. See the [current review](/sample-autonomous-cloud-coding-agents/architecture/645-p3-readiness-review) and [P3 implementation plan](/sample-autonomous-cloud-coding-agents/architecture/645-p3-implementation-plan). **Status:** proposed **Date:** 2026-07-29 @@ -70,7 +70,7 @@ Adopt **AWS Lambda MicroVMs as a third, opt-in `ComputeStrategy` backend** named The `ComputeStrategy` interface now has **mandatory** `suspendSession(handle)` / `resumeSession(handle)` methods returning `SessionLifecycleResult`, defined as `{ supported: false } | { supported: true }`. This local P3 foundation changes all three strategies together: AgentCore and ECS explicitly return false without calling AWS; MicroVM submits a command with a 10-second request bound. A future backend must implement both methods. The planned supervisor policy will use this explicit capability result. -`supported: true` means AWS acknowledged the command, not that the target state was reached. The SDK's suspend/resume response bodies are empty. All operational failures, including conflicts and not-found responses, propagate through sanitized errors; no generic conflict is normalized into success without observed-state evidence. A timed-out command may still have committed and requires reconciliation. The installed SDK has no `RESUMING` state, and the existing coarse `running` poll result also includes PENDING/unknown observations. P3 integration must obtain explicit service-state evidence before declaring wake complete. +`supported: true` means AWS acknowledged the command, not that the target state was reached. The SDK's suspend/resume response bodies are empty. All operational failures, including conflicts and not-found responses, propagate through sanitized errors; no generic conflict is normalized into success without observed-state evidence. A timed-out command may still have committed and requires reconciliation. The installed SDK has no `RESUMING` state. `SessionStatus.microvmState` now supplies explicit observations because the coarse `running` result also includes PENDING/unknown. The local policy/store uses retained generations and sticky wake intent to handle a delayed suspend after approval; the supervisor still needs to connect that policy, bound recovery and confirm wake state. See `docs/verification/645-lifecycle-intent.md`. **Poll semantics — the strategy reports, the orchestrator interprets.** `pollSession(handle)` receives only the session handle and cannot see task state, so the health rules must live where the DynamoDB status lives. `SessionStatus` gains a `'suspended'` variant; the strategy maps `GetMicrovm` state mechanically and the **orchestrator** cross-references against the task row — the same division of labor `finalPollState` already uses for ECS (substrate stopped + non-terminal DynamoDB status → failed) and `pollTaskStatus` uses for agentcore heartbeats: substrate `suspended` + task `AWAITING_APPROVAL` is healthy (orchestrator-intended suspend); `suspended` with any other task status is an anomaly to surface, not fail-fast; substrate terminal + non-terminal task status → classify failed. diff --git a/docs/verification/645-lifecycle-intent.md b/docs/verification/645-lifecycle-intent.md new file mode 100644 index 000000000..715aa7b93 --- /dev/null +++ b/docs/verification/645-lifecycle-intent.md @@ -0,0 +1,85 @@ +# MicroVM lifecycle intent: local implementation and verification + +Status: implemented locally on 2026-09-13, **not connected to automatic suspension or deployed**. This is the next foundation after `f3e684d4`. The [P3 checklist](./645-p3-implementation-plan.md) tracks the remaining integration and live gates. + +## In plain language + +The supervisor needs a saved instruction that says “this particular task's computer should sleep” or “it must wake up.” Saving it in the database lets a replacement supervisor continue after a restart. + +Each instruction has a unique revision stamp. A supervisor holding an older stamp cannot overwrite a newer instruction. Once “wake up” is saved for an approval request, that same request cannot become “sleep” again. The record stays until the task expires; deleting it would let an old supervisor mistake the empty space for permission to save its old sleep request. + +## What the code does + +- `cdk/src/handlers/shared/microvm-lifecycle.ts` reads the task and its current approval consistently, validates identity, and saves intent with database conditions. +- `cdk/src/handlers/shared/microvm-lifecycle-policy.ts` chooses an action from observations. It makes no AWS calls or human decisions. Its returned action is a recommendation for the future supervisor integration. +- `SessionStatus.microvmState` carries explicit MicroVM state alongside the existing coarse status. Known SDK states are preserved; absent/future states become local `UNKNOWN`, and a not-found response becomes local `NOT_FOUND`. These last two are observations defined by this application, not service states. Existing status/reason behavior is preserved. + +The internal `microvm_lifecycle` task attribute contains: + +| Field | Meaning | +|---|---| +| `version` | Record format, currently `1` | +| `generation` | Unique revision stamp; replaced when the gate/action changes | +| `microvm_id` | The computer this instruction belongs to | +| `request_id` | The approval request, or null when recovering a working task | +| `action` | `suspend` or `resume` | +| `requested_at_ms` | Original request time, milliseconds since the Unix epoch | +| `deadline_ms` | Original approval expiry, or null if no valid deadline is available for conservative wake | + +Repeated saves of the same action for the same VM/gate retain their generation and request time. This keeps retries from resetting the future recovery budget. Each database transaction gets its own AWS request token; reusing the persistent generation as that token would conflict when the transaction's conditions change on replay. + +This is coordinator data, with no public task-type or agent contract change. The existing worker attribute allowlist excludes it; the session-role regression now names it explicitly. Real AWS permission verification remains pending. + +## Conditions and races + +Every write checks the current task owner, status, session ID, VM ID, endpoint, approval-request identity and previous instruction generation. A first write requires the attribute to be absent. A suspend also condition-checks the **same approval row** in the transaction: still PENDING, same owner, original creation time and timeout. A concurrent decision, changed gate or cancellation rolls back the whole write. + +Resume does not require a readable approval row. If that row is missing, malformed or temporarily unreadable, waking an already-sleeping task allows the existing agent decision loop to recover or expire it. Approval-read failures are explicit observations, including a safe error class; integration must report and count them. + +A lost database reply triggers readback. Only the exact saved generation with still-eligible task/gate identity is returned as saved. A known committed write followed by cancellation or a decision returns stale. An unresolved outcome remains an error. Cancelled/closed tasks may retain their old approval pointer; reads preserve it for diagnosis without reading the approval again, and writes cannot revive those tasks. + +**Saving intent does not lock AWS.** There remains a gap between a database write and a compute command. The future caller must reread identity/intent/deadline before suspend, persist wake requests even when the VM still looks awake, and reconcile after both acknowledgements and uncertain outcomes. Guest hooks must also validate the live task/gate before allowing a freeze or resumed work. + +## Initial policy + +These are local policy choices to validate with live measurements: + +| Setting | Initial value | +|---|---| +| Wait before considering sleep | 30 seconds from gate creation | +| Wake before approval/session deadline | 60 seconds | +| Minimum useful sleep interval | 30 seconds | +| Poll during a transition/unconfirmed observation | At most 5 seconds | +| Database request/read-sequence budget | 5 seconds; write plus recovery can use two budgets | + +A caller must supply valid times, session expiry, its normal poll interval, an enable switch and an explicit compatible-image flag. Both flags are required for a new suspend. Capability must describe the image/version that launched **this VM**, backed by coordinator-owned launch metadata or a verified full drain; current deployment settings alone cannot authenticate an older VM's capability. Unknown/legacy capability keeps new suspends off. Disabling new suspends still permits wake and termination. Long poll intervals are shortened to the next grace/wake/session deadline. + +A PENDING approval alone is not a wake condition. An intended suspended VM can wait while there is sufficient time. APPROVED, DENIED, TIMED_OUT, STRANDED, deadline proximity, missing/invalid data or unintended suspension require wake/recovery. While the service reports SUSPENDING, save desired resume but return `requestReady: false`; issue ResumeMicrovm only after observing SUSPENDED. A wake acknowledgement followed by a delayed old suspend remains repairable because wake intent is retained. + +Terminal/cancelled/finalizing tasks cannot resume. Terminal VM observations go through existing task reconciliation. PENDING/UNKNOWN or a legacy coarse `running` result cannot authorize suspend or confirm wake. Persistent failures/unconfirmed states still need bounded escalation in the supervisor. + +## Verification and deployment gates + +The unit suite checks the policy and store contract. The opt-in DynamoDB Local suite checks actual transaction expressions, cross-table rollback, competing/stale writers, changed task/approval identities, later gates, lost committed replies, cancellation during recovery and a fresh module/client reading the saved intent. + +Recorded local results (2026-09-13): CDK lint/compilation passed; **158 suites / 3,738 tests** passed in the broad handler/session-role run, including **23 lifecycle** and **15 existing capacity** database tests. Five relevant suites passed **218 overlapping tests** and exited normally. The broad run exited successfully after a delay without an open-handle trace; the cause is unconfirmed. Documentation sync/build/link checks passed, and the temporary database container was removed. + +Run locally from `cdk/` with a loopback DynamoDB Local instance: + +```sh +ABCA_DDB_LOCAL_ENDPOINT=http://127.0.0.1: mise run testf -- test/handlers/shared/microvm-lifecycle-local.test.ts +``` + +The tests use dummy credentials, create uniquely named tables and delete them afterward. They do not prove AWS IAM or MicroVM behavior. + +Before enabling automatic sleep: + +1. Complete a clean P2 deployment/rerun, including the earlier bootstrap, metadata, capacity and managed-image gates. Follow the coordinated v2 drain/rollout procedure. +2. Implement guest lifecycle context, acknowledged progress durability, credential refresh preserving task identity, resume barriers and snapshot randomness handling. Reuse the original approval deadline. +3. Connect policy/store to durable supervisor polling, preserving counters, backoff, anomaly episodes and a bounded wake-recovery clock. Handle stale results by observing again. Never reset the recovery clock on repeated saves. +4. Connect approve/deny after the decision commits, with bounded best-effort wake and repair diagnostics. Preserve current decision responses on wake failure. +5. Add and verify the coordinator's required task/approval transaction permissions and scoped MicroVM lifecycle grants. Grant no lifecycle action or intent-write permission to workers. Check the total handler time budget, not only each individual call. +6. Deploy compatible hooks/image and coordinator together with automatic suspension initially disabled. Validate actual transition/conflict/timeout behavior and then the full P3 acceptance matrix in an isolated development deployment. +7. Exercise disable/rollback with sleeping tasks: stop new suspends, continue wake/expiry/termination, and drain before removing compatible code. Do not delete intent records to “reset” recovery. + +The existing approval API uses the PENDING/task/gate transaction conditions; it has no independent wall-clock expiry check. The agent owns the conditional TIMED_OUT write, and the first committed decision wins. This implementation preserves that behavior. Strict API deadline rejection would be a separate behavior change. diff --git a/docs/verification/645-p3-implementation-plan.md b/docs/verification/645-p3-implementation-plan.md index b9d812948..797f335d1 100644 --- a/docs/verification/645-p3-implementation-plan.md +++ b/docs/verification/645-p3-implementation-plan.md @@ -25,7 +25,9 @@ Prerequisite work is tracked here on `fix/645-microvm-readiness`. “Completed - [ ] Implement production nesting if included, then verify a clean P2 deployment. - [x] Add mandatory pause/wake command methods across all three compute strategies, with explicit unsupported results and bounded MicroVM requests. - [x] Keep the original approval deadline through database writes and polling, including frozen/backward clocks; preserve decision races and cancellation. -- [ ] Add durable lifecycle intent/policy, compatible agent hooks and credential/durability barriers. +- [x] Save gate/VM-bound lifecycle intent with stale-writer protection; add explicit VM observations and a tested policy helper. +- [ ] Add compatible agent hooks and credential/durability barriers. +- [ ] Persist bounded poll/recovery counters and connect lifecycle policy to the supervisor. - [ ] Connect supervisor and approval handlers, then verify the complete P3 sleep/wake lifecycle in AWS. First prerequisite batch completed locally on 2026-09-13: @@ -98,6 +100,14 @@ First P3 foundation batch completed locally (2026-09-13): - CDK lint/compilation and **5 suites / 169 tests** pass. Full agent quality passes **1,823 tests / 84.53% coverage**, including **12 new clock/race/cancellation cases**. Counts overlap earlier runs. - No callers, IAM grants or image hooks enable automatic suspension yet. The same deadline must still be registered in the future lifecycle context and checked immediately by `/resume`. Durable intent, policy, credential refresh, acknowledged progress durability and live validation remain open. +Second P3 foundation batch implemented locally (2026-09-13): + +- Added `microvm_lifecycle` coordinator records with unique generations, VM/gate identity, desired action, original request time and deadline. Writes check the current task/handle/gate/generation; suspend also transaction-checks the exact PENDING approval. Wake remains sticky within one gate, records are retained, and repeated saves preserve the recovery age. +- Added explicit `microvmState` observations without changing existing coarse status/reason semantics, plus a policy helper for grace, useful sleep, pre-deadline wake, image/enable guards, missing/unreadable data, cancellation and delayed suspend-after-wake races. +- DynamoDB Local verifies actual transaction conditions, rollback, competing writers, changed identities, lost committed replies, cancellation/decision during recovery and a fresh module/client loading saved intent. See the [lifecycle runbook](./645-lifecycle-intent.md) for protocol and remaining integration/deployment gates. +- CDK lint/compilation passed. The broad handler/session-role run passed **158 suites / 3,738 tests**, including **23 lifecycle** and **15 existing capacity** DynamoDB Local tests. Five relevant suites passed **218 overlapping tests** and exited normally. The broad run exited successfully after a delay (about 72 seconds total versus 17.5 seconds reported test execution), with no open-handle trace; its cause is not established. Documentation sync, the **77-page** build and link checks pass. No Python source changed. The temporary local database was removed. +- No production caller uses this policy/store yet. No IAM grants, image hooks or automatic suspension were enabled. Durable poll failure/recovery tracking, guest barriers, supervisor/decision-handler wiring and live AWS gates remain unfinished. + ## The result we want When the coding agent asks a human for permission, its computer may go to sleep. The human can approve or deny while it sleeps. The computer wakes, reads the saved answer, and continues or refuses the action. If nobody answers, it wakes before the deadline and applies the existing timeout-as-denial rule. Its files, task identity, permissions, progress and deadline remain correct. @@ -259,17 +269,17 @@ Follow `cdk/scripts/package-microvm-artifact.sh` and the P1/P2 runbooks: AgentCore and ECS return explicit unsupported results without an AWS request. MicroVM issues `SuspendMicrovm`/`ResumeMicrovm` with `microvmIdentifier: handle.microvmId` and a 10-second abort bound. Local tests cover wrong/empty handles, repeated requests, simulated conflicts/not-found and retriable/permanent errors. **Still required:** verify actual AWS behavior for already-target-state/terminated VMs and races before normalizing any conflict into success. -The installed SDK returns empty suspend/resume responses. Its observed states are `PENDING`, `RUNNING`, `SUSPENDING`, `SUSPENDED`, `TERMINATING` and `TERMINATED`; there is no `RESUMING` value. The current coarse poll mapping groups PENDING/unknown with running and SUSPENDING with suspended. Preserve existing consumers, but expose an explicit service-state observation for P3 reconciliation: the coarse `running` result alone cannot prove wake completion. Keep `reason` diagnostic rather than parsing it for policy. +The installed SDK returns empty suspend/resume responses. Its observed states are `PENDING`, `RUNNING`, `SUSPENDING`, `SUSPENDED`, `TERMINATING` and `TERMINATED`; there is no `RESUMING` value. The coarse poll mapping groups PENDING/unknown with running and SUSPENDING with suspended. **Implemented locally:** `SessionStatus.microvmState` carries explicit state, including local `UNKNOWN`/`NOT_FOUND` observations, while preserving existing coarse status/reason semantics. The policy uses this field; the coarse `running` result alone cannot prove wake completion. `reason` remains diagnostic. ### Durable intent and policy -Store a small typed optional lifecycle record on the task row, separate from the existing string-only `compute_metadata` handle: active `request_id`, desired action (`suspend` or `resume`), request timestamp and gate deadline. Guard writes by task status, matching gate ID and matching MicroVM handle; do not overwrite another gate's intent. Add a generation/version condition if needed to prevent an older request from undoing a newer one. Reuse the task table, not a new coordination service. +**Implemented locally:** `microvm-lifecycle.ts` stores a typed optional record on the existing task row, separate from `compute_metadata`. It includes format version, generation, VM/gate identity, desired action, original timestamp and deadline. Conditions guard owner/status/gate/handle/generation; suspend atomically checks the same PENDING approval row and unchanged deadline inputs. A wake cannot become a sleep for that gate, and retaining the record prevents an older absent-record snapshot from recreating a sleep intent. A new gate may establish a new generation. No credentials or bearer URLs are stored. -Persist failure counters/backoff and anomaly episode tracking in durable poll-loop state so Lambda replay does not reset them. Update shared/public types and sync guards only where that record is actually exposed. Store no credentials or bearer URLs in the lifecycle record. +**Still required:** persist failure counters/backoff, anomaly episodes and bounded wake recovery in durable poll-loop state so Lambda replay does not reset them. Repeated intent saves already retain the original timestamp/generation. The record is internal and has no public task API field. Database success is not a lock over a later AWS command: reread before suspend and reconcile after every command outcome. -The policy combines task status, the **specific current approval row's status**, desired action, VM state and current time. A PENDING row already exists throughout every wait: “approval row exists” is not a reason to resume. Check APPROVED/DENIED, deadline proximity, missing-row recovery or other explicit wake conditions. +**Implemented as an unwired helper:** the policy combines task status, the **specific current approval row's status**, desired action, explicit VM state and current time. PENDING alone is not a reason to resume. All terminal approval states, deadline proximity, missing/unreadable data or unintended suspension can require wake. SUSPENDING records desired wake but returns `requestReady: false` until SUSPENDED is observed. -Proposed initial tuning for the implementation: 30-second suspend grace, 60-second pre-deadline wake margin, and three consecutive poll failures before escalation. Treat these as measured/tunable policy values, not AWS facts. Only suspend after grace when enough time remains to pay for resume overhead and a useful sleep. Clamp polling/backoff to the next wake deadline and session lifetime; do not let a user-configured long poll interval oversleep it. +Initial local policy values: 30-second suspend grace, 60-second pre-deadline wake margin, 30-second minimum useful sleep and at most 5-second transition polling. Long intervals are clamped to the relevant grace/wake/session deadline. These are tunable choices requiring live measurement, not AWS facts. Three consecutive poll failures before escalation remains proposed supervisor work. The future integration must bound its whole operation, including the store's 5-second request/read-sequence budgets and any lost-reply recovery. ### State/action table @@ -297,7 +307,7 @@ Files: `agent/src/server.py`, `hooks.py`, `task_state.py`, `aws_session.py`, `pr 5. Reseed the application PRNG from fresh OS entropy on **both `/run` and `/resume`**. Do not seed it with task IDs, timestamps or an image-fixed value. Continue using cryptographic randomness for secrets. Test resumed/sibling snapshot uniqueness where meaningful; do not claim that `random` becomes cryptographically safe. 6. **Polling implemented locally:** `_ApprovalDeadline` captures the original recorded UTC expiry and a monotonic cap before database writes. Remaining time is `min(monotonic_deadline - monotonic_now, created_at + timeout_s - wall_now)`, clamped at zero, including each sleep bound. **Still required:** register this same object in the lifecycle context, check it at resume and wake the existing agent-owned decision loop. Never create a fresh timeout on resume. 7. **Preserved/tested locally:** conditional TIMED_OUT write, strongly consistent reread when that write loses, and late-decision winner behavior. Forward/backward clocks, frozen monotonic time, slow writes/reads, missing rows and cancellation have regression coverage. **Still required:** exercise these through the actual resume barrier and live AWS lifecycle. TTL is asynchronous garbage collection, not a precise alarm clock. -8. Declare `/suspend` and `/resume` as enabled image hooks only when the same source version serves them. Add shared hook-budget constants and route/contract assertions. Keep `/ready` and `/validate` AWS-silent. An old image without the new hooks must not be eligible for automatic suspension; enable policy only after deploying a compatible pinned image, with explicit capability/version gating if mixed versions can coexist. +8. Declare `/suspend` and `/resume` as enabled image hooks only when the same source version serves them. Add shared hook-budget constants and route/contract assertions. Keep `/ready` and `/validate` AWS-silent. An old image without the new hooks must not be eligible for automatic suspension. Bind capability to the image/version that **actually launched each VM**, using coordinator-owned launch metadata; current deployment settings alone cannot prove an older VM supports the hooks. Unknown/legacy capability keeps new suspends off. A verified full drain can establish a clean boundary, but must not be assumed. ## 6. Wire the supervisor and human decisions @@ -315,12 +325,14 @@ Add consecutive MicroVM poll-error tracking. Reset on successful observations; c After the existing authorization checks and decision transaction **commit**, use a shared helper to load `compute_metadata` with a strongly consistent task read. Validate compute type, complete handle and current task/gate identity. For MicroVM, request resume with a short bound. No HTTP call to the guest is necessary. -Missing handle, read failure, wrong/terminal state or resume failure must produce a warning and a structured resume-orphan event (include task ID, gate ID, VM ID when known, stage, reason and safe AWS request ID). Audit-event failure is also best-effort. **None of these post-commit failures may turn a successful decision into a 500 or undo the transaction.** Preserve the current response/status and cross-tenant/expired/wrong-gate protections. The poll loop is the repair path. +Missing handle, read failure, wrong/terminal state or resume failure must produce a warning and a structured resume-orphan event (include task ID, gate ID, VM ID when known, stage, reason and safe AWS request ID). Audit-event failure is also best-effort. **None of these post-commit failures may turn a successful decision into a 500 or undo the transaction.** Preserve the current response/status and ownership/already-decided/wrong-gate protections. The poll loop is the repair path. The current API has no independent wall-clock expiry check: the agent owns TIMED_OUT, and the first committed decision wins. Strict API expiry would be a separate behavior change. ### IAM and deployment Grant orchestrator SuspendMicrovm/ResumeMicrovm on the exact configured image ARN and required version suffix, alongside its existing lifecycle actions. Grant approve/deny ResumeMicrovm, and GetMicrovm only if the shared wake helper uses it, with the same image scope. Do not grant these actions to the agent execution role. Add no token-minting, broad role-passing or network ingress permission. +Verify the lifecycle store's DynamoDB permissions too: task GetItem/UpdateItem and approval GetItem/ConditionCheckItem for the supervisor's cross-table suspend transaction. Confirm environment wiring for both table names and test effective permissions. Worker writes to `microvm_lifecycle` must remain excluded. + Check `task-api.ts`'s lazy image-ARN wiring and no-image branch, bootstrap deployment-role coverage, tests/suppressions and CloudFormation resource counts. Image-hook changes and runtime hook serving must deploy together; automatic suspension remains off until the compatible image is ready. Provide an operational disable switch that stops **new suspends while still allowing resume, timeout handling and termination** for already-sleeping tasks. ## 7. Acceptance matrix and completion gates diff --git a/docs/verification/645-p3-readiness-review.md b/docs/verification/645-p3-readiness-review.md index 7f0c402b2..990fca24c 100644 --- a/docs/verification/645-p3-readiness-review.md +++ b/docs/verification/645-p3-readiness-review.md @@ -10,6 +10,8 @@ Further local work adds durable MicroVM start receipts, stable request tokens an The first local P3 foundation adds the supervisor's pause/wake command methods and fixes approval timing. The timer now keeps both an elapsed-time stopwatch and the original clock deadline, using whichever runs out first. A sleeping VM therefore does not get a fresh approval window. Automatic sleeping, guest safety hooks and supervisor repair logic remain unfinished. The installed AWS SDK does not expose a `RESUMING` state; receiving a wake acknowledgement alone does not prove that the VM is awake. +A second foundation adds saved sleep/wake instructions with revision stamps, so an old supervisor cannot overwrite a newer wake request. A policy helper now handles the observed state and current approval deadline. Real DynamoDB Local transaction tests cover competing writers and restart/readback cases; this is a local database test, not an AWS deployment. See the [lifecycle runbook](./645-lifecycle-intent.md). Guest hooks, durable supervisor integration and live verification remain required. + The reservation review found a separate trust boundary: the old agent role could write/replace/delete its task row, including coordinator metadata. A subsequent local fix restricts main-task writes to reporting/approval attributes, removes replacement/deletion permissions and removes unused worker counter grants. It also removes unused Python submission/session-info helpers and corrects overstated tenant-isolation comments. [Metadata verification](./645-coordinator-metadata.md) records the tests and pending AWS gate. Status reports still come from the agent, and the compute role chooses session tags; this is not complete hostile-worker isolation. ## Start here: the pieces in plain language From 1694182889139fd99e6507add7e9a3cb57d040f2 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Sun, 13 Sep 2026 21:07:06 -0400 Subject: [PATCH 019/149] fix(bootstrap): pass execution role to nested stacks (#645) Grant only the exact execution role to CloudFormation. Correct recursive policy hashing and include the generated inline policy; bump bundle to 1.7.0. --- cdk/bootstrap/BOOTSTRAP_HASH | 2 +- cdk/bootstrap/BOOTSTRAP_VERSION | 2 +- cdk/bootstrap/bootstrap-template.yaml | 23 ++++++-- cdk/scripts/generate-bootstrap-template.ts | 16 +++++- cdk/src/bootstrap/nested-stack-policy.ts | 37 +++++++++++++ cdk/src/bootstrap/version.ts | 37 +++++++++---- .../__snapshots__/version.test.ts.snap | 2 +- cdk/test/bootstrap/bootstrap-template.test.ts | 19 +++++++ cdk/test/bootstrap/version.test.ts | 54 +++++++++++++++++++ docs/design/DEPLOYMENT_ROLES.md | 23 ++++++-- .../docs/architecture/Deployment-roles.md | 23 ++++++-- 11 files changed, 210 insertions(+), 28 deletions(-) create mode 100644 cdk/src/bootstrap/nested-stack-policy.ts diff --git a/cdk/bootstrap/BOOTSTRAP_HASH b/cdk/bootstrap/BOOTSTRAP_HASH index 3ca756db5..eb86d4a5d 100644 --- a/cdk/bootstrap/BOOTSTRAP_HASH +++ b/cdk/bootstrap/BOOTSTRAP_HASH @@ -1 +1 @@ -d30eb8e63c8f6fd5551de03a72acb342231175aaf417bd98811d6a07b0be3afc +db4602a57169676471a0aa9581fdd4b34826ae1bd642dad7491529a37cfc3429 diff --git a/cdk/bootstrap/BOOTSTRAP_VERSION b/cdk/bootstrap/BOOTSTRAP_VERSION index dc1e644a1..bd8bf882d 100644 --- a/cdk/bootstrap/BOOTSTRAP_VERSION +++ b/cdk/bootstrap/BOOTSTRAP_VERSION @@ -1 +1 @@ -1.6.0 +1.7.0 diff --git a/cdk/bootstrap/bootstrap-template.yaml b/cdk/bootstrap/bootstrap-template.yaml index b4568679f..4477e703a 100644 --- a/cdk/bootstrap/bootstrap-template.yaml +++ b/cdk/bootstrap/bootstrap-template.yaml @@ -1,12 +1,13 @@ # GENERATED FILE - DO NOT EDIT DIRECTLY # This template is generated by: npx tsx scripts/generate-bootstrap-template.ts -# ABCA Bootstrap Policy Version: 1.6.0 -# ABCA Bootstrap Policy Hash: d30eb8e63c8f6fd5551de03a72acb342231175aaf417bd98811d6a07b0be3afc +# ABCA Bootstrap Policy Version: 1.7.0 +# ABCA Bootstrap Policy Hash: db4602a57169676471a0aa9581fdd4b34826ae1bd642dad7491529a37cfc3429 # # Based on the default CDK bootstrap template with the following modifications: # - BootstrapVariant set to "ABCA: Least-Privilege Bootstrap" # - ComputeTypes parameter added for compute-variant selection # - IncludeComputeEcs / IncludeComputeLambdaMicrovms conditions added +# - Execution role may pass only itself to CloudFormation for nested stacks # - 6 AWS::IAM::ManagedPolicy resources replace AdministratorAccess; each # PolicyDocument is a minified JSON string so the template stays under the # 51,200-char CloudFormation inline-template limit (#864) @@ -754,6 +755,20 @@ Resources: - PermissionsBoundarySet - Fn::Sub: arn:${AWS::Partition}:iam::${AWS::AccountId}:policy/${InputPermissionsBoundary} - Ref: AWS::NoValue + Policies: + - PolicyName: PassExecutionRoleToCloudFormation + PolicyDocument: + Version: '2012-10-17' + Statement: + - Sid: PassSelfToCloudFormation + Effect: Allow + Action: iam:PassRole + Resource: + Fn::Sub: >- + arn:${AWS::Partition}:iam::${AWS::AccountId}:role/cdk-${Qualifier}-cfn-exec-role-${AWS::AccountId}-${AWS::Region} + Condition: + StringEquals: + iam:PassedToService: cloudformation.amazonaws.com CdkBoostrapPermissionsBoundaryPolicy: Condition: ShouldCreatePermissionsBoundary Type: AWS::IAM::ManagedPolicy @@ -884,10 +899,10 @@ Outputs: Value: '32' BootstrapPolicyVersion: Description: The version of the ABCA bootstrap policy bundle - Value: 1.6.0 + Value: 1.7.0 BootstrapPolicyHash: Description: SHA-256 hash of the ABCA bootstrap policy bundle for drift detection - Value: d30eb8e63c8f6fd5551de03a72acb342231175aaf417bd98811d6a07b0be3afc + Value: db4602a57169676471a0aa9581fdd4b34826ae1bd642dad7491529a37cfc3429 BootstrapPolicySet: Description: Comma-separated list of active ABCA bootstrap policy names Value: diff --git a/cdk/scripts/generate-bootstrap-template.ts b/cdk/scripts/generate-bootstrap-template.ts index f487fae27..f2d945d51 100644 --- a/cdk/scripts/generate-bootstrap-template.ts +++ b/cdk/scripts/generate-bootstrap-template.ts @@ -31,6 +31,7 @@ import { join } from 'node:path'; import * as yaml from 'js-yaml'; +import { nestedStackExecutionPolicy } from '../src/bootstrap/nested-stack-policy'; import { applicationPolicy, computeAgentcorePolicy, @@ -193,7 +194,7 @@ export function buildTemplate(): any { } // --- Step 5: Modify CloudFormationExecutionRole ManagedPolicyArns --- - // Replace the conditional that falls back to AdministratorAccess with our inline policies. + // Replace the conditional that falls back to AdministratorAccess with our managed policies. // Keep the CloudFormationExecutionPolicies parameter override for flexibility. const coreRefs = [ { Ref: 'IaCRoleABCAInfrastructure' }, @@ -218,6 +219,17 @@ export function buildTemplate(): any { ], }; + // Nested stacks inherit the parent execution role. CloudFormation checks that + // the caller can pass it, including during change-set validation (#645). + const executionRole = template.Resources.CloudFormationExecutionRole.Properties; + executionRole.Policies = [ + ...(executionRole.Policies ?? []), + { + PolicyName: 'PassExecutionRoleToCloudFormation', + PolicyDocument: nestedStackExecutionPolicy(), + }, + ]; + // --- Step 6: Add outputs --- template.Outputs.BootstrapPolicyVersion = { Description: 'The version of the ABCA bootstrap policy bundle', @@ -283,6 +295,7 @@ export function renderTemplate(): string { '# - BootstrapVariant set to "ABCA: Least-Privilege Bootstrap"', '# - ComputeTypes parameter added for compute-variant selection', '# - IncludeComputeEcs / IncludeComputeLambdaMicrovms conditions added', + '# - Execution role may pass only itself to CloudFormation for nested stacks', '# - 6 AWS::IAM::ManagedPolicy resources replace AdministratorAccess; each', '# PolicyDocument is a minified JSON string so the template stays under the', '# 51,200-char CloudFormation inline-template limit (#864)', @@ -346,4 +359,3 @@ function main(): void { if (require.main === module) { main(); } - diff --git a/cdk/src/bootstrap/nested-stack-policy.ts b/cdk/src/bootstrap/nested-stack-policy.ts new file mode 100644 index 000000000..77332543b --- /dev/null +++ b/cdk/src/bootstrap/nested-stack-policy.ts @@ -0,0 +1,37 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +/** The parent stack passes its execution role to nested CloudFormation stacks. + * Resolve its exact ARN from bootstrap parameters, without a self-referencing + * GetAtt or a wildcard grant over other deployment/application roles. + */ +export function nestedStackExecutionPolicy() { + return { + Version: '2012-10-17', + Statement: [{ + Sid: 'PassSelfToCloudFormation', + Effect: 'Allow', + Action: 'iam:PassRole', + Resource: { + 'Fn::Sub': 'arn:${AWS::Partition}:iam::${AWS::AccountId}:role/cdk-${Qualifier}-cfn-exec-role-${AWS::AccountId}-${AWS::Region}', + }, + Condition: { StringEquals: { 'iam:PassedToService': 'cloudformation.amazonaws.com' } }, + }], + }; +} diff --git a/cdk/src/bootstrap/version.ts b/cdk/src/bootstrap/version.ts index 2bdb6caac..553f49d52 100644 --- a/cdk/src/bootstrap/version.ts +++ b/cdk/src/bootstrap/version.ts @@ -19,6 +19,7 @@ import { createHash } from 'node:crypto'; +import { nestedStackExecutionPolicy } from './nested-stack-policy'; import { allPolicies } from './policies'; /** @@ -44,25 +45,39 @@ import { allPolicies } from './policies'; * `policies/compute-lambda-microvm.ts`), which is exactly the class of breakage a * patch-level bump would under-advertise. * - * It is 1.6.0 and not 1.4.0 because #629 and #246 landed on `main` first and took + * The 1.6.0 release was not 1.4.0 because #629 and #246 landed on `main` first and took * 1.4.0 and 1.5.0 in the interim. The number is an operator-visible contract (the * `CDKToolkit` stack's `BootstrapPolicyVersion` output), so re-using a published * version would make two different bundles indistinguishable to the `>=` check * operators are told to run. + * + * 1.6.0 → 1.7.0 adds an exact-self, CloudFormation-only PassRole inline policy + * for nested stacks (#645 clean deployment). The policy hash now includes that + * inline policy and every nested JSON field; the old root-key replacer omitted + * Action/Resource/Condition changes from its input. */ -export const BOOTSTRAP_VERSION = '1.6.0'; +export const BOOTSTRAP_VERSION = '1.7.0'; + +function canonicalize(value: unknown): unknown { + if (Array.isArray(value)) return value.map(canonicalize); + if (value !== null && typeof value === 'object') { + const entries = value as Record; + return Object.fromEntries( + Object.keys(entries).sort().map((key) => [key, canonicalize(entries[key])]), + ); + } + return value; +} /** - * Computes a SHA-256 hash over all bootstrap policies. - * The hash is deterministic: policies are serialized with sorted keys - * so that object property ordering does not affect the digest. + * Hashes all ABCA managed and inline execution policies. Sort object keys + * recursively without dropping nested fields; preserve array ordering. */ export function computeBootstrapHash(): string { - const policies = allPolicies(); - const normalized = policies.map((p) => { - const json = p.toJSON(); - return JSON.stringify(json, Object.keys(json).sort()); - }); - const payload = JSON.stringify(normalized); + const policies = [ + ...allPolicies().map((policy) => policy.toJSON()), + nestedStackExecutionPolicy(), + ]; + const payload = JSON.stringify(canonicalize(policies)); return createHash('sha256').update(payload).digest('hex'); } diff --git a/cdk/test/bootstrap/__snapshots__/version.test.ts.snap b/cdk/test/bootstrap/__snapshots__/version.test.ts.snap index d4fe819d0..acd755005 100644 --- a/cdk/test/bootstrap/__snapshots__/version.test.ts.snap +++ b/cdk/test/bootstrap/__snapshots__/version.test.ts.snap @@ -1,3 +1,3 @@ // Jest Snapshot v1, https://jestjs.io/docs/snapshot-testing -exports[`bootstrap version module hash is stable 1`] = `"d30eb8e63c8f6fd5551de03a72acb342231175aaf417bd98811d6a07b0be3afc"`; +exports[`bootstrap version module hash is stable 1`] = `"db4602a57169676471a0aa9581fdd4b34826ae1bd642dad7491529a37cfc3429"`; diff --git a/cdk/test/bootstrap/bootstrap-template.test.ts b/cdk/test/bootstrap/bootstrap-template.test.ts index d4f8dc9d7..a3006fa00 100644 --- a/cdk/test/bootstrap/bootstrap-template.test.ts +++ b/cdk/test/bootstrap/bootstrap-template.test.ts @@ -154,6 +154,25 @@ describe('Bootstrap template', () => { }); describe('CloudFormationExecutionRole', () => { + it('can pass only itself to CloudFormation for nested stacks', () => { + const role = template.Resources.CloudFormationExecutionRole.Properties; + const policy = role.Policies?.find( + (item: any) => item.PolicyName === 'PassExecutionRoleToCloudFormation', + ); + expect(policy?.PolicyDocument).toEqual({ + Version: '2012-10-17', + Statement: [{ + Sid: 'PassSelfToCloudFormation', + Effect: 'Allow', + Action: 'iam:PassRole', + Resource: { + 'Fn::Sub': `arn:\${AWS::Partition}:iam::\${AWS::AccountId}:role/${role.RoleName['Fn::Sub']}`, + }, + Condition: { StringEquals: { 'iam:PassedToService': 'cloudformation.amazonaws.com' } }, + }], + }); + }); + it('exists and is an IAM Role', () => { expect(template.Resources.CloudFormationExecutionRole).toBeDefined(); expect(template.Resources.CloudFormationExecutionRole.Type).toBe('AWS::IAM::Role'); diff --git a/cdk/test/bootstrap/version.test.ts b/cdk/test/bootstrap/version.test.ts index 1c1bce5cb..d7995fd42 100644 --- a/cdk/test/bootstrap/version.test.ts +++ b/cdk/test/bootstrap/version.test.ts @@ -17,9 +17,63 @@ * SOFTWARE. */ +import { PolicyDocument } from 'aws-cdk-lib/aws-iam'; +import * as nestedStackPolicy from '../../src/bootstrap/nested-stack-policy'; +import * as bootstrapPolicies from '../../src/bootstrap/policies'; import { BOOTSTRAP_VERSION, computeBootstrapHash } from '../../src/bootstrap/version'; describe('bootstrap version module', () => { + afterEach(() => { + jest.restoreAllMocks(); + }); + + function hashDocument(document: Record): string { + const policy = new PolicyDocument(); + jest.spyOn(policy, 'toJSON').mockReturnValue(document); + jest.spyOn(bootstrapPolicies, 'allPolicies').mockReturnValue([policy]); + return computeBootstrapHash(); + } + + const statement = { + Effect: 'Allow', + Action: 'iam:PassRole', + Resource: 'arn:aws:iam::123456789012:role/example', + Condition: { StringEquals: { 'iam:PassedToService': 'cloudformation.amazonaws.com' } }, + }; + + it.each([ + { Action: 'iam:CreateRole' }, + { Effect: 'Deny' }, + { Resource: 'arn:aws:iam::123456789012:role/other' }, + { Condition: { StringEquals: { 'iam:PassedToService': 'lambda.amazonaws.com' } } }, + ])('hash changes when nested permissions change: %j', (change) => { + const before = hashDocument({ Version: '2012-10-17', Statement: [statement] }); + const after = hashDocument({ Version: '2012-10-17', Statement: [{ ...statement, ...change }] }); + expect(after).not.toBe(before); + }); + + it('ignores object key order at every depth', () => { + const before = hashDocument({ Version: '2012-10-17', Statement: [statement] }); + const after = hashDocument({ + Statement: [{ + Condition: statement.Condition, + Resource: statement.Resource, + Action: statement.Action, + Effect: statement.Effect, + }], + Version: '2012-10-17', + }); + expect(after).toBe(before); + }); + + it('includes the generated inline execution policy in the hash', () => { + const before = computeBootstrapHash(); + const policy = nestedStackPolicy.nestedStackExecutionPolicy(); + policy.Statement[0].Action = 'iam:GetRole'; + jest.spyOn(nestedStackPolicy, 'nestedStackExecutionPolicy').mockReturnValue(policy); + expect(computeBootstrapHash()).not.toBe(before); + }); + it('BOOTSTRAP_VERSION matches semver format', () => { expect(BOOTSTRAP_VERSION).toMatch(/^\d+\.\d+\.\d+$/); }); diff --git a/docs/design/DEPLOYMENT_ROLES.md b/docs/design/DEPLOYMENT_ROLES.md index 9f10f1967..95b092ac9 100644 --- a/docs/design/DEPLOYMENT_ROLES.md +++ b/docs/design/DEPLOYMENT_ROLES.md @@ -30,7 +30,7 @@ The policies are split into six IAM managed policies (each under the 6,144-chara > **Placeholder substitution**: Replace `ACCOUNT_ID` with your 12-digit AWS account ID and `REGION` with your deployment region (e.g., `us-east-1`) throughout this document. -These policies are not created or attached manually. The repository generates them — and a custom bootstrap template that wires all six into the CloudFormation execution role — from the TypeScript sources, then bootstraps with that template: +These policies are not created or attached manually. The repository generates them and a custom bootstrap template that attaches the selected policies to the CloudFormation execution role: ```bash # Regenerate artifacts (policies JSON + template YAML) and bootstrap. @@ -48,14 +48,29 @@ aws cloudformation update-stack --stack-name CDKToolkit --use-previous-template aws cloudformation describe-stacks --stack-name CDKToolkit --query 'Stacks[0].Parameters' ``` -Under the hood, `mise //cdk:bootstrap` runs `npx cdk bootstrap --template bootstrap/bootstrap-template.yaml` (see `cdk/mise.toml`). The generated template defines six inline `AWS::IAM::ManagedPolicy` resources that **replace** the default `AdministratorAccess` on the CloudFormation execution role; the `IaCRole-ABCA-Compute-ECS` and `IaCRole-ABCA-Compute-LambdaMicrovms` policies are conditional on the `ComputeTypes` parameter including their respective backend. The policy sources are `cdk/src/bootstrap/policies/{infrastructure,application,observability,compute-agentcore,compute-ecs,compute-lambda-microvm}.ts`, compiled to `cdk/bootstrap/policies/*.json` by `cdk/scripts/generate-bootstrap-artifacts.ts`. +Under the hood, `mise //cdk:bootstrap` runs `npx cdk bootstrap --template bootstrap/bootstrap-template.yaml` (see `cdk/mise.toml`). The generated template defines six `AWS::IAM::ManagedPolicy` resources that **replace** the default `AdministratorAccess` on the CloudFormation execution role; the `IaCRole-ABCA-Compute-ECS` and `IaCRole-ABCA-Compute-LambdaMicrovms` policies are conditional on the `ComputeTypes` parameter including their respective backend. The policy sources are `cdk/src/bootstrap/policies/{infrastructure,application,observability,compute-agentcore,compute-ecs,compute-lambda-microvm}.ts`, compiled to `cdk/bootstrap/policies/*.json` by `cdk/scripts/generate-bootstrap-artifacts.ts`. + +**Re-bootstrap to bundle 1.7.0 or later before deploying nested stacks.** The +execution role also needs the generated inline policy +`PassExecutionRoleToCloudFormation`, from +`cdk/src/bootstrap/nested-stack-policy.ts`. It grants `iam:PassRole` on that exact +execution role, with `iam:PassedToService=cloudformation.amazonaws.com`. The ARN +uses the bootstrap partition, account, Region and qualifier; it grants no access +to pass other roles. Without it, a fresh 1.6.0 deployment fails change-set +validation when CloudFormation tries to pass its role to the registry's nested +stacks. Preserve all existing bootstrap parameters when updating the template. + +Bundle 1.7.0 also corrects `BootstrapPolicyHash`: it includes nested policy fields +and the generated inline policy. Earlier hashes could remain unchanged after +Action, Resource or Condition changes. A matching old hash is insufficient proof +that deployed permissions match source. > **CloudFormation inline-template limit — 51,200 characters**: This is a second, independent size ceiling, distinct from the per-policy IAM 6,144-character limit above. `cdk bootstrap --template` sends the template inline as `TemplateBody`; above 51,200 characters the CLI has to stage it in S3 instead, which it **cannot** do while bootstrapping a fresh account, because that bucket is one of the resources bootstrap creates. The result is a hard `BootstrapStackRequired` failure with no way through `cdk bootstrap`, `--force` included ([#864](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/864)). > > Two consequences for anyone editing the policies: > > - **The gated size is not the file's size on disk.** The CLI parses the file, discards its formatting, and re-serialises the parsed object before measuring. Reformatting `bootstrap-template.yaml` therefore changes nothing; only the *content* moves the number. Check it with `npx cdk bootstrap --show-template --template bootstrap/bootstrap-template.yaml | wc -c`. -> - **Each `PolicyDocument` is emitted as a minified JSON string**, not a nested YAML mapping. Both are valid for this `Json`-typed property and IAM stores the string parsed, but a string scalar survives the CLI's re-serialisation on one line — which is what keeps the body under the ceiling (45,743 characters, versus 53,369 as mappings). +> - **Each managed-policy `PolicyDocument` is emitted as a minified JSON string**, not a nested YAML mapping. Both are valid for this `Json`-typed property and IAM stores the string parsed, but a string scalar survives the CLI's re-serialisation on one line. The original fix reduced the body to 45,743 characters from 53,369; subsequent policy additions remain subject to the budget. The small inline self-role policy uses a mapping to resolve its ARN with `Fn::Sub`. > > `cdk/scripts/generate-bootstrap-template.ts` fails the build when the body exceeds the budget in `cdk/src/bootstrap/template-size.ts`, so adding statements surfaces the problem at generation time rather than against somebody's fresh account. Both the guard and its regression test obtain the size by invoking `cdk bootstrap --show-template` on the committed artifact — the CLI is the component that makes the inline-vs-S3 decision, so asking it directly cannot drift the way a local copy of its serialiser would. No AWS credentials are required. @@ -92,7 +107,7 @@ Under the hood, `mise //cdk:bootstrap` runs `npx cdk bootstrap --template bootst For deploying the `backgroundagent-dev` stack. This single stack contains all platform resources including the AgentCore runtime, ECS compute (when enabled), API Gateway, Cognito, DynamoDB tables, VPC, DNS Firewall, and observability infrastructure. -> **IAM managed policy size limit**: A single managed policy cannot exceed 6,144 characters. The permissions below are split into six policies to stay under this limit (three always-applied, plus three compute-variant policies). They are wired into the CloudFormation execution role by the generated bootstrap template; see [Using these policies](#using-these-policies). +> **IAM managed policy size limit**: A single managed policy cannot exceed 6,144 characters. The permissions below are split into six policies to stay under this limit (four always applied, including AgentCore, plus optional ECS and MicroVM policies). The separate inline self-role grant is described above. They are wired into the CloudFormation execution role by the generated bootstrap template; see [Using these policies](#using-these-policies). ### IaCRole-ABCA-Infrastructure diff --git a/docs/src/content/docs/architecture/Deployment-roles.md b/docs/src/content/docs/architecture/Deployment-roles.md index 9603bf56d..8612b184e 100644 --- a/docs/src/content/docs/architecture/Deployment-roles.md +++ b/docs/src/content/docs/architecture/Deployment-roles.md @@ -34,7 +34,7 @@ The policies are split into six IAM managed policies (each under the 6,144-chara > **Placeholder substitution**: Replace `ACCOUNT_ID` with your 12-digit AWS account ID and `REGION` with your deployment region (e.g., `us-east-1`) throughout this document. -These policies are not created or attached manually. The repository generates them — and a custom bootstrap template that wires all six into the CloudFormation execution role — from the TypeScript sources, then bootstraps with that template: +These policies are not created or attached manually. The repository generates them and a custom bootstrap template that attaches the selected policies to the CloudFormation execution role: ```bash # Regenerate artifacts (policies JSON + template YAML) and bootstrap. @@ -52,14 +52,29 @@ aws cloudformation update-stack --stack-name CDKToolkit --use-previous-template aws cloudformation describe-stacks --stack-name CDKToolkit --query 'Stacks[0].Parameters' ``` -Under the hood, `mise //cdk:bootstrap` runs `npx cdk bootstrap --template bootstrap/bootstrap-template.yaml` (see `cdk/mise.toml`). The generated template defines six inline `AWS::IAM::ManagedPolicy` resources that **replace** the default `AdministratorAccess` on the CloudFormation execution role; the `IaCRole-ABCA-Compute-ECS` and `IaCRole-ABCA-Compute-LambdaMicrovms` policies are conditional on the `ComputeTypes` parameter including their respective backend. The policy sources are `cdk/src/bootstrap/policies/{infrastructure,application,observability,compute-agentcore,compute-ecs,compute-lambda-microvm}.ts`, compiled to `cdk/bootstrap/policies/*.json` by `cdk/scripts/generate-bootstrap-artifacts.ts`. +Under the hood, `mise //cdk:bootstrap` runs `npx cdk bootstrap --template bootstrap/bootstrap-template.yaml` (see `cdk/mise.toml`). The generated template defines six `AWS::IAM::ManagedPolicy` resources that **replace** the default `AdministratorAccess` on the CloudFormation execution role; the `IaCRole-ABCA-Compute-ECS` and `IaCRole-ABCA-Compute-LambdaMicrovms` policies are conditional on the `ComputeTypes` parameter including their respective backend. The policy sources are `cdk/src/bootstrap/policies/{infrastructure,application,observability,compute-agentcore,compute-ecs,compute-lambda-microvm}.ts`, compiled to `cdk/bootstrap/policies/*.json` by `cdk/scripts/generate-bootstrap-artifacts.ts`. + +**Re-bootstrap to bundle 1.7.0 or later before deploying nested stacks.** The +execution role also needs the generated inline policy +`PassExecutionRoleToCloudFormation`, from +`cdk/src/bootstrap/nested-stack-policy.ts`. It grants `iam:PassRole` on that exact +execution role, with `iam:PassedToService=cloudformation.amazonaws.com`. The ARN +uses the bootstrap partition, account, Region and qualifier; it grants no access +to pass other roles. Without it, a fresh 1.6.0 deployment fails change-set +validation when CloudFormation tries to pass its role to the registry's nested +stacks. Preserve all existing bootstrap parameters when updating the template. + +Bundle 1.7.0 also corrects `BootstrapPolicyHash`: it includes nested policy fields +and the generated inline policy. Earlier hashes could remain unchanged after +Action, Resource or Condition changes. A matching old hash is insufficient proof +that deployed permissions match source. > **CloudFormation inline-template limit — 51,200 characters**: This is a second, independent size ceiling, distinct from the per-policy IAM 6,144-character limit above. `cdk bootstrap --template` sends the template inline as `TemplateBody`; above 51,200 characters the CLI has to stage it in S3 instead, which it **cannot** do while bootstrapping a fresh account, because that bucket is one of the resources bootstrap creates. The result is a hard `BootstrapStackRequired` failure with no way through `cdk bootstrap`, `--force` included ([#864](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/864)). > > Two consequences for anyone editing the policies: > > - **The gated size is not the file's size on disk.** The CLI parses the file, discards its formatting, and re-serialises the parsed object before measuring. Reformatting `bootstrap-template.yaml` therefore changes nothing; only the *content* moves the number. Check it with `npx cdk bootstrap --show-template --template bootstrap/bootstrap-template.yaml | wc -c`. -> - **Each `PolicyDocument` is emitted as a minified JSON string**, not a nested YAML mapping. Both are valid for this `Json`-typed property and IAM stores the string parsed, but a string scalar survives the CLI's re-serialisation on one line — which is what keeps the body under the ceiling (45,743 characters, versus 53,369 as mappings). +> - **Each managed-policy `PolicyDocument` is emitted as a minified JSON string**, not a nested YAML mapping. Both are valid for this `Json`-typed property and IAM stores the string parsed, but a string scalar survives the CLI's re-serialisation on one line. The original fix reduced the body to 45,743 characters from 53,369; subsequent policy additions remain subject to the budget. The small inline self-role policy uses a mapping to resolve its ARN with `Fn::Sub`. > > `cdk/scripts/generate-bootstrap-template.ts` fails the build when the body exceeds the budget in `cdk/src/bootstrap/template-size.ts`, so adding statements surfaces the problem at generation time rather than against somebody's fresh account. Both the guard and its regression test obtain the size by invoking `cdk bootstrap --show-template` on the committed artifact — the CLI is the component that makes the inline-vs-S3 decision, so asking it directly cannot drift the way a local copy of its serialiser would. No AWS credentials are required. @@ -96,7 +111,7 @@ Under the hood, `mise //cdk:bootstrap` runs `npx cdk bootstrap --template bootst For deploying the `backgroundagent-dev` stack. This single stack contains all platform resources including the AgentCore runtime, ECS compute (when enabled), API Gateway, Cognito, DynamoDB tables, VPC, DNS Firewall, and observability infrastructure. -> **IAM managed policy size limit**: A single managed policy cannot exceed 6,144 characters. The permissions below are split into six policies to stay under this limit (three always-applied, plus three compute-variant policies). They are wired into the CloudFormation execution role by the generated bootstrap template; see [Using these policies](#using-these-policies). +> **IAM managed policy size limit**: A single managed policy cannot exceed 6,144 characters. The permissions below are split into six policies to stay under this limit (four always applied, including AgentCore, plus optional ECS and MicroVM policies). The separate inline self-role grant is described above. They are wired into the CloudFormation execution role by the generated bootstrap template; see [Using these policies](#using-these-policies). ### IaCRole-ABCA-Infrastructure From 4ea7e88b0ba11864c642755e40e2c46778104708 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Sun, 13 Sep 2026 21:26:36 -0400 Subject: [PATCH 020/149] fix(cdk): scope global resource names by region (#645) --- cdk/src/constructs/screenshot-bucket.ts | 9 +++- cdk/src/constructs/task-dashboard.ts | 4 +- cdk/test/constructs/screenshot-bucket.test.ts | 49 ++++++++++++++++++- cdk/test/constructs/task-dashboard.test.ts | 43 +++++++++++----- docs/design/OBSERVABILITY.md | 2 +- .../docs/architecture/Observability.md | 2 +- 6 files changed, 90 insertions(+), 19 deletions(-) diff --git a/cdk/src/constructs/screenshot-bucket.ts b/cdk/src/constructs/screenshot-bucket.ts index 19418bb33..00595220b 100644 --- a/cdk/src/constructs/screenshot-bucket.ts +++ b/cdk/src/constructs/screenshot-bucket.ts @@ -17,7 +17,7 @@ * SOFTWARE. */ -import { Duration, RemovalPolicy } from 'aws-cdk-lib'; +import { Duration, Names, RemovalPolicy, Stack } from 'aws-cdk-lib'; import * as cloudfront from 'aws-cdk-lib/aws-cloudfront'; import * as origins from 'aws-cdk-lib/aws-cloudfront-origins'; import * as s3 from 'aws-cdk-lib/aws-s3'; @@ -97,9 +97,14 @@ export class ScreenshotBucket extends Construct { // and grants `s3:GetObject` to the distribution's CF service principal // only — no anonymous principal in the policy, so account-level BPA // doesn't reject it. + // OAC names are account-global. Keep the construct-path hash and leave + // room for the Region within CloudFront's 64-character name limit. + const originAccessControl = new cloudfront.S3OriginAccessControl(this, 'OriginAccessControl', { + originAccessControlName: `${Names.uniqueResourceName(this, { maxLength: 40 })}-${Stack.of(this).region}`, + }); this.distribution = new cloudfront.Distribution(this, 'Distribution', { defaultBehavior: { - origin: origins.S3BucketOrigin.withOriginAccessControl(this.bucket), + origin: origins.S3BucketOrigin.withOriginAccessControl(this.bucket, { originAccessControl }), viewerProtocolPolicy: cloudfront.ViewerProtocolPolicy.REDIRECT_TO_HTTPS, // Screenshots are immutable per (repo, sha) — long TTL is safe // and minimizes origin S3 requests on hot PRs. diff --git a/cdk/src/constructs/task-dashboard.ts b/cdk/src/constructs/task-dashboard.ts index 6f4e40f7a..78c8ff4c7 100644 --- a/cdk/src/constructs/task-dashboard.ts +++ b/cdk/src/constructs/task-dashboard.ts @@ -64,7 +64,9 @@ export class TaskDashboard extends Construct { const logGroup = props.applicationLogGroup; this.dashboard = new cloudwatch.Dashboard(this, 'Dashboard', { - dashboardName: `BackgroundAgent-Tasks-${Stack.of(this).stackName}`, + // Dashboard names are account-global, so identical stack names in + // different Regions must not share one CloudFormation-owned dashboard. + dashboardName: `BackgroundAgent-Tasks-${Stack.of(this).stackName}-${Stack.of(this).region}`, defaultInterval: Duration.hours(24), }); diff --git a/cdk/test/constructs/screenshot-bucket.test.ts b/cdk/test/constructs/screenshot-bucket.test.ts index 882df1ae9..552727b3e 100644 --- a/cdk/test/constructs/screenshot-bucket.test.ts +++ b/cdk/test/constructs/screenshot-bucket.test.ts @@ -23,12 +23,59 @@ import { ScreenshotBucket } from '../../src/constructs/screenshot-bucket'; describe('ScreenshotBucket', () => { let template: Template; + let regionalTemplates: Template[]; + let longNameTemplate: Template; - beforeEach(() => { + beforeAll(() => { const app = new App(); const stack = new Stack(app, 'TestStack'); new ScreenshotBucket(stack, 'ScreenshotBucket'); template = Template.fromStack(stack); + regionalTemplates = ['us-east-1', 'us-west-2'].map((region) => { + const regionalStack = new Stack(new App(), 'TestStack', { + env: { account: '123456789012', region }, + }); + new ScreenshotBucket(regionalStack, 'ScreenshotBucket'); + return Template.fromStack(regionalStack); + }); + const longNameStack = new Stack(new App(), 'a'.repeat(128), { + env: { account: '123456789012', region: 'us-west-2' }, + }); + new ScreenshotBucket(longNameStack, 'FirstScreenshots'); + new ScreenshotBucket(longNameStack, 'SecondScreenshots'); + longNameTemplate = Template.fromStack(longNameStack); + }); + + function accessControlNames(source: Template): string[] { + return Object.values(source.findResources('AWS::CloudFront::OriginAccessControl')) + .map((resource) => resource.Properties.OriginAccessControlConfig.Name); + } + + test('same stack in different regions has distinct global access-control names', () => { + const names = regionalTemplates.flatMap(accessControlNames); + expect(names).toHaveLength(2); + expect(names[0]).toMatch(/-us-east-1$/); + expect(names[1]).toMatch(/-us-west-2$/); + expect(new Set(names).size).toBe(2); + }); + + test('long stack names preserve distinct access controls within the 64-character limit', () => { + const names = accessControlNames(longNameTemplate); + expect(new Set(names).size).toBe(2); + for (const name of names) { + expect(name.length).toBeLessThanOrEqual(64); + expect(name).toMatch(/-us-west-2$/); + } + }); + + test('access control always signs S3 requests with sigv4', () => { + template.hasResourceProperties('AWS::CloudFront::OriginAccessControl', { + OriginAccessControlConfig: { + OriginAccessControlOriginType: 's3', + SigningBehavior: 'always', + SigningProtocol: 'sigv4', + }, + }); }); // Lock in the screenshot bucket lifecycle defaults. diff --git a/cdk/test/constructs/task-dashboard.test.ts b/cdk/test/constructs/task-dashboard.test.ts index fc12dba03..185b67d28 100644 --- a/cdk/test/constructs/task-dashboard.test.ts +++ b/cdk/test/constructs/task-dashboard.test.ts @@ -22,9 +22,11 @@ import { Template } from 'aws-cdk-lib/assertions'; import * as logs from 'aws-cdk-lib/aws-logs'; import { TaskDashboard } from '../../src/constructs/task-dashboard'; -function createStack(): { stack: Stack; template: Template } { +function createStack(region?: string): { stack: Stack; template: Template } { const app = new App(); - const stack = new Stack(app, 'TestStack'); + const stack = new Stack(app, 'TestStack', region + ? { env: { account: '123456789012', region } } + : {}); const logGroup = new logs.LogGroup(stack, 'AppLogGroup'); @@ -38,16 +40,36 @@ function createStack(): { stack: Stack; template: Template } { } describe('TaskDashboard construct', () => { + let template: Template; + let regionalTemplates: Template[]; + + beforeAll(() => { + ({ template } = createStack()); + regionalTemplates = ['us-east-1', 'us-west-2'].map((region) => createStack(region).template); + }); + test('creates a CloudWatch Dashboard', () => { - const { template } = createStack(); template.resourceCountIs('AWS::CloudWatch::Dashboard', 1); }); - test('dashboard name includes stack name', () => { - const { template } = createStack(); + test('dashboard name includes stack name and deployment region', () => { template.hasResourceProperties('AWS::CloudWatch::Dashboard', { - DashboardName: 'BackgroundAgent-Tasks-TestStack', + DashboardName: { + 'Fn::Join': ['', ['BackgroundAgent-Tasks-TestStack-', { Ref: 'AWS::Region' }]], + }, + }); + }); + + test('same stack name in two regions creates distinct global dashboard names', () => { + const names = regionalTemplates.map((regionalTemplate) => { + const dashboards = regionalTemplate.findResources('AWS::CloudWatch::Dashboard'); + return Object.values(dashboards)[0].Properties.DashboardName; }); + expect(names).toEqual([ + 'BackgroundAgent-Tasks-TestStack-us-east-1', + 'BackgroundAgent-Tasks-TestStack-us-west-2', + ]); + expect(new Set(names).size).toBe(2); }); // --- Chunk 8b: Cedar HITL approval widgets (§11.3, IMPL-28) ------------ @@ -55,8 +77,8 @@ describe('TaskDashboard construct', () => { // The dashboard body is serialized as CloudFormation ``Fn::Join`` parts // with CDK tokens for the stack region/account. Use ``Match.serializedJson`` // / substring checks via template rendering. - function dashboardBodyContains(template: Template, needle: string): boolean { - const dashboards = template.findResources('AWS::CloudWatch::Dashboard'); + function dashboardBodyContains(source: Template, needle: string): boolean { + const dashboards = source.findResources('AWS::CloudWatch::Dashboard'); for (const res of Object.values(dashboards)) { const body = (res as any).Properties?.DashboardBody; if (typeof body === 'string') { @@ -72,12 +94,10 @@ describe('TaskDashboard construct', () => { } test('dashboard references ABCA/Cedar-HITL namespace for the new metrics', () => { - const { template } = createStack(); expect(dashboardBodyContains(template, 'ABCA/Cedar-HITL')).toBe(true); }); test('dashboard includes ApprovalTimeoutClipRate widget (MathExpression with IF guard)', () => { - const { template } = createStack(); expect(dashboardBodyContains(template, 'Approval Timeout Clip Rate')).toBe(true); // The IF(requested > 0, ...) guard is critical to avoid divide-by-zero // NaN renders (silent gap). Assert the expression ships in the widget. @@ -89,13 +109,11 @@ describe('TaskDashboard construct', () => { }); test('dashboard references ClippedApprovalCount and ApprovalRequestCount metrics', () => { - const { template } = createStack(); expect(dashboardBodyContains(template, 'ClippedApprovalCount')).toBe(true); expect(dashboardBodyContains(template, 'ApprovalRequestCount')).toBe(true); }); test('dashboard includes ApprovalTimeoutBreakdown widget with p50/p90/p99 on TimedOutEffectiveTimeout', () => { - const { template } = createStack(); expect(dashboardBodyContains(template, 'Approval Timeout Breakdown')).toBe(true); expect(dashboardBodyContains(template, 'TimedOutEffectiveTimeout')).toBe(true); // All three percentiles required by §11.3. @@ -105,7 +123,6 @@ describe('TaskDashboard construct', () => { }); test('dashboard includes ApprovalDecisionLatency widget with outcome dims', () => { - const { template } = createStack(); expect(dashboardBodyContains(template, 'Approval Decision Latency')).toBe(true); expect(dashboardBodyContains(template, 'ApprovalDecisionLatencyMs')).toBe(true); // Three outcome dim values — one series set per outcome per percentile. diff --git a/docs/design/OBSERVABILITY.md b/docs/design/OBSERVABILITY.md index 96441db95..5f0efbab8 100644 --- a/docs/design/OBSERVABILITY.md +++ b/docs/design/OBSERVABILITY.md @@ -139,7 +139,7 @@ Emitted as custom CloudWatch metrics and used in dashboards and alarms. ## Dashboard -A CloudWatch dashboard (`BackgroundAgent-Tasks-${stackName}`, i.e. the base name suffixed with the stack name) is deployed via the `TaskDashboard` CDK construct. It provides Logs Insights widgets for: +A CloudWatch dashboard (`BackgroundAgent-Tasks-${stackName}-${region}`) is deployed via the `TaskDashboard` CDK construct. Dashboard names are account-global, so the Region suffix lets identically named stacks coexist in different Regions. Updating a deployment from the earlier name replaces its dashboard resource; metric and log data remain in their existing stores, but saved dashboard links need the new name. It provides Logs Insights widgets for: - Task success rate and count by status - Cost per task and turns per task diff --git a/docs/src/content/docs/architecture/Observability.md b/docs/src/content/docs/architecture/Observability.md index a0991020e..841c2695a 100644 --- a/docs/src/content/docs/architecture/Observability.md +++ b/docs/src/content/docs/architecture/Observability.md @@ -143,7 +143,7 @@ Emitted as custom CloudWatch metrics and used in dashboards and alarms. ## Dashboard -A CloudWatch dashboard (`BackgroundAgent-Tasks-${stackName}`, i.e. the base name suffixed with the stack name) is deployed via the `TaskDashboard` CDK construct. It provides Logs Insights widgets for: +A CloudWatch dashboard (`BackgroundAgent-Tasks-${stackName}-${region}`) is deployed via the `TaskDashboard` CDK construct. Dashboard names are account-global, so the Region suffix lets identically named stacks coexist in different Regions. Updating a deployment from the earlier name replaces its dashboard resource; metric and log data remain in their existing stores, but saved dashboard links need the new name. It provides Logs Insights widgets for: - Task success rate and count by status - Cost per task and turns per task From edaa144cf1866c366d60395a995cbea822fdd40b Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Sun, 13 Sep 2026 22:48:40 -0400 Subject: [PATCH 021/149] fix(cdk): scope registry waiter name to deployment (#645) --- cdk/src/constructs/registry.ts | 19 ++++++- cdk/test/constructs/registry.test.ts | 49 ++++++++++++++++++- docs/design/DEPLOYMENT_ROLES.md | 6 +++ .../docs/architecture/Deployment-roles.md | 6 +++ 4 files changed, 78 insertions(+), 2 deletions(-) diff --git a/cdk/src/constructs/registry.ts b/cdk/src/constructs/registry.ts index 0332f4669..23d9e1c3a 100644 --- a/cdk/src/constructs/registry.ts +++ b/cdk/src/constructs/registry.ts @@ -23,7 +23,7 @@ // this wraps the CDK Provider framework: an `onEvent` Lambda starts the mutation // and an `isComplete` Lambda is polled until the registry reaches a stable state. import * as path from 'path'; -import { CustomResource, Duration, NestedStack, type NestedStackProps, Stack } from 'aws-cdk-lib'; +import { CfnResource, CustomResource, Duration, Names, NestedStack, type NestedStackProps, Stack } from 'aws-cdk-lib'; import * as iam from 'aws-cdk-lib/aws-iam'; import { Architecture, Runtime } from 'aws-cdk-lib/aws-lambda'; import * as lambda from 'aws-cdk-lib/aws-lambda-nodejs'; @@ -147,6 +147,23 @@ export class AgentRegistry extends Construct { totalTimeout: TOTAL_TIMEOUT, }); + // CloudFormation's generated waiter name omits the parent stack prefix, + // falling outside the bootstrap policy's backgroundagent-dev-* namespace. + // Provider has no public naming option, so use its L1 escape hatch. Include + // the outer stack name and a path hash, even inside a nested stack. + const waiters = provider.node.findAll().filter( + (node): node is CfnResource => CfnResource.isCfnResource(node) + && node.cfnResourceType === 'AWS::StepFunctions::StateMachine', + ); + if (waiters.length !== 1) { + throw new Error(`Expected one Agent Registry provider waiter, found ${waiters.length}`); + } + waiters[0].addPropertyOverride('StateMachineName', Names.uniqueResourceName(provider, { + maxLength: 80, + separator: '-', + allowedSpecialCharacters: '-', + })); + const resource = new CustomResource(this, 'Resource', { serviceToken: provider.serviceToken, // Changing the resource type forces replacement from the retired preview diff --git a/cdk/test/constructs/registry.test.ts b/cdk/test/constructs/registry.test.ts index 63f84ad72..2cc2992bf 100644 --- a/cdk/test/constructs/registry.test.ts +++ b/cdk/test/constructs/registry.test.ts @@ -19,7 +19,8 @@ import { App, Stack } from 'aws-cdk-lib'; import { Template, Match } from 'aws-cdk-lib/assertions'; -import { AgentRegistry } from '../../src/constructs/registry'; +import { applicationPolicy } from '../../src/bootstrap/policies/application'; +import { AgentRegistry, AgentRegistryStack } from '../../src/constructs/registry'; function createStack(): Template { const app = new App(); @@ -153,3 +154,49 @@ describe('AgentRegistry construct', () => { expect(JSON.stringify(statement?.Resource)).toContain('registry/*'); }); }); + +describe.each([ + 'backgroundagent-dev', + `backgroundagent-dev-${'x'.repeat(108)}`, +])('registry waiter in parent stack %s', stackName => { + let names: string[]; + + beforeAll(() => { + const app = new App(); + const parent = new Stack(app, 'Parent', { + stackName, + env: { account: '123456789012', region: 'us-west-2' }, + }); + const registries = ['FirstRegistry', 'SecondRegistry'].map( + id => new AgentRegistryStack(parent, id, { registryName: id }), + ); + names = registries.map(nested => { + const waiters = Object.values( + Template.fromStack(nested).findResources('AWS::StepFunctions::StateMachine'), + ); + expect(waiters).toHaveLength(1); + return waiters[0].Properties.StateMachineName as string; + }); + }); + + test('uses names permitted by the deployed bootstrap policy', () => { + const statement = applicationPolicy().toJSON().Statement.find( + (candidate: { Sid: string }) => candidate.Sid === 'StepFunctions', + ); + const resourcePattern = new RegExp( + `^${(statement.Resource as string).split('*').join('.*')}$`, + ); + for (const name of names) { + expect(name).toBeDefined(); + expect(`arn:aws:states:us-west-2:123456789012:stateMachine:${name}`) + .toMatch(resourcePattern); + } + }); + + test('keeps names distinct and within the Step Functions name limit', () => { + expect(new Set(names).size).toBe(2); + for (const name of names) { + expect(name).toMatch(/^[A-Za-z0-9-]{1,80}$/); + } + }); +}); diff --git a/docs/design/DEPLOYMENT_ROLES.md b/docs/design/DEPLOYMENT_ROLES.md index 95b092ac9..972f30440 100644 --- a/docs/design/DEPLOYMENT_ROLES.md +++ b/docs/design/DEPLOYMENT_ROLES.md @@ -301,6 +301,12 @@ CloudFormation stack operations, IAM roles/policies, VPC networking, and Route 5 DynamoDB tables, Lambda functions, API Gateway, Cognito, WAFv2, EventBridge, SQS, CloudFront, and Secrets Manager. When ECS Fargate compute is enabled, add the ECS statement below to this policy. +Agent Registry provisioning also uses a Step Functions workflow to wait for +asynchronous creation and deletion. Its construct explicitly names the workflow +with the parent stack's `backgroundagent-dev-` prefix so it fits this policy, +including when deployed in a nested stack. Adopting this name in an existing +deployment replaces the provider's waiter state machine. + ```json { "Version": "2012-10-17", diff --git a/docs/src/content/docs/architecture/Deployment-roles.md b/docs/src/content/docs/architecture/Deployment-roles.md index 8612b184e..2bb4ff60c 100644 --- a/docs/src/content/docs/architecture/Deployment-roles.md +++ b/docs/src/content/docs/architecture/Deployment-roles.md @@ -305,6 +305,12 @@ CloudFormation stack operations, IAM roles/policies, VPC networking, and Route 5 DynamoDB tables, Lambda functions, API Gateway, Cognito, WAFv2, EventBridge, SQS, CloudFront, and Secrets Manager. When ECS Fargate compute is enabled, add the ECS statement below to this policy. +Agent Registry provisioning also uses a Step Functions workflow to wait for +asynchronous creation and deletion. Its construct explicitly names the workflow +with the parent stack's `backgroundagent-dev-` prefix so it fits this policy, +including when deployed in a nested stack. Adopting this name in an existing +deployment replaces the provider's waiter state machine. + ```json { "Version": "2012-10-17", From 29dcaa74545b3148928b0228dcd9799113002823 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Sun, 13 Sep 2026 23:52:54 -0400 Subject: [PATCH 022/149] fix(cdk): retain scoped Jira exception in overflow policy (#645) --- cdk/src/stacks/agent.ts | 44 +++++++------ .../stacks/microvm-managed-image-nag.test.ts | 65 +++++++++++++++++++ 2 files changed, 88 insertions(+), 21 deletions(-) create mode 100644 cdk/test/stacks/microvm-managed-image-nag.test.ts diff --git a/cdk/src/stacks/agent.ts b/cdk/src/stacks/agent.ts index 77cb28fe9..76c79c72d 100644 --- a/cdk/src/stacks/agent.ts +++ b/cdk/src/stacks/agent.ts @@ -27,7 +27,7 @@ import * as iam from 'aws-cdk-lib/aws-iam'; import * as logs from 'aws-cdk-lib/aws-logs'; import * as secretsmanager from 'aws-cdk-lib/aws-secretsmanager'; import * as cr from 'aws-cdk-lib/custom-resources'; -import { NagSuppressions } from 'cdk-nag'; +import { NagSuppressions, type NagPackSuppression } from 'cdk-nag'; import { Construct, IConstruct } from 'constructs'; import { AdmissionQueuePickup } from '../constructs/admission-queue-pickup'; import { AgentMemory } from '../constructs/agent-memory'; @@ -813,24 +813,15 @@ export class AgentStack extends Stack { }, ], true); - // Chunk 10 deploy-prep: the Cedar HITL additions (TaskApprovalsTable - // grant + extra env vars) pushed the runtime - // execution role past CDK's per-inline-policy size limit, causing CDK - // to auto-split excess statements into ``OverflowPolicy1`` / etc. - // Those overflow policies inherit the same wildcard - // ``bedrock:InvokeModel*`` / CloudWatch / cross-region-inference - // actions as the base policy but live at paths that any suppression - // placed at constructor time does NOT reach (CDK creates the - // overflow policies lazily during synth ``prepare()``, after the - // construct tree has been frozen). Use an Aspect that visits every - // node during synth and matches overflow-policy children of the - // runtime ExecutionRole so any present or future overflow is - // suppressed automatically without hardcoding - // ``OverflowPolicy`` indices. - // Roles known to overflow, with the evidence for each. Keyed by a path - // fragment rather than an `OverflowPolicy` index so future splits are - // covered automatically. - const OVERFLOW_SUPPRESSIONS: readonly { readonly pathFragment: string; readonly reason: string }[] = [ + // CDK splits large role policies during synth, after constructor-time + // suppressions have visited the existing children. Apply the documented + // exceptions to those later policies before cdk-nag inspects them, matching + // the owning role without relying on a particular OverflowPolicy index. + const OVERFLOW_SUPPRESSIONS: readonly { + readonly pathFragment: string; + readonly reason: string; + readonly appliesTo?: NagPackSuppression['appliesTo']; + }[] = [ { pathFragment: '/Runtime/ExecutionRole/OverflowPolicy', reason: @@ -846,14 +837,25 @@ export class AgentStack extends Stack { reason: 'CDK-generated overflow policy on the Linear webhook processor role carries the kms:GenerateDataKey* that SNS Topic.grantPublish emits for the CMK-encrypted operational-alerts topic. Scoped to that single topic key; the wildcard only spans the GenerateDataKey/GenerateDataKeyWithoutPlaintext pair.', }, + { + // Image lifecycle grants can push the existing Jira secret grant into + // an overflow policy. Exempt only that resource pattern; other wildcard + // grants in this role's future overflow documents still require review. + pathFragment: '/TaskOrchestrator/OrchestratorFn/ServiceRole/OverflowPolicy', + reason: + 'The orchestrator reads and refreshes per-tenant Jira OAuth secrets created by bgagent jira setup. Their cloudId-based names are unknown at synth, so GetSecretValue/PutSecretValue use the account- and Region-scoped bgagent-jira-oauth-* prefix.', + appliesTo: [{ + regex: '/^Resource::arn:.*:secretsmanager:.*:secret:bgagent-jira-oauth-\\*$/', + }], + }, ]; const overflowSuppressionAspect = { visit(node: IConstruct) { const nodePath = node.node.path; if (!nodePath.endsWith('/Resource')) return; - for (const { pathFragment, reason } of OVERFLOW_SUPPRESSIONS) { + for (const { pathFragment, reason, appliesTo } of OVERFLOW_SUPPRESSIONS) { if (nodePath.includes(pathFragment)) { - NagSuppressions.addResourceSuppressions(node, [{ id: 'AwsSolutions-IAM5', reason }]); + NagSuppressions.addResourceSuppressions(node, [{ id: 'AwsSolutions-IAM5', reason, appliesTo }]); return; } } diff --git a/cdk/test/stacks/microvm-managed-image-nag.test.ts b/cdk/test/stacks/microvm-managed-image-nag.test.ts new file mode 100644 index 000000000..92be4fb56 --- /dev/null +++ b/cdk/test/stacks/microvm-managed-image-nag.test.ts @@ -0,0 +1,65 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import { ManagedPolicy, PolicyStatement, Role } from 'aws-cdk-lib/aws-iam'; +import { AGENTCORE_AZS_CONTEXT_KEY } from '../../src/constructs/agentcore-azs'; +import { TaskOrchestrator } from '../../src/constructs/task-orchestrator'; +import { buildApp } from '../../src/main'; + +describe.each([false, true])('managed MicroVM image security checks (extra wildcard: %s)', extraWildcard => { + let errors: string[]; + + beforeAll(async () => { + const app = await buildApp({ + account: '123456789012', + region: 'us-west-2', + appProps: { + context: { + compute_type: 'lambda-microvm', + microvm_base_image_arn: 'arn:aws:lambda:us-west-2:aws:microvm-image:al2023-1', + microvm_base_image_version: '1', + [AGENTCORE_AZS_CONTEXT_KEY]: ['us-west-2a', 'us-west-2b'], + }, + }, + }); + const stack = app.node.findChild('backgroundagent-dev'); + if (extraWildcard) { + const orchestrator = stack.node.findChild('TaskOrchestrator') as TaskOrchestrator; + // Model a future overflow document without relying on the policy + // splitter placing an extra statement in a particular document. + new ManagedPolicy(orchestrator.fn.role as Role, 'OverflowPolicy999', { + statements: [new PolicyStatement({ + actions: ['ssm:GetParameter'], + resources: ['arn:aws:ssm:us-west-2:123456789012:parameter/unrelated-*'], + })], + }); + } + // The real entry point installs cdk-nag. The CLI fails on these assembly + // annotations even though in-process synth itself does not throw. + const artifact = app.synth().getStackByName('backgroundagent-dev'); + errors = artifact.messages.filter(message => message.level === 'error') + .map(message => String(message.entry.data)); + }, 60_000); + + test('accepts documented Jira access while still reporting unrelated wildcard grants', () => { + expect(errors).toEqual(extraWildcard + ? [expect.stringMatching(/AwsSolutions-IAM5.*parameter\/unrelated-\*/)] + : []); + }); +}); From 98aed92dc0f74ce506e0cfacde16b0d73a678083 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Mon, 14 Sep 2026 00:49:19 -0400 Subject: [PATCH 023/149] docs: record clean P2 deployment and remaining live gates (#645) --- .../645-p2-clean-deployment-20260913.md | 340 ++++++++++++++++++ .../645-p3-implementation-plan.md | 20 +- 2 files changed, 359 insertions(+), 1 deletion(-) create mode 100644 docs/verification/645-p2-clean-deployment-20260913.md diff --git a/docs/verification/645-p2-clean-deployment-20260913.md b/docs/verification/645-p2-clean-deployment-20260913.md new file mode 100644 index 000000000..cfb9de859 --- /dev/null +++ b/docs/verification/645-p2-clean-deployment-20260913.md @@ -0,0 +1,340 @@ +# ADR-021 P2 clean deployment — 2026-09-13–14 + +Live verification of the P2 source fixes and deployment prerequisites from +[`645-p3-implementation-plan.md`](./645-p3-implementation-plan.md). + +**Status: infrastructure and managed image deployed.** CloudFormation reached +`UPDATE_COMPLETE`; image version `1.0` is `SUCCESSFUL` and `ACTIVE`. Build hooks +and authenticated API reads pass. The normal coding-task test remains pending +repository choice and GitHub token setup. P2 acceptance remains open, and the +smoke warnings have not been discharged. + +## Target and isolation + +- Source: `fix/645-microvm-readiness`, deployed commit `29dcaa74`. + The initial runtime revision was `eb7e2071a3affa09ca0fd2f0064028545537d21f`; + the four deployment fixes below are included in the final source. +- AWS profile: `sphia-dev`; account: ``; Region: `us-west-2`. +- Application stack: `backgroundagent-dev`; bootstrap stack: `CDKToolkit`. +- Compute selection: `lambda-microvm`, with the existing AgentCore resources + still provisioned. Bootstrap compute policies: `agentcore,lambda-microvm`. + This provisions a backend; each repository separately chooses its backend. +- Managed base image: + `arn:aws:lambda:us-west-2:aws:microvm-image:al2023-1`, version `1`. +- Local evidence: `/tmp/abca-645-p2-clean-20260913`. + +The existing application and shared bootstrap in `us-east-1` are outside this +deployment. That Region already has five VPCs against a quota of five, and its +Bedrock invocation logging belongs to the existing development stack. No existing +VPC was deleted and no quota increase was requested. + +Before this run, `us-west-2` had no active CloudFormation stacks, one default VPC +(`vpc-053c23d0618a71e91`), and no Bedrock invocation logging configuration. +Using this Region permits the repository's normal stack name and bootstrap +qualifier without modifying another application's resources. + +## Local prerequisites + +Docker is available on native ARM64. The installed mise `2026.2.8` cannot parse +the root monorepo task configuration. This run uses the official macOS ARM64 +mise `2026.9.7` binary in the evidence directory's `tools/` folder, with its +SHA-256 checked against the GitHub release: +`3c3f377e7123a466274a20f01502ddd8c58f76028907f471c9bc42fbf83846e1`. +The global tool installation was not changed. + +The full root build runs with `JEST_MAX_WORKERS=2` and `MISE_JOBS=2`. +The final deployed source passed in 704.66 seconds; its output is retained as +`full-build-managed-image-nag.log`. + +**Final result: passed**, exit 0. CDK: 222 suites / 4,783 tests; +CLI: 62 suites / 928 tests; Python: 1,823 tests. Compilation, lint, formatting, +type checks, the 77-page documentation build and drift checks passed. Two CDK suites / 38 +DynamoDB Local integration cases were skipped because that local service was +not running; those cases passed during the preceding implementation session. + +Earlier builds are recorded below beside the fixes they validated. Target +synthesis was run after the final build and passed. Deployment cleanup must +not run concurrently with tests/synthesis using the shared CDK temporary files. + +## Acceptance record + +- [x] Full root build passes on deployed infrastructure source (`29dcaa74`). +- [x] Fresh least-privilege bootstrap has the required policies. +- [x] No-image application infrastructure deploys from source. +- [x] Current agent artifact is packaged and uploaded. +- [x] CloudFormation creates the managed image using the pinned base. +- [x] Image version is active; `/ready` and `/validate` return HTTP 200. +- [x] Coordinator image/network/role configuration and authenticated API reads are verified. +- [ ] A normal task completes with live progress and heartbeat evidence. +- [ ] Runtime logs, payload cleanup and VM termination are verified. +- [ ] Relevant allowed/denied IAM and network cases are exercised. +- [x] Current resource state, temporary-login cleanup and remaining limits are recorded. + +Automatic P3 suspension remains disabled in this source revision. A successful +P2 deployment would not establish P3 lifecycle correctness. + +## Bootstrap and application synthesis + +`CDKToolkit` reached `CREATE_COMPLETE` with bootstrap version `32` and ABCA +policy bundle `1.6.0`. The original generated template was deployed through +CloudFormation so the custom `ComputeTypes=agentcore,lambda-microvm` parameter +could be supplied at creation. No generated template or live IAM policy was +patched. The reviewed change set contained 18 additions. + +At initial bootstrap 1.6.0 creation, the CloudFormation execution role had exactly +the five expected ABCA policies: +Infrastructure, Application, Observability, Compute-Agentcore and +Compute-LambdaMicrovms. Each deployed policy document equaled its source JSON. +There were no inline execution-role policies and no `AdministratorAccess`. +Bootstrap 1.7.0 adds the exact-role inline policy described below while +preserving those five managed policies. +Evidence: `bootstrap-policy-comparison.json`. + +The target-specific no-image synthesis passed for +`aws:///us-west-2`: 472 parent resources and a 693,882-byte parent +template. Registry resources occupy two nested stacks. Asset publication and +the application change-set preparation follow this reviewed cloud assembly. + +## First application attempt and source fix + +The agent container built and every code/image asset uploaded successfully. +Application change-set preparation then failed before resource creation: +CloudFormation's execution role lacked `iam:PassRole` on itself while validating +the registry nested stacks. Evidence: `substrate-prepare.log`. + +The source fix adds `PassExecutionRoleToCloudFormation` as a generated inline +policy, permitting only the exact execution role ARN and only the CloudFormation +service. Bootstrap bundle `1.7.0` is required. No live IAM patch was applied. + +The bootstrap hash also needed correction: its root-key JSON replacer omitted +nested Action, Effect, Resource and Condition changes. It now recursively sorts +keys, keeps those fields, and includes the new inline policy. + +Five new regressions failed on old source: the missing self-role policy and four +permission changes that left the old hash unchanged. After the fixes, the two +bootstrap suites pass all 58 tests, including object-key ordering and inline-policy +hash coverage. The generated template is 47,446 bytes on disk; the CDK CLI sends +46,307 characters, within both the budget and AWS's inline limit. + +The complete build passed again: 4,773 CDK tests, 928 CLI tests and 1,823 Python +tests; total runtime 362.54 seconds. The fix is committed as `16941828`. +Bootstrap 1.7.0 reached `UPDATE_COMPLETE`. Its actual inline policy equals the +resolved source policy, its five managed attachments are unchanged, and every +existing bootstrap parameter was preserved. AWS IAM simulation verified six +resource/service combinations, allowing only this role passed to CloudFormation. + +## Second application attempt: global names + +The application retry passed the self-PassRole check, then failed early +validation because `BackgroundAgent-Tasks-backgroundagent-dev` already exists. +CloudWatch dashboard names are account-global. The existing dashboard was last +modified on 2026-08-28 and was not changed by this run. + +An audit of the other global resources found a second collision before another +attempt: the screenshot CloudFront origin access control (OAC), +`backgroundagentdevGitHubScreOrigin1S3OriginAccessControl544B0DF8`, already exists +as `E7LV4IA65D59K`. The OAC controls how CloudFront authenticates requests to the +private screenshot bucket. It belongs to the existing deployment and must not +be deleted or repurposed. + +The source now gives the dashboard a Region suffix and gives each screenshot +OAC a bounded construct-path name plus Region. Four regressions failed the old +names; all 15 tests in the two construct suites now pass. Tests cover identical +stack names in two Regions, long names, distinct controls, and SigV4 signing. + +For existing deployments adopting this source, the dashboard and OAC resources +are replaced; the distribution references the new OAC. Dashboard bookmarks need +the new name. This run updates only the new Oregon deployment. + +The full build passed: 4,777 CDK tests, 928 CLI tests, 1,823 Python tests and +77 documentation pages; total runtime 358.95 seconds. The fix is committed as +`4ea7e88b`. Refreshed target synthesis passed with 472 parent resources. + +## Third application attempt: registry waiter + +The third change set passed validation and all 472 additions were reviewed +before execution. Actual creation failed in the registry nested stack: +CloudFormation generated the Step Functions name +`AgentRegistryProviderwaiterstatemachineE27177B2-SxOkWE4lO56U`, outside the +bootstrap policy's `backgroundagent-dev-*` namespace. The execution role therefore +denied `states:CreateStateMachine`. + +The registry construct now explicitly names this helper using the outer stack +name and a bounded construct-path hash. No bootstrap permission expansion is +needed. Four regressions failed on old source; the registry suite now passes +all 11 tests, including actual nested templates, long stack names, two distinct +registries, and comparison against the bootstrap ARN pattern. The full root +build passed in 715.27 seconds: 4,781 CDK tests, 928 CLI tests, 1,823 Python +tests and 77 documentation pages. The fix is committed as `edaa144c`. + +The target template now names the waiter +`backgroundagent-dev-AgentRegistryStack-AgentRegistry-Provider-77C0EB2A`. +IAM simulation using the actual deployed execution role allows its creation +and denies the old name. The registry nested template changes only that name; +the parent remains at 472 resources. Evidence: +`registry-waiter-iam-simulation.json` and `substrate-waiter-template-review.json`. + +Rollback initially could not delete the two MicroVM network connectors and +AgentCore memory while AWS was still creating them. A normal stack deletion +after they stabilized reached `DELETE_COMPLETE`, including the connectors, +memory and VPC. No forced deletion or VPC retention was used. This removed only +the new failed Oregon application; the Oregon bootstrap remains for the retry. + +Post-deletion inventory found one intentionally retained API Gateway logging +role and two regional logging settings. With no REST or HTTP APIs remaining, +the API Gateway setting was cleared and that exact failed-stack role deleted. +Bedrock invocation logging, still pointing to the deleted stack's log group and +role, was cleared too. Both settings now match the observed empty initial state. +Stack and VPC inventories now contain only the new bootstrap and the original +default VPC; existing resources in other Regions were not changed. + +## Fourth application attempt + +Change set `abca-645-p2-substrate-r4` passed AWS validation. Its 472 changes are +all additions, with the intended restricted CloudFormation execution role. +The proposal was executed for the new stack +`df112e20-afe6-11f1-963f-02fff900428b`. The stack and all 472 root resources +reached `CREATE_COMPLETE` at 2026-09-14 03:03 UTC. +Source fixes are `16941828`, `4ea7e88b` and `edaa144c` on top of the original +runtime revision recorded above. Bootstrap remains 1.7.0. + +The registry and its waiter both reached `CREATE_COMPLETE`. The registry ID is +`5v3pQEkpkvjdS4q6`; the waiter uses the explicit name above. This verifies the +previous failure is fixed during actual resource creation, not only simulation. +Evidence: `substrate-r4-registry-resources.json`. + +Both connectors are `ACTIVE`. `aws lambda-core` is the CLI namespace for network +connectors. Their actual security groups have no ingress rules; runtime egress +allows TCP 443, while build egress allows TCP 80 and 443. This verifies deployed +configuration; in-guest traffic tests remain outstanding. + +The DNS firewall is in the source's hardcoded observation mode. Live rules +allow baseline/additional domains at priorities 100/200 and use a catch-all +`ALERT` at priority 300. Unlisted domains are logged, not blocked; domain +allowlisting must not be claimed as an enforced isolation boundary here. +Evidence: `live-dns-firewall-rules.json`. + +## Agent artifact and API checks + +The official packaging script uploaded the artifact successfully. Downloading +that S3 object and comparing the 106 image inputs against the checkout verified +every byte. The ZIP is 1,046,220 bytes with SHA-256 +`e6f37030b0517374359927ac3c1480d85195fba37db8052ed339b9a34dcd94a1`. +Evidence: `uploaded-artifact-verification.json`. + +A temporary user in the new Cognito pool authenticated through the built CLI, +and an authenticated task list returned an empty result. Missing and invalid +tokens both returned HTTP 401. Invitation messages were suppressed. The CLI +used a separate private configuration, leaving the operator's default login +untouched. Authenticated listing passed again after the managed-image update. + +Platform doctor passes the API, Cognito, active-repo and model-catalog checks. +It fails only the GitHub token check: the new secret still contains its +placeholder. Repository choice and GitHub setup remain pending. Catalog +visibility alone does not prove runtime model invocation. +Evidence: `post-update-task-list.json` and `post-update-platform-doctor.json`. + +`bgagent runtime status` confirms that the only seeded repository, +`awslabs/agent-plugins`, still resolves to **AgentCore**, whose control plane is +`READY`. There are no repositories configured for MicroVM yet. Stack output +`ComputeSubstrate=lambda-microvm` describes provisioned infrastructure; it does +not change the seeded repository's selection. Onboard the chosen test repository +with `--compute-type lambda-microvm` before submitting a task. Evidence: +`post-update-runtime-status.json`. + +## Managed-image synthesis: overflow-policy check + +The first managed-image synth failed locally on one `AwsSolutions-IAM5` finding: +the orchestrator's Jira OAuth secret-prefix grant moved into `OverflowPolicy1` +when image lifecycle permissions enlarged the role's policy. Constructor-time +suppression metadata does not reach policies generated later during synthesis. +The source fix extends the existing overflow-policy Aspect only for that role +and that resource pattern. Both production-entry-point regression cases pass, +including an unrelated wildcard grant that still triggers an error. The fix is +committed as `29dcaa74`; the final full-build result is recorded above. + +The corrected managed-image synth passed with 474 parent resources and a +697,733-byte template. Compared with the failed synth, all IAM Role, Policy +and ManagedPolicy resource properties are identical: the fix changes the narrow +security-check exception metadata, not permissions. Evidence: +`managed-image-template-review.json` and `managed-image-synth-r2.log`. + +## Managed-image deployment and live checks + +Change set `abca-645-p2-managed-image-r1` contained 42 changes: four additions, +36 modifications and two removals. The removals were old immutable Lambda and +guardrail versions. No existing data bucket, table or VPC was replaced. +CloudFormation executed the update successfully; `UPDATE_COMPLETE` was observed +at 2026-09-14 04:39:46 UTC. The stack has 474 root resources. + +| Item | Observed result | +|---|---| +| Stack | `backgroundagent-dev`, ID suffix `df112e20-afe6-11f1-963f-02fff900428b`, `UPDATE_COMPLETE` | +| Image | `arn:aws:lambda:us-west-2::microvm-image:backgroundagent-dev-abca-agent` | +| Image record | `CREATED`, `latestActiveImageVersion=1.0` | +| Image version `1.0` | Build state `SUCCESSFUL`, status `ACTIVE` | +| Base | `al2023-1`; input version `1` is reported by AWS as `1.0` | +| Configuration | `ARM_64`, 8,192 MiB minimum memory, hook port 8080 | +| Build hooks | Ready enabled / 300 seconds; validate enabled / 60 seconds | +| Runtime hooks | Run enabled / 60 seconds; terminate enabled / 15 seconds; no suspend/resume hooks | +| CloudWatch logs | `/aws/lambda-microvms/backgroundagent-dev-abca-agent` | +| Active task VMs | None; no coding task has been submitted | + +The actual `/ready` log records warm-up of Claude CLI 2.1.191, git 2.47.3 and +Node 24.21.0, followed by HTTP 200. `/validate` reports Python 3.13.13, +13 supported platform configuration keys and zero validator warnings, followed +by HTTP 200. Some generic Uvicorn “Invalid HTTP request received” warnings +precede validation; their cause was not established. These build-hook results +do not exercise runtime AWS credentials, model invocation or task execution. + +The live coordinator alias points to version `2`, with Lambda state `Active` +and last update `Successful`. Its actual environment includes the managed image +ARN, the runtime execution role, runtime egress connector, AWS `NO_INGRESS` +connector and the dedicated payload bucket. No image-version override is set, +so launch resolves the latest active version. + +Evidence: `managed-image-change-set-r1.json`, `managed-image-r1-state.json`, +`managed-image-r1-resources.json`, `final-managed-image.json`, +`final-managed-image-version.json`, `final-orchestrator-configuration.json`, +`final-microvms.json` and `image-build-log-snapshot.json`. + +## Retained deployment, cleanup and next verification + +The successful application, managed image, artifact and bootstrap remain live. +The default Oregon VPC and existing deployment in other Regions were preserved. +The failed earlier application was fully removed as described above. +The final read at 2026-09-14 04:48:54 UTC confirms `UPDATE_COMPLETE`, no task +MicroVMs and zero objects in the payload bucket. Since no task ran, an empty +bucket does not prove the cleanup path works. Evidence: +`final-deployment-status.json`. + +After the final authenticated API check, the temporary Cognito user +`p2-verification-20260913@example.invalid` was deleted from the new pool. +`AdminGetUser` confirmed `UserNotFoundException`. Its three private request files +and cached CLI credentials were removed. The non-secret isolated CLI +configuration and deployment evidence remain. Evidence: +`verification-user-cleanup.json`. + +The next verification sequence is: + +1. Select the test repository and populate the new deployment's GitHub token + using `bgagent github set-token --region us-west-2 --stack-name backgroundagent-dev`. + No GitHub token was copied and no repository, branch or PR was created by this run. +2. Onboard that repository with `bgagent repo onboard OWNER/REPO --compute-type + lambda-microvm --region us-west-2 --stack-name backgroundagent-dev`. Confirm + its effective backend with `bgagent runtime status --repo OWNER/REPO`. + Use this deployment's isolated CLI configuration and an authorized login. +3. Submit the normal clone/change/test/PR task from the P2 runbook and capture + progress, heartbeat, model invocation, Memory writes and logs while the VM + is running. The image build and empty API list are not substitutes. +4. Verify success, failure and cancellation cleanup, including S3 payload + deletion and service-reported VM termination. Exercise the pending IAM and + in-guest network positive/negative cases. The DNS observation-mode limit + remains applicable. +5. Record those results before discharging P2 warnings or enabling automatic + P3 suspension. + +A separate rebuild follow-up remains: overwriting the fixed artifact S3 key does +not itself change the CloudFormation image properties. Code-only redeployments +need a content/version-based image update trigger. This first image creation is +unaffected; later automatic rebuilds have not been verified. diff --git a/docs/verification/645-p3-implementation-plan.md b/docs/verification/645-p3-implementation-plan.md index 797f335d1..125810d8c 100644 --- a/docs/verification/645-p3-implementation-plan.md +++ b/docs/verification/645-p3-implementation-plan.md @@ -6,6 +6,17 @@ Prepared 2026-09-13 from `main` `5e10038c7e28179b302ac4de78b709795aeba3ce`. Read Prerequisite work is tracked here on `fix/645-microvm-readiness`. “Completed” means implemented and checked locally; AWS deployment and live verification have separate completion gates below. +**Live infrastructure and image deployed (2026-09-14):** the +[clean P2 deployment record](./645-p2-clean-deployment-20260913.md) tracks the new +Oregon environment, four deployment fixes and actual verification results. +Bootstrap 1.7.0 and application source `29dcaa74` are deployed. CloudFormation +reached `UPDATE_COMPLETE`; managed image version `1.0` is active, its ready and +validate hooks passed, and authenticated API reads passed after the update. +The normal task test still needs a selected repository, GitHub token setup and +explicit per-repository MicroVM routing. P2 acceptance and all P3 live gates +remain open. The batch notes below record what was verified at their original +completion; their deployment status is superseded by this record. + - [x] Review current implementation, clean up verified stale comments, and prototype nesting. - [x] Fix server-test thread isolation (#841). - [x] Grant scoped coordinator payload deletion (#817). @@ -22,6 +33,7 @@ Prerequisite work is tracked here on `fix/645-microvm-readiness`. “Completed - [x] Restrict agent task updates to reporting fields; remove replacement/deletion and worker counter grants. - [ ] Verify metadata restrictions with real AWS sessions/transactions; retain status/tag trust limits. - [x] Replace unused logging-failure bookkeeping with structured stdout diagnostics (#810); document shared runtime networking and verify large registry payload delivery (#818). +- [x] Deploy a fresh bootstrap, application and managed image from current source; verify build hooks and API reads. - [ ] Implement production nesting if included, then verify a clean P2 deployment. - [x] Add mandatory pause/wake command methods across all three compute strategies, with explicit unsupported results and bounded MicroVM requests. - [x] Keep the original approval deadline through database writes and polling, including frozen/backward clocks; preserve decision races and cancellation. @@ -247,7 +259,13 @@ The writer inventory found no production callers of Python `write_submitted` or ## 3. Re-prove P2 on the final infrastructure -Use a supported Region and an isolated development repository/account deployment. Record the actual deployed bootstrap bundle (at least 1.6.0, or the newer bundle produced by nesting). Compare effective policies as well as the displayed version. Update bootstrap deliberately when required; a command that skips an already bootstrapped stack is not evidence of refresh. +Use a supported Region and an isolated development repository/account deployment. Record the actual deployed bootstrap bundle (at least 1.7.0 for this source, including exact-self CloudFormation PassRole, or a newer required bundle). Compare effective policies as well as the displayed version. Update bootstrap deliberately when required; a command that skips an already bootstrapped stack is not evidence of refresh. + +The [2026-09-13–14 clean deployment](./645-p2-clean-deployment-20260913.md) +completed the infrastructure, managed-image creation and build-hook checks in +steps 1–2 below. Live runtime credential behavior and steps 3–6 remain +unverified. The stack is configured to offer MicroVM, but the seeded repository +still uses AgentCore; explicitly select `lambda-microvm` for the test repository. Follow `cdk/scripts/package-microvm-artifact.sh` and the P1/P2 runbooks: From e04f85c7d01b00fd989c26ea212b2a22cc8ce787 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Mon, 14 Sep 2026 08:48:02 -0400 Subject: [PATCH 024/149] docs: record P2 live task and cancellation checks (#645) --- .../645-p2-clean-deployment-20260913.md | 20 +- .../verification/645-p2-live-task-20260914.md | 208 ++++++++++++++++++ .../645-p3-implementation-plan.md | 22 +- 3 files changed, 237 insertions(+), 13 deletions(-) create mode 100644 docs/verification/645-p2-live-task-20260914.md diff --git a/docs/verification/645-p2-clean-deployment-20260913.md b/docs/verification/645-p2-clean-deployment-20260913.md index cfb9de859..7e387b58f 100644 --- a/docs/verification/645-p2-clean-deployment-20260913.md +++ b/docs/verification/645-p2-clean-deployment-20260913.md @@ -5,9 +5,14 @@ Live verification of the P2 source fixes and deployment prerequisites from **Status: infrastructure and managed image deployed.** CloudFormation reached `UPDATE_COMPLETE`; image version `1.0` is `SUCCESSFUL` and `ACTIVE`. Build hooks -and authenticated API reads pass. The normal coding-task test remains pending -repository choice and GitHub token setup. P2 acceptance remains open, and the -smoke warnings have not been discharged. +and authenticated API reads pass. The later +[live task verification](./645-p2-live-task-20260914.md) also passed normal-task, +PR-iteration and cancellation checks. Remaining P2 acceptance conditions are +listed there; the broad smoke warnings have not been discharged. + +The detailed chronology below records the infrastructure handoff at +2026-09-14 04:48 UTC. Repository/PAT setup and task execution occurred afterward +and are documented in the linked follow-up. ## Target and isolation @@ -65,8 +70,9 @@ not run concurrently with tests/synthesis using the shared CDK temporary files. - [x] CloudFormation creates the managed image using the pinned base. - [x] Image version is active; `/ready` and `/validate` return HTTP 200. - [x] Coordinator image/network/role configuration and authenticated API reads are verified. -- [ ] A normal task completes with live progress and heartbeat evidence. -- [ ] Runtime logs, payload cleanup and VM termination are verified. +- [x] A normal task completes with live progress and heartbeat evidence (follow-up). +- [x] Runtime logs, successful/canceled-task payload cleanup and VM termination are verified (follow-up). +- [ ] Failure/recovery cleanup and automatic repository build/lint gates are verified. - [ ] Relevant allowed/denied IAM and network cases are exercised. - [x] Current resource state, temporary-login cleanup and remaining limits are recorded. @@ -315,7 +321,9 @@ and cached CLI credentials were removed. The non-secret isolated CLI configuration and deployment evidence remain. Evidence: `verification-user-cleanup.json`. -The next verification sequence is: +At infrastructure handoff, the planned verification sequence was the following. +The [live task record](./645-p2-live-task-20260914.md) records the completed +steps and the remaining limits: 1. Select the test repository and populate the new deployment's GitHub token using `bgagent github set-token --region us-west-2 --stack-name backgroundagent-dev`. diff --git a/docs/verification/645-p2-live-task-20260914.md b/docs/verification/645-p2-live-task-20260914.md new file mode 100644 index 000000000..e877c12a0 --- /dev/null +++ b/docs/verification/645-p2-live-task-20260914.md @@ -0,0 +1,208 @@ +# ADR-021 P2 live task verification — 2026-09-14 + +Follow-up to the [clean infrastructure and image deployment](./645-p2-clean-deployment-20260913.md). + +**Result:** a normal coding task and a PR-update task completed on managed +MicroVM image `1.0`. Both ran the repository's npm checks successfully, wrote +Memory events and terminated automatically. A separate cancellation test +removed its own VM, payload and capacity reservation while the other task +continued running. + +The reviewed result is [isadeks/vercel-abca-linear PR #584](https://github.com/isadeks/vercel-abca-linear/pull/584), +open and unmerged at commit `fd509fa63fa089df356574bdd654123c8621b12e`. +Only `README.md` changes. + +This verifies the observed success, PR-iteration and cancellation paths. +It does **not** discharge every P2 warning: automatic build/lint configuration +for this repository, failure/recovery cases and the wider IAM/network matrix +remain outstanding. Automatic P3 suspension remains disabled. + +## Environment and repository setup + +- AWS profile/account/Region: `sphia-dev` / `` / `us-west-2`. +- Stack: `backgroundagent-dev`; bootstrap bundle: `1.7.0`. +- Deployed source: `29dcaa74`; no runtime, image or IAM changes were needed for + these tests. The later local commits contain verification documentation. +- Image: `arn:aws:lambda:us-west-2::microvm-image:backgroundagent-dev-abca-agent`, + version `1.0`, `ACTIVE`. +- User-selected repository: `isadeks/vercel-abca-linear`. +- Local evidence: `/tmp/abca-645-p2-clean-20260913`. + +Oregon was chosen because the existing east deployment already occupied its +regional resources and `us-east-1` had five VPCs against a quota of five. This +run preserved that deployment and its resources. + +The existing effective PAT for the selected repository was checked without +printing its value. GitHub `/user` and `/repos/isadeks/vercel-abca-linear` both +returned HTTP 200 as `isadeks`, with repository push permission. GitHub reported +expiration **2026-09-27 12:52:45 UTC**. The verified value was copied into the +new west deployment's still-empty platform secret, using the built CLI helper. +Read-back equality passed; the east secret was not modified. + +The built CLI onboarded the selected repository with +`--compute-type lambda-microvm`. Runtime status confirmed that selection, and +platform doctor passed every check, including MicroVM service availability. +The previously seeded `awslabs/agent-plugins` repository retained its AgentCore +selection. + +Evidence: `selected-repository-pat-check.json`, `github-token-setup.json`, +`smoke-runtime-status.json` and `smoke-platform-doctor.json`. + +## Normal task and review correction + +| Run | Task ID | Workflow | Outcome | +|---|---|---|---| +| Initial README task | `01M2FYAHE8TDF8JZ9EB432F3SX` | `coding/new-task-v1@1.0.0` | `COMPLETED`, created PR #584 | +| README review correction | `01M2FYS5DVZH60J9TJWX3ZE9CA` | `coding/pr-iteration-v1@1.0.0` | `COMPLETED`, updated the same PR | +| Cancellation canary | `01M2FYV0SVT4YP3VKFDWTB3S7R` | `coding/new-task-v1@1.0.0` | `CANCELLED`, VM and payload cleaned up | + +The first task preserved the project title and documented `npm ci`, +`npm run lint` and `npm test`. It used a 40-turn / $5 limit. The actual VM +`microvm-86421c36-72e8-38cb-9cb3-b5f29222f497` reached `RUNNING` on image `1.0`. +The `/run` hook returned HTTP 200, scoped tenant credentials initialized, +repository cloning succeeded and heartbeats advanced throughout execution. + +Claude Opus 5 executed tools, changed only the README and pushed commit +`20c0717eec36b858b12a40a0b43cb13ea651feb1`. The task completed at +12:34:21 UTC with reported duration 202.3 seconds, 17 turns and model cost +$0.2655996. These duration/cost fields are the task's reported metrics, not +total AWS infrastructure cost. + +Review then caught two incorrect claims: `api/_lib/` contains only a design +README, so the booking API modules are planned rather than implemented; and +there is no repository CI workflow automatically running the npm checks. +The second task corrected those statements through the normal PR-iteration +workflow, with a 30-turn / $3 limit. + +Its VM was `microvm-d552daeb-dbd6-3300-8b2a-1ef7fae46429`. The task completed at +12:41:09 UTC with reported duration 143.2 seconds, nine turns and model cost +$0.2111514. The final diff correctly describes the current static site and +planned API. The PR description was shortened after review to state the final +change, actual validation and configuration limitation. Vercel checks passed +on the final head. + +For both runs, the explicit npm commands exited 0. The correction run's +CloudWatch tool result records: + +```text +npm ci exit=0 npm run lint exit=0 npm test exit=0 +``` + +Vitest ran one baseline placeholder test. That verifies the shipped test +command ran successfully; it does not establish application feature coverage. +`bgagent watch` observed the first task through completion and exited 0. + +Evidence: `smoke-task-{request,submit}.json`, +`smoke-review-task-{request,submit}.json`, `smoke-watch.ndjson`, +`smoke-observations.ndjson`, `smoke-review-validation-evidence.json`, +`smoke-pr-final-diff.patch` and the timestamped task/VM/log snapshots. + +## Automatic build/lint configuration remains incomplete + +The default blueprint checks are `mise run build` and `mise run lint`. +This repository has no mise tasks. The agent's explicit npm checks passed, +but the platform's automatic checks still used those defaults. + +Both task records therefore show `build_passed=true` and `lint_passed=false`. +The build result reflects the existing inert/default-build behavior; it must +not be presented as a successful project build. The lint result reflects the +failing default mise command, not a failing `npm run lint`. + +Configure suitable `pipeline.buildCommand` / `pipeline.lintCommand` values on +the repository's canonical Blueprint, including dependency installation where +needed, then verify those automatic gates on another normal task. The runtime +onboarding command used here does not configure those fields. No placeholder +mise task, no-op build or unrelated repository change was added to hide the +mismatch. + +## Logging and Memory evidence + +Two separate logging paths were observed while the initial VM was running: + +1. Service/runtime logs in `/aws/lambda-microvms/backgroundagent-dev-abca-agent`, + under the VM's versioned log stream. +2. Direct server writes to + `/aws/vendedlogs/bedrock-agentcore/runtime/APPLICATION_LOGS/backgroundagent-dev`, + stream `server_debug/01M2FYAHE8TDF8JZ9EB432F3SX`. + +The second path contained configuration/thread/run-acceptance records. It +exercises actual application `CreateLogStream` / `PutLogEvents` calls under +the deployed runtime credentials, beyond checking logs after shutdown. + +AgentCore `ListEvents` independently returned: + +- Initial task: a `task_episode` and `repo_learnings` event, timestamp + 12:34:21 UTC, with schema version 3 and content hashes. +- Correction task: a `task_episode` event, timestamp 12:41:09 UTC, referencing + the same PR. + +These records verify Memory writes. The correction's hydration event reported +`has_memory_context=false`; this run does not establish successful retrieval +of prior Memory context. + +Evidence: `smoke-application-logs-running.json`, `smoke-memory-events.json`, +`smoke-review-memory-events.json` and `smoke-review-events.json`. + +## Termination, payload deletion and cancellation + +For the first successful task, S3 initially contained its `payload.json`, +private `launch.json` and the shared bootstrap manifest. After automatic +finalization, both task files were gone. The service reported `TERMINATED`, +and the `/terminate` hook returned HTTP 200 at 12:34:35 UTC with zero active +pipeline threads. + +The correction run also terminated automatically. Its hook returned HTTP 200 +at 12:41:31 UTC with zero active pipeline threads. An empty `microvm_id` in +these terminate records matches the service behavior already documented in +`server.py`; the versioned log stream and service state provide the VM identity. +The hook acknowledgement is not a P3 suspend durability barrier. + +The cancellation canary requested no coding or GitHub publication work and +used a one-turn / $0.05 limit. At 12:39:32 UTC: + +- Its VM `microvm-1a9f0024-2a4b-3b80-a172-b6b93c35af77` was `RUNNING`. +- The correction VM was also `RUNNING`. +- A strongly consistent read showed the user's `active_count=2`. + +Normal CLI cancellation returned `CANCELLED` at 12:39:33.958 UTC. By the +12:40:24 UTC verification, its VM was `TERMINATED`, its S3 task prefix was +empty, and its reservation was `released`. The count was **1**, with the +correction VM still running. That observed cancellation did not release the +other task's seat. + +After the correction completed, `active_count=0`, all three VMs were +`TERMINATED`, and only the shared bootstrap manifest remained in S3. +That manifest intentionally expires through the one-day lifecycle rule; +it is not a retained task payload. + +Evidence: `smoke-payload-objects-{running,after}.json`, +`smoke-cancel-{before,response,after}.json`, `smoke-final-counter.json`, +`smoke-final-payload-inventory.json` and `smoke-observations.ndjson`. + +## Handoff and remaining gates + +The temporary Cognito user `p2-smoke-20260914@example.invalid` was deleted after +verification; `AdminGetUser` confirmed `UserNotFoundException`. Its cached CLI +tokens were removed. The generated password was never saved or sent in an +invitation. Evidence: `smoke-verification-user-cleanup.json`. + +The intended deployment, repository/PAT configuration, open PR, task audit +records and non-secret local evidence remain. No application code, IAM policy, +agent image or resource in the existing east deployment was modified. + +- [x] Normal task and PR iteration complete with live heartbeat/model/tool evidence. +- [x] Real npm checks pass and the final README/PR is reviewed. +- [x] Application logging works while running; actual Memory events are stored. +- [x] Successful and canceled tasks terminate and delete their task payloads. +- [x] Cancellation preserves the other task's seat; final active count is zero. +- [ ] Configure and re-prove automatic repository build/lint gates. +- [ ] Exercise failure cleanup, lost/replayed start responses and recovery cases. +- [ ] Exercise the full real-session IAM and in-guest/public-ingress network matrix. +- [ ] Verify Memory retrieval if claiming that behavior. +- [ ] Complete the separate P3 hooks, credential/durability barriers, supervisor + integration and live suspend/resume matrix. + +The clean deployment's DNS firewall remains in observation mode: unlisted +domains are logged rather than blocked. The code-only managed-image rebuild +trigger follow-up also remains open. None of these remaining conditions is +discharged by the successful tasks above. diff --git a/docs/verification/645-p3-implementation-plan.md b/docs/verification/645-p3-implementation-plan.md index 125810d8c..39da0abcb 100644 --- a/docs/verification/645-p3-implementation-plan.md +++ b/docs/verification/645-p3-implementation-plan.md @@ -12,10 +12,14 @@ Oregon environment, four deployment fixes and actual verification results. Bootstrap 1.7.0 and application source `29dcaa74` are deployed. CloudFormation reached `UPDATE_COMPLETE`; managed image version `1.0` is active, its ready and validate hooks passed, and authenticated API reads passed after the update. -The normal task test still needs a selected repository, GitHub token setup and -explicit per-repository MicroVM routing. P2 acceptance and all P3 live gates -remain open. The batch notes below record what was verified at their original -completion; their deployment status is superseded by this record. +The [live task verification](./645-p2-live-task-20260914.md) subsequently passed +normal coding, PR iteration and cancellation on `isadeks/vercel-abca-linear`, +including heartbeat, npm checks, Memory writes and automatic cleanup. The +repository's automatic mise build/lint defaults still need proper npm command +configuration; failure/recovery and the wider IAM/network matrix also remain. +Full P2 acceptance and all P3 live gates remain open. The batch notes below +record what was verified at their original completion; their deployment status +is superseded by these records. - [x] Review current implementation, clean up verified stale comments, and prototype nesting. - [x] Fix server-test thread isolation (#841). @@ -34,6 +38,7 @@ completion; their deployment status is superseded by this record. - [ ] Verify metadata restrictions with real AWS sessions/transactions; retain status/tag trust limits. - [x] Replace unused logging-failure bookkeeping with structured stdout diagnostics (#810); document shared runtime networking and verify large registry payload delivery (#818). - [x] Deploy a fresh bootstrap, application and managed image from current source; verify build hooks and API reads. +- [x] Verify normal coding, PR iteration, Memory writes, live logging and successful/canceled-task cleanup in AWS; observe cancellation preserve another task's capacity. - [ ] Implement production nesting if included, then verify a clean P2 deployment. - [x] Add mandatory pause/wake command methods across all three compute strategies, with explicit unsupported results and bounded MicroVM requests. - [x] Keep the original approval deadline through database writes and polling, including frozen/backward clocks; preserve decision races and cancellation. @@ -263,9 +268,12 @@ Use a supported Region and an isolated development repository/account deployment The [2026-09-13–14 clean deployment](./645-p2-clean-deployment-20260913.md) completed the infrastructure, managed-image creation and build-hook checks in -steps 1–2 below. Live runtime credential behavior and steps 3–6 remain -unverified. The stack is configured to offer MicroVM, but the seeded repository -still uses AgentCore; explicitly select `lambda-microvm` for the test repository. +steps 1–2 below. The [live task follow-up](./645-p2-live-task-20260914.md) +provides positive runtime evidence for steps 3–5 and success/cancellation cleanup +in step 6. Automatic repository build/lint command configuration, failure +cleanup, recovery and negative IAM/network cases remain unverified. +`isadeks/vercel-abca-linear` explicitly selects `lambda-microvm`; the original +seeded repository retains AgentCore. Follow `cdk/scripts/package-microvm-artifact.sh` and the P1/P2 runbooks: From e8c2961b81d5625ae88ab89f5180d2e3c3dc6c18 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Mon, 14 Sep 2026 09:16:36 -0400 Subject: [PATCH 025/149] fix(cli): configure and preserve repository verification commands --- cli/README.md | 20 ++++++- cli/src/commands/repo.ts | 4 ++ cli/src/repo-display.ts | 18 +++++- cli/src/repo-lookup.ts | 2 + cli/src/repo-onboard.ts | 7 +++ cli/test/commands/repo-display.test.ts | 30 ++++++++++ .../commands/repo-onboard-command.test.ts | 21 ++++++- cli/test/commands/repo-onboard.test.ts | 57 +++++++++++++++++++ 8 files changed, 156 insertions(+), 3 deletions(-) diff --git a/cli/README.md b/cli/README.md index 9c331530d..96e613083 100644 --- a/cli/README.md +++ b/cli/README.md @@ -258,15 +258,33 @@ Register or re-activate a repository in `RepoTable` without a CDK redeploy. With ``` bgagent repo onboard owner/repo \ - --compute-type \ + --compute-type \ --runtime-arn AgentCore runtime override (agentcore only) \ --model \ --token-secret-arn \ --max-turns \ --poll-interval Default agent poll interval in milliseconds \ + --build-command Build verification override \ + --lint-command Lint verification override \ --output ``` +Verification defaults to `mise run build` and `mise run lint`, using tasks in the +target repository's `mise.toml`. For an npm repository without mise tasks, set +the commands explicitly: + +```bash +bgagent repo onboard owner/repo \ + --build-command 'npm ci && npm test' \ + --lint-command 'npm run lint' +``` + +Re-onboarding preserves existing commands when these flags are omitted. Pass an +empty string to restore a command's mise default. `bgagent repo show owner/repo` +shows the effective commands. This operator command updates RepoTable; for a +repository managed by CDK, also set `Blueprint.pipeline.buildCommand` and +`lintCommand` in its source so future Blueprint updates retain the configuration. + ### `bgagent repo offboard ` Soft-delete a repository (`status=removed` + TTL), matching Blueprint delete semantics. An existing Blueprint will re-activate the repo on the next CDK deploy. diff --git a/cli/src/commands/repo.ts b/cli/src/commands/repo.ts index 626c4cbdb..b8a291759 100644 --- a/cli/src/commands/repo.ts +++ b/cli/src/commands/repo.ts @@ -159,6 +159,8 @@ export function makeRepoCommand(): Command { .option('--token-secret-arn ', 'Per-repo GitHub token Secrets Manager ARN') .option('--max-turns ', 'Default max turns for tasks', parseInt) .option('--poll-interval ', 'Default agent poll interval in milliseconds', parseInt) + .option('--build-command ', 'Build verification command (empty restores mise run build)') + .option('--lint-command ', 'Lint verification command (empty restores mise run lint)') .option('--output ', 'Output format: text or json', 'text') .action(async (repoId: string, opts) => { assertRepoFormat(repoId); @@ -210,6 +212,8 @@ export function makeRepoCommand(): Command { githubTokenSecretArn: opts.tokenSecretArn, maxTurns: opts.maxTurns, pollIntervalMs: opts.pollInterval, + buildCommand: opts.buildCommand, + lintCommand: opts.lintCommand, }); const notes = buildRepoOnboardNotes({ config, diff --git a/cli/src/repo-display.ts b/cli/src/repo-display.ts index fa7b04c9d..1f9880035 100644 --- a/cli/src/repo-display.ts +++ b/cli/src/repo-display.ts @@ -48,6 +48,8 @@ export const PLATFORM_REPO_DEFAULTS = { model_id: 'us.anthropic.claude-opus-5', max_turns: 200, poll_interval_ms: 30_000, + build_command: 'mise run build', + lint_command: 'mise run lint', approval_gate_cap: 50, } as const; @@ -70,6 +72,8 @@ export interface RepoConfigDisplay { readonly max_turns: number; readonly max_budget_usd: string; readonly poll_interval_ms: number; + readonly build_command: string; + readonly lint_command: string; readonly approval_gate_cap: number; readonly github_token_source: GithubTokenSource; readonly github_token_secret_arn?: string; @@ -108,6 +112,8 @@ export function formatRepoConfigForDisplay( mark('max_turns', config.max_turns !== undefined); mark('max_budget_usd', config.max_budget_usd !== undefined); mark('poll_interval_ms', config.poll_interval_ms !== undefined); + mark('build_command', Boolean(config.build_command?.trim())); + mark('lint_command', Boolean(config.lint_command?.trim())); mark('approval_gate_cap', config.approval_gate_cap !== undefined); fieldSources.github_token_secret_arn = usesBlueprintToken ? 'blueprint' : 'platform'; @@ -115,7 +121,7 @@ export function formatRepoConfigForDisplay( for (const key of [ 'compute_type', 'runtime_arn', 'model_id', 'max_turns', 'max_budget_usd', 'poll_interval_ms', 'approval_gate_cap', 'system_prompt_overrides', - 'egress_allowlist', 'cedar_policies', 'github_token_secret_arn', + 'egress_allowlist', 'cedar_policies', 'github_token_secret_arn', 'build_command', 'lint_command', ] as const) { const value = config[key]; if (value !== undefined && value !== null && !(Array.isArray(value) && value.length === 0)) { @@ -143,6 +149,8 @@ export function formatRepoConfigForDisplay( ? String(config.max_budget_usd) : 'unlimited', poll_interval_ms: config.poll_interval_ms ?? PLATFORM_REPO_DEFAULTS.poll_interval_ms, + build_command: config.build_command?.trim() || PLATFORM_REPO_DEFAULTS.build_command, + lint_command: config.lint_command?.trim() || PLATFORM_REPO_DEFAULTS.lint_command, approval_gate_cap: config.approval_gate_cap ?? PLATFORM_REPO_DEFAULTS.approval_gate_cap, github_token_source: usesBlueprintToken ? 'blueprint' : 'platform', github_token_secret_arn: effectiveTokenArn @@ -197,6 +205,14 @@ export function buildRepoShowLines(display: RepoConfigDisplay): RepoShowLine[] { display.field_sources.poll_interval_ms, ), }, + { + key: 'build_command', + text: formatSourcedValue(display.effective.build_command, display.field_sources.build_command), + }, + { + key: 'lint_command', + text: formatSourcedValue(display.effective.lint_command, display.field_sources.lint_command), + }, { key: 'approval_gate_cap', text: formatSourcedValue( diff --git a/cli/src/repo-lookup.ts b/cli/src/repo-lookup.ts index 7301d5fdc..17513fe87 100644 --- a/cli/src/repo-lookup.ts +++ b/cli/src/repo-lookup.ts @@ -37,6 +37,8 @@ export interface RepoConfigRow { readonly model_id?: string; readonly max_turns?: number; readonly max_budget_usd?: number; + readonly build_command?: string; + readonly lint_command?: string; readonly system_prompt_overrides?: string; readonly github_token_secret_arn?: string; readonly poll_interval_ms?: number; diff --git a/cli/src/repo-onboard.ts b/cli/src/repo-onboard.ts index 4e30009fd..976880f8d 100644 --- a/cli/src/repo-onboard.ts +++ b/cli/src/repo-onboard.ts @@ -41,6 +41,8 @@ export interface OnboardRepoOptions { readonly maxTurns?: number; readonly githubTokenSecretArn?: string; readonly pollIntervalMs?: number; + readonly buildCommand?: string; + readonly lintCommand?: string; } export interface OnboardRepoDependencies { @@ -114,6 +116,11 @@ export async function onboardRepo( item.approval_gate_cap = existing.approval_gate_cap; } if (existing?.max_budget_usd !== undefined) item.max_budget_usd = existing.max_budget_usd; + // Retain verification recipes when the operator leaves their flags unset. + if (options.buildCommand !== undefined) item.build_command = options.buildCommand; + else if (existing?.build_command !== undefined) item.build_command = existing.build_command; + if (options.lintCommand !== undefined) item.lint_command = options.lintCommand; + else if (existing?.lint_command !== undefined) item.lint_command = existing.lint_command; await ddb.send(new PutCommand({ TableName: tableName, diff --git a/cli/test/commands/repo-display.test.ts b/cli/test/commands/repo-display.test.ts index 6d45cf7e0..e868101d6 100644 --- a/cli/test/commands/repo-display.test.ts +++ b/cli/test/commands/repo-display.test.ts @@ -82,6 +82,36 @@ describe('formatRepoConfigForDisplay', () => { }); describe('buildRepoShowLines', () => { + test.each([undefined, '', ' '])('shows actual worker defaults for unset verification commands: %p', (command) => { + const display = formatRepoConfigForDisplay({ + repo: 'acme/foo', status: 'active', build_command: command, lint_command: command, + }, PLATFORM); + const lines = buildRepoShowLines(display); + const postHooks = fs.readFileSync( + path.resolve(__dirname, '../../../agent/src/post_hooks.py'), 'utf8', + ); + for (const kind of ['build', 'lint'] as const) { + const field = `${kind}_command` as const; + const match = postHooks.match(new RegExp(`DEFAULT_${kind.toUpperCase()}_COMMAND = "([^"]+)"`)); + expect(match).not.toBeNull(); + expect(display.effective[field]).toBe(match![1]); + expect(lines.find((line) => line.key === field)?.text).toBe(`(platform default) ${match![1]}`); + } + }); + + test('shows verification overrides in text and JSON', () => { + const commands = { build_command: 'npm ci && npm test', lint_command: 'npm run lint' }; + const display = formatRepoConfigForDisplay({ + repo: 'acme/foo', status: 'active', ...commands, + }, PLATFORM); + expect(display.blueprint_overrides).toMatchObject(commands); + expect(display.effective).toMatchObject(commands); + const lines = buildRepoShowLines(display); + for (const [key, value] of Object.entries(commands)) { + expect(lines.find((line) => line.key === key)?.text).toBe(`${value} (per-blueprint override)`); + } + }); + test('shows platform defaults instead of dash for unset blueprint fields', () => { const display = formatRepoConfigForDisplay( { repo: 'awslabs/agent-plugins', status: 'active' }, diff --git a/cli/test/commands/repo-onboard-command.test.ts b/cli/test/commands/repo-onboard-command.test.ts index 56524c510..e78e8fae2 100644 --- a/cli/test/commands/repo-onboard-command.test.ts +++ b/cli/test/commands/repo-onboard-command.test.ts @@ -22,7 +22,10 @@ import { onboardRepo, offboardRepo } from '../../src/repo-onboard'; import { getStackOutput } from '../../src/stack-outputs'; jest.mock('../../src/repo-onboard'); -jest.mock('../../src/stack-outputs'); +jest.mock('../../src/stack-outputs', () => ({ + ...jest.requireActual('../../src/stack-outputs'), + getStackOutput: jest.fn(), +})); describe('repo onboard/offboard commands', () => { let consoleSpy: jest.SpiedFunction; @@ -127,4 +130,20 @@ describe('repo onboard/offboard commands', () => { expect(offboardRepo).toHaveBeenCalled(); expect(consoleSpy.mock.calls[0][0]).toContain('offboarded'); }); + + test.each([ + ['npm ci && npm test', 'npm run lint'], + ['', ''], + ])('repo onboard forwards build/lint commands without interpreting their contents', async (build, lint) => { + const cmd = makeRepoCommand(); + await cmd.parseAsync([ + 'node', 'test', 'onboard', 'acme/a', '--region', 'us-east-1', + '--build-command', build, '--lint-command', lint, + ]); + + expect(onboardRepo).toHaveBeenLastCalledWith( + 'us-east-1', 'RepoTable', 'acme/a', + expect.objectContaining({ buildCommand: build, lintCommand: lint }), + ); + }); }); diff --git a/cli/test/commands/repo-onboard.test.ts b/cli/test/commands/repo-onboard.test.ts index d6114e3a7..a2a1bd3ca 100644 --- a/cli/test/commands/repo-onboard.test.ts +++ b/cli/test/commands/repo-onboard.test.ts @@ -178,6 +178,37 @@ describe('repo onboard/offboard', () => { expect(put.input.Item?.poll_interval_ms).toBe(12345); }); + test.each(['active', 'removed'])( + 'onboardRepo keeps Blueprint verification commands when status is %s', + async (status) => { + const { loadRepoConfig } = jest.requireMock('../../src/repo-lookup') as { + loadRepoConfig: jest.Mock; + }; + const commands = { + build_command: 'npm ci && npm test', + lint_command: 'npm run lint', + }; + loadRepoConfig.mockResolvedValueOnce({ + repo: 'acme/a', + status, + compute_type: 'agentcore', + ...commands, + }); + + const result = await onboardRepo('us-east-1', 'RepoTable', 'acme/a', { + computeType: 'ecs', + }); + + const put = ddbSend.mock.calls[0][0] as PutCommand; + expect(put.input.Item).toMatchObject({ + status: 'active', + compute_type: 'ecs', + ...commands, + }); + expect(result).toMatchObject(commands); + }, + ); + test('onboardRepo treats a missing row as a fresh onboard', async () => { const { loadRepoConfig } = jest.requireMock('../../src/repo-lookup') as { loadRepoConfig: jest.Mock; @@ -186,6 +217,32 @@ describe('repo onboard/offboard', () => { await expect(onboardRepo('us-east-1', 'RepoTable', 'acme/a')).resolves.toBeDefined(); expect(ddbSend).toHaveBeenCalledTimes(1); + const put = ddbSend.mock.calls[0][0] as PutCommand; + expect(put.input.Item).not.toHaveProperty('build_command'); + expect(put.input.Item).not.toHaveProperty('lint_command'); + }); + + test.each([ + { buildCommand: 'npm ci && npm test', lintCommand: 'npm run lint' }, + { buildCommand: '', lintCommand: '' }, + ])('onboardRepo replaces verification commands, including resetting to defaults: %p', async (options) => { + const { loadRepoConfig } = jest.requireMock('../../src/repo-lookup') as { + loadRepoConfig: jest.Mock; + }; + loadRepoConfig.mockResolvedValueOnce({ + repo: 'acme/a', + status: 'active', + build_command: 'old-build', + lint_command: 'old-lint', + }); + + await onboardRepo('us-east-1', 'RepoTable', 'acme/a', options); + + const put = ddbSend.mock.calls[0][0] as PutCommand; + expect(put.input.Item).toMatchObject({ + build_command: options.buildCommand, + lint_command: options.lintCommand, + }); }); test('onboardRepo re-throws non-not-found load errors instead of wiping overrides', async () => { From 70f1bf5ebcf6465b936afb7075f928dacab5450b Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Mon, 14 Sep 2026 09:18:34 -0400 Subject: [PATCH 026/149] docs: record live repository checks and remaining MicroVM gates --- ...ADR-021-lambda-microvms-compute-backend.md | 4 +- ...Adr-021-lambda-microvms-compute-backend.md | 4 +- .../verification/645-p2-live-task-20260914.md | 17 ++- .../645-p2-repository-config-20260914.md | 131 ++++++++++++++++++ .../645-p3-implementation-plan.md | 15 +- docs/verification/645-p3-readiness-review.md | 2 + 6 files changed, 159 insertions(+), 14 deletions(-) create mode 100644 docs/verification/645-p2-repository-config-20260914.md diff --git a/docs/decisions/ADR-021-lambda-microvms-compute-backend.md b/docs/decisions/ADR-021-lambda-microvms-compute-backend.md index 475c74c24..16124ef33 100644 --- a/docs/decisions/ADR-021-lambda-microvms-compute-backend.md +++ b/docs/decisions/ADR-021-lambda-microvms-compute-backend.md @@ -1,6 +1,6 @@ # ADR-021: AWS Lambda MicroVMs as a third ComputeStrategy backend -> **Implementation status (2026-09-13):** P1 ([#689](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/689)) and P2 ([#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733)) are merged. P2 completed real coding tasks with an IAM workaround; a clean rerun after re-bootstrap remains outstanding. Local P3 foundations include suspend/resume commands, explicit VM observations, guarded persistent intent, a policy helper and approval deadlines that survive a frozen clock. Agent lifecycle hooks and supervisor/approval-handler integration remain unimplemented; automatic suspension is not enabled. This ADR defines P1–P3, not a P4. See the [current review](../verification/645-p3-readiness-review.md) and [P3 implementation plan](../verification/645-p3-implementation-plan.md). +> **Implementation status (2026-09-14):** P1 ([#689](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/689)) and P2 ([#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733)) are merged. The takeover branch now has a [clean deployment](../verification/645-p2-clean-deployment-20260913.md) with bootstrap 1.7.0 and managed image 1.0, plus [live coding, PR iteration and cancellation evidence](../verification/645-p2-live-task-20260914.md), without manual IAM workarounds. Remaining P2 failure/recovery and permission/network checks prevent full acceptance. Local P3 foundations include suspend/resume commands, explicit VM observations, guarded persistent intent, a policy helper and approval deadlines that survive a frozen clock. Agent lifecycle hooks and supervisor/approval-handler integration remain unimplemented; automatic suspension is not enabled. This ADR defines P1–P3, not a P4. See the [current review](../verification/645-p3-readiness-review.md) and [P3 implementation plan](../verification/645-p3-implementation-plan.md). **Status:** proposed **Date:** 2026-07-29 @@ -414,7 +414,7 @@ Lambda MicroVMs launched in 5 regions (us-east-1/2, us-west-2, eu-west-1, ap-nor - `/run` pre-install silence: with a **baked `LOG_GROUP_NAME`** (the hostile case — without it the assertions pass vacuously) every AWS/credential seam (`boto3.client`/`Session`, the `aws_session` factories, `_debug_cw`/`_warn_cw`) is armed to raise until the install succeeds. Asserted on the accepted path, on all three rejection paths (bad envelope, `platform_config` invalid, `platform_config` incomplete) and on the failed-fetch 500 — where the seams stay armed for the whole request, because a rejected run installed nothing and so earns no AWS call. The permitted exception is asserted POSITIVELY: exactly one client is built pre-install, for `s3`, through the attributed factory. - `/ready` warm-up tests (P2-F5): the hook exec's each configured binary exactly once with a generous timeout; `claude` is the only REQUIRED entry; a timeout, a missing binary, a non-zero exit and an unexpected `OSError` each produce **503 with the reason logged to stdout** rather than a 200 or a 500; a best-effort failure still reports ready; the warm-up makes zero AWS calls with `LOG_GROUP_NAME` baked. Plus the backstop half: the `claude --version` probe's bound is asserted to be ≥ 60 s and to be applied to the *exec* rather than to the PATH lookup, and a missing CLI warns instead of raising. - CDK assertions (P2-F1/F2/F4): no source-condition key on any of the three MicroVM-facing role trusts, and no `aws:SourceAccount`/`aws:SourceArn` string anywhere in them; hook properties are `ENABLED` and the architecture is `ARM_64`, with a negative assertion that **no** hook route string appears anywhere in the rendered image resource; the agent hook routes are asserted against their own dedicated constant (the template no longer carries a path to compare); the execution role holds `logs:CreateLogStream`/`PutLogEvents` on the application log group and the two logs grants stay separate; the stack wires the SAME log group it delivers as `platform_config.log_group_name`. -- Smoke (gated like the ECS backend): clone → change → PR with `bgagent watch` progress; Memory write parity (no AccessDenied no-op). **Run 1 (2026-08-06) FAILED at `implement`, turn 0 — no PR. Run 2 (2026-08-07) PASSED: two tasks clone → change → commit → push → PR, `COMPLETED`, 12 turns / $0.279 / 153 s** (`docs/verification/645-p2-smoke-runbook.md`), which also discharged P2-F1, P2-F2, P2-F4, P2-F5 and the dual-signal-liveness item (45 s heartbeat cadence observed across a 181 s `RUNNING` window). **The row is not yet fully closed:** run 2 needed one live IAM workaround, and establishing why produced P2r2-F10 (the identity-side `iam:PassedToService`) and P2r2-F9 (its CloudFormation twin). Both are fixed in source above and neither has been re-exercised live, so what remains is a re-run on a re-bootstrapped account with no workarounds. +- Smoke (gated like the ECS backend): clone → change → PR with `bgagent watch` progress; Memory write parity (no AccessDenied no-op). **Run 1 (2026-08-06) FAILED at `implement`, turn 0 — no PR. Run 2 (2026-08-07) PASSED: two tasks clone → change → commit → push → PR, `COMPLETED`, 12 turns / $0.279 / 153 s** (`docs/verification/645-p2-smoke-runbook.md`), which also discharged P2-F1, P2-F2, P2-F4, P2-F5 and the dual-signal-liveness item (45 s heartbeat cadence observed across a 181 s `RUNNING` window). Run 2 needed one live IAM workaround, producing P2r2-F10 (the identity-side `iam:PassedToService`) and P2r2-F9 (its CloudFormation twin). **The 2026-09-14 takeover-branch rerun exercised both corrected paths:** CloudFormation created the managed image and the orchestrator launched real coding tasks using source-defined permissions after a fresh bootstrap, with no manual IAM workaround. See the [deployment](../verification/645-p2-clean-deployment-20260913.md) and [task evidence](../verification/645-p2-live-task-20260914.md). This closes that historical clean-rerun requirement; the broader failure/recovery and permission/network acceptance matrix remains open. **P3 (suspend/resume):** diff --git a/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md b/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md index 31eb5f718..b28960894 100644 --- a/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md +++ b/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md @@ -4,7 +4,7 @@ title: Adr 021 lambda microvms compute backend # ADR-021: AWS Lambda MicroVMs as a third ComputeStrategy backend -> **Implementation status (2026-09-13):** P1 ([#689](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/689)) and P2 ([#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733)) are merged. P2 completed real coding tasks with an IAM workaround; a clean rerun after re-bootstrap remains outstanding. Local P3 foundations include suspend/resume commands, explicit VM observations, guarded persistent intent, a policy helper and approval deadlines that survive a frozen clock. Agent lifecycle hooks and supervisor/approval-handler integration remain unimplemented; automatic suspension is not enabled. This ADR defines P1–P3, not a P4. See the [current review](/sample-autonomous-cloud-coding-agents/architecture/645-p3-readiness-review) and [P3 implementation plan](/sample-autonomous-cloud-coding-agents/architecture/645-p3-implementation-plan). +> **Implementation status (2026-09-14):** P1 ([#689](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/689)) and P2 ([#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733)) are merged. The takeover branch now has a [clean deployment](/sample-autonomous-cloud-coding-agents/architecture/645-p2-clean-deployment-20260913) with bootstrap 1.7.0 and managed image 1.0, plus [live coding, PR iteration and cancellation evidence](/sample-autonomous-cloud-coding-agents/architecture/645-p2-live-task-20260914), without manual IAM workarounds. Remaining P2 failure/recovery and permission/network checks prevent full acceptance. Local P3 foundations include suspend/resume commands, explicit VM observations, guarded persistent intent, a policy helper and approval deadlines that survive a frozen clock. Agent lifecycle hooks and supervisor/approval-handler integration remain unimplemented; automatic suspension is not enabled. This ADR defines P1–P3, not a P4. See the [current review](/sample-autonomous-cloud-coding-agents/architecture/645-p3-readiness-review) and [P3 implementation plan](/sample-autonomous-cloud-coding-agents/architecture/645-p3-implementation-plan). **Status:** proposed **Date:** 2026-07-29 @@ -418,7 +418,7 @@ Lambda MicroVMs launched in 5 regions (us-east-1/2, us-west-2, eu-west-1, ap-nor - `/run` pre-install silence: with a **baked `LOG_GROUP_NAME`** (the hostile case — without it the assertions pass vacuously) every AWS/credential seam (`boto3.client`/`Session`, the `aws_session` factories, `_debug_cw`/`_warn_cw`) is armed to raise until the install succeeds. Asserted on the accepted path, on all three rejection paths (bad envelope, `platform_config` invalid, `platform_config` incomplete) and on the failed-fetch 500 — where the seams stay armed for the whole request, because a rejected run installed nothing and so earns no AWS call. The permitted exception is asserted POSITIVELY: exactly one client is built pre-install, for `s3`, through the attributed factory. - `/ready` warm-up tests (P2-F5): the hook exec's each configured binary exactly once with a generous timeout; `claude` is the only REQUIRED entry; a timeout, a missing binary, a non-zero exit and an unexpected `OSError` each produce **503 with the reason logged to stdout** rather than a 200 or a 500; a best-effort failure still reports ready; the warm-up makes zero AWS calls with `LOG_GROUP_NAME` baked. Plus the backstop half: the `claude --version` probe's bound is asserted to be ≥ 60 s and to be applied to the *exec* rather than to the PATH lookup, and a missing CLI warns instead of raising. - CDK assertions (P2-F1/F2/F4): no source-condition key on any of the three MicroVM-facing role trusts, and no `aws:SourceAccount`/`aws:SourceArn` string anywhere in them; hook properties are `ENABLED` and the architecture is `ARM_64`, with a negative assertion that **no** hook route string appears anywhere in the rendered image resource; the agent hook routes are asserted against their own dedicated constant (the template no longer carries a path to compare); the execution role holds `logs:CreateLogStream`/`PutLogEvents` on the application log group and the two logs grants stay separate; the stack wires the SAME log group it delivers as `platform_config.log_group_name`. -- Smoke (gated like the ECS backend): clone → change → PR with `bgagent watch` progress; Memory write parity (no AccessDenied no-op). **Run 1 (2026-08-06) FAILED at `implement`, turn 0 — no PR. Run 2 (2026-08-07) PASSED: two tasks clone → change → commit → push → PR, `COMPLETED`, 12 turns / $0.279 / 153 s** (`docs/verification/645-p2-smoke-runbook.md`), which also discharged P2-F1, P2-F2, P2-F4, P2-F5 and the dual-signal-liveness item (45 s heartbeat cadence observed across a 181 s `RUNNING` window). **The row is not yet fully closed:** run 2 needed one live IAM workaround, and establishing why produced P2r2-F10 (the identity-side `iam:PassedToService`) and P2r2-F9 (its CloudFormation twin). Both are fixed in source above and neither has been re-exercised live, so what remains is a re-run on a re-bootstrapped account with no workarounds. +- Smoke (gated like the ECS backend): clone → change → PR with `bgagent watch` progress; Memory write parity (no AccessDenied no-op). **Run 1 (2026-08-06) FAILED at `implement`, turn 0 — no PR. Run 2 (2026-08-07) PASSED: two tasks clone → change → commit → push → PR, `COMPLETED`, 12 turns / $0.279 / 153 s** (`docs/verification/645-p2-smoke-runbook.md`), which also discharged P2-F1, P2-F2, P2-F4, P2-F5 and the dual-signal-liveness item (45 s heartbeat cadence observed across a 181 s `RUNNING` window). Run 2 needed one live IAM workaround, producing P2r2-F10 (the identity-side `iam:PassedToService`) and P2r2-F9 (its CloudFormation twin). **The 2026-09-14 takeover-branch rerun exercised both corrected paths:** CloudFormation created the managed image and the orchestrator launched real coding tasks using source-defined permissions after a fresh bootstrap, with no manual IAM workaround. See the [deployment](/sample-autonomous-cloud-coding-agents/architecture/645-p2-clean-deployment-20260913) and [task evidence](/sample-autonomous-cloud-coding-agents/architecture/645-p2-live-task-20260914). This closes that historical clean-rerun requirement; the broader failure/recovery and permission/network acceptance matrix remains open. **P3 (suspend/resume):** diff --git a/docs/verification/645-p2-live-task-20260914.md b/docs/verification/645-p2-live-task-20260914.md index e877c12a0..983c3f64e 100644 --- a/docs/verification/645-p2-live-task-20260914.md +++ b/docs/verification/645-p2-live-task-20260914.md @@ -13,9 +13,11 @@ open and unmerged at commit `fd509fa63fa089df356574bdd654123c8621b12e`. Only `README.md` changes. This verifies the observed success, PR-iteration and cancellation paths. -It does **not** discharge every P2 warning: automatic build/lint configuration -for this repository, failure/recovery cases and the wider IAM/network matrix -remain outstanding. Automatic P3 suspension remains disabled. +It does **not** discharge every P2 warning. The later +[configuration follow-up](./645-p2-repository-config-20260914.md) verifies real +automatic build/lint commands and a worker-reported failure cleanup; additional +failure/recovery cases and the wider IAM/network matrix remain outstanding. +Automatic P3 suspension remains disabled. ## Environment and repository setup @@ -97,7 +99,12 @@ Evidence: `smoke-task-{request,submit}.json`, `smoke-observations.ndjson`, `smoke-review-validation-evidence.json`, `smoke-pr-final-diff.patch` and the timestamped task/VM/log snapshots. -## Automatic build/lint configuration remains incomplete +## Automatic build/lint configuration in the original runs + +**Follow-up:** the [repository configuration record](./645-p2-repository-config-20260914.md) +supersedes this gap. Per-repository npm commands are now configured and all four +automatic pre/post checks passed live. The findings below describe the original +two runs before that configuration was applied. The default blueprint checks are `mise run build` and `mise run lint`. This repository has no mise tasks. The agent's explicit npm checks passed, @@ -195,7 +202,7 @@ agent image or resource in the existing east deployment was modified. - [x] Application logging works while running; actual Memory events are stored. - [x] Successful and canceled tasks terminate and delete their task payloads. - [x] Cancellation preserves the other task's seat; final active count is zero. -- [ ] Configure and re-prove automatic repository build/lint gates. +- [x] Configure and re-prove automatic repository build/lint gates — see the [follow-up](./645-p2-repository-config-20260914.md). - [ ] Exercise failure cleanup, lost/replayed start responses and recovery cases. - [ ] Exercise the full real-session IAM and in-guest/public-ingress network matrix. - [ ] Verify Memory retrieval if claiming that behavior. diff --git a/docs/verification/645-p2-repository-config-20260914.md b/docs/verification/645-p2-repository-config-20260914.md new file mode 100644 index 000000000..dc7fae330 --- /dev/null +++ b/docs/verification/645-p2-repository-config-20260914.md @@ -0,0 +1,131 @@ +# ADR-021 repository verification configuration and phase audit + +Date: 2026-09-14. Follows the [clean deployment](./645-p2-clean-deployment-20260913.md) +and [first live tasks](./645-p2-live-task-20260914.md). The user selected +`isadeks/vercel-abca-linear` and authorized completing its configuration. + +## Configuration + +The selected repository has npm lint/test scripts but no mise tasks. All three +compute backends use the same worker defaults (`mise run build` and +`mise run lint`), and all accept per-repository command overrides. + +The operator CLI now exposes those existing settings: + +```bash +bgagent repo onboard isadeks/vercel-abca-linear \ + --region us-west-2 --stack-name backgroundagent-dev \ + --compute-type lambda-microvm \ + --build-command 'npm ci && npm test' \ + --lint-command 'npm run lint' +``` + +This command was executed against account ``. A separate +`repo show` confirmed both effective commands and `lambda-microvm` routing. +The east deployment was not changed. No infrastructure or image update is +required: the deployed coordinator already forwards these fields to the worker. +The build command installs the pinned dependencies before running tests, so the +following lint command has its tools available. + +The CLI also preserves existing commands on re-onboarding and displays their +effective values. Previously, its replacement row silently omitted both fields. +Two regressions failed before the preservation fix. Tests now cover active and +removed repositories, backend changes, explicit overrides, empty-string resets, +fresh defaults, command parsing and text/JSON display. CLI compile/lint and all +**62 suites / 938 tests** pass. The implementation is commit `62cce37d`. + +For a repository managed through CDK, put the same values in +`Blueprint.pipeline.buildCommand` / `lintCommand`; CLI changes alone do not edit +that source. This test repository was onboarded through the operator CLI. + +## Local mise alternative + +A local checkout of PR #584 contains commit `7bf72d3`, adding `mise.toml` and +updating README. The tasks share an `install` dependency (`npm ci`); `build` +runs `npm test`, and `lint` runs `npm run lint`. `mise run build ::: lint` +passed, including the repository's one baseline test. It does not compile site +assets or establish functional application coverage. + +Code Defender rejected the push because it considers the public repository +unapproved. The hook was not bypassed and the commit was not published through +another channel. PR #584 remains at its previously published README-only head. +The live RepoTable configuration above works independently of that local file. + +`npm ci` reported nine existing dependency vulnerabilities (three moderate, +five high and one critical). Dependencies and the lockfile were not changed by +this configuration work; successful lint/tests are not a security audit. + +## Live verification + +An initial PR-review attempt, `01M2G0SN8CN8H2WJBK70S5GC60`, failed during hydration +because Bedrock Guardrails classified its PR context as +`CONTENT/PROMPT_ATTACK (MEDIUM)`. It had no session ID and launched no MicroVM. +No guardrail settings were changed. This is not runtime verification. The +platform's notification handler nevertheless posted a failed-task status comment +to PR #584 before agent execution; a read-only agent instruction does not disable +those platform notifications. The PR's source head remains +`fd509fa63fa089df356574bdd654123c8621b12e`. + +A separate default-branch inspection, `01M2G0VC4TY37SJ7GGE2JQ7YXF`, was submitted +at 13:14:06 UTC with a 12-turn / $2 limit. The request keeps files and GitHub +records unchanged. It launched +`microvm-f858bc9f-1235-3052-9792-2d9b263cd33b` using managed image `1.0`. +Live logs already confirm the automatic pre-agent build and lint commands +returned `OK` at 13:14:22 UTC. Automatic post-agent build and lint also returned +`OK` at 13:15:56 UTC. These are the platform's own four verification invocations, +not just commands the model chose to run. The task stored `build_passed=true` +and `lint_passed=true`. + +The overall task ended **FAILED** at 13:15:58 UTC: `coding/new-task-v1` requires a +commit/PR, and the inspection intentionally produced neither. Its error records +`agent_status=success, deliverable=lost`. This is evidence that configuration +works and a worker-reported failure is finalized, **not** another successful +coding/PR smoke test. The earlier successful coding/PR tests remain the evidence +for that path. Reported model cost was $0.14489115 and duration 95.1 seconds. + +Cleanup was verified independently: + +- Service state `TERMINATED`; no active MicroVMs remained in the region. +- `/terminate` returned 200 at 13:16:09 UTC with zero active pipeline threads. +- The task's payload prefix was empty. +- The task-owned reservation was `released` at 13:16:09.292 UTC, and the + temporary user's strongly read counter was zero. +- The temporary Cognito user was deleted after checking its recorded subject ID; + a follow-up read returned `UserNotFoundException` at 13:17:11 UTC. The isolated + cached credentials were removed. + +This failure-path observation does not cover a crashed worker, rejected run hook, +lost start reply, or failed cleanup API. Those need their own cases. + +## P2/P3 completion audit + +| Area | Current evidence | Remaining work | +|---|---|---| +| P1 start/poll/stop | Merged; exercised again by the clean deployment | Keep service-fact limits in the original runbook explicit | +| P2 managed build and normal work | Clean bootstrap/image deployment; coding, PR iteration, Memory writes, live logs, cancellation and a worker-reported failure cleanup observed | Complete failure/recovery and effective permission/network matrix | +| Repository command configuration | CLI configured and independently read back; all four automatic pre/post commands pass live; cleanup verified | Local mise file remains unpublished; CLI source still needs upstream review/merge | +| #817 review fixes | Error classification, deletion grants, trusted configuration, malformed-byte handling and comment fixes on takeover branch | Effective-role and ingress negatives; upstream review/merge | +| #700 task-scoped payloads | v2 signed references and deployment manifests implemented; real launches work | Cross-task/public-object denials, expiry, conditional-write and coordinated migration cases | +| #818 registry portability | Large-payload transport/local loader tests; shared HTTPS/443 limits documented | Real remote-tool DNS/TLS/auth/connectivity | +| #841 thread isolation | Regression and Python suite passed | Upstream merge and requested downstream #680 build confirmation | +| #810 logging diagnostics | Unused counter removed; structured failure events tested; normal live logs work | Do not equate normal log delivery with injected live writer-failure evidence | +| Start/capacity/metadata protocols | Local race/replay tests; normal live reservation release and cancellation work | AWS replay/token semantics, unknown-start recovery, effective session permissions, migration/drain and scan-scale checks | +| Managed image updates | First image build succeeded | Make code-only artifact changes trigger a managed-image rebuild and verify it | +| Optional nesting / #857 | Offline prototype only; clean deployment uses current root layout | Production split and full feature/migration validation if adopted | +| P3 sleep/wake | Strategy methods, persistent intent, policy helper and original-deadline handling implemented | Guest hooks, credential/durability barriers, bounded coordinator recovery, approval wiring, scoped grants, compatible-image control and live lifecycle matrix | + +All seven inspected upstream issues (#645, #817, #700, #818, #841, #857 and #810) +remain open. A local fix or one positive AWS run does not establish upstream +completion. ADR-021 defines P1–P3; it has no P4. + +The ordered [P3 implementation plan](./645-p3-implementation-plan.md) remains the +working checklist. Automatic suspension remains disabled. P3 is not complete. + +## Evidence + +Private evidence is retained under `/tmp/abca-645-p2-clean-20260913/`, including +CLI/docs build logs, redacted configuration output, task requests and timestamped +AWS/task/log observations. `config-command-execution-evidence.json` contains the +four commands and their successful completion lines; `config-final-*.json` and +`config-verification-user-cleanup.json` record cleanup. The documentation sync +and **77-page** site build also passed. diff --git a/docs/verification/645-p3-implementation-plan.md b/docs/verification/645-p3-implementation-plan.md index 39da0abcb..1df7b8bfa 100644 --- a/docs/verification/645-p3-implementation-plan.md +++ b/docs/verification/645-p3-implementation-plan.md @@ -15,8 +15,10 @@ validate hooks passed, and authenticated API reads passed after the update. The [live task verification](./645-p2-live-task-20260914.md) subsequently passed normal coding, PR iteration and cancellation on `isadeks/vercel-abca-linear`, including heartbeat, npm checks, Memory writes and automatic cleanup. The -repository's automatic mise build/lint defaults still need proper npm command -configuration; failure/recovery and the wider IAM/network matrix also remain. +repository's [automatic verification configuration](./645-p2-repository-config-20260914.md) +now runs real npm checks through per-repository overrides; all four pre/post +commands passed live, and a worker-reported failure was cleaned up. +Failure/recovery and the wider IAM/network matrix still remain. Full P2 acceptance and all P3 live gates remain open. The batch notes below record what was verified at their original completion; their deployment status is superseded by these records. @@ -39,7 +41,8 @@ is superseded by these records. - [x] Replace unused logging-failure bookkeeping with structured stdout diagnostics (#810); document shared runtime networking and verify large registry payload delivery (#818). - [x] Deploy a fresh bootstrap, application and managed image from current source; verify build hooks and API reads. - [x] Verify normal coding, PR iteration, Memory writes, live logging and successful/canceled-task cleanup in AWS; observe cancellation preserve another task's capacity. -- [ ] Implement production nesting if included, then verify a clean P2 deployment. +- [x] Configure repository verification commands, preserve them through re-onboarding, and verify actual automatic pre/post checks and worker-reported failure cleanup in AWS. +- Optional: production nesting remains unimplemented; the clean deployment uses the existing root layout. Validate the split and migration if adopted. - [x] Add mandatory pause/wake command methods across all three compute strategies, with explicit unsupported results and bounded MicroVM requests. - [x] Keep the original approval deadline through database writes and polling, including frozen/backward clocks; preserve decision races and cancellation. - [x] Save gate/VM-bound lifecycle intent with stale-writer protection; add explicit VM observations and a tested policy helper. @@ -270,8 +273,10 @@ The [2026-09-13–14 clean deployment](./645-p2-clean-deployment-20260913.md) completed the infrastructure, managed-image creation and build-hook checks in steps 1–2 below. The [live task follow-up](./645-p2-live-task-20260914.md) provides positive runtime evidence for steps 3–5 and success/cancellation cleanup -in step 6. Automatic repository build/lint command configuration, failure -cleanup, recovery and negative IAM/network cases remain unverified. +in step 6. The [configuration follow-up](./645-p2-repository-config-20260914.md) +also verifies automatic pre/post npm checks and cleanup after a worker-reported +delivery failure. Crash/rejected-hook/cleanup-error paths, recovery and negative +IAM/network cases remain unverified. `isadeks/vercel-abca-linear` explicitly selects `lambda-microvm`; the original seeded repository retains AgentCore. diff --git a/docs/verification/645-p3-readiness-review.md b/docs/verification/645-p3-readiness-review.md index 990fca24c..718c2b046 100644 --- a/docs/verification/645-p3-readiness-review.md +++ b/docs/verification/645-p3-readiness-review.md @@ -4,6 +4,8 @@ Reviewed 2026-09-13 against `main` commit `5e10038c7e28179b302ac4de78b709795aeba This review covers the existing MicroVM implementation, related P2 follow-ups, comment accuracy and nested-stack feasibility. It includes local experiments, not an AWS deployment or a new live smoke run. The associated [implementation plan](./645-p3-implementation-plan.md) turns the findings into ordered work. +**Live update (2026-09-14):** subsequent work completed a [clean deployment](./645-p2-clean-deployment-20260913.md) and [real coding, PR iteration and cancellation tests](./645-p2-live-task-20260914.md), including Memory writes and runtime logging. Those records supersede the clean-deployment/stdout-ingestion gaps in the historical notes below. Full P2 acceptance and integrated P3 sleep/wake remain open; the original review findings are retained as a dated baseline. + **Implementation update (2026-09-13):** subsequent local batches fix thread isolation, deletion/error/byte/contract bugs, approval heartbeat, stable MicroVM start recovery, atomic capacity reservations and coordinator metadata permissions. The latest batch implements v2 authenticated deployment manifests and single-object payload links for both ECS and MicroVM (#817/#700), with no old unsigned fallback. See the [bootstrap runbook](./645-payload-bootstrap.md) and [implementation progress](./645-p3-implementation-plan.md#implementation-progress). A further local batch removes unused logging counters in favor of structured stdout failures and verifies large registry assets through v2 delivery and the local loader. Effective AWS policies, expiry/networking, stdout ingestion, remote-tool connectivity, clean deployment and P3 sleep/wake remain open. Findings below preserve the original reviewed baseline, rather than describing all of them as current defects. Further local work adds durable MicroVM start receipts, stable request tokens and bounded handle recovery. AWS token-retention/conflict behavior and unknown-ID cleanup remain live verification gates. The shared finalizer's repeated-decrement risk was subsequently reproduced and fixed locally with task-owned reservation transactions, unified counter writers and revision-guarded reconciliation, including approval waits. DynamoDB Local exercises the real transaction conditions; deployed IAM, scan scale and upgrade/drain behavior still need AWS verification. From 5394571d4cf4f20d80b885e7afc7c2afde92d789 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Mon, 14 Sep 2026 09:28:38 -0400 Subject: [PATCH 027/149] Revert "fix(cli): configure and preserve repository verification commands" This reverts commit 62cce37d8f942560be2bbee08e2a046acfb0fd8d. --- cli/README.md | 20 +------ cli/src/commands/repo.ts | 4 -- cli/src/repo-display.ts | 18 +----- cli/src/repo-lookup.ts | 2 - cli/src/repo-onboard.ts | 7 --- cli/test/commands/repo-display.test.ts | 30 ---------- .../commands/repo-onboard-command.test.ts | 21 +------ cli/test/commands/repo-onboard.test.ts | 57 ------------------- 8 files changed, 3 insertions(+), 156 deletions(-) diff --git a/cli/README.md b/cli/README.md index 96e613083..9c331530d 100644 --- a/cli/README.md +++ b/cli/README.md @@ -258,33 +258,15 @@ Register or re-activate a repository in `RepoTable` without a CDK redeploy. With ``` bgagent repo onboard owner/repo \ - --compute-type \ + --compute-type \ --runtime-arn AgentCore runtime override (agentcore only) \ --model \ --token-secret-arn \ --max-turns \ --poll-interval Default agent poll interval in milliseconds \ - --build-command Build verification override \ - --lint-command Lint verification override \ --output ``` -Verification defaults to `mise run build` and `mise run lint`, using tasks in the -target repository's `mise.toml`. For an npm repository without mise tasks, set -the commands explicitly: - -```bash -bgagent repo onboard owner/repo \ - --build-command 'npm ci && npm test' \ - --lint-command 'npm run lint' -``` - -Re-onboarding preserves existing commands when these flags are omitted. Pass an -empty string to restore a command's mise default. `bgagent repo show owner/repo` -shows the effective commands. This operator command updates RepoTable; for a -repository managed by CDK, also set `Blueprint.pipeline.buildCommand` and -`lintCommand` in its source so future Blueprint updates retain the configuration. - ### `bgagent repo offboard ` Soft-delete a repository (`status=removed` + TTL), matching Blueprint delete semantics. An existing Blueprint will re-activate the repo on the next CDK deploy. diff --git a/cli/src/commands/repo.ts b/cli/src/commands/repo.ts index b8a291759..626c4cbdb 100644 --- a/cli/src/commands/repo.ts +++ b/cli/src/commands/repo.ts @@ -159,8 +159,6 @@ export function makeRepoCommand(): Command { .option('--token-secret-arn ', 'Per-repo GitHub token Secrets Manager ARN') .option('--max-turns ', 'Default max turns for tasks', parseInt) .option('--poll-interval ', 'Default agent poll interval in milliseconds', parseInt) - .option('--build-command ', 'Build verification command (empty restores mise run build)') - .option('--lint-command ', 'Lint verification command (empty restores mise run lint)') .option('--output ', 'Output format: text or json', 'text') .action(async (repoId: string, opts) => { assertRepoFormat(repoId); @@ -212,8 +210,6 @@ export function makeRepoCommand(): Command { githubTokenSecretArn: opts.tokenSecretArn, maxTurns: opts.maxTurns, pollIntervalMs: opts.pollInterval, - buildCommand: opts.buildCommand, - lintCommand: opts.lintCommand, }); const notes = buildRepoOnboardNotes({ config, diff --git a/cli/src/repo-display.ts b/cli/src/repo-display.ts index 1f9880035..fa7b04c9d 100644 --- a/cli/src/repo-display.ts +++ b/cli/src/repo-display.ts @@ -48,8 +48,6 @@ export const PLATFORM_REPO_DEFAULTS = { model_id: 'us.anthropic.claude-opus-5', max_turns: 200, poll_interval_ms: 30_000, - build_command: 'mise run build', - lint_command: 'mise run lint', approval_gate_cap: 50, } as const; @@ -72,8 +70,6 @@ export interface RepoConfigDisplay { readonly max_turns: number; readonly max_budget_usd: string; readonly poll_interval_ms: number; - readonly build_command: string; - readonly lint_command: string; readonly approval_gate_cap: number; readonly github_token_source: GithubTokenSource; readonly github_token_secret_arn?: string; @@ -112,8 +108,6 @@ export function formatRepoConfigForDisplay( mark('max_turns', config.max_turns !== undefined); mark('max_budget_usd', config.max_budget_usd !== undefined); mark('poll_interval_ms', config.poll_interval_ms !== undefined); - mark('build_command', Boolean(config.build_command?.trim())); - mark('lint_command', Boolean(config.lint_command?.trim())); mark('approval_gate_cap', config.approval_gate_cap !== undefined); fieldSources.github_token_secret_arn = usesBlueprintToken ? 'blueprint' : 'platform'; @@ -121,7 +115,7 @@ export function formatRepoConfigForDisplay( for (const key of [ 'compute_type', 'runtime_arn', 'model_id', 'max_turns', 'max_budget_usd', 'poll_interval_ms', 'approval_gate_cap', 'system_prompt_overrides', - 'egress_allowlist', 'cedar_policies', 'github_token_secret_arn', 'build_command', 'lint_command', + 'egress_allowlist', 'cedar_policies', 'github_token_secret_arn', ] as const) { const value = config[key]; if (value !== undefined && value !== null && !(Array.isArray(value) && value.length === 0)) { @@ -149,8 +143,6 @@ export function formatRepoConfigForDisplay( ? String(config.max_budget_usd) : 'unlimited', poll_interval_ms: config.poll_interval_ms ?? PLATFORM_REPO_DEFAULTS.poll_interval_ms, - build_command: config.build_command?.trim() || PLATFORM_REPO_DEFAULTS.build_command, - lint_command: config.lint_command?.trim() || PLATFORM_REPO_DEFAULTS.lint_command, approval_gate_cap: config.approval_gate_cap ?? PLATFORM_REPO_DEFAULTS.approval_gate_cap, github_token_source: usesBlueprintToken ? 'blueprint' : 'platform', github_token_secret_arn: effectiveTokenArn @@ -205,14 +197,6 @@ export function buildRepoShowLines(display: RepoConfigDisplay): RepoShowLine[] { display.field_sources.poll_interval_ms, ), }, - { - key: 'build_command', - text: formatSourcedValue(display.effective.build_command, display.field_sources.build_command), - }, - { - key: 'lint_command', - text: formatSourcedValue(display.effective.lint_command, display.field_sources.lint_command), - }, { key: 'approval_gate_cap', text: formatSourcedValue( diff --git a/cli/src/repo-lookup.ts b/cli/src/repo-lookup.ts index 17513fe87..7301d5fdc 100644 --- a/cli/src/repo-lookup.ts +++ b/cli/src/repo-lookup.ts @@ -37,8 +37,6 @@ export interface RepoConfigRow { readonly model_id?: string; readonly max_turns?: number; readonly max_budget_usd?: number; - readonly build_command?: string; - readonly lint_command?: string; readonly system_prompt_overrides?: string; readonly github_token_secret_arn?: string; readonly poll_interval_ms?: number; diff --git a/cli/src/repo-onboard.ts b/cli/src/repo-onboard.ts index 976880f8d..4e30009fd 100644 --- a/cli/src/repo-onboard.ts +++ b/cli/src/repo-onboard.ts @@ -41,8 +41,6 @@ export interface OnboardRepoOptions { readonly maxTurns?: number; readonly githubTokenSecretArn?: string; readonly pollIntervalMs?: number; - readonly buildCommand?: string; - readonly lintCommand?: string; } export interface OnboardRepoDependencies { @@ -116,11 +114,6 @@ export async function onboardRepo( item.approval_gate_cap = existing.approval_gate_cap; } if (existing?.max_budget_usd !== undefined) item.max_budget_usd = existing.max_budget_usd; - // Retain verification recipes when the operator leaves their flags unset. - if (options.buildCommand !== undefined) item.build_command = options.buildCommand; - else if (existing?.build_command !== undefined) item.build_command = existing.build_command; - if (options.lintCommand !== undefined) item.lint_command = options.lintCommand; - else if (existing?.lint_command !== undefined) item.lint_command = existing.lint_command; await ddb.send(new PutCommand({ TableName: tableName, diff --git a/cli/test/commands/repo-display.test.ts b/cli/test/commands/repo-display.test.ts index e868101d6..6d45cf7e0 100644 --- a/cli/test/commands/repo-display.test.ts +++ b/cli/test/commands/repo-display.test.ts @@ -82,36 +82,6 @@ describe('formatRepoConfigForDisplay', () => { }); describe('buildRepoShowLines', () => { - test.each([undefined, '', ' '])('shows actual worker defaults for unset verification commands: %p', (command) => { - const display = formatRepoConfigForDisplay({ - repo: 'acme/foo', status: 'active', build_command: command, lint_command: command, - }, PLATFORM); - const lines = buildRepoShowLines(display); - const postHooks = fs.readFileSync( - path.resolve(__dirname, '../../../agent/src/post_hooks.py'), 'utf8', - ); - for (const kind of ['build', 'lint'] as const) { - const field = `${kind}_command` as const; - const match = postHooks.match(new RegExp(`DEFAULT_${kind.toUpperCase()}_COMMAND = "([^"]+)"`)); - expect(match).not.toBeNull(); - expect(display.effective[field]).toBe(match![1]); - expect(lines.find((line) => line.key === field)?.text).toBe(`(platform default) ${match![1]}`); - } - }); - - test('shows verification overrides in text and JSON', () => { - const commands = { build_command: 'npm ci && npm test', lint_command: 'npm run lint' }; - const display = formatRepoConfigForDisplay({ - repo: 'acme/foo', status: 'active', ...commands, - }, PLATFORM); - expect(display.blueprint_overrides).toMatchObject(commands); - expect(display.effective).toMatchObject(commands); - const lines = buildRepoShowLines(display); - for (const [key, value] of Object.entries(commands)) { - expect(lines.find((line) => line.key === key)?.text).toBe(`${value} (per-blueprint override)`); - } - }); - test('shows platform defaults instead of dash for unset blueprint fields', () => { const display = formatRepoConfigForDisplay( { repo: 'awslabs/agent-plugins', status: 'active' }, diff --git a/cli/test/commands/repo-onboard-command.test.ts b/cli/test/commands/repo-onboard-command.test.ts index e78e8fae2..56524c510 100644 --- a/cli/test/commands/repo-onboard-command.test.ts +++ b/cli/test/commands/repo-onboard-command.test.ts @@ -22,10 +22,7 @@ import { onboardRepo, offboardRepo } from '../../src/repo-onboard'; import { getStackOutput } from '../../src/stack-outputs'; jest.mock('../../src/repo-onboard'); -jest.mock('../../src/stack-outputs', () => ({ - ...jest.requireActual('../../src/stack-outputs'), - getStackOutput: jest.fn(), -})); +jest.mock('../../src/stack-outputs'); describe('repo onboard/offboard commands', () => { let consoleSpy: jest.SpiedFunction; @@ -130,20 +127,4 @@ describe('repo onboard/offboard commands', () => { expect(offboardRepo).toHaveBeenCalled(); expect(consoleSpy.mock.calls[0][0]).toContain('offboarded'); }); - - test.each([ - ['npm ci && npm test', 'npm run lint'], - ['', ''], - ])('repo onboard forwards build/lint commands without interpreting their contents', async (build, lint) => { - const cmd = makeRepoCommand(); - await cmd.parseAsync([ - 'node', 'test', 'onboard', 'acme/a', '--region', 'us-east-1', - '--build-command', build, '--lint-command', lint, - ]); - - expect(onboardRepo).toHaveBeenLastCalledWith( - 'us-east-1', 'RepoTable', 'acme/a', - expect.objectContaining({ buildCommand: build, lintCommand: lint }), - ); - }); }); diff --git a/cli/test/commands/repo-onboard.test.ts b/cli/test/commands/repo-onboard.test.ts index a2a1bd3ca..d6114e3a7 100644 --- a/cli/test/commands/repo-onboard.test.ts +++ b/cli/test/commands/repo-onboard.test.ts @@ -178,37 +178,6 @@ describe('repo onboard/offboard', () => { expect(put.input.Item?.poll_interval_ms).toBe(12345); }); - test.each(['active', 'removed'])( - 'onboardRepo keeps Blueprint verification commands when status is %s', - async (status) => { - const { loadRepoConfig } = jest.requireMock('../../src/repo-lookup') as { - loadRepoConfig: jest.Mock; - }; - const commands = { - build_command: 'npm ci && npm test', - lint_command: 'npm run lint', - }; - loadRepoConfig.mockResolvedValueOnce({ - repo: 'acme/a', - status, - compute_type: 'agentcore', - ...commands, - }); - - const result = await onboardRepo('us-east-1', 'RepoTable', 'acme/a', { - computeType: 'ecs', - }); - - const put = ddbSend.mock.calls[0][0] as PutCommand; - expect(put.input.Item).toMatchObject({ - status: 'active', - compute_type: 'ecs', - ...commands, - }); - expect(result).toMatchObject(commands); - }, - ); - test('onboardRepo treats a missing row as a fresh onboard', async () => { const { loadRepoConfig } = jest.requireMock('../../src/repo-lookup') as { loadRepoConfig: jest.Mock; @@ -217,32 +186,6 @@ describe('repo onboard/offboard', () => { await expect(onboardRepo('us-east-1', 'RepoTable', 'acme/a')).resolves.toBeDefined(); expect(ddbSend).toHaveBeenCalledTimes(1); - const put = ddbSend.mock.calls[0][0] as PutCommand; - expect(put.input.Item).not.toHaveProperty('build_command'); - expect(put.input.Item).not.toHaveProperty('lint_command'); - }); - - test.each([ - { buildCommand: 'npm ci && npm test', lintCommand: 'npm run lint' }, - { buildCommand: '', lintCommand: '' }, - ])('onboardRepo replaces verification commands, including resetting to defaults: %p', async (options) => { - const { loadRepoConfig } = jest.requireMock('../../src/repo-lookup') as { - loadRepoConfig: jest.Mock; - }; - loadRepoConfig.mockResolvedValueOnce({ - repo: 'acme/a', - status: 'active', - build_command: 'old-build', - lint_command: 'old-lint', - }); - - await onboardRepo('us-east-1', 'RepoTable', 'acme/a', options); - - const put = ddbSend.mock.calls[0][0] as PutCommand; - expect(put.input.Item).toMatchObject({ - build_command: options.buildCommand, - lint_command: options.lintCommand, - }); }); test('onboardRepo re-throws non-not-found load errors instead of wiping overrides', async () => { From 011d8f0184c914d667b7e3f69ab6059df56f5771 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Mon, 14 Sep 2026 09:31:12 -0400 Subject: [PATCH 028/149] docs: record CLI rollback and restored mise defaults --- .../verification/645-p2-live-task-20260914.md | 12 +++-- .../645-p2-repository-config-20260914.md | 54 ++++++++++++------- .../645-p3-implementation-plan.md | 11 ++-- 3 files changed, 50 insertions(+), 27 deletions(-) diff --git a/docs/verification/645-p2-live-task-20260914.md b/docs/verification/645-p2-live-task-20260914.md index 983c3f64e..f1bbe4ae3 100644 --- a/docs/verification/645-p2-live-task-20260914.md +++ b/docs/verification/645-p2-live-task-20260914.md @@ -17,6 +17,8 @@ It does **not** discharge every P2 warning. The later [configuration follow-up](./645-p2-repository-config-20260914.md) verifies real automatic build/lint commands and a worker-reported failure cleanup; additional failure/recovery cases and the wider IAM/network matrix remain outstanding. +The temporary CLI configuration was subsequently removed at user request; +repository mise tasks are still needed for the restored defaults. Automatic P3 suspension remains disabled. ## Environment and repository setup @@ -102,9 +104,10 @@ Evidence: `smoke-task-{request,submit}.json`, ## Automatic build/lint configuration in the original runs **Follow-up:** the [repository configuration record](./645-p2-repository-config-20260914.md) -supersedes this gap. Per-repository npm commands are now configured and all four -automatic pre/post checks passed live. The findings below describe the original -two runs before that configuration was applied. +records all four automatic pre/post npm checks passing under temporary overrides. +Those overrides and the CLI addition were subsequently removed at user request; +repository mise tasks are still needed. The findings below describe the original +two runs before the temporary configuration was applied. The default blueprint checks are `mise run build` and `mise run lint`. This repository has no mise tasks. The agent's explicit npm checks passed, @@ -202,7 +205,8 @@ agent image or resource in the existing east deployment was modified. - [x] Application logging works while running; actual Memory events are stored. - [x] Successful and canceled tasks terminate and delete their task payloads. - [x] Cancellation preserves the other task's seat; final active count is zero. -- [x] Configure and re-prove automatic repository build/lint gates — see the [follow-up](./645-p2-repository-config-20260914.md). +- [x] Verify automatic npm build/lint checks under temporary overrides — see the [follow-up](./645-p2-repository-config-20260914.md). +- [ ] Make repository mise tasks available and verify default build/lint commands after withdrawal of the CLI configuration. - [ ] Exercise failure cleanup, lost/replayed start responses and recovery cases. - [ ] Exercise the full real-session IAM and in-guest/public-ingress network matrix. - [ ] Verify Memory retrieval if claiming that behavior. diff --git a/docs/verification/645-p2-repository-config-20260914.md b/docs/verification/645-p2-repository-config-20260914.md index dc7fae330..ea59f399e 100644 --- a/docs/verification/645-p2-repository-config-20260914.md +++ b/docs/verification/645-p2-repository-config-20260914.md @@ -4,35 +4,40 @@ Date: 2026-09-14. Follows the [clean deployment](./645-p2-clean-deployment-20260 and [first live tasks](./645-p2-live-task-20260914.md). The user selected `isadeks/vercel-abca-linear` and authorized completing its configuration. -## Configuration +**Current status — withdrawn at user request:** commit `4e616679` reverts the +entire CLI addition `62cce37d`. At 13:29:11 UTC, the two live command overrides +were also removed from the west repository row using a conditional update. +A strongly consistent read confirmed their absence and retained +`lambda-microvm` routing. Future tasks use the existing `mise run build` / +`mise run lint` defaults. The remote repository still has no `mise.toml`, so +verification setup is open again. The results below preserve the earlier test +evidence; they do not describe the current configuration. + +## Historical configuration The selected repository has npm lint/test scripts but no mise tasks. All three compute backends use the same worker defaults (`mise run build` and `mise run lint`), and all accept per-repository command overrides. -The operator CLI now exposes those existing settings: - -```bash -bgagent repo onboard isadeks/vercel-abca-linear \ - --region us-west-2 --stack-name backgroundagent-dev \ - --compute-type lambda-microvm \ - --build-command 'npm ci && npm test' \ - --lint-command 'npm run lint' -``` +The temporary operator CLI exposed those existing settings and was used to set +`build_command = npm ci && npm test` and `lint_command = npm run lint` for +this repository in `us-west-2`, stack `backgroundagent-dev`. -This command was executed against account ``. A separate +The configuration was applied to account ``. A separate `repo show` confirmed both effective commands and `lambda-microvm` routing. The east deployment was not changed. No infrastructure or image update is required: the deployed coordinator already forwards these fields to the worker. The build command installs the pinned dependencies before running tests, so the following lint command has its tools available. -The CLI also preserves existing commands on re-onboarding and displays their -effective values. Previously, its replacement row silently omitted both fields. -Two regressions failed before the preservation fix. Tests now cover active and +The temporary CLI also preserved existing commands on re-onboarding and displayed +their effective values. Its original replacement row silently omitted both fields. +Two regressions failed before the preservation fix. Tests covered active and removed repositories, backend changes, explicit overrides, empty-string resets, fresh defaults, command parsing and text/JSON display. CLI compile/lint and all -**62 suites / 938 tests** pass. The implementation is commit `62cce37d`. +**62 suites / 938 tests** passed on commit `62cce37d`. The full revert also removes +that preservation fix; the known omission remains a separate follow-up, not a +completed fix on the current branch. For a repository managed through CDK, put the same values in `Blueprint.pipeline.buildCommand` / `lintCommand`; CLI changes alone do not edit @@ -49,7 +54,8 @@ assets or establish functional application coverage. Code Defender rejected the push because it considers the public repository unapproved. The hook was not bypassed and the commit was not published through another channel. PR #584 remains at its previously published README-only head. -The live RepoTable configuration above works independently of that local file. +The historical RepoTable configuration worked independently of that local file; +it is now removed. No publication was attempted as part of the CLI revert. `npm ci` reported nine existing dependency vulnerabilities (three moderate, five high and one critical). Dependencies and the lockfile were not changed by @@ -78,8 +84,8 @@ and `lint_passed=true`. The overall task ended **FAILED** at 13:15:58 UTC: `coding/new-task-v1` requires a commit/PR, and the inspection intentionally produced neither. Its error records -`agent_status=success, deliverable=lost`. This is evidence that configuration -works and a worker-reported failure is finalized, **not** another successful +`agent_status=success, deliverable=lost`. This is evidence that the tested +configuration worked and a worker-reported failure was finalized, **not** another successful coding/PR smoke test. The earlier successful coding/PR tests remain the evidence for that path. Reported model cost was $0.14489115 and duration 95.1 seconds. @@ -103,7 +109,7 @@ lost start reply, or failed cleanup API. Those need their own cases. |---|---|---| | P1 start/poll/stop | Merged; exercised again by the clean deployment | Keep service-fact limits in the original runbook explicit | | P2 managed build and normal work | Clean bootstrap/image deployment; coding, PR iteration, Memory writes, live logs, cancellation and a worker-reported failure cleanup observed | Complete failure/recovery and effective permission/network matrix | -| Repository command configuration | CLI configured and independently read back; all four automatic pre/post commands pass live; cleanup verified | Local mise file remains unpublished; CLI source still needs upstream review/merge | +| Repository command configuration | Temporary npm overrides passed all four pre/post checks; CLI addition and live overrides subsequently removed at user request | Repository mise tasks must be available before the restored default commands can pass | | #817 review fixes | Error classification, deletion grants, trusted configuration, malformed-byte handling and comment fixes on takeover branch | Effective-role and ingress negatives; upstream review/merge | | #700 task-scoped payloads | v2 signed references and deployment manifests implemented; real launches work | Cross-task/public-object denials, expiry, conditional-write and coordinated migration cases | | #818 registry portability | Large-payload transport/local loader tests; shared HTTPS/443 limits documented | Real remote-tool DNS/TLS/auth/connectivity | @@ -129,3 +135,13 @@ AWS/task/log observations. `config-command-execution-evidence.json` contains the four commands and their successful completion lines; `config-final-*.json` and `config-verification-user-cleanup.json` record cleanup. The documentation sync and **77-page** site build also passed. + +Rollback evidence is in `cli-revert-config-{before,update,after}.json`. The +update required the old command values and timestamp to match before removing +only those two settings and refreshing `updated_at`. No test task was launched +for the revert, and the existing deployment/image and east environment were +unchanged. + +After the revert, CLI compile/lint and **62 suites / 928 tests** passed. The +entire `cli/` tree matches its pre-addition state at `396a31e0`, and the rebuilt +`repo onboard --help` no longer includes either new command flag. diff --git a/docs/verification/645-p3-implementation-plan.md b/docs/verification/645-p3-implementation-plan.md index 1df7b8bfa..9711f5506 100644 --- a/docs/verification/645-p3-implementation-plan.md +++ b/docs/verification/645-p3-implementation-plan.md @@ -15,9 +15,11 @@ validate hooks passed, and authenticated API reads passed after the update. The [live task verification](./645-p2-live-task-20260914.md) subsequently passed normal coding, PR iteration and cancellation on `isadeks/vercel-abca-linear`, including heartbeat, npm checks, Memory writes and automatic cleanup. The -repository's [automatic verification configuration](./645-p2-repository-config-20260914.md) -now runs real npm checks through per-repository overrides; all four pre/post -commands passed live, and a worker-reported failure was cleaned up. +repository's [temporary verification configuration](./645-p2-repository-config-20260914.md) +passed all four pre/post npm commands live, and a worker-reported failure was +cleaned up. The user subsequently requested removal of the CLI addition; its +live overrides were also removed. Repository mise tasks are still needed for +the restored default commands. Failure/recovery and the wider IAM/network matrix still remain. Full P2 acceptance and all P3 live gates remain open. The batch notes below record what was verified at their original completion; their deployment status @@ -41,7 +43,8 @@ is superseded by these records. - [x] Replace unused logging-failure bookkeeping with structured stdout diagnostics (#810); document shared runtime networking and verify large registry payload delivery (#818). - [x] Deploy a fresh bootstrap, application and managed image from current source; verify build hooks and API reads. - [x] Verify normal coding, PR iteration, Memory writes, live logging and successful/canceled-task cleanup in AWS; observe cancellation preserve another task's capacity. -- [x] Configure repository verification commands, preserve them through re-onboarding, and verify actual automatic pre/post checks and worker-reported failure cleanup in AWS. +- [x] Verify automatic pre/post npm checks under temporary overrides and worker-reported failure cleanup in AWS. +- [ ] Make repository mise tasks available and verify the restored default commands; the CLI addition and temporary overrides were withdrawn at user request. - Optional: production nesting remains unimplemented; the clean deployment uses the existing root layout. Validate the split and migration if adopted. - [x] Add mandatory pause/wake command methods across all three compute strategies, with explicit unsupported results and bounded MicroVM requests. - [x] Keep the original approval deadline through database writes and polling, including frozen/backward clocks; preserve decision races and cancellation. From 76341d4abcb13b62f850e0b21a1d12ee6c1d7e3a Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Mon, 14 Sep 2026 11:44:46 -0400 Subject: [PATCH 029/149] fix(microvm): rebuild managed images from immutable artifacts --- cdk/scripts/build-microvm-artifact.py | 107 +++++++++ cdk/scripts/package-microvm-artifact.sh | 145 ++++++----- cdk/src/constructs/lambda-microvm-compute.ts | 107 +++++---- cdk/src/stacks/agent.ts | 5 + .../constructs/lambda-microvm-compute.test.ts | 45 +++- .../scripts/package-microvm-artifact.test.ts | 227 ++++++++++++++++++ cdk/test/stacks/agent.test.ts | 11 + .../stacks/microvm-managed-image-nag.test.ts | 1 + docs/design/COMPUTE.md | 4 +- docs/src/content/docs/architecture/Compute.md | 4 +- .../645-microvm-image-rebuild-20260914.md | 91 +++++++ .../645-p3-implementation-plan.md | 4 +- 12 files changed, 628 insertions(+), 123 deletions(-) create mode 100644 cdk/scripts/build-microvm-artifact.py create mode 100644 cdk/test/scripts/package-microvm-artifact.test.ts create mode 100644 docs/verification/645-microvm-image-rebuild-20260914.md diff --git a/cdk/scripts/build-microvm-artifact.py b/cdk/scripts/build-microvm-artifact.py new file mode 100644 index 000000000..3ff0a9cb8 --- /dev/null +++ b/cdk/scripts/build-microvm-artifact.py @@ -0,0 +1,107 @@ +#!/usr/bin/env python3 +# MIT No Attribution +# +# Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. +# +# Permission is hereby granted, free of charge, to any person obtaining a copy of +# the Software without restriction, including without limitation the rights to +# use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of +# the Software, and to permit persons to whom the Software is furnished to do so. +# +# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# SOFTWARE. + +"""Package the Dockerfile's local inputs into a reproducible MicroVM build ZIP.""" + +import argparse +import base64 +import hashlib +import json +from pathlib import Path +import shutil +import stat +import zipfile + + +# Keep these in step with the local COPY sources in agent/Dockerfile. The +# packaging regression checks both directions; COPY --from uses remote stages. +INPUTS = ( + "agent/pyproject.toml", + "agent/uv.lock", + "agent/src", + "agent/policies", + "agent/workflows", + "agent/prepare-commit-msg.sh", + "agent/managed-settings.json", + "contracts", +) +IGNORED_DIRS = {"__pycache__", ".pytest_cache", ".ruff_cache", ".mypy_cache", "node_modules", ".git"} + + +def source_files(root: Path): + """Yield only regular build inputs; never follow a link outside the tree.""" + def walk(path: Path): + if path.is_symlink(): + raise ValueError(f"MicroVM build inputs must not contain symlinks: {path}") + if path.is_dir(): + for child in sorted(path.iterdir()): + if child.name in IGNORED_DIRS or child.name == ".DS_Store": + continue + if child.suffix in {".pyc", ".pyo"}: + continue + yield from walk(child) + elif path.is_file(): + yield path + else: + raise ValueError(f"Missing or unsupported MicroVM build input: {path}") + + dockerfile = root / "agent/Dockerfile" + if dockerfile.is_symlink() or not dockerfile.is_file(): + raise ValueError("agent/Dockerfile must be a regular file") + yield "Dockerfile", dockerfile + for name in INPUTS: + for path in walk(root / name): + yield path.relative_to(root).as_posix(), path + + +def build(root: Path, output: Path): + files = sorted(source_files(root)) + output.parent.mkdir(parents=True, exist_ok=True) + with zipfile.ZipFile(output, "w", compression=zipfile.ZIP_DEFLATED) as archive: + for name, path in files: + # File dates, checkout locations, uid/gid and umask must not cause + # needless image versions. Preserve just the executable permission. + info = zipfile.ZipInfo(name, date_time=(1980, 1, 1, 0, 0, 0)) + info.create_system = 3 + mode = 0o755 if path.stat().st_mode & 0o111 else 0o644 + info.external_attr = (stat.S_IFREG | mode) << 16 + info.compress_type = zipfile.ZIP_DEFLATED + with path.open("rb") as source, archive.open(info, "w") as target: + shutil.copyfileobj(source, target) + digest = hashlib.sha256() + with output.open("rb") as source: + for block in iter(lambda: source.read(1024 * 1024), b""): + digest.update(block) + return { + "sha256": digest.hexdigest(), + "checksum_sha256": base64.b64encode(digest.digest()).decode("ascii"), + "file_count": len(files), + "size_bytes": output.stat().st_size, + } + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--repo-root", required=True, type=Path) + parser.add_argument("--output", required=True, type=Path) + args = parser.parse_args() + print(json.dumps(build(args.repo_root.resolve(), args.output.resolve()))) + + +if __name__ == "__main__": + main() diff --git a/cdk/scripts/package-microvm-artifact.sh b/cdk/scripts/package-microvm-artifact.sh index f825063ca..6f40a26d8 100755 --- a/cdk/scripts/package-microvm-artifact.sh +++ b/cdk/scripts/package-microvm-artifact.sh @@ -25,8 +25,8 @@ # MISE_EXPERIMENTAL=1 mise //cdk:deploy -- --context compute_type=lambda-microvm # # 2. Package + upload the artifact (this script). It reads the bucket name and -# object key straight from the stack outputs, so there is nothing to copy -# by hand: +# base object key from stack outputs, uploads an immutable hash-suffixed +# artifact, and prints the digest required by the subsequent deployment: # # cdk/scripts/package-microvm-artifact.sh --stack-name backgroundagent-dev # @@ -45,11 +45,13 @@ # MISE_EXPERIMENTAL=1 mise //cdk:deploy -- \ # --context compute_type=lambda-microvm \ # --context microvm_base_image_arn= \ -# --context microvm_base_image_version= +# --context microvm_base_image_version= \ +# --context microvm_artifact_sha256= # -# On subsequent agent changes only step 2 is needed, followed by a CloudFormation -# update of the image resource (the service builds a NEW image version from the -# refreshed artifact). +# On subsequent agent changes repeat step 2 and deploy with the NEW digest. +# Changing the digest changes CodeArtifact.Uri, causing a new image version. +# Retain the digest in the deployment's context for later unrelated deploys; +# overwriting a fixed key alone never signals a CloudFormation image update. # # --------------------------------------------------------------------------- # THE OUT-OF-BAND ALTERNATIVE (--create-image) @@ -74,15 +76,15 @@ # is therefore: # # Dockerfile <- verbatim copy of agent/Dockerfile (MicroVM needs it at the root) -# agent/ <- minus .venv/, __pycache__, test caches +# agent/ <- only local Dockerfile COPY inputs, without bytecode/caches # contracts/ <- cross-language constants the agent reads at runtime # # --------------------------------------------------------------------------- -# !! A P2 IMAGE IS FULLY WIRED, BUT NOT SMOKE-VERIFIED !! +# P2 VERIFICATION STATUS # --------------------------------------------------------------------------- -# This script packages and uploads a real artifact, and the image the service -# builds from it will reach ACTIVE, accept a `runHookPayload`, and launch. What -# it does NOT have is any smoke-parity guarantee. +# The 2026-09-14 clean deployment and coding/iteration/cancellation runs passed +# without the earlier manual IAM workaround. The wider failure/security matrix +# remains open; see docs/verification/645-p2-live-task-20260914.md. # # ADR-021 sub-decision 3's hook-phasing table (corrected after the live P1 # verification run, then completed in P2) is now: @@ -106,19 +108,10 @@ # answers fails its lifecycle transition, so # each is enabled only once it is served. # -# A P2 smoke run HAS now completed clone → change → PR on this substrate -# (2026-08-07: two tasks COMPLETED with pull requests, live progress events, and -# the 45 s agent heartbeat observed). What is still missing is a run with NO manual -# intervention: that smoke needed a live IAM workaround, and the two defects behind -# it (ADR-021 P2r2-F9 / P2r2-F10 — the `iam:PassedToService` condition on both -# `iam:PassRole` paths) are fixed in source but not yet re-exercised live. Keep -# production repos on compute_type=agentcore or ecs until a clean run is on record, -# and note that the CDK-managed image path additionally needs bootstrap policy -# bundle >= 1.6.0 (see the banner after upload). The Dockerfile is the P2 tuned base -# and is copied unmodified — further customization (e.g., Alpine adoption) would be -# P2.5 work. +# The clean P2 deployment used bootstrap policy bundle 1.7.0. The Dockerfile is +# copied unmodified; image build success does not establish full P2 acceptance. # -# Requires: awscli v2, zip, rsync, python3 (none of which are installed by this script). +# Requires: awscli v2 with conditional PutObject/checksum support, python3. set -euo pipefail @@ -196,7 +189,7 @@ case " ${SUPPORTED_MEMORY_MIB} " in ;; esac -for tool in aws zip python3 rsync; do +for tool in aws python3; do command -v "$tool" >/dev/null 2>&1 || { echo "error: '$tool' is required but not on PATH" >&2; exit 1; } done @@ -224,6 +217,9 @@ for output in stacks[0].get("Outputs", []): ARTIFACT_BUCKET="$(stack_output MicrovmArtifactBucketName)" ARTIFACT_KEY="$(stack_output MicrovmArtifactObjectKey)" +ARTIFACT_BASE_KEY="$(stack_output MicrovmArtifactBaseObjectKey)" +# Before hashed-artifact support, ObjectKey was always the unsuffixed base. +ARTIFACT_BASE_KEY="${ARTIFACT_BASE_KEY:-${ARTIFACT_KEY}}" BUILD_ROLE_ARN="$(stack_output MicrovmBuildRoleArn)" EGRESS_CONNECTORS="$(stack_output MicrovmEgressConnectorArns)" # BUILD-time connectors (TCP 443 + 80). `agent/Dockerfile` runs `apt-get`, which @@ -265,18 +261,14 @@ echo " log group : ${LOG_GROUP}" print_p1_reminder() { cat <<'EOF' -!! REMINDER (ADR-021 P2): smoke-verified ONCE, and only WITH a manual workaround. - The image is creatable and launchable, the agent serves all four declared hooks - (/ready + /validate on the build path, /run + /terminate at runtime), the - execution role holds its full runtime permission set, and a 2026-08-07 run took - two tasks clone -> change -> PR to COMPLETED with a live 45 s heartbeat. - NOT verified: an UNATTENDED run. That smoke needed a live IAM workaround, and the - two defects behind it (ADR-021 P2r2-F9 / P2r2-F10) are fixed in source but not - re-exercised. The CDK-managed image path also needs bootstrap bundle >= 1.6.0. - Keep production repos on compute_type=agentcore or ecs until a clean run is on - record. /suspend and /resume stay disabled until P3. CDK synth emits the same - warning (abca:microvm-image-p1-smoke-unverified) on every deploy that configures - an image. +REMINDER (ADR-021 P2): clean deployment and coding, iteration and cancellation + passed on 2026-09-14 with bootstrap bundle 1.7.0, without manual IAM changes. + Ready/validate hooks, heartbeat, logging, Memory writes and cleanup have live + evidence. Full P2 acceptance still needs the failure/recovery, effective IAM + and networking matrix in docs/verification/645-p3-implementation-plan.md. + /suspend and /resume remain undeclared until compatible P3 agent hooks land. + CDK retains warning ID abca:microvm-image-p1-smoke-unverified for compatibility; + its text describes the current verification gaps. EOF } @@ -291,44 +283,64 @@ cleanup() { } trap cleanup EXIT -echo "==> Staging agent tree in ${STAGE_DIR}" -# The MicroVM build context is the zip root, and agent/Dockerfile COPYs -# repo-root-relative paths — so the staged tree mirrors the repo, with the -# Dockerfile additionally promoted to the root where the service looks for it. -cp "${REPO_ROOT}/agent/Dockerfile" "${STAGE_DIR}/Dockerfile" - -# Excludes match the build inputs the Dockerfile never COPYs but which dominate -# the zip size: the local virtualenv, Python/pytest caches, and node_modules. -rsync -a \ - --exclude '.venv/' \ - --exclude '__pycache__/' \ - --exclude '.pytest_cache/' \ - --exclude '.ruff_cache/' \ - --exclude '.mypy_cache/' \ - --exclude 'node_modules/' \ - --exclude '*.pyc' \ - "${REPO_ROOT}/agent" "${STAGE_DIR}/" -rsync -a --exclude 'node_modules/' "${REPO_ROOT}/contracts" "${STAGE_DIR}/" - -ARTIFACT_ZIP="${STAGE_DIR}.zip" -rm -f "${ARTIFACT_ZIP}" -echo "==> Zipping to ${ARTIFACT_ZIP}" -( cd "${STAGE_DIR}" && zip -q -r "${ARTIFACT_ZIP}" . ) -echo " $(du -h "${ARTIFACT_ZIP}" | cut -f1) artifact" +ARTIFACT_ZIP="${STAGE_DIR}/agent-artifact.zip" +echo "==> Packaging reproducible build inputs to ${ARTIFACT_ZIP}" +ARTIFACT_INFO="$(python3 "${REPO_ROOT}/cdk/scripts/build-microvm-artifact.py" \ + --repo-root "${REPO_ROOT}" --output "${ARTIFACT_ZIP}")" +ARTIFACT_SHA256="$(python3 -c 'import json,sys; print(json.load(sys.stdin)["sha256"])' <<<"${ARTIFACT_INFO}")" +ARTIFACT_CHECKSUM="$(python3 -c 'import json,sys; print(json.load(sys.stdin)["checksum_sha256"])' <<<"${ARTIFACT_INFO}")" +echo " ${ARTIFACT_INFO}" # --- Upload ------------------------------------------------------------------- +if [[ "${CREATE_IMAGE}" -eq 0 ]]; then + ARTIFACT_KEY="${ARTIFACT_BASE_KEY%.zip}-${ARTIFACT_SHA256}.zip" +else + # The manual create API starts a build explicitly; retain its fixed-key path. + ARTIFACT_KEY="${ARTIFACT_BASE_KEY}" +fi echo "==> Uploading to s3://${ARTIFACT_BUCKET}/${ARTIFACT_KEY}" -aws s3 cp "${ARTIFACT_ZIP}" "s3://${ARTIFACT_BUCKET}/${ARTIFACT_KEY}" -rm -f "${ARTIFACT_ZIP}" +if [[ "${CREATE_IMAGE}" -eq 0 ]]; then + # S3 verifies the bytes against the checksum. Never overwrite a managed build's + # inputs, including a repeated invocation or concurrent publisher. + if aws s3api put-object --bucket "${ARTIFACT_BUCKET}" --key "${ARTIFACT_KEY}" \ + --body "${ARTIFACT_ZIP}" --if-none-match '*' \ + --checksum-algorithm SHA256 --checksum-sha256 "${ARTIFACT_CHECKSUM}" \ + >"${STAGE_DIR}/upload.json" 2>"${STAGE_DIR}/upload.err"; then + echo " Created immutable artifact" + else + UPLOAD_STATUS=$? + case "$(cat "${STAGE_DIR}/upload.err")" in + *"(PreconditionFailed)"*) + EXISTING_CHECKSUM="$(aws s3api head-object --bucket "${ARTIFACT_BUCKET}" \ + --key "${ARTIFACT_KEY}" --checksum-mode ENABLED \ + --query ChecksumSHA256 --output text)" + if [[ "${EXISTING_CHECKSUM}" != "${ARTIFACT_CHECKSUM}" ]]; then + echo "error: existing artifact checksum does not match; refusing to overwrite ${ARTIFACT_KEY}" >&2 + exit 1 + fi + echo " Reusing checksum-verified existing artifact" + ;; + *) + cat "${STAGE_DIR}/upload.err" >&2 + exit "${UPLOAD_STATUS}" + ;; + esac + fi +else + aws s3api put-object --bucket "${ARTIFACT_BUCKET}" --key "${ARTIFACT_KEY}" \ + --body "${ARTIFACT_ZIP}" \ + --checksum-algorithm SHA256 --checksum-sha256 "${ARTIFACT_CHECKSUM}" +fi if [[ "${CREATE_IMAGE}" -eq 0 ]]; then cat < Artifact uploaded. Next: create (or update) the image. - CDK-managed (recommended) — redeploy with the base image pinned. + CDK-managed (recommended) — redeploy with the base image and artifact pinned. + Keep this digest with the deployment's context; it identifies these exact ZIP bytes. - !! RE-BOOTSTRAP REQUIRED (bootstrap policy bundle >= 1.6.0) !! + !! BOOTSTRAP POLICY BUNDLE >= 1.7.0 REQUIRED !! This path took two live-verified fixes to work. The first (ADR-021 P2-F2: the L1 sent hook paths and \`arm64\` where CloudFormation wants ENABLED / ARM_64) is DISCHARGED — change-set early validation now passes. The second (ADR-021 @@ -341,7 +353,7 @@ if [[ "${CREATE_IMAGE}" -eq 0 ]]; then aws cloudformation describe-stacks --stack-name CDKToolkit \\ --query 'Stacks[0].Outputs[?OutputKey==\`BootstrapPolicyVersion\`].OutputValue' --output text - # if that is below 1.6.0: + # if that is below 1.7.0: MISE_EXPERIMENTAL=1 mise //cdk:bootstrap # ComputeTypes must include lambda-microvm Without it the image resource fails with @@ -353,7 +365,8 @@ if [[ "${CREATE_IMAGE}" -eq 0 ]]; then MISE_EXPERIMENTAL=1 mise //cdk:deploy -- \\ --context compute_type=lambda-microvm \\ --context microvm_base_image_arn= \\ - --context microvm_base_image_version= + --context microvm_base_image_version= \\ + --context microvm_artifact_sha256=${ARTIFACT_SHA256} Out of band — re-run this script with: diff --git a/cdk/src/constructs/lambda-microvm-compute.ts b/cdk/src/constructs/lambda-microvm-compute.ts index 6fb0c882b..e004deaac 100644 --- a/cdk/src/constructs/lambda-microvm-compute.ts +++ b/cdk/src/constructs/lambda-microvm-compute.ts @@ -75,10 +75,9 @@ export const MICROVM_BACKEND_TAG_VALUE = 'lambda-microvm'; export const MICROVM_LOG_GROUP_PREFIX = '/aws/lambda-microvms'; /** - * Default S3 key the packaging helper (`cdk/scripts/package-microvm-artifact.sh`) - * uploads the zip + Dockerfile artifact to, inside the artifact bucket this - * construct creates. Kept in one place so the script, the `CfnMicrovmImage` - * `codeArtifact.uri`, and the build role's `s3:GetObject` scope cannot drift. + * Base S3 key for image artifacts. Managed builds insert the ZIP's SHA-256 + * before `.zip`, so changing code changes CodeArtifact.Uri. The unsuffixed key + * remains available to the explicit out-of-band image builder. */ export const MICROVM_ARTIFACT_OBJECT_KEY = 'microvm-images/agent-artifact.zip'; @@ -442,7 +441,7 @@ export function assertLambdaMicrovmRegionSupported(scope: Construct): void { } /** - * The four operator-supplied image inputs, read from CDK context by the stack. + * Operator-supplied image inputs, read from CDK context by the stack. * * Extracted into a type so the stack can resolve them ONCE, before `TaskApi` is * constructed, and hand the same object to this construct — see @@ -451,6 +450,7 @@ export function assertLambdaMicrovmRegionSupported(scope: Construct): void { export interface LambdaMicrovmImageInputs { readonly baseImageArn?: string; readonly baseImageVersion?: string; + readonly artifactSha256?: string; readonly externalImageIdentifier?: string; readonly externalImageVersion?: string; } @@ -574,6 +574,12 @@ export interface LambdaMicrovmComputeProps extends LambdaMicrovmImageInputs { */ readonly baseImageVersion?: string; + /** + * SHA-256 of the uploaded ZIP, printed by package-microvm-artifact.sh. + * Required for managed images: a mutable fixed key does not trigger updates. + */ + readonly artifactSha256?: string; + /** * Identifier (name or ARN) of a MicroVM image built **out of band** — i.e. by * running `cdk/scripts/package-microvm-artifact.sh` and then @@ -601,7 +607,8 @@ export interface LambdaMicrovmComputeProps extends LambdaMicrovmImageInputs { readonly imageName?: string; /** - * S3 key of the zip + Dockerfile artifact inside the artifact bucket. + * Base S3 key of the zip + Dockerfile artifact. Managed builds append the + * artifact digest before `.zip`; the manual builder uses this base key. * @default MICROVM_ARTIFACT_OBJECT_KEY */ readonly artifactObjectKey?: string; @@ -679,7 +686,7 @@ export interface LambdaMicrovmComputeProps extends LambdaMicrovmImageInputs { * * | Props supplied | What happens | When to use it | * |---|---|---| - * | `baseImageArn` + `baseImageVersion` | `AWS::Lambda::MicrovmImage` L1 is synthesized from `s3:///`; {@link imageIdentifier} is its ARN | steady state | + * | `baseImageArn` + `baseImageVersion` + `artifactSha256` | `AWS::Lambda::MicrovmImage` L1 uses the immutable hash-suffixed artifact key; {@link imageIdentifier} is its ARN | steady state | * | `externalImageIdentifier` | no image resource; the supplied identifier is resolved to its exact ARN and handed to the orchestrator | iterating on the snapshot out of band | * | neither | roles + buckets + connectors only; a synth-time **warning**, no image, and no `MICROVM_IMAGE_IDENTIFIER` for the orchestrator | first deploy — you cannot upload the artifact before the bucket that holds it exists | * @@ -687,12 +694,13 @@ export interface LambdaMicrovmComputeProps extends LambdaMicrovmImageInputs { * stack, so the very first `--context compute_type=lambda-microvm` deploy has * nowhere to have put the zip yet. It is a **warning rather than a throw** * precisely so the bootstrap sequence (deploy → run the packaging script - * against the now-existing bucket → redeploy with `microvm_base_image_arn`) is + * against the now-existing bucket → redeploy with the base image and printed + * `microvm_artifact_sha256`) is * possible at all. A `lambda-microvm` task submitted in that interim window * fails fast with the strategy's own "stack deployed without the MicroVM * substrate" error, which names the remedy. * - * ## ⚠️ P2 smoke succeeded with an IAM workaround; clean verification is pending + * ## P2 clean smoke passed; broader acceptance remains open * * Reaching state 1 or 2 provisions a complete substrate, a buildable image, and * a payload-deliverable `/run` path: P1 declares AND the agent serves `/ready` @@ -713,13 +721,14 @@ export interface LambdaMicrovmComputeProps extends LambdaMicrovmImageInputs { * what it can assert) and `/terminate` (in-guest teardown breadcrumb — * {@link TERMINATE_HOOK_TIMEOUT_SECONDS}). * - * The 2026-08-07 smoke completed clone → change → PR with live progress and - * heartbeats, but required a manual IAM workaround. The permanent PassRole - * fixes still need a clean rerun after re-bootstrap to policy bundle >=1.6.0, - * including runtime log verification. The stable - * `abca:microvm-image-p1-smoke-unverified` warning below records that remaining - * work, as does `cdk/scripts/package-microvm-artifact.sh`. Only `/suspend` and `/resume` - * remain undeclared, until P3 implements them: a hook the service calls but + * The 2026-09-14 clean deployment with bootstrap bundle 1.7.0 and subsequent + * coding, iteration and cancellation runs passed without manual IAM changes, + * including heartbeat, runtime logs, Memory writes and cleanup. The full + * failure/recovery, effective IAM and network matrix remains open; see + * docs/verification/645-p3-implementation-plan.md. The stable + * `abca:microvm-image-p1-smoke-unverified` warning below records that scope. + * Only `/suspend` and `/resume` remain undeclared, until P3 implements compatible + * agent hooks: a hook the service calls but * nothing answers fails the corresponding lifecycle transition. * * ## Deliberately NOT here @@ -740,6 +749,8 @@ export class LambdaMicrovmCompute extends Construct { /** Key of the artifact object inside {@link artifactBucket}. */ public readonly artifactObjectKey: string; + /** Unsuffixed key used by the manual builder and packaging helper. */ + public readonly artifactBaseObjectKey: string; /** S3 bucket holding bootstrap manifests, task payloads and private launch references. */ public readonly payloadBucket: s3.Bucket; @@ -833,7 +844,18 @@ export class LambdaMicrovmCompute extends Construct { assertLambdaMicrovmRegionSupported(this); const stack = Stack.of(this); - this.artifactObjectKey = props.artifactObjectKey ?? MICROVM_ARTIFACT_OBJECT_KEY; + const managedImage = Boolean(props.baseImageArn && props.baseImageVersion); + if ((managedImage || props.artifactSha256 !== undefined) + && !/^[a-f0-9]{64}$/.test(props.artifactSha256 ?? '')) { + throw new Error( + 'Managed MicroVM images require microvm_artifact_sha256 (64 lowercase hex characters). ' + + 'Run cdk/scripts/package-microvm-artifact.sh and deploy with its printed artifact digest.', + ); + } + this.artifactBaseObjectKey = props.artifactObjectKey ?? MICROVM_ARTIFACT_OBJECT_KEY; + this.artifactObjectKey = props.artifactSha256 + ? `${this.artifactBaseObjectKey.replace(/\.zip$/, '')}-${props.artifactSha256}.zip` + : this.artifactBaseObjectKey; this.imageName = props.imageName ?? sanitizeImageName(`${stack.stackName}-abca-agent`); // Fail at SYNTH on an unsupported memory size. The service enumerates the @@ -1036,8 +1058,8 @@ export class LambdaMicrovmCompute extends Construct { { id: 'microvm-artifact-mpu-abort', enabled: true, - // The artifact is tens/hundreds of MB, so uploads are multipart; a - // failed `aws s3 cp` otherwise leaves billable parts forever. + // Abort abandoned multipart uploads from alternative publishers. + // The packaging helper currently uses a single PutObject. abortIncompleteMultipartUploadAfter: Duration.days(1), }, ], @@ -1084,12 +1106,12 @@ export class LambdaMicrovmCompute extends Construct { }); grantTagSession(this.buildRole, microvmAssumedBy); - // s3:GetObject only, scoped to the single artifact key — not the bucket. - // The build role runs the `/ready` and `/validate` build hooks, i.e. code - // from the repo under build, so it gets the narrowest possible read. + // Exact selected artifact plus the legacy manual-build key, never the + // bucket or every hash. The managed image only reads its immutable key. this.buildRole.addToPrincipalPolicy(new iam.PolicyStatement({ actions: ['s3:GetObject'], - resources: [this.artifactBucket.arnForObjects(this.artifactObjectKey)], + resources: [...new Set([this.artifactObjectKey, this.artifactBaseObjectKey])] + .map(key => this.artifactBucket.arnForObjects(key)), })); // Build role: keeps `logs:CreateLogGroup` — see `grantMicrovmLogWrites`. this.grantMicrovmLogWrites(this.buildRole, { allowCreateLogGroup: true }); @@ -1291,8 +1313,8 @@ export class LambdaMicrovmCompute extends Construct { // so each is enabled only once it is served. // // `/suspend` and `/resume` stay OMITTED (not `DISABLED`) until P3, - // where the suspend/resume interface widening lands across all three - // strategies. Note that termination does NOT depend on this hook: + // where compatible agent hooks and durability barriers are integrated. + // Note that termination does NOT depend on this hook: // `TerminateMicrovm` removes the VM with or without in-guest // cooperation, which is what makes a best-effort `/terminate` safe to // declare. @@ -1366,47 +1388,40 @@ export class LambdaMicrovmCompute extends Construct { + 'deploy (the artifact bucket must exist before the artifact can be uploaded). Next: run ' + 'cdk/scripts/package-microvm-artifact.sh to upload the zip+Dockerfile, then redeploy with ' + '--context microvm_base_image_arn= --context microvm_base_image_version= ' + + '--context microvm_artifact_sha256= ' + '(or point at an image you built by hand with --context microvm_image_identifier=).', ); } if (this.imageIdentifier) { // Emitted on EVERY deploy that configures an image, in both image states. - // A successful deploy does not establish a clean end-to-end run. The - // successful smoke used a manual IAM workaround; the warning identifies - // the source fixes and re-bootstrap/live verification still outstanding. + // Clean coding runs exist; they do not cover the broader failure/security + // matrix. Keep the warning scoped to those remaining acceptance gates. // // The id is deliberately UNCHANGED across P1→P2 (operators grep for it, and a // rename would read as "the old warning is gone, so it must be fine"). Annotations.of(this).addWarningV2( 'abca:microvm-image-p1-smoke-unverified', - 'A MicroVM image is configured. A P2 smoke run HAS now completed clone -> change -> PR on ' - + 'this substrate (2026-08-07: two tasks COMPLETED with pull requests, progress streaming to ' - + 'bgagent watch, and the 45s agent heartbeat observed live), the agent serves the /ready, ' - + '/validate, /run and /terminate hooks, and the execution role holds its full runtime ' - + 'permission set. What is still MISSING is a run with no manual intervention: that smoke ' - + 'needed a live IAM workaround, and the two defects behind it (ADR-021 P2r2-F9 / P2r2-F10 — ' - + 'the iam:PassedToService condition on both PassRole paths) are fixed in source but NOT yet ' - + 're-exercised live. ALSO REQUIRED: re-bootstrap to policy bundle 1.6.0 or the CDK-managed ' - + 'image path fails with iam:PassRole AccessDenied on the build role. So the backend still ' - + 'carries no smoke-parity guarantee for an unattended deployment - keep production repos on ' - + 'compute_type=agentcore or ecs until a clean run is on record. Only the /suspend and ' - + '/resume runtime hooks remain undeclared, until P3 implements them: a hook the service ' - + 'calls but nothing answers fails the corresponding lifecycle transition. ' - + "(This warning's id still reads p1- by design: it is frozen across phases so operator " - + 'greps and suppression lists keep matching — read the text, not the id, for the phase.)', + 'A MicroVM image is configured. Clean P2 deployment with bootstrap bundle 1.7.0 and ' + + 'coding, iteration and cancellation runs passed on 2026-09-14 without manual IAM changes. ' + + 'The agent serves the declared /ready, /validate, /run and /terminate hooks. ' + + 'Heartbeat, logs, Memory writes and cleanup have live evidence. Full P2 acceptance ' + + 'still needs the failure/recovery, effective IAM and networking matrix in ' + + 'docs/verification/645-p3-implementation-plan.md. The /suspend and /resume hooks remain ' + + 'undeclared until compatible P3 agent hooks are integrated. The warning ID is retained ' + + 'across phases for existing operator filters.', ); } NagSuppressions.addResourceSuppressions([this.artifactBucket, this.payloadBucket], [ { id: 'AwsSolutions-S1', - reason: 'Artifact bucket holds a single build input (the agent zip+Dockerfile) read only by ' + reason: 'Artifact bucket holds versioned agent zip+Dockerfile build inputs read only by ' + 'the Lambda MicroVMs build role; the payload bucket holds ephemeral per-task /run payloads ' + `with a ${MICROVM_PAYLOAD_TTL_DAYS}-day TTL, written only by the orchestrator and ` + 'read through single-object signed URLs; the worker reads only bootstrap manifests. Object-level access ' - + 'logging (a second log bucket + CloudTrail data events) is not justified for a single ' - + 'build input or for transient boot payloads.', + + 'logging (a second log bucket + CloudTrail data events) is not justified for these ' + + 'build inputs or for transient boot payloads.', }, ], true); diff --git a/cdk/src/stacks/agent.ts b/cdk/src/stacks/agent.ts index 76c79c72d..40a8120c4 100644 --- a/cdk/src/stacks/agent.ts +++ b/cdk/src/stacks/agent.ts @@ -400,6 +400,7 @@ export class AgentStack extends Stack { const microvmImageInputs: LambdaMicrovmImageInputs = { baseImageArn: this.node.tryGetContext('microvm_base_image_arn'), baseImageVersion: this.node.tryGetContext('microvm_base_image_version'), + artifactSha256: this.node.tryGetContext('microvm_artifact_sha256'), externalImageIdentifier: this.node.tryGetContext('microvm_image_identifier'), externalImageVersion: this.node.tryGetContext('microvm_image_version'), }; @@ -1155,6 +1156,10 @@ export class AgentStack extends Stack { value: lambdaMicrovm.artifactObjectKey, description: 'S3 key the Lambda MicroVMs artifact must be uploaded to (matches the build role\'s s3:GetObject scope)', }); + new CfnOutput(this, 'MicrovmArtifactBaseObjectKey', { + value: lambdaMicrovm.artifactBaseObjectKey, + description: 'Base artifact key; managed packaging adds the ZIP SHA-256, manual builds use this key', + }); new CfnOutput(this, 'MicrovmBuildRoleArn', { value: lambdaMicrovm.buildRole.roleArn, description: 'IAM role for `aws lambda-microvms create-microvm-image --build-role-arn`', diff --git a/cdk/test/constructs/lambda-microvm-compute.test.ts b/cdk/test/constructs/lambda-microvm-compute.test.ts index e263a35c2..b13079aa9 100644 --- a/cdk/test/constructs/lambda-microvm-compute.test.ts +++ b/cdk/test/constructs/lambda-microvm-compute.test.ts @@ -53,6 +53,7 @@ import { LAMBDA_MICROVM_SUPPORTED_REGIONS } from '../../src/handlers/shared/micr // keep passing after someone lowered the real budget. const BASE_IMAGE_ARN = 'arn:aws:lambda:us-east-1:aws:microvm-image:al2023-1'; +const ARTIFACT_SHA256 = 'a'.repeat(64); const GITHUB_TOKEN_SECRET_ARN = 'arn:aws:secretsmanager:us-east-1:123456789012:secret:abca/github-token-AbCdEf'; /** @@ -67,6 +68,7 @@ interface BuildOptions { readonly region?: string; readonly context?: Record; readonly withImage?: boolean; + readonly artifactSha256?: string | null; readonly externalImageIdentifier?: string; readonly externalImageVersion?: string; readonly withSessionRole?: boolean; @@ -140,6 +142,7 @@ function instantiate(options: BuildOptions = {}): Omit { ...(options.withImage && { baseImageArn: BASE_IMAGE_ARN, baseImageVersion: '1', + artifactSha256: options.artifactSha256 === null ? undefined : options.artifactSha256 ?? ARTIFACT_SHA256, }), externalImageIdentifier: options.externalImageIdentifier, externalImageVersion: options.externalImageVersion, @@ -174,13 +177,33 @@ describe('LambdaMicrovmCompute — image provisioned from a managed base image', 'Fn::Join': ['', [ 's3://', { Ref: Match.stringLikeRegexp('LambdaMicrovmComputeArtifactBucket') }, - `/${MICROVM_ARTIFACT_OBJECT_KEY}`, + `/microvm-images/agent-artifact-${ARTIFACT_SHA256}.zip`, ]], }, }, }); }); + test('a changed artifact updates the URI without replacing the image identity', () => { + const next = build({ withImage: true, artifactSha256: 'b'.repeat(64) }); + const firstImages = template.findResources('AWS::Lambda::MicrovmImage'); + const nextImages = next.template.findResources('AWS::Lambda::MicrovmImage'); + expect(Object.keys(nextImages)).toEqual(Object.keys(firstImages)); + const first = Object.values(firstImages)[0]!.Properties; + const second = Object.values(nextImages)[0]!.Properties; + expect(second.Name).toEqual(first.Name); + expect(second.CodeArtifact.Uri).not.toEqual(first.CodeArtifact.Uri); + expect(JSON.stringify(second.CodeArtifact.Uri)).toContain(`agent-artifact-${'b'.repeat(64)}.zip`); + }); + + test.each([null, '', 'not-a-sha256', 'A'.repeat(64), 'a'.repeat(63)])( + 'rejects managed builds without an exact artifact digest: %p', + artifactSha256 => { + expect(() => instantiate({ withImage: true, artifactSha256 })) + .toThrow(/microvm_artifact_sha256/); + }, + ); + test('builds an ARM64 image at the largest ACCEPTED BASELINE (8 GiB)', () => { // 32768 was rejected live: "The requested memory size of 32768 MiB is not // supported by base MicroVM image …al2023-1. Supported memory sizes in MiB @@ -662,7 +685,7 @@ describe('LambdaMicrovmCompute — image provisioned from a managed base image', } }); - test('build role reads exactly the one artifact object and writes MicroVM logs', () => { + test('build role reads exactly the selected and manual artifacts and writes MicroVM logs', () => { const policies = Object.entries(template.findResources('AWS::IAM::Policy')) .filter(([id]) => id.includes('LambdaMicrovmComputeBuildRole')); expect(policies).toHaveLength(1); @@ -678,6 +701,12 @@ describe('LambdaMicrovmCompute — image provisioned from a managed base image', // Object-scoped, not bucket-scoped. const s3Statement = statements.find((s: { Action: string }) => s.Action === 's3:GetObject'); expect(JSON.stringify(s3Statement.Resource)).toContain(MICROVM_ARTIFACT_OBJECT_KEY); + expect(s3Statement.Resource).toEqual([ + built.stack.resolve(built.construct.artifactBucket.arnForObjects( + `microvm-images/agent-artifact-${ARTIFACT_SHA256}.zip`, + )), + built.stack.resolve(built.construct.artifactBucket.arnForObjects(MICROVM_ARTIFACT_OBJECT_KEY)), + ]); }); test('execution role gets only bootstrap reads, explicit payload/list denies and no writes', () => { @@ -949,11 +978,9 @@ describe('LambdaMicrovmCompute — image provisioned from a managed base image', expect(JSON.stringify(template.toJSON())).not.toContain('CreateMicrovmAuthToken'); }); - test('warns that a configured image has no smoke-parity guarantee (hook phasing)', () => { - // ADR-021 sub-decision 3, as corrected by the live P1 run and completed in P2: - // all four served hooks are declared, so the image is creatable, launchable and - // payload-deliverable — which makes it look even MORE like a working backend, - // while nothing has exercised clone → change → PR on it. + test('distinguishes clean P2 smoke evidence from remaining acceptance and P3 hooks', () => { + // Coding, iteration and cancellation have live evidence. The warning must + // identify the remaining matrix and retain the served/undeclared hook list. const warnings = built.construct.node.metadata.filter(m => m.type === 'aws:cdk:warning'); const message = warnings.map(w => String(w.data)).join('\n'); expect(JSON.stringify(built.construct.node.metadata)) @@ -961,7 +988,9 @@ describe('LambdaMicrovmCompute — image provisioned from a managed base image', // The superseded id must be gone, not merely reworded — operators grep for it. expect(JSON.stringify(built.construct.node.metadata)) .not.toContain('abca:microvm-image-p1-not-runnable'); - expect(message).toContain('smoke'); + expect(message).toContain('Clean P2 deployment'); + expect(message).toContain('2026-09-14 without manual IAM changes'); + expect(message).toContain('failure/recovery, effective IAM and networking matrix'); expect(message).toContain('P2'); // It must state what IS true now, or it reads as the old (wrong) claim — and // the hook list here is what an operator compares against a failed build or a diff --git a/cdk/test/scripts/package-microvm-artifact.test.ts b/cdk/test/scripts/package-microvm-artifact.test.ts new file mode 100644 index 000000000..5e44862de --- /dev/null +++ b/cdk/test/scripts/package-microvm-artifact.test.ts @@ -0,0 +1,227 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import { spawnSync } from 'child_process'; +import { createHash } from 'crypto'; +import * as fs from 'fs'; +import * as os from 'os'; +import * as path from 'path'; + +const REPO = path.resolve(__dirname, '../..'); +const BASE_KEY = 'microvm-images/agent-artifact.zip'; +const BUILD_INPUTS = [ + 'agent/pyproject.toml', 'agent/uv.lock', 'agent/src/runner.py', + 'agent/policies/test.cedar', 'agent/workflows/test.yaml', + 'agent/prepare-commit-msg.sh', 'agent/managed-settings.json', 'contracts/constants.json', +]; + +// An executable fake CLI exercises the real shell/Python packagers without +// network access. It validates PUT checksums and enforces If-None-Match against +// a local object store; unexpected AWS operations fail instead of falling back. +const AWS = `#!/usr/bin/env python3 +import base64, hashlib, json, os, pathlib, sys +args = sys.argv[1:] +store = pathlib.Path(os.environ["MICROVM_TEST_STORE"]) +with (store / "calls.jsonl").open("a") as f: + f.write(json.dumps(args) + "\\n") +def arg(name): + return args[args.index(name) + 1] +if args[:2] == ["cloudformation", "describe-stacks"]: + outputs = { + "MicrovmArtifactBucketName": "test-artifacts", + "MicrovmArtifactObjectKey": "microvm-images/agent-artifact.zip", + "MicrovmBuildRoleArn": "arn:aws:iam::123456789012:role/test", + "MicrovmEgressConnectorArns": "runtime-connector", + "MicrovmBuildEgressConnectorArns": "build-connector", + "MicrovmLogGroupName": "/aws/lambda-microvms/test", + } + if os.environ.get("MICROVM_TEST_BASE_KEY"): + outputs["MicrovmArtifactBaseObjectKey"] = os.environ["MICROVM_TEST_BASE_KEY"] + print(json.dumps({"Stacks": [{"Outputs": [ + {"OutputKey": k, "OutputValue": v} for k,v in outputs.items() + ]}]})) +elif args[:2] == ["s3api", "put-object"]: + if os.environ.get("MICROVM_TEST_PUT_ERROR"): + print("An error occurred (AccessDenied) when calling PutObject", file=sys.stderr) + sys.exit(254) + data = pathlib.Path(arg("--body")).read_bytes() + checksum = base64.b64encode(hashlib.sha256(data).digest()).decode() + assert arg("--checksum-sha256") == checksum + target = store / pathlib.PurePosixPath(arg("--key")).name + if "--if-none-match" in args: + assert arg("--if-none-match") == "*" + if target.exists(): + print("An error occurred (PreconditionFailed) when calling PutObject", file=sys.stderr) + sys.exit(254) + target.write_bytes(data) + print(json.dumps({"ChecksumSHA256": checksum})) +elif args[:2] == ["s3api", "head-object"]: + data = (store / pathlib.PurePosixPath(arg("--key")).name).read_bytes() + print("invalid-checksum" if os.environ.get("MICROVM_TEST_BAD_CHECKSUM") else + base64.b64encode(hashlib.sha256(data).digest()).decode()) +elif args[:2] == ["lambda-microvms", "create-microvm-image"]: + print(json.dumps({"imageArn": "arn:aws:lambda:us-west-2:123456789012:microvm-image:test", + "imageVersion": "1.0"})) +else: + print("Unexpected AWS operation: " + repr(args), file=sys.stderr) + sys.exit(2) +`; + +describe('MicroVM artifact packaging and immutable publication', () => { + let fixture: string; + let store: string; + + function write(relative: string, contents: string): void { + const filename = path.join(fixture, relative); + fs.mkdirSync(path.dirname(filename), { recursive: true }); + fs.writeFileSync(filename, contents); + } + + function packageArtifact(extraEnv: Record = {}, args: string[] = []) { + return spawnSync('bash', [ + path.join(fixture, 'cdk/scripts/package-microvm-artifact.sh'), ...args, + ], { + encoding: 'utf8', + timeout: 20_000, + env: { + ...process.env, + PATH: `${path.join(fixture, 'bin')}${path.delimiter}${process.env.PATH}`, + MICROVM_TEST_STORE: store, + ...extraEnv, + }, + }); + } + + function uploads(): string[][] { + return fs.readFileSync(path.join(store, 'calls.jsonl'), 'utf8').trim().split('\n') + .map(line => JSON.parse(line) as string[]) + .filter(args => args[0] === 's3api' && args[1] === 'put-object'); + } + + beforeEach(() => { + fixture = fs.mkdtempSync(path.join(os.tmpdir(), 'abca-microvm-package-test-')); + store = path.join(fixture, 'store'); + fs.mkdirSync(store); + for (const name of ['package-microvm-artifact.sh', 'build-microvm-artifact.py']) { + write(`cdk/scripts/${name}`, fs.readFileSync(path.join(REPO, 'scripts', name), 'utf8')); + } + write('agent/Dockerfile', 'FROM scratch\nCOPY agent/src/ /app/src/\n'); + for (const file of BUILD_INPUTS) write(file, `contents of ${file}\n`); + write('bin/aws', AWS); + fs.chmodSync(path.join(fixture, 'bin/aws'), 0o755); + fs.chmodSync(path.join(fixture, 'agent/prepare-commit-msg.sh'), 0o755); + }); + + afterEach(() => { + fs.rmSync(fixture, { recursive: true, force: true }); + }); + + test('same inputs reuse identical bytes despite dates/caches; changed source gets a new key', () => { + const first = packageArtifact(); + expect(first.status).toBe(0); + const firstArgs = uploads()[0]!; + const firstKey = firstArgs[firstArgs.indexOf('--key') + 1]!; + const bytes = fs.readFileSync(path.join(store, path.basename(firstKey))); + const hash = createHash('sha256').update(bytes).digest('hex'); + expect(firstKey).toBe(`microvm-images/agent-artifact-${hash}.zip`); + expect(first.stdout).toContain(`--context microvm_artifact_sha256=${hash}`); + + fs.utimesSync(path.join(fixture, 'agent/src/runner.py'), new Date(0), new Date(0)); + write('agent/src/__pycache__/runner.pyc', 'cache'); + write('agent/.coverage', 'not a build input'); + write('agent/tests/test_runner.py', 'not a build input'); + const repeated = packageArtifact(); + expect(repeated.status).toBe(0); + expect(repeated.stdout).toContain('Reusing checksum-verified existing artifact'); + expect(uploads()[1]).toContain(firstKey); + expect(fs.readFileSync(path.join(store, path.basename(firstKey)))).toEqual(bytes); + + write('agent/src/runner.py', 'changed runtime source'); + const changed = packageArtifact(); + expect(changed.status).toBe(0); + expect(uploads()[2]).not.toContain(firstKey); + + const archive = spawnSync('python3', ['-c', ` +import json, sys, zipfile +with zipfile.ZipFile(sys.argv[1]) as z: + print(json.dumps({i.filename: {"date": i.date_time, "mode": (i.external_attr >> 16) & 511} + for i in z.infolist()})) +`, path.join(store, path.basename(firstKey))], { encoding: 'utf8' }); + expect(archive.status).toBe(0); + const entries = JSON.parse(archive.stdout); + expect(Object.keys(entries).sort()).toEqual(['Dockerfile', ...BUILD_INPUTS].sort()); + expect(entries['agent/prepare-commit-msg.sh'].mode).toBe(0o755); + expect(entries['agent/src/runner.py']).toEqual({ date: [1980, 1, 1, 0, 0, 0], mode: 0o644 }); + }); + + test('rejects an existing object whose verified checksum differs', () => { + expect(packageArtifact().status).toBe(0); + const second = packageArtifact({ MICROVM_TEST_BAD_CHECKSUM: '1' }); + expect(second.status).not.toBe(0); + expect(second.stderr).toContain('refusing to overwrite'); + }); + + test('does not treat an authorization failure as an existing artifact', () => { + const result = packageArtifact({ MICROVM_TEST_PUT_ERROR: '1' }); + expect(result.status).not.toBe(0); + expect(result.stderr).toContain('AccessDenied'); + expect(fs.readFileSync(path.join(store, 'calls.jsonl'), 'utf8')).not.toContain('head-object'); + }); + + test('rejects symlinked inputs before any upload', () => { + fs.symlinkSync(path.join(fixture, 'agent/uv.lock'), path.join(fixture, 'agent/src/linked.py')); + const result = packageArtifact(); + expect(result.status).not.toBe(0); + expect(result.stderr).toContain('must not contain symlinks'); + expect(uploads()).toEqual([]); + }); + + test('uses the base-key output without nesting a previous digest into the next key', () => { + expect(packageArtifact({ MICROVM_TEST_BASE_KEY: 'custom/input.zip' }).status).toBe(0); + const args = uploads()[0]!; + expect(args[args.indexOf('--key') + 1]).toMatch(/^custom\/input-[a-f0-9]{64}\.zip$/); + }); + + test('keeps explicit out-of-band creation on the legacy fixed key', () => { + const result = packageArtifact({}, [ + '--create-image', '--base-image-arn', 'arn:aws:lambda:us-west-2:aws:microvm-image:al2023-1', + '--base-image-version', '1', + ]); + expect(result.status).toBe(0); + expect(uploads()[0]).toContain(BASE_KEY); + expect(uploads()[0]).not.toContain('--if-none-match'); + }); +}); + +test('the packager includes every local Dockerfile COPY source and no extra source tree', () => { + const dockerfile = fs.readFileSync(path.join(REPO, '../agent/Dockerfile'), 'utf8'); + const copied = dockerfile.split('\n') + .filter(line => line.startsWith('COPY ') && !line.includes('--from=')) + .flatMap(line => line.trim().split(/\s+/).slice(1, -1)) + .map(source => source.replace(/\/$/, '')); + const helper = spawnSync('python3', ['-c', ` +import importlib.util, json, sys +spec = importlib.util.spec_from_file_location("artifact", sys.argv[1]) +module = importlib.util.module_from_spec(spec) +spec.loader.exec_module(module) +print(json.dumps(module.INPUTS)) +`, path.join(REPO, 'scripts/build-microvm-artifact.py')], { encoding: 'utf8' }); + expect(helper.status).toBe(0); + expect(JSON.parse(helper.stdout).sort()).toEqual(copied.sort()); +}); diff --git a/cdk/test/stacks/agent.test.ts b/cdk/test/stacks/agent.test.ts index 8156d64cb..a41c6129c 100644 --- a/cdk/test/stacks/agent.test.ts +++ b/cdk/test/stacks/agent.test.ts @@ -1086,6 +1086,7 @@ describe('AgentStack with the Lambda MicroVMs substrate gate (--context compute_ compute_type: 'lambda-microvm', microvm_base_image_arn: BASE_IMAGE_ARN, microvm_base_image_version: '1', + microvm_artifact_sha256: 'a'.repeat(64), }, }); const stack = new AgentStack(app, 'TestAgentStackMicrovm', { @@ -1114,6 +1115,7 @@ describe('AgentStack with the Lambda MicroVMs substrate gate (--context compute_ for (const output of [ 'MicrovmArtifactBucketName', 'MicrovmArtifactObjectKey', + 'MicrovmArtifactBaseObjectKey', 'MicrovmBuildRoleArn', 'MicrovmExecutionRoleArn', 'MicrovmEgressConnectorArns', @@ -1126,6 +1128,14 @@ describe('AgentStack with the Lambda MicroVMs substrate gate (--context compute_ } }); + test('forwards the uploaded artifact digest into the image URI and outputs', () => { + const key = `microvm-images/agent-artifact-${'a'.repeat(64)}.zip`; + template.hasOutput('MicrovmArtifactObjectKey', { Value: key }); + template.hasOutput('MicrovmArtifactBaseObjectKey', { Value: 'microvm-images/agent-artifact.zip' }); + const image = Object.values(template.findResources('AWS::Lambda::MicrovmImage'))[0]!; + expect(JSON.stringify(image.Properties.CodeArtifact.Uri)).toContain(key); + }); + test('the build and runtime egress connector outputs are DIFFERENT connectors', () => { const outputs = template.toJSON().Outputs as Record; expect(JSON.stringify(outputs.MicrovmEgressConnectorArns.Value)) @@ -1319,6 +1329,7 @@ describe('AgentStack with the Lambda MicroVMs substrate gate (--context compute_ microvm_region_override: true, microvm_base_image_arn: BASE_IMAGE_ARN, microvm_base_image_version: '1', + microvm_artifact_sha256: 'a'.repeat(64), }, }); overriddenTemplate = Template.fromStack(new AgentStack(app, 'TestAgentStackMicrovmOverride', { diff --git a/cdk/test/stacks/microvm-managed-image-nag.test.ts b/cdk/test/stacks/microvm-managed-image-nag.test.ts index 92be4fb56..9d0944f79 100644 --- a/cdk/test/stacks/microvm-managed-image-nag.test.ts +++ b/cdk/test/stacks/microvm-managed-image-nag.test.ts @@ -34,6 +34,7 @@ describe.each([false, true])('managed MicroVM image security checks (extra wildc compute_type: 'lambda-microvm', microvm_base_image_arn: 'arn:aws:lambda:us-west-2:aws:microvm-image:al2023-1', microvm_base_image_version: '1', + microvm_artifact_sha256: 'a'.repeat(64), [AGENTCORE_AZS_CONTEXT_KEY]: ['us-west-2a', 'us-west-2b'], }, }, diff --git a/docs/design/COMPUTE.md b/docs/design/COMPUTE.md index 213c9ba1a..d50b215bf 100644 --- a/docs/design/COMPUTE.md +++ b/docs/design/COMPUTE.md @@ -77,7 +77,9 @@ See [ORCHESTRATOR.md](./ORCHESTRATOR.md) for how the orchestrator handles these ## Lambda MicroVMs backend -Lambda MicroVMs are an opt-in third backend, selected per repository with `compute_type: lambda-microvm`; AgentCore remains the default. Image configuration has three states: a managed base-image ARN and version creates the snapshot image in CDK; an external image identifier uses a snapshot built out of band; and supplying neither provisions only the roles, buckets, and connectors needed for the bootstrap deploy. `cdk/scripts/package-microvm-artifact.sh` packages the agent as zip + Dockerfile, uploads it to the artifact bucket, and can create the external image. Lambda MicroVMs are available in five launch regions (us-east-1, us-east-2, us-west-2, eu-west-1, ap-northeast-1) and will expand; the platform enforces regional availability in layers via a synth-time constant, onboarding live probes, and orchestration-time classification. +Lambda MicroVMs are an opt-in third backend, selected per repository with `compute_type: lambda-microvm`; AgentCore remains the default. Image configuration has three states: a managed base-image ARN, version and artifact digest create the snapshot image in CDK; an external image identifier uses a snapshot built out of band; and supplying neither image provisions only the roles, buckets, and connectors needed for the bootstrap deploy. Lambda MicroVMs are available in five launch regions (us-east-1, us-east-2, us-west-2, eu-west-1, ap-northeast-1) and will expand; the platform enforces regional availability in layers via a synth-time constant, onboarding live probes, and orchestration-time classification. + +For managed images, run `cdk/scripts/package-microvm-artifact.sh --stack-name ` after each agent change. It packages the Dockerfile's local inputs into a deterministic ZIP and uploads it under `microvm-images/agent-artifact-.zip`. The checksum is a fingerprint of the uploaded bytes: identical inputs reuse the same verified object, and changed inputs produce a new filename. Deploy with the printed `--context microvm_artifact_sha256=` alongside `microvm_base_image_arn` and `microvm_base_image_version`, and retain these inputs for later deployments. The changed S3 URI tells CloudFormation to update the existing image; overwriting the old fixed filename alone does not. Missing or malformed digests fail synthesis. An initial deployment without an image must create the bucket first. The script's explicit `--create-image` alternative retains the fixed base key and calls the image API directly. Because a snapshot freezes its build-time environment, current deployment identifiers arrive through the v2 payload bootstrap. The coordinator publishes a non-secret deployment manifest and sends a single-object signed download URL for the task. The worker reads only its deployment's `bootstrap/*` with ambient credentials; other object reads and payload-bucket listing are explicitly denied. The downloaded task identity and configuration must match the authenticated manifest before configuration installation. The serialized reference fits the verified 4,096-byte hook limit; all payload sizes use S3. ECS shares this transport through `AGENT_PAYLOAD_REF`. [ADR-021 §3](../decisions/ADR-021-lambda-microvms-compute-backend.md#3-packaging-same-agent-image-source-new-build-path) defines the wire format, compatibility change and pending live gates. diff --git a/docs/src/content/docs/architecture/Compute.md b/docs/src/content/docs/architecture/Compute.md index ef65554e9..0e2580e43 100644 --- a/docs/src/content/docs/architecture/Compute.md +++ b/docs/src/content/docs/architecture/Compute.md @@ -81,7 +81,9 @@ See [ORCHESTRATOR.md](/sample-autonomous-cloud-coding-agents/architecture/orches ## Lambda MicroVMs backend -Lambda MicroVMs are an opt-in third backend, selected per repository with `compute_type: lambda-microvm`; AgentCore remains the default. Image configuration has three states: a managed base-image ARN and version creates the snapshot image in CDK; an external image identifier uses a snapshot built out of band; and supplying neither provisions only the roles, buckets, and connectors needed for the bootstrap deploy. `cdk/scripts/package-microvm-artifact.sh` packages the agent as zip + Dockerfile, uploads it to the artifact bucket, and can create the external image. Lambda MicroVMs are available in five launch regions (us-east-1, us-east-2, us-west-2, eu-west-1, ap-northeast-1) and will expand; the platform enforces regional availability in layers via a synth-time constant, onboarding live probes, and orchestration-time classification. +Lambda MicroVMs are an opt-in third backend, selected per repository with `compute_type: lambda-microvm`; AgentCore remains the default. Image configuration has three states: a managed base-image ARN, version and artifact digest create the snapshot image in CDK; an external image identifier uses a snapshot built out of band; and supplying neither image provisions only the roles, buckets, and connectors needed for the bootstrap deploy. Lambda MicroVMs are available in five launch regions (us-east-1, us-east-2, us-west-2, eu-west-1, ap-northeast-1) and will expand; the platform enforces regional availability in layers via a synth-time constant, onboarding live probes, and orchestration-time classification. + +For managed images, run `cdk/scripts/package-microvm-artifact.sh --stack-name ` after each agent change. It packages the Dockerfile's local inputs into a deterministic ZIP and uploads it under `microvm-images/agent-artifact-.zip`. The checksum is a fingerprint of the uploaded bytes: identical inputs reuse the same verified object, and changed inputs produce a new filename. Deploy with the printed `--context microvm_artifact_sha256=` alongside `microvm_base_image_arn` and `microvm_base_image_version`, and retain these inputs for later deployments. The changed S3 URI tells CloudFormation to update the existing image; overwriting the old fixed filename alone does not. Missing or malformed digests fail synthesis. An initial deployment without an image must create the bucket first. The script's explicit `--create-image` alternative retains the fixed base key and calls the image API directly. Because a snapshot freezes its build-time environment, current deployment identifiers arrive through the v2 payload bootstrap. The coordinator publishes a non-secret deployment manifest and sends a single-object signed download URL for the task. The worker reads only its deployment's `bootstrap/*` with ambient credentials; other object reads and payload-bucket listing are explicitly denied. The downloaded task identity and configuration must match the authenticated manifest before configuration installation. The serialized reference fits the verified 4,096-byte hook limit; all payload sizes use S3. ECS shares this transport through `AGENT_PAYLOAD_REF`. [ADR-021 §3](/sample-autonomous-cloud-coding-agents/architecture/adr-021-lambda-microvms-compute-backend#3-packaging-same-agent-image-source-new-build-path) defines the wire format, compatibility change and pending live gates. diff --git a/docs/verification/645-microvm-image-rebuild-20260914.md b/docs/verification/645-microvm-image-rebuild-20260914.md new file mode 100644 index 000000000..1ce20aa09 --- /dev/null +++ b/docs/verification/645-microvm-image-rebuild-20260914.md @@ -0,0 +1,91 @@ +# Managed MicroVM image rebuild verification + +## Purpose and status + +The original managed image referenced `microvm-images/agent-artifact.zip`. +Uploading new bytes to that filename did not change the CloudFormation +template, so an ordinary deployment could leave the old agent image active. + +The source fix is implemented and four targeted suites pass 229 tests. +Full-build and live-update results are recorded separately below. This work +does not complete the remaining P2 acceptance matrix or enable P3 suspension. + +## Deployment workflow + +1. Run the full root `mise run build` before deployment. +2. Use the intended AWS profile and Region, then run + `cdk/scripts/package-microvm-artifact.sh --stack-name `. + An initial deployment without an image must create the artifact bucket first. +3. The script packages the Dockerfile's local inputs and prints their ZIP's + SHA-256 digest. It uploads + `microvm-images/agent-artifact-.zip`, verifying the checksum with S3. + Repeating the upload reuses the object only after verifying its checksum. + It refuses to overwrite an existing object with a different checksum. +4. Synthesize and review the deployment using `compute_type=lambda-microvm`, + the existing `microvm_base_image_arn` and `microvm_base_image_version`, and + the printed `microvm_artifact_sha256`. Execute the reviewed change set. +5. Verify both CloudFormation completion and the actual image build/version. + Retain these context inputs for future deployments, including unrelated + infrastructure changes. After changing agent source, package again and use + the newly printed digest. + +The digest is a fingerprint of the ZIP bytes. Source file timestamps, checkout +paths and caches do not affect it. Changed runtime inputs change the filename, +which changes `CodeArtifact.Uri` and requests an update to the existing image. +The packaging step remains explicit; CDK does not upload this artifact itself. +Base-image changes can also request a rebuild while using the same artifact. + +The build role reads the selected immutable object and the legacy base object +only. The script's explicit `--create-image` mode keeps using the base object +and directly requests an out-of-band build. The new +`MicrovmArtifactBaseObjectKey` output prevents repeated managed packaging from +adding a second digest onto the previous filename. Older stack outputs work +for the first upgrade because their object key is the unsuffixed base. + +For an artifact rollback, retain the previous ZIP and deploy its digest after +reviewing the change set. This requests another build; it does not promise that +AWS retains or instantly reactivates a previous image version. Artifact objects +have no automatic expiration. Missing or malformed managed-image digests fail +synthesis with packaging instructions. + +## Local verification + +- The pre-fix regression reproduced identical image URIs for two artifact + revisions and acceptance of missing/invalid digests. +- Four targeted suites passed 229 tests: the real shell/Python packagers with + an isolated fake AWS CLI, construct image/IAM assertions, stack wiring and + the production CDK-nag path. +- Packaging cases cover byte-for-byte reuse despite changed timestamps/caches, + a changed runtime source, checksum mismatch, authorization failure, + symlink rejection, custom base keys and manual image creation. A guard + compares the packager's input manifest against every local Dockerfile COPY. +- Image assertions keep the logical ID and name stable while changing the URI. + Build-role reads remain exact object ARNs. +- Shell and Python syntax checks pass. +- Full root build: passed, exit 0 in 897.66 seconds. CDK: 223 suites / + 4,797 tests; CLI: 62 suites / 928 tests; Python: 1,823 tests. Compilation, + lint, formatting, types, drift checks, the 77-page docs build and links pass. + Two existing DynamoDB Local suites / 38 cases were skipped because that + service was not running; this change does not modify the capacity protocol. + The first full run caught an omitted hook list in the revised warning; the + final run includes the corrected warning and its passing regression. + +## Live verification + +Target: existing `backgroundagent-dev`, account ``, `us-west-2`, +profile `sphia-dev`, bootstrap policy bundle `1.7.0`. + +Before this update, CloudFormation was `UPDATE_COMPLETE`, the managed image +`backgroundagent-dev-abca-agent` had latest active version `1.0`, and all four +listed task MicroVMs were terminated. AWS's resource schema marks only `Name` +as create-only; `CodeArtifact.Uri` supports an in-place update. + +- [ ] Upload and checksum-verify the immutable artifact in S3. +- [ ] Review a change set preserving image identity and existing infrastructure. +- [ ] Execute the normal update and verify a successful new active image version. +- [ ] Verify repeating the same artifact digest requests no further image change. + +Private command output and AWS responses are retained under +`/tmp/abca-645-p2-clean-20260913/image-rebuild-*`. +The [P3 implementation plan](./645-p3-implementation-plan.md) tracks the +remaining failure/recovery, permission/network and sleep/wake work. diff --git a/docs/verification/645-p3-implementation-plan.md b/docs/verification/645-p3-implementation-plan.md index 9711f5506..026f1e79f 100644 --- a/docs/verification/645-p3-implementation-plan.md +++ b/docs/verification/645-p3-implementation-plan.md @@ -44,8 +44,10 @@ is superseded by these records. - [x] Deploy a fresh bootstrap, application and managed image from current source; verify build hooks and API reads. - [x] Verify normal coding, PR iteration, Memory writes, live logging and successful/canceled-task cleanup in AWS; observe cancellation preserve another task's capacity. - [x] Verify automatic pre/post npm checks under temporary overrides and worker-reported failure cleanup in AWS. +- [x] Give managed image builds immutable, checksum-verified artifacts and require their digest in deployment context; packaging, construct, stack and CDK-nag regressions pass locally. +- [ ] Verify a normal CloudFormation update builds a new managed image version from the changed artifact URI, then verify repeating the same digest makes no image change. - [ ] Make repository mise tasks available and verify the restored default commands; the CLI addition and temporary overrides were withdrawn at user request. -- Optional: production nesting remains unimplemented; the clean deployment uses the existing root layout. Validate the split and migration if adopted. +- Optional: production nesting remains unimplemented; the clean deployment uses 474 of the root stack's 500 resource slots. P3 does not inherently require nesting. Recheck the count for supported feature combinations and validate the split/migration if adopted. - [x] Add mandatory pause/wake command methods across all three compute strategies, with explicit unsupported results and bounded MicroVM requests. - [x] Keep the original approval deadline through database writes and polling, including frozen/backward clocks; preserve decision races and cancellation. - [x] Save gate/VM-bound lifecycle intent with stale-writer protection; add explicit VM observations and a tested policy helper. From 364ea18a1334b2896315129871bf27a5e5b5df52 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Mon, 14 Sep 2026 12:17:50 -0400 Subject: [PATCH 030/149] docs(microvm): record clean image update and repeat deployment --- .../645-microvm-image-rebuild-20260914.md | 55 +++++++++++++++++-- .../645-p2-clean-deployment-20260913.md | 11 ++-- .../645-p3-implementation-plan.md | 13 +++-- 3 files changed, 65 insertions(+), 14 deletions(-) diff --git a/docs/verification/645-microvm-image-rebuild-20260914.md b/docs/verification/645-microvm-image-rebuild-20260914.md index 1ce20aa09..9af17a9a7 100644 --- a/docs/verification/645-microvm-image-rebuild-20260914.md +++ b/docs/verification/645-microvm-image-rebuild-20260914.md @@ -6,8 +6,9 @@ The original managed image referenced `microvm-images/agent-artifact.zip`. Uploading new bytes to that filename did not change the CloudFormation template, so an ordinary deployment could leave the old agent image active. -The source fix is implemented and four targeted suites pass 229 tests. -Full-build and live-update results are recorded separately below. This work +The fix is committed as `e1d5debe`. The full build passed, the normal +CloudFormation update built and activated image version `2.0`, and a repeat +deployment reported no changes. This work does not complete the remaining P2 acceptance matrix or enable P3 suspension. ## Deployment workflow @@ -80,10 +81,52 @@ Before this update, CloudFormation was `UPDATE_COMPLETE`, the managed image listed task MicroVMs were terminated. AWS's resource schema marks only `Name` as create-only; `CodeArtifact.Uri` supports an in-place update. -- [ ] Upload and checksum-verify the immutable artifact in S3. -- [ ] Review a change set preserving image identity and existing infrastructure. -- [ ] Execute the normal update and verify a successful new active image version. -- [ ] Verify repeating the same artifact digest requests no further image change. +- [x] Upload and checksum-verify the immutable artifact in S3. +- [x] Review a change set preserving image identity and existing infrastructure. +- [x] Execute the normal update and verify a successful new active image version. +- [x] Verify repeating the same artifact digest requests no further image change. + +The uploaded ZIP contains **106 files / 467,322 bytes**. Every archived file +matches its source bytes; ZIP integrity checks pass. S3 reports the matching +SHA-256 checksum and AES256 encryption. The artifact digest is +`86219317d92fc501b58d21df02dcb298955576c28751703f0753cc655c1c0a21`. + +CDK prepared change set `abca-645-image-rebuild-20260914`. AWS reported ten +modifications and no additions/removals: the image URI and build-role policy, +six metadata-only changes, the AgentCore container reference and the existing +`awslabs/agent-plugins` blueprint timestamp refresh. The image has +`Replacement: False`. The blueprint's custom-resource physical ID is stable; +its update refreshes active status/time for that blueprint. No target-repository +verification overrides were restored. + +The incidental AgentCore update comes from its existing repository-root asset +fingerprint; `agent/` and `contracts/` source still match the earlier deployed +`29dcaa74`. A textual `GetTemplate` comparison also showed Unicode characters +as question marks, including a layer description, although the original +synthesized template contains Unicode. The actual AWS change set excludes +those apparent description changes and does not replace that layer. + +The build-role policy completed its update at **16:08:15 UTC**, before the image +update started at **16:08:17 UTC**. Version `2.0` used the hash-suffixed URI; +`/ready` and `/validate` returned HTTP 200 in its version-specific log streams. +Validation reported zero warnings. The version reached `SUCCESSFUL` / `ACTIVE`, +and the image's `latestActiveImageVersion` became `2.0` under its original ARN: +`arn:aws:lambda:us-west-2::microvm-image:backgroundagent-dev-abca-agent`. +CloudFormation reached `UPDATE_COMPLETE`, retaining **474 root resources**. +No manual IAM changes or image API updates were needed. + +Running the packager again against the upgraded stack outputs reused the same +checksum-verified object and printed the same digest. A normal CDK deployment +of the identical reviewed cloud assembly exited 0 with **`(no changes)`** and +zero deployment time. The subsequent version list contains `1.0` and `2.0`; +no `3.0` was created. Fresh synthesis can still refresh the unrelated blueprint +timestamps described above, while unchanged artifact input preserves the image URI. + +All four listed task MicroVMs remain terminated. This verification covers image +rebuild and activation; no new coding task was launched. The earlier coding, +iteration and cancellation evidence remains in the +[version 1.0 live task record](./645-p2-live-task-20260914.md). +P3 `/suspend` and `/resume` hooks remain undeclared. Private command output and AWS responses are retained under `/tmp/abca-645-p2-clean-20260913/image-rebuild-*`. diff --git a/docs/verification/645-p2-clean-deployment-20260913.md b/docs/verification/645-p2-clean-deployment-20260913.md index 7e387b58f..eb271e03f 100644 --- a/docs/verification/645-p2-clean-deployment-20260913.md +++ b/docs/verification/645-p2-clean-deployment-20260913.md @@ -342,7 +342,10 @@ steps and the remaining limits: 5. Record those results before discharging P2 warnings or enabling automatic P3 suspension. -A separate rebuild follow-up remains: overwriting the fixed artifact S3 key does -not itself change the CloudFormation image properties. Code-only redeployments -need a content/version-based image update trigger. This first image creation is -unaffected; later automatic rebuilds have not been verified. +The rebuild follow-up identified here is now resolved in the +[managed image update record](./645-microvm-image-rebuild-20260914.md): +overwriting the fixed S3 key did not change CloudFormation image properties. +Commit `e1d5debe` adds immutable hash-suffixed artifacts and requires their digest +in deployment context. A normal update built and activated image `2.0`; repeat +packaging reused the verified object and a same-assembly deployment made no +changes. Packaging and passing the printed digest remain explicit operator steps. diff --git a/docs/verification/645-p3-implementation-plan.md b/docs/verification/645-p3-implementation-plan.md index 026f1e79f..8fde8e73d 100644 --- a/docs/verification/645-p3-implementation-plan.md +++ b/docs/verification/645-p3-implementation-plan.md @@ -9,9 +9,14 @@ Prerequisite work is tracked here on `fix/645-microvm-readiness`. “Completed **Live infrastructure and image deployed (2026-09-14):** the [clean P2 deployment record](./645-p2-clean-deployment-20260913.md) tracks the new Oregon environment, four deployment fixes and actual verification results. -Bootstrap 1.7.0 and application source `29dcaa74` are deployed. CloudFormation -reached `UPDATE_COMPLETE`; managed image version `1.0` is active, its ready and +The clean deployment used bootstrap 1.7.0 and application source `29dcaa74`. +CloudFormation reached `UPDATE_COMPLETE`; managed image version `1.0` became active, its ready and validate hooks passed, and authenticated API reads passed after the update. +The subsequent [image rebuild fix](./645-microvm-image-rebuild-20260914.md) +deployed `e1d5debe` through a normal reviewed update and activated version `2.0`. +Its ready/validate hooks passed; repeated packaging reused the artifact, and +redeploying the same cloud assembly reported no changes. The root still has +474 resources. The [live task verification](./645-p2-live-task-20260914.md) subsequently passed normal coding, PR iteration and cancellation on `isadeks/vercel-abca-linear`, including heartbeat, npm checks, Memory writes and automatic cleanup. The @@ -44,8 +49,8 @@ is superseded by these records. - [x] Deploy a fresh bootstrap, application and managed image from current source; verify build hooks and API reads. - [x] Verify normal coding, PR iteration, Memory writes, live logging and successful/canceled-task cleanup in AWS; observe cancellation preserve another task's capacity. - [x] Verify automatic pre/post npm checks under temporary overrides and worker-reported failure cleanup in AWS. -- [x] Give managed image builds immutable, checksum-verified artifacts and require their digest in deployment context; packaging, construct, stack and CDK-nag regressions pass locally. -- [ ] Verify a normal CloudFormation update builds a new managed image version from the changed artifact URI, then verify repeating the same digest makes no image change. +- [x] Give managed image builds immutable, checksum-verified artifacts and require their digest in deployment context; packaging, construct, stack and CDK-nag regressions and the full build pass. +- [x] Verify a normal CloudFormation update builds and activates image `2.0` from the changed artifact URI; repeat packaging reuses the verified object and a same-assembly redeploy reports no changes. - [ ] Make repository mise tasks available and verify the restored default commands; the CLI addition and temporary overrides were withdrawn at user request. - Optional: production nesting remains unimplemented; the clean deployment uses 474 of the root stack's 500 resource slots. P3 does not inherently require nesting. Recheck the count for supported feature combinations and validate the split/migration if adopted. - [x] Add mandatory pause/wake command methods across all three compute strategies, with explicit unsupported results and bounded MicroVM requests. From 2e0262c009ca9dcc458f89f31519aa45ada2336f Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Mon, 14 Sep 2026 13:09:22 -0400 Subject: [PATCH 031/149] test(microvm): verify live P2 payload failure boundaries --- cdk/test/live/verify-microvm-payload.live.ts | 429 ++++++++++++++++++ .../645-p2-payload-live-20260914.md | 199 ++++++++ .../645-p3-implementation-plan.md | 41 +- docs/verification/645-p3-readiness-review.md | 4 +- docs/verification/645-payload-bootstrap.md | 13 +- 5 files changed, 674 insertions(+), 12 deletions(-) create mode 100644 cdk/test/live/verify-microvm-payload.live.ts create mode 100644 docs/verification/645-p2-payload-live-20260914.md diff --git a/cdk/test/live/verify-microvm-payload.live.ts b/cdk/test/live/verify-microvm-payload.live.ts new file mode 100644 index 000000000..c15754206 --- /dev/null +++ b/cdk/test/live/verify-microvm-payload.live.ts @@ -0,0 +1,429 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +// Runs against an ALREADY deployed image; no infrastructure/permission changes. +// Every synthetic payload deliberately fails before installing configuration or +// starting a pipeline. Signed URLs stay in memory. Never persist request bodies. +import assert from 'node:assert/strict'; +import { execFile } from 'node:child_process'; +import { createHash, randomUUID } from 'node:crypto'; +import { mkdir, writeFile } from 'node:fs/promises'; +import { join } from 'node:path'; +import { setTimeout as delay } from 'node:timers/promises'; +import { parseArgs, promisify } from 'node:util'; +import { GetFunctionConfigurationCommand, LambdaClient } from '@aws-sdk/client-lambda'; +import { + GetMicrovmCommand, GetMicrovmImageCommand, LambdaMicrovmsClient, + RunMicrovmCommand, TerminateMicrovmCommand, +} from '@aws-sdk/client-lambda-microvms'; +import { DeleteObjectCommand, GetObjectCommand, PutObjectCommand, S3Client } from '@aws-sdk/client-s3'; +import { GetCallerIdentityCommand, STSClient } from '@aws-sdk/client-sts'; +import { getSignedUrl } from '@aws-sdk/s3-request-presigner'; +import { + deletePayloadReference, PAYLOAD_BOOTSTRAP, type PayloadReference, preparePayloadReference, redactPayloadUrls, +} from '../../src/handlers/shared/payload-bootstrap'; +import { makeClient } from '../../src/handlers/shared/ua'; + +const CASES = [ + 'valid-transport', 'wrong-task', 'wrong-config', 'bad-json', + 'bad-signature', 'expired-url', 'revoked-url', 'foreign-manifest', + 'wrong-path', 'bad-manifest-digest', 'large-payload', +] as const; +const command = promisify(execFile); +type Case = typeof CASES[number]; +const EXPECTED: Record = { + 'valid-transport': 'platform_config is missing or blank for required key(s)', + 'wrong-task': 'downloaded payload does not belong to the referenced task', + 'wrong-config': 'payload configuration does not match the authenticated deployment manifest', + 'bad-json': 'task payload is not valid JSON', + 'bad-signature': 'task payload download returned HTTP 403', + 'expired-url': 'task payload download returned HTTP 403', + 'revoked-url': 'task payload download returned HTTP 404', + 'foreign-manifest': 'deployment manifest read failed (AccessDenied)', + 'wrong-path': 'payload reference must sign this task\'s exact S3 object', + 'bad-manifest-digest': 'deployment manifest digest does not match its key', + 'large-payload': 'platform_config is missing or blank for required key(s)', +}; + +function safeError(error: unknown): string { + return redactPayloadUrls(error instanceof Error ? `${error.name}: ${error.message}` : String(error)); +} + +async function main(): Promise { + const { values } = parseArgs({ + options: { + 'execute': { type: 'boolean', default: false }, + 'account': { type: 'string' }, + 'region': { type: 'string' }, + 'stack': { type: 'string' }, + 'image-version': { type: 'string' }, + 'output': { type: 'string' }, + 'cases': { type: 'string', default: CASES.join(',') }, + }, + }); + const cases = values.cases!.split(',') as Case[]; + assert(cases.length > 0 && cases.every(name => CASES.includes(name)), 'Unknown probe case'); + if (!values.execute) { + process.stdout.write(`${JSON.stringify({ cases, effects: 'Synthetic S3 objects and short-lived NO_INGRESS MicroVMs; exact cleanup' })}\n`); + return; + } + for (const key of ['account', 'region', 'stack', 'image-version', 'output'] as const) { + assert(values[key], `--${key} is required with --execute`); + } + const region = values.region!; + // The production producer resolves its client region from the environment. + process.env.AWS_REGION = region; + process.env.AWS_DEFAULT_REGION = region; + const aws = async (args: string[]): Promise => { + const response = await command('aws', [...args, '--region', region, '--output', 'json'], + { timeout: 15_000, maxBuffer: 2 * 1024 * 1024 }); + return JSON.parse(response.stdout) as T; + }; + const cfg = { region, maxAttempts: 2 }; + const s3 = makeClient(S3Client, cfg); + const mv = makeClient(LambdaMicrovmsClient, cfg); + const lambda = makeClient(LambdaClient, cfg); + const sts = makeClient(STSClient, cfg); + const identity = await sts.send(new GetCallerIdentityCommand({})); + assert.equal(identity.Account, values.account, 'Wrong AWS account'); + const stack = (await aws<{ + Stacks: { + StackId?: string; StackStatus?: string; Outputs?: { OutputKey: string; OutputValue: string }[]; + }[]; + }>(['cloudformation', 'describe-stacks', '--stack-name', values.stack!])).Stacks[0]; + assert(stack, 'Stack does not exist'); + assert(stack.StackStatus?.endsWith('_COMPLETE') && !stack.StackStatus.includes('ROLLBACK'), 'Stack is not ready'); + const outputs = Object.fromEntries((stack.Outputs ?? []).map(o => [o.OutputKey!, o.OutputValue!])); + const resources = (await aws<{ + StackResourceSummaries: { + ResourceType?: string; LogicalResourceId?: string; PhysicalResourceId?: string; + }[]; + }>(['cloudformation', 'list-stack-resources', '--stack-name', values.stack!])).StackResourceSummaries; + const fn = resources.find(r => r.ResourceType === 'AWS::Lambda::Function' + && r.LogicalResourceId?.startsWith('TaskOrchestratorOrchestratorFn'))?.PhysicalResourceId; + assert(fn, 'Cannot identify this stack’s coordinator'); + const env = (await lambda.send(new GetFunctionConfigurationCommand({ FunctionName: fn }))).Environment?.Variables ?? {}; + const bucket = env.MICROVM_PAYLOAD_BUCKET; + const image = env.MICROVM_IMAGE_IDENTIFIER; + const executionRole = env.MICROVM_EXECUTION_ROLE_ARN; + const ingress = env.MICROVM_INGRESS_CONNECTOR_ARNS?.split(',') ?? []; + const egress = env.MICROVM_EGRESS_CONNECTOR_ARNS?.split(',') ?? []; + const logGroup = outputs.MicrovmLogGroupName; + const foreignBucket = outputs.MicrovmArtifactBucketName; + assert(bucket && image && executionRole && logGroup && foreignBucket && egress.length, 'Missing MicroVM settings'); + assert.equal(executionRole, outputs.MicrovmExecutionRoleArn); + assert.equal(ingress.length, 1); + assert(ingress[0]!.endsWith(':NO_INGRESS'), 'Probe requires deployed NO_INGRESS'); + const imageInfo = await mv.send(new GetMicrovmImageCommand({ imageIdentifier: image })); + assert.equal(imageInfo.latestActiveImageVersion, values['image-version'], 'Unexpected active image'); + + const runId = `p2-bootstrap-${randomUUID()}`; + const directory = values.output!; + await mkdir(directory, { recursive: false, mode: 0o700 }); + const results: Record[] = []; + const owned = new Map(); + const active = new Set(); + const vmIds: string[] = []; + const taskIds: string[] = []; + const config = { log_group_name: runId }; // valid key, deliberately incomplete configuration + const manifestBody = JSON.stringify({ backend: 'lambda-microvm', platform_config: config, version: PAYLOAD_BOOTSTRAP.version }); + const expectedManifestKey = `${PAYLOAD_BOOTSTRAP.manifest_prefix}${createHash('sha256').update(manifestBody).digest('hex')}.json`; + const remember = (b: string, key: string) => owned.set(`${b}/${key}`, { bucket: b, key }); + const report = async (row: Record) => { + results.push(row); + process.stdout.write(`${JSON.stringify(row)}\n`); + await writeFile(join(directory, 'results.json'), JSON.stringify(results, null, 2), { mode: 0o600 }); + }; + const read = async (b: string, key: string) => { + const response = await s3.send(new GetObjectCommand({ Bucket: b, Key: key })); + assert(response.Body); + return { body: await response.Body.transformToString(), etag: response.ETag, requestId: response.$metadata.requestId }; + }; + const replace = async (key: string, body: string) => { + assert(owned.has(`${bucket}/${key}`), 'Refuse to change an unowned object'); + const current = await read(bucket, key); + await s3.send(new PutObjectCommand({ Bucket: bucket, Key: key, Body: body, IfMatch: current.etag })); + }; + const stop = async (id: string) => { + const initial = await mv.send(new GetMicrovmCommand({ microvmIdentifier: id })); + if (initial.state === 'TERMINATED') { + active.delete(id); + return; + } + if (initial.state !== 'TERMINATING') { + await mv.send(new TerminateMicrovmCommand({ microvmIdentifier: id })); + } + for (let i = 0; i < 30; i++) { + const state = await mv.send(new GetMicrovmCommand({ microvmIdentifier: id })); + if (state.state === 'TERMINATED') { + active.delete(id); + return; + } + await delay(2_000); + } + throw new Error(`Termination not confirmed for ${id}`); + }; + const awaitDiagnostic = async (id: string, started: number, expected: string, status: number) => { + const day = new Date(started).toISOString().slice(0, 10).replaceAll('-', '/'); + const prefix = `${day}[${values['image-version']}]${id}`; + for (let i = 0; i < 45; i++) { + // Reuse the installed AWS CLI for logs; this live test adds no runtime SDK dependency. + const page = await aws<{ events?: { message?: string }[] }>([ + 'logs', 'filter-log-events', '--log-group-name', logGroup, + '--log-stream-name-prefix', prefix, '--start-time', String(started), + '--limit', '1000', + ]); + const messages = (page.events ?? []).map(e => e.message ?? ''); + assert(!messages.some(m => /X-Amz-(?:Signature|Credential|Security-Token)=/.test(m)), 'Bearer URL leaked to worker logs'); + assert(!messages.some(m => m.includes('hook accepted task_id=') || m.includes('installed platform_config env')), + 'Probe unexpectedly installed configuration or started a pipeline'); + if (messages.some(m => m.includes(expected)) + && messages.some(m => m.includes(`/run HTTP/1.1" ${status}`))) { + await writeFile(join(directory, `${id}.log`), messages.map(redactPayloadUrls).join(''), { mode: 0o600 }); + return messages.filter(m => m.includes('/run hook') || m.includes('/run HTTP/1.1')) + .map(m => redactPayloadUrls(m.trim())); + } + await delay(2_000); + } + throw new Error(`Expected worker diagnostic not observed for ${id}`); + }; + + await writeFile(join(directory, 'context.json'), JSON.stringify({ + runId, + account: identity.Account, + caller: identity.Arn, + region, + stackId: stack.StackId, + image, + imageVersion: values['image-version'], + executionRole, + ingress, + egress, + bucket, + maximumDurationInSeconds: 180, + cases, + // Recovery coordinates contain no signed URLs. The service lifetime bounds + // workers if this process is killed; cleanup uses only these synthetic keys. + plannedCleanupObjects: [ + { bucket, key: expectedManifestKey }, + ...cases.flatMap(name => ['payload.json', 'launch.json'].map(filename => ({ + bucket, key: `${runId}-${name}/${filename}`, + }))), + ...(cases.includes('foreign-manifest') ? [{ bucket: foreignBucket, key: expectedManifestKey }] : []), + ], + scope: 'Production producer/S3 semantics with operator credentials; actual MicroVM consumer with unchanged worker role', + }, null, 2), { mode: 0o600 }); + + let failed = false; + try { + // This is the exact canonical manifest for our one-key synthetic config. + // Record ownership before preparing, so a lost S3 reply cannot leak it. + await assert.rejects(read(bucket, expectedManifestKey), (error: unknown) => + (error as { name?: string }).name === 'NoSuchKey'); + remember(bucket, expectedManifestKey); + for (const name of cases) { + const taskId = `${runId}-${name}`; + taskIds.push(taskId); + const key = `${taskId}/payload.json`; + remember(bucket, key); + remember(bucket, `${taskId}/launch.json`); + const input = { + bucket, + taskId, + backend: 'lambda-microvm' as const, + payload: { + task_id: taskId, + description: name === 'large-payload' ? 'x'.repeat(1024 * 1024) : `Synthetic ${runId}; never start a pipeline`, + }, + platformConfig: config, + }; + // A unique config makes this run's shared manifest unique too. Discover + // its exact key from the real producer, never delete a prefix/bucket. + // Wait for BOTH writers even if one fails, before any cleanup can run. + const prepared = await Promise.allSettled([preparePayloadReference(input), preparePayloadReference(input)]); + const refs = prepared.map(result => { + if (result.status === 'rejected') throw result.reason; + return result.value; + }); + assert.deepEqual(refs[0], refs[1], 'Competing preparations produced different capabilities'); + let reference: PayloadReference = refs[0]!; + const manifestKey = new URL(reference.bootstrap_s3_uri).pathname.slice(1); + assert.equal(manifestKey, expectedManifestKey, 'Producer manifest contract changed'); + assert.deepEqual(await preparePayloadReference(input), reference, 'Replay changed the saved capability'); + const original = await read(bucket, key); + await assert.rejects(preparePayloadReference({ + ...input, payload: { ...input.payload, description: 'Conflicting synthetic instructions' }, + }), /PAYLOAD_BOOTSTRAP_CONFLICT/); + assert.equal((await read(bucket, key)).body, original.body, 'Conflict overwrote instructions'); + // Positive control for every signed object before injecting a failure. + const download = await fetch(reference.payload_url, { redirect: 'manual', signal: AbortSignal.timeout(10_000) }); + assert.equal(download.status, 200, 'Signed URL positive control failed'); + assert.equal(await download.text(), original.body); + + switch (name) { + case 'wrong-task': { + const doc = JSON.parse(original.body); + doc.task_id = `${taskId}-different`; + await replace(key, JSON.stringify(doc)); + break; + } + case 'wrong-config': { + const doc = JSON.parse(original.body); + doc.platform_config = { ...config, github_token_secret_arn: `arn:aws:secretsmanager:${region}:${identity.Account}:secret:synthetic-other-workspace` }; + await replace(key, JSON.stringify(doc)); + break; + } + case 'bad-json': + await replace(key, '{ invalid synthetic JSON'); + break; + case 'bad-signature': { + const url = new URL(reference.payload_url); + const signature = url.searchParams.get('X-Amz-Signature')!; + url.searchParams.set('X-Amz-Signature', `${signature[0] === '0' ? '1' : '0'}${signature.slice(1)}`); + reference = { ...reference, payload_url: url.toString() }; + break; + } + case 'wrong-path': { + const url = new URL(reference.payload_url); + url.pathname = url.pathname.replace(/\/payload\.json$/, '/different.json'); + reference = { ...reference, payload_url: url.toString() }; + break; + } + case 'bad-manifest-digest': + // JSON meaning stays identical; the fingerprint must reject different bytes. + await replace(manifestKey, `${manifestBody} `); + break; + case 'expired-url': { + const url = await getSignedUrl(s3, new GetObjectCommand({ Bucket: bucket, Key: key }), { expiresIn: 1 }); + await delay(2_100); + const expired = await fetch(url, { redirect: 'manual', signal: AbortSignal.timeout(10_000) }); + assert.equal(expired.status, 403); + await expired.body?.cancel(); + reference = { ...reference, payload_url: url, expires_at: Date.now() - 1 }; + break; + } + case 'revoked-url': + await deletePayloadReference(bucket, taskId); + for (const filename of ['payload.json', 'launch.json']) { + await assert.rejects(read(bucket, `${taskId}/${filename}`), (error: unknown) => + (error as { name?: string }).name === 'NoSuchKey'); + } + break; + case 'foreign-manifest': { + const manifest = await read(bucket, manifestKey); + await assert.rejects(read(foreignBucket, manifestKey), (error: unknown) => + (error as { name?: string }).name === 'NoSuchKey'); + remember(foreignBucket, manifestKey); + await s3.send(new PutObjectCommand({ + Bucket: foreignBucket, Key: manifestKey, Body: manifest.body, IfNoneMatch: '*', + })); + assert.equal((await read(foreignBucket, manifestKey)).body, manifest.body); + reference = { ...reference, bootstrap_s3_uri: `s3://${foreignBucket}/${manifestKey}` }; + break; + } + case 'valid-transport': + case 'large-payload': + break; + } + const request = { + imageIdentifier: image, + imageVersion: values['image-version'], + executionRoleArn: executionRole, + ingressNetworkConnectors: ingress, + egressNetworkConnectors: egress, + logging: { cloudWatch: { logGroup } }, + maximumDurationInSeconds: 180, + clientToken: randomUUID(), + runHookPayload: JSON.stringify(reference), + }; + assert(Buffer.byteLength(request.runHookPayload) <= 4_096, 'Reference exceeds the verified hook limit'); + const started = Date.now(); + const startedVm = await mv.send(new RunMicrovmCommand(request)); + assert(startedVm.microvmId, 'Run returned no handle; bounded service lifetime is the backstop'); + const id = startedVm.microvmId; + active.add(id); + vmIds.push(id); + await report({ + case: name, + stage: 'started', + vmId: id, + taskId, + runRequestId: startedVm.$metadata.requestId, + getObjectRequestId: original.requestId, + signedDownloadRequestId: download.headers.get('x-amz-request-id'), + storedPayloadBytes: Buffer.byteLength(original.body), + hookReferenceBytes: Buffer.byteLength(request.runHookPayload), + requestFingerprint: createHash('sha256').update(JSON.stringify(request)).digest('hex'), + }); + try { + const replay = await mv.send(new RunMicrovmCommand(request)); + if (replay.microvmId && replay.microvmId !== id) { + active.add(replay.microvmId); + vmIds.push(replay.microvmId); + } + assert.equal(replay.microvmId, id, 'Same token/request created another MicroVM'); + const status = ['valid-transport', 'large-payload', 'wrong-task', 'wrong-config', 'wrong-path'].includes(name) ? 400 : 500; + const diagnostics = await awaitDiagnostic(id, started, EXPECTED[name], status); + const observed = await mv.send(new GetMicrovmCommand({ microvmIdentifier: id })); + assert.deepEqual(observed.ingressNetworkConnectors, ingress); + await report({ + case: name, + stage: 'passed', + vmId: id, + diagnostics, + observedState: observed.state, + repeatRunSameId: true, + preparationReplayAndConflict: true, + }); + } finally { + await stop(id); + } + } + } catch (error) { + failed = true; + await report({ stage: 'failed', error: safeError(error) }); + } finally { + for (const id of [...active]) { + try { await stop(id); } catch (error) { + failed = true; + await report({ stage: 'cleanup-failed', vmId: id, error: safeError(error) }); + } + } + // Call the real finalizer helper first (revoked-url separately asserts its + // effect); operator cleanup removes only this probe's exact synthetic keys. + for (const taskId of taskIds) await deletePayloadReference(bucket, taskId); + for (const object of owned.values()) { + await s3.send(new DeleteObjectCommand({ Bucket: object.bucket, Key: object.key })); + await assert.rejects(read(object.bucket, object.key), (error: unknown) => + (error as { name?: string }).name === 'NoSuchKey'); + } + await report({ + stage: active.size ? 'cleanup-incomplete' : 'cleanup-complete', + vmIds, + activeVmIds: [...active], + deletedObjects: owned.size, + }); + } + if (failed) process.exitCode = 1; +} + +main().catch(error => { + process.stderr.write(`${safeError(error)}\n`); + process.exitCode = 1; +}); diff --git a/docs/verification/645-p2-payload-live-20260914.md b/docs/verification/645-p2-payload-live-20260914.md new file mode 100644 index 000000000..2b0e22ff9 --- /dev/null +++ b/docs/verification/645-p2-payload-live-20260914.md @@ -0,0 +1,199 @@ +# P2 live MicroVM payload verification + +## Result + +On 2026-09-14, **11 distinct cases passed against the deployed MicroVM image +`2.0`**. They verify delivery of synthetic task instructions, rejection of +invalid instructions/download references, S3 URL expiry and revocation, and +immediate identical-request replay. All **12 disposable workers terminated** +and all **29 synthetic file locations were confirmed absent** after cleanup. + +This is a subset of P2 acceptance. P3 sleep/wake remains unfinished. The +[implementation checklist](./645-p3-implementation-plan.md) tracks the remaining +work; the [bootstrap runbook](./645-payload-bootstrap.md) explains the protocol. + +## What the test does + +A worker needs two things before starting: a trusted settings file and a +temporary download ticket for its task instructions. The settings file is the +**manifest**; the ticket is a **presigned URL**. A bucket is an S3 storage +container, and an object is a file inside it. + +The [live runner](../../cdk/test/live/verify-microvm-payload.live.ts) uses the +real TypeScript producer to create both files and the saved launch reference. +It then starts a disposable worker through the real AWS `RunMicrovm` API. +The worker's `/run` hook is its startup check. + +Every case uses deliberately incomplete settings: +`{log_group_name: }`. When delivery succeeds, the startup +check reaches required-settings validation and returns HTTP **400** before +installing configuration or starting the coding pipeline. That expected +rejection is the positive control: it proves the files were fetched and +validated without making repository changes. + +The producer runs with the operator's AWS credentials. The worker uses the +deployment's unchanged execution role and runtime network connector, with +explicit `NO_INGRESS`. No shell connector, new image, IAM change, application +deployment, task row or capacity reservation is involved. + +## Deployment identity + +| Item | Value | +|---|---| +| Account / profile / Region | `` / `sphia-dev` / `us-west-2` | +| Stack | `backgroundagent-dev` | +| Deployed source | `e1d5debe`; the verifier was added on top of documentation commit `9bceb764` | +| Bootstrap policy bundle | `1.7.0` | +| Image | `backgroundagent-dev-abca-agent`, version `2.0` | +| Execution role | `backgroundagent-dev-LambdaMicrovmComputeExecutionRo-wYoHnFJEUj9S` | +| Runtime egress connector | `nc-e1601dbe-8821-45b8-bdee-3a59e7393031` | +| Ingress | AWS `NO_INGRESS` | +| Worker lifetime cap | 180 seconds per probe | +| Worker log group | `/aws/lambda-microvms/backgroundagent-dev-abca-agent` | + +The [image rebuild record](./645-microvm-image-rebuild-20260914.md) identifies +the deployed artifact and its full-build validation. + +## Observed cases + +All rows below passed the expected diagnostic and HTTP-status assertions. +HTTP 400 means the startup input was rejected; HTTP 500 here means the worker +could not read or authenticate the supplied files. These are expected test +outcomes. + +| Case | Observed startup result | HTTP | +|---|---|---| +| Valid transport | Manifest and signed task download passed; deliberately missing required settings rejected | 400 | +| Wrong task | Downloaded document did not belong to the referenced task | 400 | +| Wrong configuration | Same-account, other-workspace secret address did not match the authenticated manifest | 400 | +| Invalid JSON | Stored task bytes could not be parsed as JSON | 500 | +| Changed signature | S3 rejected the download with HTTP 403 | 500 | +| Expired URL | A real one-second URL expired; both operator control and worker received S3 HTTP 403 | 500 | +| Revoked URL | Production deletion helper removed both task files; signed download received S3 HTTP 404 | 500 | +| Foreign manifest | Same valid manifest bytes in the private artifact bucket could be read by the operator; worker received `AccessDenied` | 500 | +| Wrong download path | Reference did not sign this task's exact `payload.json` path; rejected before download | 400 | +| Corrupted manifest bytes | Appended whitespace left valid JSON but broke the SHA-256 fingerprint in its filename | 500 | +| Large task | **1,048,865 stored bytes** traveled through an **825-byte** hook reference and reached required-settings validation | 400 | + +Each case also exercised two concurrent producer preparations, a later +identical preparation, and changed instructions. The preparations returned +the same saved reference; changed instructions raised +`PAYLOAD_BOOTSTRAP_CONFLICT` without overwriting the original object. +The original signed URL returned HTTP 200 and the exact stored bytes before +fault injection. These observations exercise real S3 conditional writes with +operator credentials. + +Each passing case immediately repeated the exact `RunMicrovm` request with +the same client token. AWS returned the same worker ID. This proves immediate +identical-request replay only. + +The first foreign-manifest run correctly returned `AccessDenied`, but the +runner initially expected the generic exception name `ClientError`. That +assertion timed out and the batch exited 1 after cleanup. Boto3 exposes an +`AccessDenied` subclass here. The corrected expectation passed in a second +worker; this accounts for 12 workers across 11 distinct cases. + +## Permissions, logs and cleanup + +The live execution-role policy was read without modification. It contains: + +- `Allow s3:GetObject` on this deployment's payload-bucket `bootstrap/*`. +- `Deny s3:GetObject*` with `NotResource` set to that same prefix. +- `Deny s3:List*` on the payload bucket. + +The foreign-manifest case exercises the real worker's rejection of another +private bucket. The broader effective-permission matrix below remains open. + +The runner verifies rejection before configuration installation and pipeline +acceptance, and checks diagnostics for signed-URL credential/signature markers. +A final independent read of all 12 terminated workers' log streams found three +events per worker, no such markers, and no configuration-installed or +pipeline-accepted message. + +Cleanup uses the production task-file deletion helper, followed by operator +deletion of exact synthetic keys, including the run's unique manifests. +The revoked-URL case verifies both task files are absent **before** that +fallback. These direct probes do not exercise automatic coordinator +finalization. + +Fresh AWS reads confirmed all 12 owned worker IDs were `TERMINATED` and all +29 exact object locations returned not found. The payload bucket has +versioning disabled. The stack remained `UPDATE_COMPLETE`, with its last +update still `2026-09-14T16:07:52.421Z`; the latest active image remained `2.0`. +The test retains the intended development deployment. + +## Reproduce + +Use the repository's installed dependencies, Node 22, AWS CLI and credentials +for the intended development account. The output directory must not already +exist. The script makes no AWS calls unless `--execute` is present. + +```bash +cd cdk +mise exec -- npx tsx test/live/verify-microvm-payload.live.ts + +AWS_PROFILE=sphia-dev mise exec -- npx tsx test/live/verify-microvm-payload.live.ts \ + --execute \ + --account \ + --region us-west-2 \ + --stack backgroundagent-dev \ + --image-version 2.0 \ + --output /tmp/abca-microvm-payload-new-run +``` + +Optional `--cases` accepts comma-separated names shown by the dry run. +The script checks the actual account, stack, active image version and deployed +`NO_INGRESS` configuration before creating fixtures. Its output records worker +IDs, AWS request IDs, request fingerprints, statuses and cleanup coordinates; +signed URLs and request bodies stay in memory. + +The output directory is private. If the process is interrupted, use its +`context.json` exact `plannedCleanupObjects` and recorded worker IDs to finish +cleanup; do not delete whole buckets or prefixes. The 180-second service +lifetime bounds workers even if the process dies, but does not delete S3 files. +This is an operator-run live verifier, outside the normal Jest test pattern. + +Validation: ESLint, focused strict TypeScript checking and dry run passed. +The live batches used `--cases valid-transport`, then the seven original +negative cases, then +`--cases foreign-manifest,wrong-path,bad-manifest-digest,large-payload`. +The second batch's harness-only failure and corrected rerun are recorded above. + +## Evidence index + +Private evidence root: `/tmp/abca-645-p2-clean-20260913`. + +| Directory | Run ID | Result / cleanup | +|---|---|---| +| `payload-live-canary-20260914` | `p2-bootstrap-14632317-f022-4cb5-aad6-be45e6cb0144` | Control passed; 1 worker / 3 object locations | +| `payload-live-failures-20260914` | `p2-bootstrap-ff822345-11fc-47a7-8fb1-17bc46ed8b89` | Six assertions passed; foreign diagnostic misasserted; 7 workers / 16 locations cleaned | +| `payload-live-extended-20260914` | `p2-bootstrap-62ee9df3-eacf-4338-a43e-88999f66e3ee` | Four passed; 4 workers / 10 object locations | + +Each directory contains `context.json`, `results.json` and per-worker logs for +passing assertions. `results.json` links each case to its worker and Run/S3 +request IDs. The first foreign worker's actual rejection is separately retained +in `payload-live-foreign-investigation-logs.json` and its state file. +`payload-live-worker-s3-policy.json`, `payload-live-final-audit.json` and +`payload-live-final-log-audit.json` retain the policy and independent audits. + +For example, the passing foreign-manifest worker is +`microvm-9b27a667-8780-3439-91b0-55f8e00bf1a4`, Run request +`e0a364de-8178-4cf6-9567-004e4e6f24d3`. The large-payload worker is +`microvm-080080da-8637-3d8d-8bf7-eb65e25dface`, Run request +`fdb09b32-a2d7-4c2c-8a95-bab0fdd0be9a`. + +## Still required + +- Effective worker/session-role tests for listing, task/launch reads, + writes/deletes, public-bucket resource-policy grants, task-table restrictions + and transactions; equivalent ECS delivery and authorization cases. +- Expired **signer credentials**, coordinated upgrade/rollback, lost committed + S3 replies and coordinator restart under the deployed coordinator role. +- Lost Run replies, simultaneous or changed requests with the same token, + delayed replay/retention, recovery after process death, and unknown-ID cleanup. +- Coordinator terminal classification and automatic cleanup after rejected + startup hooks, plus injected cleanup failures and capacity migration/repair. +- Active public-ingress and port-denial attempts, remote MCP connectivity, and + end-to-end registry/tool use. The large case proves byte transport only. +- P3 guest suspend/resume hooks, safe credential refresh and durable state, + supervisor/approval integration, bounded recovery and live sleep/wake. diff --git a/docs/verification/645-p3-implementation-plan.md b/docs/verification/645-p3-implementation-plan.md index 8fde8e73d..7f0743218 100644 --- a/docs/verification/645-p3-implementation-plan.md +++ b/docs/verification/645-p3-implementation-plan.md @@ -17,7 +17,7 @@ deployed `e1d5debe` through a normal reviewed update and activated version `2.0` Its ready/validate hooks passed; repeated packaging reused the artifact, and redeploying the same cloud assembly reported no changes. The root still has 474 resources. -The [live task verification](./645-p2-live-task-20260914.md) subsequently passed +The [live task verification](./645-p2-live-task-20260914.md) on image `1.0` passed normal coding, PR iteration and cancellation on `isadeks/vercel-abca-linear`, including heartbeat, npm checks, Memory writes and automatic cleanup. The repository's [temporary verification configuration](./645-p2-repository-config-20260914.md) @@ -25,7 +25,13 @@ passed all four pre/post npm commands live, and a worker-reported failure was cleaned up. The user subsequently requested removal of the CLI addition; its live overrides were also removed. Repository mise tasks are still needed for the restored default commands. -Failure/recovery and the wider IAM/network matrix still remain. +The [image 2.0 payload verification](./645-p2-payload-live-20260914.md) passed +11 transport/failure cases, including URL expiry/revocation, a foreign-manifest +denial and a payload over 1 MiB. Concurrent/repeated preparation and immediate +identical Run replay passed; all 12 disposable workers and 29 synthetic object +locations were cleaned up. These direct probes use operator credentials for +preparation and bypass coordinator admission/finalization. +Crash/recovery and the wider IAM/network matrix still remain. Full P2 acceptance and all P3 live gates remain open. The batch notes below record what was verified at their original completion; their deployment status is superseded by these records. @@ -38,9 +44,11 @@ is superseded by these records. - [x] Exercise real S3 bad-byte paths and fix closed-stream error classification (#817). - [x] Require new ARN fields to participate in validation; pin contract fields and anchor (#817). - [x] Bind configuration to IAM-authenticated deployment manifests and use single-object payload links for ECS/MicroVM (#817 / #700). -- [ ] Verify v2 bootstrap policies, S3 conditional writes, expiry, networking and coordinated rollout in AWS. +- [x] Verify MicroVM manifest/download transport, malformed or mismatched inputs, URL expiry/revocation, a foreign private-bucket denial and >1 MiB transport in AWS; verify concurrent/repeated/conflicting S3 preparation with operator credentials. +- [ ] Complete v2 effective-role/public-bucket tests, expired signer credentials, coordinator recovery and the ECS/coordinated-rollout matrix in AWS. - [x] Implement saved MicroVM start receipts, stable tokens, input fingerprints and handle recovery. -- [ ] Verify AWS token retention/conflicts and unknown-start cleanup on a live deployment. +- [x] Verify immediate identical `RunMicrovm` replay returns the same worker ID in the live payload probes. +- [ ] Verify delayed AWS token retention, simultaneous/changed-request conflicts, lost Run replies and unknown-start cleanup on a live deployment. - [x] Make capacity acquisition/release atomic per task across crash replay; unify counter writers and repair. - [ ] Verify the capacity protocol's upgrade/drain procedure, deployed IAM and scan scale in AWS. - [x] Restrict agent task updates to reporting fields; remove replacement/deletion and worker counter grants. @@ -216,7 +224,17 @@ Keep nesting in a separate change from lifecycle logic. Developing P3 locally ne The [bootstrap runbook](./645-payload-bootstrap.md) records wire/storage shapes, caps, credential lifetime, the coordinator's `ListBucket` requirement for missing-object detection, coordinated drain/image/controller/policy upgrade and rollback, and the real AWS allow/deny matrix. Old unsigned envelopes are intentionally rejected; this is a coordinated contract change, not a rolling mixed-version deployment. -**Still required:** effective-role cross-task/public-bucket negatives, actual S3 conditional writes and missing-object behavior, signer/URL expiry, runtime DNS/HTTPS and clean launches for both backends. The role/tag and other platform-grant limits in 1G remain; this boot-path fix does not establish complete hostile-worker isolation. +**Live subset completed:** [image 2.0 probes](./645-p2-payload-live-20260914.md) +verify MicroVM manifest/download access through the runtime connector, invalid +task/config/path/bytes/signature rejection, URL expiry/revocation, foreign +private-bucket denial and >1 MiB transport. Concurrent/repeated preparation and +changed-input conflicts exercise real S3 with operator credentials. + +**Still required:** effective-role cross-task/list/write/public-bucket negatives, +expired signer credentials, lost committed replies and restart under the +coordinator role, and equivalent ECS/upgrade evidence. The role/tag and other +platform-grant limits in 1G remain; this boot-path fix does not establish complete +hostile-worker isolation. ### 1C. Fix approval/heartbeat ordering @@ -232,7 +250,11 @@ The receipt works like an order number: when the reply gets lost, the next call Fault tests exercise a successful simulated service creation followed by a lost response, a second application call with the same token, a fresh strategy instance, saved-handle replay, changed input, expired recovery, confirmed rejection, cancellation before/during/after creation, and lost DynamoDB responses. The handler recovers committed registration, treats start-audit failures as non-fatal, and routes start failures through one finalization path. Finalization reads the latest committed task, avoiding a stale cancellation/failure report. An unknown first outcome stays unknown even when the second call gets a definite rejection; HTTP 408 and named service timeouts remain uncertain even with a 4xx status. -**Still required:** the installed SDK documents `clientToken` idempotency but gives no retention period. Public AWS API documentation URLs did not provide a usable RunMicrovm reference during this review. The local 120-second limit is a conservative application cutoff, not evidence of AWS's retention window. Verify same-token replay, changed parameters, simultaneous requests/conflicts, token expiry and returned handles after termination against AWS before accepting this prerequisite. The service emulator proves client behavior only. +**Live subset completed:** every passing [payload probe](./645-p2-payload-live-20260914.md) +immediately repeated the exact Run request and received the same worker ID. +The probes do not simulate a lost reply or coordinator restart. + +**Still required:** the installed SDK documents `clientToken` idempotency but gives no retention period. Public AWS API documentation URLs did not provide a usable RunMicrovm reference during this review. The local 120-second limit is a conservative application cutoff, not evidence of AWS's retention window. Verify delayed replay, changed parameters, simultaneous requests/conflicts, token expiry and returned handles after termination against AWS before accepting this prerequisite. The service emulator proves client behavior only. For an unknown outcome with no returned ID, the task error or cancellation event identifies the saved token for investigation. Do not automatically submit a replacement task. Verify how operators find and terminate that VM in the deployed service; if they cannot recover an ID, the eight-hour lifetime cap is the remaining bound. Keep this limitation explicit in live evidence. @@ -285,8 +307,11 @@ steps 1–2 below. The [live task follow-up](./645-p2-live-task-20260914.md) provides positive runtime evidence for steps 3–5 and success/cancellation cleanup in step 6. The [configuration follow-up](./645-p2-repository-config-20260914.md) also verifies automatic pre/post npm checks and cleanup after a worker-reported -delivery failure. Crash/rejected-hook/cleanup-error paths, recovery and negative -IAM/network cases remain unverified. +delivery failure. The [image 2.0 payload follow-up](./645-p2-payload-live-20260914.md) +verifies direct startup-hook transport/rejections, immediate Run replay and +operator cleanup. Coordinator classification/finalization after rejected hooks, +crash/recovery/cleanup-error paths and the wider negative IAM/network matrix +remain unverified. `isadeks/vercel-abca-linear` explicitly selects `lambda-microvm`; the original seeded repository retains AgentCore. diff --git a/docs/verification/645-p3-readiness-review.md b/docs/verification/645-p3-readiness-review.md index 718c2b046..2921e1f1c 100644 --- a/docs/verification/645-p3-readiness-review.md +++ b/docs/verification/645-p3-readiness-review.md @@ -4,9 +4,9 @@ Reviewed 2026-09-13 against `main` commit `5e10038c7e28179b302ac4de78b709795aeba This review covers the existing MicroVM implementation, related P2 follow-ups, comment accuracy and nested-stack feasibility. It includes local experiments, not an AWS deployment or a new live smoke run. The associated [implementation plan](./645-p3-implementation-plan.md) turns the findings into ordered work. -**Live update (2026-09-14):** subsequent work completed a [clean deployment](./645-p2-clean-deployment-20260913.md) and [real coding, PR iteration and cancellation tests](./645-p2-live-task-20260914.md), including Memory writes and runtime logging. Those records supersede the clean-deployment/stdout-ingestion gaps in the historical notes below. Full P2 acceptance and integrated P3 sleep/wake remain open; the original review findings are retained as a dated baseline. +**Live update (2026-09-14):** subsequent work completed a [clean deployment](./645-p2-clean-deployment-20260913.md) and [real coding, PR iteration and cancellation tests](./645-p2-live-task-20260914.md), including Memory writes and runtime logging. A normal [image rebuild](./645-microvm-image-rebuild-20260914.md) activated version `2.0`; [11 live payload cases](./645-p2-payload-live-20260914.md) then verified transport/rejection, URL expiry/revocation and immediate Run replay. Those records supersede the corresponding gaps in the historical notes below. Full P2 acceptance and integrated P3 sleep/wake remain open; the original review findings are retained as a dated baseline. -**Implementation update (2026-09-13):** subsequent local batches fix thread isolation, deletion/error/byte/contract bugs, approval heartbeat, stable MicroVM start recovery, atomic capacity reservations and coordinator metadata permissions. The latest batch implements v2 authenticated deployment manifests and single-object payload links for both ECS and MicroVM (#817/#700), with no old unsigned fallback. See the [bootstrap runbook](./645-payload-bootstrap.md) and [implementation progress](./645-p3-implementation-plan.md#implementation-progress). A further local batch removes unused logging counters in favor of structured stdout failures and verifies large registry assets through v2 delivery and the local loader. Effective AWS policies, expiry/networking, stdout ingestion, remote-tool connectivity, clean deployment and P3 sleep/wake remain open. Findings below preserve the original reviewed baseline, rather than describing all of them as current defects. +**Implementation update (2026-09-13):** subsequent local batches fix thread isolation, deletion/error/byte/contract bugs, approval heartbeat, stable MicroVM start recovery, atomic capacity reservations and coordinator metadata permissions. The latest batch implements v2 authenticated deployment manifests and single-object payload links for both ECS and MicroVM (#817/#700), with no old unsigned fallback. See the [bootstrap runbook](./645-payload-bootstrap.md) and [implementation progress](./645-p3-implementation-plan.md#implementation-progress). A further local batch removes unused logging counters in favor of structured stdout failures and verifies large registry assets through v2 delivery and the local loader. At that batch's completion, effective AWS policies, expiry/networking, stdout ingestion, remote-tool connectivity, clean deployment and P3 sleep/wake were pending; the live update above records later evidence. Findings below preserve the original reviewed baseline, rather than describing all of them as current defects. Further local work adds durable MicroVM start receipts, stable request tokens and bounded handle recovery. AWS token-retention/conflict behavior and unknown-ID cleanup remain live verification gates. The shared finalizer's repeated-decrement risk was subsequently reproduced and fixed locally with task-owned reservation transactions, unified counter writers and revision-guarded reconciliation, including approval waits. DynamoDB Local exercises the real transaction conditions; deployed IAM, scan scale and upgrade/drain behavior still need AWS verification. diff --git a/docs/verification/645-payload-bootstrap.md b/docs/verification/645-payload-bootstrap.md index e8e0cc4c2..04672b33b 100644 --- a/docs/verification/645-payload-bootstrap.md +++ b/docs/verification/645-payload-bootstrap.md @@ -1,6 +1,6 @@ # #645: trusted task delivery for ECS and MicroVM -Implementation date: 2026-09-13. Tracks the local prerequisites in #817 and #700. **Implemented locally; AWS authorization, networking, expiry and deployment checks below are still pending.** This does not implement P3 sleep/wake. +Implementation date: 2026-09-13. Tracks the prerequisites in #817 and #700. **Deployed for MicroVM; 11 live transport/failure cases passed on 2026-09-14.** The [live evidence](./645-p2-payload-live-20260914.md) covers real worker reads/rejections, URL expiry/revocation, operator-side conditional preparation and immediate Run replay. The broader authorization/recovery matrix and ECS rollout below remain open. This does not implement P3 sleep/wake. ## What changed, in plain language @@ -61,7 +61,7 @@ AWS documents this in [GetObject permissions](https://docs.aws.amazon.com/Amazon The worker reads the manifest through the attributed `platform_client("s3")` before installing configuration. It then downloads the payload over HTTPS without adding worker credentials. Only the exact bucket/task key on a regional S3 host is accepted; redirects, environment proxies, alternate hosts, credentials in URLs, custom ports, duplicate query parameters and non-HTTPS URLs are rejected. AWS validates the actual signature and expiry; local URL checks are not a cryptographic verifier. -The runtime still needs HTTPS/443 and DNS access to the selected S3 region. The source checks do not establish that the deployed connector permits it. Build hooks remain AWS-silent; before configuration installation, runtime bootstrap uses manifest-read and payload-download operations, and diagnostics go to stdout. +The runtime needs HTTPS/443 and DNS access to the selected S3 region. The live MicroVM probes verified manifest and signed-task downloads through the deployed runtime connector, including a payload over 1 MiB. Other connectors, ECS and remote-tool connectivity still need their own evidence. Build hooks remain AWS-silent; before configuration installation, runtime bootstrap uses manifest-read and payload-download operations, and diagnostics go to stdout. ## Retry, expiry and cleanup @@ -106,6 +106,15 @@ The normal stack already creates distinct payload buckets for the selected backe For **both ECS and MicroVM**, retain source/image identifiers, relevant policy snippets, sanitized error codes, AWS request IDs and timing. Never retain signed URLs, credentials or downloaded customer prompts in evidence. +**Partial completion, 2026-09-14:** [11 live MicroVM cases](./645-p2-payload-live-20260914.md) +verify own-manifest/download access, task/config/path checks, bad bytes and +signature, URL expiry/revocation, another private bucket's manifest denial, +manifest digest validation and large-file transport. Every passing case also +checks concurrent/repeated producer preparation, changed-input conflict and +immediate identical Run replay. Producer calls use operator credentials; +direct Runs bypass coordinator admission/finalization. The table remains the +full acceptance target, including combinations those probes do not cover. + | Check | Expected result | |---|---| | New task with no stored launch | Coordinator gets `NoSuchKey`, creates immutable payload/reference, worker starts successfully. | From 770df63f41134a79a6b189382c9da4bd23561f03 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Mon, 14 Sep 2026 13:53:50 -0400 Subject: [PATCH 032/149] test(microvm): verify live start replay and recovery --- agent/src/server.py | 19 +- .../strategies/lambda-microvm-strategy.ts | 16 +- cdk/test/live/microvm-live-support.ts | 108 ++++++ cdk/test/live/microvm-start-child.ts | 128 +++++++ cdk/test/live/verify-microvm-replay.live.ts | 172 ++++++++++ cdk/test/live/verify-microvm-start.live.ts | 323 ++++++++++++++++++ .../645-p2-start-recovery-live-20260914.md | 209 ++++++++++++ .../645-p3-implementation-plan.md | 37 +- docs/verification/645-payload-bootstrap.md | 6 + 9 files changed, 990 insertions(+), 28 deletions(-) create mode 100644 cdk/test/live/microvm-live-support.ts create mode 100644 cdk/test/live/microvm-start-child.ts create mode 100644 cdk/test/live/verify-microvm-replay.live.ts create mode 100644 cdk/test/live/verify-microvm-start.live.ts create mode 100644 docs/verification/645-p2-start-recovery-live-20260914.md diff --git a/agent/src/server.py b/agent/src/server.py index 79da35a20..a47784941 100644 --- a/agent/src/server.py +++ b/agent/src/server.py @@ -1150,10 +1150,10 @@ def _install_platform_config(raw: Any) -> list[str]: Returns the sorted env var names actually installed. Rules, all deliberate: - * ``None`` / absent → install nothing and return ``[]``. This is the P1 - envelope (no ``platform_config`` sibling), where the snapshot's own env is - all there is; a MicroVM image can be launched by an orchestrator that - predates Stage B, and the two deploy on independent cadences. + * ``None`` → install nothing and return ``[]`` for direct helper callers. + The v2 ``/run`` path requires a dictionary from the authenticated manifest + before calling this helper. This no-op does not accept legacy unsigned + envelopes or permit an independent coordinator/image contract rollout. * present but not an object, or carrying ANY key outside the allowlist, or carrying a non-string value, or carrying a control character in a value → reject the run (``…_INVALID``). Unknown keys are an env-injection attempt, @@ -1161,13 +1161,12 @@ def _install_platform_config(raw: Any) -> list[str]: block is refused rather than filtered. Control characters are refused for the reason in ``_PLATFORM_CONFIG_FORBIDDEN_VALUE_CHARS``. * ``None`` / blank / whitespace-only values are treated as ABSENT, not as an - instruction to clear the variable: the natural TypeScript producer - (``process.env.X ?? ''``) emits an empty string for a resource the - deployment does not have, and clobbering an image value with ``""`` would - turn "not configured over there" into "unconfigured here". + instruction to clear the variable. The current TypeScript producer omits + unconfigured values; this helper also filters explicit empty values + instead of installing them into the environment. * every required key must survive that filter, else reject - (``…_INCOMPLETE``). An explicitly-sent-but-empty ``{}`` therefore fails — - a producer with nothing to say must omit the key entirely. + (``…_INCOMPLETE``). An explicitly-sent-but-empty ``{}`` therefore fails; + the live v2 boot path always requires the complete required subset. * every ARN-shaped value must agree with the anchor ARN's partition + account, else reject (``…_INVALID``) — see :func:`_reject_foreign_arns`, which is explicit that this is internal consistency plus fail-fast, not an diff --git a/cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts b/cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts index a924844a0..158c5ffbb 100644 --- a/cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts +++ b/cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts @@ -176,9 +176,10 @@ const MICROVM_BENIGN_STATE_REASON = 'Success.'; * ## What may and may not go in here * * NON-SECRET IDENTIFIERS ONLY — table names, bucket names, log-group names, and - * secret/role **ARNs**. Never a token, never a secret *value*: the envelope is - * written to an S3 object and echoed into MicroVM logs on a hook failure, and - * the agent resolves an ARN itself through its own (SessionRole / + * secret/role **ARNs**. Never a token, never a secret *value*: configuration is + * stored in the worker-readable deployment manifest and task object. Hook + * diagnostics omit payloads and redact signed URLs. The agent resolves an ARN + * itself through its own (SessionRole / * execution-role) credentials. The producer below is a map over exactly the * contract's keys, so a value can only reach the wire by being added to the * contract — an unrelated `process.env` entry (`GITHUB_TOKEN`, @@ -579,11 +580,10 @@ export class LambdaMicrovmComputeStrategy implements ComputeStrategy { const { microvmId, endpoint } = result; if (!microvmId || !endpoint) { - // A malformed response means a MicroVM may ALREADY BE RUNNING (and billing) - // that no caller will ever receive a handle for — nothing self-terminates on - // this substrate. Reap it here, best-effort, before failing: this is the one - // orphan window the orchestrator's own catch cannot cover, because - // `startSession` never returned a handle to it. + // A malformed response may describe an existing VM without a usable handle. + // The eight-hour service lifetime is a backstop, not prompt cleanup. + // Clean up any available ID here: the caller will not receive a handle + // it can use to stop this VM. if (microvmId) { await this.terminateBestEffort(microvmId, 'incomplete RunMicrovm response'); } diff --git a/cdk/test/live/microvm-live-support.ts b/cdk/test/live/microvm-live-support.ts new file mode 100644 index 000000000..5f37db0ce --- /dev/null +++ b/cdk/test/live/microvm-live-support.ts @@ -0,0 +1,108 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import assert from 'node:assert/strict'; +import { execFile } from 'node:child_process'; +import { setTimeout as delay } from 'node:timers/promises'; +import { promisify } from 'node:util'; +import { GetFunctionConfigurationCommand, LambdaClient } from '@aws-sdk/client-lambda'; +import { + GetMicrovmCommand, GetMicrovmImageCommand, LambdaMicrovmsClient, + ListMicrovmsCommand, TerminateMicrovmCommand, +} from '@aws-sdk/client-lambda-microvms'; +import { GetCallerIdentityCommand, STSClient } from '@aws-sdk/client-sts'; +import { redactPayloadUrls } from '../../src/handlers/shared/payload-bootstrap'; +import { makeClient } from '../../src/handlers/shared/ua'; + +export interface LiveTarget { + account: string; + region: string; + stack: string; + imageVersion: string; +} + +const command = promisify(execFile); + +export function safeError(error: unknown): string { + return redactPayloadUrls(error instanceof Error ? `${error.name}: ${error.message}` : String(error)); +} + +export async function connectMicrovm(target: LiveTarget) { + process.env.AWS_REGION = target.region; + process.env.AWS_DEFAULT_REGION = target.region; + const aws = async (args: string[]): Promise => { + const response = await command('aws', [...args, '--region', target.region, '--output', 'json'], { + timeout: 20_000, maxBuffer: 2 * 1024 * 1024, + }); + return JSON.parse(response.stdout) as T; + }; + // A single SDK attempt exposes the actual response to each probe. + const cfg = { region: target.region, maxAttempts: 1 }; + const identity = await makeClient(STSClient, cfg).send(new GetCallerIdentityCommand({})); + assert.equal(identity.Account, target.account, 'Wrong AWS account'); + const stack = (await aws<{ + Stacks: { StackId: string; StackStatus: string; Outputs: { OutputKey: string; OutputValue: string }[] }[]; + }>(['cloudformation', 'describe-stacks', '--stack-name', target.stack])).Stacks[0]!; + assert(['CREATE_COMPLETE', 'UPDATE_COMPLETE'].includes(stack.StackStatus), 'Stack is not ready'); + const outputs = Object.fromEntries(stack.Outputs.map(o => [o.OutputKey, o.OutputValue])); + const resources = (await aws<{ + StackResourceSummaries: { ResourceType: string; LogicalResourceId: string; PhysicalResourceId: string }[]; + }>(['cloudformation', 'list-stack-resources', '--stack-name', target.stack])).StackResourceSummaries; + const fn = resources.find(r => r.ResourceType === 'AWS::Lambda::Function' + && r.LogicalResourceId.startsWith('TaskOrchestratorOrchestratorFn'))?.PhysicalResourceId; + assert(fn, 'Cannot identify coordinator'); + const configuration = await makeClient(LambdaClient, cfg).send(new GetFunctionConfigurationCommand({ FunctionName: fn })); + const env = configuration.Environment?.Variables ?? {}; + const image = env.MICROVM_IMAGE_IDENTIFIER; + const executionRole = env.MICROVM_EXECUTION_ROLE_ARN; + const ingress = env.MICROVM_INGRESS_CONNECTOR_ARNS?.split(',') ?? []; + const egress = env.MICROVM_EGRESS_CONNECTOR_ARNS?.split(',') ?? []; + const logGroup = outputs.MicrovmLogGroupName; + assert(image && executionRole && logGroup && egress.length, 'Missing MicroVM configuration'); + assert.equal(executionRole, outputs.MicrovmExecutionRoleArn); + assert.equal(ingress.length, 1); + assert(ingress[0]!.endsWith(':NO_INGRESS'), 'Live probes require NO_INGRESS'); + const mv = makeClient(LambdaMicrovmsClient, cfg); + const info = await mv.send(new GetMicrovmImageCommand({ imageIdentifier: image })); + assert.equal(info.latestActiveImageVersion, target.imageVersion, 'Unexpected active image'); + return { aws, mv, image, executionRole, ingress, egress, logGroup, env, stackId: stack.StackId }; +} + +export async function listWorkers(mv: LambdaMicrovmsClient, image: string) { + const workers = []; + let nextToken: string | undefined; + do { + const page = await mv.send(new ListMicrovmsCommand({ imageIdentifier: image, nextToken, maxResults: 50 })); + workers.push(...(page.items ?? [])); + nextToken = page.nextToken; + } while (nextToken); + return workers; +} + +export async function stopWorker(mv: LambdaMicrovmsClient, id: string): Promise { + const first = await mv.send(new GetMicrovmCommand({ microvmIdentifier: id })); + if (first.state === 'TERMINATED') return; + if (first.state !== 'TERMINATING') await mv.send(new TerminateMicrovmCommand({ microvmIdentifier: id })); + for (let attempt = 0; attempt < 30; attempt++) { + const state = await mv.send(new GetMicrovmCommand({ microvmIdentifier: id })); + if (state.state === 'TERMINATED') return; + await delay(2_000); + } + throw new Error(`Termination not confirmed for ${id}`); +} diff --git a/cdk/test/live/microvm-start-child.ts b/cdk/test/live/microvm-start-child.ts new file mode 100644 index 000000000..715bfa179 --- /dev/null +++ b/cdk/test/live/microvm-start-child.ts @@ -0,0 +1,128 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import assert from 'node:assert/strict'; +import { createHash } from 'node:crypto'; +import { appendFile } from 'node:fs/promises'; +import { LambdaMicrovmsClient, RunMicrovmCommand, type RunMicrovmCommandOutput } from '@aws-sdk/client-lambda-microvms'; +import { PutObjectCommand, S3Client } from '@aws-sdk/client-s3'; +import { DynamoDBDocumentClient, UpdateCommand } from '@aws-sdk/lib-dynamodb'; +import { safeError } from './microvm-live-support'; + +export interface ProbeContext { + taskId: string; + userId: string; + environment: Record; + manifestKey: string; + logGroup: string; +} +export interface Trace { + kind: string; + id?: string; + fingerprint?: string; + error?: string; + autoRetried?: boolean; + [key: string]: unknown; +} + +/** SDK fault injection is confined to a fresh local process, never the deployment. */ +export async function runStartChild(context: ProbeContext, fault: string, action: string, traceFile: string): Promise { + Object.assign(process.env, context.environment); + const trace = async (row: Trace) => appendFile(traceFile, `${JSON.stringify(row)}\n`, { mode: 0o600 }); + let injected = false; + const inject = async (point: string) => { + if (injected || ![`lost-${point}-reply`, `crash-after-${point}`].includes(fault)) return; + injected = true; + await trace({ kind: 'fault', point, fault }); + if (fault.startsWith('crash-')) process.exit(73); + throw Object.assign(new Error(`Synthetic lost ${point} response after AWS success`), { name: 'TimeoutError' }); + }; + // eslint-disable-next-line @typescript-eslint/unbound-method -- Reflect.apply supplies each actual client as this. + const originalVmSend = LambdaMicrovmsClient.prototype.send; + Object.defineProperty(LambdaMicrovmsClient.prototype, 'send', { + value: async function(this: LambdaMicrovmsClient, command: RunMicrovmCommand) { + if (!(command instanceof RunMicrovmCommand)) return Reflect.apply(originalVmSend, this, [command]); + // The deterministic test transport uses 180s instead of 8h and adds logs. + // Production hashing/receipt/payload/retry logic remains unchanged. + command.input.maximumDurationInSeconds = 180; + command.input.logging = { cloudWatch: { logGroup: context.logGroup } }; + const fingerprint = createHash('sha256').update(JSON.stringify(command.input)).digest('hex'); + await trace({ kind: 'run-request', fingerprint, clientToken: command.input.clientToken }); + const response = await Reflect.apply(originalVmSend, this, [command]) as RunMicrovmCommandOutput; + await trace({ kind: 'run-success', id: response.microvmId, fingerprint, requestId: response.$metadata.requestId }); + await inject('run'); + return response; + }, + }); + // eslint-disable-next-line @typescript-eslint/unbound-method -- Reflect.apply supplies each actual client as this. + const originalS3Send = S3Client.prototype.send; + Object.defineProperty(S3Client.prototype, 'send', { + value: async function(this: S3Client, command: PutObjectCommand) { + if (!(command instanceof PutObjectCommand)) return Reflect.apply(originalS3Send, this, [command]); + assert.equal(command.input.Bucket, context.environment.MICROVM_PAYLOAD_BUCKET); + const key = command.input.Key!; + assert([context.manifestKey, `${context.taskId}/payload.json`, `${context.taskId}/launch.json`].includes(key)); + const response = await Reflect.apply(originalS3Send, this, [command]); + await trace({ kind: 's3-committed', key }); + if (key.endsWith('/payload.json')) await inject('payload'); + if (key.endsWith('/launch.json')) await inject('launch'); + return response; + }, + }); + // eslint-disable-next-line @typescript-eslint/unbound-method -- Reflect.apply supplies each actual client as this. + const originalDdbSend = DynamoDBDocumentClient.prototype.send; + Object.defineProperty(DynamoDBDocumentClient.prototype, 'send', { + value: async function(this: DynamoDBDocumentClient, command: UpdateCommand) { + const handleWrite = command instanceof UpdateCommand && command.input.UpdateExpression?.includes('microvm_start.#handle'); + if (handleWrite && fault === 'handle-write-rejected') { + await trace({ kind: 'fault', point: 'before-handle-write', fault }); + throw Object.assign(new Error('Synthetic denied handle write'), { name: 'AccessDeniedException' }); + } + const response = await Reflect.apply(originalDdbSend, this, [command]); + if (handleWrite) { + await trace({ kind: 'handle-committed' }); + await inject('handle'); + } + return response; + }, + }); + // These production modules capture their environment at import time. + const { LambdaMicrovmComputeStrategy } = await import('../../src/handlers/shared/strategies/lambda-microvm-strategy.js'); + const { startSessionWithRetry } = await import('../../src/handlers/shared/session-start-retry.js'); + const input = { + taskId: context.taskId, + userId: context.userId, + payload: { task_id: context.taskId, description: action === 'changed' ? 'Changed synthetic instructions' : 'Synthetic recovery probe' }, + blueprintConfig: { compute_type: 'lambda-microvm' as const, runtime_arn: '' }, + }; + try { + const strategy = new LambdaMicrovmComputeStrategy(); + const result = action === 'auto-retry' + ? await startSessionWithRetry(strategy, input, { + taskId: context.taskId, + emitRetryEvent: async () => { await trace({ kind: 'retry-event' }); }, + logger: { warn: () => { /* Retry event and final result are captured above/below. */ } }, + }) + : { handle: await strategy.startSession(input), autoRetried: false }; + await trace({ kind: 'result', id: result.handle.sessionId, autoRetried: result.autoRetried }); + } catch (error) { + await trace({ kind: 'application-error', error: safeError(error) }); + process.exitCode = 1; + } +} diff --git a/cdk/test/live/verify-microvm-replay.live.ts b/cdk/test/live/verify-microvm-replay.live.ts new file mode 100644 index 000000000..f0d4f80f1 --- /dev/null +++ b/cdk/test/live/verify-microvm-replay.live.ts @@ -0,0 +1,172 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import assert from 'node:assert/strict'; +import { createHash, randomUUID } from 'node:crypto'; +import { mkdir, writeFile } from 'node:fs/promises'; +import { join } from 'node:path'; +import { setTimeout as delay } from 'node:timers/promises'; +import { parseArgs } from 'node:util'; +import { GetMicrovmCommand, RunMicrovmCommand, type RunMicrovmCommandInput } from '@aws-sdk/client-lambda-microvms'; +import { connectMicrovm, listWorkers, safeError, stopWorker } from './microvm-live-support'; + +async function main(): Promise { + const { values } = parseArgs({ + options: { + 'execute': { type: 'boolean', default: false }, + 'account': { type: 'string' }, + 'region': { type: 'string' }, + 'stack': { type: 'string' }, + 'image-version': { type: 'string' }, + 'output': { type: 'string' }, + }, + }); + if (!values.execute) { + process.stdout.write(`${JSON.stringify({ + cases: ['simultaneous-identical', 'changed-parameters', 'replay-after-termination', 'replay-at-30-130-305-seconds'], + effects: 'Disposable NO_INGRESS workers with rejected startup input; no S3/task rows; maximum lifetime 180 seconds', + })}\n`); + return; + } + for (const key of ['account', 'region', 'stack', 'image-version', 'output'] as const) assert(values[key], `--${key} is required`); + const target = { account: values.account!, region: values.region!, stack: values.stack!, imageVersion: values['image-version']! }; + const live = await connectMicrovm(target); + const directory = values.output!; + await mkdir(directory, { mode: 0o700 }); + const before = await listWorkers(live.mv, live.image); + const runId = `p2-replay-${randomUUID()}`; + const request: RunMicrovmCommandInput = { + imageIdentifier: live.image, + imageVersion: target.imageVersion, + executionRoleArn: live.executionRole, + ingressNetworkConnectors: live.ingress, + egressNetworkConnectors: live.egress, + logging: { cloudWatch: { logGroup: live.logGroup } }, + // Missing v2 reference: guaranteed rejection before S3/config/pipeline. + runHookPayload: JSON.stringify({ probe: runId }), + maximumDurationInSeconds: 180, + clientToken: runId, + }; + await writeFile(join(directory, 'context.json'), JSON.stringify({ + ...target, + runId, + stackId: live.stackId, + request, + beforeIds: before.map(w => w.microvmId), + scope: 'Direct service idempotency; no coordinator or application retry code', + }, null, 2), { mode: 0o600 }); + const results: Record[] = []; + const ids = new Set(); + const started = Date.now(); + const report = async (row: Record) => { + const timed = { ...row, elapsedMs: Date.now() - started }; + results.push(timed); + process.stdout.write(`${JSON.stringify(timed)}\n`); + await writeFile(join(directory, 'results.json'), JSON.stringify(results, null, 2), { mode: 0o600 }); + }; + const launch = async (label: string, input = request) => { + try { + const result = await live.mv.send(new RunMicrovmCommand(input)); + if (result.microvmId) ids.add(result.microvmId); + await report({ + label, + outcome: 'accepted', + id: result.microvmId, + state: result.state, + requestId: result.$metadata.requestId, + duration: result.maximumDurationInSeconds, + requestFingerprint: createHash('sha256').update(JSON.stringify(input)).digest('hex'), + }); + assert(result.microvmId, 'Run returned no handle'); + return result.microvmId; + } catch (error) { + await report({ + label, + outcome: 'error', + error: safeError(error), + requestId: (error as { $metadata?: { requestId?: string } }).$metadata?.requestId, + }); + throw error; + } + }; + try { + const simultaneous = await Promise.allSettled([launch('simultaneous-a'), launch('simultaneous-b')]); + const successes = simultaneous.filter(r => r.status === 'fulfilled'); + assert(successes.length, 'Neither simultaneous request succeeded'); + assert.equal(ids.size, 1, 'Identical requests created different workers'); + for (const rejected of simultaneous.filter(r => r.status === 'rejected')) { + assert.equal((rejected.reason as { name: string }).name, 'ConflictException', 'Unexpected simultaneous failure'); + } + const id = [...ids][0]!; + assert.equal(await launch('immediate-replay'), id); + // Observe whether AWS rejects changed parameters or returns the existing + // worker. Either way it must not create a second worker for this token. + try { + assert.equal(await launch('changed-duration', { ...request, maximumDurationInSeconds: 181 }), id); + await report({ label: 'changed-parameters', outcome: 'existing-worker-returned' }); + } catch (error) { + assert.equal((error as { name: string }).name, 'ValidationException', 'Unexpected changed-request failure'); + assert.match((error as Error).message, /clientToken was used with different request parameters/); + await report({ label: 'changed-parameters', outcome: 'conflict-rejected' }); + } + const state = await live.mv.send(new GetMicrovmCommand({ microvmIdentifier: id })); + assert.equal(state.maximumDurationInSeconds, 180, 'Changed request mutated the original worker'); + // Give the intentionally invalid startup hook time to reject on its own. + for (let i = 0; i < 30; i++) { + const current = await live.mv.send(new GetMicrovmCommand({ microvmIdentifier: id })); + if (current.state === 'TERMINATED') { + assert(current.stateReason?.includes('HTTP status 400'), 'Worker did not reject the invalid startup input'); + await report({ label: 'hook-rejected', id, state: current.state, reason: current.stateReason }); + break; + } + assert(i < 29, 'Worker did not terminate after startup rejection'); + await delay(2_000); + } + assert.equal(await launch('replay-after-termination'), id); + for (const seconds of [30, 130, 305]) { + await report({ label: 'waiting-for-replay', seconds }); + while (Date.now() - started < seconds * 1_000) { + await delay(Math.min(10_000, seconds * 1_000 - (Date.now() - started))); + } + assert.equal(await launch(`replay-at-${seconds}s`), id); + } + assert.equal(ids.size, 1); + await report({ stage: 'passed', boundedRetentionEvidenceSeconds: 305, id }); + } finally { + const cleanup = await Promise.allSettled([...ids].map(id => stopWorker(live.mv, id))); + const after = await listWorkers(live.mv, live.image); + const unaccounted = after.filter(w => !before.some(b => b.microvmId === w.microvmId) && !ids.has(w.microvmId!)); + await report({ + stage: 'cleanup', + knownIds: [...ids], + cleanup: cleanup.map((r, i) => ({ + id: [...ids][i], confirmed: r.status === 'fulfilled', ...(r.status === 'rejected' && { error: safeError(r.reason) }), + })), + unaccountedIds: unaccounted.map(w => w.microvmId), + }); + assert(cleanup.every(r => r.status === 'fulfilled'), 'Cleanup incomplete'); + // Unknown rows can belong to a concurrent caller: report, never delete them. + assert.equal(unaccounted.length, 0, 'Another worker appeared during this probe; attribution needs review'); + } +} + +main().catch(error => { + process.stderr.write(`${safeError(error)}\n`); + process.exitCode = 1; +}); diff --git a/cdk/test/live/verify-microvm-start.live.ts b/cdk/test/live/verify-microvm-start.live.ts new file mode 100644 index 000000000..74de7a7ae --- /dev/null +++ b/cdk/test/live/verify-microvm-start.live.ts @@ -0,0 +1,323 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import assert from 'node:assert/strict'; +import { spawn } from 'node:child_process'; +import { createHash, randomUUID } from 'node:crypto'; +import { mkdir, readFile, readdir, writeFile } from 'node:fs/promises'; +import { join } from 'node:path'; +import { setTimeout as delay } from 'node:timers/promises'; +import { parseArgs } from 'node:util'; +import { GetMicrovmCommand } from '@aws-sdk/client-lambda-microvms'; +import { DeleteObjectCommand, GetObjectCommand, S3Client } from '@aws-sdk/client-s3'; +import { DeleteCommand, GetCommand, PutCommand, UpdateCommand } from '@aws-sdk/lib-dynamodb'; +import { connectMicrovm, listWorkers, safeError, stopWorker } from './microvm-live-support'; +import { runStartChild, type ProbeContext, type Trace } from './microvm-start-child'; +import constants from '../../../contracts/constants.json'; +import { TaskStatus } from '../../src/constructs/task-status'; +import { deletePayloadReference, PAYLOAD_BOOTSTRAP, redactPayloadUrls } from '../../src/handlers/shared/payload-bootstrap'; +import { makeClient, makeDocClient } from '../../src/handlers/shared/ua'; + +const CASES = [ + 'control', 'lost-run-reply', 'crash-after-run', 'crash-after-payload', 'crash-after-launch', + 'lost-payload-reply', 'lost-launch-reply', 'lost-handle-reply', 'crash-after-handle', + 'handle-write-rejected', 'changed-input', 'canceled-handle', 'expired-receipt', +] as const; +type Case = typeof CASES[number]; + +async function main(): Promise { + const { values } = parseArgs({ + options: { + 'execute': { type: 'boolean', default: false }, + 'account': { type: 'string' }, + 'region': { type: 'string' }, + 'stack': { type: 'string' }, + 'image-version': { type: 'string' }, + 'output': { type: 'string' }, + 'cases': { type: 'string', default: CASES.join(',') }, + 'child-context': { type: 'string' }, + 'fault': { type: 'string', default: 'none' }, + 'action': { type: 'string', default: 'direct' }, + 'trace': { type: 'string' }, + }, + }); + if (!values.execute) { + process.stdout.write(`${JSON.stringify({ cases: CASES, effects: 'Synthetic S3 files/task rows and NO_INGRESS workers limited to 180 seconds; local process faults against AWS' })}\n`); + return; + } + if (values['child-context']) { + assert(values.trace); + await runStartChild(JSON.parse(await readFile(values['child-context'], 'utf8')) as ProbeContext, values.fault!, values.action!, values.trace); + return; + } + for (const key of ['account', 'region', 'stack', 'image-version', 'output'] as const) assert(values[key], `--${key} is required`); + const cases = values.cases!.split(',') as Case[]; + assert(cases.every(c => CASES.includes(c)) && new Set(cases).size === cases.length); + const target = { account: values.account!, region: values.region!, stack: values.stack!, imageVersion: values['image-version']! }; + const live = await connectMicrovm(target); + const bucket = live.env.MICROVM_PAYLOAD_BUCKET!; + const table = live.env.TASK_TABLE_NAME!; + assert(bucket && table); + const s3 = makeClient(S3Client, { region: target.region }); + const ddb = makeDocClient({ region: target.region }); + const directory = values.output!; + await mkdir(directory, { mode: 0o700 }); + const runId = `p2-start-${randomUUID()}`; + const environment: Record = { + ...Object.fromEntries(Object.values(constants.microvm_platform_config.env_by_key).map(name => [name, ''])), + AWS_REGION: target.region, + AWS_DEFAULT_REGION: target.region, + AWS_MAX_ATTEMPTS: '1', + TASK_TABLE_NAME: table, + TASK_EVENTS_TABLE_NAME: `${runId}-events`, + // Nonempty for the producer, deliberately malformed for guest validation. + // The guest rejects this before installing config, fetching secrets or work. + AGENT_SESSION_ROLE_ARN: 'synthetic-invalid-role', + GITHUB_TOKEN_SECRET_ARN: `arn:aws:secretsmanager:${target.region}:${target.account}:secret:synthetic`, + LOG_GROUP_NAME: runId, + MICROVM_IMAGE_IDENTIFIER: live.image, + MICROVM_IMAGE_VERSION: target.imageVersion, + MICROVM_PAYLOAD_BUCKET: bucket, + MICROVM_EXECUTION_ROLE_ARN: live.executionRole, + MICROVM_INGRESS_CONNECTOR_ARNS: live.ingress.join(','), + MICROVM_EGRESS_CONNECTOR_ARNS: live.egress.join(','), + }; + const config = Object.fromEntries(Object.entries(constants.microvm_platform_config.env_by_key) + .filter(([, name]) => environment[name]).map(([key, name]) => [key, environment[name]])); + const canonical = JSON.stringify({ version: PAYLOAD_BOOTSTRAP.version, backend: 'lambda-microvm', platform_config: config }, + (_key, value: unknown) => value && typeof value === 'object' && !Array.isArray(value) + ? Object.fromEntries(Object.entries(value).sort(([a], [b]) => a.localeCompare(b))) : value); + const manifestKey = `${PAYLOAD_BOOTSTRAP.manifest_prefix}${createHash('sha256').update(canonical).digest('hex')}.json`; + const plans = cases.map(name => ({ name, taskId: randomUUID() })); + const before = await listWorkers(live.mv, live.image); + const readObject = async (key: string) => { + const result = await s3.send(new GetObjectCommand({ Bucket: bucket, Key: key })); + assert(result.Body); + return result.Body.transformToString(); + }; + const absent = async (key: string) => assert.rejects(readObject(key), + (error: unknown) => (error as { name?: string }).name === 'NoSuchKey'); + await absent(manifestKey); + await writeFile(join(directory, 'context.json'), JSON.stringify({ + ...target, + runId, + environment, + manifestKey, + plans, + beforeIds: before.map(w => w.microvmId), + scope: 'Production strategy/storage under operator credentials; local child faults; 180s lifetime and logging transport overrides', + }, null, 2), { mode: 0o600 }); + const results: Record[] = []; + const report = async (row: Record) => { + results.push(row); + process.stdout.write(`${JSON.stringify(row)}\n`); + await writeFile(join(directory, 'results.json'), JSON.stringify(results, null, 2), { mode: 0o600 }); + }; + const allTraces = async (): Promise => { + const traces: Trace[] = []; + for (const path of (await readdir(directory)).filter(p => p.endsWith('.jsonl'))) { + traces.push(...(await readFile(join(directory, path), 'utf8')) + .trim().split('\n').filter(Boolean).map(line => JSON.parse(line) as Trace)); + } + return traces; + }; + const task = async (id: string) => (await ddb.send(new GetCommand({ + TableName: table, Key: { task_id: id }, ConsistentRead: true, + }))).Item; + const cleanupTask = async (id: string) => { + await deletePayloadReference(bucket, id); + await absent(`${id}/payload.json`); + await absent(`${id}/launch.json`); + if (await task(id)) { + await ddb.send(new DeleteCommand({ + TableName: table, + Key: { task_id: id }, + ConditionExpression: 'user_id = :owner', + ExpressionAttributeValues: { ':owner': runId }, + })); + } + assert.equal(await task(id), undefined); + }; + let attempt = 0; + const execute = async (context: ProbeContext, fault = 'none', action = 'direct') => { + const contextFile = join(directory, `${context.taskId}.json`); + await writeFile(contextFile, JSON.stringify(context, null, 2), { mode: 0o600 }); + const traceFile = join(directory, `${context.taskId}-${++attempt}.jsonl`); + await writeFile(traceFile, '', { mode: 0o600 }); + const outcome = await new Promise<{ code: number | null; signal: string | null; output: string }>((resolve, reject) => { + // Reuse the parent's resolved loader; npx may supply tsx from its cache. + const processChild = spawn(process.execPath, [...process.execArgv, __filename, '--execute', + '--child-context', contextFile, '--fault', fault, '--action', action, '--trace', traceFile], + { stdio: ['ignore', 'pipe', 'pipe'], timeout: 90_000 }); + let output = ''; + processChild.stdout.on('data', chunk => { output += String(chunk); }); + processChild.stderr.on('data', chunk => { output += String(chunk); }); + processChild.on('error', reject); + processChild.on('close', (code, signal) => resolve({ code, signal, output })); + }); + await writeFile(`${traceFile}.log`, redactPayloadUrls(outcome.output), { mode: 0o600 }); + assert.equal(outcome.signal, null, 'Child exceeded its time budget'); + const trace = (await readFile(traceFile, 'utf8')).trim().split('\n').filter(Boolean).map(line => JSON.parse(line) as Trace); + await report({ stage: 'child', taskId: context.taskId, fault, action, code: outcome.code, trace }); + return { code: outcome.code, trace }; + }; + let failed = false; + try { + for (const { name, taskId } of plans) { + await ddb.send(new PutCommand({ + TableName: table, + Item: { + task_id: taskId, + user_id: runId, + status: TaskStatus.HYDRATING, + created_at: new Date().toISOString(), + ttl: Math.floor(Date.now() / 1000) + 3_600, + }, + ConditionExpression: 'attribute_not_exists(task_id)', + })); + const context = { taskId, userId: runId, environment, manifestKey, logGroup: live.logGroup }; + const initialFault = ['changed-input', 'expired-receipt'].includes(name) ? 'crash-after-run' + : ['control', 'canceled-handle'].includes(name) ? 'none' : name; + const first = await execute(context, initialFault, name === 'lost-run-reply' ? 'auto-retry' : 'direct'); + const traces = [...first.trace]; + if (initialFault !== 'none') assert(first.trace.some(t => t.kind === 'fault'), 'Requested fault was not injected'); + if (initialFault.startsWith('crash-')) { + assert.equal(first.code, 73, 'Process did not reach the requested crash point'); + const beforeRetry = await task(taskId); + assert(beforeRetry?.microvm_start); + const storedPayload = await readObject(`${taskId}/payload.json`); + if (name === 'changed-input') { + const changed = await execute(context, 'none', 'changed'); + traces.push(...changed.trace); + assert.equal(changed.code, 1); + assert(changed.trace.some(t => t.error?.includes('MICROVM_START_INPUT_CHANGED'))); + assert(!changed.trace.some(t => t.kind === 'run-request')); + assert.equal(await readObject(`${taskId}/payload.json`), storedPayload); + } + if (name === 'expired-receipt') { + await report({ stage: 'waiting-for-receipt-expiry', taskId, expiresAt: beforeRetry.microvm_start.expiresAt }); + while (Date.now() <= beforeRetry.microvm_start.expiresAt) await delay(2_000); + const expired = await execute(context); + traces.push(...expired.trace); + assert.equal(expired.code, 1); + assert(expired.trace.some(t => t.error?.includes('MICROVM_START_OUTCOME_UNKNOWN'))); + assert(!expired.trace.some(t => t.kind === 'run-request')); + } else { + const recovered = await execute(context); + traces.push(...recovered.trace); + assert.equal(recovered.code, 0); + if (name === 'crash-after-handle') assert(!recovered.trace.some(t => t.kind === 'run-request')); + assert.equal(await readObject(`${taskId}/payload.json`), storedPayload); + } + } else if (name === 'handle-write-rejected') { + assert.equal(first.code, 1); + assert(first.trace.some(t => t.error?.includes('MICROVM_START_RECEIPT_SAVE_FAILED'))); + assert.equal((await task(taskId))?.microvm_start?.handle, undefined); + } else { + assert.equal(first.code, 0); + } + if (name === 'lost-run-reply') assert(first.trace.some(t => t.kind === 'result' && t.autoRetried)); + const launches = traces.filter(t => t.kind === 'run-success'); + const ids = new Set(launches.map(t => t.id)); + assert.equal(ids.size, 1, 'Application recovery created different workers'); + assert.equal(new Set(launches.map(t => t.fingerprint)).size, 1, 'Recovery changed the service request'); + const id = launches[0]!.id!; + if (!['expired-receipt', 'handle-write-rejected'].includes(name)) { + const saved = await task(taskId); + assert.equal(saved?.microvm_start?.clientToken, taskId); + assert.equal(saved?.microvm_start?.handle?.microvmId, id); + assert.equal(saved?.session_id, id); + } + if (name === 'canceled-handle') { + await ddb.send(new UpdateCommand({ + TableName: table, + Key: { task_id: taskId }, + UpdateExpression: 'SET #s = :c', + ExpressionAttributeNames: { '#s': 'status' }, + ExpressionAttributeValues: { ':c': TaskStatus.CANCELLED }, + })); + const closed = await execute(context); + assert.equal(closed.code, 1); + assert(closed.trace.some(t => t.error?.includes('MICROVM_START_TASK_CLOSED'))); + assert(!closed.trace.some(t => t.kind === 'run-request')); + } + for (let i = 0; i < 30; i++) { + const observed = await live.mv.send(new GetMicrovmCommand({ microvmIdentifier: id })); + if (observed.state === 'TERMINATED') break; + assert(i < 29, 'Synthetic worker did not terminate'); + await delay(2_000); + } + // Rejected writes/cancellation may stop the VM before its startup hook. + // Other cases must reach the malformed-anchor barrier, before installation. + let messages: string[] = []; + for (let i = 0; i < 30; i++) { + const worker = await live.mv.send(new GetMicrovmCommand({ microvmIdentifier: id })); + const day = worker.startedAt!.toISOString().slice(0, 10).replaceAll('-', '/'); + const logs = await live.aws<{ events?: { message?: string }[] }>([ + 'logs', 'filter-log-events', '--log-group-name', live.logGroup, + '--log-stream-name-prefix', `${day}[${target.imageVersion}]${id}`, + ]); + messages = (logs.events ?? []).map(e => e.message ?? ''); + assert(!messages.some(m => /X-Amz-(Signature|Credential|Security-Token)=/i.test(m)), 'Signed URL in logs'); + assert(!messages.some(m => /installed platform_config env|hook accepted task_id=/.test(m)), 'Pipeline unexpectedly started'); + if (['handle-write-rejected', 'canceled-handle'].includes(name)) break; + if (messages.some(m => m.includes('must be a well-formed IAM role ARN'))) break; + assert(i < 29, 'Expected guest configuration rejection not observed'); + await delay(2_000); + } + await writeFile(join(directory, `${id}.log`), messages.map(redactPayloadUrls).join(''), { mode: 0o600 }); + await cleanupTask(taskId); + await report({ + stage: 'passed', + case: name, + taskId, + id, + runRequests: launches.length, + requestFingerprint: launches[0]!.fingerprint, + taskAndPayloadDeleted: true, + }); + } + } catch (error) { + failed = true; + await report({ stage: 'failed', error: safeError(error) }); + } finally { + const ids = [...new Set((await allTraces()).filter(t => t.kind === 'run-success' && t.id).map(t => t.id!))]; + const workerCleanup = await Promise.allSettled(ids.map(id => stopWorker(live.mv, id))); + const taskCleanup = await Promise.allSettled(plans.map(p => cleanupTask(p.taskId))); + const manifestCleanup = await Promise.allSettled([s3.send(new DeleteObjectCommand({ Bucket: bucket, Key: manifestKey })).then(() => absent(manifestKey))]); + const cleanup = [...workerCleanup, ...taskCleanup, ...manifestCleanup]; + const after = await listWorkers(live.mv, live.image); + const unaccounted = after.filter(w => !before.some(b => b.microvmId === w.microvmId) && !ids.includes(w.microvmId!)); + await report({ + stage: 'cleanup', + ids, + tasks: plans.map(p => p.taskId), + confirmed: cleanup.every(r => r.status === 'fulfilled'), + unaccountedIds: unaccounted.map(w => w.microvmId), + errors: cleanup.filter(r => r.status === 'rejected').map(r => safeError(r.reason)), + }); + if (cleanup.some(r => r.status === 'rejected') || unaccounted.length) failed = true; + } + if (failed) process.exitCode = 1; +} + +main().catch(error => { + process.stderr.write(`${safeError(error)}\n`); + process.exitCode = 1; +}); diff --git a/docs/verification/645-p2-start-recovery-live-20260914.md b/docs/verification/645-p2-start-recovery-live-20260914.md new file mode 100644 index 000000000..6c4a93b7b --- /dev/null +++ b/docs/verification/645-p2-start-recovery-live-20260914.md @@ -0,0 +1,209 @@ +# P2 live MicroVM start and recovery verification + +## Result + +On 2026-09-14, the AWS service replay checks and **13 application start/recovery +cases passed** against the existing `backgroundagent-dev` deployment in +`us-west-2`, account ``, profile `sphia-dev`, image `2.0`. +The deployed source remains `e1d5debe`, with bootstrap policy bundle `1.7.0`. + +This extends the [payload verification](./645-p2-payload-live-20260914.md). +It does not complete the full P2 matrix or implement P3 sleep/wake. +The [implementation checklist](./645-p3-implementation-plan.md) tracks those gates. + +## What “recovery” means here + +The coordinator starts a task's worker. Before asking AWS, it saves a +**receipt** in DynamoDB, the task database. The receipt contains a stable +request number, called the **client token**, and a fingerprint of the request. +Once AWS returns the worker ID, that ID is saved too. + +If a reply gets lost, retrying with the same receipt should find the same +worker. It should not change the instructions or buy another computer. +If the application no longer has enough information to retry safely, it must +stop and report uncertainty. + +These checks use two separate runners: + +- [Service replay](../../cdk/test/live/verify-microvm-replay.live.ts) calls + `RunMicrovm` directly and measures AWS's actual token behavior. +- [Application recovery](../../cdk/test/live/verify-microvm-start.live.ts) runs + the production strategy, S3 producer and DynamoDB receipt code in fresh local + child processes against real AWS. Its + [fault injector](../../cdk/test/live/microvm-start-child.ts) discards successful + replies or exits the child after an acknowledged operation. The next child + receives only the original task/settings, and recovers through AWS storage. + +The fault injector retains worker IDs separately for the test's cleanup audit. +The recovering application does not read that observer trace. This tests a lost +reply/process boundary, but it does not prove recovery when neither the +application nor an operator has any record of the worker ID. + +## AWS service observations + +The test uses an invalid startup reference, so the real guest rejects it before +S3 reads, configuration installation or pipeline startup. + +| Request | Observed result | +|---|---| +| Two simultaneous identical starts | Both returned the same worker ID | +| Immediate identical replay | Same ID and original response | +| Same token, duration changed from 180 to 181 seconds | `ValidationException`: “The provided clientToken was used with different request parameters.” | +| Replay after `GetMicrovm` reported termination | Same ID; no replacement worker | +| Replay at elapsed 30.194, 144.855 and 305.207 seconds | Same ID and request fingerprint | + +The worker was `microvm-c87a143e-1132-36ee-987f-ea778e23f955`. +The last replay's AWS request ID was +`1cc4e9be-b3c9-492d-95e4-e92a30cef006`. + +The replay response kept saying **`PENDING`**, even after termination. +An independent `GetMicrovm` read confirmed the original start time +`17:25:58.707Z`, termination time `17:26:04.155Z`, and state `TERMINATED`. +The replay returns the original response; callers must poll current state +separately. These observations through roughly five minutes do not establish +AWS's maximum token-retention period or its behavior after that period expires. +The application's 120-second cutoff remains a separate conservative limit. + +## Application observations + +Each task uses synthetic configuration with a deliberately malformed +`agent_session_role_arn`. The producer accepts the nonempty identifier, but the +guest rejects its shape before installing configuration, fetching secrets or +starting a pipeline. The control verifies this barrier through real worker logs. + +The operator runs the control-plane code. The guest uses the unchanged deployed +MicroVM execution role and runtime egress connector, with explicit `NO_INGRESS`. +The test changes two outgoing Run fields consistently: it caps worker lifetime +at **180 seconds** instead of the production eight hours, and sets the existing +MicroVM log group. Receipt hashing, payload preparation and application retry +logic remain unchanged. No deployed Lambda is killed or modified. + +| Case | Verified behavior | +|---|---| +| Control | Real S3 files, start receipt and worker-ID registration completed | +| Successful Run reply discarded | Production automatic retry recovered the same ID; `autoRetried` was true | +| Process exits after Run success, before receiving its result | Fresh process reused the saved token and exact signed request; same worker ID | +| Process exits after task-file write | Fresh process reused the immutable task bytes and completed the launch | +| Process exits after private launch-record write | Fresh process recovered the saved launch reference and completed the launch | +| Task-file write reply discarded | Producer read back the committed object and continued | +| Launch-record write reply discarded | Producer recovered the committed reference and continued | +| Worker-ID database write reply discarded | Strategy read back the committed handle and succeeded | +| Process exits after worker-ID write | Fresh process returned the saved handle without another Run request | +| Worker-ID write deliberately rejected before submission | Strategy stopped the known worker and reported `MICROVM_START_RECEIPT_SAVE_FAILED` | +| Changed instructions after interrupted start | `MICROVM_START_INPUT_CHANGED`; no Run call and no overwrite; original request could still recover | +| Cancellation with a saved handle | `MICROVM_START_TASK_CLOSED`; no new Run request; worker stopped | +| Actual receipt deadline passes after interrupted start | `MICROVM_START_OUTCOME_UNKNOWN`; no further Run request | + +Each case that reached AWS used exactly one worker ID. Repeated Run requests +had identical fingerprints, including the signed URL. Successful registration +saved matching `microvm_start.handle`, `session_id` and task-stable client token. +No capacity reservation or production channel/repository operation was created. + +The denied-write case injects an exception locally; it is not a test of an +actual IAM denial. Cancellation is observed through a fresh strategy invocation, +not a deployed approval/cancel handler racing a durable checkpoint. + +## Cleanup, logs and test corrections + +Independent reads confirmed **17 owned workers terminated**, all **36 planned +S3 object locations absent**, and all **16 planned task IDs absent**, including +setup failures and cases skipped when an earlier assertion stopped a batch. +Those counts include three service-probe workers and fourteen application-probe +workers across the control, fault batch and corrected cancellation rerun. +They are cleanup coordinates, not counts of distinct passing test cases. + +The final log audit found no signed-URL credential/signature markers and no +configuration-installed or pipeline-accepted messages. Four workers stopped +before emitting guest logs; the other thirteen had three events each. The test +does not use an empty log stream alone to prove a successful rejection. + +The runner calls the production task-file deletion helper, verifies both +files are absent, deletes only its own synthetic task rows, and removes its +unique manifest. This is operator cleanup; it does not replace the remaining +deployed-coordinator finalization checks. + +The stack remains `UPDATE_COMPLETE`, with last update +`2026-09-14T16:07:52.421Z`. No image, role policy or network configuration changed. + +Several harness expectations were corrected during calibration: + +- `ListMicrovms` permits at most 50 results per page. +- Parameter mismatch uses `ValidationException`; the state reason says + `HTTP status 400`. +- Child processes must reuse the parent's resolved `tsx` loader, since `npx` + can supply it from its cache. +- The task status is `TaskStatus.CANCELLED`. The misspelled fixture `CANCELED` + correctly failed as an unknown state. The corrected cancellation test passed. + +All affected fixtures were cleaned, and the corrected cases were rerun. +The production behavior required no fix in this batch. Comment cleanup in +`agent/src/server.py` and the MicroVM strategy removes stale claims about P1 +envelope compatibility, logged payloads and absence of automatic termination. +The helper's direct `None` no-op is not a legacy v2 startup path; the eight-hour +service lifetime is a backstop, not prompt cleanup. + +## Reproduce and validate + +Run from `cdk/` with the installed repository dependencies, Node 22 and AWS CLI. +Both scripts make no AWS calls without `--execute`. Output directories must +not already exist. + +```bash +mise exec -- npx tsx test/live/verify-microvm-replay.live.ts +mise exec -- npx tsx test/live/verify-microvm-start.live.ts + +AWS_PROFILE=sphia-dev mise exec -- npx tsx test/live/verify-microvm-replay.live.ts \ + --execute --account --region us-west-2 \ + --stack backgroundagent-dev --image-version 2.0 \ + --output /tmp/abca-p2-replay-new-run + +AWS_PROFILE=sphia-dev mise exec -- npx tsx test/live/verify-microvm-start.live.ts \ + --execute --account --region us-west-2 \ + --stack backgroundagent-dev --image-version 2.0 \ + --output /tmp/abca-p2-start-new-run +``` + +The application runner supports a comma-separated `--cases` selection. +Do not run independent worker-creating probes concurrently: each checks its +before/after inventory for unaccounted workers and reports them without deleting +them. If interrupted, use the private context, task plans and observer traces +for exact cleanup. The 180-second worker cap does not delete database/S3 files. +Context and traces contain identifiers and fingerprints, not signed URLs or +credentials. Both runners are outside the normal Jest test pattern. + +Validation: three existing CDK suites passed **134 tests**, and the selected +Python platform-config tests passed **5 tests**. Focused strict TypeScript, +ESLint and both dry runs passed. Documentation sync and changed-file link +checks accompany this record. No full application redeployment was needed. + +## Evidence and remaining gates + +Private evidence root: `/tmp/abca-645-p2-clean-20260913`. + +- `start-service-replay-v4-20260914`: successful service probe; earlier + `start-service-replay*` records retain calibration and cleanup. +- `start-recovery-control-v2-20260914`: successful application control. +- `start-recovery-faults-20260914`: ten passing fault cases and the misspelled + cancellation fixture; cleanup confirmed. +- `start-recovery-final-20260914`: corrected cancellation and real expiry passed. +- `start-recovery-final-audit.json`, `start-recovery-final-log-audit.json`, + `start-recovery-stack-status.json`: independent final audits. + +Per-run `results.json` links cases to task/worker IDs and AWS Run request IDs; +child traces record committed effects, injected faults and request fingerprints. +The five-minute service probe and the 120-second application cutoff are measured +separately. + +Still required: deployed durable-Lambda interruption/checkpoint recovery, +registration/cancellation races, automatic finalization after rejected hooks, +injected cleanup failures, and an operator recovery procedure for a genuinely +unknown worker ID. A CloudTrail Event History lookup for `RunMicrovm` during +this test interval returned no events; it did not establish such a procedure. +`ListMicrovms`/`GetMicrovm` expose no task token in the installed SDK shapes. + +The wider effective-role/session/transaction matrix, expired signer credentials, +ECS, public-bucket policy grants, network negatives, capacity migration and P3 +sleep/wake also remain open. Read-only trust inspection confirmed the session +role accepts the exact worker roles and the MicroVM execution role trusts +`lambda.amazonaws.com`; an isolated Lambda using that unchanged role is a +possible follow-up for actual IAM requests. No such function was created. diff --git a/docs/verification/645-p3-implementation-plan.md b/docs/verification/645-p3-implementation-plan.md index 7f0743218..9c602bf40 100644 --- a/docs/verification/645-p3-implementation-plan.md +++ b/docs/verification/645-p3-implementation-plan.md @@ -31,7 +31,13 @@ denial and a payload over 1 MiB. Concurrent/repeated preparation and immediate identical Run replay passed; all 12 disposable workers and 29 synthetic object locations were cleaned up. These direct probes use operator credentials for preparation and bypass coordinator admission/finalization. -Crash/recovery and the wider IAM/network matrix still remain. +The [start/recovery follow-up](./645-p2-start-recovery-live-20260914.md) +passed simultaneous/changed-request service checks, terminated-worker replay +through roughly five minutes, and 13 production-code cases against AWS storage +with local process/reply faults. Saved-handle recovery, changed-input refusal, +cancellation and the actual 120-second cutoff passed. These tests use operator +credentials and do not interrupt the deployed durable Lambda. +Deployed-coordinator recovery and the wider IAM/network matrix still remain. Full P2 acceptance and all P3 live gates remain open. The batch notes below record what was verified at their original completion; their deployment status is superseded by these records. @@ -48,7 +54,9 @@ is superseded by these records. - [ ] Complete v2 effective-role/public-bucket tests, expired signer credentials, coordinator recovery and the ECS/coordinated-rollout matrix in AWS. - [x] Implement saved MicroVM start receipts, stable tokens, input fingerprints and handle recovery. - [x] Verify immediate identical `RunMicrovm` replay returns the same worker ID in the live payload probes. -- [ ] Verify delayed AWS token retention, simultaneous/changed-request conflicts, lost Run replies and unknown-start cleanup on a live deployment. +- [x] Verify simultaneous identical Run calls, changed-parameter rejection, and replay after termination through roughly five minutes against AWS; distinguish cached Run responses from fresh VM state. +- [x] Verify production start/receipt/payload code against AWS with lost replies and local process death, saved-handle recovery, changed-input/cancellation refusal and actual receipt expiry. +- [ ] Verify deployed durable-Lambda checkpoint/registration recovery and races, AWS behavior after token retention expires, and operator cleanup of genuinely unknown worker IDs. - [x] Make capacity acquisition/release atomic per task across crash replay; unify counter writers and repair. - [ ] Verify the capacity protocol's upgrade/drain procedure, deployed IAM and scan scale in AWS. - [x] Restrict agent task updates to reporting fields; remove replacement/deletion and worker counter grants. @@ -229,10 +237,13 @@ verify MicroVM manifest/download access through the runtime connector, invalid task/config/path/bytes/signature rejection, URL expiry/revocation, foreign private-bucket denial and >1 MiB transport. Concurrent/repeated preparation and changed-input conflicts exercise real S3 with operator credentials. +The [start/recovery follow-up](./645-p2-start-recovery-live-20260914.md) +also verifies recovery after committed S3 replies are lost or the local process +exits after payload/launch writes. **Still required:** effective-role cross-task/list/write/public-bucket negatives, -expired signer credentials, lost committed replies and restart under the -coordinator role, and equivalent ECS/upgrade evidence. The role/tag and other +expired signer credentials, recovery under the deployed coordinator role and +durable execution, and equivalent ECS/upgrade evidence. The role/tag and other platform-grant limits in 1G remain; this boot-path fix does not establish complete hostile-worker isolation. @@ -250,11 +261,15 @@ The receipt works like an order number: when the reply gets lost, the next call Fault tests exercise a successful simulated service creation followed by a lost response, a second application call with the same token, a fresh strategy instance, saved-handle replay, changed input, expired recovery, confirmed rejection, cancellation before/during/after creation, and lost DynamoDB responses. The handler recovers committed registration, treats start-audit failures as non-fatal, and routes start failures through one finalization path. Finalization reads the latest committed task, avoiding a stale cancellation/failure report. An unknown first outcome stays unknown even when the second call gets a definite rejection; HTTP 408 and named service timeouts remain uncertain even with a 4xx status. -**Live subset completed:** every passing [payload probe](./645-p2-payload-live-20260914.md) -immediately repeated the exact Run request and received the same worker ID. -The probes do not simulate a lost reply or coordinator restart. +**Live subset completed:** [service and application recovery probes](./645-p2-start-recovery-live-20260914.md) +extend immediate replay with simultaneous identical requests, changed-parameter +rejection and terminated-worker replay through roughly five minutes. Production +strategy/storage code recovered lost successful replies and local process death +using real AWS; it also reused saved handles without Run, refused changed input +and cancellation, and stopped replay after its actual 120-second deadline. +These use operator credentials and do not kill the deployed durable Lambda. -**Still required:** the installed SDK documents `clientToken` idempotency but gives no retention period. Public AWS API documentation URLs did not provide a usable RunMicrovm reference during this review. The local 120-second limit is a conservative application cutoff, not evidence of AWS's retention window. Verify delayed replay, changed parameters, simultaneous requests/conflicts, token expiry and returned handles after termination against AWS before accepting this prerequisite. The service emulator proves client behavior only. +**Still required:** the installed SDK documents `clientToken` idempotency but gives no retention period. Public AWS API documentation URLs did not provide a usable RunMicrovm reference during this review. The local 120-second limit is a conservative application cutoff; roughly five minutes of observed AWS replay does not establish maximum retention or post-expiry behavior. Verify deployed durable-Lambda interruption, registration/cancellation races and operator recovery of an unknown worker ID before accepting this prerequisite. Cached `RunMicrovm` replay responses can still say `PENDING` after the worker terminated; use `GetMicrovm` for current state. For an unknown outcome with no returned ID, the task error or cancellation event identifies the saved token for investigation. Do not automatically submit a replacement task. Verify how operators find and terminate that VM in the deployed service; if they cannot recover an ID, the eight-hour lifetime cap is the remaining bound. Keep this limitation explicit in live evidence. @@ -309,8 +324,10 @@ in step 6. The [configuration follow-up](./645-p2-repository-config-20260914.md) also verifies automatic pre/post npm checks and cleanup after a worker-reported delivery failure. The [image 2.0 payload follow-up](./645-p2-payload-live-20260914.md) verifies direct startup-hook transport/rejections, immediate Run replay and -operator cleanup. Coordinator classification/finalization after rejected hooks, -crash/recovery/cleanup-error paths and the wider negative IAM/network matrix +operator cleanup. The [recovery follow-up](./645-p2-start-recovery-live-20260914.md) +adds service replay through five minutes and real-storage application tests with +local process/reply faults. Deployed coordinator classification/finalization +after rejected hooks, durable restart/cleanup-error paths and the wider IAM/network matrix remain unverified. `isadeks/vercel-abca-linear` explicitly selects `lambda-microvm`; the original seeded repository retains AgentCore. diff --git a/docs/verification/645-payload-bootstrap.md b/docs/verification/645-payload-bootstrap.md index 04672b33b..48adb73b6 100644 --- a/docs/verification/645-payload-bootstrap.md +++ b/docs/verification/645-payload-bootstrap.md @@ -115,6 +115,12 @@ immediate identical Run replay. Producer calls use operator credentials; direct Runs bypass coordinator admission/finalization. The table remains the full acceptance target, including combinations those probes do not cover. +The [start/recovery follow-up](./645-p2-start-recovery-live-20260914.md) also +verifies real S3 recovery after committed payload/launch replies are lost or a +local process exits between writes. Production start code preserves the saved +capability through a lost Run reply and a fresh process. These use operator +credentials; deployed durable-Lambda recovery and effective-role tests remain. + | Check | Expected result | |---|---| | New task with no stored launch | Coordinator gets `NoSuchKey`, creates immutable payload/reference, worker starts successfully. | From 6ea275cba6f6982d09e4674597aeacd046759ad7 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Mon, 14 Sep 2026 15:03:08 -0400 Subject: [PATCH 033/149] feat(agent): add MicroVM approval pause barrier --- agent/.vulture_allowlist.py | 5 + agent/src/bedrock_creds_helper.py | 5 +- agent/src/hooks.py | 172 ++++--- agent/src/microvm_lifecycle.py | 325 ++++++++++++ agent/src/progress_writer.py | 29 +- agent/src/server.py | 42 +- agent/tests/test_hooks.py | 63 +++ agent/tests/test_microvm_lifecycle.py | 472 ++++++++++++++++++ agent/tests/test_server.py | 35 ++ ...ADR-021-lambda-microvms-compute-backend.md | 2 +- ...Adr-021-lambda-microvms-compute-backend.md | 2 +- docs/verification/645-p3-guest-barrier.md | 118 +++++ .../645-p3-implementation-plan.md | 23 +- 13 files changed, 1214 insertions(+), 79 deletions(-) create mode 100644 agent/src/microvm_lifecycle.py create mode 100644 agent/tests/test_microvm_lifecycle.py create mode 100644 docs/verification/645-p3-guest-barrier.md diff --git a/agent/.vulture_allowlist.py b/agent/.vulture_allowlist.py index 7f59bc591..e4a3a38c7 100644 --- a/agent/.vulture_allowlist.py +++ b/agent/.vulture_allowlist.py @@ -21,3 +21,8 @@ hook_context # unused variable (src/hooks.py:132) hook_context # unused variable (src/hooks.py:918) hook_context # unused variable (src/hooks.py:1258) + +# urllib's HTTPRedirectHandler supplies this named parameter to redirect_request. +# The bootstrap override intentionally rejects every redirect without inspecting +# its destination; retain the standard-library callback signature. +newurl # unused variable (src/payload_bootstrap.py:_NoRedirect.redirect_request) diff --git a/agent/src/bedrock_creds_helper.py b/agent/src/bedrock_creds_helper.py index 5f5e4d022..40ad13d70 100644 --- a/agent/src/bedrock_creds_helper.py +++ b/agent/src/bedrock_creds_helper.py @@ -7,7 +7,10 @@ ``awsCredentialExport`` setting (in the image's managed-settings layer) runs this script, captures its JSON stdout, and signs Bedrock requests with the returned credentials. With a real ``Expiration`` it re-runs ~5 min before -expiry, so an 8 h task survives the 1 h role-chaining cap. +expiry during active use. Pinned Claude 2.1.191 returns its previous cached +value while that refresh runs in the background: this is NOT an acknowledged +refresh barrier after MicroVM sleep. P3 must cover that subprocess cache before +enabling suspension; refreshing Python's boto3 session does not clear it. Goal: assume the per-task SessionRole with ``{user_id, repo, task_id}`` STS session tags so Bedrock spend is attributable per user/repo in AWS Cost diff --git a/agent/src/hooks.py b/agent/src/hooks.py index 6db262f75..ddcc357bd 100644 --- a/agent/src/hooks.py +++ b/agent/src/hooks.py @@ -24,12 +24,14 @@ import re import time from collections.abc import Callable +from contextlib import nullcontext from dataclasses import dataclass from datetime import UTC, datetime from typing import TYPE_CHECKING, Any import nudge_reader import task_state +from microvm_lifecycle import get_context from nudge_reader import _xml_escape from output_scanner import scan_tool_output from policy import APPROVAL_RATE_LIMIT, FLOOR_TIMEOUT_S, Outcome @@ -538,6 +540,7 @@ async def pre_tool_use_hook( user_id=user_id, progress=progress, ts=ts_module, + tool_use_id=tool_use_id, ) @@ -551,6 +554,7 @@ async def _handle_require_approval( user_id: str | None, progress: Any, ts: Any, + tool_use_id: str | None = None, ) -> dict: """REQUIRE_APPROVAL branch of ``pre_tool_use_hook``. @@ -728,14 +732,24 @@ async def _handle_require_approval( matching_rule_ids=list(decision.matching_rule_ids), ) - # Step 9 — poll for a decision. - outcome = await _poll_for_decision( - task_id=task_id, - request_id=request_id, - deadline=deadline, - progress=progress, - ts=ts, - ) + # Step 9 — register the exact persisted gate and its ORIGINAL deadline. + # The SDK wrapper already tracks this tool; direct/legacy callers without a + # lifecycle context keep their existing approval behavior. + lifecycle = get_context(task_id) + park = lifecycle.park_approval(request_id, tool_use_id, deadline) if lifecycle else None + try: + outcome = await _poll_for_decision( + task_id=task_id, + request_id=request_id, + deadline=deadline, + progress=progress, + ts=ts, + ) + finally: + if lifecycle and park: + # Clear the safe point before ANY approval/task-state mutation or + # hook return. A concurrent suspend owns the barrier until resume. + await lifecycle.leave_approval(park) # Step 10 — VM-throttle + late-approval race. Best-effort flip to # TIMED_OUT; if ConditionCheckFailed, the user beat us — read and honor. @@ -989,56 +1003,58 @@ async def _poll_for_decision( consecutive_fails = 0 degraded_emitted = False + lifecycle = get_context(task_id) while True: - if deadline.remaining_s() <= 0: - return {"status": "TIMED_OUT", "reason": None} + async with lifecycle.approval_poll() if lifecycle else nullcontext(): + if deadline.remaining_s() <= 0: + return {"status": "TIMED_OUT", "reason": None} - try: - row = await asyncio.to_thread( - ts.get_approval_row, - task_id, - request_id, - consistent_read=True, - ) - consecutive_fails = 0 - except Exception as exc: - consecutive_fails += 1 - log( - "WARN", - f"approval poll get_item raised ({consecutive_fails}/" - f"{POLL_MAX_CONSECUTIVE_FAILS}): {type(exc).__name__}: {exc}", - ) - if consecutive_fails >= POLL_DEGRADED_FAILS and not degraded_emitted: - if progress is not None: - _try_progress( - progress, - "write_approval_poll_degraded", - request_id=request_id, - consecutive_failures=consecutive_fails, - ) - degraded_emitted = True - if consecutive_fails >= POLL_MAX_CONSECUTIVE_FAILS: - return { - "status": "TIMED_OUT", - "reason": f"poll failed {consecutive_fails} consecutive times", - } - row = None # force sleep below - - if row is not None: - status = row.get("status") - if status == "APPROVED": - return { - "status": "APPROVED", - "scope": row.get("scope"), - "decided_at": row.get("decided_at"), - "decided_by": row.get("user_id"), - } - if status == "DENIED": - return { - "status": "DENIED", - "reason": row.get("deny_reason") or "denied", - "decided_at": row.get("decided_at"), - } + try: + row = await asyncio.to_thread( + ts.get_approval_row, + task_id, + request_id, + consistent_read=True, + ) + consecutive_fails = 0 + except Exception as exc: + consecutive_fails += 1 + log( + "WARN", + f"approval poll get_item raised ({consecutive_fails}/" + f"{POLL_MAX_CONSECUTIVE_FAILS}): {type(exc).__name__}: {exc}", + ) + if consecutive_fails >= POLL_DEGRADED_FAILS and not degraded_emitted: + if progress is not None: + _try_progress( + progress, + "write_approval_poll_degraded", + request_id=request_id, + consecutive_failures=consecutive_fails, + ) + degraded_emitted = True + if consecutive_fails >= POLL_MAX_CONSECUTIVE_FAILS: + return { + "status": "TIMED_OUT", + "reason": f"poll failed {consecutive_fails} consecutive times", + } + row = None # force sleep below + + if row is not None: + status = row.get("status") + if status == "APPROVED": + return { + "status": "APPROVED", + "scope": row.get("scope"), + "decided_at": row.get("decided_at"), + "decided_by": row.get("user_id"), + } + if status == "DENIED": + return { + "status": "DENIED", + "reason": row.get("deny_reason") or "denied", + "decided_at": row.get("decided_at"), + } # Compute sleep interval based on elapsed since poll started. elapsed = time.monotonic() - start @@ -1781,6 +1797,9 @@ def build_hook_matchers( # PostToolUse closure feeds it every tool result; the Stop closure reads it # between turns to steer / bail on a repeating failing command. _stuck_guard = StuckGuard() + # Retain the controller even after registry removal. A late callback must + # see its CLOSED barrier, not fall back to the non-MicroVM path. + lifecycle = get_context(task_id) # Closure-based wrapper matches the HookCallback signature exactly: # (HookInput, str | None, HookContext) -> Awaitable[HookJSONOutput] @@ -1794,7 +1813,17 @@ async def _pre( # undefined — we MUST NOT trust it to fail closed. Mapping every # uncaught exception to a DENY here makes the security posture # explicit at the SDK boundary. + allowed = False try: + if lifecycle: + await lifecycle.tool_started(tool_use_id) + if isinstance(hook_input, dict): + tool_input = hook_input.get("tool_input") + # Legacy serialized inputs are normalized deeper in the + # policy hook. Keep their tool behavior, but do not infer a + # safe foreground-only execution from an opaque value here. + if not isinstance(tool_input, dict) or tool_input.get("run_in_background"): + lifecycle.disable_suspend() result = await pre_tool_use_hook( hook_input, tool_use_id, @@ -1806,6 +1835,10 @@ async def _pre( progress=progress, repo_url=repo_url or None, ) + if result.get("hookSpecificOutput", {}).get("permissionDecision") == "allow": + if lifecycle: + await lifecycle.wait_until_open() + allowed = True except Exception as exc: log( "ERROR", @@ -1818,6 +1851,9 @@ async def _pre( return SyncHookJSONOutput( **_deny_response("Hook error — fail-closed deny"), ) + finally: + if lifecycle and not allowed: + lifecycle.tool_finished(tool_use_id) return SyncHookJSONOutput(**result) async def _post( @@ -1840,6 +1876,25 @@ async def _post( "updatedMCPToolOutput": "[Output redacted: hook error — fail-closed]", } return SyncHookJSONOutput(hookSpecificOutput=fail_closed) + finally: + if lifecycle: + # Background Bash may be promoted after invocation. A returned + # background identifier means the tool's post hook is not a + # reliable indication that its child stopped. + if isinstance(hook_input, dict): + response = hook_input.get("tool_response") + if isinstance(response, dict) and ( + response.get("backgroundTaskId") or response.get("background_task_id") + ): + lifecycle.disable_suspend() + lifecycle.tool_finished(tool_use_id) + + async def _post_failure( + hook_input: HookInput, tool_use_id: str | None, ctx: HookContext + ) -> HookJSONOutput: + if lifecycle: + lifecycle.tool_finished(tool_use_id) + return SyncHookJSONOutput() async def _stop( hook_input: HookInput, tool_use_id: str | None, ctx: HookContext @@ -1866,8 +1921,11 @@ async def _stop( # Empty dict == allow stop. SyncHookJSONOutput(**{}) is fine. return SyncHookJSONOutput(**result) - return { + matchers = { "PreToolUse": [HookMatcher(matcher=None, hooks=[_pre])], "PostToolUse": [HookMatcher(matcher=None, hooks=[_post])], "Stop": [HookMatcher(matcher=None, hooks=[_stop])], } + if lifecycle: + matchers["PostToolUseFailure"] = [HookMatcher(matcher=None, hooks=[_post_failure])] + return matchers diff --git a/agent/src/microvm_lifecycle.py b/agent/src/microvm_lifecycle.py new file mode 100644 index 000000000..f7cf3a3fe --- /dev/null +++ b/agent/src/microvm_lifecycle.py @@ -0,0 +1,325 @@ +# Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. +# SPDX-License-Identifier: MIT-0 + +"""Guest-side pause barrier for ADR-021. + +This controller does not call AWS or decide approvals. The server owns one +controller per running MicroVM task; tool hooks register work and the *original* +approval deadline. Future HTTP lifecycle handlers must supply the durable +checkpoint and credential-refresh operations before declaring image capability. + +Returning from a suspend callback is not proof that AWS actually froze the VM. +Once acknowledged, the barrier opens only after a successful resume callback. +There is deliberately no timer that releases coding after an ambiguous suspend: +the supervisor must wake or terminate within its bounded recovery window. +""" + +from __future__ import annotations + +import asyncio +import math +import os +import random +import threading +import time +from contextlib import asynccontextmanager, contextmanager +from dataclasses import dataclass +from typing import TYPE_CHECKING, Protocol + +if TYPE_CHECKING: + from collections.abc import Callable + + +class ApprovalDeadline(Protocol): + def remaining_s(self) -> float: ... + + +class LifecycleUnavailable(RuntimeError): + """The guest cannot establish a safe lifecycle boundary.""" + + +def reseed_random() -> None: + """Give each run/wake fresh application PRNG state; secrets still use OS RNG.""" + random.seed(os.urandom(32)) + + +@dataclass(frozen=True) +class ApprovalPark: + task_id: str + microvm_id: str + request_id: str + tool_use_id: str + deadline: ApprovalDeadline + + +class MicrovmLifecycle: + """Synchronize tool execution, approval exit, and bounded lifecycle work. + + Locks protect only local state; never hold one across an await or AWS call. + Timed-out callbacks can finish in a worker thread, but only this controller + can commit a transition, and its generation check rejects late completion. + """ + + def __init__(self, task_id: str, microvm_id: str) -> None: + if not task_id or not microvm_id: + raise ValueError("Lifecycle requires task and MicroVM identity") + self.task_id = task_id + self.microvm_id = microvm_id + self._lock = threading.Lock() + self._tools: set[str] = set() + self._park: ApprovalPark | None = None + self._phase = "active" + self._generation = 0 + self._activities = 0 + self._progress_failed = False + self._suspend_ineligible = False + self._slept_request_id: str | None = None + + def _check_open(self) -> bool: + if self._phase in {"closed", "failed"}: + raise LifecycleUnavailable("Lifecycle barrier is closed") + return self._phase in {"active", "parked"} + + async def wait_until_open(self) -> None: + # A thread-safe local predicate works across the server and pipeline's + # separate event loops. No AWS calls or new approval timeout are needed. + while True: + with self._lock: + if self._check_open(): + return + await asyncio.sleep(0.02) + + async def tool_started(self, tool_use_id: str | None) -> None: + while True: + await self.wait_until_open() + with self._lock: + if not self._check_open(): + continue + if not tool_use_id or tool_use_id in self._tools: + # Preserve existing tool behavior, but unknown/duplicate + # identities cannot prove that every parallel tool stopped. + self._suspend_ineligible = True + else: + self._tools.add(tool_use_id) + return + + def tool_finished(self, tool_use_id: str | None) -> None: + with self._lock: + self._tools.discard(tool_use_id or "") + + def disable_suspend(self) -> None: + """Retain normal execution when detached work cannot be accounted for.""" + with self._lock: + self._suspend_ineligible = True + + def park_approval( + self, request_id: str, tool_use_id: str | None, deadline: ApprovalDeadline + ) -> ApprovalPark | None: + with self._lock: + if ( + not request_id + or not tool_use_id + or tool_use_id not in self._tools + or self._park is not None + or self._phase != "active" + ): + self._suspend_ineligible = True + return None + park = ApprovalPark(self.task_id, self.microvm_id, request_id, tool_use_id, deadline) + self._park = park + self._phase = "parked" + return park + + async def leave_approval(self, park: ApprovalPark) -> None: + """Remove the safe point *before* the hook changes task state/returns.""" + while True: + await self.wait_until_open() + with self._lock: + if not self._check_open(): + continue + if self._park is not park: + raise LifecycleUnavailable("Approval park changed") + self._park = None + self._phase = "active" + return + + @contextmanager + def activity(self): + """Drain synchronous progress and heartbeat work before suspension. + + Best-effort writers must explicitly report missing/failed acknowledgments + with progress_write_failed(). A later successful event cannot recover a + dropped earlier event, so that failure remains latched for the task. + """ + with self._lock: + if not self._check_open(): + # Do not silently write with pre-wake credentials. The progress + # writer catches this like its existing best-effort failures. + raise LifecycleUnavailable("Guest activity is paused") + self._activities += 1 + try: + yield + finally: + with self._lock: + self._activities -= 1 + + @asynccontextmanager + async def approval_poll(self): + """Pause new approval reads and drain an already-running read safely.""" + while True: + await self.wait_until_open() + with self._lock: + if not self._check_open(): + continue + self._activities += 1 + break + try: + yield + finally: + with self._lock: + self._activities -= 1 + + def progress_write_failed(self) -> None: + with self._lock: + self._progress_failed = True + + @staticmethod + def _budget(seconds: float) -> float: + if not math.isfinite(seconds) or seconds <= 0: + raise ValueError("Lifecycle budget must be finite and positive") + return time.monotonic() + seconds + + async def suspend( + self, checkpoint: Callable[[ApprovalPark], None], *, budget_s: float + ) -> ApprovalPark: + """Drain progress, then require an acknowledged gate/checkpoint check. + + ``checkpoint`` must synchronously verify durable task/gate identity and + deadline and raise on any failed or uncertain write/read. Returning None + means acknowledged success, never a best-effort event method. + """ + end = self._budget(budget_s) + with self._lock: + park = self._park + if ( + park is None + or self._phase != "parked" + or self._suspend_ineligible + or park.request_id == self._slept_request_id + or self._progress_failed + or self._tools != {park.tool_use_id} + or park.deadline.remaining_s() <= 0 + ): + raise LifecycleUnavailable("Task is not safely parked for suspend") + self._phase = "suspending" + self._generation += 1 + generation = self._generation + try: + while True: + with self._lock: + self._assert_transition(generation, "suspending") + if self._progress_failed: + raise LifecycleUnavailable("Progress was not acknowledged") + drained = self._activities == 0 + if drained: + break + if time.monotonic() >= end: + raise TimeoutError("Progress did not drain within lifecycle budget") + await asyncio.sleep(min(0.02, max(0, end - time.monotonic()))) + await self._run_bounded(checkpoint, park, end) + with self._lock: + self._assert_transition(generation, "suspending") + if self._progress_failed or park.deadline.remaining_s() <= 0: + raise LifecycleUnavailable("Suspend checkpoint is no longer safe") + self._phase = "suspend-ready" + return park + except BaseException: + # An unacknowledged checkpoint never authorizes a freeze. In-flight + # progress may finish later; new suspension stays off after failure. + with self._lock: + if self._generation == generation and self._phase == "suspending": + self._phase = "parked" + self._suspend_ineligible = True + self._generation += 1 + raise + + async def resume( + self, refresh_and_reconcile: Callable[[ApprovalPark], None], *, budget_s: float + ) -> ApprovalPark: + """Release the original wait only after credentials and state are safe. + + Callback ownership is intentionally narrow: refresh credentials and read + the existing gate, without deciding it, changing its deadline, or calling + leave_approval. Timeout/cancellation closes the barrier permanently; a + late thread must not release coding. The supervisor owns termination. + """ + end = self._budget(budget_s) + with self._lock: + park = self._park + if self._phase != "suspend-ready" or park is None: + raise LifecycleUnavailable("No acknowledged suspend to resume") + self._phase = "resuming" + self._generation += 1 + generation = self._generation + try: + await self._run_bounded(refresh_and_reconcile, park, end) + reseed_random() + with self._lock: + self._assert_transition(generation, "resuming") + self._phase = "parked" + # One sleep per approval gate. A duplicate suspend must not + # race the newly released decision loop. + self._slept_request_id = park.request_id + return park + except BaseException: + with self._lock: + if self._generation == generation and self._phase == "resuming": + self._phase = "failed" + self._generation += 1 + raise + + def _assert_transition(self, generation: int, phase: str) -> None: + if self._generation != generation or self._phase != phase: + raise LifecycleUnavailable("Lifecycle transition was superseded") + + @staticmethod + async def _run_bounded( + callback: Callable[[ApprovalPark], None], park: ApprovalPark, end: float + ) -> None: + remaining = end - time.monotonic() + if remaining <= 0: + raise TimeoutError("Lifecycle budget expired") + await asyncio.wait_for(asyncio.to_thread(callback, park), timeout=remaining) + + def close(self) -> None: + """Invalidate in-flight callbacks before pipeline teardown.""" + with self._lock: + self._phase = "closed" + self._generation += 1 + self._park = None + self._tools.clear() + + +_registry_lock = threading.Lock() +_contexts: dict[str, MicrovmLifecycle] = {} + + +def register_task(task_id: str, microvm_id: str) -> MicrovmLifecycle: + with _registry_lock: + if _contexts: + raise LifecycleUnavailable("A MicroVM pipeline is already registered") + context = MicrovmLifecycle(task_id, microvm_id) + _contexts[task_id] = context + return context + + +def get_context(task_id: str | None) -> MicrovmLifecycle | None: + with _registry_lock: + return _contexts.get(task_id or "") + + +def unregister_task(context: MicrovmLifecycle) -> None: + with _registry_lock: + context.close() + if _contexts.get(context.task_id) is context: + del _contexts[context.task_id] diff --git a/agent/src/progress_writer.py b/agent/src/progress_writer.py index f7dbea692..f885174ab 100644 --- a/agent/src/progress_writer.py +++ b/agent/src/progress_writer.py @@ -455,6 +455,25 @@ def _ensure_table(self): # -- core write ------------------------------------------------------------ def _put_event(self, event_type: str, metadata: dict) -> None: + from microvm_lifecycle import LifecycleUnavailable, get_context + + lifecycle = get_context(self._task_id) + if lifecycle is None: + self._put_event_best_effort(event_type, metadata) + return + try: + with lifecycle.activity(): + acknowledged = False + try: + acknowledged = self._put_event_best_effort(event_type, metadata) + finally: + if not acknowledged: + lifecycle.progress_write_failed() + except LifecycleUnavailable: + lifecycle.progress_write_failed() + print("[progress] lifecycle barrier closed — event not acknowledged", flush=True) + + def _put_event_best_effort(self, event_type: str, metadata: dict) -> bool: """Write a single progress event item to DynamoDB. Error handling splits three ways: @@ -472,12 +491,12 @@ def _put_event(self, event_type: str, metadata: dict) -> None: louder ERROR level so unexpected codes surface in reviews. """ if not self._table_name or self._disabled: - return + return False try: self._ensure_table() if self._table is None: self._disabled = True - return + return False now = datetime.now(UTC) # Correlation envelope (#245): trace_id is read per-event from the @@ -509,6 +528,7 @@ def _put_event(self, event_type: str, metadata: dict) -> None: # for the rest of the task (see ``_SharedCircuitBreaker`` # docstring). _CIRCUIT_BREAKERS.record_success(self._task_id) + return True except ImportError: self._disabled = True @@ -556,7 +576,7 @@ def _put_event(self, event_type: str, metadata: dict) -> None: f"({exc_type}: {code}); breaker NOT incremented: {e}", flush=True, ) - return + return False if classification == "transient": new_count, now_disabled = _CIRCUIT_BREAKERS.record_failure( @@ -575,7 +595,7 @@ def _put_event(self, event_type: str, metadata: dict) -> None: f"{self._MAX_FAILURES}, transient): {exc_type}: {e}", flush=True, ) - return + return False # Unknown: count like transient but flag loudly so operators # can add the new code to the classifier next release. @@ -596,6 +616,7 @@ def _put_event(self, event_type: str, metadata: dict) -> None: f"adding {exc_type} to the classifier: {e}", flush=True, ) + return False # -- public event methods -------------------------------------------------- diff --git a/agent/src/server.py b/agent/src/server.py index a47784941..98b5bb359 100644 --- a/agent/src/server.py +++ b/agent/src/server.py @@ -30,6 +30,13 @@ import task_state from config import resolve_github_token +from microvm_lifecycle import ( + LifecycleUnavailable, + get_context, + register_task, + reseed_random, + unregister_task, +) from models import TaskResult from observability import propagate_correlation_context from pipeline import run_task @@ -265,7 +272,16 @@ def _heartbeat_worker(task_id: str, stop: threading.Event) -> None: """Periodically refresh ``agent_heartbeat_at`` so the orchestrator can detect crashes.""" while not stop.wait(timeout=_HEARTBEAT_INTERVAL_SECONDS): try: - task_state.write_heartbeat(task_id) + lifecycle = get_context(task_id) + if lifecycle: + with lifecycle.activity(): + task_state.write_heartbeat(task_id) + else: + task_state.write_heartbeat(task_id) + except LifecycleUnavailable: + # An approval waiter has no liveness obligation while frozen. + # Resume must refresh credentials before another heartbeat writes. + continue except Exception as e: print( f"[heartbeat] write_heartbeat error (will retry): {type(e).__name__}: {e}", @@ -409,6 +425,7 @@ def _run_task_background( workload_access_token: str = "", attachments: list[dict] | None = None, resolved_assets: list[dict] | None = None, + microvm_id: str = "", ) -> None: """Run the agent task in a background thread.""" global _background_pipeline_failed @@ -448,18 +465,18 @@ def _run_task_background( task_id=task_id, ) + lifecycle = register_task(task_id, microvm_id) if microvm_id else None stop_heartbeat = threading.Event() hb_thread: threading.Thread | None = None - if task_id: - hb_thread = threading.Thread( - target=_heartbeat_worker, - args=(task_id, stop_heartbeat), - name=f"heartbeat-{task_id}", - daemon=True, - ) - hb_thread.start() - try: + if task_id: + hb_thread = threading.Thread( + target=_heartbeat_worker, + args=(task_id, stop_heartbeat), + name=f"heartbeat-{task_id}", + daemon=True, + ) + hb_thread.start() # Propagate the correlation envelope into this thread's OTEL context # so spans are correlated with the AgentCore session and the platform # identity in CloudWatch (#245). Runs whenever any field is present — @@ -513,6 +530,8 @@ def _run_task_background( ) task_state.write_terminal(task_id, "FAILED", backup.model_dump()) finally: + if lifecycle: + unregister_task(lifecycle) stop_heartbeat.set() if hb_thread is not None and hb_thread.is_alive(): hb_thread.join(timeout=3) @@ -1962,7 +1981,8 @@ def microvm_run(request: Request, body: MicrovmRunHookRequest): }, ) - _spawn_background(params) + reseed_random() + _spawn_background({**params, "microvm_id": body.microvmId}) task_id = params["task_id"] # Carries microvm_id as well as task_id: the "/run hook received" line that # used to correlate the two is stdout-only now (pre-install), so this is the diff --git a/agent/tests/test_hooks.py b/agent/tests/test_hooks.py index 29627b327..3e02708c0 100644 --- a/agent/tests/test_hooks.py +++ b/agent/tests/test_hooks.py @@ -2236,3 +2236,66 @@ def test_build_hook_matchers_creates_a_guard_without_crashing(self): engine = PolicyEngine(task_type="new_task", repo="owner/repo") matchers = build_hook_matchers(engine, task_id="t") assert "PostToolUse" in matchers and "Stop" in matchers + + +@pytest.mark.parametrize("cancelled", [False, True]) +def test_microvm_approval_cannot_return_until_resume_barrier_opens( + monkeypatch, engine_with_soft_gate, fake_task_state, progress, cancelled +): + import threading + + from microvm_lifecycle import register_task, unregister_task + + context = register_task("lifecycle-hook-task", "microvm-hook") + entered = threading.Event() + answer = threading.Event() + original_read = fake_task_state.get_approval_row + + def read(*args, **kwargs): + entered.set() + return {"status": "APPROVED", "scope": "once"} if answer.is_set() else {"status": "PENDING"} + + fake_task_state.get_approval_row = read + monkeypatch.setattr(hooks, "task_state", fake_task_state) + monkeypatch.setattr(hooks, "POLL_FAST_INTERVAL_S", 0.01) + if cancelled: + fake_task_state.resume_raises = _FakeApprovalResumeError("cancel won") + matchers = build_hook_matchers( + engine=engine_with_soft_gate, + task_id=context.task_id, + progress=progress, + ) + + async def scenario(): + pending = asyncio.create_task(matchers["PreToolUse"][0].hooks[0](_hook_input(), "tu-1", {})) + try: + assert await asyncio.to_thread(entered.wait, 1) + checkpoints = [] + await context.suspend(checkpoints.append, budget_s=1) + park = checkpoints[0] + assert park.request_id == fake_task_state.write_calls[0][1] + assert park.deadline.remaining_s() > 0 + answer.set() + await asyncio.sleep(0.04) + assert not pending.done() + assert fake_task_state.resume_calls == [] + refreshed = [] + await context.resume(refreshed.append, budget_s=1) + assert refreshed == [park] + result = await asyncio.wait_for(pending, 1) + expected = "deny" if cancelled else "allow" + assert result["hookSpecificOutput"]["permissionDecision"] == expected + if not cancelled: + await matchers["PostToolUse"][0].hooks[0]( + {"tool_name": "Bash", "tool_response": "ok"}, "tu-1", {} + ) + finally: + answer.set() + context.close() + await asyncio.gather(pending, return_exceptions=True) + + try: + _run(scenario()) + finally: + fake_task_state.get_approval_row = original_read + unregister_task(context) diff --git a/agent/tests/test_microvm_lifecycle.py b/agent/tests/test_microvm_lifecycle.py new file mode 100644 index 000000000..e0cf73c33 --- /dev/null +++ b/agent/tests/test_microvm_lifecycle.py @@ -0,0 +1,472 @@ +# Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. +# SPDX-License-Identifier: MIT-0 + +"""Concurrency regressions for the guest approval/sleep boundary.""" + +import asyncio +import threading +from unittest.mock import Mock + +import pytest + +from hooks import _ApprovalDeadline +from microvm_lifecycle import ( + LifecycleUnavailable, + MicrovmLifecycle, + get_context, + register_task, + unregister_task, +) + + +@pytest.fixture +def anyio_backend(): + return "asyncio" + + +@pytest.fixture(autouse=True) +def reset_progress_state(): + from progress_writer import _reset_circuit_breakers + + _reset_circuit_breakers() + yield + _reset_circuit_breakers() + + +async def parked(context=None): + context = context or MicrovmLifecycle("task", "microvm") + deadline = Mock(remaining_s=Mock(return_value=300)) + await context.tool_started("tool") + park = context.park_approval("request", "tool", deadline) + assert park is not None + return context, park + + +@pytest.mark.anyio +async def test_suspend_blocks_decision_and_new_tool_until_refresh_finishes(): + context, park = await parked() + checkpoint = Mock() + await context.suspend(checkpoint, budget_s=1) + checkpoint.assert_called_once_with(park) + decision = asyncio.create_task(context.leave_approval(park)) + new_tool = asyncio.create_task(context.tool_started("parallel")) + await asyncio.sleep(0.04) + assert not decision.done() + assert not new_tool.done() + refreshing = threading.Event() + release = threading.Event() + + def refresh(same_park): + assert same_park is park + refreshing.set() + assert release.wait(2) + + wake = asyncio.create_task(context.resume(refresh, budget_s=1)) + try: + assert await asyncio.to_thread(refreshing.wait, 1) + assert not decision.done() + assert not new_tool.done() + finally: + release.set() + await wake + await asyncio.wait_for(asyncio.gather(decision, new_tool), 1) + + +@pytest.mark.anyio +async def test_parallel_tool_prevents_suspend_until_its_post_hook(): + context, park = await parked() + await context.tool_started("other") + checkpoint = Mock() + with pytest.raises(LifecycleUnavailable): + await context.suspend(checkpoint, budget_s=1) + checkpoint.assert_not_called() + context.tool_finished("other") + assert await context.suspend(checkpoint, budget_s=1) is park + + +@pytest.mark.anyio +@pytest.mark.parametrize( + "reason", ["unknown-tool", "duplicate-tool", "detached-tool", "dropped-event"] +) +async def test_unaccounted_work_or_progress_disables_sleep_without_blocking_tools(reason): + context, park = await parked() + if reason == "unknown-tool": + await context.tool_started(None) + elif reason == "duplicate-tool": + await context.tool_started("tool") + elif reason == "detached-tool": + context.disable_suspend() + else: + context.progress_write_failed() + with context.activity(): + pass # A later success does not recover the lost event. + with pytest.raises(LifecycleUnavailable): + await context.suspend(Mock(), budget_s=1) + await context.leave_approval(park) + await context.tool_started("next") + + +@pytest.mark.anyio +async def test_suspend_drains_inflight_progress_before_checkpoint(): + context, park = await parked() + entered = threading.Event() + release = threading.Event() + + def write(): + with context.activity(): + entered.set() + assert release.wait(2) + + writer = asyncio.create_task(asyncio.to_thread(write)) + assert await asyncio.to_thread(entered.wait, 1) + checkpoint = Mock() + suspend = asyncio.create_task(context.suspend(checkpoint, budget_s=1)) + try: + await asyncio.sleep(0.04) + checkpoint.assert_not_called() + finally: + release.set() + await writer + assert await suspend is park + checkpoint.assert_called_once() + + +@pytest.mark.anyio +async def test_failed_inflight_progress_rejects_suspend(): + context, park = await parked() + entered = threading.Event() + release = threading.Event() + + def write(): + with context.activity(): + entered.set() + assert release.wait(2) + context.progress_write_failed() + + writer = asyncio.create_task(asyncio.to_thread(write)) + assert await asyncio.to_thread(entered.wait, 1) + checkpoint = Mock() + suspend = asyncio.create_task(context.suspend(checkpoint, budget_s=1)) + await asyncio.sleep(0.04) + release.set() + await writer + with pytest.raises(LifecycleUnavailable): + await suspend + checkpoint.assert_not_called() + await context.leave_approval(park) + + +@pytest.mark.anyio +async def test_checkpoint_failure_releases_unsuspended_wait_and_disables_retry(): + context, park = await parked() + with pytest.raises(OSError, match="durability"): + await context.suspend(Mock(side_effect=OSError("durability")), budget_s=1) + with pytest.raises(LifecycleUnavailable): + await context.suspend(Mock(), budget_s=1) + await context.leave_approval(park) + + +@pytest.mark.anyio +async def test_expiry_during_checkpoint_refuses_freeze(): + context, park = await parked() + + def checkpoint(_): + park.deadline.remaining_s.return_value = 0 + + with pytest.raises(LifecycleUnavailable): + await context.suspend(checkpoint, budget_s=1) + await context.leave_approval(park) + + +@pytest.mark.anyio +async def test_late_refresh_after_timeout_cannot_release_coding(): + context, park = await parked() + await context.suspend(Mock(), budget_s=1) + entered = threading.Event() + release = threading.Event() + completed = threading.Event() + + def refresh(_): + entered.set() + assert release.wait(2) + completed.set() + + wake = asyncio.create_task(context.resume(refresh, budget_s=0.1)) + try: + assert await asyncio.to_thread(entered.wait, 1) + with pytest.raises(TimeoutError): + await wake + with pytest.raises(LifecycleUnavailable): + await context.leave_approval(park) + finally: + release.set() + assert await asyncio.to_thread(completed.wait, 1) + with pytest.raises(LifecycleUnavailable): + await context.tool_started("next") + with pytest.raises(LifecycleUnavailable): + await context.resume(Mock(), budget_s=1) + + +@pytest.mark.anyio +async def test_close_during_refresh_invalidates_completion(): + context, park = await parked() + await context.suspend(Mock(), budget_s=1) + with pytest.raises(LifecycleUnavailable, match="superseded"): + await context.resume(lambda _: context.close(), budget_s=1) + with pytest.raises(LifecycleUnavailable): + await context.leave_approval(park) + + +@pytest.mark.anyio +async def test_resume_failure_and_cancellation_stay_closed(): + for failure in (OSError("STS unavailable"), asyncio.CancelledError()): + context, park = await parked() + await context.suspend(Mock(), budget_s=1) + with pytest.raises(type(failure)): + await context.resume(Mock(side_effect=failure), budget_s=1) + with pytest.raises(LifecycleUnavailable): + await context.leave_approval(park) + + +@pytest.mark.anyio +async def test_concurrent_lifecycle_calls_do_not_share_transition_ownership(): + context, park = await parked() + entered = threading.Event() + release = threading.Event() + + def checkpoint(_): + entered.set() + assert release.wait(2) + + suspend = asyncio.create_task(context.suspend(checkpoint, budget_s=1)) + try: + assert await asyncio.to_thread(entered.wait, 1) + with pytest.raises(LifecycleUnavailable): + await context.suspend(Mock(), budget_s=1) + with pytest.raises(LifecycleUnavailable): + await context.resume(Mock(), budget_s=1) + finally: + release.set() + await suspend + await context.resume(Mock(), budget_s=1) + # The same gate cannot sleep again while the released waiter catches up. + with pytest.raises(LifecycleUnavailable): + await context.suspend(Mock(), budget_s=1) + await context.leave_approval(park) + context.tool_finished("tool") + await context.tool_started("next") + next_park = context.park_approval("next-gate", "next", park.deadline) + assert await context.suspend(Mock(), budget_s=1) is next_park + + +@pytest.mark.anyio +async def test_original_deadline_survives_freeze_and_expired_resume(monkeypatch): + clock = {"wall": 1000.0, "mono": 50.0} + # Patch only the module method through a proxy; replacing global monotonic + # would also freeze asyncio's own timeout scheduler. + monkeypatch.setattr( + "hooks.time", + Mock(time=lambda: clock["wall"], monotonic=lambda: clock["mono"]), + ) + deadline = _ApprovalDeadline.from_recorded("1970-01-01T00:16:40Z", 300) + context = MicrovmLifecycle("task", "microvm") + await context.tool_started("tool") + park = context.park_approval("request", "tool", deadline) + await context.suspend(Mock(), budget_s=1) + clock["wall"] += 400 # Guest monotonic clock did not advance while frozen. + refresh = Mock() + assert await context.resume(refresh, budget_s=1) is park + assert park.deadline is deadline + assert park.deadline.remaining_s() == 0 + refresh.assert_called_once_with(park) + await context.leave_approval(park) + + +@pytest.mark.anyio +async def test_leaving_approval_before_suspend_wins_without_checkpoint(): + context, park = await parked() + await context.leave_approval(park) + checkpoint = Mock() + with pytest.raises(LifecycleUnavailable): + await context.suspend(checkpoint, budget_s=1) + checkpoint.assert_not_called() + + +def test_registry_rejects_second_pipeline_and_unregister_uses_identity(): + context = register_task("task", "microvm") + try: + assert get_context("task") is context + assert get_context("other") is None + with pytest.raises(LifecycleUnavailable): + register_task("other", "other-vm") + unregister_task(MicrovmLifecycle("task", "other-vm")) + assert get_context("task") is context + finally: + unregister_task(context) + assert get_context("task") is None + + +@pytest.mark.anyio +async def test_progress_writer_acknowledgments_are_shared_and_missing_table_refuses_sleep( + monkeypatch, +): + from progress_writer import _ProgressWriter + + context = register_task("progress-task", "microvm") + try: + await context.tool_started("tool") + context.park_approval("gate", "tool", Mock(remaining_s=Mock(return_value=300))) + monkeypatch.setenv("TASK_EVENTS_TABLE_NAME", "events") + writer = _ProgressWriter("progress-task") + writer._table = Mock() + writer._put_event("agent_milestone", {"message": "saved"}) + writer._table.put_item.assert_called_once() + + # A second writer's dropped event must invalidate the same task's + # barrier, even after the first writer successfully writes again. + missing = _ProgressWriter("progress-task") + missing._table_name = None + missing._put_event("agent_milestone", {"message": "lost"}) + writer._put_event("agent_milestone", {"message": "later success"}) + with pytest.raises(LifecycleUnavailable): + await context.suspend(Mock(), budget_s=1) + finally: + unregister_task(context) + + +@pytest.mark.anyio +async def test_real_progress_writer_failure_cannot_acknowledge_suspend(monkeypatch): + from progress_writer import _ProgressWriter + + context = register_task("write-error-task", "microvm") + try: + await context.tool_started("tool") + context.park_approval("gate", "tool", Mock(remaining_s=Mock(return_value=300))) + monkeypatch.setenv("TASK_EVENTS_TABLE_NAME", "events") + writer = _ProgressWriter("write-error-task") + writer._table = Mock() + writer._table.put_item.side_effect = OSError("write reply lost") + writer._put_event("agent_milestone", {"message": "uncertain"}) + with pytest.raises(LifecycleUnavailable): + await context.suspend(Mock(), budget_s=1) + finally: + unregister_task(context) + + +@pytest.mark.anyio +async def test_resume_reseeds_from_fresh_os_entropy_after_refresh(monkeypatch): + from microvm_lifecycle import reseed_random + + entropy = Mock(side_effect=[b"r" * 32, b"w" * 32]) + seed = Mock() + monkeypatch.setattr("microvm_lifecycle.os.urandom", entropy) + monkeypatch.setattr("microvm_lifecycle.random.seed", seed) + reseed_random() # Same entry point used by /run. + context, _ = await parked() + await context.suspend(Mock(), budget_s=1) + + def refresh(_): + seed.assert_called_once_with(b"r" * 32) + + await context.resume(refresh, budget_s=1) + assert [call.args for call in entropy.call_args_list] == [(32,), (32,)] + assert [call.args for call in seed.call_args_list] == [(b"r" * 32,), (b"w" * 32,)] + + +@pytest.mark.anyio +async def test_heartbeat_does_not_write_until_resume_finishes(monkeypatch): + import server + + context = register_task("heartbeat-task", "microvm") + write = Mock() + monkeypatch.setattr(server.task_state, "write_heartbeat", write) + try: + await context.tool_started("tool") + context.park_approval("gate", "tool", Mock(remaining_s=Mock(return_value=300))) + await context.suspend(Mock(), budget_s=1) + server._heartbeat_worker(context.task_id, Mock(wait=Mock(side_effect=[False, True]))) + write.assert_not_called() + await context.resume(Mock(), budget_s=1) + server._heartbeat_worker(context.task_id, Mock(wait=Mock(side_effect=[False, True]))) + write.assert_called_once_with(context.task_id) + finally: + unregister_task(context) + + +@pytest.mark.anyio +@pytest.mark.parametrize("background", [False, True, "serialized"]) +async def test_sdk_failure_hook_releases_tool_and_detached_work_disables_sleep( + monkeypatch, background +): + import hooks + + async def allow(*args, **kwargs): + return {"hookSpecificOutput": {"permissionDecision": "allow"}} + + monkeypatch.setattr(hooks, "pre_tool_use_hook", allow) + context = register_task("sdk-tools-task", "microvm") + try: + matchers = hooks.build_hook_matchers(engine=Mock(), task_id=context.task_id) + pre = matchers["PreToolUse"][0].hooks[0] + tool_input = ( + '{"run_in_background": true}' + if background == "serialized" + else {"run_in_background": background} + ) + await pre({"tool_name": "Bash", "tool_input": tool_input}, "first", {}) + await matchers["PostToolUseFailure"][0].hooks[0]({}, "first", {}) + await pre({"tool_name": "Bash", "tool_input": {}}, "approval", {}) + context.park_approval("gate", "approval", Mock(remaining_s=Mock(return_value=300))) + if background: + with pytest.raises(LifecycleUnavailable): + await context.suspend(Mock(), budget_s=1) + else: + await context.suspend(Mock(), budget_s=1) + finally: + unregister_task(context) + + +@pytest.mark.anyio +@pytest.mark.parametrize("already_started", [False, True]) +async def test_late_sdk_callback_keeps_closed_context_after_registry_removal( + monkeypatch, already_started +): + import hooks + + entered = asyncio.Event() + release = asyncio.Event() + + async def allow(*args, **kwargs): + entered.set() + await release.wait() + return {"hookSpecificOutput": {"permissionDecision": "allow"}} + + monkeypatch.setattr(hooks, "pre_tool_use_hook", allow) + monkeypatch.setattr(hooks, "log_error_cw", Mock()) + context = register_task("late-sdk-task", "microvm") + matchers = hooks.build_hook_matchers(engine=Mock(), task_id=context.task_id) + pre = matchers["PreToolUse"][0].hooks[0] + pending = None + try: + if already_started: + pending = asyncio.create_task(pre({"tool_input": {}}, "tool", {})) + await asyncio.wait_for(entered.wait(), 1) + unregister_task(context) + release.set() + result = await pending if pending else await pre({"tool_input": {}}, "tool", {}) + assert result["hookSpecificOutput"]["permissionDecision"] == "deny" + assert entered.is_set() is already_started + finally: + release.set() + unregister_task(context) + if pending: + await asyncio.gather(pending, return_exceptions=True) + + +@pytest.mark.anyio +@pytest.mark.parametrize("budget", [0, -1, float("nan"), float("inf")]) +async def test_bad_budget_does_not_close_approval_gate(budget): + context, park = await parked() + with pytest.raises(ValueError): + await context.suspend(Mock(), budget_s=budget) + await context.leave_approval(park) diff --git a/agent/tests/test_server.py b/agent/tests/test_server.py index 08d395896..1c2dbbb10 100644 --- a/agent/tests/test_server.py +++ b/agent/tests/test_server.py @@ -2933,3 +2933,38 @@ def test_every_unusable_body_yields_empty(self, raw): def test_ignores_unrelated_fields(self): assert server._parse_terminate_microvm_id(b'{"reason": "idle", "x": 1}') == "" + + +@pytest.mark.parametrize("crash", [False, True]) +def test_microvm_pipeline_registers_identity_and_always_removes_lifecycle(monkeypatch, crash): + from microvm_lifecycle import get_context + + observed = [] + + def run_task(**kwargs): + context = get_context("lifecycle-server-task") + assert context is not None + assert context.microvm_id == "microvm-server" + assert "microvm_id" not in kwargs + observed.append(context) + if crash: + raise RuntimeError("pipeline crashed") + + monkeypatch.setattr(server, "run_task", run_task) + monkeypatch.setattr(server, "_heartbeat_worker", lambda *_: None) + monkeypatch.setattr(server.task_state, "write_terminal", MagicMock()) + server._run_task_background( + repo_url="owner/repo", + task_description="test", + issue_number="", + github_token="test-token", + anthropic_model="model", + max_turns=1, + max_budget_usd=None, + aws_region="us-west-2", + task_id="lifecycle-server-task", + microvm_id="microvm-server", + ) + assert len(observed) == 1 + assert get_context("lifecycle-server-task") is None + assert server._background_pipeline_failed is crash diff --git a/docs/decisions/ADR-021-lambda-microvms-compute-backend.md b/docs/decisions/ADR-021-lambda-microvms-compute-backend.md index 16124ef33..424cb7d96 100644 --- a/docs/decisions/ADR-021-lambda-microvms-compute-backend.md +++ b/docs/decisions/ADR-021-lambda-microvms-compute-backend.md @@ -1,6 +1,6 @@ # ADR-021: AWS Lambda MicroVMs as a third ComputeStrategy backend -> **Implementation status (2026-09-14):** P1 ([#689](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/689)) and P2 ([#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733)) are merged. The takeover branch now has a [clean deployment](../verification/645-p2-clean-deployment-20260913.md) with bootstrap 1.7.0 and managed image 1.0, plus [live coding, PR iteration and cancellation evidence](../verification/645-p2-live-task-20260914.md), without manual IAM workarounds. Remaining P2 failure/recovery and permission/network checks prevent full acceptance. Local P3 foundations include suspend/resume commands, explicit VM observations, guarded persistent intent, a policy helper and approval deadlines that survive a frozen clock. Agent lifecycle hooks and supervisor/approval-handler integration remain unimplemented; automatic suspension is not enabled. This ADR defines P1–P3, not a P4. See the [current review](../verification/645-p3-readiness-review.md) and [P3 implementation plan](../verification/645-p3-implementation-plan.md). +> **Implementation status (2026-09-14):** P1 ([#689](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/689)) and P2 ([#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733)) are merged. The takeover branch now has a [clean deployment](../verification/645-p2-clean-deployment-20260913.md) with bootstrap 1.7.0 and a [verified managed-image update to 2.0](../verification/645-microvm-image-rebuild-20260914.md), plus [live coding, PR iteration and cancellation evidence](../verification/645-p2-live-task-20260914.md), without manual IAM workarounds. Remaining P2 failure/recovery and permission/network checks prevent full acceptance. Local P3 foundations include suspend/resume commands, explicit VM observations, guarded persistent intent, a policy helper, approval deadlines that survive a frozen clock, and the [guest pause controller](../verification/645-p3-guest-barrier.md). Agent lifecycle hooks and supervisor/approval-handler integration remain unimplemented; automatic suspension is not enabled. This ADR defines P1–P3, not a P4. See the [current review](../verification/645-p3-readiness-review.md) and [P3 implementation plan](../verification/645-p3-implementation-plan.md). **Status:** proposed **Date:** 2026-07-29 diff --git a/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md b/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md index b28960894..b77e94d2d 100644 --- a/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md +++ b/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md @@ -4,7 +4,7 @@ title: Adr 021 lambda microvms compute backend # ADR-021: AWS Lambda MicroVMs as a third ComputeStrategy backend -> **Implementation status (2026-09-14):** P1 ([#689](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/689)) and P2 ([#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733)) are merged. The takeover branch now has a [clean deployment](/sample-autonomous-cloud-coding-agents/architecture/645-p2-clean-deployment-20260913) with bootstrap 1.7.0 and managed image 1.0, plus [live coding, PR iteration and cancellation evidence](/sample-autonomous-cloud-coding-agents/architecture/645-p2-live-task-20260914), without manual IAM workarounds. Remaining P2 failure/recovery and permission/network checks prevent full acceptance. Local P3 foundations include suspend/resume commands, explicit VM observations, guarded persistent intent, a policy helper and approval deadlines that survive a frozen clock. Agent lifecycle hooks and supervisor/approval-handler integration remain unimplemented; automatic suspension is not enabled. This ADR defines P1–P3, not a P4. See the [current review](/sample-autonomous-cloud-coding-agents/architecture/645-p3-readiness-review) and [P3 implementation plan](/sample-autonomous-cloud-coding-agents/architecture/645-p3-implementation-plan). +> **Implementation status (2026-09-14):** P1 ([#689](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/689)) and P2 ([#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733)) are merged. The takeover branch now has a [clean deployment](/sample-autonomous-cloud-coding-agents/architecture/645-p2-clean-deployment-20260913) with bootstrap 1.7.0 and a [verified managed-image update to 2.0](/sample-autonomous-cloud-coding-agents/architecture/645-microvm-image-rebuild-20260914), plus [live coding, PR iteration and cancellation evidence](/sample-autonomous-cloud-coding-agents/architecture/645-p2-live-task-20260914), without manual IAM workarounds. Remaining P2 failure/recovery and permission/network checks prevent full acceptance. Local P3 foundations include suspend/resume commands, explicit VM observations, guarded persistent intent, a policy helper, approval deadlines that survive a frozen clock, and the [guest pause controller](/sample-autonomous-cloud-coding-agents/architecture/645-p3-guest-barrier). Agent lifecycle hooks and supervisor/approval-handler integration remain unimplemented; automatic suspension is not enabled. This ADR defines P1–P3, not a P4. See the [current review](/sample-autonomous-cloud-coding-agents/architecture/645-p3-readiness-review) and [P3 implementation plan](/sample-autonomous-cloud-coding-agents/architecture/645-p3-implementation-plan). **Status:** proposed **Date:** 2026-07-29 diff --git a/docs/verification/645-p3-guest-barrier.md b/docs/verification/645-p3-guest-barrier.md new file mode 100644 index 000000000..0b4d6c091 --- /dev/null +++ b/docs/verification/645-p3-guest-barrier.md @@ -0,0 +1,118 @@ +# #645 P3 guest pause controller + +Date: 2026-09-14. Local implementation; deployed image `2.0` still has only +ready, validate, run and terminate hooks. Automatic sleeping remains disabled. + +## What is implemented + +`agent/src/microvm_lifecycle.py` owns the local pause boundary. It does not call +AWS, authorize a human decision or extend an approval's deadline. + +- The MicroVM pipeline registers task and VM identity, then removes the context + on completion or failure. Hook closures retain that closed controller and + recheck it before allowing a tool, so registry removal cannot release late + callbacks. AgentCore/ECS do not register a MicroVM context. +- SDK pre/post/failure hooks track parallel tool calls. Suspension requires + exactly the tool currently parked for approval. Missing/duplicate identities + and known background work conservatively disable suspension. +- The approval hook registers the same `_ApprovalDeadline` created before its + database transaction. Before leaving the approval wait, it must pass the + barrier and remove the safe point. Approval/cancellation still use the existing + conditional task-state transition. +- Approval reads and heartbeat writes already in progress drain before the + checkpoint callback. New approval reads wait; heartbeat ticks skip while + paused. +- All progress writer instances report their actual write acknowledgments to + the same context. Missing tables, open circuit breakers and uncertain/failed + writes prevent suspension. A later event cannot repair a previously dropped + event, so this latch is separate from the transient failure counter. +- Suspend/resume callbacks run within a caller-supplied budget. Locks protect + local state only. Generation checks prevent a callback from committing a + transition after teardown or a superseding operation. +- A failed pre-suspend checkpoint leaves the unfrozen waiter able to proceed + and disables further suspension. An acknowledged suspend keeps the gate + closed until successful resume. Failed/timed-out/cancelled resume closes the + barrier; a late callback cannot release coding. +- `/run` and successful controller resume reseed Python's application PRNG with + `os.urandom(32)`. This does not make `random` suitable for secrets. + +Concurrent lifecycle requests receive a controlled rejection rather than +sharing ownership or holding a lock across network work. HTTP retry/duplicate +acknowledgment semantics still need integration with the future routes. + +## Why the image is not eligible for sleep yet + +The controller's checkpoint and refresh operations are currently test-supplied +callbacks. Production HTTP handlers must supply bounded, acknowledged database +checks and credential renewal before returning success. + +The worker has several independent credential consumers: + +| Consumer | Current owner | Required wake work | +|---|---|---| +| Task/approval/events/nudges and tenant S3 | `aws_session` tenant session; existing clients retain its credential object | Refresh without losing tags or leaving existing clients attached to stale credentials | +| Memory, Logs, trajectory and platform secrets | Ambient `platform_client` factory, including cached clients | Refresh the actual provider used by those clients | +| Claude Bedrock requests | Separate Claude subprocess and `awsCredentialExport` cache | Prove a blocking refresh path before its first request after sleep | +| Gateway request signing | Fresh botocore session per signing operation | Verify runtime-provider renewal and signing after wake | + +Inspection of the installed SDK `0.2.110`'s bundled Claude `2.1.191` found that +its helper cache uses wall-clock expiry but returns the previous value while +refreshing in the background. It also falls back to a one-hour cache lifetime +when the helper expiration is missing or six minutes or less away. Therefore: + +- Refreshing Python does not clear Claude's cache. +- Returning a very short helper expiration does not force a safe refresh. +- Current documentation for newer Claude default-chain caching is not evidence + that the pinned helper path has those semantics. + +Next, exercise the actual pinned CLI against a local fake Bedrock endpoint with +synthetic credentials and controlled expiry. Evaluate a supported strict +credential-provider path, such as `credential_process`, which must await renewal +before signing, or a supported explicit cache invalidation/continuation method. +Check this using the actual SDK/CLI, preserve task/user/repo attribution, and +keep provider failures from falling back to stale or unscoped credentials. +Do not patch internal minified CLI functions. + +Known background tool flags are tracked, but arbitrary shell/MCP subprocesses +may also detach work. Their safe-point behavior still requires a conservative +policy or process-level proof. Tool completion tracking alone must not be +advertised as proof that every process in the VM is idle. + +## Verification + +The agent quality gate passes: Ruff lint/format, type checking and **1,857 Python +tests** (34 added for this milestone; total branch coverage **85.02%**). The +configured Bandit high-severity gate and Vulture dead-code gate pass. A pre-existing +Vulture false positive for urllib's required `newurl` callback argument was +documented in the existing narrow allowlist; redirect behavior is unchanged. +New regressions exercise: + +- Parallel tools, original gate/deadline identity and an approval arriving during + the pause boundary. +- Cancellation winning the existing conditional resume transition. +- In-flight progress drain, failed/missing acknowledgments shared across + writers, and heartbeat/read suppression while paused. +- Credential callback failure, cancellation, timeout and late completion; + teardown during refresh; competing lifecycle requests. +- Frozen monotonic time with elapsed UTC approval deadline, fresh OS entropy and + pipeline registration/cleanup. + +These are local synchronization and existing-hook integration tests. +They do not prove actual AWS freeze/resume, credential renewal, hook HTTP +budgets, durable supervisor recovery or complete P3 acceptance. + +## Remaining integration order + +1. Prove the Claude credential path, then implement tenant/ambient refresh for + existing clients while preserving attribution. +2. Implement the production checkpoint and HTTP suspend/resume routes with + shared service budgets, duplicate-request handling and deterministic teardown. +3. Bind lifecycle capability to the image/version that launched each worker; + declare compatible hooks, initially leaving automatic suspension disabled. +4. Wire persistent intent/policy into durable supervisor polling with bounded + failure/wake recovery. Wake after committed approve/deny decisions. +5. Deploy through the normal CDK/bootstrap path and complete the P3 live matrix, + including delayed suspend, expiry, cancellation, refresh failure and rollback. + +The [implementation plan](./645-p3-implementation-plan.md) remains the complete +task list, including unfinished P2 live gates. diff --git a/docs/verification/645-p3-implementation-plan.md b/docs/verification/645-p3-implementation-plan.md index 9c602bf40..f7df5cb0c 100644 --- a/docs/verification/645-p3-implementation-plan.md +++ b/docs/verification/645-p3-implementation-plan.md @@ -72,6 +72,7 @@ is superseded by these records. - [x] Add mandatory pause/wake command methods across all three compute strategies, with explicit unsupported results and bounded MicroVM requests. - [x] Keep the original approval deadline through database writes and polling, including frozen/backward clocks; preserve decision races and cancellation. - [x] Save gate/VM-bound lifecycle intent with stale-writer protection; add explicit VM observations and a tested policy helper. +- [x] Add the guest pause controller, original-gate registration, parallel-tool tracking, progress acknowledgment tracking, heartbeat/read drain and generation-guarded wake completion; reseed the application PRNG at run and controller resume. - [ ] Add compatible agent hooks and credential/durability barriers. - [ ] Persist bounded poll/recovery counters and connect lifecycle policy to the supervisor. - [ ] Connect supervisor and approval handlers, then verify the complete P3 sleep/wake lifecycle in AWS. @@ -383,12 +384,26 @@ Initial local policy values: 30-second suspend grace, 60-second pre-deadline wak Files: `agent/src/server.py`, `hooks.py`, `task_state.py`, `aws_session.py`, `progress_writer.py`, credential-helper/cache owners, `contracts/constants.json`, and focused tests. -1. Create a per-task lifecycle context holding task/VM identity, active approval gate, durable deadline and synchronization primitives. The current server does not hand `/resume` a live gate object automatically. Register/unregister this context with the pipeline and approval hook, including failures and teardown. +**Guest controller implemented locally (2026-09-14):** `microvm_lifecycle.py` +now registers the MicroVM task, tracks SDK tool completion (including failures), +parks the exact approval and original deadline, and drains approval reads, +heartbeat calls and progress writes before a supplied checkpoint callback. +Every progress writer shares the task's acknowledgment history; a dropped event +keeps suspension disabled even after later successful writes. Failed or timed-out +wake callbacks cannot release coding through a late thread completion. The +controller and `/run` reseed `random` using fresh OS entropy. + +This is **not the completed hook implementation**. No HTTP suspend/resume routes, +image capability, production checkpoint/refresh callbacks, IAM changes or +automatic sleep have been enabled. See the [guest barrier review](./645-p3-guest-barrier.md) +for the implemented boundary, tests and remaining credential/subprocess work. + +1. **Implemented:** a per-task context holds task/VM identity, active approval and the original deadline. MicroVM background startup registers it; pipeline exit/crash unregisters it. The SDK hook removes the approval safe point before changing task state or returning a permission result. **Still required:** supply this context to the HTTP lifecycle handlers and finish duplicate-hook admission/acknowledgment behavior. 2. `/suspend` validates that the task is still parked on the intended gate. Wait for lifecycle/progress work already in progress to finish, establish an acknowledged durability barrier, and return success only within the hook budget. Existing best-effort event methods cannot establish that barrier. On timeout/write failure, report a hook failure so the coordinator keeps/reconciles the running VM. Test approval or cancellation arriving during this boundary. -3. `/resume` refreshes ambient/runtime credential providers as necessary, then ensures tenant-scoped assumed credentials are usable **with the same task/user/repo tags**. Inventory cached DynamoDB/S3/Memory/Logs clients and the Claude Bedrock credential helper; replacing one global session does not replace every already-created client or subprocess cache. Do not call the test-only `reset_session_cache()` and lose identity. Fail closed on refresh failure. +3. `/resume` refreshes ambient/runtime credential providers as necessary, then ensures tenant-scoped assumed credentials are usable **with the same task/user/repo tags**. **Inventory completed:** cached DynamoDB/S3 clients retain the tenant credential object; Memory/Logs/trajectory clients use the ambient factory; the Claude subprocess has a separate cache. Pinned Claude 2.1.191's `awsCredentialExport` refresh returns the old cached value to its first caller while renewing in the background, so an expired token can survive into a post-wake request. A shorter helper expiration alone does not fix this. Prove a supported, blocking credential-provider/continuation path for the subprocess before enabling sleep. Do not call the test-only `reset_session_cache()` and lose identity. Fail closed on refresh failure. 4. Keep the coding action blocked behind the resume barrier until refresh and gate reconciliation finish. Handle duplicate hook calls and concurrent lifecycle requests without deadlocks. Expired credentials or a slow AWS call must not hold the hook beyond its service budget. -5. Reseed the application PRNG from fresh OS entropy on **both `/run` and `/resume`**. Do not seed it with task IDs, timestamps or an image-fixed value. Continue using cryptographic randomness for secrets. Test resumed/sibling snapshot uniqueness where meaningful; do not claim that `random` becomes cryptographically safe. -6. **Polling implemented locally:** `_ApprovalDeadline` captures the original recorded UTC expiry and a monotonic cap before database writes. Remaining time is `min(monotonic_deadline - monotonic_now, created_at + timeout_s - wall_now)`, clamped at zero, including each sleep bound. **Still required:** register this same object in the lifecycle context, check it at resume and wake the existing agent-owned decision loop. Never create a fresh timeout on resume. +5. **Implemented locally:** `/run` and successful controller resume reseed the application PRNG from fresh OS entropy. The future `/resume` HTTP route must use this controller. Do not seed it with task IDs, timestamps or an image-fixed value. Continue using cryptographic randomness for secrets. Test resumed/sibling snapshot uniqueness where meaningful; do not claim that `random` becomes cryptographically safe. +6. **Polling implemented locally:** `_ApprovalDeadline` captures the original recorded UTC expiry and a monotonic cap before database writes. Remaining time is `min(monotonic_deadline - monotonic_now, created_at + timeout_s - wall_now)`, clamped at zero, including each sleep bound. **Registered locally:** the lifecycle context retains this exact object and releases the existing approval loop after successful controller resume. **Still required:** the production resume callback must reconcile the gate and check the original deadline. Never create a fresh timeout on resume. 7. **Preserved/tested locally:** conditional TIMED_OUT write, strongly consistent reread when that write loses, and late-decision winner behavior. Forward/backward clocks, frozen monotonic time, slow writes/reads, missing rows and cancellation have regression coverage. **Still required:** exercise these through the actual resume barrier and live AWS lifecycle. TTL is asynchronous garbage collection, not a precise alarm clock. 8. Declare `/suspend` and `/resume` as enabled image hooks only when the same source version serves them. Add shared hook-budget constants and route/contract assertions. Keep `/ready` and `/validate` AWS-silent. An old image without the new hooks must not be eligible for automatic suspension. Bind capability to the image/version that **actually launched each VM**, using coordinator-owned launch metadata; current deployment settings alone cannot prove an older VM supports the hooks. Unknown/legacy capability keeps new suspends off. A verified full drain can establish a clean boundary, but must not be assumed. From 585d545edf5023a2294b3bc03ce57f0ef721cffc Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Mon, 14 Sep 2026 15:49:29 -0400 Subject: [PATCH 034/149] fix(agent): renew MicroVM credentials without ambient fallback --- agent/Dockerfile | 4 +- agent/README.md | 9 + agent/scripts/verify_microvm_credentials.py | 395 ++++++++++++++++++ agent/src/aws_session.py | 102 ++++- agent/src/bedrock_creds_helper.py | 15 +- agent/src/microvm_credentials.py | 137 ++++++ agent/src/runner.py | 30 +- agent/tests/test_bedrock_creds_helper.py | 13 + agent/tests/test_microvm_credentials.py | 314 ++++++++++++++ agent/tests/test_runner.py | 79 +++- agent/tests/test_server.py | 9 +- ...ADR-021-lambda-microvms-compute-backend.md | 2 +- ...Adr-021-lambda-microvms-compute-backend.md | 2 +- docs/verification/645-p3-credentials.md | 125 ++++++ docs/verification/645-p3-guest-barrier.md | 18 +- .../645-p3-implementation-plan.md | 9 +- docs/verification/645-p3-readiness-review.md | 10 +- 17 files changed, 1235 insertions(+), 38 deletions(-) create mode 100644 agent/scripts/verify_microvm_credentials.py create mode 100644 agent/src/microvm_credentials.py create mode 100644 agent/tests/test_microvm_credentials.py create mode 100644 docs/verification/645-p3-credentials.md diff --git a/agent/Dockerfile b/agent/Dockerfile index cc280daf1..100fc5611 100644 --- a/agent/Dockerfile +++ b/agent/Dockerfile @@ -128,7 +128,9 @@ COPY agent/prepare-commit-msg.sh /app/ # Claude Code managed settings (#215). The highest-precedence settings layer — # loaded regardless of setting_sources and unoverridable by the untrusted cloned # repo's project .claude/settings.json. Carries awsCredentialExport so Bedrock -# calls use session-tagged, refreshable credentials for cost attribution. +# calls use session-tagged credentials for cost attribution. For MicroVM workers, +# the helper returns no exported keys: the parent's scoped container provider +# handles renewal without Claude's stale awsCredentialExport cache. # Placing awsCredentialExport (an arbitrary command) anywhere the target repo # can influence would be RCE with the compute role, so it lives ONLY here. COPY agent/managed-settings.json /etc/claude-code/managed-settings.json diff --git a/agent/README.md b/agent/README.md index fffb8070c..185e3b955 100644 --- a/agent/README.md +++ b/agent/README.md @@ -139,6 +139,15 @@ tenant's OAuth or Forge credential to the next task. † You need valid Bedrock credentials in the container: export keys (Option A), let `run.sh` inject keys from the AWS CLI after `aws sso login` or similar (Option B), or mount `~/.aws` (Option C). `run.sh` also sets `CLAUDE_CODE_USE_BEDROCK=1` so Claude Code uses Bedrock. +MicroVM workers configure Claude's AWS provider internally at runtime. The parent +sets `ABCA_MICROVM_CREDENTIAL_BROKER=1` **only in the Claude child**, points +`AWS_CONTAINER_CREDENTIALS_FULL_URI` at an authenticated loopback endpoint, and +clears alternate credential sources in that child. Operators should not set this +internal flag themselves. The endpoint serves the current task's scoped session; +the parent's runtime credentials and other backends' attribution path remain +separate. See [P3 credential verification](../docs/verification/645-p3-credentials.md) +for implementation and live-verification limits. + ### Examples ```bash diff --git a/agent/scripts/verify_microvm_credentials.py b/agent/scripts/verify_microvm_credentials.py new file mode 100644 index 000000000..941285a78 --- /dev/null +++ b/agent/scripts/verify_microvm_credentials.py @@ -0,0 +1,395 @@ +#!/usr/bin/env python3 +# Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. +# SPDX-License-Identifier: MIT-0 + +"""Opt-in pinned Claude SDK/CLI credential probe; model and credentials are fake. + +Run with agent/.venv/bin/python. This uses a loopback Bedrock event-stream server, +a temporary AWS profile and synthetic keys. It makes no paid model invocation. +It does not simulate a real VM snapshot or establish the AWS runtime provider's +refresh behavior. Keep it out of the ordinary unit suite. +""" + +from __future__ import annotations + +import argparse +import asyncio +import base64 +import importlib.metadata +import json +import os +import re +import shlex +import struct +import subprocess +import sys +import tempfile +import threading +import time +import zlib +from datetime import UTC, datetime +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer +from pathlib import Path +from typing import TypedDict + +MODEL = "us.anthropic.claude-sonnet-4-20250514-v1:0" + + +class ProbeState(TypedDict): + key: str + expiry: float + fail: bool + + +def event_frame(event: dict) -> bytes: + """Encode one AWS event-stream chunk containing an Anthropic streaming event.""" + headers = bytearray() + for name, value in { + ":event-type": "chunk", + ":content-type": "application/json", + ":message-type": "event", + }.items(): + key, val = name.encode(), value.encode() + headers += bytes([len(key)]) + key + b"\x07" + struct.pack(">H", len(val)) + val + body = json.dumps({"bytes": base64.b64encode(json.dumps(event).encode()).decode()}).encode() + prelude = struct.pack(">II", 16 + len(headers) + len(body), len(headers)) + frame = prelude + struct.pack(">I", zlib.crc32(prelude)) + headers + body + return frame + struct.pack(">I", zlib.crc32(frame)) + + +def model_response() -> bytes: + events = [ + { + "type": "message_start", + "message": { + "id": "msg_offline", + "type": "message", + "role": "assistant", + "model": MODEL, + "content": [], + "stop_reason": None, + "stop_sequence": None, + "usage": {"input_tokens": 1, "output_tokens": 0}, + }, + }, + {"type": "content_block_start", "index": 0, "content_block": {"type": "text", "text": ""}}, + { + "type": "content_block_delta", + "index": 0, + "delta": {"type": "text_delta", "text": "Offline credential probe."}, + }, + {"type": "content_block_stop", "index": 0}, + { + "type": "message_delta", + "delta": {"stop_reason": "end_turn", "stop_sequence": None}, + "usage": {"output_tokens": 1}, + }, + {"type": "message_stop"}, + ] + return b"".join(event_frame(event) for event in events) + + +HELPER = """\ +import json, sys +from datetime import UTC, datetime +from pathlib import Path +state = json.loads(Path(sys.argv[1]).read_text()) +if state["fail"]: + print("synthetic credential renewal failure", file=sys.stderr) + raise SystemExit(7) +creds = { + "AccessKeyId": state["key"], + "SecretAccessKey": "synthetic-secret-not-valid-in-aws", + "SessionToken": "synthetic-session-not-valid-in-aws", + "Expiration": datetime.fromtimestamp(state["expiry"], UTC).isoformat(), +} +with open(sys.argv[2], "a") as log: + log.write(json.dumps({"key": state["key"]}) + "\\n") +print(json.dumps({"Credentials": creds} if sys.argv[3] == "export" else {"Version": 1, **creds})) +""" + + +async def probe(mode: str, *, ambient_fallback: bool) -> dict: + import claude_agent_sdk + from claude_agent_sdk import ClaudeAgentOptions, ClaudeSDKClient, ResultMessage + + sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "src")) + from microvm_credentials import ScopedCredentialBroker + from microvm_lifecycle import MicrovmLifecycle + + requests: list[dict] = [] + initial_expiry = int(time.time()) + 12 + state: ProbeState = {"key": "SYNTHETIC_INITIAL", "expiry": initial_expiry, "fail": False} + stream = model_response() + sdk_version = importlib.metadata.version("claude-agent-sdk") + cli = Path(claude_agent_sdk.__file__).parent / "_bundled" / "claude" + cli_version = subprocess.run( + [str(cli), "--version"], capture_output=True, text=True, check=True, timeout=5 + ).stdout.strip() + if sdk_version != "0.2.110" or cli_version != "2.1.191 (Claude Code)": + raise RuntimeError("Probe expectations require reviewed SDK 0.2.110 / Claude 2.1.191 pins") + + class Handler(BaseHTTPRequestHandler): + def log_message(self, format, *args): + del format, args + + def do_GET(self): + if self.path != "/credentials": + self.send_error(404) + return + body = json.dumps( + { + "AccessKeyId": "SYNTHETIC_AMBIENT", + "SecretAccessKey": "synthetic-ambient-secret", + "Token": "synthetic-ambient-token", + "SessionToken": "synthetic-ambient-token", + "Expiration": datetime.fromtimestamp(time.time() + 3600, UTC).strftime( + "%Y-%m-%dT%H:%M:%SZ" + ), + } + ).encode() + requests.append({"kind": "ambient-provider"}) + self.send_response(200) + self.send_header("Content-Type", "application/json") + self.send_header("Content-Length", str(len(body))) + self.end_headers() + self.wfile.write(body) + + def do_POST(self): + self.rfile.read(int(self.headers.get("Content-Length", "0"))) + key = re.search(r"Credential=([^/]+)", self.headers.get("Authorization", "")) + requests.append( + { + "kind": "model", + "key": key.group(1) if key else "missing", + "phase": state["key"], + "at": time.time(), + "path": self.path, + } + ) + self.send_response(200) + self.send_header("Content-Type", "application/vnd.amazon.eventstream") + self.send_header("Content-Length", str(len(stream))) + self.end_headers() + self.wfile.write(stream) + + server = ThreadingHTTPServer(("127.0.0.1", 0), Handler) + server_thread = threading.Thread(target=server.serve_forever, daemon=True) + server_thread.start() + endpoint = f"http://127.0.0.1:{server.server_port}" + broker = None + try: + with tempfile.TemporaryDirectory(prefix="abca-645-credentials-") as directory: + temp = Path(directory) + helper, state_file, audit = ( + temp / name for name in ("helper.py", "state.json", "audit") + ) + helper.write_text(HELPER) + state_file.write_text(json.dumps(state)) + settings = temp / "settings.json" + provider = ( + "export" + if mode == "export" + else "none" + if mode == "ambient" or mode.startswith("broker") + else "process" + ) + command = shlex.join( + [sys.executable, str(helper), str(state_file), str(audit), provider] + ) + export_command = ( + shlex.join( + [ + sys.executable, + str(Path(__file__).resolve().parents[1] / "src/bedrock_creds_helper.py"), + ] + ) + if mode.startswith("broker") + else command + ) + settings.write_text( + json.dumps( + {"awsCredentialExport": export_command} + if mode == "export" or mode.startswith("broker") + else {} + ) + ) + config = temp / "aws-config" + config.write_text( + "[profile probe]\nregion = us-west-2\n" + + (f"credential_process = {command}\n" if provider == "process" else "") + ) + credentials = temp / "aws-credentials" + credentials.write_text("") + config_dir = temp / "claude-config" + config_dir.mkdir() + child_env = { + "CLAUDE_CONFIG_DIR": str(config_dir), + "CLAUDE_CODE_USE_BEDROCK": "1", + "CLAUDE_CODE_DISABLE_NONESSENTIAL_TRAFFIC": "1", + "CLAUDE_CODE_MAX_RETRIES": "0", + "DISABLE_TELEMETRY": "1", + "DISABLE_ERROR_REPORTING": "1", + "DISABLE_AUTOUPDATER": "1", + "ANTHROPIC_BEDROCK_BASE_URL": endpoint, + "AWS_ENDPOINT_URL": endpoint, + "AWS_REGION": "us-west-2", + "AWS_DEFAULT_REGION": "us-west-2", + "AWS_PROFILE": "probe", + "AWS_CONFIG_FILE": str(config), + "AWS_SHARED_CREDENTIALS_FILE": str(credentials), + "AWS_EC2_METADATA_DISABLED": "true", + "AWS_CONTAINER_CREDENTIALS_FULL_URI": endpoint + "/credentials" + if ambient_fallback or mode.startswith("broker") + else "", + "ABCA_MICROVM_CREDENTIAL_BROKER": "1" if mode.startswith("broker") else "", + "AWS_METADATA_SERVICE_TIMEOUT": "1", + "AWS_METADATA_SERVICE_NUM_ATTEMPTS": "1", + "AWS_MAX_ATTEMPTS": "1", + } + if mode.startswith("broker"): + + def scoped_provider(): + requests.append( + {"kind": "broker-failure" if state["fail"] else "broker-provider"} + ) + if state["fail"]: + raise RuntimeError("synthetic broker renewal failure") + return { + "AccessKeyId": state["key"], + "SecretAccessKey": "synthetic-scoped-secret", + "Token": "synthetic-scoped-token", + "Expiration": datetime.fromtimestamp(state["expiry"], UTC).strftime( + "%Y-%m-%dT%H:%M:%SZ" + ), + } + + broker = ScopedCredentialBroker( + MicrovmLifecycle("probe", "probe-vm"), provider=scoped_provider + ) + child_env.update(broker.environment) + errors: list[str] = [] + options = ClaudeAgentOptions( + model=MODEL, + max_turns=1, + cwd=directory, + setting_sources=[], + settings=str(settings), + env=child_env, + stderr=errors.append, + ) + results = [] + async with ClaudeSDKClient(options=options) as client: + for phase in ("initial", "after-expiry"): + if phase == "after-expiry": + await asyncio.sleep(max(0, state["expiry"] - time.time()) + 0.3) + state.update( + key="SYNTHETIC_RENEWED", + expiry=time.time() + 3600, + fail=mode in {"process-failure", "broker-failure"}, + ) + pending = state_file.with_suffix(".pending") + pending.write_text(json.dumps(state)) + pending.replace(state_file) + await client.query("Return one short sentence. Do not call tools.") + async for message in client.receive_response(): + if isinstance(message, ResultMessage): + results.append( + { + "phase": phase, + "error": message.is_error, + **({"detail": message.result} if message.is_error else {}), + } + ) + issued = ( + [json.loads(line) for line in audit.read_text().splitlines()] + if audit.exists() + else [] + ) + return { + "mode": mode, + "sdk_version": sdk_version, + "cli_version": cli_version, + "ambient_fallback": ambient_fallback, + "initial_expiry": initial_expiry, + "requests": requests, + "issued": issued, + "results": results, + "stderr_lines": len(errors), + } + finally: + if broker is not None: + broker.close() + server.shutdown() + server.server_close() + server_thread.join(timeout=2) + + +def verify(result: dict) -> None: + """Positive controls prevent a broken fake endpoint from proving safety.""" + model = [ + request + for request in result["requests"] + if request.get("path") == f"/model/{MODEL}/invoke-with-response-stream" + ] + initial = [request for request in model if request["phase"] == "SYNTHETIC_INITIAL"] + after = [request for request in model if request["phase"] == "SYNTHETIC_RENEWED"] + mode = result["mode"] + expected_initial = "SYNTHETIC_AMBIENT" if mode == "ambient" else "SYNTHETIC_INITIAL" + if ( + len(initial) != 1 + or initial[0]["key"] != expected_initial + or initial[0]["at"] >= result["initial_expiry"] + ): + raise RuntimeError("Initial query must succeed with the expected key before its expiry") + failed = mode == "broker-failure" or ( + mode == "process-failure" and not result["ambient_fallback"] + ) + if [item["error"] for item in result["results"]] != [False, failed]: + raise RuntimeError("Unexpected SDK result; credential-path proof failed") + if failed: + if after: + raise RuntimeError("Renewal failure allowed a model request") + else: + expected = ( + "SYNTHETIC_INITIAL" + if mode == "export" + else "SYNTHETIC_AMBIENT" + if mode in {"ambient", "process-failure"} + else "SYNTHETIC_RENEWED" + ) + if ( + len(after) != 1 + or after[0]["key"] != expected + or after[0]["at"] <= result["initial_expiry"] + ): + raise RuntimeError("Post-expiry query did not use the expected credential path") + result["verified"] = True + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument( + "--mode", + choices=["export", "process", "process-failure", "ambient", "broker", "broker-failure"], + required=True, + ) + parser.add_argument("--ambient-fallback", action="store_true") + args = parser.parse_args() + if args.mode == "ambient" and not args.ambient_fallback: + parser.error("ambient positive control requires --ambient-fallback") + # This process exists only for the probe. Do not inherit operator credentials + # or auth/telemetry settings into the real CLI. HOME is never reassigned. + for key in tuple(os.environ): + if key.startswith(("AWS_", "ANTHROPIC_", "CLAUDE_", "OTEL_", "BEDROCK_")): + del os.environ[key] + result = asyncio.run( + asyncio.wait_for(probe(args.mode, ambient_fallback=args.ambient_fallback), 45) + ) + verify(result) + print(json.dumps(result, indent=2)) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/agent/src/aws_session.py b/agent/src/aws_session.py index 89e926546..8c05da9e8 100644 --- a/agent/src/aws_session.py +++ b/agent/src/aws_session.py @@ -48,6 +48,7 @@ import os import threading +from datetime import UTC from typing import Any # Env var holding the per-task SessionRole ARN. Set by the orchestrator on the @@ -64,6 +65,8 @@ _lock = threading.Lock() _session: Any = None # cached boto3.Session (scoped or plain) _scoped: bool | None = None # None until first resolution; True if tag-scoped +_ambient_lock = threading.Lock() +_ambient_credentials: dict[int, Any] = {} # Session-tag values, set once at startup by ``configure_session`` from the # resolved TaskConfig. Kept in private module state — NOT os.environ — so the @@ -105,11 +108,15 @@ def configure_session(user_id: str, repo: str, task_id: str) -> None: spawned subprocesses. """ global _tags - _tags = { + tags = { key: value for key, value in (("user_id", user_id), ("repo", repo), ("task_id", task_id)) if value } + with _lock: + if _session is not None and _tags != tags: + raise SessionScopingError("Cannot change identity after creating the task session") + _tags = tags def reset_session_cache() -> None: @@ -119,6 +126,8 @@ def reset_session_cache() -> None: _session = None _scoped = None _tags = {} + with _ambient_lock: + _ambient_credentials.clear() def _session_tags() -> list[dict[str, str]]: @@ -144,12 +153,14 @@ def _build_scoped_session(role_arn: str) -> Any: running past the 1-hour role-chaining cap keeps working. """ import boto3 + from botocore.config import Config from botocore.credentials import ( DeferredRefreshableCredentials, ) from botocore.session import get_session as get_botocore_session import ua + from microvm_lifecycle import get_context region = os.environ.get("AWS_REGION") or os.environ.get("AWS_DEFAULT_REGION") task_id = _tags.get("task_id", "") @@ -163,14 +174,21 @@ def _build_scoped_session(role_arn: str) -> Any: # This is the role-chaining caller; the assumed SessionRole credentials it # returns must NOT be used to build it, or refresh would recurse. Carries # the static md/ UA segment so the assume-role call is attributed too. - sts_client = boto3.client("sts", region_name=region, config=ua.client_config()) + sts_config = ( + Config(connect_timeout=2, read_timeout=2, retries={"total_max_attempts": 1}) + if get_context(task_id) is not None + else Config() + ) + sts_client = platform_client("sts", region_name=region, config=sts_config) + # A retained client's refresh must never pick up another task's identity. + tags = _session_tags() def _refresh() -> dict[str, str]: resp = sts_client.assume_role( RoleArn=role_arn, RoleSessionName=session_name, DurationSeconds=_CHAINED_SESSION_DURATION_S, - Tags=_session_tags(), + Tags=tags, ) creds = resp["Credentials"] return { @@ -334,4 +352,80 @@ def platform_client(service_name: str, **kwargs: Any) -> Any: """ import boto3 - return boto3.client(service_name, **_merge_ua_config(kwargs)) + client = boto3.client(service_name, **_merge_ua_config(kwargs)) + credentials = client._request_signer._credentials + with _ambient_lock: + _ambient_credentials[id(credentials)] = credentials + return client + + +def _microvm_scoped_credentials(task_id: str) -> Any: + """Require the existing task identity; never reset/rebuild a cached session.""" + session = get_session() + with _lock: + if not _scoped or not task_id or _tags.get("task_id") != task_id: + raise SessionScopingError("MicroVM credentials require the original scoped task") + return session.get_credentials() + + +def _locked_refresh(credentials: Any, *, force: bool) -> dict[str, str]: + """Refresh and export one coherent key/expiry pair on the retained object. + + Botocore has no public forced-refresh operation. Keep its lock/private API + use here and exercise it against the installed botocore in regression tests. + A mandatory refresh propagates errors even while the old keys remain valid. + The enclosing lifecycle callback supplies the overall wall-clock budget. + """ + from botocore.credentials import RefreshableCredentials + + if not isinstance(credentials, RefreshableCredentials): + # Static env keys cannot prove renewal after sleep. Do not invent a TTL + # or claim that rereading the same environment refreshed the runtime role. + raise SessionScopingError("MicroVM resume requires a refreshable credential provider") + if not credentials._refresh_lock.acquire(timeout=2): + raise TimeoutError("Credential refresh lock did not become available") + try: + if force or credentials.refresh_needed(): + credentials._protected_refresh(is_mandatory=True) + frozen = credentials._frozen_credentials + expiry = credentials._expiry_time + if ( + frozen is None + or not frozen.access_key + or not frozen.secret_key + or not frozen.token + or expiry is None + or credentials._is_expired() + ): + raise SessionScopingError("Credential provider did not return usable temporary keys") + return { + "AccessKeyId": frozen.access_key, + "SecretAccessKey": frozen.secret_key, + "Token": frozen.token, + "Expiration": expiry.astimezone(UTC).strftime("%Y-%m-%dT%H:%M:%SZ"), + } + finally: + credentials._refresh_lock.release() + + +def export_microvm_credentials(task_id: str) -> dict[str, str]: + """Serve only the original task's temporary keys to the local Claude broker.""" + return _locked_refresh(_microvm_scoped_credentials(task_id), force=False) + + +def refresh_microvm_credentials(task_id: str) -> None: + """Refresh ambient callers first, then the same tag-scoped tenant object. + + Call only behind the closed guest resume barrier. Cached DynamoDB/S3 and + platform clients keep their credential references; replacing a boto3 session + would leave those references stale. Unknown/static ambient providers fail + closed until their runtime renewal path has been established. + """ + tenant = _microvm_scoped_credentials(task_id) + with _ambient_lock: + ambient = tuple(_ambient_credentials.values()) + if not ambient: + raise SessionScopingError("No runtime credential provider was recorded") + for credentials in ambient: + _locked_refresh(credentials, force=True) + _locked_refresh(tenant, force=True) diff --git a/agent/src/bedrock_creds_helper.py b/agent/src/bedrock_creds_helper.py index 40ad13d70..b7f0b9f01 100644 --- a/agent/src/bedrock_creds_helper.py +++ b/agent/src/bedrock_creds_helper.py @@ -9,8 +9,9 @@ returned credentials. With a real ``Expiration`` it re-runs ~5 min before expiry during active use. Pinned Claude 2.1.191 returns its previous cached value while that refresh runs in the background: this is NOT an acknowledged -refresh barrier after MicroVM sleep. P3 must cover that subprocess cache before -enabling suspension; refreshing Python's boto3 session does not clear it. +refresh barrier after MicroVM sleep. MicroVM workers instead leave this export +empty and use the scoped container-credential provider installed by the parent. +That provider renews synchronously before signing an expired-key request. Goal: assume the per-task SessionRole with ``{user_id, repo, task_id}`` STS session tags so Bedrock spend is attributable per user/repo in AWS Cost @@ -18,7 +19,7 @@ the cost-allocation tags). The same role already carries the tenant-data grants; Track-1 only adds ``bedrock:InvokeModel*`` to it (see ``agent-session-role.ts``). -**Fails OPEN.** Bedrock attribution is a billing/observability control, not a +**Other backends fail OPEN.** Bedrock attribution is a billing/observability control, not a tenant-isolation one (contrast ``aws_session.py``, which fails closed). If the attribution config is absent or the assume-role fails, this helper emits the **ambient** compute-role credentials so Bedrock keeps working untagged — losing @@ -125,7 +126,13 @@ def _ambient_credentials() -> dict[str, str]: def resolve_credentials() -> dict[str, str]: - """Return tagged assumed-role creds, or ambient creds on any failure.""" + """Use the MicroVM default provider, otherwise retain attribution fallback.""" + # The parent supplies a single scoped loopback provider and removes other + # credential sources from the MicroVM Claude child. Exporting even those + # scoped credentials here would reintroduce Claude's stale export cache. + # Do not read the attribution file or initialize boto3 in this branch. + if os.environ.get("ABCA_MICROVM_CREDENTIAL_BROKER") == "1": + return {} path = attribution_file_path() try: with open(path) as fh: diff --git a/agent/src/microvm_credentials.py b/agent/src/microvm_credentials.py new file mode 100644 index 000000000..d6ed1c06e --- /dev/null +++ b/agent/src/microvm_credentials.py @@ -0,0 +1,137 @@ +# Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. +# SPDX-License-Identifier: MIT-0 + +"""Runtime-only scoped credential endpoint for the MicroVM Claude subprocess. + +AWS's container provider waits for replacement keys before signing after expiry. +The managed export helper deliberately returns no keys in this mode. The child +has no alternate AWS provider; the parent's runtime credential chain is untouched. +This protects provider selection, not isolation from code running as the same OS +user. Never start this server during image warm-up or snapshot validation. +""" + +from __future__ import annotations + +import hmac +import json +import os +import secrets +import threading +from http.server import BaseHTTPRequestHandler, HTTPServer +from typing import TYPE_CHECKING + +from aws_session import export_microvm_credentials + +if TYPE_CHECKING: + from collections.abc import Callable + + from microvm_lifecycle import MicrovmLifecycle + + +class ScopedCredentialBroker: + """One task's loopback endpoint, owned and closed by its Claude session.""" + + def __init__( + self, + lifecycle: MicrovmLifecycle, + *, + provider: Callable[[], dict[str, str]] | None = None, + ) -> None: + self._closed = threading.Event() + token = secrets.token_urlsafe(32) + resolve = provider or (lambda: export_microvm_credentials(lifecycle.task_id)) + closed = self._closed + + class Handler(BaseHTTPRequestHandler): + def log_message(self, format, *args): + # Authorization headers and response bodies contain credentials. + del format, args + + def do_GET(self): + if self.path != "/credentials": + self.send_error(404) + return + if not hmac.compare_digest(self.headers.get("Authorization", ""), token): + self.send_error(403) + return + try: + # Count the entire response in the suspend drain. Paused or + # failed lifecycle controllers reject before consulting AWS. + with lifecycle.activity(): + if closed.is_set(): + raise RuntimeError("Credential broker is closed") + body = json.dumps(resolve()).encode() + if closed.is_set(): + raise RuntimeError("Credential broker closed during renewal") + self.send_response(200) + self.send_header("Content-Type", "application/json") + self.send_header("Cache-Control", "no-store") + self.send_header("Content-Length", str(len(body))) + self.end_headers() + self.wfile.write(body) + except Exception: + # Deliberately omit provider exception text/credentials. + # The caller fails closed; no ambient fallback is available. + self.send_error(503, "Scoped credentials unavailable") + + class Server(HTTPServer): + def get_request(self): + connection, address = super().get_request() + connection.settimeout(2) + return connection, address + + def handle_error(self, request, client_address): + # A disconnected client may interrupt the generic 503 response. + # No request/exception dumping from this credential endpoint. + del request, client_address + + self._server = Server(("127.0.0.1", 0), Handler) + self._thread = threading.Thread( + target=self._server.serve_forever, + kwargs={"poll_interval": 0.05}, + name="microvm-scoped-credentials", + daemon=True, + ) + # ClaudeAgentOptions.env overlays the parent environment; empty strings + # suppress inherited providers. Controlled empty files also suppress + # ~/.aws profile/SSO/process credentials without changing HOME. + self.environment = { + key: "" + for key in os.environ + if key.startswith("AWS_") + and key not in {"AWS_REGION", "AWS_DEFAULT_REGION", "AWS_SDK_UA_APP_ID"} + } + self.environment.update( + { + "AWS_ACCESS_KEY_ID": "", + "AWS_SECRET_ACCESS_KEY": "", + "AWS_SESSION_TOKEN": "", + "AWS_PROFILE": "", + "AWS_DEFAULT_PROFILE": "", + "AWS_CONFIG_FILE": os.devnull, + "AWS_SHARED_CREDENTIALS_FILE": os.devnull, + "AWS_WEB_IDENTITY_TOKEN_FILE": "", + "AWS_ROLE_ARN": "", + "AWS_CONTAINER_CREDENTIALS_RELATIVE_URI": "", + "AWS_CONTAINER_CREDENTIALS_FULL_URI": ( + f"http://127.0.0.1:{self._server.server_port}/credentials" + ), + "AWS_CONTAINER_AUTHORIZATION_TOKEN": token, + "AWS_CONTAINER_AUTHORIZATION_TOKEN_FILE": "", + "AWS_EC2_METADATA_DISABLED": "true", + "AWS_BEARER_TOKEN_BEDROCK": "", + "ANTHROPIC_API_KEY": "", + "ANTHROPIC_AUTH_TOKEN": "", + "ABCA_MICROVM_CREDENTIAL_BROKER": "1", + } + ) + self._thread.start() + + def close(self) -> None: + """Stop serving keys, then stop the loop before the task is torn down.""" + if self._closed.is_set(): + return + self._closed.set() + self._server.shutdown() + self._server.server_close() + self._thread.join(timeout=2) diff --git a/agent/src/runner.py b/agent/src/runner.py index d4d83c8d3..45743be30 100644 --- a/agent/src/runner.py +++ b/agent/src/runner.py @@ -638,12 +638,20 @@ def _on_stderr(line: str) -> None: # standalone query() function. This matches the official AWS sample: # https://github.com/aws-samples/sample-deploy-ClaudeAgentSDK-based-agents-to-AgentCore-Runtime client = ClaudeSDKClient(options=options) - log("AGENT", "Connecting to Claude Code CLI subprocess...") - await client.connect() - log("AGENT", "Connected. Sending prompt...") - await client.query(prompt=prompt) - log("AGENT", "Prompt sent. Receiving messages...") + from microvm_credentials import ScopedCredentialBroker + from microvm_lifecycle import get_context + + broker = None try: + lifecycle = get_context(config.task_id or "") + if lifecycle is not None: + broker = ScopedCredentialBroker(lifecycle) + options.env.update(broker.environment) + log("AGENT", "Connecting to Claude Code CLI subprocess...") + await client.connect() + log("AGENT", "Connected. Sending prompt...") + await client.query(prompt=prompt) + log("AGENT", "Prompt sent. Receiving messages...") async for message in client.receive_response(): if isinstance(message, SystemMessage): message_counts["system"] += 1 @@ -864,13 +872,21 @@ def _on_stderr(line: str) -> None: # see it on the dashboard widget + ``bgagent status`` and not # just on the runtime-DEFAULT stream. log_error_cw( - f"Exception during receive_response(): {type(e).__name__}: {e}", + f"Exception during Claude session: {type(e).__name__}: {e}", task_id=config.task_id or None, ) progress.write_agent_error(error_type=type(e).__name__, message=str(e)) if result.status == "unknown": result.status = "error" - result.error = f"receive_response() failed: {e}" + result.error = f"Claude session failed: {e}" + finally: + # Also cover startup/query failure and cancellation, before pipeline + # teardown removes the task's lifecycle registry entry. + try: + await client.disconnect() + finally: + if broker is not None: + broker.close() log("AGENT", f"Generator finished. Messages received: {message_counts}") log("AGENT", f"CLI stderr lines received: {stderr_line_count}") diff --git a/agent/tests/test_bedrock_creds_helper.py b/agent/tests/test_bedrock_creds_helper.py index 23f4c2a7d..e2f242935 100644 --- a/agent/tests/test_bedrock_creds_helper.py +++ b/agent/tests/test_bedrock_creds_helper.py @@ -32,6 +32,19 @@ def attr_file(tmp_path, monkeypatch): return path +def test_microvm_export_is_empty_without_reading_files_or_resolving_ambient(monkeypatch, capsys): + monkeypatch.setenv("ABCA_MICROVM_CREDENTIAL_BROKER", "1") + with ( + patch("builtins.open", side_effect=AssertionError("must not read attribution")), + patch.object( + helper, "_ambient_credentials", side_effect=AssertionError("must not resolve") + ), + patch("boto3.client", side_effect=AssertionError("must not call STS")), + ): + assert helper.main() == 0 + assert json.loads(capsys.readouterr().out) == {"Credentials": {}} + + def test_write_attribution_file_is_0600(attr_file): tags = build_session_tags("u1", "owner/repo", "task123") written = helper.write_attribution_file("arn:aws:iam::1:role/SR", tags, attr_file) diff --git a/agent/tests/test_microvm_credentials.py b/agent/tests/test_microvm_credentials.py new file mode 100644 index 000000000..baaf888b1 --- /dev/null +++ b/agent/tests/test_microvm_credentials.py @@ -0,0 +1,314 @@ +# Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. +# SPDX-License-Identifier: MIT-0 + +"""Real botocore objects and loopback HTTP exercise the credential boundary.""" + +from __future__ import annotations + +import asyncio +import json +import threading +from datetime import UTC, datetime, timedelta, timezone +from unittest.mock import MagicMock +from urllib.error import HTTPError +from urllib.request import ProxyHandler, Request, build_opener + +import pytest +from botocore.config import Config +from botocore.credentials import Credentials, DeferredRefreshableCredentials + +import aws_session +from microvm_credentials import ScopedCredentialBroker +from microvm_lifecycle import LifecycleUnavailable, MicrovmLifecycle + + +@pytest.fixture(autouse=True) +def _reset(): + aws_session.reset_session_cache() + yield + aws_session.reset_session_cache() + + +@pytest.fixture +def anyio_backend(): + return "asyncio" + + +def _metadata(key: str) -> dict[str, str]: + return { + "access_key": key, + "secret_key": "synthetic-secret", + "token": "synthetic-token", + "expiry_time": (datetime.now(UTC) + timedelta(hours=1)).isoformat(), + } + + +def _envelope() -> dict[str, str]: + return { + "AccessKeyId": "SYNTHETIC", + "SecretAccessKey": "synthetic-secret", + "Token": "synthetic-token", + "Expiration": (datetime.now(UTC) + timedelta(hours=1)).strftime("%Y-%m-%dT%H:%M:%SZ"), + } + + +def _fetch(broker: ScopedCredentialBroker, *, token: str | None = None, path: str = ""): + url = broker.environment["AWS_CONTAINER_CREDENTIALS_FULL_URI"] + path + assert url.startswith("http://127.0.0.1:") + request = Request(url) # noqa: S310 — endpoint comes from this test's loopback broker + if token is not None: + request.add_header("Authorization", token) + with build_opener(ProxyHandler({})).open(request, timeout=3) as response: + assert response.headers["Cache-Control"] == "no-store" + return json.load(response) + + +class TestRetainedCredentials: + def _configure(self, monkeypatch): + monkeypatch.setenv( + aws_session.SESSION_ROLE_ARN_ENV, "arn:aws:iam::123456789012:role/session" + ) + monkeypatch.setenv("AWS_REGION", "us-west-2") + aws_session.configure_session("user", "owner/repo", "task") + order = [] + + def ambient_refresh(): + order.append("ambient") + return _metadata(f"AMBIENT_{len(order)}") + + ambient = DeferredRefreshableCredentials( + method="container-role", refresh_using=ambient_refresh + ) + sts = MagicMock() + sts._request_signer._credentials = ambient + + def assume(**_kwargs): + ambient.get_frozen_credentials() + order.append("tenant") + return { + "Credentials": { + "AccessKeyId": f"TENANT_{sts.assume_role.call_count}", + "SecretAccessKey": "synthetic-secret", + "SessionToken": "synthetic-token", + "Expiration": datetime.now(UTC) + timedelta(hours=1), + } + } + + sts.assume_role.side_effect = assume + monkeypatch.setattr("boto3.client", lambda *_a, **_kw: sts) + return sts, ambient, order + + def test_resume_refreshes_retained_clients_in_order_with_identical_tags(self, monkeypatch): + sts, ambient, order = self._configure(monkeypatch) + client = aws_session.tenant_client("s3", config=Config(signature_version="s3v4")) + resource = aws_session.tenant_resource("dynamodb") + original = aws_session.get_session().get_credentials() + before = aws_session.export_microvm_credentials("task") + tags = sts.assume_role.call_args.kwargs["Tags"] + assert order == ["ambient", "tenant"] + order.clear() + + aws_session.refresh_microvm_credentials("task") + + assert order == ["ambient", "tenant"] + assert client._request_signer._credentials is original + assert resource.meta.client._request_signer._credentials is original + assert sts._request_signer._credentials is ambient + assert aws_session.get_session().get_credentials() is original + assert ( + sts.assume_role.call_args.kwargs["Tags"] + == tags + == [ + {"Key": "user_id", "Value": "user"}, + {"Key": "repo", "Value": "owner/repo"}, + {"Key": "task_id", "Value": "task"}, + ] + ) + after = aws_session.export_microvm_credentials("task") + assert before["AccessKeyId"] != after["AccessKeyId"] + signed = client.generate_presigned_url( + "get_object", Params={"Bucket": "synthetic", "Key": "synthetic"} + ) + assert after["AccessKeyId"] in signed + assert before["AccessKeyId"] not in signed + + @pytest.mark.parametrize("microvm", [False, True]) + def test_short_sts_network_budget_applies_only_to_microvm(self, monkeypatch, microvm): + import microvm_lifecycle + + sts, _, _ = self._configure(monkeypatch) + make_client = MagicMock(return_value=sts) + monkeypatch.setattr("boto3.client", make_client) + context = microvm_lifecycle.register_task("task", "vm") if microvm else None + try: + aws_session.export_microvm_credentials("task") + config = make_client.call_args.kwargs["config"] + assert config.connect_timeout == (2 if microvm else 60) + assert config.read_timeout == (2 if microvm else 60) + assert config.retries == ({"total_max_attempts": 1} if microvm else None) + finally: + if context is not None: + microvm_lifecycle.unregister_task(context) + + def test_failed_ambient_refresh_never_attempts_tenant_renewal(self, monkeypatch): + sts, ambient, _ = self._configure(monkeypatch) + aws_session.export_microvm_credentials("task") + count = sts.assume_role.call_count + ambient._refresh_using = MagicMock(side_effect=RuntimeError("synthetic outage")) + with pytest.raises(RuntimeError, match="synthetic outage"): + aws_session.refresh_microvm_credentials("task") + assert sts.assume_role.call_count == count + + def test_failed_tenant_refresh_is_mandatory_even_before_expiry(self, monkeypatch): + sts, _, _ = self._configure(monkeypatch) + aws_session.export_microvm_credentials("task") + sts.assume_role.side_effect = RuntimeError("synthetic denial") + with pytest.raises(RuntimeError, match="synthetic denial"): + aws_session.refresh_microvm_credentials("task") + + def test_identity_cannot_change_after_clients_exist(self, monkeypatch): + self._configure(monkeypatch) + aws_session.export_microvm_credentials("task") + aws_session.configure_session("user", "owner/repo", "task") + with pytest.raises(aws_session.SessionScopingError, match="Cannot change identity"): + aws_session.configure_session("another", "owner/other", "another-task") + with pytest.raises(aws_session.SessionScopingError, match="original scoped task"): + aws_session.export_microvm_credentials("another-task") + assert aws_session._tags["user_id"] == "user" + + def test_static_runtime_credentials_cannot_claim_wake_renewal(self, monkeypatch): + sts, _, _ = self._configure(monkeypatch) + sts._request_signer._credentials = Credentials("STATIC", "synthetic") + aws_session.export_microvm_credentials("task") + with pytest.raises(aws_session.SessionScopingError, match="refreshable"): + aws_session.refresh_microvm_credentials("task") + + def test_absent_scoping_or_ambient_provider_is_rejected(self, monkeypatch): + self._configure(monkeypatch) + aws_session.export_microvm_credentials("task") + aws_session._ambient_credentials.clear() + with pytest.raises(aws_session.SessionScopingError, match="No runtime"): + aws_session.refresh_microvm_credentials("task") + monkeypatch.setattr(aws_session, "_scoped", False) + with pytest.raises(aws_session.SessionScopingError, match="original scoped task"): + aws_session.export_microvm_credentials("task") + + def test_expired_replacement_is_rejected(self): + metadata = _metadata("EXPIRED") + metadata["expiry_time"] = (datetime.now(UTC) - timedelta(seconds=1)).isoformat() + credentials = DeferredRefreshableCredentials(method="test", refresh_using=lambda: metadata) + with pytest.raises(RuntimeError, match="still expired"): + aws_session._locked_refresh(credentials, force=True) + + def test_expiration_is_utc_and_keys_are_a_coherent_pair(self): + metadata = _metadata("PAIR") + metadata["expiry_time"] = ( + datetime.now(UTC).astimezone(timezone(timedelta(hours=5))) + timedelta(hours=1) + ).isoformat() + credentials = DeferredRefreshableCredentials(method="test", refresh_using=lambda: metadata) + result = aws_session._locked_refresh(credentials, force=False) + expiry = datetime.fromisoformat(result["Expiration"]) + assert 3598 < (expiry - datetime.now(UTC)).total_seconds() <= 3600 + assert result["AccessKeyId"] == "PAIR" + + +class TestScopedBroker: + def test_requires_auth_and_scrubs_only_child_environment(self, monkeypatch): + monkeypatch.setenv("AWS_ACCESS_KEY_ID", "PARENT") + monkeypatch.setenv("AWS_PROFILE", "operator") + monkeypatch.setenv("AWS_WEB_IDENTITY_TOKEN_FILE", "/synthetic/token") + monkeypatch.setenv("AWS_CONTAINER_AUTHORIZATION_TOKEN_FILE", "/synthetic/ambient-token") + monkeypatch.setenv("AWS_FUTURE_CREDENTIAL_SOURCE", "synthetic") + monkeypatch.setenv("AWS_REGION", "us-west-2") + provider = MagicMock(side_effect=_envelope) + broker = ScopedCredentialBroker(MicrovmLifecycle("task", "vm"), provider=provider) + try: + import os + + assert os.environ["AWS_ACCESS_KEY_ID"] == "PARENT" + assert os.environ["AWS_PROFILE"] == "operator" + for key in ( + "AWS_ACCESS_KEY_ID", + "AWS_PROFILE", + "AWS_WEB_IDENTITY_TOKEN_FILE", + "AWS_CONTAINER_AUTHORIZATION_TOKEN_FILE", + "AWS_FUTURE_CREDENTIAL_SOURCE", + ): + assert broker.environment[key] == "" + assert "HOME" not in broker.environment + assert "AWS_REGION" not in broker.environment + for token, path, status in [(None, "", 403), ("wrong", "", 403), ("wrong", "/x", 404)]: + with pytest.raises(HTTPError) as error: + _fetch(broker, token=token, path=path) + assert error.value.code == status + provider.assert_not_called() + result = _fetch(broker, token=broker.environment["AWS_CONTAINER_AUTHORIZATION_TOKEN"]) + assert result["AccessKeyId"] == "SYNTHETIC" + finally: + broker.close() + broker.close() + assert not broker._thread.is_alive() + + def test_provider_failure_has_no_secret_details_or_fallback(self): + provider = MagicMock(side_effect=RuntimeError("do-not-leak-this")) + broker = ScopedCredentialBroker(MicrovmLifecycle("task", "vm"), provider=provider) + try: + with pytest.raises(HTTPError) as error: + _fetch(broker, token=broker.environment["AWS_CONTAINER_AUTHORIZATION_TOKEN"]) + assert error.value.code == 503 + assert b"do-not-leak-this" not in error.value.read() + provider.assert_called_once() + finally: + broker.close() + + @pytest.mark.anyio + async def test_suspend_drains_broker_and_resume_failure_keeps_it_closed(self): + class Deadline: + def remaining_s(self) -> float: + return 60 + + lifecycle = MicrovmLifecycle("task", "vm") + await lifecycle.tool_started("tool") + lifecycle.park_approval("gate", "tool", Deadline()) + entered, release = threading.Event(), threading.Event() + + def provider(): + entered.set() + assert release.wait(2) + return _envelope() + + broker = ScopedCredentialBroker(lifecycle, provider=provider) + try: + fetch = asyncio.create_task( + asyncio.to_thread( + _fetch, broker, token=broker.environment["AWS_CONTAINER_AUTHORIZATION_TOKEN"] + ) + ) + assert await asyncio.to_thread(entered.wait, 1) + checkpoint = MagicMock() + suspend = asyncio.create_task(lifecycle.suspend(checkpoint, budget_s=2)) + await asyncio.sleep(0.03) + checkpoint.assert_not_called() + release.set() + await fetch + await suspend + checkpoint.assert_called_once() + with pytest.raises(HTTPError) as error: + await asyncio.to_thread( + _fetch, broker, token=broker.environment["AWS_CONTAINER_AUTHORIZATION_TOKEN"] + ) + assert error.value.code == 503 + with pytest.raises(RuntimeError, match="renewal failed"): + await lifecycle.resume( + MagicMock(side_effect=RuntimeError("renewal failed")), budget_s=1 + ) + with pytest.raises(LifecycleUnavailable): + await lifecycle.wait_until_open() + with pytest.raises(HTTPError) as error: + await asyncio.to_thread( + _fetch, broker, token=broker.environment["AWS_CONTAINER_AUTHORIZATION_TOKEN"] + ) + assert error.value.code == 503 + finally: + release.set() + broker.close() diff --git a/agent/tests/test_runner.py b/agent/tests/test_runner.py index 9320f9633..7bc96bac8 100644 --- a/agent/tests/test_runner.py +++ b/agent/tests/test_runner.py @@ -1,7 +1,7 @@ """Unit tests for runner.py helpers. -The full ``run_agent`` path is integration-tested via test_pipeline.py -with a mocked ``pipeline.run_agent``. This module covers the narrower +Pipeline tests mock ``pipeline.run_agent``; they do not exercise its SDK loop. +This module covers client/broker ownership and the narrower ``_initialize_policy_engine_and_hooks`` helper extracted in Chunk 7 so the policy-engine bootstrap + ``pre_approvals_loaded`` emission can be verified without spinning up the Claude Agent SDK client. @@ -12,7 +12,7 @@ import asyncio import subprocess from typing import Any -from unittest.mock import MagicMock, patch +from unittest.mock import AsyncMock, MagicMock, patch import pytest @@ -48,6 +48,79 @@ def _config(**overrides: Any) -> TaskConfig: return TaskConfig(**base) +class TestClaudeSessionOwnership: + @pytest.mark.parametrize("microvm", [False, True]) + @pytest.mark.parametrize("failure", [None, "connect", "query", "receive", "cancel"]) + def test_broker_selection_and_cleanup_on_every_session_exit( + self, monkeypatch, microvm, failure + ): + import claude_agent_sdk + + import microvm_credentials + import microvm_lifecycle + + config = _config() + context = microvm_lifecycle.register_task(config.task_id, "vm") if microvm else None + client = MagicMock() + client.connect = AsyncMock() + client.query = AsyncMock() + client.disconnect = AsyncMock() + if failure in {"connect", "query"}: + getattr(client, failure).side_effect = RuntimeError("synthetic failure") + + async def messages(): + if failure == "receive": + raise RuntimeError("synthetic receive failure") + if failure == "cancel": + raise asyncio.CancelledError + yield claude_agent_sdk.ResultMessage( + subtype="success", + duration_ms=1, + duration_api_ms=1, + is_error=False, + num_turns=0, + session_id="synthetic", + total_cost_usd=0, + usage={}, + ) + + client.receive_response = messages + make_client = MagicMock(return_value=client) + monkeypatch.setattr(claude_agent_sdk, "ClaudeSDKClient", make_client) + broker = MagicMock() + broker.environment = {"ABCA_MICROVM_CREDENTIAL_BROKER": "1"} + make_broker = MagicMock(return_value=broker) + monkeypatch.setattr(microvm_credentials, "ScopedCredentialBroker", make_broker) + monkeypatch.setattr(runner, "_setup_agent_env", lambda _config: None) + monkeypatch.setattr(runner, "_log_claude_cli_version", lambda: None) + monkeypatch.setattr(runner, "_initialize_policy_engine_and_hooks", lambda **_kw: (None, {})) + monkeypatch.setattr(runner, "_register_gateway_server", lambda _servers: None) + monkeypatch.setattr(runner, "build_clarification_server", lambda: None) + monkeypatch.setattr(runner, "_ProgressWriter", MagicMock()) + monkeypatch.setattr(runner, "log_error_cw", MagicMock()) + try: + if failure == "cancel": + with pytest.raises(asyncio.CancelledError): + asyncio.run(runner.run_agent("probe", "probe", config, trajectory=MagicMock())) + else: + result = asyncio.run( + runner.run_agent("probe", "probe", config, trajectory=MagicMock()) + ) + assert result.status == ("error" if failure else "success") + options = make_client.call_args.kwargs["options"] + if microvm: + make_broker.assert_called_once_with(context) + assert options.env == broker.environment + broker.close.assert_called_once() + else: + make_broker.assert_not_called() + assert options.env == {} + client.disconnect.assert_awaited_once() + finally: + if context is not None: + microvm_lifecycle.unregister_task(context) + + class TestInitializePolicyEngineAndHooks: """Bootstrap the per-task PolicyEngine + hooks without the SDK loop. diff --git a/agent/tests/test_server.py b/agent/tests/test_server.py index 1c2dbbb10..511420ba8 100644 --- a/agent/tests/test_server.py +++ b/agent/tests/test_server.py @@ -608,12 +608,9 @@ def create_log_stream(self, *, logGroupName, logStreamName): def put_log_events(self, *, logGroupName, logStreamName, logEvents): captured_streams.append(logStreamName) - class _FakeBoto3: - @staticmethod - def client(*args, **kwargs): - return _FakeLogs() - - monkeypatch.setitem(__import__("sys").modules, "boto3", _FakeBoto3) + # This test owns stream routing. Credential/signing behavior belongs to the + # aws_session tests, so stub the attributed client factory at its boundary. + monkeypatch.setattr("aws_session.platform_client", lambda *_args, **_kwargs: _FakeLogs()) server._warn_cw_write_blocking( log_group="/some/log-group", diff --git a/docs/decisions/ADR-021-lambda-microvms-compute-backend.md b/docs/decisions/ADR-021-lambda-microvms-compute-backend.md index 424cb7d96..a08caff8a 100644 --- a/docs/decisions/ADR-021-lambda-microvms-compute-backend.md +++ b/docs/decisions/ADR-021-lambda-microvms-compute-backend.md @@ -1,6 +1,6 @@ # ADR-021: AWS Lambda MicroVMs as a third ComputeStrategy backend -> **Implementation status (2026-09-14):** P1 ([#689](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/689)) and P2 ([#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733)) are merged. The takeover branch now has a [clean deployment](../verification/645-p2-clean-deployment-20260913.md) with bootstrap 1.7.0 and a [verified managed-image update to 2.0](../verification/645-microvm-image-rebuild-20260914.md), plus [live coding, PR iteration and cancellation evidence](../verification/645-p2-live-task-20260914.md), without manual IAM workarounds. Remaining P2 failure/recovery and permission/network checks prevent full acceptance. Local P3 foundations include suspend/resume commands, explicit VM observations, guarded persistent intent, a policy helper, approval deadlines that survive a frozen clock, and the [guest pause controller](../verification/645-p3-guest-barrier.md). Agent lifecycle hooks and supervisor/approval-handler integration remain unimplemented; automatic suspension is not enabled. This ADR defines P1–P3, not a P4. See the [current review](../verification/645-p3-readiness-review.md) and [P3 implementation plan](../verification/645-p3-implementation-plan.md). +> **Implementation status (2026-09-14):** P1 ([#689](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/689)) and P2 ([#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733)) are merged. The takeover branch now has a [clean deployment](../verification/645-p2-clean-deployment-20260913.md) with bootstrap 1.7.0 and a [verified managed-image update to 2.0](../verification/645-microvm-image-rebuild-20260914.md), plus [live coding, PR iteration and cancellation evidence](../verification/645-p2-live-task-20260914.md), without manual IAM workarounds. Remaining P2 failure/recovery and permission/network checks prevent full acceptance. Local P3 foundations include suspend/resume commands, explicit VM observations, guarded persistent intent, a policy helper, approval deadlines that survive a frozen clock, the [guest pause controller](../verification/645-p3-guest-barrier.md), and [scoped Claude credentials with retained-client refresh](../verification/645-p3-credentials.md). Production HTTP lifecycle hooks and supervisor/approval-handler integration remain unimplemented; automatic suspension is not enabled. This ADR defines P1–P3, not a P4. See the [current review](../verification/645-p3-readiness-review.md) and [P3 implementation plan](../verification/645-p3-implementation-plan.md). **Status:** proposed **Date:** 2026-07-29 diff --git a/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md b/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md index b77e94d2d..a515c3030 100644 --- a/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md +++ b/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md @@ -4,7 +4,7 @@ title: Adr 021 lambda microvms compute backend # ADR-021: AWS Lambda MicroVMs as a third ComputeStrategy backend -> **Implementation status (2026-09-14):** P1 ([#689](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/689)) and P2 ([#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733)) are merged. The takeover branch now has a [clean deployment](/sample-autonomous-cloud-coding-agents/architecture/645-p2-clean-deployment-20260913) with bootstrap 1.7.0 and a [verified managed-image update to 2.0](/sample-autonomous-cloud-coding-agents/architecture/645-microvm-image-rebuild-20260914), plus [live coding, PR iteration and cancellation evidence](/sample-autonomous-cloud-coding-agents/architecture/645-p2-live-task-20260914), without manual IAM workarounds. Remaining P2 failure/recovery and permission/network checks prevent full acceptance. Local P3 foundations include suspend/resume commands, explicit VM observations, guarded persistent intent, a policy helper, approval deadlines that survive a frozen clock, and the [guest pause controller](/sample-autonomous-cloud-coding-agents/architecture/645-p3-guest-barrier). Agent lifecycle hooks and supervisor/approval-handler integration remain unimplemented; automatic suspension is not enabled. This ADR defines P1–P3, not a P4. See the [current review](/sample-autonomous-cloud-coding-agents/architecture/645-p3-readiness-review) and [P3 implementation plan](/sample-autonomous-cloud-coding-agents/architecture/645-p3-implementation-plan). +> **Implementation status (2026-09-14):** P1 ([#689](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/689)) and P2 ([#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733)) are merged. The takeover branch now has a [clean deployment](/sample-autonomous-cloud-coding-agents/architecture/645-p2-clean-deployment-20260913) with bootstrap 1.7.0 and a [verified managed-image update to 2.0](/sample-autonomous-cloud-coding-agents/architecture/645-microvm-image-rebuild-20260914), plus [live coding, PR iteration and cancellation evidence](/sample-autonomous-cloud-coding-agents/architecture/645-p2-live-task-20260914), without manual IAM workarounds. Remaining P2 failure/recovery and permission/network checks prevent full acceptance. Local P3 foundations include suspend/resume commands, explicit VM observations, guarded persistent intent, a policy helper, approval deadlines that survive a frozen clock, the [guest pause controller](/sample-autonomous-cloud-coding-agents/architecture/645-p3-guest-barrier), and [scoped Claude credentials with retained-client refresh](/sample-autonomous-cloud-coding-agents/architecture/645-p3-credentials). Production HTTP lifecycle hooks and supervisor/approval-handler integration remain unimplemented; automatic suspension is not enabled. This ADR defines P1–P3, not a P4. See the [current review](/sample-autonomous-cloud-coding-agents/architecture/645-p3-readiness-review) and [P3 implementation plan](/sample-autonomous-cloud-coding-agents/architecture/645-p3-implementation-plan). **Status:** proposed **Date:** 2026-07-29 diff --git a/docs/verification/645-p3-credentials.md b/docs/verification/645-p3-credentials.md new file mode 100644 index 000000000..b763902ec --- /dev/null +++ b/docs/verification/645-p3-credentials.md @@ -0,0 +1,125 @@ +# #645 P3: scoped credentials across sleep + +Date: 2026-09-14. Local implementation and actual pinned Claude process probes. +No AWS deployment or suspension was performed for this milestone. Deployed +image **2.0** remains unchanged; P3 is not complete. + +## What changed + +Think of AWS credentials as a visitor badge with an expiry time. A sleeping +worker may wake up holding an expired badge. Every part of the worker needs a +working replacement before continuing, with the same task permissions. + +`microvm_credentials.py` now serves the task's scoped credentials through an +authenticated HTTP endpoint bound only to `127.0.0.1`, on a runtime-selected port. +The Claude child receives that endpoint as its AWS container credential provider. +Its alternate AWS key, profile, SSO/process, web-identity, metadata and bearer-token +sources are cleared; controlled empty configuration files suppress local profiles. +The parent environment is preserved. + +The image's managed `awsCredentialExport` command stays in place. In this internal +MicroVM mode, `bedrock_creds_helper.py` emits `{"Credentials": {}}` without reading +attribution files or resolving AWS credentials. That leaves resolution to the +scoped container provider and keeps keys out of Claude's stale export cache. +AgentCore/ECS/local attribution retains its existing helper behavior. + +`aws_session.py` exports a coherent key/expiry pair under botocore's refresh lock. +The future resume callback can force recorded ambient providers to refresh first, +then force the original tenant credential object to renew with the original STS +tags. Existing clients keep their references. Changing the configured identity +after session construction is rejected. Mandatory renewal propagates failure +even if an older cached key remains valid. + +Normal deployments already launch a fresh AgentCore runtime session or ECS/MicroVM +worker per task. Manually reusing one Python process for a different identity after +credential construction now fails explicitly. MicroVM STS calls use short network +timeouts; the other backends retain their default timeout/retry behavior. + +The broker participates in the guest activity drain. Paused, failed or closed +controllers reject credential requests before consulting AWS. SDK success, startup +failure, query failure, receive failure and cancellation close the client and broker. +The earlier runner did not disconnect its SDK client. + +This is provider selection, not isolation from arbitrary code running as the same +OS user. The bearer token is created from OS entropy at runtime; it and the real +credential responses must never be logged or included in verification evidence. +The broker requires no ingress connector or additional AWS IAM permission. + +## Actual CLI evidence + +The opt-in verifier runs **claude-agent-sdk 0.2.110 / Claude 2.1.191** against a +loopback fake Bedrock server. It returns valid AWS event-stream frames containing +synthetic model text. All keys are synthetic. There are no paid model calls. + +Each case requires a successful initial model query before a 12-second credential +expiry, waits past that actual wall-clock expiry, and checks both the next request's +signing key and the SDK result. The verifier rejects unreviewed version pins. +`CLAUDE_CODE_MAX_RETRIES=0` bounds this experiment's error result; production model +retry settings are unchanged. + +| Case | First query after expiry | Result | +|---|---|---| +| Existing `awsCredentialExport` | Still signs with `SYNTHETIC_INITIAL` | Reproduces stale-key reuse | +| `credential_process` renewal | Signs with `SYNTHETIC_RENEWED` | Renewal itself works | +| Working ambient-only control | Signs with `SYNTHETIC_AMBIENT` | Calibrates fallback endpoint | +| Failed process renewal plus working ambient provider | Signs with `SYNTHETIC_AMBIENT` | Demonstrates unsafe fallback | +| Production scoped broker plus production helper | Signs with `SYNTHETIC_RENEWED` | Waits for replacement before signing | +| Failed production broker renewal | No second model request; SDK reports credential error | Fails closed | + +The fake server accepts synthetic signatures. The export case deliberately proves +which key Claude sends; its successful fake response does not mean AWS would accept +an expired key. Non-streaming startup model-availability checks are recorded +separately from the two main queries. + +Run from `agent/`: + +```bash +.venv/bin/python scripts/verify_microvm_credentials.py --mode export +.venv/bin/python scripts/verify_microvm_credentials.py --mode process +.venv/bin/python scripts/verify_microvm_credentials.py --mode ambient --ambient-fallback +.venv/bin/python scripts/verify_microvm_credentials.py --mode process-failure --ambient-fallback +.venv/bin/python scripts/verify_microvm_credentials.py --mode broker +.venv/bin/python scripts/verify_microvm_credentials.py --mode broker-failure +``` + +Each successful verification prints `"verified": true`; failed expectations exit +nonzero. All temporary files, CLI processes and loopback servers are cleaned up. +Reports for these six runs were retained locally as +`/tmp/abca-645-p2-clean-20260913/p3-credentials--20260914.json`. + +## Regression coverage and remaining work + +The agent quality gate passes **1,881 tests** (24 added), with **86.14%** total +branch coverage, plus Ruff lint/format and type checking. The configured Bandit +high-severity gate and Vulture dead-code gate pass. + +The full monorepo build also passes: **4,797 CDK tests** (38 existing skips across +two suites), **928 CLI tests**, **11 Forge tests**, CDK compile/lint/synth, +documentation build/link checks and contract drift checks. The final agent quality +run includes the two additional backend-timeout tests added during that build. +Logs are retained as `p3-credentials-build-20260914.log` and +`p3-credentials-agent-quality-20260914.log` in the same private evidence directory. + +Focused tests use real botocore refreshable credentials and retained S3/DynamoDB +clients. They verify ambient-before-tenant ordering, exact tag preservation, +signing with replaced keys, mandatory failure, identity mismatch, unknown/static +providers, expired replacements and UTC expiry formatting. Loopback tests verify +authentication, child-only environment changes, no secret error details, suspend +drain and continued denial after failed resume. Runner tests cover cleanup on +normal completion and failure/cancellation for MicroVM and other backends. + +Before automatic suspension can ship: + +1. Connect the refresh operation to the bounded production `/resume` callback, + followed by durable gate/deadline reconciliation. Finish acknowledged checkpoint + writes and `/suspend` admission. +2. Verify actual MicroVM runtime credential-provider type and renewal after sleep. + Static or unknown providers currently reject resume; rereading an unchanged + environment is not proof of renewal. Botocore's forced-refresh private API is + isolated in one adapter and requires review when the SDK changes. +3. Verify the built image's managed-settings path, Gateway signing, detached + subprocess behavior and long-expiry sleep in AWS. The local CLI probe uses an + explicit settings file containing the production helper command; it does not + install `/etc/claude-code/managed-settings.json` on the developer machine. +4. Finish image capability, durable supervisor recovery and approval-triggered wake, + then complete the [P3 plan](./645-p3-implementation-plan.md), including its P2 gates. diff --git a/docs/verification/645-p3-guest-barrier.md b/docs/verification/645-p3-guest-barrier.md index 0b4d6c091..f5d454343 100644 --- a/docs/verification/645-p3-guest-barrier.md +++ b/docs/verification/645-p3-guest-barrier.md @@ -65,13 +65,13 @@ when the helper expiration is missing or six minutes or less away. Therefore: - Current documentation for newer Claude default-chain caching is not evidence that the pinned helper path has those semantics. -Next, exercise the actual pinned CLI against a local fake Bedrock endpoint with -synthetic credentials and controlled expiry. Evaluate a supported strict -credential-provider path, such as `credential_process`, which must await renewal -before signing, or a supported explicit cache invalidation/continuation method. -Check this using the actual SDK/CLI, preserve task/user/repo attribution, and -keep provider failures from falling back to stale or unscoped credentials. -Do not patch internal minified CLI functions. +**Follow-up implemented locally:** the [credential verification](./645-p3-credentials.md) +now exercises that actual CLI. `credential_process` renewed successfully but +fell through to ambient credentials on failure. The chosen MicroVM path uses a +single authenticated, scoped container provider; renewal waits before signing, +and failure sends no model request. Python refresh updates retained credential +objects with the original identity. Production HTTP callbacks and live runtime +provider/snapshot verification are still open. No minified CLI internals were patched. Known background tool flags are tracked, but arbitrary shell/MCP subprocesses may also detach work. Their safe-point behavior still requires a conservative @@ -103,8 +103,8 @@ budgets, durable supervisor recovery or complete P3 acceptance. ## Remaining integration order -1. Prove the Claude credential path, then implement tenant/ambient refresh for - existing clients while preserving attribution. +1. **Local credential implementation complete:** verify the deployed runtime + provider and snapshot behavior; connect retained-client refresh to the resume callback. 2. Implement the production checkpoint and HTTP suspend/resume routes with shared service budgets, duplicate-request handling and deterministic teardown. 3. Bind lifecycle capability to the image/version that launched each worker; diff --git a/docs/verification/645-p3-implementation-plan.md b/docs/verification/645-p3-implementation-plan.md index f7df5cb0c..d319c31ab 100644 --- a/docs/verification/645-p3-implementation-plan.md +++ b/docs/verification/645-p3-implementation-plan.md @@ -398,9 +398,16 @@ image capability, production checkpoint/refresh callbacks, IAM changes or automatic sleep have been enabled. See the [guest barrier review](./645-p3-guest-barrier.md) for the implemented boundary, tests and remaining credential/subprocess work. +**Credential implementation added locally (2026-09-14):** the +[credential verification](./645-p3-credentials.md) records actual pinned-CLI +expiry/failure probes, the MicroVM-only scoped loopback provider, retained +ambient/tenant refresh and SDK/broker teardown. Static/unknown runtime providers +fail closed; their real renewal path must be verified before enabling suspension. +This refresh function is not yet wired to a production HTTP resume callback. + 1. **Implemented:** a per-task context holds task/VM identity, active approval and the original deadline. MicroVM background startup registers it; pipeline exit/crash unregisters it. The SDK hook removes the approval safe point before changing task state or returning a permission result. **Still required:** supply this context to the HTTP lifecycle handlers and finish duplicate-hook admission/acknowledgment behavior. 2. `/suspend` validates that the task is still parked on the intended gate. Wait for lifecycle/progress work already in progress to finish, establish an acknowledged durability barrier, and return success only within the hook budget. Existing best-effort event methods cannot establish that barrier. On timeout/write failure, report a hook failure so the coordinator keeps/reconciles the running VM. Test approval or cancellation arriving during this boundary. -3. `/resume` refreshes ambient/runtime credential providers as necessary, then ensures tenant-scoped assumed credentials are usable **with the same task/user/repo tags**. **Inventory completed:** cached DynamoDB/S3 clients retain the tenant credential object; Memory/Logs/trajectory clients use the ambient factory; the Claude subprocess has a separate cache. Pinned Claude 2.1.191's `awsCredentialExport` refresh returns the old cached value to its first caller while renewing in the background, so an expired token can survive into a post-wake request. A shorter helper expiration alone does not fix this. Prove a supported, blocking credential-provider/continuation path for the subprocess before enabling sleep. Do not call the test-only `reset_session_cache()` and lose identity. Fail closed on refresh failure. +3. `/resume` refreshes ambient/runtime credential providers as necessary, then ensures tenant-scoped assumed credentials are usable **with the same task/user/repo tags**. **Implemented/tested locally:** `refresh_microvm_credentials` forces mandatory ambient renewal before renewing the same tenant credential object. Existing DynamoDB/S3/platform clients retain their references; session identity cannot change after construction. The Claude child uses only the scoped container provider and its managed export helper returns no cached keys. Actual pinned CLI probes establish first-request renewal and failure without fallback; a `credential_process` fallback control demonstrates why provider isolation matters. **Remaining:** wire refresh behind the HTTP resume barrier; verify actual runtime provider, long sleep, Gateway signing and managed image behavior in AWS. Never use test-only `reset_session_cache()` to simulate renewal. 4. Keep the coding action blocked behind the resume barrier until refresh and gate reconciliation finish. Handle duplicate hook calls and concurrent lifecycle requests without deadlocks. Expired credentials or a slow AWS call must not hold the hook beyond its service budget. 5. **Implemented locally:** `/run` and successful controller resume reseed the application PRNG from fresh OS entropy. The future `/resume` HTTP route must use this controller. Do not seed it with task IDs, timestamps or an image-fixed value. Continue using cryptographic randomness for secrets. Test resumed/sibling snapshot uniqueness where meaningful; do not claim that `random` becomes cryptographically safe. 6. **Polling implemented locally:** `_ApprovalDeadline` captures the original recorded UTC expiry and a monotonic cap before database writes. Remaining time is `min(monotonic_deadline - monotonic_now, created_at + timeout_s - wall_now)`, clamped at zero, including each sleep bound. **Registered locally:** the lifecycle context retains this exact object and releases the existing approval loop after successful controller resume. **Still required:** the production resume callback must reconcile the gate and check the original deadline. Never create a fresh timeout on resume. diff --git a/docs/verification/645-p3-readiness-review.md b/docs/verification/645-p3-readiness-review.md index 2921e1f1c..0315b123f 100644 --- a/docs/verification/645-p3-readiness-review.md +++ b/docs/verification/645-p3-readiness-review.md @@ -14,6 +14,14 @@ The first local P3 foundation adds the supervisor's pause/wake command methods a A second foundation adds saved sleep/wake instructions with revision stamps, so an old supervisor cannot overwrite a newer wake request. A policy helper now handles the observed state and current approval deadline. Real DynamoDB Local transaction tests cover competing writers and restart/readback cases; this is a local database test, not an AWS deployment. See the [lifecycle runbook](./645-lifecycle-intent.md). Guest hooks, durable supervisor integration and live verification remain required. +**Further P3 work (2026-09-14):** the [guest pause controller](./645-p3-guest-barrier.md) +now keeps coding behind a controlled door while approval work is paused. The +[credential implementation](./645-p3-credentials.md) renews the existing AWS key +objects and gives Claude one source of task-specific keys. Actual pinned Claude +tests with fake AWS responses prove renewal before the next request and safe +failure without borrowing the parent's keys. Production HTTP pause/wake handlers, +supervisor wiring and real AWS sleep/wake verification remain open. + The reservation review found a separate trust boundary: the old agent role could write/replace/delete its task row, including coordinator metadata. A subsequent local fix restricts main-task writes to reporting/approval attributes, removes replacement/deletion permissions and removes unused worker counter grants. It also removes unused Python submission/session-info helpers and corrects overstated tenant-isolation comments. [Metadata verification](./645-coordinator-metadata.md) records the tests and pending AWS gate. Status reports still come from the agent, and the compute role chooses session tags; this is not complete hostile-worker isolation. ## Start here: the pieces in plain language @@ -34,7 +42,7 @@ An **IAM role** is a permission badge. A **trust policy** says who may wear that |---|---|---| | P1 | Build the computer, start it, deliver a task, check it and stop it | Merged in [#689](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/689). Strategy, infrastructure, bootstrap permissions, packaging, types, `/ready` and `/run` exist. | | P2 | Make a real coding task work with configuration, permissions, logs and progress | Merged in [#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733). `/validate`, `/terminate`, warm-up, runtime grants and heartbeat support exist. Two recorded tasks completed and opened PRs on August 7, using a manual IAM workaround. Permanent fixes still need a clean rerun. | -| P3 | Sleep during a human approval wait, wake correctly, and keep deadlines/credentials safe | Not implemented. AWS's suspend API was exercised manually; ABCA's integrated pause/resume lifecycle is missing. | +| P3 | Sleep during a human approval wait, wake correctly, and keep deadlines/credentials safe | Local foundations now include intent/policy, the guest pause controller and scoped credential renewal. The integrated HTTP/supervisor lifecycle and live acceptance remain open. | | P4 | — | ADR-021 defines no P4. Verification runbooks have their own numbered phases; those are not extra ADR milestones. | The old unchecked checklist and the word “proposed” do not erase the merged work. Conversely, merged code is not proof that the final deployment path works unattended. From c6aee5c0c3cc47b6588ce2d69b182ff8925bab04 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Mon, 14 Sep 2026 16:55:55 -0400 Subject: [PATCH 035/149] feat(agent): checkpoint MicroVM approval sleep and wake --- agent/src/hooks.py | 18 +- agent/src/microvm_checkpoint.py | 242 ++++++++++ agent/src/microvm_http.py | 101 +++++ agent/src/microvm_lifecycle.py | 50 ++- agent/src/progress_writer.py | 42 +- agent/src/server.py | 38 +- agent/tests/test_microvm_checkpoint.py | 416 ++++++++++++++++++ agent/tests/test_microvm_http.py | 318 +++++++++++++ agent/tests/test_server.py | 15 +- cdk/src/constructs/lambda-microvm-compute.ts | 11 +- .../constructs/lambda-microvm-compute.test.ts | 16 +- cdk/test/scripts/check-constants-sync.test.ts | 24 + contracts/constants.json | 4 +- contracts/constants.md | 16 +- ...ADR-021-lambda-microvms-compute-backend.md | 2 +- ...Adr-021-lambda-microvms-compute-backend.md | 2 +- docs/verification/645-p3-credentials.md | 8 +- docs/verification/645-p3-guest-barrier.md | 26 +- .../645-p3-implementation-plan.md | 40 +- docs/verification/645-p3-lifecycle-hooks.md | 157 +++++++ docs/verification/645-p3-readiness-review.md | 8 +- scripts/check-constants-sync.ts | 23 +- 22 files changed, 1499 insertions(+), 78 deletions(-) create mode 100644 agent/src/microvm_checkpoint.py create mode 100644 agent/src/microvm_http.py create mode 100644 agent/tests/test_microvm_checkpoint.py create mode 100644 agent/tests/test_microvm_http.py create mode 100644 docs/verification/645-p3-lifecycle-hooks.md diff --git a/agent/src/hooks.py b/agent/src/hooks.py index ddcc357bd..e4b035fcc 100644 --- a/agent/src/hooks.py +++ b/agent/src/hooks.py @@ -31,7 +31,7 @@ import nudge_reader import task_state -from microvm_lifecycle import get_context +from microvm_lifecycle import ApprovalRecord, get_context from nudge_reader import _xml_escape from output_scanner import scan_tool_output from policy import APPROVAL_RATE_LIMIT, FLOOR_TIMEOUT_S, Outcome @@ -736,7 +736,21 @@ async def _handle_require_approval( # The SDK wrapper already tracks this tool; direct/legacy callers without a # lifecycle context keep their existing approval behavior. lifecycle = get_context(task_id) - park = lifecycle.park_approval(request_id, tool_use_id, deadline) if lifecycle else None + park = ( + lifecycle.park_approval( + request_id, + tool_use_id, + deadline, + record=ApprovalRecord( + user_id=row["user_id"], + repo=row["repo"], + created_at=row["created_at"], + timeout_s=effective_timeout, + ), + ) + if lifecycle + else None + ) try: outcome = await _poll_for_decision( task_id=task_id, diff --git a/agent/src/microvm_checkpoint.py b/agent/src/microvm_checkpoint.py new file mode 100644 index 000000000..7a42982e9 --- /dev/null +++ b/agent/src/microvm_checkpoint.py @@ -0,0 +1,242 @@ +# Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. +# SPDX-License-Identifier: MIT-0 + +"""Durable guest pause checks; the coordinator retains lifecycle intent ownership.""" + +from __future__ import annotations + +import os +from datetime import UTC, datetime +from decimal import Decimal +from typing import TYPE_CHECKING, Any, Literal + +from microvm_lifecycle import ApprovalPark, LifecycleUnavailable +from progress_writer import _ProgressWriter + +if TYPE_CHECKING: + from microvm_lifecycle import ApprovalRecord + + +def _integer(value: Any) -> bool: + return ( + isinstance(value, (int, Decimal)) + and not isinstance(value, bool) + and value == int(value) + and 0 <= value <= 2**53 - 1 + ) + + +def _record(park: ApprovalPark) -> tuple[ApprovalRecord, int]: + record = park.record + if ( + record is None + or not record.user_id + or not isinstance(record.repo, str) + or not _integer(record.timeout_s) + or record.timeout_s <= 0 + ): + raise LifecycleUnavailable("Original approval identity is unavailable") + try: + created = datetime.strptime(record.created_at, "%Y-%m-%dT%H:%M:%SZ").replace(tzinfo=UTC) + except (TypeError, ValueError) as exc: + raise LifecycleUnavailable("Original approval timestamp is invalid") from exc + if created.strftime("%Y-%m-%dT%H:%M:%SZ") != record.created_at: + raise LifecycleUnavailable("Original approval timestamp is not canonical") + return record, int(created.timestamp() * 1000) + record.timeout_s * 1000 + + +def _read( + park: ApprovalPark, action: Literal["suspend", "resume"] +) -> tuple[Any, str, str, dict, int]: + """Read current task/gate identity strongly; absence and mismatch fail closed.""" + from boto3.dynamodb.types import TypeDeserializer + from botocore.config import Config + + from aws_session import tenant_client + + record, deadline_ms = _record(park) + task_table = os.environ.get("TASK_TABLE_NAME", "").strip() + approvals_table = os.environ.get("TASK_APPROVALS_TABLE_NAME", "").strip() + if not task_table or not approvals_table: + raise RuntimeError("Lifecycle task/approval tables are unavailable") + client = tenant_client( + "dynamodb", + region_name=os.environ.get("AWS_REGION") or os.environ.get("AWS_DEFAULT_REGION"), + config=Config(connect_timeout=2, read_timeout=2, retries={"total_max_attempts": 1}), + ) + deserialize = TypeDeserializer().deserialize + + def read_item(table: str, key: dict) -> dict: + item = client.get_item(TableName=table, Key=key, ConsistentRead=True).get("Item") + if not isinstance(item, dict): + raise LifecycleUnavailable("Lifecycle task or approval is missing") + return {key: deserialize(value) for key, value in item.items()} + + task = read_item(task_table, {"task_id": {"S": park.task_id}}) + metadata = task.get("compute_metadata") + intent = task.get("microvm_lifecycle") + if ( + task.get("task_id") != park.task_id + or task.get("user_id") != record.user_id + or task.get("repo", "") != record.repo + or task.get("compute_type") != "lambda-microvm" + or task.get("session_id") != park.microvm_id + or not isinstance(metadata, dict) + or metadata.get("microvmId") != park.microvm_id + or task.get("status") != "AWAITING_APPROVAL" + or task.get("awaiting_approval_request_id") != park.request_id + ): + raise LifecycleUnavailable("Lifecycle task identity or approval state changed") + if ( + not isinstance(intent, dict) + or not _integer(intent.get("version")) + or intent["version"] != 1 + or not isinstance(intent.get("generation"), str) + or not intent["generation"].strip() + or intent.get("microvm_id") != park.microvm_id + or intent.get("request_id") != park.request_id + or intent.get("action") != action + or not _integer(intent.get("requested_at_ms")) + or not _integer(intent.get("deadline_ms")) + or intent["deadline_ms"] != deadline_ms + ): + raise LifecycleUnavailable("Coordinator lifecycle intent does not match this approval") + + approval = read_item( + approvals_table, + {"task_id": {"S": park.task_id}, "request_id": {"S": park.request_id}}, + ) + statuses = ( + {"PENDING"} + if action == "suspend" + else {"PENDING", "APPROVED", "DENIED", "TIMED_OUT", "STRANDED"} + ) + if ( + approval.get("task_id") != park.task_id + or approval.get("request_id") != park.request_id + or approval.get("user_id") != record.user_id + or approval.get("repo") != record.repo + or approval.get("created_at") != record.created_at + or not _integer(approval.get("timeout_s")) + or approval["timeout_s"] != record.timeout_s + or approval.get("status") not in statuses + ): + raise LifecycleUnavailable("Original approval changed or cannot be reconciled") + return client, task_table, approvals_table, intent, deadline_ms + + +def _transaction_checks( + park: ApprovalPark, task_table: str, approvals_table: str, intent: dict +) -> list[dict]: + """Recheck both rows atomically after validating their complete read shapes.""" + from boto3.dynamodb.types import TypeSerializer + + record, _ = _record(park) + serialize = TypeSerializer().serialize + values = { + ":task": park.task_id, + ":user": record.user_id, + ":repo": record.repo, + ":vm": park.microvm_id, + ":request": park.request_id, + ":awaiting": "AWAITING_APPROVAL", + ":backend": "lambda-microvm", + ":intent": intent, + } + task_check = { + "TableName": task_table, + "Key": {"task_id": {"S": park.task_id}}, + "ConditionExpression": ( + "task_id = :task AND user_id = :user AND " + + ( + "(attribute_not_exists(repo) OR repo = :repo)" + if not record.repo + else "repo = :repo" + ) + + " AND compute_type = :backend AND session_id = :vm" + " AND compute_metadata.microvmId = :vm AND #status = :awaiting" + " AND awaiting_approval_request_id = :request AND microvm_lifecycle = :intent" + ), + "ExpressionAttributeNames": {"#status": "status"}, + "ExpressionAttributeValues": {key: serialize(value) for key, value in values.items()}, + } + approval_check = { + "TableName": approvals_table, + "Key": {"task_id": {"S": park.task_id}, "request_id": {"S": park.request_id}}, + "ConditionExpression": ( + "task_id = :task AND request_id = :request AND user_id = :user AND repo = :repo" + " AND created_at = :created AND timeout_s = :timeout AND " + + ( + "#status = :pending" + if intent["action"] == "suspend" + else "#status IN (:pending, :approved, :denied, :timed_out, :stranded)" + ) + ), + "ExpressionAttributeNames": {"#status": "status"}, + "ExpressionAttributeValues": { + key: serialize(value) + for key, value in { + ":task": park.task_id, + ":request": park.request_id, + ":user": record.user_id, + ":repo": record.repo, + ":created": record.created_at, + ":timeout": record.timeout_s, + ":pending": "PENDING", + **( + { + ":approved": "APPROVED", + ":denied": "DENIED", + ":timed_out": "TIMED_OUT", + ":stranded": "STRANDED", + } + if intent["action"] == "resume" + else {} + ), + }.items() + }, + } + return [task_check, approval_check] + + +def checkpoint_before_suspend(park: ApprovalPark) -> None: + """Commit a marker only if the same task, sleep intent and pending gate hold.""" + record, _ = _record(park) + client, task_table, approvals_table, intent, deadline_ms = _read(park, "suspend") + if park.deadline.remaining_s() <= 0: + raise LifecycleUnavailable("Approval deadline elapsed before checkpoint") + # Never reuse an old transaction client token across HTTP requests: cached + # success must not bypass conditions after a concurrent approval/cancellation. + _ProgressWriter( + park.task_id, user_id=record.user_id, repo=record.repo + ).write_microvm_checkpoint( + client=client, + condition_checks=_transaction_checks(park, task_table, approvals_table, intent), + metadata={ + "request_id": park.request_id, + "microvm_id": park.microvm_id, + "generation": intent["generation"], + "approval_deadline_ms": deadline_ms, + }, + ) + + +def refresh_and_reconcile_after_resume(park: ApprovalPark) -> None: + """Refresh before reading AWS; the original approval loop owns any decision.""" + from aws_session import refresh_microvm_credentials + + _record(park) + refresh_microvm_credentials(park.task_id) + client, task_table, approvals_table, intent, _ = _read(park, "resume") + # This transaction writes no task/approval state. It acknowledges that both + # identities still hold, including cancellation or intent changes after reads. + # A concurrent valid decision is allowed; the original loop observes it. + client.transact_write_items( + TransactItems=[ + {"ConditionCheck": check} + for check in _transaction_checks(park, task_table, approvals_table, intent) + ] + ) + # The existing approval loop checks its original stopwatch/UTC cap as soon + # as the barrier opens. Expiry enters its timeout/late-decision path; a timely + # approval already recorded must still win that conditional race. diff --git a/agent/src/microvm_http.py b/agent/src/microvm_http.py new file mode 100644 index 000000000..d51210a65 --- /dev/null +++ b/agent/src/microvm_http.py @@ -0,0 +1,101 @@ +# Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. +# SPDX-License-Identifier: MIT-0 + +"""Bounded service-owned MicroVM pause/wake routes; no public ingress is added.""" + +from __future__ import annotations + +import asyncio +import json +import time +from typing import Literal + +from fastapi import Request # noqa: TC002 — FastAPI resolves this annotation at registration. +from fastapi.responses import JSONResponse + +from microvm_checkpoint import checkpoint_before_suspend, refresh_and_reconcile_after_resume +from microvm_lifecycle import LifecycleUnavailable, get_registered_context +from shared_constants import SHARED_CONSTANTS + +_BUDGETS = SHARED_CONSTANTS["microvm_hook_budgets"] +LIFECYCLE_HANDLER_BUDGET_S: float = _BUDGETS["lifecycle_handler_budget_seconds"] +LIFECYCLE_HOOK_TIMEOUT_S: float = _BUDGETS["lifecycle_hook_timeout_seconds"] +if not 0 < LIFECYCLE_HANDLER_BUDGET_S < LIFECYCLE_HOOK_TIMEOUT_S: + raise ValueError("Lifecycle handler must leave time for the service hook response") +_MAX_BODY_BYTES = 4096 + + +class _InvalidBody(ValueError): + def __init__(self, status: int): + self.status = status + + +async def _microvm_id(request: Request) -> str: + body = bytearray() + async for chunk in request.stream(): + body.extend(chunk) + if len(body) > _MAX_BODY_BYTES: + raise _InvalidBody(413) + try: + payload = json.loads(body) if body else {} + except (ValueError, UnicodeError) as exc: + raise _InvalidBody(400) from exc + if not isinstance(payload, dict): + raise _InvalidBody(400) + microvm_id = payload.get("microvmId", "") + if not isinstance(microvm_id, str) or microvm_id != microvm_id.strip(): + raise _InvalidBody(400) + # The service sends an empty id on /terminate; tolerate an absent/empty id + # here too, using only the local /run registration. A supplied id must match. + return microvm_id + + +async def _transition(request: Request, action: Literal["suspend", "resume"]) -> JSONResponse: + try: + end = time.monotonic() + LIFECYCLE_HANDLER_BUDGET_S + async with asyncio.timeout(LIFECYCLE_HANDLER_BUDGET_S): + microvm_id = await _microvm_id(request) + lifecycle = get_registered_context() + if lifecycle is None or (microvm_id and microvm_id != lifecycle.microvm_id): + raise LifecycleUnavailable("No matching MicroVM task is registered") + remaining = end - time.monotonic() + if remaining <= 0: + raise TimeoutError("Lifecycle body consumed its budget") + if action == "suspend": + park = await lifecycle.suspend(checkpoint_before_suspend, budget_s=remaining) + else: + park = await lifecycle.resume( + refresh_and_reconcile_after_resume, budget_s=remaining + ) + return JSONResponse( + content={ + "status": "acknowledged", + "action": action, + "task_id": park.task_id, + "microvm_id": park.microvm_id, + "request_id": park.request_id, + } + ) + except _InvalidBody as exc: + return JSONResponse( + status_code=exc.status, content={"code": "MICROVM_LIFECYCLE_BODY_INVALID"} + ) + except LifecycleUnavailable: + return JSONResponse(status_code=409, content={"code": "MICROVM_LIFECYCLE_UNAVAILABLE"}) + except TimeoutError: + return JSONResponse(status_code=503, content={"code": "MICROVM_LIFECYCLE_TIMEOUT"}) + except Exception as exc: + # Neither service-hook responses nor logs may echo AWS exception details. + # A failed/uncertain wake stays behind the controller's closed barrier. + return JSONResponse( + status_code=503, + content={"code": "MICROVM_LIFECYCLE_FAILED", "error_type": type(exc).__name__}, + ) + + +async def microvm_suspend(request: Request) -> JSONResponse: + return await _transition(request, "suspend") + + +async def microvm_resume(request: Request) -> JSONResponse: + return await _transition(request, "resume") diff --git a/agent/src/microvm_lifecycle.py b/agent/src/microvm_lifecycle.py index f7cf3a3fe..a0dd40c58 100644 --- a/agent/src/microvm_lifecycle.py +++ b/agent/src/microvm_lifecycle.py @@ -5,8 +5,8 @@ This controller does not call AWS or decide approvals. The server owns one controller per running MicroVM task; tool hooks register work and the *original* -approval deadline. Future HTTP lifecycle handlers must supply the durable -checkpoint and credential-refresh operations before declaring image capability. +approval deadline. HTTP lifecycle handlers supply the durable checkpoint and +credential-refresh operations; this controller owns permission to release work. Returning from a suspend callback is not proof that AWS actually froze the VM. Once acknowledged, the barrier opens only after a successful resume callback. @@ -43,6 +43,16 @@ def reseed_random() -> None: random.seed(os.urandom(32)) +@dataclass(frozen=True) +class ApprovalRecord: + """Original durable gate fields, captured when its request is written.""" + + user_id: str + repo: str + created_at: str + timeout_s: int + + @dataclass(frozen=True) class ApprovalPark: task_id: str @@ -50,6 +60,7 @@ class ApprovalPark: request_id: str tool_use_id: str deadline: ApprovalDeadline + record: ApprovalRecord | None = None class MicrovmLifecycle: @@ -74,6 +85,7 @@ def __init__(self, task_id: str, microvm_id: str) -> None: self._progress_failed = False self._suspend_ineligible = False self._slept_request_id: str | None = None + self._last_resume_park: ApprovalPark | None = None def _check_open(self) -> bool: if self._phase in {"closed", "failed"}: @@ -113,7 +125,12 @@ def disable_suspend(self) -> None: self._suspend_ineligible = True def park_approval( - self, request_id: str, tool_use_id: str | None, deadline: ApprovalDeadline + self, + request_id: str, + tool_use_id: str | None, + deadline: ApprovalDeadline, + *, + record: ApprovalRecord | None = None, ) -> ApprovalPark | None: with self._lock: if ( @@ -125,8 +142,13 @@ def park_approval( ): self._suspend_ineligible = True return None - park = ApprovalPark(self.task_id, self.microvm_id, request_id, tool_use_id, deadline) + park = ApprovalPark( + self.task_id, self.microvm_id, request_id, tool_use_id, deadline, record + ) self._park = park + # A new gate cannot acknowledge a wake using the previous gate's + # cached result. Its own suspend must establish a fresh safe point. + self._last_resume_park = None self._phase = "parked" return park @@ -201,6 +223,15 @@ async def suspend( end = self._budget(budget_s) with self._lock: park = self._park + if ( + self._phase == "suspend-ready" + and park is not None + and not self._progress_failed + and park.deadline.remaining_s() > 0 + ): + # An already-acknowledged HTTP retry neither checkpoints again + # nor opens the barrier. Expired/unsafe retries remain closed. + return park if ( park is None or self._phase != "parked" @@ -256,6 +287,10 @@ async def resume( end = self._budget(budget_s) with self._lock: park = self._park + if self._phase in {"active", "parked"} and self._last_resume_park is not None: + # A duplicate wake acknowledgment cannot renew the approval + # timeout or re-run credential refresh on an executing task. + return self._last_resume_park if self._phase != "suspend-ready" or park is None: raise LifecycleUnavailable("No acknowledged suspend to resume") self._phase = "resuming" @@ -270,6 +305,7 @@ async def resume( # One sleep per approval gate. A duplicate suspend must not # race the newly released decision loop. self._slept_request_id = park.request_id + self._last_resume_park = park return park except BaseException: with self._lock: @@ -318,6 +354,12 @@ def get_context(task_id: str | None) -> MicrovmLifecycle | None: return _contexts.get(task_id or "") +def get_registered_context() -> MicrovmLifecycle | None: + """The service hook belongs to the sole task registered by this VM's /run.""" + with _registry_lock: + return next(iter(_contexts.values()), None) + + def unregister_task(context: MicrovmLifecycle) -> None: with _registry_lock: context.close() diff --git a/agent/src/progress_writer.py b/agent/src/progress_writer.py index f885174ab..ff2fdc1e8 100644 --- a/agent/src/progress_writer.py +++ b/agent/src/progress_writer.py @@ -360,7 +360,7 @@ def _reset_circuit_breakers() -> None: class _ProgressWriter: """Write AG-UI-style progress events to the existing DynamoDB TaskEventsTable. - Fail-open: a DDB write failure is logged but never raises. After + Ordinary event methods fail open: a DDB write failure is logged but never raises. After ``_MAX_FAILURES`` consecutive *transient* failures the task's stream is permanently disabled (circuit breaker). Permanent errors (``ValidationException`` et al.) drop the individual event without @@ -454,6 +454,46 @@ def _ensure_table(self): # -- core write ------------------------------------------------------------ + def write_microvm_checkpoint( + self, *, metadata: dict, condition_checks: list[dict], client + ) -> None: + """Atomically acknowledge a pause marker and its task/gate preconditions. + + Called only by the lifecycle checkpoint callback after activity drains. + This deliberately bypasses the best-effort event path: absent tables, + disabled progress, failed conditions and uncertain writes must raise. + A saved marker records a safe point, not proof that AWS froze the VM. + """ + if not self._table_name or self._disabled: + raise RuntimeError("Checkpoint progress table is unavailable") + from boto3.dynamodb.types import TypeSerializer + + now = datetime.now(UTC) + item = { + "task_id": self._task_id, + "event_id": _generate_ulid(), + "event_type": "agent_milestone", + "metadata": {"milestone": "microvm_suspend_checkpoint", **metadata}, + "timestamp": now.isoformat(), + "ttl": int(now.timestamp()) + _TTL_SECONDS, + "user_id": self._user_id, + } + if self._repo: + item["repo"] = self._repo + serializer = TypeSerializer() + client.transact_write_items( + TransactItems=[ + *({"ConditionCheck": check} for check in condition_checks), + { + "Put": { + "TableName": self._table_name, + "Item": {key: serializer.serialize(value) for key, value in item.items()}, + "ConditionExpression": "attribute_not_exists(task_id)", + } + }, + ] + ) + def _put_event(self, event_type: str, metadata: dict) -> None: from microvm_lifecycle import LifecycleUnavailable, get_context diff --git a/agent/src/server.py b/agent/src/server.py index 98b5bb359..95cf5dbf1 100644 --- a/agent/src/server.py +++ b/agent/src/server.py @@ -30,9 +30,11 @@ import task_state from config import resolve_github_token +from microvm_http import microvm_resume, microvm_suspend from microvm_lifecycle import ( LifecycleUnavailable, get_context, + get_registered_context, register_task, reseed_random, unregister_task, @@ -848,7 +850,8 @@ async def invoke_agent(request: Request, body: InvocationRequest): # (8080 — the same uvicorn process that serves /invocations and /ping), so the # hooks live here rather than in a sidecar. # -# Four hooks are served; ``/suspend`` + ``/resume`` are still P3: +# Six hooks are served; image declaration/capability and supervisor wiring for +# ``/suspend`` + ``/resume`` remain a separate P3 rollout step: # * ``/ready`` (build, P1) is MANDATORY. ``CreateMicrovmImage`` refuses an image # that enables ANY lifecycle hook without it ("The ready (/ready) MicroVM # image hook must be enabled when any MicroVM lifecycle hook … is enabled"), @@ -861,13 +864,13 @@ async def invoke_agent(request: Request, body: InvocationRequest): # BUILD role and makes ZERO AWS calls — see ``microvm_validate``. # * ``/terminate`` (runtime, P2) is a best-effort final flush. It never writes # terminal task status — the orchestrator owns terminal state. -# ``/suspend`` and ``/resume`` are P3 (they need the ComputeStrategy interface -# widening). Declaring a hook the agent does not answer fails the corresponding -# build or lifecycle transition, which is why the construct declares exactly the -# hooks served here. +# ``/suspend`` and ``/resume`` use a drained, acknowledged checkpoint and mandatory +# credential/gate reconciliation. The image must declare them with the shared +# service budget before it can advertise lifecycle capability. MICROVM_HOOK_PREFIX = "/aws/lambda-microvms/runtime/v1" -#: ``s3://`` scheme prefix for the out-of-band payload pointer. +app.add_api_route(f"{MICROVM_HOOK_PREFIX}/suspend", microvm_suspend, methods=["POST"]) +app.add_api_route(f"{MICROVM_HOOK_PREFIX}/resume", microvm_resume, methods=["POST"]) # --- platform_config allowlist (ADR-021 P2) -------------------------------- # WHY the agent's platform env arrives in the ``/run`` payload at all, instead of @@ -1720,7 +1723,8 @@ def microvm_validate(): would fail every build. Names only — never values. """ expected_routes = { - f"{MICROVM_HOOK_PREFIX}/{hook}" for hook in ("ready", "validate", "run", "terminate") + f"{MICROVM_HOOK_PREFIX}/{hook}" + for hook in ("ready", "validate", "run", "terminate", "suspend", "resume") } registered = {getattr(route, "path", None) for route in app.routes} missing_routes = sorted(expected_routes - registered) @@ -1774,6 +1778,11 @@ def microvm_validate(): return body +# The terminate body is only optional correlation data. Leave ample headroom +# inside the image's 15-second hook timeout if the service stream stalls. +_TERMINATE_BODY_BUDGET_SECONDS = 1.0 + + @app.post(f"{MICROVM_HOOK_PREFIX}/terminate") async def microvm_terminate(request: Request): """MicroVM ``/terminate`` runtime hook — best-effort flush, always 200. @@ -1803,12 +1812,21 @@ async def microvm_terminate(request: Request): There is no progress queue to drain. ``ProgressWriter`` writes synchronously but catches and drops failures; a return from its event method is not proof - of durability. This hook only logs and acknowledges teardown. P3's - ``/suspend`` needs a separate acknowledged durability barrier. + of durability. This hook closes the coding barrier, logs and acknowledges + teardown. ``/suspend`` uses a separate acknowledged checkpoint transaction. """ + # Close the local barrier before reading the body or emitting diagnostics. + # A slow checkpoint/refresh thread must not release coding during teardown. + try: + lifecycle = get_registered_context() + if lifecycle is not None: + lifecycle.close() + except Exception as exc: + _emit_stdout_line(f"[server/warn] /terminate barrier close failed: {type(exc).__name__}") raw = b"" try: - raw = await request.body() + async with asyncio.timeout(_TERMINATE_BODY_BUDGET_SECONDS): + raw = await request.body() except Exception as exc: # A truncated/aborted body must not become a 5xx: the VM is going away and # the id is only a correlation string. Logged, not swallowed. diff --git a/agent/tests/test_microvm_checkpoint.py b/agent/tests/test_microvm_checkpoint.py new file mode 100644 index 000000000..694eeac3e --- /dev/null +++ b/agent/tests/test_microvm_checkpoint.py @@ -0,0 +1,416 @@ +# Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. +# SPDX-License-Identifier: MIT-0 + +"""Guest durable-state checks; optional DynamoDB Local cases execute conditions.""" + +from __future__ import annotations + +import asyncio +import copy +import os +import uuid +from dataclasses import dataclass +from datetime import UTC, datetime +from unittest.mock import MagicMock +from urllib.parse import urlsplit + +import pytest +from boto3.dynamodb.types import TypeDeserializer, TypeSerializer +from botocore.exceptions import ClientError + +import microvm_checkpoint as checkpoint +from microvm_lifecycle import ApprovalPark, ApprovalRecord, LifecycleUnavailable +from progress_writer import _reset_circuit_breakers + + +@dataclass +class Deadline: + remaining: float = 60 + + def remaining_s(self) -> float: + return self.remaining + + +def make_park() -> ApprovalPark: + return ApprovalPark( + task_id="task", + microvm_id="vm", + request_id="request", + tool_use_id="tool", + deadline=Deadline(), + record=ApprovalRecord( + "user", "owner/repo", datetime.now(UTC).strftime("%Y-%m-%dT%H:%M:%SZ"), 60 + ), + ) + + +def make_rows(park: ApprovalPark) -> tuple[dict, dict]: + record, deadline_ms = checkpoint._record(park) + task = { + "task_id": park.task_id, + "user_id": record.user_id, + "repo": record.repo, + "status": "AWAITING_APPROVAL", + "compute_type": "lambda-microvm", + "session_id": park.microvm_id, + "compute_metadata": {"microvmId": park.microvm_id, "endpoint": "https://synthetic.invalid"}, + "awaiting_approval_request_id": park.request_id, + "microvm_lifecycle": { + "version": 1, + "generation": "suspend-generation", + "microvm_id": park.microvm_id, + "request_id": park.request_id, + "action": "suspend", + "requested_at_ms": int(datetime.now(UTC).timestamp() * 1000), + "deadline_ms": deadline_ms, + }, + } + approval = { + "task_id": park.task_id, + "request_id": park.request_id, + "user_id": record.user_id, + "repo": record.repo, + "created_at": record.created_at, + "timeout_s": record.timeout_s, + "status": "PENDING", + } + return task, approval + + +def serialize(item: dict) -> dict: + return {key: TypeSerializer().serialize(value) for key, value in item.items()} + + +@pytest.fixture(autouse=True) +def _progress(): + _reset_circuit_breakers() + yield + _reset_circuit_breakers() + + +@pytest.fixture +def state(monkeypatch): + monkeypatch.setenv("TASK_TABLE_NAME", "tasks") + monkeypatch.setenv("TASK_APPROVALS_TABLE_NAME", "approvals") + monkeypatch.setenv("TASK_EVENTS_TABLE_NAME", "events") + park = make_park() + task, approval = make_rows(park) + client = MagicMock() + client.get_item.side_effect = lambda **kw: { + "Item": serialize(task if kw["TableName"] == "tasks" else approval) + } + monkeypatch.setattr("aws_session.tenant_client", lambda *_a, **_kw: client) + monkeypatch.setattr("aws_session.refresh_microvm_credentials", MagicMock()) + return park, task, approval, client + + +class TestCheckpoint: + def test_checkpoint_is_an_acknowledged_cross_table_transaction(self, state): + park, task, approval, client = state + checkpoint.checkpoint_before_suspend(park) + assert all(call.kwargs["ConsistentRead"] for call in client.get_item.call_args_list) + operations = client.transact_write_items.call_args.kwargs["TransactItems"] + assert len(operations) == 3 + assert [op["ConditionCheck"]["TableName"] for op in operations[:2]] == [ + "tasks", + "approvals", + ] + assert operations[0]["ConditionCheck"]["ExpressionAttributeValues"][":intent"] == ( + TypeSerializer().serialize(task["microvm_lifecycle"]) + ) + assert ":pending" in operations[1]["ConditionCheck"]["ConditionExpression"] + put = operations[2]["Put"] + assert put["TableName"] == "events" + item = {key: TypeDeserializer().deserialize(value) for key, value in put["Item"].items()} + assert item["task_id"] == park.task_id + assert item["metadata"]["milestone"] == "microvm_suspend_checkpoint" + assert item["metadata"]["generation"] == task["microvm_lifecycle"]["generation"] + assert item["metadata"]["request_id"] == approval["request_id"] + assert "ClientRequestToken" not in client.transact_write_items.call_args.kwargs + assert task["status"] == "AWAITING_APPROVAL" + assert approval["status"] == "PENDING" + + @pytest.mark.parametrize( + ("row", "field", "value"), + [ + ("task", "task_id", "other"), + ("task", "user_id", "other"), + ("task", "repo", "other/repo"), + ("task", "status", "CANCELLED"), + ("task", "compute_type", "agentcore"), + ("task", "session_id", "other-vm"), + ("task", "compute_metadata", {"microvmId": "other-vm"}), + ("task", "awaiting_approval_request_id", "other-request"), + ("intent", "version", 2), + ("intent", "version", True), + ("intent", "generation", ""), + ("intent", "microvm_id", "other-vm"), + ("intent", "request_id", "other-request"), + ("intent", "action", "resume"), + ("intent", "deadline_ms", 1), + ("intent", "requested_at_ms", -1), + ("approval", "task_id", "other"), + ("approval", "request_id", "other-request"), + ("approval", "user_id", "other"), + ("approval", "repo", "other/repo"), + ("approval", "created_at", "2020-01-01T00:00:00Z"), + ("approval", "timeout_s", 61), + ("approval", "timeout_s", True), + ("approval", "status", "APPROVED"), + ], + ) + def test_changed_identity_or_deadline_never_writes(self, state, row, field, value): + park, task, approval, client = state + target = {"task": task, "approval": approval, "intent": task["microvm_lifecycle"]}[row] + target[field] = value + with pytest.raises(LifecycleUnavailable): + checkpoint.checkpoint_before_suspend(park) + client.transact_write_items.assert_not_called() + + @pytest.mark.parametrize("stage", ["task", "approval", "write"]) + def test_missing_rows_and_uncertain_write_do_not_acknowledge(self, state, stage): + park, task, _, client = state + if stage == "write": + client.transact_write_items.side_effect = TimeoutError("lost response") + expected = TimeoutError + else: + client.get_item.side_effect = ( + [{}] if stage == "task" else [{"Item": serialize(task)}, {}] + ) + expected = LifecycleUnavailable + with pytest.raises(expected): + checkpoint.checkpoint_before_suspend(park) + + def test_unconfigured_or_disabled_progress_cannot_checkpoint(self, state, monkeypatch): + from progress_writer import _ProgressWriter + + park, _, _, client = state + monkeypatch.delenv("TASK_EVENTS_TABLE_NAME") + with pytest.raises(RuntimeError, match="progress table"): + checkpoint.checkpoint_before_suspend(park) + monkeypatch.setenv("TASK_EVENTS_TABLE_NAME", "events") + _ProgressWriter(park.task_id)._disabled = True + with pytest.raises(RuntimeError, match="progress table"): + checkpoint.checkpoint_before_suspend(park) + client.transact_write_items.assert_not_called() + + def test_expired_original_deadline_cannot_suspend(self, state): + park, _, _, client = state + assert isinstance(park.deadline, Deadline) + park.deadline.remaining = 0 + with pytest.raises(LifecycleUnavailable, match="deadline elapsed"): + checkpoint.checkpoint_before_suspend(park) + client.transact_write_items.assert_not_called() + + @pytest.mark.parametrize("status", ["PENDING", "APPROVED", "DENIED", "TIMED_OUT", "STRANDED"]) + def test_expired_wake_retains_original_gate_for_existing_decision_loop( + self, state, status, monkeypatch + ): + park, task, approval, client = state + task["microvm_lifecycle"].update(action="resume", generation="resume-generation") + approval["status"] = status + before = copy.deepcopy(approval) + assert isinstance(park.deadline, Deadline) + park.deadline.remaining = 0 + original_deadline = park.deadline + refresh = MagicMock() + monkeypatch.setattr("aws_session.refresh_microvm_credentials", refresh) + client.get_item.side_effect = lambda **kw: ( + {"Item": serialize(task if kw["TableName"] == "tasks" else approval)} + if refresh.called + else pytest.fail("AWS read occurred before credential refresh") + ) + checkpoint.refresh_and_reconcile_after_resume(park) + refresh.assert_called_once_with(park.task_id) + assert park.deadline is original_deadline + assert park.deadline.remaining_s() == 0 + assert approval == before + operations = client.transact_write_items.call_args.kwargs["TransactItems"] + assert len(operations) == 2 + assert all(set(operation) == {"ConditionCheck"} for operation in operations) + assert "IN (:pending, :approved" in operations[1]["ConditionCheck"]["ConditionExpression"] + + def test_failed_refresh_prevents_any_aws_reconciliation(self, state, monkeypatch): + park, _, _, client = state + monkeypatch.setattr( + "aws_session.refresh_microvm_credentials", MagicMock(side_effect=RuntimeError("denied")) + ) + with pytest.raises(RuntimeError, match="denied"): + checkpoint.refresh_and_reconcile_after_resume(park) + client.get_item.assert_not_called() + client.transact_write_items.assert_not_called() + + +LOCAL_ENDPOINT = os.environ.get("ABCA_DDB_LOCAL_ENDPOINT", "") + + +@pytest.fixture +def local_tables(monkeypatch): + import boto3 + from botocore.config import Config + + # Never let an accidentally set endpoint create tables in an AWS account. + endpoint = urlsplit(LOCAL_ENDPOINT) + assert endpoint.scheme == "http" and endpoint.hostname == "127.0.0.1" + client = boto3.client( + "dynamodb", + endpoint_url=LOCAL_ENDPOINT, + region_name="us-west-2", + aws_access_key_id="SYNTHETIC", + aws_secret_access_key="synthetic", + config=Config(connect_timeout=1, read_timeout=2, retries={"total_max_attempts": 1}), + ) + names = {} + try: + for kind, sort_key, env in [ + ("tasks", None, "TASK_TABLE_NAME"), + ("approvals", "request_id", "TASK_APPROVALS_TABLE_NAME"), + ("events", "event_id", "TASK_EVENTS_TABLE_NAME"), + ]: + name = f"p3-{kind}-{uuid.uuid4().hex}" + keys = ["task_id", *([sort_key] if sort_key else [])] + client.create_table( + TableName=name, + KeySchema=[ + {"AttributeName": key, "KeyType": "HASH" if index == 0 else "RANGE"} + for index, key in enumerate(keys) + ], + AttributeDefinitions=[{"AttributeName": key, "AttributeType": "S"} for key in keys], + BillingMode="PAY_PER_REQUEST", + ) + names[kind] = name + monkeypatch.setenv(env, name) + monkeypatch.setattr("aws_session.tenant_client", lambda *_a, **_kw: client) + monkeypatch.setattr("aws_session.refresh_microvm_credentials", MagicMock()) + yield client, names + finally: + for name in names.values(): + client.delete_table(TableName=name) + + +@pytest.mark.skipif(not LOCAL_ENDPOINT, reason="opt-in DynamoDB Local condition verification") +class TestDynamoLocalCheckpoint: + @pytest.mark.parametrize("race", [None, "cancel", "approve", "deadline", "intent"]) + def test_real_suspend_transaction_guards_races(self, local_tables, monkeypatch, race): + client, names = local_tables + park = make_park() + task, approval = make_rows(park) + client.put_item(TableName=names["tasks"], Item=serialize(task)) + client.put_item(TableName=names["approvals"], Item=serialize(approval)) + transact = client.transact_write_items + + def race_then_transact(**kwargs): + if race == "cancel": + task["status"] = "CANCELLED" + elif race == "approve": + approval["status"] = "APPROVED" + elif race == "deadline": + approval["timeout_s"] += 1 + elif race == "intent": + task["microvm_lifecycle"].update(action="resume", generation="new-generation") + client.put_item(TableName=names["tasks"], Item=serialize(task)) + client.put_item(TableName=names["approvals"], Item=serialize(approval)) + return transact(**kwargs) + + monkeypatch.setattr(client, "transact_write_items", race_then_transact) + if race: + with pytest.raises(ClientError) as error: + checkpoint.checkpoint_before_suspend(park) + assert error.value.response["Error"]["Code"] == "TransactionCanceledException" + else: + checkpoint.checkpoint_before_suspend(park) + assert client.scan(TableName=names["events"], ConsistentRead=True)["Count"] == ( + 0 if race else 1 + ) + + @pytest.mark.parametrize("race", ["approve", "cancel"]) + def test_real_resume_transaction_accepts_decision_but_rejects_cancellation( + self, local_tables, monkeypatch, race + ): + client, names = local_tables + park = make_park() + task, approval = make_rows(park) + task["microvm_lifecycle"].update(action="resume", generation="wake") + client.put_item(TableName=names["tasks"], Item=serialize(task)) + client.put_item(TableName=names["approvals"], Item=serialize(approval)) + transact = client.transact_write_items + + def race_then_transact(**kwargs): + if race == "approve": + approval["status"] = "APPROVED" + client.put_item(TableName=names["approvals"], Item=serialize(approval)) + else: + task["status"] = "CANCELLED" + client.put_item(TableName=names["tasks"], Item=serialize(task)) + return transact(**kwargs) + + monkeypatch.setattr(client, "transact_write_items", race_then_transact) + if race == "cancel": + with pytest.raises(ClientError) as error: + checkpoint.refresh_and_reconcile_after_resume(park) + assert error.value.response["Error"]["Code"] == "TransactionCanceledException" + else: + checkpoint.refresh_and_reconcile_after_resume(park) + assert client.scan(TableName=names["events"], ConsistentRead=True)["Count"] == 0 + + @pytest.mark.parametrize("wake", ["approved", "expired_pending", "cancelled", "changed_intent"]) + def test_http_hooks_use_real_transactions_and_preserve_original_gate(self, local_tables, wake): + import httpx + + import server + from microvm_lifecycle import register_task, unregister_task + + client, names = local_tables + original = make_park() + task, approval = make_rows(original) + client.put_item(TableName=names["tasks"], Item=serialize(task)) + client.put_item(TableName=names["approvals"], Item=serialize(approval)) + context = register_task(original.task_id, original.microvm_id) + + async def exercise(): + await context.tool_started(original.tool_use_id) + park = context.park_approval( + original.request_id, original.tool_use_id, original.deadline, record=original.record + ) + assert park is not None + async with httpx.AsyncClient( + transport=httpx.ASGITransport(app=server.app), base_url="http://test" + ) as http: + prefix = server.MICROVM_HOOK_PREFIX + for _ in range(2): + assert (await http.post(prefix + "/suspend", json={})).status_code == 200 + assert client.scan(TableName=names["events"], ConsistentRead=True)["Count"] == 1 + task["microvm_lifecycle"].update(action="resume", generation="wake") + if wake == "approved": + approval["status"] = "APPROVED" + elif wake == "expired_pending": + assert isinstance(original.deadline, Deadline) + original.deadline.remaining = 0 + elif wake == "cancelled": + task["status"] = "CANCELLED" + else: + task["microvm_lifecycle"]["request_id"] = "another-gate" + client.put_item(TableName=names["tasks"], Item=serialize(task)) + client.put_item(TableName=names["approvals"], Item=serialize(approval)) + + response = await http.post(prefix + "/resume", json={"microvmId": "vm"}) + if wake in {"cancelled", "changed_intent"}: + assert response.status_code == 409 + with pytest.raises(LifecycleUnavailable): + await context.wait_until_open() + else: + assert response.status_code == 200 + await asyncio.wait_for(context.leave_approval(park), timeout=1) + assert park.deadline is original.deadline + if wake == "expired_pending": + assert park.deadline.remaining_s() == 0 + assert (await http.post(prefix + "/resume", json={})).status_code == 200 + for kind, expected in [("tasks", task), ("approvals", approval)]: + rows = client.scan(TableName=names[kind], ConsistentRead=True)["Items"] + assert rows == [serialize(expected)] + assert client.scan(TableName=names["events"], ConsistentRead=True)["Count"] == 1 + + try: + asyncio.run(exercise()) + finally: + unregister_task(context) diff --git a/agent/tests/test_microvm_http.py b/agent/tests/test_microvm_http.py new file mode 100644 index 000000000..350a8f154 --- /dev/null +++ b/agent/tests/test_microvm_http.py @@ -0,0 +1,318 @@ +# Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. +# SPDX-License-Identifier: MIT-0 + +"""Exercise production HTTP routes with real controller state and bounded work.""" + +from __future__ import annotations + +import asyncio +import threading +import time +from dataclasses import dataclass +from unittest.mock import MagicMock + +import httpx +import pytest +from fastapi import Request + +import microvm_http +import microvm_lifecycle as lifecycle +import server + +PREFIX = server.MICROVM_HOOK_PREFIX + + +@dataclass +class Deadline: + remaining: float = 60 + + def remaining_s(self) -> float: + return self.remaining + + +@pytest.fixture +def anyio_backend(): + return "asyncio" + + +@pytest.fixture +def context(): + context = lifecycle.register_task("http-task", "http-vm") + yield context + lifecycle.unregister_task(context) + + +@pytest.fixture +def callbacks(monkeypatch): + suspend, resume = MagicMock(), MagicMock() + monkeypatch.setattr(microvm_http, "checkpoint_before_suspend", suspend) + monkeypatch.setattr(microvm_http, "refresh_and_reconcile_after_resume", resume) + return suspend, resume + + +@pytest.fixture +async def client(): + async with httpx.AsyncClient( + transport=httpx.ASGITransport(app=server.app), base_url="http://test" + ) as client: + yield client + + +async def park(context): + await context.tool_started("tool") + deadline = Deadline() + parked = context.park_approval("gate", "tool", deadline) + assert parked is not None + return parked, deadline + + +@pytest.mark.anyio +class TestLifecycleHttp: + async def test_duplicate_hooks_acknowledge_without_repeating_work( + self, context, callbacks, client + ): + suspend, resume = callbacks + parked, deadline = await park(context) + for body in [{"microvmId": "http-vm"}, {"microvmId": ""}]: + response = await client.post(PREFIX + "/suspend", json=body) + assert response.status_code == 200 + assert response.json()["request_id"] == "gate" + suspend.assert_called_once_with(parked) + leave = asyncio.create_task(context.leave_approval(parked)) + await asyncio.sleep(0.03) + assert not leave.done() + deadline.remaining = 0 + response = await client.post(PREFIX + "/resume", content=b"") + assert response.status_code == 200 + await leave + response = await client.post(PREFIX + "/resume", json={"microvmId": "http-vm"}) + assert response.status_code == 200 + resume.assert_called_once_with(parked) + assert parked.deadline is deadline + assert deadline.remaining_s() == 0 + assert (await client.post(PREFIX + "/suspend", json={})).status_code == 409 + + async def test_new_gate_cannot_reuse_an_old_wake_acknowledgment( + self, context, callbacks, client + ): + parked, _ = await park(context) + assert (await client.post(PREFIX + "/suspend", json={})).status_code == 200 + assert (await client.post(PREFIX + "/resume", json={})).status_code == 200 + await context.leave_approval(parked) + context.tool_finished("tool") + await context.tool_started("next-tool") + next_park = context.park_approval("next-gate", "next-tool", Deadline()) + assert next_park is not None + assert (await client.post(PREFIX + "/resume", json={})).status_code == 409 + assert (await client.post(PREFIX + "/suspend", json={})).status_code == 200 + response = await client.post(PREFIX + "/resume", json={}) + assert response.status_code == 200 + assert response.json()["request_id"] == "next-gate" + callbacks[1].assert_called_with(next_park) + assert callbacks[1].call_count == 2 + + @pytest.mark.parametrize( + ("content", "status"), + [ + (b"{", 400), + (b"[]", 400), + (b'{"microvmId":1}', 400), + (b'{"microvmId":" http-vm"}', 400), + (b"x" * 4097, 413), + ], + ) + async def test_invalid_body_never_reaches_lifecycle_work( + self, context, callbacks, client, content, status + ): + await park(context) + assert (await client.post(PREFIX + "/suspend", content=content)).status_code == status + assert (await client.post(PREFIX + "/resume", content=content)).status_code == status + for callback in callbacks: + callback.assert_not_called() + + async def test_wrong_vm_or_no_registered_task_rejects(self, context, callbacks, client): + await park(context) + assert ( + await client.post(PREFIX + "/suspend", json={"microvmId": "another-vm"}) + ).status_code == 409 + lifecycle.unregister_task(context) + assert (await client.post(PREFIX + "/suspend", json={})).status_code == 409 + assert (await client.post(PREFIX + "/resume", json={})).status_code == 409 + for callback in callbacks: + callback.assert_not_called() + + async def test_unparked_resume_or_parallel_tools_never_acknowledges( + self, context, callbacks, client + ): + assert (await client.post(PREFIX + "/suspend", json={})).status_code == 409 + await park(context) + assert (await client.post(PREFIX + "/resume", json={})).status_code == 409 + await context.tool_started("parallel") + assert (await client.post(PREFIX + "/suspend", json={})).status_code == 409 + for callback in callbacks: + callback.assert_not_called() + + async def test_checkpoint_failure_keeps_guest_awake_and_disables_another_suspend( + self, context, callbacks, client + ): + suspend, _ = callbacks + parked, _ = await park(context) + suspend.side_effect = RuntimeError("do-not-leak-secret") + response = await client.post(PREFIX + "/suspend", json={}) + assert response.status_code == 503 + assert "do-not-leak-secret" not in response.text + assert (await client.post(PREFIX + "/suspend", json={})).status_code == 409 + await context.leave_approval(parked) + + async def test_failed_refresh_closes_barrier_without_retrying(self, context, callbacks, client): + _, resume = callbacks + await park(context) + assert (await client.post(PREFIX + "/suspend", json={})).status_code == 200 + resume.side_effect = RuntimeError("do-not-leak-secret") + response = await client.post(PREFIX + "/resume", json={}) + assert response.status_code == 503 + assert "do-not-leak-secret" not in response.text + with pytest.raises(lifecycle.LifecycleUnavailable): + await context.wait_until_open() + assert (await client.post(PREFIX + "/resume", json={})).status_code == 409 + resume.assert_called_once() + + @pytest.mark.parametrize("action", ["suspend", "resume"]) + async def test_timed_out_callback_cannot_acknowledge_late( + self, context, callbacks, client, monkeypatch, action + ): + await park(context) + if action == "resume": + assert (await client.post(PREFIX + "/suspend", json={})).status_code == 200 + entered, release, finished = threading.Event(), threading.Event(), threading.Event() + + def block(_park): + entered.set() + try: + assert release.wait(2) + finally: + finished.set() + + callbacks[action == "resume"].side_effect = block + monkeypatch.setattr(microvm_http, "LIFECYCLE_HANDLER_BUDGET_S", 0.05) + started = time.monotonic() + try: + response = await client.post(PREFIX + "/" + action, json={}) + assert response.status_code == 503 + assert response.json()["code"] == "MICROVM_LIFECYCLE_TIMEOUT" + assert time.monotonic() - started < 0.5 + assert entered.is_set() + finally: + release.set() + assert await asyncio.to_thread(finished.wait, 1) + assert (await client.post(PREFIX + "/" + action, json={})).status_code == 409 + if action == "resume": + with pytest.raises(lifecycle.LifecycleUnavailable): + await context.wait_until_open() + else: + await context.wait_until_open() + + async def test_terminate_invalidates_an_inflight_refresh( + self, context, callbacks, client, monkeypatch + ): + await park(context) + assert (await client.post(PREFIX + "/suspend", json={})).status_code == 200 + entered, release = threading.Event(), threading.Event() + + def refresh(_park): + entered.set() + assert release.wait(2) + + callbacks[1].side_effect = refresh + monkeypatch.setattr(server, "_debug_cw", MagicMock()) + request = asyncio.create_task(client.post(PREFIX + "/resume", json={})) + try: + assert await asyncio.to_thread(entered.wait, 1) + assert ( + await client.post(PREFIX + "/terminate", content=b"bad body") + ).status_code == 200 + with pytest.raises(lifecycle.LifecycleUnavailable): + await context.wait_until_open() + finally: + release.set() + assert (await request).status_code == 409 + with pytest.raises(lifecycle.LifecycleUnavailable): + await context.wait_until_open() + + async def test_concurrent_resume_reports_busy_while_first_finishes( + self, context, callbacks, client + ): + await park(context) + assert (await client.post(PREFIX + "/suspend", json={})).status_code == 200 + entered, release = threading.Event(), threading.Event() + + def refresh(_park): + entered.set() + assert release.wait(2) + + callbacks[1].side_effect = refresh + request = asyncio.create_task(client.post(PREFIX + "/resume", json={})) + try: + assert await asyncio.to_thread(entered.wait, 1) + assert (await client.post(PREFIX + "/resume", json={})).status_code == 409 + finally: + release.set() + assert (await request).status_code == 200 + callbacks[1].assert_called_once() + + async def test_request_body_read_shares_the_total_budget(self, context, callbacks, monkeypatch): + await park(context) + monkeypatch.setattr(microvm_http, "LIFECYCLE_HANDLER_BUDGET_S", 0.02) + + async def slow_receive(): + await asyncio.sleep(1) + return {"type": "http.request", "body": b"{}", "more_body": False} + + request = Request({"type": "http", "method": "POST", "headers": []}, slow_receive) + response = await microvm_http.microvm_suspend(request) + assert response.status_code == 503 + for callback in callbacks: + callback.assert_not_called() + + async def test_terminate_closes_barrier_and_answers_even_if_body_stalls( + self, context, monkeypatch + ): + await park(context) + monkeypatch.setattr(server, "_TERMINATE_BODY_BUDGET_SECONDS", 0.02) + monkeypatch.setattr(server, "_debug_cw", MagicMock()) + + async def slow_receive(): + with pytest.raises(lifecycle.LifecycleUnavailable): + await context.wait_until_open() + await asyncio.sleep(1) + return {"type": "http.request", "body": b"{}", "more_body": False} + + request = Request({"type": "http", "method": "POST", "headers": []}, slow_receive) + started = time.monotonic() + response = await server.microvm_terminate(request) + assert response["status"] == "acknowledged" + assert time.monotonic() - started < 0.5 + + async def test_expired_or_failed_progress_cannot_repeat_suspend_ack( + self, context, callbacks, client + ): + _, deadline = await park(context) + assert (await client.post(PREFIX + "/suspend", json={})).status_code == 200 + deadline.remaining = 0 + assert (await client.post(PREFIX + "/suspend", json={})).status_code == 409 + deadline.remaining = 60 + context.progress_write_failed() + assert (await client.post(PREFIX + "/suspend", json={})).status_code == 409 + callbacks[0].assert_called_once() + + async def test_validate_checks_new_routes_without_creating_aws_clients( + self, monkeypatch, client + ): + forbidden = MagicMock(side_effect=AssertionError("build hook initialized AWS")) + monkeypatch.setattr("aws_session.tenant_client", forbidden) + monkeypatch.setattr("aws_session.platform_client", forbidden) + response = await client.post(PREFIX + "/validate") + assert response.status_code == 200 + assert response.json()["checks"]["hook_routes_registered"] is True + forbidden.assert_not_called() + assert 0 < microvm_http.LIFECYCLE_HANDLER_BUDGET_S < microvm_http.LIFECYCLE_HOOK_TIMEOUT_S diff --git a/agent/tests/test_server.py b/agent/tests/test_server.py index 511420ba8..a0e98d3bd 100644 --- a/agent/tests/test_server.py +++ b/agent/tests/test_server.py @@ -1012,14 +1012,13 @@ def test_ready_does_not_start_a_pipeline(self, client, monkeypatch, warm_ready): with server._threads_lock: assert server._active_threads == [] - def test_suspend_and_resume_are_NOT_served(self, client): - # Declaring a hook nothing answers fails the corresponding build or - # lifecycle transition, so the construct declares exactly the hooks the - # agent serves. /validate + /terminate joined that set in P2; /suspend + - # /resume need the ComputeStrategy interface widening (P3), so they must - # still 404 — the assertion that keeps the construct honest. + def test_suspend_and_resume_require_a_registered_task(self, client): + # Runtime routes exist before the image declares the P3 capability. + # Snapshot warm-up has no task and cannot authorize a lifecycle change. for hook in ("suspend", "resume"): - assert client.post(f"{server.MICROVM_HOOK_PREFIX}/{hook}").status_code == 404 + response = client.post(f"{server.MICROVM_HOOK_PREFIX}/{hook}") + assert response.status_code == 409 + assert response.json()["code"] == "MICROVM_LIFECYCLE_UNAVAILABLE" class TestMicrovmReadyHookWarmUp: @@ -2546,7 +2545,9 @@ def test_reports_a_missing_hook_route(self, client, monkeypatch): assert "hook_routes_registered" in r.json()["failed_checks"] assert r.json()["missing_routes"] == [ "/typo/prefix/ready", + "/typo/prefix/resume", "/typo/prefix/run", + "/typo/prefix/suspend", "/typo/prefix/terminate", "/typo/prefix/validate", ] diff --git a/cdk/src/constructs/lambda-microvm-compute.ts b/cdk/src/constructs/lambda-microvm-compute.ts index e004deaac..7a7c3b2b5 100644 --- a/cdk/src/constructs/lambda-microvm-compute.ts +++ b/cdk/src/constructs/lambda-microvm-compute.ts @@ -108,9 +108,9 @@ const AGENT_HOOK_PORT = 8080; * surface: the service calls fixed well-known routes, which * {@link MICROVM_AGENT_HOOK_ROUTES} records and the live build/run logs confirm. * - * `DISABLED` is never emitted: a hook the agent does not serve is OMITTED rather - * than disabled, so the exactness test can assert the declared set in both - * directions (see `/suspend` + `/resume`, P3). + * `DISABLED` is never emitted: hooks outside the image's enabled capability are + * omitted. The guest now serves `/suspend` + `/resume`, but their image + * declaration remains gated on the P3 capability rollout and live verification. */ const HOOK_ENABLED = 'ENABLED'; @@ -121,8 +121,9 @@ const HOOK_ENABLED = 'ENABLED'; const MICROVM_HOOK_ROUTE_PREFIX = '/aws/lambda-microvms/runtime/v1'; /** - * The service's fixed hook ROUTES, keyed by hook name — the paths - * `agent/src/server.py` must serve. + * The service's fixed routes for the hooks this image currently enables. + * `agent/src/server.py` must serve each; additional guest routes alone do not + * enable an image capability. * * ## ⚠️ These are AGENT ROUTE CONSTANTS ONLY. Never send them to an AWS API. * diff --git a/cdk/test/constructs/lambda-microvm-compute.test.ts b/cdk/test/constructs/lambda-microvm-compute.test.ts index b13079aa9..8d6ce8e96 100644 --- a/cdk/test/constructs/lambda-microvm-compute.test.ts +++ b/cdk/test/constructs/lambda-microvm-compute.test.ts @@ -226,13 +226,11 @@ describe('LambdaMicrovmCompute — image provisioned from a managed base image', expect(Math.max(...MICROVM_SUPPORTED_MEMORY_MIB)).toBe(DEFAULT_MINIMUM_MEMORY_MIB); }); - test('declares EXACTLY the four hooks the agent serves (P2), and no more', () => { - // `toEqual` on the whole object, not per-key assertions: the invariant runs in - // BOTH directions. A hook the agent serves but the image does not declare is - // never called (the P2 R2 regression this replaces — the agent gained - // /validate and /terminate while the construct still advertised two hooks); - // a hook the image declares but the agent does not serve fails the - // corresponding build or lifecycle transition. Only an exact set catches both. + test('declares exactly the four enabled P2 image hooks until the P3 capability rollout', () => { + // Compare the whole enabled set. A declared hook must be served or the + // corresponding build/lifecycle transition fails. The guest additionally + // serves /suspend and /resume, but they stay undeclared until the P3 image + // capability rollout; a route alone must not opt existing workers into sleep. const images = template.findResources('AWS::Lambda::MicrovmImage'); const hooks = Object.values(images)[0]!.Properties.Hooks; expect(hooks.Port).toBe(8080); @@ -248,8 +246,8 @@ describe('LambdaMicrovmCompute — image provisioned from a managed base image', RunTimeoutInSeconds: 60, Terminate: 'ENABLED', // Near the BOTTOM of the service's 1–60 s window on purpose: the handler is - // a log-and-acknowledge with nothing to drain (progress writes are already - // durable per event), and the budget bounds how long teardown waits on a + // a barrier close plus log-and-acknowledge with no progress queue to drain + // (ordinary event writes are best effort), and the budget bounds teardown on a // WEDGED guest that is still holding admission-gating memory quota. TerminateTimeoutInSeconds: 15, }); diff --git a/cdk/test/scripts/check-constants-sync.test.ts b/cdk/test/scripts/check-constants-sync.test.ts index 4f2f88e17..84515bd2e 100644 --- a/cdk/test/scripts/check-constants-sync.test.ts +++ b/cdk/test/scripts/check-constants-sync.test.ts @@ -63,6 +63,7 @@ const FIXTURE_FILES = [ 'agent/src/server.py', 'agent/src/config.py', 'agent/src/payload_bootstrap.py', + 'agent/src/microvm_http.py', 'cdk/src/handlers/shared/payload-bootstrap.ts', 'cdk/src/constructs/lambda-microvm-compute.ts', ]; @@ -466,6 +467,29 @@ describe('check-constants-sync', () => { }); describe('hook-budget invariant', () => { + test.each(['LIFECYCLE_HANDLER_BUDGET_S', 'LIFECYCLE_HOOK_TIMEOUT_S'])( + 'rejects a hardcoded lifecycle budget %s', + (name) => { + const result = runInMutatedRepo((root) => { + write(root, 'agent/src/microvm_http.py', + `${read(root, 'agent/src/microvm_http.py')}\n${name}: float = 20\n`); + }); + expect(result.status).toBe(1); + expect(result.stderr).toContain(name); + }, + ); + + test.each([0, 1])('rejects a lifecycle handler budget %s seconds beyond the service timeout', (offset) => { + const result = runInMutatedRepo((root) => { + patchContract(root, (json) => { + json.microvm_hook_budgets.lifecycle_handler_budget_seconds = + json.microvm_hook_budgets.lifecycle_hook_timeout_seconds + offset; + }); + }); + expect(result.status).toBe(1); + expect(result.stderr).toContain('lifecycle_handler_budget_seconds must be <'); + }); + test('rejects a warm-up budget that does not fit inside the hook timeout', () => { // The relationship the two-sided contract exists for: a warm-up that cannot // answer inside the service's hook budget turns a runtime fix into a build diff --git a/contracts/constants.json b/contracts/constants.json index e8e9df1ad..546d7098a 100644 --- a/contracts/constants.json +++ b/contracts/constants.json @@ -73,7 +73,9 @@ "microvm_hook_budgets": { "ready_hook_timeout_seconds": 300, "warmup_total_budget_seconds": 240, - "warmup_required_timeout_seconds": 120 + "warmup_required_timeout_seconds": 120, + "lifecycle_hook_timeout_seconds": 30, + "lifecycle_handler_budget_seconds": 20 }, "linear_vault": { "_why": "The vault caches a grant keyed by the WHOLE token request, customParameters included. A one-token divergence between any two of these copies makes every resolve a cache miss, which post-#812 is reported as consent-required and can latch a healthy workspace revoked.", diff --git a/contracts/constants.md b/contracts/constants.md index 3bf857283..aca047b0f 100644 --- a/contracts/constants.md +++ b/contracts/constants.md @@ -24,6 +24,7 @@ the contract. This is the neutral location both runtimes read. | `agent/src/payload_bootstrap.py` | `SHARED_CONSTANTS["payload_bootstrap"]` | import-time | | `cdk/src/handlers/shared/payload-bootstrap.ts`, `cdk/src/constructs/payload-bootstrap-permissions.ts` | `payload_bootstrap` | import-time | | `agent/src/server.py` | `SHARED_CONSTANTS["microvm_platform_config"]`, `SHARED_CONSTANTS["microvm_hook_budgets"]` | import-time | +| `agent/src/microvm_http.py` | `SHARED_CONSTANTS["microvm_hook_budgets"]` | import-time | | `cdk/src/handlers/shared/types.ts`, `jira-app-actor.ts` | `../../../../contracts/constants.json` | synth-time `import` | | `cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts` | `microvm_platform_config` | synth-time `import`, read per session start | | `cdk/src/constructs/lambda-microvm-compute.ts` | `microvm_hook_budgets` | synth-time `import` | @@ -91,7 +92,9 @@ JSON at TypeScript compile time via `resolveJsonModule`. "microvm_hook_budgets": { "ready_hook_timeout_seconds": 300, "warmup_total_budget_seconds": 240, - "warmup_required_timeout_seconds": 120 + "warmup_required_timeout_seconds": 120, + "lifecycle_hook_timeout_seconds": 30, + "lifecycle_handler_budget_seconds": 20 } } ``` @@ -196,7 +199,7 @@ gate. purpose: a cold 225 MiB `exec` has no predictable duration, which is the lesson of P2-F5. -Unlike every other block here, these three are not independent tuning bounds — they +These three warm-up values are not independent tuning bounds — they are a **relationship**: `warmup_required < warmup_total < ready_hook`. The warm-up must finish inside the budget the service holds the hook to, or a fix for a runtime failure turns into a build failure. A relationship cannot be enforced from one side, @@ -207,6 +210,15 @@ literal re-declaration on **either** side — the Python constants *and* `agent/src/server.py` re-checks the same ordering at import time, so a bad contract fails the drift check *and* the image build. +The runtime lifecycle pair has its own relationship: +`lifecycle_handler_budget_seconds < lifecycle_hook_timeout_seconds`. The 20-second +handler limit covers reading the body, draining activity and checkpoint/refresh +work together. The planned 30-second service hook timeout leaves response headroom. +`microvm_http.py` checks the ordering at import time; the drift script checks +positive integer values, ordering and hardcoded Python redeclarations. The image +does not declare suspend/resume hooks yet; its later capability rollout must use +this service timeout. These values do not enable automatic suspension. + The published CLI package contains only `lib/`, so it cannot load the repository contract at runtime. It mirrors these values as literals and `cli/test/constants-parity.test.ts` makes drift a CI failure. The standalone diff --git a/docs/decisions/ADR-021-lambda-microvms-compute-backend.md b/docs/decisions/ADR-021-lambda-microvms-compute-backend.md index a08caff8a..b0a9a89d9 100644 --- a/docs/decisions/ADR-021-lambda-microvms-compute-backend.md +++ b/docs/decisions/ADR-021-lambda-microvms-compute-backend.md @@ -1,6 +1,6 @@ # ADR-021: AWS Lambda MicroVMs as a third ComputeStrategy backend -> **Implementation status (2026-09-14):** P1 ([#689](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/689)) and P2 ([#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733)) are merged. The takeover branch now has a [clean deployment](../verification/645-p2-clean-deployment-20260913.md) with bootstrap 1.7.0 and a [verified managed-image update to 2.0](../verification/645-microvm-image-rebuild-20260914.md), plus [live coding, PR iteration and cancellation evidence](../verification/645-p2-live-task-20260914.md), without manual IAM workarounds. Remaining P2 failure/recovery and permission/network checks prevent full acceptance. Local P3 foundations include suspend/resume commands, explicit VM observations, guarded persistent intent, a policy helper, approval deadlines that survive a frozen clock, the [guest pause controller](../verification/645-p3-guest-barrier.md), and [scoped Claude credentials with retained-client refresh](../verification/645-p3-credentials.md). Production HTTP lifecycle hooks and supervisor/approval-handler integration remain unimplemented; automatic suspension is not enabled. This ADR defines P1–P3, not a P4. See the [current review](../verification/645-p3-readiness-review.md) and [P3 implementation plan](../verification/645-p3-implementation-plan.md). +> **Implementation status (2026-09-14):** P1 ([#689](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/689)) and P2 ([#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733)) are merged. The takeover branch now has a [clean deployment](../verification/645-p2-clean-deployment-20260913.md) with bootstrap 1.7.0 and a [verified managed-image update to 2.0](../verification/645-microvm-image-rebuild-20260914.md), plus [live coding, PR iteration and cancellation evidence](../verification/645-p2-live-task-20260914.md), without manual IAM workarounds. Remaining P2 failure/recovery and permission/network checks prevent full acceptance. Local P3 foundations include suspend/resume commands, explicit VM observations, guarded persistent intent, a policy helper, approval deadlines that survive a frozen clock, the [guest pause controller](../verification/645-p3-guest-barrier.md), and [scoped Claude credentials with retained-client refresh](../verification/645-p3-credentials.md). The [production HTTP hooks](../verification/645-p3-lifecycle-hooks.md) now connect the guest barrier to atomic checkpoints and credential/gate reconciliation locally. Image capability and supervisor/approval-handler integration remain unfinished; automatic suspension is not enabled. This ADR defines P1–P3, not a P4. See the [current review](../verification/645-p3-readiness-review.md) and [P3 implementation plan](../verification/645-p3-implementation-plan.md). **Status:** proposed **Date:** 2026-07-29 diff --git a/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md b/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md index a515c3030..05949bab8 100644 --- a/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md +++ b/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md @@ -4,7 +4,7 @@ title: Adr 021 lambda microvms compute backend # ADR-021: AWS Lambda MicroVMs as a third ComputeStrategy backend -> **Implementation status (2026-09-14):** P1 ([#689](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/689)) and P2 ([#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733)) are merged. The takeover branch now has a [clean deployment](/sample-autonomous-cloud-coding-agents/architecture/645-p2-clean-deployment-20260913) with bootstrap 1.7.0 and a [verified managed-image update to 2.0](/sample-autonomous-cloud-coding-agents/architecture/645-microvm-image-rebuild-20260914), plus [live coding, PR iteration and cancellation evidence](/sample-autonomous-cloud-coding-agents/architecture/645-p2-live-task-20260914), without manual IAM workarounds. Remaining P2 failure/recovery and permission/network checks prevent full acceptance. Local P3 foundations include suspend/resume commands, explicit VM observations, guarded persistent intent, a policy helper, approval deadlines that survive a frozen clock, the [guest pause controller](/sample-autonomous-cloud-coding-agents/architecture/645-p3-guest-barrier), and [scoped Claude credentials with retained-client refresh](/sample-autonomous-cloud-coding-agents/architecture/645-p3-credentials). Production HTTP lifecycle hooks and supervisor/approval-handler integration remain unimplemented; automatic suspension is not enabled. This ADR defines P1–P3, not a P4. See the [current review](/sample-autonomous-cloud-coding-agents/architecture/645-p3-readiness-review) and [P3 implementation plan](/sample-autonomous-cloud-coding-agents/architecture/645-p3-implementation-plan). +> **Implementation status (2026-09-14):** P1 ([#689](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/689)) and P2 ([#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733)) are merged. The takeover branch now has a [clean deployment](/sample-autonomous-cloud-coding-agents/architecture/645-p2-clean-deployment-20260913) with bootstrap 1.7.0 and a [verified managed-image update to 2.0](/sample-autonomous-cloud-coding-agents/architecture/645-microvm-image-rebuild-20260914), plus [live coding, PR iteration and cancellation evidence](/sample-autonomous-cloud-coding-agents/architecture/645-p2-live-task-20260914), without manual IAM workarounds. Remaining P2 failure/recovery and permission/network checks prevent full acceptance. Local P3 foundations include suspend/resume commands, explicit VM observations, guarded persistent intent, a policy helper, approval deadlines that survive a frozen clock, the [guest pause controller](/sample-autonomous-cloud-coding-agents/architecture/645-p3-guest-barrier), and [scoped Claude credentials with retained-client refresh](/sample-autonomous-cloud-coding-agents/architecture/645-p3-credentials). The [production HTTP hooks](/sample-autonomous-cloud-coding-agents/architecture/645-p3-lifecycle-hooks) now connect the guest barrier to atomic checkpoints and credential/gate reconciliation locally. Image capability and supervisor/approval-handler integration remain unfinished; automatic suspension is not enabled. This ADR defines P1–P3, not a P4. See the [current review](/sample-autonomous-cloud-coding-agents/architecture/645-p3-readiness-review) and [P3 implementation plan](/sample-autonomous-cloud-coding-agents/architecture/645-p3-implementation-plan). **Status:** proposed **Date:** 2026-07-29 diff --git a/docs/verification/645-p3-credentials.md b/docs/verification/645-p3-credentials.md index b763902ec..9a876460d 100644 --- a/docs/verification/645-p3-credentials.md +++ b/docs/verification/645-p3-credentials.md @@ -24,7 +24,7 @@ scoped container provider and keeps keys out of Claude's stale export cache. AgentCore/ECS/local attribution retains its existing helper behavior. `aws_session.py` exports a coherent key/expiry pair under botocore's refresh lock. -The future resume callback can force recorded ambient providers to refresh first, +The production resume callback forces recorded ambient providers to refresh first, then force the original tenant credential object to renew with the original STS tags. Existing clients keep their references. Changing the configured identity after session construction is rejected. Mandatory renewal propagates failure @@ -110,9 +110,9 @@ normal completion and failure/cancellation for MicroVM and other backends. Before automatic suspension can ship: -1. Connect the refresh operation to the bounded production `/resume` callback, - followed by durable gate/deadline reconciliation. Finish acknowledged checkpoint - writes and `/suspend` admission. +1. **Implemented locally in the [HTTP hook milestone](./645-p3-lifecycle-hooks.md):** + bounded `/resume` invokes refresh before atomic task/gate reconciliation; + `/suspend` requires an acknowledged checkpoint. 2. Verify actual MicroVM runtime credential-provider type and renewal after sleep. Static or unknown providers currently reject resume; rereading an unchanged environment is not proof of renewal. Botocore's forced-refresh private API is diff --git a/docs/verification/645-p3-guest-barrier.md b/docs/verification/645-p3-guest-barrier.md index f5d454343..356ef6f98 100644 --- a/docs/verification/645-p3-guest-barrier.md +++ b/docs/verification/645-p3-guest-barrier.md @@ -36,15 +36,15 @@ AWS, authorize a human decision or extend an approval's deadline. - `/run` and successful controller resume reseed Python's application PRNG with `os.urandom(32)`. This does not make `random` suitable for secrets. -Concurrent lifecycle requests receive a controlled rejection rather than -sharing ownership or holding a lock across network work. HTTP retry/duplicate -acknowledgment semantics still need integration with the future routes. +Concurrent lifecycle requests receive a controlled rejection without holding a +lock across network work. The later [HTTP hook milestone](./645-p3-lifecycle-hooks.md) +adds cached successful acknowledgments and clears an old wake result for each new gate. ## Why the image is not eligible for sleep yet -The controller's checkpoint and refresh operations are currently test-supplied -callbacks. Production HTTP handlers must supply bounded, acknowledged database -checks and credential renewal before returning success. +The later [HTTP hook milestone](./645-p3-lifecycle-hooks.md) supplies production +callbacks for atomic checkpoint writes, credential renewal and gate reconciliation. +Image capability, supervisor integration and live service verification remain open. The worker has several independent credential consumers: @@ -70,8 +70,9 @@ now exercises that actual CLI. `credential_process` renewed successfully but fell through to ambient credentials on failure. The chosen MicroVM path uses a single authenticated, scoped container provider; renewal waits before signing, and failure sends no model request. Python refresh updates retained credential -objects with the original identity. Production HTTP callbacks and live runtime -provider/snapshot verification are still open. No minified CLI internals were patched. +objects with the original identity. The HTTP resume callback now invokes this +operation; live runtime provider/snapshot verification remains open. +No minified CLI internals were patched. Known background tool flags are tracked, but arbitrary shell/MCP subprocesses may also detach work. Their safe-point behavior still requires a conservative @@ -103,10 +104,11 @@ budgets, durable supervisor recovery or complete P3 acceptance. ## Remaining integration order -1. **Local credential implementation complete:** verify the deployed runtime - provider and snapshot behavior; connect retained-client refresh to the resume callback. -2. Implement the production checkpoint and HTTP suspend/resume routes with - shared service budgets, duplicate-request handling and deterministic teardown. +1. **Local credential implementation complete:** retained-client refresh is now + connected to resume; verify the deployed runtime provider and snapshot behavior. +2. **Local HTTP integration complete:** acknowledged production checkpoints, + shared budgets, duplicate handling and teardown are covered in the + [hook verification](./645-p3-lifecycle-hooks.md). 3. Bind lifecycle capability to the image/version that launched each worker; declare compatible hooks, initially leaving automatic suspension disabled. 4. Wire persistent intent/policy into durable supervisor polling with bounded diff --git a/docs/verification/645-p3-implementation-plan.md b/docs/verification/645-p3-implementation-plan.md index d319c31ab..0f8a79b48 100644 --- a/docs/verification/645-p3-implementation-plan.md +++ b/docs/verification/645-p3-implementation-plan.md @@ -6,6 +6,15 @@ Prepared 2026-09-13 from `main` `5e10038c7e28179b302ac4de78b709795aeba3ce`. Read Prerequisite work is tracked here on `fix/645-microvm-readiness`. “Completed” means implemented and checked locally; AWS deployment and live verification have separate completion gates below. +**Latest local P3 milestone (2026-09-14):** production +[worker suspend/resume hooks](./645-p3-lifecycle-hooks.md) now connect the guest +barrier to atomic checkpoint writes and retained-credential refresh followed by +task/gate reconciliation. Duplicate acknowledgments stay within one approval +generation; a new gate cannot reuse an old wake result. Shared handler/service +budgets are 20/30 seconds. Image declaration/capability, durable supervisor +integration, approval-triggered wake and live sleep/wake verification remain open. +No deployment or automatic suspension was enabled in this milestone. + **Live infrastructure and image deployed (2026-09-14):** the [clean P2 deployment record](./645-p2-clean-deployment-20260913.md) tracks the new Oregon environment, four deployment fixes and actual verification results. @@ -67,13 +76,14 @@ is superseded by these records. - [x] Verify automatic pre/post npm checks under temporary overrides and worker-reported failure cleanup in AWS. - [x] Give managed image builds immutable, checksum-verified artifacts and require their digest in deployment context; packaging, construct, stack and CDK-nag regressions and the full build pass. - [x] Verify a normal CloudFormation update builds and activates image `2.0` from the changed artifact URI; repeat packaging reuses the verified object and a same-assembly redeploy reports no changes. -- [ ] Make repository mise tasks available and verify the restored default commands; the CLI addition and temporary overrides were withdrawn at user request. +- Deferred at user request: publish mise tasks in the target repository and verify its default commands. The CLI addition and temporary overrides were withdrawn; this repository configuration work is outside the current P3 implementation. - Optional: production nesting remains unimplemented; the clean deployment uses 474 of the root stack's 500 resource slots. P3 does not inherently require nesting. Recheck the count for supported feature combinations and validate the split/migration if adopted. - [x] Add mandatory pause/wake command methods across all three compute strategies, with explicit unsupported results and bounded MicroVM requests. - [x] Keep the original approval deadline through database writes and polling, including frozen/backward clocks; preserve decision races and cancellation. - [x] Save gate/VM-bound lifecycle intent with stale-writer protection; add explicit VM observations and a tested policy helper. - [x] Add the guest pause controller, original-gate registration, parallel-tool tracking, progress acknowledgment tracking, heartbeat/read drain and generation-guarded wake completion; reseed the application PRNG at run and controller resume. -- [ ] Add compatible agent hooks and credential/durability barriers. +- [x] Add production agent hooks with acknowledged checkpoints, retained ambient/tenant credential renewal, a sole scoped Claude provider and atomic task/gate reconciliation; verify duplicates, original deadlines, timeout and teardown behavior locally. +- [ ] Declare compatible image hooks using shared budgets and bind lifecycle capability to the actual image/version used by each worker; keep automatic sleep disabled until integration/live acceptance. - [ ] Persist bounded poll/recovery counters and connect lifecycle policy to the supervisor. - [ ] Connect supervisor and approval handlers, then verify the complete P3 sleep/wake lifecycle in AWS. @@ -393,26 +403,26 @@ keeps suspension disabled even after later successful writes. Failed or timed-ou wake callbacks cannot release coding through a late thread completion. The controller and `/run` reseed `random` using fresh OS entropy. -This is **not the completed hook implementation**. No HTTP suspend/resume routes, -image capability, production checkpoint/refresh callbacks, IAM changes or -automatic sleep have been enabled. See the [guest barrier review](./645-p3-guest-barrier.md) -for the implemented boundary, tests and remaining credential/subprocess work. +The [guest barrier review](./645-p3-guest-barrier.md) records that controller +milestone. The subsequent [HTTP hook implementation](./645-p3-lifecycle-hooks.md) +now supplies production checkpoint and refresh/reconciliation callbacks. No image +capability, new IAM grants or automatic sleep have been enabled. **Credential implementation added locally (2026-09-14):** the [credential verification](./645-p3-credentials.md) records actual pinned-CLI expiry/failure probes, the MicroVM-only scoped loopback provider, retained ambient/tenant refresh and SDK/broker teardown. Static/unknown runtime providers fail closed; their real renewal path must be verified before enabling suspension. -This refresh function is not yet wired to a production HTTP resume callback. - -1. **Implemented:** a per-task context holds task/VM identity, active approval and the original deadline. MicroVM background startup registers it; pipeline exit/crash unregisters it. The SDK hook removes the approval safe point before changing task state or returning a permission result. **Still required:** supply this context to the HTTP lifecycle handlers and finish duplicate-hook admission/acknowledgment behavior. -2. `/suspend` validates that the task is still parked on the intended gate. Wait for lifecycle/progress work already in progress to finish, establish an acknowledged durability barrier, and return success only within the hook budget. Existing best-effort event methods cannot establish that barrier. On timeout/write failure, report a hook failure so the coordinator keeps/reconciles the running VM. Test approval or cancellation arriving during this boundary. -3. `/resume` refreshes ambient/runtime credential providers as necessary, then ensures tenant-scoped assumed credentials are usable **with the same task/user/repo tags**. **Implemented/tested locally:** `refresh_microvm_credentials` forces mandatory ambient renewal before renewing the same tenant credential object. Existing DynamoDB/S3/platform clients retain their references; session identity cannot change after construction. The Claude child uses only the scoped container provider and its managed export helper returns no cached keys. Actual pinned CLI probes establish first-request renewal and failure without fallback; a `credential_process` fallback control demonstrates why provider isolation matters. **Remaining:** wire refresh behind the HTTP resume barrier; verify actual runtime provider, long sleep, Gateway signing and managed image behavior in AWS. Never use test-only `reset_session_cache()` to simulate renewal. -4. Keep the coding action blocked behind the resume barrier until refresh and gate reconciliation finish. Handle duplicate hook calls and concurrent lifecycle requests without deadlocks. Expired credentials or a slow AWS call must not hold the hook beyond its service budget. -5. **Implemented locally:** `/run` and successful controller resume reseed the application PRNG from fresh OS entropy. The future `/resume` HTTP route must use this controller. Do not seed it with task IDs, timestamps or an image-fixed value. Continue using cryptographic randomness for secrets. Test resumed/sibling snapshot uniqueness where meaningful; do not claim that `random` becomes cryptographically safe. -6. **Polling implemented locally:** `_ApprovalDeadline` captures the original recorded UTC expiry and a monotonic cap before database writes. Remaining time is `min(monotonic_deadline - monotonic_now, created_at + timeout_s - wall_now)`, clamped at zero, including each sleep bound. **Registered locally:** the lifecycle context retains this exact object and releases the existing approval loop after successful controller resume. **Still required:** the production resume callback must reconcile the gate and check the original deadline. Never create a fresh timeout on resume. +The production HTTP resume callback now invokes this refresh before any AWS reads. + +1. **Implemented locally:** a per-task context holds task/VM identity, active approval and the original deadline. MicroVM background startup registers it; pipeline exit/crash unregisters it. The SDK hook removes the approval safe point before changing task state or returning a permission result. HTTP handlers use this context, cache completed acknowledgments and reject conflicting transitions. A new gate clears the previous wake result. +2. **Implemented locally:** `/suspend` drains tracked activity, strongly reads the task/gate and atomically checks current coordinator intent plus the original PENDING approval while writing an acknowledged TaskEvents checkpoint. Failed/uncertain writes never acknowledge suspension. Real DynamoDB Local transactions cover approval, cancellation, deadline and intent changes between reads and commit. Existing best-effort event methods are not used for this barrier. +3. **Implemented/tested locally:** `/resume` invokes `refresh_microvm_credentials` behind the closed barrier, forcing ambient renewal before renewing the same tenant credential object **with the same task/user/repo tags**. Existing DynamoDB/S3/platform clients retain their references; session identity cannot change after construction. The Claude child uses only the scoped container provider and its managed export helper returns no cached keys. Actual pinned CLI probes establish first-request renewal and failure without fallback. **Remaining:** verify actual runtime provider, long sleep, Gateway signing and managed image behavior in AWS. Never use test-only `reset_session_cache()` to simulate renewal. +4. **Implemented locally:** coding remains blocked until refresh and atomic task/gate reconciliation finish. Completed duplicates acknowledge cached results; concurrent requests receive 409. The 20-second handler budget includes body reads and all lifecycle work. A timed-out or terminated callback cannot release work through a late thread completion. +5. **Implemented locally:** `/run` and successful HTTP/controller resume reseed the application PRNG from fresh OS entropy. Do not seed it with task IDs, timestamps or an image-fixed value. Continue using cryptographic randomness for secrets. Test resumed/sibling snapshot uniqueness where meaningful; do not claim that `random` becomes cryptographically safe. +6. **Implemented locally:** `_ApprovalDeadline` captures the original recorded UTC expiry and a monotonic cap before database writes. Remaining time is `min(monotonic_deadline - monotonic_now, created_at + timeout_s - wall_now)`, clamped at zero, including each sleep bound. Resume verifies the original recorded creation time/timeout and coordinator deadline, then releases the same approval loop with this exact object. Expired wake enters the existing timeout/late-decision path; it never creates a fresh window. 7. **Preserved/tested locally:** conditional TIMED_OUT write, strongly consistent reread when that write loses, and late-decision winner behavior. Forward/backward clocks, frozen monotonic time, slow writes/reads, missing rows and cancellation have regression coverage. **Still required:** exercise these through the actual resume barrier and live AWS lifecycle. TTL is asynchronous garbage collection, not a precise alarm clock. -8. Declare `/suspend` and `/resume` as enabled image hooks only when the same source version serves them. Add shared hook-budget constants and route/contract assertions. Keep `/ready` and `/validate` AWS-silent. An old image without the new hooks must not be eligible for automatic suspension. Bind capability to the image/version that **actually launched each VM**, using coordinator-owned launch metadata; current deployment settings alone cannot prove an older VM supports the hooks. Unknown/legacy capability keeps new suspends off. A verified full drain can establish a clean boundary, but must not be assumed. +8. **Shared budgets and served-route checks implemented locally; image declaration still pending.** Declare `/suspend` and `/resume` as enabled image hooks only when the same source version serves them, using the shared 30-second service timeout. `/ready` and `/validate` remain AWS-silent. An old image without the new hooks must not be eligible for automatic suspension. Bind capability to the image/version that **actually launched each VM**, using coordinator-owned launch metadata; current deployment settings alone cannot prove an older VM supports the hooks. Unknown/legacy capability keeps new suspends off. A verified full drain can establish a clean boundary, but must not be assumed. ## 6. Wire the supervisor and human decisions diff --git a/docs/verification/645-p3-lifecycle-hooks.md b/docs/verification/645-p3-lifecycle-hooks.md new file mode 100644 index 000000000..4797bcc2d --- /dev/null +++ b/docs/verification/645-p3-lifecycle-hooks.md @@ -0,0 +1,157 @@ +# #645 P3: worker suspend/resume hooks + +Date: 2026-09-14. Local implementation and DynamoDB Local verification. +Nothing in this milestone was deployed. Managed image **2.0** and automatic +suspension remain unchanged. P3 still requires image capability, supervisor +integration and the live acceptance matrix in the [plan](./645-p3-implementation-plan.md). + +## Behavior + +Think of a checkpoint as a signed-off bookmark: “this worker stopped here, waiting +for this answer.” It is saved outside the worker so the supervisor can inspect it. +It does not prove that AWS put the worker to sleep. + +`server.py` serves POST `/suspend` and `/resume` under +`/aws/lambda-microvms/runtime/v1`. They use the sole context registered by `/run`. +An absent/empty body, `{}`, or empty `microvmId` uses that context; a supplied +nonempty ID must match. Actual AWS suspend/resume body behavior still needs live +verification. This tolerance comes from observed empty-ID terminate requests, +not a claim about a verified sleep/wake service payload. + +The handler shares a **20-second total limit** across reading the body, draining +activity and AWS work. Bodies over 4,096 bytes are rejected. The shared contract +reserves a **30-second service timeout** for the later image hook declaration. +Invalid bodies return 400/413, unavailable or conflicting local state returns +409, and failed/uncertain work returns 503. Error responses contain a code and, +for unexpected failures, the exception type; they never echo AWS exception text. + +## Before sleep + +1. The controller requires the original approval park and exactly its blocked + tool. Parallel or unaccounted background work prevents admission. +2. It blocks new activity and waits for already-running approval reads, + credential requests, progress writes and heartbeat calls to drain. + Any previously lost progress acknowledgment keeps suspension disabled. +3. The checkpoint callback strongly reads the task and approval. It verifies + task/user/repository/VM/gate identity, the original creation time and timeout, + task status `AWAITING_APPROVAL`, a matching coordinator `suspend` intent and + approval status `PENDING`. The original deadline must still have time left. +4. One DynamoDB transaction checks both records again and writes a TaskEvents + `agent_milestone` with `milestone: microvm_suspend_checkpoint`. Metadata records + the VM, request, intent generation and original deadline. A change between the + reads and transaction aborts the whole save. +5. Only an acknowledged transaction can produce HTTP 200. The controller checks + the deadline and acknowledgment latch again before returning success. + +The strict `_ProgressWriter.write_microvm_checkpoint` path raises on missing +storage, disabled progress or failed/uncertain writes. Ordinary progress remains +best effort. Existing task-scoped IAM permissions already permit the event Put +and task/approval ConditionChecks; this milestone adds no grants or task metadata +writes. Real deployed permission validation remains a P2/P3 acceptance gate. + +A lost transaction reply can leave a bookmark in DynamoDB while the HTTP hook +fails. That bookmark alone never authorizes the supervisor to assume suspension. +There is no transaction token reused across separate HTTP attempts: DynamoDB's +cached success must not bypass a later changed approval or cancellation. + +## After wake + +The coding barrier stays closed while the callback forces renewal of the retained +ambient providers, then the same tenant credential object with the original +task/user/repository tags. Only then does it read AWS state. The +[credential milestone](./645-p3-credentials.md) covers the Claude subprocess +provider and actual pinned-CLI renewal/failure probes. + +The callback requires the current task and original gate plus a matching +coordinator `resume` intent. A transaction containing only two ConditionChecks +rechecks task and approval together. It changes neither record. A concurrent +valid approval/denial is allowed; cancellation or a changed identity, intent or +original deadline fails reconciliation. + +Successful resume releases the **same** approval loop with the **same** deadline +object. Waking grants no extra time. If the window expired, that loop applies its +existing conditional timeout and late-decision rules, including honoring a timely +decision already saved. Rejecting every expired wake would strand those decisions. + +## Retries and teardown + +- A completed duplicate suspend acknowledges the same parked checkpoint while it + remains safe; it does not write a second event. A completed duplicate resume + returns its cached result without rerunning credential refresh on active coding. +- Concurrent lifecycle requests receive 409 while the first owns the transition. + Each new approval gate clears the old wake acknowledgment. One gate can sleep + only once after a successful wake. +- After admission, failed suspend leaves the unfrozen approval waiter able to + continue and disables another suspend. Failed/timed-out resume closes the + barrier permanently. A rejected body or mismatched VM does not begin a transition. +- A thread can finish a network call after its handler times out. The controller's + generation check prevents that late completion from releasing work. `/terminate` + closes the controller before processing its body and retains its best-effort + 200 cleanup response. Its optional body read now has a one-second limit inside + the existing 15-second service timeout; a stalled stream cannot hold teardown + indefinitely. A regression reproduced the unbounded wait before this fix. +- There is no timer that opens an acknowledged suspend barrier. The supervisor + must repair or terminate an ambiguous lifecycle transition within a bound. + +Build hooks remain AWS-silent. `/validate` checks all six served routes. Direct +FastAPI route registration keeps this check valid with the installed FastAPI +version, whose included routers are lazy objects without a top-level `path`. +Image declaration remains a separate rollout gate. + +## Verification + +The focused Python tests exercise real HTTP dispatch and controller state, +duplicate/concurrent requests, slow bodies, timed-out threads, failed refresh, +termination during refresh, identity/deadline mismatch and expired wake. + +Opt-in DynamoDB Local cases execute the actual condition expressions and +transactions. They inject cancellation, approval, deadline changes and replacement +intent after strong reads. Additional cases connect HTTP → controller → production +callbacks → local database for approval, expiry, cancellation and changed intent. +They assert that wake leaves task and approval records untouched and duplicate +suspend creates only one checkpoint. + +Run from `agent/`, with your own DynamoDB Local container listening on a loopback +port: + +```bash +ABCA_DDB_LOCAL_ENDPOINT=http://127.0.0.1:8000 \ + .venv/bin/pytest tests/test_microvm_checkpoint.py tests/test_microvm_http.py \ + tests/test_microvm_lifecycle.py tests/test_server.py tests/test_hooks.py \ + tests/test_progress_writer.py -q --no-cov +``` + +The fixture accepts only `http://127.0.0.1`, supplies synthetic credentials and +creates/deletes uniquely named temporary tables. Without the opt-in endpoint, +local database cases skip. This verifies DynamoDB expression semantics, not AWS +IAM enforcement, service hook ordering, snapshot clocks or actual frozen threads. + +The CLI already renders arbitrary `agent_milestone` metadata through its generic +milestone path; the new checkpoint needs no CLI configuration changes. + +The initial full agent quality run passed **1,946 tests**, including all **11** +DynamoDB Local cases. After the additional teardown-timeout regression, the final +run passed **1,936 tests** with those 11 opt-in cases skipped because the database +container had been removed. There are **66 new tests** in total; total coverage +including branches is **86.28%**. Ruff lint/format, type checking, Vulture and the +configured Bandit high-severity gate pass. Local tables were verified empty and +the dedicated container was stopped. Evidence is retained privately as +`p3-http-agent-quality-20260914.log` and `p3-http-final-agent-quality-20260914.log`. + +The full monorepo build passed in about 13 minutes: **4,801 CDK tests** (38 +existing optional skips), **928 CLI tests**, **11 Forge tests**, infrastructure +compile/lint/synthesis, documentation build/links and contract drift checks. +The final targeted constants suite passed **37 tests** after making its negative +budget cases independent of the configured service timeout. Its lint check passed. +Evidence: `p3-http-build-20260914.log`, `p3-http-constants-final-20260914.log` and +`p3-http-final-eslint-20260914.log`. The final Python run above includes the +teardown regression added during the build. + +## Next integration + +Bind capability to the actual image/version used by each worker and declare the +matching hooks, initially with automatic sleep disabled. Wire persistent intent +and policy into supervisor polling with bounded retries/recovery, and request +wake after an approval/denial transaction commits. Then deploy and verify real +hook bodies, runtime credential renewal, managed Claude settings, Gateway signing, +long sleep, delayed transitions, cancellation, expiry and rollback. diff --git a/docs/verification/645-p3-readiness-review.md b/docs/verification/645-p3-readiness-review.md index 0315b123f..1b8fa10d5 100644 --- a/docs/verification/645-p3-readiness-review.md +++ b/docs/verification/645-p3-readiness-review.md @@ -19,8 +19,10 @@ now keeps coding behind a controlled door while approval work is paused. The [credential implementation](./645-p3-credentials.md) renews the existing AWS key objects and gives Claude one source of task-specific keys. Actual pinned Claude tests with fake AWS responses prove renewal before the next request and safe -failure without borrowing the parent's keys. Production HTTP pause/wake handlers, -supervisor wiring and real AWS sleep/wake verification remain open. +failure without borrowing the parent's keys. The subsequent +[HTTP hook milestone](./645-p3-lifecycle-hooks.md) connects pause to an atomic +checkpoint and wake to credential renewal plus task/gate reconciliation. +Image capability, supervisor wiring and real AWS sleep/wake verification remain open. The reservation review found a separate trust boundary: the old agent role could write/replace/delete its task row, including coordinator metadata. A subsequent local fix restricts main-task writes to reporting/approval attributes, removes replacement/deletion permissions and removes unused worker counter grants. It also removes unused Python submission/session-info helpers and corrects overstated tenant-isolation comments. [Metadata verification](./645-coordinator-metadata.md) records the tests and pending AWS gate. Status reports still come from the agent, and the compute role chooses session tags; this is not complete hostile-worker isolation. @@ -42,7 +44,7 @@ An **IAM role** is a permission badge. A **trust policy** says who may wear that |---|---|---| | P1 | Build the computer, start it, deliver a task, check it and stop it | Merged in [#689](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/689). Strategy, infrastructure, bootstrap permissions, packaging, types, `/ready` and `/run` exist. | | P2 | Make a real coding task work with configuration, permissions, logs and progress | Merged in [#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733). `/validate`, `/terminate`, warm-up, runtime grants and heartbeat support exist. Two recorded tasks completed and opened PRs on August 7, using a manual IAM workaround. Permanent fixes still need a clean rerun. | -| P3 | Sleep during a human approval wait, wake correctly, and keep deadlines/credentials safe | Local foundations now include intent/policy, the guest pause controller and scoped credential renewal. The integrated HTTP/supervisor lifecycle and live acceptance remain open. | +| P3 | Sleep during a human approval wait, wake correctly, and keep deadlines/credentials safe | Local foundations include intent/policy, guest pause control, scoped credential renewal and production HTTP checkpoint/wake hooks. Image capability, supervisor integration and live acceptance remain open. | | P4 | — | ADR-021 defines no P4. Verification runbooks have their own numbered phases; those are not extra ADR milestones. | The old unchecked checklist and the word “proposed” do not erase the merged work. Conversely, merged code is not proof that the final deployment path works unattended. diff --git a/scripts/check-constants-sync.ts b/scripts/check-constants-sync.ts index 1fd1f3a3f..fe53707c3 100644 --- a/scripts/check-constants-sync.ts +++ b/scripts/check-constants-sync.ts @@ -50,8 +50,11 @@ const JIRA_REACTIONS_PY = path.join(REPO_ROOT, 'agent/src/jira_reactions.py'); const SERVER_PY = path.join(REPO_ROOT, 'agent/src/server.py'); const CONFIG_PY = path.join(REPO_ROOT, 'agent/src/config.py'); const PAYLOAD_BOOTSTRAP_PY = path.join(REPO_ROOT, 'agent/src/payload_bootstrap.py'); +const MICROVM_HTTP_PY = path.join(REPO_ROOT, 'agent/src/microvm_http.py'); const PAYLOAD_BOOTSTRAP_TS = path.join(REPO_ROOT, 'cdk/src/handlers/shared/payload-bootstrap.ts'); -const PYTHON_CONSUMERS = [POLICY_PY, JIRA_REACTIONS_PY, SERVER_PY, CONFIG_PY, PAYLOAD_BOOTSTRAP_PY]; +const PYTHON_CONSUMERS = [ + POLICY_PY, JIRA_REACTIONS_PY, SERVER_PY, CONFIG_PY, PAYLOAD_BOOTSTRAP_PY, MICROVM_HTTP_PY, +]; const MICROVM_COMPUTE_TS = path.join(REPO_ROOT, 'cdk/src/constructs/lambda-microvm-compute.ts'); const TS_CONSUMERS = [MICROVM_COMPUTE_TS]; @@ -120,6 +123,14 @@ const OWNED_PYTHON_PATTERNS: ReadonlyArray<{ name: string; regex: RegExp }> = [ name: '_READY_WARMUP_REQUIRED_TIMEOUT_SECONDS', regex: /^\s*_READY_WARMUP_REQUIRED_TIMEOUT_SECONDS\s*(?::\s*(?:int|float))?\s*=\s*-?\d+\b/m, }, + { + name: 'LIFECYCLE_HANDLER_BUDGET_S', + regex: /^\s*LIFECYCLE_HANDLER_BUDGET_S\s*(?::\s*(?:int|float))?\s*=\s*-?\d+\b/m, + }, + { + name: 'LIFECYCLE_HOOK_TIMEOUT_S', + regex: /^\s*LIFECYCLE_HOOK_TIMEOUT_S\s*(?::\s*(?:int|float))?\s*=\s*-?\d+\b/m, + }, ]; /** @@ -186,6 +197,8 @@ function main(): number { ready_hook_timeout_seconds: number; warmup_total_budget_seconds: number; warmup_required_timeout_seconds: number; + lifecycle_hook_timeout_seconds: number; + lifecycle_handler_budget_seconds: number; }; payload_bootstrap?: { version: number; @@ -349,6 +362,8 @@ function main(): number { 'ready_hook_timeout_seconds', 'warmup_total_budget_seconds', 'warmup_required_timeout_seconds', + 'lifecycle_hook_timeout_seconds', + 'lifecycle_handler_budget_seconds', ] as const; if (!mhb || BUDGET_FIELDS.some(field => !Number.isInteger(mhb[field]))) { console.error( @@ -373,6 +388,12 @@ function main(): number { 'best-effort ones something to share)', ); } + if (mhb.lifecycle_handler_budget_seconds >= mhb.lifecycle_hook_timeout_seconds) { + invariantErrors.push( + 'microvm_hook_budgets.lifecycle_handler_budget_seconds must be < ' + + 'lifecycle_hook_timeout_seconds (pause/wake must leave time to answer)', + ); + } const bootstrap = json.payload_bootstrap; const bootstrapNumbers = [ From de926a5ead1fed077849d7dfdf5a8aeedb4f5bbb Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Tue, 15 Sep 2026 10:26:40 -0400 Subject: [PATCH 036/149] feat(microvm): verify lifecycle support for each launched image --- agent/README.md | 4 +- agent/src/server.py | 6 ++ agent/tests/test_server.py | 23 ++++ cdk/scripts/package-microvm-artifact.sh | 20 ++-- cdk/src/constructs/lambda-microvm-compute.ts | 45 +++++--- cdk/src/constructs/task-orchestrator.ts | 12 +-- cdk/src/handlers/shared/compute-strategy.ts | 12 +-- .../shared/microvm-image-capability.ts | 82 ++++++++++++++ .../shared/microvm-lifecycle-policy.ts | 5 +- cdk/src/handlers/shared/microvm-lifecycle.ts | 19 +++- cdk/src/handlers/shared/microvm-start.ts | 30 +++++- cdk/src/handlers/shared/orchestrator.ts | 3 +- .../strategies/lambda-microvm-strategy.ts | 48 +++++++-- .../constructs/lambda-microvm-compute.test.ts | 71 +++++++------ cdk/test/constructs/task-orchestrator.test.ts | 3 +- .../handlers/orchestrate-task-microvm.test.ts | 6 +- .../shared/microvm-image-capability.test.ts | 95 +++++++++++++++++ .../shared/microvm-lifecycle-local.test.ts | 100 +++++++++++++++++- .../handlers/shared/microvm-lifecycle.test.ts | 15 ++- cdk/test/handlers/shared/orchestrator.test.ts | 13 ++- .../lambda-microvm-strategy.test.ts | 90 +++++++++++++++- .../start-session-composition.test.ts | 24 +++-- cdk/test/scripts/check-constants-sync.test.ts | 29 +++++ cdk/test/stacks/agent.test.ts | 5 +- contracts/constants.json | 5 + contracts/constants.md | 28 +++-- ...ADR-021-lambda-microvms-compute-backend.md | 12 +-- ...Adr-021-lambda-microvms-compute-backend.md | 12 +-- docs/verification/645-lifecycle-intent.md | 2 +- docs/verification/645-p3-credentials.md | 2 + docs/verification/645-p3-guest-barrier.md | 2 + docs/verification/645-p3-image-capability.md | 92 ++++++++++++++++ .../645-p3-implementation-plan.md | 19 +++- docs/verification/645-p3-lifecycle-hooks.md | 2 + docs/verification/645-p3-readiness-review.md | 4 +- scripts/check-constants-sync.ts | 26 ++++- 36 files changed, 834 insertions(+), 132 deletions(-) create mode 100644 cdk/src/handlers/shared/microvm-image-capability.ts create mode 100644 cdk/test/handlers/shared/microvm-image-capability.test.ts create mode 100644 docs/verification/645-p3-image-capability.md diff --git a/agent/README.md b/agent/README.md index 185e3b955..bc214705e 100644 --- a/agent/README.md +++ b/agent/README.md @@ -249,7 +249,7 @@ It runs under the **build role**, which has no Bedrock, Secrets Manager or Dynam Baked secrets are **reported, not enforced**: `warnings` lists the names (never values) of any credential-shaped env var present in the snapshot, because the build environment's own credentials may legitimately be in that env and failing here would fail every build. -**`POST /aws/lambda-microvms/runtime/v1/terminate`** — Runtime hook (P2). Best-effort: emits one final structured log line and returns 200 — always, inside the hook budget, even with nothing running, and for **any body**: malformed JSON, a wrong content-type, an empty body or no body at all. That is why the handler takes the raw request instead of a typed body model — FastAPI validates a typed body *before* the handler runs, so a truncated body would answer 422 and report a hook failure for a teardown that actually succeeded. It does **not** join the pipeline thread (that is `lifespan`'s job on graceful shutdown) and it **never writes terminal task status**: the orchestrator finalizes the task and *then* calls `TerminateMicrovm`, so a status write here would race that finalization. `ProgressWriter` writes each event synchronously, but catches and drops failures; returning from an event method is not a durability guarantee. P3 needs an acknowledged barrier for `/suspend`, not merely a call to this best-effort writer. +**`POST /aws/lambda-microvms/runtime/v1/terminate`** — Runtime hook (P2). Best-effort: emits one final structured log line and returns 200 — always, inside the hook budget, even with nothing running, and for **any body**: malformed JSON, a wrong content-type, an empty body or no body at all. That is why the handler takes the raw request instead of a typed body model — FastAPI validates a typed body *before* the handler runs, so a truncated body would answer 422 and report a hook failure for a teardown that actually succeeded. It does **not** join the pipeline thread (that is `lifespan`'s job on graceful shutdown) and it **never writes terminal task status**: the orchestrator finalizes the task and *then* calls `TerminateMicrovm`, so a status write here would race that finalization. `ProgressWriter` writes each event synchronously, but catches and drops failures; returning from an event method is not a durability guarantee. The P3 `/suspend` hook uses its own atomic acknowledged checkpoint after draining tracked activity. `microvmId` is parsed defensively and **arrives empty in practice**: the service sends `""` here, unlike `/run` where it is populated (live-verified, ADR-021 P2-F8). So an empty id is expected-normal, not a degraded read — and this hook therefore **cannot** join the guest's record to the control-plane one. `/run`'s `hook accepted task_id=… microvm_id=…` line carries that correlation; `/terminate`'s value is the pipeline-state snapshot it reports. @@ -303,7 +303,7 @@ Values are **non-secret identifiers only** — secrets are still fetched at `/ru Rejections are structured so they are readable in the MicroVM log group: `400 MICROVM_RUN_PAYLOAD_INVALID` (unusable envelope — retrying the same body cannot help), `500 MICROVM_RUN_PAYLOAD_UNREADABLE` (manifest/payload read or stored bytes failed), `400 MICROVM_RUN_PLATFORM_CONFIG_INVALID` (key off the allowlist, non-object block, or non-string value — fix the producer), `400 MICROVM_RUN_PLATFORM_CONFIG_INCOMPLETE` (a required key missing or blank — fix the deployment wiring), `400 TASK_RECORD_INCOMPLETE` (same validator and vocabulary as `/invocations`). -`/suspend` and `/resume` are deliberately **not** served — declaring a hook nothing answers fails the corresponding lifecycle transition, so the CDK construct declares exactly the hooks the agent serves. They land in P3 with the ComputeStrategy interface widening. +`/suspend` and `/resume` are served by `microvm_http.py` and declared in managed images with 30-second service timeouts and a shared non-secret protocol marker. Pause requires an active original approval gate, matching coordinator intent, drained activity and an acknowledged checkpoint; wake renews credentials and atomically rechecks the original task/gate before allowing coding. Each handler has a 20-second total budget. `/validate` rejects a supplied incompatible image marker without contacting AWS. The coordinator checks the actual launched image version and persists support on that worker; missing support disables new suspension. See [image capability verification](../docs/verification/645-p3-image-capability.md). Supervisor integration and live P3 acceptance remain open. ### Testing Server Mode Locally diff --git a/agent/src/server.py b/agent/src/server.py index 95cf5dbf1..65da4920d 100644 --- a/agent/src/server.py +++ b/agent/src/server.py @@ -1734,6 +1734,12 @@ def microvm_validate(): "hook_routes_registered": not missing_routes, "python_version_supported": sys.version_info[:2] >= _MIN_PYTHON_VERSION, "platform_config_contract_loaded": bool(MICROVM_PLATFORM_CONFIG_ENV_BY_KEY), + # Absent markers support ordinary/legacy images. A marked image must + # actually contain the protocol it advertises, before AWS snapshots it. + "image_lifecycle_protocol_supported": os.environ.get( + SHARED_CONSTANTS["microvm_lifecycle"]["image_protocol_env"] + ) + in (None, str(SHARED_CONSTANTS["microvm_lifecycle"]["protocol_version"])), } failed = sorted(name for name, ok in checks.items() if not ok) diff --git a/agent/tests/test_server.py b/agent/tests/test_server.py index a0e98d3bd..ceda0147d 100644 --- a/agent/tests/test_server.py +++ b/agent/tests/test_server.py @@ -2458,10 +2458,33 @@ def test_returns_200_with_the_individual_checks(self, client): "hook_routes_registered": True, "python_version_supported": True, "platform_config_contract_loaded": True, + "image_lifecycle_protocol_supported": True, } assert body["hook_prefix"] == server.MICROVM_HOOK_PREFIX assert body["platform_config_keys"] == len(server.MICROVM_PLATFORM_CONFIG_ENV_BY_KEY) + @pytest.mark.parametrize( + "marker", + [None, str(server.SHARED_CONSTANTS["microvm_lifecycle"]["protocol_version"]), "", "999"], + ) + def test_build_rejects_an_image_marker_the_source_does_not_support( + self, client, monkeypatch, marker + ): + contract = server.SHARED_CONSTANTS["microvm_lifecycle"] + key = contract["image_protocol_env"] + if marker is None: + monkeypatch.delenv(key, raising=False) + else: + monkeypatch.setenv(key, marker) + forbidden = MagicMock(side_effect=AssertionError("image validation contacted AWS")) + monkeypatch.setattr("aws_session.platform_client", forbidden) + monkeypatch.setattr("aws_session.tenant_client", forbidden) + response = client.post(VALIDATE_HOOK) + expected = marker in (None, str(contract["protocol_version"])) + assert response.status_code == (200 if expected else 503) + assert response.json()["checks"]["image_lifecycle_protocol_supported"] is expected + forbidden.assert_not_called() + def test_makes_zero_aws_calls_even_with_a_log_group_configured( self, client, monkeypatch, capfd ): diff --git a/cdk/scripts/package-microvm-artifact.sh b/cdk/scripts/package-microvm-artifact.sh index 6f40a26d8..48b89486c 100755 --- a/cdk/scripts/package-microvm-artifact.sh +++ b/cdk/scripts/package-microvm-artifact.sh @@ -104,9 +104,9 @@ # breadcrumb that must not write terminal task # status — the orchestrator finalizes the task # and THEN calls TerminateMicrovm. -# /suspend, /resume P3. A hook the service calls but nothing -# answers fails its lifecycle transition, so -# each is enabled only once it is served. +# /suspend, /resume declared AND served in P3, with the image +# protocol marker. The coordinator verifies the +# actual launched version before allowing sleep. # # The clean P2 deployment used bootstrap policy bundle 1.7.0. The Dockerfile is # copied unmodified; image build success does not establish full P2 acceptance. @@ -266,7 +266,7 @@ REMINDER (ADR-021 P2): clean deployment and coding, iteration and cancellation Ready/validate hooks, heartbeat, logging, Memory writes and cleanup have live evidence. Full P2 acceptance still needs the failure/recovery, effective IAM and networking matrix in docs/verification/645-p3-implementation-plan.md. - /suspend and /resume remain undeclared until compatible P3 agent hooks land. + /suspend and /resume are declared; supervisor integration and live P3 acceptance remain open. CDK retains warning ID abca:microvm-image-p1-smoke-unverified for compatibility; its text describes the current verification gaps. EOF @@ -417,12 +417,11 @@ echo "==> Creating MicroVM image '${IMAGE_NAME}' (${MEMORY_MIB} MiB baseline)" # * `/ready` is MANDATORY whenever any lifecycle hook is enabled: # "The ready (/ready) MicroVM image hook must be enabled when any MicroVM # lifecycle hook (run, resume, suspend, or terminate) is enabled." -# * all four hooks the agent serves are enabled: `/ready` + `/validate` (build) -# and `/run` + `/terminate` (runtime). `/suspend` and `/resume` stay DISABLED -# until P3 implements them — a hook the service calls but nothing answers -# fails the corresponding build or lifecycle transition. +# * all six served hooks are enabled: `/ready` + `/validate` (build), `/run`, +# `/terminate`, `/suspend` and `/resume` (runtime). The non-secret lifecycle +# protocol marker describes the source's checkpoint/credential/gate protocol. # * the timeouts mirror the construct's constants -# (`RUN_/READY_/VALIDATE_/TERMINATE_HOOK_TIMEOUT_SECONDS` in +# (`RUN_/READY_/VALIDATE_/TERMINATE_/LIFECYCLE_HOOK_TIMEOUT_SECONDS` in # `cdk/src/constructs/lambda-microvm-compute.ts`), which carry the rationale # for each value. A bash helper cannot import them, and "keep the two in step" # as prose already FAILED once — `readyTimeoutInSeconds` stayed at 60 here when @@ -449,7 +448,8 @@ CREATE_RESPONSE="$(aws lambda-microvms create-microvm-image \ --resources "[{\"minimumMemoryInMiB\":${MEMORY_MIB}}]" \ --egress-network-connectors "${BUILD_EGRESS_CONNECTORS}" \ --logging "{\"cloudWatch\":{\"logGroup\":\"${LOG_GROUP}\"}}" \ - --hooks '{"port":8080,"microvmHooks":{"run":"ENABLED","runTimeoutInSeconds":60,"terminate":"ENABLED","terminateTimeoutInSeconds":15},"microvmImageHooks":{"ready":"ENABLED","readyTimeoutInSeconds":300,"validate":"ENABLED","validateTimeoutInSeconds":60}}' \ + --hooks '{"port":8080,"microvmHooks":{"run":"ENABLED","runTimeoutInSeconds":60,"terminate":"ENABLED","terminateTimeoutInSeconds":15,"suspend":"ENABLED","suspendTimeoutInSeconds":30,"resume":"ENABLED","resumeTimeoutInSeconds":30},"microvmImageHooks":{"ready":"ENABLED","readyTimeoutInSeconds":300,"validate":"ENABLED","validateTimeoutInSeconds":60}}' \ + --environment-variables '{"ABCA_MICROVM_LIFECYCLE_PROTOCOL":"1"}' \ --tags "abca:compute-backend=lambda-microvm" \ --output json)" diff --git a/cdk/src/constructs/lambda-microvm-compute.ts b/cdk/src/constructs/lambda-microvm-compute.ts index 7a7c3b2b5..a9c635769 100644 --- a/cdk/src/constructs/lambda-microvm-compute.ts +++ b/cdk/src/constructs/lambda-microvm-compute.ts @@ -86,7 +86,8 @@ export const MICROVM_ARTIFACT_OBJECT_KEY = 'microvm-images/agent-artifact.zip'; * (`agent/Dockerfile` → `EXPOSE 8080`), and therefore the port the MicroVM * lifecycle-hook listener is configured for. */ -const AGENT_HOOK_PORT = 8080; +const AGENT_HOOK_PORT = sharedConstants.microvm_lifecycle.hook_port; +const LIFECYCLE_HOOK_TIMEOUT_SECONDS = sharedConstants.microvm_hook_budgets.lifecycle_hook_timeout_seconds; /** * Value every hook field on `AWS::Lambda::MicrovmImage` takes to turn a hook ON. @@ -109,8 +110,8 @@ const AGENT_HOOK_PORT = 8080; * {@link MICROVM_AGENT_HOOK_ROUTES} records and the live build/run logs confirm. * * `DISABLED` is never emitted: hooks outside the image's enabled capability are - * omitted. The guest now serves `/suspend` + `/resume`, but their image - * declaration remains gated on the P3 capability rollout and live verification. + * omitted. All six guest hooks are now declared for managed images. The separate + * supervisor enable flag still controls whether new automatic suspends are allowed. */ const HOOK_ENABLED = 'ENABLED'; @@ -149,6 +150,8 @@ export const MICROVM_AGENT_HOOK_ROUTES = { validate: `${MICROVM_HOOK_ROUTE_PREFIX}/validate`, run: `${MICROVM_HOOK_ROUTE_PREFIX}/run`, terminate: `${MICROVM_HOOK_ROUTE_PREFIX}/terminate`, + suspend: `${MICROVM_HOOK_ROUTE_PREFIX}/suspend`, + resume: `${MICROVM_HOOK_ROUTE_PREFIX}/resume`, } as const; /** @@ -635,7 +638,7 @@ export interface LambdaMicrovmComputeProps extends LambdaMicrovmImageInputs { /** * Non-secret environment variables baked into the snapshot at build time. * - * Deliberately empty by default, and expected to STAY empty. ADR-021 + * Empty by default; the construct adds only its invariant lifecycle protocol marker. ADR-021 * sub-decision 3 forbids secrets, tokens, and per-task identity in the snapshot * — and P2 resolved the remaining question (where the agent's non-secret * configuration parity with the ECS container comes from) in favour of the @@ -728,9 +731,10 @@ export interface LambdaMicrovmComputeProps extends LambdaMicrovmImageInputs { * failure/recovery, effective IAM and network matrix remains open; see * docs/verification/645-p3-implementation-plan.md. The stable * `abca:microvm-image-p1-smoke-unverified` warning below records that scope. - * Only `/suspend` and `/resume` remain undeclared, until P3 implements compatible - * agent hooks: a hook the service calls but - * nothing answers fails the corresponding lifecycle transition. + * P3 also declares the served `/suspend` and `/resume` hooks and bakes a non-secret + * protocol marker into the image. The coordinator verifies the actual launched + * version before allowing suspension; supervisor integration and live acceptance + * remain separate rollout gates. * * ## Deliberately NOT here * @@ -1274,6 +1278,10 @@ export class LambdaMicrovmCompute extends Construct { // ad-hoc condition, so this construct and the stack's pre-TaskApi decision // (isLambdaMicrovmImageConfigured) can never disagree. if (props.baseImageArn && props.baseImageVersion) { + const requestedProtocol = props.imageEnvironmentVariables?.[sharedConstants.microvm_lifecycle.image_protocol_env]; + if (requestedProtocol !== undefined && requestedProtocol !== String(sharedConstants.microvm_lifecycle.protocol_version)) { + throw new Error('The managed MicroVM lifecycle protocol marker is owned by the image source'); + } this.image = new lambda.CfnMicrovmImage(this, 'Image', { name: this.imageName, description: `ABCA agent snapshot for ${stack.stackName} (ADR-021 lambda-microvm backend)`, @@ -1295,8 +1303,13 @@ export class LambdaMicrovmCompute extends Construct { logging: { cloudWatch: { logGroup: this.logGroup.logGroupName } }, // No extra OS capabilities: the agent runs ordinary user-space tooling. additionalOsCapabilities: [], - // Nothing baked in — see `imageEnvironmentVariables`. - environmentVariables: Object.entries(props.imageEnvironmentVariables ?? {}) + // The non-secret protocol marker belongs to this immutable image version. + // Task/deployment identity still arrives only through /run. + environmentVariables: Object.entries({ + ...props.imageEnvironmentVariables, + [sharedConstants.microvm_lifecycle.image_protocol_env]: + String(sharedConstants.microvm_lifecycle.protocol_version), + }) .map(([key, value]) => ({ key, value })), hooks: { port: AGENT_HOOK_PORT, @@ -1313,8 +1326,8 @@ export class LambdaMicrovmCompute extends Construct { // hook nothing answers fails the corresponding lifecycle transition, // so each is enabled only once it is served. // - // `/suspend` and `/resume` stay OMITTED (not `DISABLED`) until P3, - // where compatible agent hooks and durability barriers are integrated. + // The served P3 hooks use the shared service budget. Declaring them + // enables service callbacks; automatic suspension remains supervisor-gated. // Note that termination does NOT depend on this hook: // `TerminateMicrovm` removes the VM with or without in-guest // cooperation, which is what makes a best-effort `/terminate` safe to @@ -1323,6 +1336,10 @@ export class LambdaMicrovmCompute extends Construct { runTimeoutInSeconds: RUN_HOOK_TIMEOUT_SECONDS, terminate: HOOK_ENABLED, terminateTimeoutInSeconds: TERMINATE_HOOK_TIMEOUT_SECONDS, + suspend: HOOK_ENABLED, + suspendTimeoutInSeconds: LIFECYCLE_HOOK_TIMEOUT_SECONDS, + resume: HOOK_ENABLED, + resumeTimeoutInSeconds: LIFECYCLE_HOOK_TIMEOUT_SECONDS, }, microvmImageHooks: { // `/ready` is MANDATORY whenever any lifecycle hook is enabled — the @@ -1405,11 +1422,11 @@ export class LambdaMicrovmCompute extends Construct { 'abca:microvm-image-p1-smoke-unverified', 'A MicroVM image is configured. Clean P2 deployment with bootstrap bundle 1.7.0 and ' + 'coding, iteration and cancellation runs passed on 2026-09-14 without manual IAM changes. ' - + 'The agent serves the declared /ready, /validate, /run and /terminate hooks. ' + + 'The agent serves /ready, /validate, /run, /terminate, /suspend and /resume; managed images declare all six. ' + 'Heartbeat, logs, Memory writes and cleanup have live evidence. Full P2 acceptance ' + 'still needs the failure/recovery, effective IAM and networking matrix in ' - + 'docs/verification/645-p3-implementation-plan.md. The /suspend and /resume hooks remain ' - + 'undeclared until compatible P3 agent hooks are integrated. The warning ID is retained ' + + 'docs/verification/645-p3-implementation-plan.md. P3 checks the actual launched image version; ' + + 'supervisor integration and live sleep/wake acceptance remain open. The warning ID is retained ' + 'across phases for existing operator filters.', ); } diff --git a/cdk/src/constructs/task-orchestrator.ts b/cdk/src/constructs/task-orchestrator.ts index a505533b2..eeaff9fc9 100644 --- a/cdk/src/constructs/task-orchestrator.ts +++ b/cdk/src/constructs/task-orchestrator.ts @@ -610,20 +610,19 @@ export class TaskOrchestrator extends Construct { // Lambda MicroVMs compute strategy permissions (only when configured). // - // EXACTLY the four control-plane actions the P1 strategy calls, per + // Control-plane actions used by the strategy, per // ADR-021's "only the MicroVM lifecycle actions it calls" requirement: // RunMicrovm — startSession // GetMicrovm — pollSession + // GetMicrovmImageVersion — attest the actual launched snapshot's lifecycle hooks // TerminateMicrovm — stopSession / finalize (the active cleanup path) // PassNetworkConnector — required to attach egress connectors, even the // AWS-managed ones // // NOT granted, deliberately: // - lambda:SuspendMicrovm / lambda:ResumeMicrovm — the ADR's grant list - // names them, but P1 has no suspend/resume code path. They land with the - // P3 interface widening (mandatory suspendSession/resumeSession across - // all three strategies) together with the approve/deny Lambdas' - // conditional ResumeMicrovm + GetMicrovm. + // names them and strategy methods exist, but no supervisor caller is + // wired yet. Grant them when that integration adds actual calls. // - lambda:CreateMicrovmAuthToken — granted to no role in any phase; no // JWE consumer exists (ADR-021 sub-decision 3). if (props.microvmConfig) { @@ -656,6 +655,7 @@ export class TaskOrchestrator extends Construct { actions: [ 'lambda:RunMicrovm', 'lambda:GetMicrovm', + 'lambda:GetMicrovmImageVersion', 'lambda:TerminateMicrovm', ], resources: microvmImageResources, @@ -786,7 +786,7 @@ export class TaskOrchestrator extends Construct { }, { id: 'AwsSolutions-IAM5', - reason: 'DynamoDB index/* wildcards generated by CDK grantReadWriteData; AgentCore runtime/* required for sub-resource invocation; Secrets Manager wildcards generated by CDK grantRead; AgentCore Memory wildcards generated by CDK grantRead/grantWrite; ECS RunTask/DescribeTasks/StopTask conditioned on cluster ARN; iam:PassRole scoped to ECS task/execution roles and conditioned on ecs-tasks.amazonaws.com; S3 writes restricted to bootstrap manifests and task payload/launch objects; GetObject and DeleteObject restricted to */payload.json and */launch.json for signing, replay and cleanup; ListBucket is scoped to each payload bucket so absent launch records return NoSuchKey; MicroVM lifecycle actions (RunMicrovm/GetMicrovm/TerminateMicrovm) are scoped to the single platform MicroVM image ARN plus a :* version-suffix sibling (every one of them authorizes against the image resource, not the per-session instance; no account-wide wildcard is used); lambda:PassNetworkConnector requires Resource:* because the action supports no resource-level permissions and the AWS-managed connectors live outside this account; iam:PassRole is scoped to the exact MicroVM execution role without iam:PassedToService (ADR-021 P2r2-F10); Agent Registry read scoped to the wired registry ARN, with a record/* suffix wildcard because record ids are server-assigned and unknown at synth (#246)', + reason: 'DynamoDB index/* wildcards generated by CDK grantReadWriteData; AgentCore runtime/* required for sub-resource invocation; Secrets Manager wildcards generated by CDK grantRead; AgentCore Memory wildcards generated by CDK grantRead/grantWrite; ECS RunTask/DescribeTasks/StopTask conditioned on cluster ARN; iam:PassRole scoped to ECS task/execution roles and conditioned on ecs-tasks.amazonaws.com; S3 writes restricted to bootstrap manifests and task payload/launch objects; GetObject and DeleteObject restricted to */payload.json and */launch.json for signing, replay and cleanup; ListBucket is scoped to each payload bucket so absent launch records return NoSuchKey; MicroVM launch/state/cleanup and image-capability actions (RunMicrovm/GetMicrovm/TerminateMicrovm/GetMicrovmImageVersion) are scoped to the single platform MicroVM image ARN plus a :* version-suffix sibling (every one of them authorizes against the image resource, not the per-session instance; no account-wide wildcard is used); lambda:PassNetworkConnector requires Resource:* because the action supports no resource-level permissions and the AWS-managed connectors live outside this account; iam:PassRole is scoped to the exact MicroVM execution role without iam:PassedToService (ADR-021 P2r2-F10); Agent Registry read scoped to the wired registry ARN, with a record/* suffix wildcard because record ids are server-assigned and unknown at synth (#246)', }, ], true); } diff --git a/cdk/src/handlers/shared/compute-strategy.ts b/cdk/src/handlers/shared/compute-strategy.ts index 5b96cc302..25208878d 100644 --- a/cdk/src/handlers/shared/compute-strategy.ts +++ b/cdk/src/handlers/shared/compute-strategy.ts @@ -18,6 +18,7 @@ */ import type { MicrovmState } from '@aws-sdk/client-lambda-microvms'; +import type { MicrovmImageMetadata } from './microvm-image-capability'; import type { BlueprintConfig, ComputeType } from './repo-config'; import { AgentCoreComputeStrategy } from './strategies/agentcore-strategy'; import { EcsComputeStrategy } from './strategies/ecs-strategy'; @@ -37,16 +38,15 @@ import { LambdaMicrovmComputeStrategy } from './strategies/lambda-microvm-strate * ADR-021 sub-decision 1: the MicroVM variant carries ``microvmId`` (every * lifecycle API — suspend/resume/terminate/get — takes only that identifier) * and ``endpoint`` (minted per session by ``RunMicrovm``, required for any - * future orchestrator→agent HTTP interaction). The image ARN/version is - * deliberately NOT in the handle: like the ECS task-definition ARN it is - * deployment-time configuration consumed by ``startSession`` from the - * construct-injected environment and recorded in the session-start log entry - * for diagnostics, not per-session lifecycle state. + * future orchestrator→agent HTTP interaction). P3 additionally retains the actual + * image ARN/version and verified lifecycle protocol. These describe the snapshot + * that launched this worker; current deployment settings cannot substitute for it. + * Legacy handles remain usable for cleanup, with new suspension disabled. */ export type SessionHandle = | { readonly sessionId: string; readonly strategyType: 'agentcore'; readonly runtimeArn: string } | { readonly sessionId: string; readonly strategyType: 'ecs'; readonly clusterArn: string; readonly taskArn: string } - | { readonly sessionId: string; readonly strategyType: 'lambda-microvm'; readonly microvmId: string; readonly endpoint: string }; + | ({ readonly sessionId: string; readonly strategyType: 'lambda-microvm'; readonly microvmId: string; readonly endpoint: string } & MicrovmImageMetadata); /** * Substrate-observed session state. Deliberately mechanical: the strategy diff --git a/cdk/src/handlers/shared/microvm-image-capability.ts b/cdk/src/handlers/shared/microvm-image-capability.ts new file mode 100644 index 000000000..9b338b327 --- /dev/null +++ b/cdk/src/handlers/shared/microvm-image-capability.ts @@ -0,0 +1,82 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +// SPDX-License-Identifier: MIT-0 + +import type { GetMicrovmImageVersionOutput } from '@aws-sdk/client-lambda-microvms'; +import sharedConstants from '../../../../contracts/constants.json'; + +export const MICROVM_LIFECYCLE_PROTOCOL = String(sharedConstants.microvm_lifecycle.protocol_version); +export const MICROVM_IMAGE_PROTOCOL_ENV = sharedConstants.microvm_lifecycle.image_protocol_env; +// Optional discovery/enrichment must not hold up an already registered worker. +export const MICROVM_IMAGE_CAPABILITY_REQUEST_TIMEOUT_MS = 3_000; +const MAX_IMAGE_IDENTITY_LENGTH = 2048; + +/** Coordinator-owned evidence for the image version that actually launched a VM. */ +export interface MicrovmImageMetadata { + readonly imageArn?: string; + readonly imageVersion?: string; + readonly lifecycleProtocol?: string; +} + +function nonblank(value: unknown): value is string { + return typeof value === 'string' && value.length > 0 && value.length <= MAX_IMAGE_IDENTITY_LENGTH + && value === value.trim() && !/[\u0000-\u001f\u007f]/.test(value); +} + +/** Legacy or malformed identity never acquires capability from deployment settings. */ +export function readMicrovmImageMetadata(value: unknown): Record { + if (!value || typeof value !== 'object' || Array.isArray(value)) return {}; + const metadata = value as MicrovmImageMetadata; + if (!nonblank(metadata.imageArn) || !nonblank(metadata.imageVersion)) return {}; + return { + imageArn: metadata.imageArn, + imageVersion: metadata.imageVersion, + ...(metadata.lifecycleProtocol === MICROVM_LIFECYCLE_PROTOCOL + && { lifecycleProtocol: MICROVM_LIFECYCLE_PROTOCOL }), + }; +} + +export function supportsMicrovmLifecycle(value: unknown): boolean { + return readMicrovmImageMetadata(value).lifecycleProtocol === MICROVM_LIFECYCLE_PROTOCOL; +} + +/** Check the exact immutable version returned by Run, never a latest-version alias. */ +export function verifyMicrovmImageLifecycle( + identity: Required>, + version: Pick, +): boolean { + const hooks = version.hooks; + const runtime = hooks?.microvmHooks; + const timeout = sharedConstants.microvm_hook_budgets.lifecycle_hook_timeout_seconds; + return version.imageArn === identity.imageArn + && version.imageVersion === identity.imageVersion + && version.environmentVariables?.[MICROVM_IMAGE_PROTOCOL_ENV] === MICROVM_LIFECYCLE_PROTOCOL + && hooks?.port === sharedConstants.microvm_lifecycle.hook_port + && hooks.microvmImageHooks?.ready === 'ENABLED' + && hooks.microvmImageHooks?.validate === 'ENABLED' + && runtime?.run === 'ENABLED' + && runtime.terminate === 'ENABLED' + && runtime.suspend === 'ENABLED' + && runtime.resume === 'ENABLED' + && Number.isInteger(runtime.suspendTimeoutInSeconds) + && runtime.suspendTimeoutInSeconds! >= timeout + && Number.isInteger(runtime.resumeTimeoutInSeconds) + && runtime.resumeTimeoutInSeconds! >= timeout; +} diff --git a/cdk/src/handlers/shared/microvm-lifecycle-policy.ts b/cdk/src/handlers/shared/microvm-lifecycle-policy.ts index feac21fc6..fafacee87 100644 --- a/cdk/src/handlers/shared/microvm-lifecycle-policy.ts +++ b/cdk/src/handlers/shared/microvm-lifecycle-policy.ts @@ -18,6 +18,7 @@ */ import type { SessionStatus } from './compute-strategy'; +import { supportsMicrovmLifecycle } from './microvm-image-capability'; import { intentMatchesGate, type MicrovmLifecycleSnapshot } from './microvm-lifecycle'; import { TaskStatus, TERMINAL_STATUSES } from '../../constructs/task-status'; @@ -35,8 +36,6 @@ export interface MicrovmLifecyclePolicyInput { readonly pollIntervalMs: number; /** Stops new suspends; already-sleeping VMs can still wake or terminate. */ readonly suspendEnabled: boolean; - /** True only for the compatible pinned image once hooks/barriers are deployed. */ - readonly imageSupportsLifecycle: boolean; } export type MicrovmLifecycleDecision = ( @@ -98,7 +97,7 @@ export function decideMicrovmLifecycle(input: MicrovmLifecyclePolicyInput): Micr } return { action: 'wait', reason: 'intentionally-suspended', nextPollInMs: Math.min(nextPollInMs, wakeAt - nowMs) }; } - if (!input.suspendEnabled || !input.imageSupportsLifecycle) { + if (!input.suspendEnabled || !supportsMicrovmLifecycle(snapshot.handle)) { return { action: 'wait', reason: 'suspend-disabled', nextPollInMs }; } // A prior gate's in-flight suspend must be resolved conservatively. Persist a diff --git a/cdk/src/handlers/shared/microvm-lifecycle.ts b/cdk/src/handlers/shared/microvm-lifecycle.ts index 124885869..b62d428f4 100644 --- a/cdk/src/handlers/shared/microvm-lifecycle.ts +++ b/cdk/src/handlers/shared/microvm-lifecycle.ts @@ -20,6 +20,7 @@ import { randomUUID } from 'node:crypto'; import { GetCommand, TransactWriteCommand } from '@aws-sdk/lib-dynamodb'; import type { SessionHandle } from './compute-strategy'; +import { readMicrovmImageMetadata, supportsMicrovmLifecycle } from './microvm-image-capability'; import type { ApprovalStatus } from './types'; import { makeDocClient } from './ua'; import { TaskStatus, type TaskStatusType } from '../../constructs/task-status'; @@ -155,7 +156,13 @@ export async function readMicrovmLifecycleSnapshot(taskId: string, userId: strin requestId, approval, intent: task.microvm_lifecycle, - handle: { strategyType: 'lambda-microvm', sessionId: task.session_id, microvmId: metadata.microvmId, endpoint: metadata.endpoint }, + handle: { + strategyType: 'lambda-microvm', + sessionId: task.session_id, + microvmId: metadata.microvmId, + endpoint: metadata.endpoint, + ...readMicrovmImageMetadata(metadata), + }, }; } @@ -166,7 +173,8 @@ export function intentMatchesGate(snapshot: MicrovmLifecycleSnapshot): boolean { function eligible(snapshot: MicrovmLifecycleSnapshot, action: LifecycleAction, nowMs: number): boolean { if (!LIVE_TASK_STATUSES.includes(snapshot.status)) return false; if (action === 'resume') return true; - return snapshot.status === TaskStatus.AWAITING_APPROVAL && snapshot.requestId !== null + return supportsMicrovmLifecycle(snapshot.handle) + && snapshot.status === TaskStatus.AWAITING_APPROVAL && snapshot.requestId !== null && snapshot.approval.kind === 'present' && snapshot.approval.status === 'PENDING' && nowMs >= snapshot.approval.createdAtMs && nowMs < snapshot.approval.deadlineMs && !(intentMatchesGate(snapshot) && (snapshot.intent?.action === 'resume' @@ -206,6 +214,13 @@ export async function saveMicrovmLifecycleIntent( }; let condition = 'user_id = :user AND #status = :status AND compute_type = :type AND session_id = :id ' + 'AND compute_metadata.microvmId = :id AND compute_metadata.endpoint = :endpoint'; + if (action === 'suspend') { + condition += ' AND compute_metadata.imageArn = :imageArn AND compute_metadata.imageVersion = :imageVersion' + + ' AND compute_metadata.lifecycleProtocol = :protocol'; + values[':imageArn'] = snapshot.handle.imageArn; + values[':imageVersion'] = snapshot.handle.imageVersion; + values[':protocol'] = snapshot.handle.lifecycleProtocol; + } if (snapshot.requestId === null) { condition += ' AND attribute_not_exists(awaiting_approval_request_id)'; } else { diff --git a/cdk/src/handlers/shared/microvm-start.ts b/cdk/src/handlers/shared/microvm-start.ts index ad3f2aebe..8aaa2bffc 100644 --- a/cdk/src/handlers/shared/microvm-start.ts +++ b/cdk/src/handlers/shared/microvm-start.ts @@ -20,6 +20,7 @@ import { createHash } from 'node:crypto'; import { GetCommand, UpdateCommand } from '@aws-sdk/lib-dynamodb'; import type { SessionHandle } from './compute-strategy'; +import { MICROVM_IMAGE_CAPABILITY_REQUEST_TIMEOUT_MS, readMicrovmImageMetadata, supportsMicrovmLifecycle } from './microvm-image-capability'; import { makeDocClient } from './ua'; import { TaskStatus, TERMINAL_STATUSES } from '../../constructs/task-status'; @@ -83,6 +84,7 @@ function recordedHandle(record: StartRecord): MicrovmHandle | undefined { sessionId: record.session_id, microvmId: metadata.microvmId, endpoint: metadata.endpoint, + ...readMicrovmImageMetadata(metadata), }; } return undefined; @@ -185,7 +187,33 @@ export async function saveMicrovmStartHandle( ':id': handle.microvmId, ':handle': handle, ':type': 'lambda-microvm', - ':metadata': { microvmId: handle.microvmId, endpoint: handle.endpoint }, + ':metadata': { + microvmId: handle.microvmId, endpoint: handle.endpoint, ...readMicrovmImageMetadata(handle), + }, }, })); } + +/** Enrich only the same durably saved launch; never replace its identity or task state. */ +export async function saveMicrovmImageCapability( + taskId: string, clientToken: string, handle: MicrovmHandle, +): Promise { + if (!supportsMicrovmLifecycle(handle)) throw new Error('MicroVM image capability is incomplete'); + await ddb.send(new UpdateCommand({ + TableName: TABLE_NAME, + Key: { task_id: taskId }, + UpdateExpression: 'SET microvm_start.#handle.lifecycleProtocol = :protocol, compute_metadata.lifecycleProtocol = :protocol', + ConditionExpression: 'microvm_start.clientToken = :token AND session_id = :id AND ' + + 'microvm_start.#handle.microvmId = :id AND compute_metadata.microvmId = :id AND ' + + 'microvm_start.#handle.imageArn = :arn AND compute_metadata.imageArn = :arn AND ' + + 'microvm_start.#handle.imageVersion = :version AND compute_metadata.imageVersion = :version', + ExpressionAttributeNames: { '#handle': 'handle' }, + ExpressionAttributeValues: { + ':token': clientToken, + ':id': handle.microvmId, + ':arn': handle.imageArn, + ':version': handle.imageVersion, + ':protocol': handle.lifecycleProtocol, + }, + }), { abortSignal: AbortSignal.timeout(MICROVM_IMAGE_CAPABILITY_REQUEST_TIMEOUT_MS) }); +} diff --git a/cdk/src/handlers/shared/orchestrator.ts b/cdk/src/handlers/shared/orchestrator.ts index c70252771..f41f8804e 100644 --- a/cdk/src/handlers/shared/orchestrator.ts +++ b/cdk/src/handlers/shared/orchestrator.ts @@ -25,6 +25,7 @@ import { AttachmentBudgetExceededError, AttachmentConfigurationError, Attachment import { formatMicrovmTerminalFailure } from './error-classifier'; import { logger, type Logger } from './logger'; import { writeMinimalEpisode } from './memory'; +import { readMicrovmImageMetadata } from './microvm-image-capability'; import { coerceNumericOrNull } from './numeric'; import { computePromptVersion } from './prompt-version'; import { makeRegistryClient } from './registry/factory'; @@ -340,7 +341,7 @@ export function buildComputeMetadata(handle: SessionHandle): Record = { - sessionId: microvmId, strategyType: 'lambda-microvm', microvmId, endpoint, + let handle: Extract = { + sessionId: microvmId, + strategyType: 'lambda-microvm', + microvmId, + endpoint, + ...readMicrovmImageMetadata({ imageArn: result.imageArn, imageVersion: result.imageVersion }), }; try { await saveMicrovmStartHandle(taskId, latest.clientToken, handle); @@ -621,16 +630,41 @@ export class LambdaMicrovmComputeStrategy implements ComputeStrategy { throw new Error(`MICROVM_START_RECEIPT_SAVE_FAILED: ${String(err)}`, { cause: err }); } - // Image ARN/version is logged, NOT carried in the handle (ADR-021 - // sub-decision 1) — it is deployment-time config, and this line is the - // diagnostic record of which snapshot a given session actually booted. + // Persist the known worker above BEFORE optional image discovery. A crash or + // failed lookup must not widen the orphan window or break ordinary coding. + // Never infer capability from a requested pin or the deployment's latest image. + if (handle.imageArn === MICROVM_IMAGE_IDENTIFIER && handle.imageVersion) { + try { + const identity = { imageArn: handle.imageArn, imageVersion: handle.imageVersion }; + const version = await getClient().send(new GetMicrovmImageVersionCommand({ + imageIdentifier: identity.imageArn, imageVersion: identity.imageVersion, + }), { abortSignal: AbortSignal.timeout(MICROVM_IMAGE_CAPABILITY_REQUEST_TIMEOUT_MS) }); + if (verifyMicrovmImageLifecycle(identity, version)) { + const capable = { ...handle, lifecycleProtocol: MICROVM_LIFECYCLE_PROTOCOL }; + await saveMicrovmImageCapability(taskId, latest.clientToken, capable); + handle = capable; + } + } catch (error) { + // Explicit degraded mode: the saved worker remains usable, with new + // suspension disabled. Do not expose image environment or AWS error text. + const name = (error as { name?: unknown })?.name; + logger.warn('MicroVM image capability unavailable; automatic suspension remains disabled', { + task_id: taskId, + microvm_id: microvmId, + error_type: typeof name === 'string' && /^[A-Za-z0-9_]{1,100}$/.test(name) ? name : 'Error', + }); + } + } + + // The durable handle carries actual identity/capability for later decisions. logger.info('Lambda MicroVM session started', { task_id: taskId, microvm_id: microvmId, state: result.state, image_identifier: MICROVM_IMAGE_IDENTIFIER, image_arn: result.imageArn, - image_version: result.imageVersion ?? MICROVM_IMAGE_VERSION, + image_version: handle.imageVersion ?? null, + lifecycle_protocol: handle.lifecycleProtocol ?? 'unverified', maximum_duration_seconds: MICROVM_MAX_DURATION_SECONDS, payload_delivery: 'signed_reference', // KEY NAMES only, never values: this is the one operator-visible record of diff --git a/cdk/test/constructs/lambda-microvm-compute.test.ts b/cdk/test/constructs/lambda-microvm-compute.test.ts index 8d6ce8e96..a0549d3c3 100644 --- a/cdk/test/constructs/lambda-microvm-compute.test.ts +++ b/cdk/test/constructs/lambda-microvm-compute.test.ts @@ -68,6 +68,7 @@ interface BuildOptions { readonly region?: string; readonly context?: Record; readonly withImage?: boolean; + readonly imageEnvironmentVariables?: Record; readonly artifactSha256?: string | null; readonly externalImageIdentifier?: string; readonly externalImageVersion?: string; @@ -146,6 +147,7 @@ function instantiate(options: BuildOptions = {}): Omit { }), externalImageIdentifier: options.externalImageIdentifier, externalImageVersion: options.externalImageVersion, + imageEnvironmentVariables: options.imageEnvironmentVariables, minimumMemoryInMiB: options.minimumMemoryInMiB, }); @@ -226,11 +228,9 @@ describe('LambdaMicrovmCompute — image provisioned from a managed base image', expect(Math.max(...MICROVM_SUPPORTED_MEMORY_MIB)).toBe(DEFAULT_MINIMUM_MEMORY_MIB); }); - test('declares exactly the four enabled P2 image hooks until the P3 capability rollout', () => { - // Compare the whole enabled set. A declared hook must be served or the - // corresponding build/lifecycle transition fails. The guest additionally - // serves /suspend and /resume, but they stay undeclared until the P3 image - // capability rollout; a route alone must not opt existing workers into sleep. + test('declares all six served hooks with the shared P3 lifecycle budget', () => { + // Compare the whole set; declaration and runtime support must move together. + // The supervisor still gates automatic sleep on each worker's saved capability. const images = template.findResources('AWS::Lambda::MicrovmImage'); const hooks = Object.values(images)[0]!.Properties.Hooks; expect(hooks.Port).toBe(8080); @@ -250,6 +250,10 @@ describe('LambdaMicrovmCompute — image provisioned from a managed base image', // (ordinary event writes are best effort), and the budget bounds teardown on a // WEDGED guest that is still holding admission-gating memory quota. TerminateTimeoutInSeconds: 15, + Suspend: 'ENABLED', + SuspendTimeoutInSeconds: sharedConstants.microvm_hook_budgets.lifecycle_hook_timeout_seconds, + Resume: 'ENABLED', + ResumeTimeoutInSeconds: sharedConstants.microvm_hook_budgets.lifecycle_hook_timeout_seconds, }); // BUILD (image) hooks. /ready is MANDATORY: create-microvm-image refuses ANY @@ -323,27 +327,25 @@ describe('LambdaMicrovmCompute — image provisioned from a managed base image', }; const image = Object.values(template.findResources('AWS::Lambda::MicrovmImage'))[0]!; - // Hooks: all four states AND all four timeouts, in one comparison — which is + // Hooks: all six states AND all six timeouts, in one comparison — which is // precisely the assertion the missing one would have been. expect(toCfnKeys(flagJson('hooks'))).toEqual(image.Properties.Hooks); // ...and the architecture enum, the other half of P2-F2. expect(toCfnKeys(flagJson('cpu-configurations'))).toEqual(image.Properties.CpuConfigurations); + expect(Object.entries(flagJson('environment-variables') as Record) + .map(([Key, Value]) => ({ Key, Value }))).toEqual(image.Properties.EnvironmentVariables); }); - test('does NOT declare /suspend or /resume — they are P3 and nothing answers them yet', () => { - // The remaining half of the exactness rule, called out separately because it - // is the one that must survive P3 landing suspend/resume in ONE commit across - // all three strategies: until then, declaring either fails the corresponding - // lifecycle transition on a real suspend attempt. + test('pause/wake handlers leave response headroom inside declared service timeouts', () => { const images = template.findResources('AWS::Lambda::MicrovmImage'); const hooks = Object.values(images)[0]!.Properties.Hooks; - expect(hooks.MicrovmHooks.Suspend).toBeUndefined(); - expect(hooks.MicrovmHooks.SuspendTimeoutInSeconds).toBeUndefined(); - expect(hooks.MicrovmHooks.Resume).toBeUndefined(); - expect(hooks.MicrovmHooks.ResumeTimeoutInSeconds).toBeUndefined(); + expect(hooks.MicrovmHooks.SuspendTimeoutInSeconds) + .toBeGreaterThan(sharedConstants.microvm_hook_budgets.lifecycle_handler_budget_seconds); + expect(hooks.MicrovmHooks.ResumeTimeoutInSeconds) + .toBeGreaterThan(sharedConstants.microvm_hook_budgets.lifecycle_handler_budget_seconds); }); - test('the agent hook routes are exactly the four the service calls, under one prefix', () => { + test('the six declared agent hook routes share the service-owned prefix', () => { // The cross-package contract that used to be checked against the rendered // template. It cannot be any more: the template carries `ENABLED`, not a path // (P2-F2), so the routes now have a dedicated source — MICROVM_AGENT_HOOK_ROUTES @@ -354,13 +356,15 @@ describe('LambdaMicrovmCompute — image provisioned from a managed base image', // service POSTs to exactly these paths ("POST /aws/lambda-microvms/runtime/v1/ // ready HTTP/1.1" 200 OK, and the same for the other three). const routes = Object.values(MICROVM_AGENT_HOOK_ROUTES); - expect(routes).toHaveLength(4); + expect(routes).toHaveLength(6); for (const route of routes) { - expect(route).toMatch(/^\/aws\/lambda-microvms\/runtime\/v1\/(ready|validate|run|terminate)$/); + expect(route).toMatch(/^\/aws\/lambda-microvms\/runtime\/v1\/(ready|validate|run|terminate|suspend|resume)$/); } expect([...routes].sort()).toEqual([ '/aws/lambda-microvms/runtime/v1/ready', + '/aws/lambda-microvms/runtime/v1/resume', '/aws/lambda-microvms/runtime/v1/run', + '/aws/lambda-microvms/runtime/v1/suspend', '/aws/lambda-microvms/runtime/v1/terminate', '/aws/lambda-microvms/runtime/v1/validate', ]); @@ -368,7 +372,7 @@ describe('LambdaMicrovmCompute — image provisioned from a managed base image', // Hooks properties use — so "the agent serves every hook the image enables" // stays checkable from one place. expect(Object.keys(MICROVM_AGENT_HOOK_ROUTES).sort()) - .toEqual(['ready', 'run', 'terminate', 'validate']); + .toEqual(['ready', 'resume', 'run', 'suspend', 'terminate', 'validate']); }); test('every declared hook timeout sits inside the service window for its kind', () => { @@ -415,11 +419,20 @@ describe('LambdaMicrovmCompute — image provisioned from a managed base image', .toBeGreaterThanOrEqual(30); }); - test('bakes NO environment variables into the snapshot (ADR-021: no secrets in the image)', () => { + test('bakes only the non-secret image protocol marker by default', () => { template.hasResourceProperties('AWS::Lambda::MicrovmImage', { - EnvironmentVariables: [], + EnvironmentVariables: [{ + Key: sharedConstants.microvm_lifecycle.image_protocol_env, + Value: String(sharedConstants.microvm_lifecycle.protocol_version), + }], }); }); + test('rejects an override of the image source protocol', () => { + expect(() => instantiate({ + withImage: true, + imageEnvironmentVariables: { [sharedConstants.microvm_lifecycle.image_protocol_env]: '999' }, + })).toThrow('protocol marker is owned by the image source'); + }); test('routes image build-time egress through the BUILD connector, not the runtime one', () => { // The runtime connector is 443-only, and `agent/Dockerfile` runs `apt-get` @@ -992,14 +1005,11 @@ describe('LambdaMicrovmCompute — image provisioned from a managed base image', expect(message).toContain('P2'); // It must state what IS true now, or it reads as the old (wrong) claim — and // the hook list here is what an operator compares against a failed build or a - // failed lifecycle transition, so all four have to be named. - for (const hook of ['/ready', '/validate', '/run', '/terminate']) { + // failed lifecycle transition, so all six have to be named. + for (const hook of ['/ready', '/validate', '/run', '/terminate', '/suspend', '/resume']) { expect(message).toContain(hook); } - // ...and it must still say which two are NOT declared, or the enumeration - // above reads as "everything is wired". - expect(message).toContain('/suspend'); - expect(message).toContain('/resume'); + expect(message).toContain('supervisor integration and live sleep/wake acceptance remain open'); }); test('enables every hook the agent serves, and only those (rendered form)', () => { @@ -1008,14 +1018,9 @@ describe('LambdaMicrovmCompute — image provisioned from a managed base image', // the shape CloudFormation validates. `"ENABLED"`, never a path (P2-F2). const images = template.findResources('AWS::Lambda::MicrovmImage'); const rendered = JSON.stringify(Object.values(images)[0]!.Properties.Hooks); - for (const hook of ['Run', 'Terminate', 'Ready', 'Validate']) { + for (const hook of ['Run', 'Terminate', 'Ready', 'Validate', 'Suspend', 'Resume']) { expect(rendered).toContain(`"${hook}":"ENABLED"`); } - // P3, and nothing answers them yet. OMITTED rather than "DISABLED", so the - // absence assertion stays meaningful. - for (const hook of ['Suspend', 'Resume']) { - expect(rendered).not.toContain(hook); - } expect(rendered).not.toContain('DISABLED'); }); diff --git a/cdk/test/constructs/task-orchestrator.test.ts b/cdk/test/constructs/task-orchestrator.test.ts index 214cda176..4a523fae4 100644 --- a/cdk/test/constructs/task-orchestrator.test.ts +++ b/cdk/test/constructs/task-orchestrator.test.ts @@ -733,12 +733,13 @@ describe('TaskOrchestrator with the Lambda MicroVMs backend (ADR-021)', () => { expect(env.MICROVM_INGRESS_CONNECTOR_ARNS).not.toContain('NO_INGRESS'); }); - test('grants exactly the four P1 lifecycle actions and nothing more', () => { + test('grants lifecycle calls and image-version discovery without enabling pause/wake callers yet', () => { const actions = microvmStatements(template) .flatMap(s => Array.isArray(s.Action) ? s.Action : [s.Action]) .filter(a => a.startsWith('lambda:')); expect(actions.sort()).toEqual([ 'lambda:GetMicrovm', + 'lambda:GetMicrovmImageVersion', 'lambda:PassNetworkConnector', 'lambda:RunMicrovm', 'lambda:TerminateMicrovm', diff --git a/cdk/test/handlers/orchestrate-task-microvm.test.ts b/cdk/test/handlers/orchestrate-task-microvm.test.ts index abb33908f..3a4cdc43e 100644 --- a/cdk/test/handlers/orchestrate-task-microvm.test.ts +++ b/cdk/test/handlers/orchestrate-task-microvm.test.ts @@ -487,7 +487,7 @@ describe('orchestrate-task for a lambda-microvm task', () => { expect(mockFailTask).not.toHaveBeenCalled(); }); - test('persists microvmId and endpoint in compute_metadata on the RUNNING transition', async () => { + test('persists the worker and actual image identity on the RUNNING transition', async () => { runMicrovmOk(); const { ctx } = fakeContext(); @@ -502,7 +502,9 @@ describe('orchestrate-task for a lambda-microvm task', () => { expect(to).toBe(TaskStatus.RUNNING); expect(attrs.compute_type).toBe('lambda-microvm'); // ADR-021: the P3 approve/deny Lambdas resume from exactly these two keys. - expect(attrs.compute_metadata).toEqual({ microvmId: MICROVM_ID, endpoint: ENDPOINT }); + expect(attrs.compute_metadata).toEqual({ + microvmId: MICROVM_ID, endpoint: ENDPOINT, imageArn: 'arn:image', imageVersion: '7', + }); // sessionId is the microvmId (substrate identifier, mirroring ECS). expect(attrs.session_id).toBe(MICROVM_ID); // agent_runtime_arn is an AgentCore-only attribute and must not appear. diff --git a/cdk/test/handlers/shared/microvm-image-capability.test.ts b/cdk/test/handlers/shared/microvm-image-capability.test.ts new file mode 100644 index 000000000..68bc1dd7c --- /dev/null +++ b/cdk/test/handlers/shared/microvm-image-capability.test.ts @@ -0,0 +1,95 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +// SPDX-License-Identifier: MIT-0 + +import sharedConstants from '../../../../contracts/constants.json'; +import { + MICROVM_IMAGE_PROTOCOL_ENV, MICROVM_LIFECYCLE_PROTOCOL, readMicrovmImageMetadata, + supportsMicrovmLifecycle, verifyMicrovmImageLifecycle, +} from '../../../src/handlers/shared/microvm-image-capability'; + +const identity = { + imageArn: 'arn:aws:lambda:us-east-1:123456789012:microvm-image:test', + imageVersion: '3.0', +}; +type Version = Parameters[1]; +const version = (): Version => ({ + ...identity, + environmentVariables: { [MICROVM_IMAGE_PROTOCOL_ENV]: MICROVM_LIFECYCLE_PROTOCOL }, + hooks: { + port: sharedConstants.microvm_lifecycle.hook_port, + microvmImageHooks: { ready: 'ENABLED', validate: 'ENABLED' }, + microvmHooks: { + run: 'ENABLED', + terminate: 'ENABLED', + suspend: 'ENABLED', + resume: 'ENABLED', + suspendTimeoutInSeconds: sharedConstants.microvm_hook_budgets.lifecycle_hook_timeout_seconds, + resumeTimeoutInSeconds: sharedConstants.microvm_hook_budgets.lifecycle_hook_timeout_seconds, + }, + }, +}); + +test('verifies the exact launched version and its compatible hooks and marker', () => { + expect(verifyMicrovmImageLifecycle(identity, version())).toBe(true); + // Deactivation only blocks new launches; it does not rewrite an existing snapshot. + const inactive = { ...version(), status: 'INACTIVE' }; + expect(verifyMicrovmImageLifecycle(identity, inactive)).toBe(true); +}); + +test.each([ + ['different image', (v: Version) => { v.imageArn += '-other'; }], + ['different version', (v: Version) => { v.imageVersion = '4.0'; }], + ['no marker', (v: Version) => { v.environmentVariables = {}; }], + ['wrong protocol', (v: Version) => { v.environmentVariables![MICROVM_IMAGE_PROTOCOL_ENV] = '999'; }], + ['wrong port', (v: Version) => { v.hooks!.port = 8081; }], + ['no hooks', (v: Version) => { v.hooks = undefined; }], + ['no ready', (v: Version) => { v.hooks!.microvmImageHooks!.ready = 'DISABLED'; }], + ['no validate', (v: Version) => { v.hooks!.microvmImageHooks!.validate = 'DISABLED'; }], + ['no run', (v: Version) => { v.hooks!.microvmHooks!.run = 'DISABLED'; }], + ['no terminate', (v: Version) => { v.hooks!.microvmHooks!.terminate = 'DISABLED'; }], + ['no suspend', (v: Version) => { v.hooks!.microvmHooks!.suspend = 'DISABLED'; }], + ['no resume', (v: Version) => { v.hooks!.microvmHooks!.resume = 'DISABLED'; }], + ['short suspend', (v: Version) => { v.hooks!.microvmHooks!.suspendTimeoutInSeconds = 1; }], + ['short resume', (v: Version) => { v.hooks!.microvmHooks!.resumeTimeoutInSeconds = 1; }], + ['unknown suspend budget', (v: Version) => { v.hooks!.microvmHooks!.suspendTimeoutInSeconds = undefined; }], + ['fractional resume budget', (v: Version) => { v.hooks!.microvmHooks!.resumeTimeoutInSeconds = 30.5; }], +] as const)('rejects %s', (_label, mutate) => { + const changed = version(); + mutate(changed); + expect(verifyMicrovmImageLifecycle(identity, changed)).toBe(false); +}); + +test.each([undefined, null, [], {}, { lifecycleProtocol: '1' }, { ...identity, imageArn: 1 }, + { ...identity, imageVersion: ' 3.0' }, { ...identity, imageArn: 'a\nb' }])('malformed metadata never invents image capability: %j', value => { + expect(readMicrovmImageMetadata(value)).toEqual({}); + expect(supportsMicrovmLifecycle(value)).toBe(false); +}); + +test('legacy and unsupported metadata preserve identity without claiming lifecycle support', () => { + for (const lifecycleProtocol of [undefined, '999']) { + const metadata = { ...identity, lifecycleProtocol }; + expect(readMicrovmImageMetadata(metadata)).toEqual(identity); + expect(supportsMicrovmLifecycle(metadata)).toBe(false); + } + const capable = { ...identity, lifecycleProtocol: MICROVM_LIFECYCLE_PROTOCOL }; + expect(readMicrovmImageMetadata({ ...capable, untrustedExtra: 'drop-me' })).toEqual(capable); + expect(supportsMicrovmLifecycle(capable)).toBe(true); +}); diff --git a/cdk/test/handlers/shared/microvm-lifecycle-local.test.ts b/cdk/test/handlers/shared/microvm-lifecycle-local.test.ts index 9c446a9a9..21a82c959 100644 --- a/cdk/test/handlers/shared/microvm-lifecycle-local.test.ts +++ b/cdk/test/handlers/shared/microvm-lifecycle-local.test.ts @@ -56,6 +56,7 @@ const tasks = `lifecycle-tasks-${suffix}`; const approvals = `lifecycle-approvals-${suffix}`; Object.assign(process.env, { TASK_TABLE_NAME: tasks, TASK_APPROVALS_TABLE_NAME: approvals }); import { readMicrovmLifecycleSnapshot, saveMicrovmLifecycleIntent } from '../../../src/handlers/shared/microvm-lifecycle'; +import { claimMicrovmStart, saveMicrovmImageCapability, saveMicrovmStartHandle } from '../../../src/handlers/shared/microvm-start'; const raw = new DynamoDBClient({ endpoint: endpoint ?? 'http://127.0.0.1:1', @@ -102,7 +103,13 @@ local('MicroVM lifecycle against DynamoDB Local', () => { status: 'AWAITING_APPROVAL', compute_type: 'lambda-microvm', session_id: 'vm', - compute_metadata: { microvmId: 'vm', endpoint: 'https://vm.example' }, + compute_metadata: { + microvmId: 'vm', + endpoint: 'https://vm.example', + imageArn: 'arn:aws:lambda:us-east-1:123456789012:microvm-image:test', + imageVersion: '3.0', + lifecycleProtocol: '1', + }, awaiting_approval_request_id: 'gate', }, })); @@ -148,7 +155,13 @@ local('MicroVM lifecycle against DynamoDB Local', () => { expect(result.status).toBe('saved'); const saved = await taskRow(); expect(saved.status).toBe('AWAITING_APPROVAL'); - expect(saved.compute_metadata).toEqual({ microvmId: 'vm', endpoint: 'https://vm.example' }); + expect(saved.compute_metadata).toEqual({ + microvmId: 'vm', + endpoint: 'https://vm.example', + imageArn: 'arn:aws:lambda:us-east-1:123456789012:microvm-image:test', + imageVersion: '3.0', + lifecycleProtocol: '1', + }); expect(saved.microvm_lifecycle).toMatchObject({ action: 'suspend', request_id: 'gate', @@ -156,6 +169,89 @@ local('MicroVM lifecycle against DynamoDB Local', () => { deadline_ms: observed.approval.kind === 'present' ? observed.approval.deadlineMs : NaN, }); }); + test.each([ + ['imageArn', 'arn:aws:lambda:us-east-1:123456789012:microvm-image:replacement'], + ['imageVersion', '4.0'], ['lifecycleProtocol', '999'], ['lifecycleProtocol', undefined], + ])('changed image %s fences an already planned suspend', async (field, value) => { + const old = await current(); + const metadata = { ...(await taskRow()).compute_metadata, [field]: value }; + if (value === undefined) delete metadata[field]; + await set(tasks, 'compute_metadata', metadata); + expect(await saveMicrovmLifecycleIntent(old, 'suspend')).toEqual({ status: 'stale' }); + expect((await taskRow()).microvm_lifecycle).toBeUndefined(); + }); + test('legacy image cannot sleep but can be recovered with a wake', async () => { + await set(tasks, 'compute_metadata', { microvmId: 'vm', endpoint: 'https://vm.example' }); + const legacy = await current(); + expect(await saveMicrovmLifecycleIntent(legacy, 'suspend')).toEqual({ status: 'ineligible' }); + expect((await saveMicrovmLifecycleIntent(legacy, 'resume')).status).toBe('saved'); + }); + describe('image capability enrichment', () => { + const handle = { + strategyType: 'lambda-microvm' as const, + sessionId: 'vm', + microvmId: 'vm', + endpoint: 'https://vm.example', + imageArn: 'arn:aws:lambda:us-east-1:123456789012:microvm-image:test', + imageVersion: '3.0', + }; + const capable = { ...handle, lifecycleProtocol: '1' }; + beforeEach(async () => { + await set(tasks, 'status', 'HYDRATING'); + await set(tasks, 'microvm_start', { + clientToken: 'task', requestHash: 'original', expiresAt: Date.now() + 120_000, + }); + await saveMicrovmStartHandle('task', 'task', handle); + }); + test('persists support in both records without changing a concurrent terminal state', async () => { + await set(tasks, 'status', 'CANCELLED'); + await saveMicrovmImageCapability('task', 'task', capable); + const saved = await taskRow(); + expect(saved.status).toBe('CANCELLED'); + expect(saved.microvm_start.handle).toEqual(capable); + expect(saved.compute_metadata).toEqual({ + microvmId: handle.microvmId, + endpoint: handle.endpoint, + imageArn: handle.imageArn, + imageVersion: handle.imageVersion, + lifecycleProtocol: '1', + }); + expect(await claimMicrovmStart('task', 'user', 'changed-request')).toEqual({ + clientToken: 'task', closed: true, handle: capable, + }); + }); + test.each([ + 'session_id', 'microvm_start.clientToken', 'microvm_start.handle.microvmId', + 'microvm_start.handle.imageArn', 'microvm_start.handle.imageVersion', + 'compute_metadata.microvmId', 'compute_metadata.imageArn', 'compute_metadata.imageVersion', + ])('changed %s rejects the entire capability update', async path => { + const changed = await taskRow(); + const fields = path.split('.'); + let parent = changed; + for (const field of fields.slice(0, -1)) parent = parent[field]; + parent[fields[fields.length - 1]] = 'replacement'; + await admin.send(new PutCommand({ TableName: tasks, Item: changed })); + await expect(saveMicrovmImageCapability('task', 'task', capable)) + .rejects.toMatchObject({ name: 'ConditionalCheckFailedException' }); + expect(await taskRow()).toEqual(changed); + }); + test('lost committed reply is recovered by the original start receipt', async () => { + mockAfterSend.mockImplementationOnce(() => { + throw Object.assign(new Error('lost reply'), { name: 'TimeoutError' }); + }); + await expect(saveMicrovmImageCapability('task', 'task', capable)).rejects.toThrow('lost reply'); + expect(await claimMicrovmStart('task', 'user', 'original')).toEqual({ + clientToken: 'task', closed: false, handle: capable, + }); + }); + test('incomplete capability rejects before a database mutation', async () => { + const saved = await taskRow(); + mockBeforeSend.mockClear(); + await expect(saveMicrovmImageCapability('task', 'task', handle)).rejects.toThrow('incomplete'); + expect(mockBeforeSend).not.toHaveBeenCalled(); + expect(await taskRow()).toEqual(saved); + }); + }); test('a wake blocks an older absent-record sleep and remains sticky on fresh reads', async () => { const old = await current(); await saveMicrovmLifecycleIntent(await current(), 'resume'); diff --git a/cdk/test/handlers/shared/microvm-lifecycle.test.ts b/cdk/test/handlers/shared/microvm-lifecycle.test.ts index 6436d6bf1..2888a41ba 100644 --- a/cdk/test/handlers/shared/microvm-lifecycle.test.ts +++ b/cdk/test/handlers/shared/microvm-lifecycle.test.ts @@ -33,13 +33,18 @@ import { decideMicrovmLifecycle, type MicrovmLifecyclePolicyInput } from '../../ const NOW = 1_800_000_000_000; const CREATED = new Date(NOW - 45_000).toISOString(); const DEADLINE = NOW + 555_000; +const imageMetadata = { + imageArn: 'arn:aws:lambda:us-east-1:123456789012:microvm-image:test', + imageVersion: '3.0', + lifecycleProtocol: '1', +}; const task = { task_id: 'task', user_id: 'user', status: 'AWAITING_APPROVAL', compute_type: 'lambda-microvm', session_id: 'vm', - compute_metadata: { microvmId: 'vm', endpoint: 'https://vm.example' }, + compute_metadata: { microvmId: 'vm', endpoint: 'https://vm.example', ...imageMetadata }, awaiting_approval_request_id: 'gate', }; const row = { task_id: 'task', user_id: 'user', request_id: 'gate', status: 'PENDING', created_at: CREATED, timeout_s: 600 }; @@ -57,7 +62,7 @@ const snapshot = (overrides: Partial = {}): MicrovmLif userId: 'user', status: 'AWAITING_APPROVAL', requestId: 'gate', - handle: { strategyType: 'lambda-microvm', sessionId: 'vm', microvmId: 'vm', endpoint: 'https://vm.example' }, + handle: { strategyType: 'lambda-microvm', sessionId: 'vm', microvmId: 'vm', endpoint: 'https://vm.example', ...imageMetadata }, approval: { kind: 'present', status: 'PENDING', created_at: CREATED, timeout_s: 600, createdAtMs: NOW - 45_000, deadlineMs: DEADLINE }, ...overrides, }); @@ -68,7 +73,6 @@ const policy = (state: MicrovmObservedState = 'RUNNING', change: Partial { test('legacy coarse running without a service observation cannot trigger suspend', () => { expect(policy('RUNNING', { substrate: { status: 'running' } })).toMatchObject({ action: 'wait', reason: 'unconfirmed-state' }); }); - test.each([{ suspendEnabled: false }, { imageSupportsLifecycle: false }])('requires both enable and compatible image: %j', change => { + test.each([ + { suspendEnabled: false }, + { snapshot: snapshot({ handle: { ...snapshot().handle, lifecycleProtocol: undefined } }) }, + ])('requires both enable and the actual worker capability: %j', change => { expect(policy('RUNNING', change)).toMatchObject({ action: 'wait', reason: 'suspend-disabled' }); }); test('long poll setting is clamped to the end of grace', () => { diff --git a/cdk/test/handlers/shared/orchestrator.test.ts b/cdk/test/handlers/shared/orchestrator.test.ts index 5b8759613..590b40655 100644 --- a/cdk/test/handlers/shared/orchestrator.test.ts +++ b/cdk/test/handlers/shared/orchestrator.test.ts @@ -134,14 +134,23 @@ describe('buildComputeMetadata', () => { expect(buildComputeMetadata(handle)).toEqual({ microvmId: MICROVM_ID, endpoint: ENDPOINT }); }); - test('never carries the MicroVM image ARN (deployment config, not session state)', () => { + test('preserves actual image identity and verified capability for later policy decisions', () => { const metadata = buildComputeMetadata({ sessionId: MICROVM_ID, strategyType: 'lambda-microvm', microvmId: MICROVM_ID, endpoint: ENDPOINT, + imageArn: 'arn:aws:lambda:us-east-1:123456789012:microvm-image:test', + imageVersion: '3.0', + lifecycleProtocol: '1', + }); + expect(metadata).toEqual({ + microvmId: MICROVM_ID, + endpoint: ENDPOINT, + imageArn: 'arn:aws:lambda:us-east-1:123456789012:microvm-image:test', + imageVersion: '3.0', + lifecycleProtocol: '1', }); - expect(Object.keys(metadata).sort()).toEqual(['endpoint', 'microvmId']); }); test('produces only string values (compute_metadata is Record in DDB)', () => { diff --git a/cdk/test/handlers/shared/strategies/lambda-microvm-strategy.test.ts b/cdk/test/handlers/shared/strategies/lambda-microvm-strategy.test.ts index d5b69136c..383dc47b5 100644 --- a/cdk/test/handlers/shared/strategies/lambda-microvm-strategy.test.ts +++ b/cdk/test/handlers/shared/strategies/lambda-microvm-strategy.test.ts @@ -79,15 +79,18 @@ for (const optional of [ const mockSend = jest.fn(); const mockClaimStart = jest.fn(); const mockSaveHandle = jest.fn(); +const mockSaveCapability = jest.fn(); jest.mock('../../../../src/handlers/shared/microvm-start', () => ({ ...jest.requireActual('../../../../src/handlers/shared/microvm-start'), claimMicrovmStart: (...args: unknown[]) => mockClaimStart(...args), saveMicrovmStartHandle: (...args: unknown[]) => mockSaveHandle(...args), + saveMicrovmImageCapability: (...args: unknown[]) => mockSaveCapability(...args), })); jest.mock('@aws-sdk/client-lambda-microvms', () => ({ LambdaMicrovmsClient: jest.fn(() => ({ send: mockSend })), RunMicrovmCommand: jest.fn((input: unknown) => ({ _type: 'RunMicrovm', input })), GetMicrovmCommand: jest.fn((input: unknown) => ({ _type: 'GetMicrovm', input })), + GetMicrovmImageVersionCommand: jest.fn((input: unknown) => ({ _type: 'GetMicrovmImageVersion', input })), TerminateMicrovmCommand: jest.fn((input: unknown) => ({ _type: 'TerminateMicrovm', input })), // Mirrors the real SDK's const-object enum so the strategy's switch keys on // the same literals the service returns. @@ -234,10 +237,93 @@ async function withoutEnvAsync(keys: string[], body: () => Promise): Promi beforeEach(() => { jest.clearAllMocks(); + mockSend.mockReset(); mockPrepare.mockReset().mockImplementation(async ({ taskId }: { taskId: string }) => ({ version: 2, task_id: taskId, bootstrap_s3_uri: 's3://b/bootstrap/example.json', payload_url: 'https://signed.example/task', expires_at: Date.now()+900000 })); mockDelete.mockResolvedValue(undefined); mockClaimStart.mockReset().mockImplementation(async (taskId: string) => ({ clientToken: taskId, closed: false })); mockSaveHandle.mockReset().mockResolvedValue(undefined); + mockSaveCapability.mockReset().mockResolvedValue(undefined); +}); + +describe('per-worker image capability', () => { + const input = { + taskId: 'CAP001', + userId: 'cognito-test', + payload: { task_id: 'CAP001' }, + blueprintConfig: BLUEPRINT, + }; + const identity = { imageArn: IMAGE_IDENTIFIER, imageVersion: 'actual-3.0' }; + const version = () => ({ + ...identity, + environmentVariables: { + [sharedConstants.microvm_lifecycle.image_protocol_env]: String(sharedConstants.microvm_lifecycle.protocol_version), + }, + hooks: { + port: sharedConstants.microvm_lifecycle.hook_port, + microvmImageHooks: { ready: 'ENABLED', validate: 'ENABLED' }, + microvmHooks: { + run: 'ENABLED', + terminate: 'ENABLED', + suspend: 'ENABLED', + resume: 'ENABLED', + suspendTimeoutInSeconds: sharedConstants.microvm_hook_budgets.lifecycle_hook_timeout_seconds, + resumeTimeoutInSeconds: sharedConstants.microvm_hook_budgets.lifecycle_hook_timeout_seconds, + }, + }, + }); + + test('saves the known worker before querying its actual image and conditionally enriching it', async () => { + mockSend.mockResolvedValueOnce({ ...makeHandle(), ...identity }) + .mockImplementationOnce(async (command, options) => { + expect(mockSaveHandle).toHaveBeenCalledWith(input.taskId, input.taskId, { ...makeHandle(), ...identity }); + expect(command).toEqual({ + _type: 'GetMicrovmImageVersion', + input: { imageIdentifier: IMAGE_IDENTIFIER, imageVersion: identity.imageVersion }, + }); + expect(options.abortSignal).toBeInstanceOf(AbortSignal); + return version(); + }); + const result = await new LambdaMicrovmComputeStrategy().startSession(input); + const expected = { ...makeHandle(), ...identity, lifecycleProtocol: '1' }; + expect(result).toEqual(expected); + expect(mockSaveCapability).toHaveBeenCalledWith(input.taskId, input.taskId, expected); + expect(mockSaveHandle.mock.invocationCallOrder[0]).toBeLessThan(mockSaveCapability.mock.invocationCallOrder[0]); + }); + + test.each(['lookup', 'marker', 'persistence'])('failed %s keeps the saved worker usable without enabling sleep', async failure => { + mockSend.mockResolvedValueOnce({ ...makeHandle(), ...identity, lifecycleProtocol: '1' }); + if (failure === 'lookup') { + mockSend.mockRejectedValueOnce(Object.assign(new Error('do-not-log-secret'), { name: 'AbortError' })); + } else { + const response = version(); + if (failure === 'marker') response.environmentVariables = {}; + mockSend.mockResolvedValueOnce(response); + if (failure === 'persistence') mockSaveCapability.mockRejectedValueOnce(new Error('do-not-log-secret')); + } + expect(await new LambdaMicrovmComputeStrategy().startSession(input)).toEqual({ ...makeHandle(), ...identity }); + expect(mockSaveHandle.mock.calls[0][2]).not.toHaveProperty('lifecycleProtocol'); + expect(mockSend.mock.calls.filter(([command]) => command._type === 'RunMicrovm')).toHaveLength(1); + expect(mockSend.mock.calls.filter(([command]) => command._type === 'TerminateMicrovm')).toHaveLength(0); + expect(JSON.stringify(mockLogger.warn.mock.calls)).not.toContain('do-not-log-secret'); + }); + + test.each([{}, { imageArn: IMAGE_IDENTIFIER }, { ...identity, imageArn: 'arn:other-image' }])( + 'missing or mismatched returned identity cannot borrow deployment capability: %j', async returned => { + mockSend.mockResolvedValueOnce({ ...makeHandle(), ...returned }); + const result = await new LambdaMicrovmComputeStrategy().startSession(input); + expect(result).not.toHaveProperty('lifecycleProtocol'); + expect(mockSend).toHaveBeenCalledTimes(1); + expect(mockSaveCapability).not.toHaveBeenCalled(); + }, + ); + + test.each([undefined, '1'])('replay preserves saved capability %s without querying the current deployment', async lifecycleProtocol => { + const saved = lifecycleProtocol ? { ...makeHandle(), ...identity, lifecycleProtocol } : makeHandle(); + mockClaimStart.mockResolvedValue({ clientToken: input.taskId, handle: saved, closed: false }); + expect(await new LambdaMicrovmComputeStrategy().startSession(input)).toEqual(saved); + expect(mockSend).not.toHaveBeenCalled(); + expect(mockSaveCapability).not.toHaveBeenCalled(); + }); }); describe('LambdaMicrovmComputeStrategy', () => { @@ -256,7 +342,7 @@ describe('LambdaMicrovmComputeStrategy', () => { blueprintConfig: BLUEPRINT, }); - expect(mockSend).toHaveBeenCalledTimes(1); + expect(mockSend).toHaveBeenCalledTimes(2); const call = mockSend.mock.calls[0][0]; expect(call._type).toBe('RunMicrovm'); expect(call.input.imageIdentifier).toBe(IMAGE_IDENTIFIER); @@ -269,6 +355,8 @@ describe('LambdaMicrovmComputeStrategy', () => { strategyType: 'lambda-microvm', microvmId: MICROVM_ID, endpoint: ENDPOINT, + imageArn: IMAGE_IDENTIFIER, + imageVersion: IMAGE_VERSION, }); }); diff --git a/cdk/test/handlers/start-session-composition.test.ts b/cdk/test/handlers/start-session-composition.test.ts index f1ea4ba40..fac6ea82c 100644 --- a/cdk/test/handlers/start-session-composition.test.ts +++ b/cdk/test/handlers/start-session-composition.test.ts @@ -45,6 +45,7 @@ jest.mock('@aws-sdk/client-lambda-microvms', () => ({ LambdaMicrovmsClient: jest.fn(() => ({ send: mockMicrovmSend })), RunMicrovmCommand: jest.fn((input: unknown) => ({ _type: 'RunMicrovm', input })), GetMicrovmCommand: jest.fn((input: unknown) => ({ _type: 'GetMicrovm', input })), + GetMicrovmImageVersionCommand: jest.fn((input: unknown) => ({ _type: 'GetMicrovmImageVersion', input })), TerminateMicrovmCommand: jest.fn((input: unknown) => ({ _type: 'TerminateMicrovm', input })), MicrovmState: { PENDING: 'PENDING', @@ -218,7 +219,7 @@ describe('start-session step composition — lambda-microvm (ADR-021)', () => { }); }); - test('startSession → buildComputeMetadata → transitionTask persists microvmId and endpoint', async () => { + test('startSession → buildComputeMetadata → transitionTask persists the worker and actual image', async () => { mockMicrovmSend.mockResolvedValueOnce({ microvmId: MICROVM_ID, endpoint: ENDPOINT, @@ -246,27 +247,38 @@ describe('start-session step composition — lambda-microvm (ADR-021)', () => { c[0]._type === 'Update' && c[0].input.ExpressionAttributeValues[':toStatus'] === TaskStatus.RUNNING)![0]; const values = update.input.ExpressionAttributeValues as Record; expect(values[':attr_compute_type']).toBe('lambda-microvm'); - expect(values[':attr_compute_metadata']).toEqual({ microvmId: MICROVM_ID, endpoint: ENDPOINT }); + expect(values[':attr_compute_metadata']).toEqual({ + microvmId: MICROVM_ID, endpoint: ENDPOINT, imageArn: 'arn:image', imageVersion: '7', + }); expect(values[':attr_session_id']).toBe(MICROVM_ID); expect(values[':toStatus']).toBe(TaskStatus.RUNNING); }); - test('compute_metadata carries ONLY the two lifecycle keys (no image ARN)', async () => { + test('image identity alone does not claim verified lifecycle support', async () => { mockMicrovmSend.mockResolvedValueOnce({ microvmId: MICROVM_ID, endpoint: ENDPOINT, state: 'RUNNING', imageArn: 'arn:aws:lambda:us-east-1:123456789012:microvm-image/abca-agent', imageVersion: '7', + }).mockResolvedValueOnce({ + imageArn: 'arn:aws:lambda:us-east-1:123456789012:microvm-image/abca-agent', + imageVersion: '7', + hooks: {}, }); const strategy = resolveComputeStrategy(blueprintConfig); const handle = await strategy.startSession({ taskId, userId: 'cognito-test', payload, blueprintConfig }); const metadata = buildComputeMetadata(handle); - // ADR-021: the image ARN is deployment-time config, logged not persisted. - expect(Object.keys(metadata).sort()).toEqual(['endpoint', 'microvmId']); - expect(JSON.stringify(metadata)).not.toContain('microvm-image'); + expect(metadata).toEqual({ + microvmId: MICROVM_ID, + endpoint: ENDPOINT, + imageArn: 'arn:aws:lambda:us-east-1:123456789012:microvm-image/abca-agent', + imageVersion: '7', + }); + expect(metadata.lifecycleProtocol).toBeUndefined(); + expect(mockMicrovmSend.mock.calls[1][0]._type).toBe('GetMicrovmImageVersion'); }); test('a rejected RunMicrovm preserves its marked AWS exception for the caller', async () => { diff --git a/cdk/test/scripts/check-constants-sync.test.ts b/cdk/test/scripts/check-constants-sync.test.ts index 84515bd2e..e232e4a33 100644 --- a/cdk/test/scripts/check-constants-sync.test.ts +++ b/cdk/test/scripts/check-constants-sync.test.ts @@ -66,6 +66,7 @@ const FIXTURE_FILES = [ 'agent/src/microvm_http.py', 'cdk/src/handlers/shared/payload-bootstrap.ts', 'cdk/src/constructs/lambda-microvm-compute.ts', + 'cdk/src/handlers/shared/microvm-image-capability.ts', ]; interface RunResult { @@ -129,6 +130,34 @@ describe('check-constants-sync', () => { // subprocess spawns, so give it room on a cold cache. jest.setTimeout(60_000); + describe('MicroVM lifecycle image contract', () => { + test.each([ + ['protocol_version', 0], ['protocol_version', 1.5], ['protocol_version', '1'], + ['hook_port', 0], ['hook_port', 65536], ['hook_port', 8080.5], + ['image_protocol_env', 'AWS_ACCESS_KEY_ID'], ['image_protocol_env', 'ABCA_MICROVM_bad'], + ])('rejects invalid %s=%s', (key, value) => { + const result = runInMutatedRepo(root => patchContract(root, json => { + json.microvm_lifecycle[key] = value; + })); + expect(result.status).toBe(1); + expect(result.stderr).toContain('microvm_lifecycle'); + }); + test.each([ + ['cdk/src/constructs/lambda-microvm-compute.ts', 'AGENT_HOOK_PORT', '8080'], + ['cdk/src/constructs/lambda-microvm-compute.ts', 'LIFECYCLE_HOOK_TIMEOUT_SECONDS', '30'], + ['cdk/src/handlers/shared/microvm-image-capability.ts', 'MICROVM_LIFECYCLE_PROTOCOL', '"1"'], + ['cdk/src/handlers/shared/microvm-image-capability.ts', 'MICROVM_LIFECYCLE_PROTOCOL', 'String(1)'], + ['cdk/src/handlers/shared/microvm-image-capability.ts', 'MICROVM_IMAGE_PROTOCOL_ENV', '"ABCA_MICROVM_LIFECYCLE_PROTOCOL"'], + ])('rejects a literal %s/%s', (file, name, value) => { + const result = runInMutatedRepo(root => { + write(root, file, `${read(root, file)}\nexport const ${name} = ${value};\n`); + }); + expect(result.status).toBe(1); + expect(result.stderr).toContain(name); + expect(result.stderr).toContain('Cross-language constants drift detected'); + }); + }); + describe('payload bootstrap contract', () => { test.each([ ['max_payload_bytes', 0], diff --git a/cdk/test/stacks/agent.test.ts b/cdk/test/stacks/agent.test.ts index a41c6129c..962d334d5 100644 --- a/cdk/test/stacks/agent.test.ts +++ b/cdk/test/stacks/agent.test.ts @@ -1176,7 +1176,7 @@ describe('AgentStack with the Lambda MicroVMs substrate gate (--context compute_ .toMatch(/"Fn::GetAtt":\["LambdaMicrovmComputeImage[^"]*","ImageArn"\]/); }); - test('grants the orchestrator exactly the P1 lifecycle actions, image-scoped', () => { + test('grants the orchestrator its launch, state, cleanup and image-capability actions, image-scoped', () => { const policies = Object.entries(template.findResources('AWS::IAM::Policy')) .filter(([id]) => id.includes('TaskOrchestrator')); const statements = policies.flatMap(([, p]) => p.Properties.PolicyDocument.Statement as Array<{ @@ -1189,6 +1189,7 @@ describe('AgentStack with the Lambda MicroVMs substrate gate (--context compute_ expect(lifecycle.Action).toEqual([ 'lambda:RunMicrovm', 'lambda:GetMicrovm', + 'lambda:GetMicrovmImageVersion', 'lambda:TerminateMicrovm', ]); // Every MicroVM lifecycle action authorizes against the *image* resource, @@ -1202,7 +1203,7 @@ describe('AgentStack with the Lambda MicroVMs substrate gate (--context compute_ expect(pass.Action).toBe('lambda:PassNetworkConnector'); expect(pass.Resource).toBe('*'); - // iam:PassRole for the execution role hand-off, service-conditioned. + // iam:PassRole for the execution role hand-off is scoped to the exact role. const passRole = statements.find(s => s.Sid === 'MicrovmPassExecutionRole')!; expect(passRole.Action).toBe('iam:PassRole'); }); diff --git a/contracts/constants.json b/contracts/constants.json index 546d7098a..49a6064df 100644 --- a/contracts/constants.json +++ b/contracts/constants.json @@ -77,6 +77,11 @@ "lifecycle_hook_timeout_seconds": 30, "lifecycle_handler_budget_seconds": 20 }, + "microvm_lifecycle": { + "protocol_version": 1, + "image_protocol_env": "ABCA_MICROVM_LIFECYCLE_PROTOCOL", + "hook_port": 8080 + }, "linear_vault": { "_why": "The vault caches a grant keyed by the WHOLE token request, customParameters included. A one-token divergence between any two of these copies makes every resolve a cache miss, which post-#812 is reported as consent-required and can latch a healthy workspace revoked.", "scopes": [ diff --git a/contracts/constants.md b/contracts/constants.md index aca047b0f..e03bd5783 100644 --- a/contracts/constants.md +++ b/contracts/constants.md @@ -23,11 +23,12 @@ the contract. This is the neutral location both runtimes read. | `agent/src/policy.py`, `agent/src/jira_reactions.py` | `SHARED_CONSTANTS` | import-time | | `agent/src/payload_bootstrap.py` | `SHARED_CONSTANTS["payload_bootstrap"]` | import-time | | `cdk/src/handlers/shared/payload-bootstrap.ts`, `cdk/src/constructs/payload-bootstrap-permissions.ts` | `payload_bootstrap` | import-time | -| `agent/src/server.py` | `SHARED_CONSTANTS["microvm_platform_config"]`, `SHARED_CONSTANTS["microvm_hook_budgets"]` | import-time | +| `agent/src/server.py` | `SHARED_CONSTANTS["microvm_platform_config"]`, `SHARED_CONSTANTS["microvm_hook_budgets"]`, `SHARED_CONSTANTS["microvm_lifecycle"]` | import-time | | `agent/src/microvm_http.py` | `SHARED_CONSTANTS["microvm_hook_budgets"]` | import-time | | `cdk/src/handlers/shared/types.ts`, `jira-app-actor.ts` | `../../../../contracts/constants.json` | synth-time `import` | | `cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts` | `microvm_platform_config` | synth-time `import`, read per session start | -| `cdk/src/constructs/lambda-microvm-compute.ts` | `microvm_hook_budgets` | synth-time `import` | +| `cdk/src/constructs/lambda-microvm-compute.ts` | `microvm_hook_budgets`, `microvm_lifecycle` | synth-time `import` | +| `cdk/src/handlers/shared/microvm-image-capability.ts` | `microvm_hook_budgets`, `microvm_lifecycle` | runtime `import` | | `cdk/src/constructs/blueprint.ts` | re-exports from `types.ts` | synth-time | | `cli/test/constants-parity.test.ts` | package-safe literal parity | test-time | @@ -95,6 +96,11 @@ JSON at TypeScript compile time via `resolveJsonModule`. "warmup_required_timeout_seconds": 120, "lifecycle_hook_timeout_seconds": 30, "lifecycle_handler_budget_seconds": 20 + }, + "microvm_lifecycle": { + "protocol_version": 1, + "image_protocol_env": "ABCA_MICROVM_LIFECYCLE_PROTOCOL", + "hook_port": 8080 } } ``` @@ -213,11 +219,21 @@ fails the drift check *and* the image build. The runtime lifecycle pair has its own relationship: `lifecycle_handler_budget_seconds < lifecycle_hook_timeout_seconds`. The 20-second handler limit covers reading the body, draining activity and checkpoint/refresh -work together. The planned 30-second service hook timeout leaves response headroom. +work together. The declared 30-second service hook timeout leaves response headroom. `microvm_http.py` checks the ordering at import time; the drift script checks -positive integer values, ordering and hardcoded Python redeclarations. The image -does not declare suspend/resume hooks yet; its later capability rollout must use -this service timeout. These values do not enable automatic suspension. +positive integer values, ordering and hardcoded Python/TypeScript redeclarations. +The image declares suspend/resume using this service timeout. These values do not +enable automatic suspension. + +`microvm_lifecycle` owns protocol version `1`, marker name +`ABCA_MICROVM_LIFECYCLE_PROTOCOL` and hook port `8080`. The marker is baked into +the immutable image; it is neither a credential nor task/deployment configuration. +`/validate` rejects a supplied unsupported marker. The coordinator checks the exact +image ARN/version returned by Run, including all six enabled hooks and lifecycle +budgets, before persisting `lifecycleProtocol` alongside that worker's `imageArn` +and `imageVersion`. Legacy or unverified workers cannot start a new suspension. +The drift gate validates this shape and rejects literal copies in its TypeScript +consumers; the artifact-script parity test checks the shell hook/environment JSON. The published CLI package contains only `lib/`, so it cannot load the repository contract at runtime. It mirrors these values as literals and diff --git a/docs/decisions/ADR-021-lambda-microvms-compute-backend.md b/docs/decisions/ADR-021-lambda-microvms-compute-backend.md index b0a9a89d9..690e5e046 100644 --- a/docs/decisions/ADR-021-lambda-microvms-compute-backend.md +++ b/docs/decisions/ADR-021-lambda-microvms-compute-backend.md @@ -1,6 +1,6 @@ # ADR-021: AWS Lambda MicroVMs as a third ComputeStrategy backend -> **Implementation status (2026-09-14):** P1 ([#689](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/689)) and P2 ([#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733)) are merged. The takeover branch now has a [clean deployment](../verification/645-p2-clean-deployment-20260913.md) with bootstrap 1.7.0 and a [verified managed-image update to 2.0](../verification/645-microvm-image-rebuild-20260914.md), plus [live coding, PR iteration and cancellation evidence](../verification/645-p2-live-task-20260914.md), without manual IAM workarounds. Remaining P2 failure/recovery and permission/network checks prevent full acceptance. Local P3 foundations include suspend/resume commands, explicit VM observations, guarded persistent intent, a policy helper, approval deadlines that survive a frozen clock, the [guest pause controller](../verification/645-p3-guest-barrier.md), and [scoped Claude credentials with retained-client refresh](../verification/645-p3-credentials.md). The [production HTTP hooks](../verification/645-p3-lifecycle-hooks.md) now connect the guest barrier to atomic checkpoints and credential/gate reconciliation locally. Image capability and supervisor/approval-handler integration remain unfinished; automatic suspension is not enabled. This ADR defines P1–P3, not a P4. See the [current review](../verification/645-p3-readiness-review.md) and [P3 implementation plan](../verification/645-p3-implementation-plan.md). +> **Implementation status (2026-09-15):** P1 ([#689](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/689)) and P2 ([#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733)) are merged. The takeover branch now has a [clean deployment](../verification/645-p2-clean-deployment-20260913.md) with bootstrap 1.7.0 and a [verified managed-image update to 2.0](../verification/645-microvm-image-rebuild-20260914.md), plus [live coding, PR iteration and cancellation evidence](../verification/645-p2-live-task-20260914.md), without manual IAM workarounds. Remaining P2 failure/recovery and permission/network checks prevent full acceptance. Local P3 foundations include suspend/resume commands, explicit VM observations, guarded persistent intent, a policy helper, approval deadlines that survive a frozen clock, the [guest pause controller](../verification/645-p3-guest-barrier.md), and [scoped Claude credentials with retained-client refresh](../verification/645-p3-credentials.md). The [production HTTP hooks](../verification/645-p3-lifecycle-hooks.md) now connect the guest barrier to atomic checkpoints and credential/gate reconciliation locally. [Per-worker image capability](../verification/645-p3-image-capability.md) now declares the six hooks and retains/verifies the actual launched image version locally. Supervisor/approval-handler integration remains unfinished; automatic suspension is not enabled. This ADR defines P1–P3, not a P4. See the [current review](../verification/645-p3-readiness-review.md) and [P3 implementation plan](../verification/645-p3-implementation-plan.md). **Status:** proposed **Date:** 2026-07-29 @@ -60,7 +60,7 @@ Adopt **AWS Lambda MicroVMs as a third, opt-in `ComputeStrategy` backend** named ### 1. Strategy shape: extend the interface with mandatory suspend/resume -`ComputeType` widens to `'agentcore' | 'ecs' | 'lambda-microvm'` (mirrored in `cli/src/types.ts` and the CLI's inline unions). `SessionHandle` gains a `{ strategyType: 'lambda-microvm', microvmId, endpoint }` variant — `microvmId` because every lifecycle API (`suspend-microvm`, `resume-microvm`, `terminate-microvm`, `get-microvm`, and `create-microvm-auth-token` — the latter not called in P1–P3, see sub-decision 3) takes only the MicroVM identifier, and `endpoint` because it is per-session (minted by `RunMicrovm`) and required for any orchestrator→agent HTTP interaction. Note the naming seam: the handle field is `microvmId` (matching `RunMicrovmResponse.microvmId`), while the request key on every lifecycle command is `microvmIdentifier` — the strategy is the only place that translates between the two. The image ARN is deliberately **not** in the handle: like the ECS task definition ARN, it is deployment-time configuration consumed by `startSession` (from construct-injected environment) and recorded in the session-start log entry for diagnostics, not per-session lifecycle state. +`ComputeType` widens to `'agentcore' | 'ecs' | 'lambda-microvm'` (mirrored in `cli/src/types.ts` and the CLI's inline unions). `SessionHandle` gains a `{ strategyType: 'lambda-microvm', microvmId, endpoint }` variant — `microvmId` because every lifecycle API (`suspend-microvm`, `resume-microvm`, `terminate-microvm`, `get-microvm`, and `create-microvm-auth-token` — the latter not called in P1–P3, see sub-decision 3) takes only the MicroVM identifier, and `endpoint` because it is per-session (minted by `RunMicrovm`) and required for any orchestrator→agent HTTP interaction. Note the naming seam: the handle field is `microvmId` (matching `RunMicrovmResponse.microvmId`), while the request key on every lifecycle command is `microvmIdentifier` — the strategy is the only place that translates between the two. **P3 refinement (2026-09-15):** the handle also retains the actual `imageArn` and `imageVersion` returned by Run, plus `lifecycleProtocol` after exact-version verification. The requested image remains deployment configuration, but the version that launched an existing worker is per-session evidence. Save the known handle before optional discovery; check that exact version with `GetMicrovmImageVersion`, then conditionally persist support in the receipt and `compute_metadata`. Current deployment settings cannot substitute for missing legacy evidence. **A second, sharper naming seam: `imageIdentifier` must be an ARN.** The name suggests a bare image name is acceptable — `create-microvm-image --name` takes one, and this ADR originally assumed `run-microvm --image-identifier` would too. It does not: a bare name is rejected with `ValidationException: Malformed ARN - doesn't start with 'arn:'`, and so is `list-microvm-image-builds --image-identifier ` (`Invalid ARN format`). The construct therefore resolves an operator-supplied name to its exact `arn:${Partition}:lambda:${Region}:${Account}:microvm-image:${Name}` ARN **once** — the same value it scopes the lifecycle IAM grant to — and injects THAT as `MICROVM_IMAGE_IDENTIFIER`. One derivation, two consumers, so a request field and an IAM resource can never disagree. The strategy validates the invariant and fails fast with the remedy, because the service's own error names neither the env var nor the fix. @@ -100,7 +100,7 @@ Normative requirements (EARS, per [ADR-020](./ADR-020-ears-requirements-syntax.m - When `startSession` is invoked, the strategy shall call `RunMicrovm` with `maximumDurationInSeconds` set to 28 800 (the service maximum, matching AgentCore's 8-hour session cap and sitting inside the orchestrator's ~8.5 h safety-net poll window). - When `startSession` is invoked, the strategy shall pass a fully-qualified MicroVM image ARN as `imageIdentifier`. - If the configured image identifier is not an ARN, then the strategy shall fail the session start with an error naming the environment variable and the redeploy remedy, before performing any AWS call. -- When `startSession` returns, the orchestrator shall persist the MicroVM handle (`microvmId`, `endpoint`) in the task row's `compute_metadata` (the field `cancel-task.ts` already reads ECS handles from). +- When `startSession` returns, the orchestrator shall persist the MicroVM handle (`microvmId`, `endpoint`, actual `imageArn`/`imageVersion` when returned, and verified `lifecycleProtocol` when supported) in the task row's `compute_metadata` (the field `cancel-task.ts` already reads ECS handles from). - The strategy shall omit `idlePolicy` on every `RunMicrovm` call, in every phase. - The orchestrator shall be the sole initiator of suspension, via `suspendSession`. - When `pollSession` observes MicroVM state `SUSPENDED` or `SUSPENDING`, the strategy shall report `suspended` without interpreting task state. @@ -173,10 +173,10 @@ The phasing is therefore: | `/ready` | **P1** (construct enables `hooks.microvmImageHooks.ready`) | **P1** | MANDATORY, not a quality nicety — see above. A 200 proves uvicorn is bound and `server` imported cleanly (pulling in `pipeline` → `runner` → the policy engine), so a missing policy file fails the BUILD instead of the first task. **Since P2-F5 it also WARMS the snapshot** — the hook's 200 is what the service waits for before capturing the snapshot, making this the only place a warm page can be created, and the 225 MiB `claude` binary was cold in it (see the P2-F5 correction below). A required warm-up failure answers 503, so a snapshot that cannot exec the agent's own CLI fails the image build instead of every task. Still makes ZERO AWS calls, logging included (a `--version` exec is neither an AWS call nor a network call). | | `/run` | **P1** (construct sets `hooks.microvmHooks.run`) | **P1** | The payload-delivery channel. Must be served in P1 because `/ready` forces hooks to exist at all, and a hook-less image cannot accept `runHookPayload`. Since P2 it is also the **platform-configuration** channel (see "Platform configuration delivery" below). | | `/validate` | **P2** (construct sets `hooks.microvmImageHooks.validate`) | **P2** | An **image** (build-time) hook, and a **shallow self-check only**: server alive, hook routes registered, interpreter + contract sanity. It runs under the BUILD role, which deliberately holds no Bedrock / Secrets Manager / DynamoDB grants, so it must make **zero AWS API calls** and must not touch credential resolution — the "deeper warm-up assertions (Bedrock reachability, Memory access, tool availability)" this ADR originally assigned here are **not implementable**: every one of them would `AccessDenied` and fail every image build. They belong to the first task's own error handling. 200 when the checks pass, 503 while still initialising. | -| `/terminate` | **P2** (construct sets `hooks.microvmHooks.terminate`) | **P2** | Best-effort final flush: a final structured log line, then 200 — always, inside the hook budget, even with nothing running. It must **not** write terminal task status (the orchestrator finalizes the task and *then* calls `TerminateMicrovm`, so a terminate hook that wrote a status would race that finalization and could clobber the real outcome) and must not join the pipeline thread. `ProgressWriter` writes synchronously but drops failed events, so this hook cannot guarantee that every progress write succeeded. P3 needs an acknowledged durability barrier for suspension. "Always 200" also covers the BODY: the handler reads the raw request rather than a typed model, because a typed body is validated before the handler runs and would answer 422 to malformed JSON — a reported hook failure on a successful teardown. Safe to declare because `TerminateMicrovm` removes the VM with or without in-guest cooperation. **Correction (P2-F8):** the service sends `microvmId: ""` on this hook, unlike `/run` where it is populated, so an empty id is expected-normal and this hook cannot join the guest record to the control-plane one — `/run`'s accepted line carries that correlation instead. | -| `/suspend`, `/resume` | **P3** | **P3** | Declaring a runtime hook the agent does not serve fails the corresponding lifecycle transition, so each is declared only in the phase that implements it. P1 termination is the orchestrator's `TerminateMicrovm`, which needs no in-guest cooperation. | +| `/terminate` | **P2** (construct sets `hooks.microvmHooks.terminate`) | **P2** | Best-effort final flush: a final structured log line, then 200 — always, inside the hook budget, even with nothing running. It must **not** write terminal task status (the orchestrator finalizes the task and *then* calls `TerminateMicrovm`, so a terminate hook that wrote a status would race that finalization and could clobber the real outcome) and must not join the pipeline thread. `ProgressWriter` writes synchronously but drops failed events, so this hook cannot guarantee that every progress write succeeded. P3 supplies a separate acknowledged checkpoint for suspension. "Always 200" also covers the BODY: the handler reads the raw request rather than a typed model, because a typed body is validated before the handler runs and would answer 422 to malformed JSON — a reported hook failure on a successful teardown. Safe to declare because `TerminateMicrovm` removes the VM with or without in-guest cooperation. **Correction (P2-F8):** the service sends `microvmId: ""` on this hook, unlike `/run` where it is populated, so an empty id is expected-normal and this hook cannot join the guest record to the control-plane one — `/run`'s accepted line carries that correlation instead. | +| `/suspend`, `/resume` | **P3**, implemented locally | **P3**, implemented locally | Managed images declare the served checkpoint/credential/gate hooks with the shared 30-second service timeout and protocol marker. The coordinator verifies the actual launched version before allowing new suspension. Supervisor integration and live acceptance remain open. | -Consequence to state plainly, replacing the original "a P1-built MicroVM image is not runnable end to end": **a P1 image is creatable, launchable and payload-deliverable, but carries no smoke-parity guarantee.** P1 delivers the strategy, the construct, the roles/buckets/connectors, the image resource, the packaging script, and the `/ready` + `/run` endpoints — so a `lambda-microvm` task can start a MicroVM and hand it a payload. What P1 has **not** established is anything P2 owns: AgentCore Memory grants and `MEMORY_ID` delivery, the agent's non-secret env parity inside the snapshot, egress specifics from a running MicroVM, and heartbeat/progress behaviour end to end. At the P1 handoff, no clone → change → PR run had happened. P2 subsequently demonstrated that path with an IAM workaround; clean unattended verification is still pending. The construct and the packaging script both surface exactly this at synth/run time (`abca:microvm-image-p1-smoke-unverified`) so an operator cannot mistake a launchable substrate for a verified one. +Consequence to state plainly, replacing the original "a P1-built MicroVM image is not runnable end to end": **a P1 image is creatable, launchable and payload-deliverable, but carries no smoke-parity guarantee.** P1 delivers the strategy, the construct, the roles/buckets/connectors, the image resource, the packaging script, and the `/ready` + `/run` endpoints — so a `lambda-microvm` task can start a MicroVM and hand it a payload. What P1 has **not** established is anything P2 owns: AgentCore Memory grants and `MEMORY_ID` delivery, the agent's non-secret env parity inside the snapshot, egress specifics from a running MicroVM, and heartbeat/progress behaviour end to end. At the P1 handoff, no clone → change → PR run had happened. P2 initially demonstrated that path with an IAM workaround; the 2026-09-14 clean deployment and coding/iteration/cancellation runs later passed without manual IAM changes. The broader failure/recovery, effective IAM and networking matrix remains open. The construct and the packaging script surface that remaining scope at synth/run time (`abca:microvm-image-p1-smoke-unverified`) so an operator cannot mistake a launchable substrate for a verified one. **The `AWS::Lambda::MicrovmImage` L1 enforces the API's enums (P2-F2, live 2026-08-06).** This closes the one item P1 left explicitly open, and it closes it against the construct's own stated reasoning. CloudFormation's generated types make `cpuConfigurations[].architecture` and all four `hooks.*` fields plain strings and document no allowed values, from which P1 concluded that the CloudFormation surface takes a *hook path* while the API takes an `ENABLED`/`DISABLED` flag, and that both were correct for their own surface. CloudFormation refused the change set at **early validation** — the stack was never touched, so there was no rollback and no runtime symptom to trace back — on five values: diff --git a/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md b/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md index 05949bab8..4bf9fad74 100644 --- a/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md +++ b/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md @@ -4,7 +4,7 @@ title: Adr 021 lambda microvms compute backend # ADR-021: AWS Lambda MicroVMs as a third ComputeStrategy backend -> **Implementation status (2026-09-14):** P1 ([#689](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/689)) and P2 ([#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733)) are merged. The takeover branch now has a [clean deployment](/sample-autonomous-cloud-coding-agents/architecture/645-p2-clean-deployment-20260913) with bootstrap 1.7.0 and a [verified managed-image update to 2.0](/sample-autonomous-cloud-coding-agents/architecture/645-microvm-image-rebuild-20260914), plus [live coding, PR iteration and cancellation evidence](/sample-autonomous-cloud-coding-agents/architecture/645-p2-live-task-20260914), without manual IAM workarounds. Remaining P2 failure/recovery and permission/network checks prevent full acceptance. Local P3 foundations include suspend/resume commands, explicit VM observations, guarded persistent intent, a policy helper, approval deadlines that survive a frozen clock, the [guest pause controller](/sample-autonomous-cloud-coding-agents/architecture/645-p3-guest-barrier), and [scoped Claude credentials with retained-client refresh](/sample-autonomous-cloud-coding-agents/architecture/645-p3-credentials). The [production HTTP hooks](/sample-autonomous-cloud-coding-agents/architecture/645-p3-lifecycle-hooks) now connect the guest barrier to atomic checkpoints and credential/gate reconciliation locally. Image capability and supervisor/approval-handler integration remain unfinished; automatic suspension is not enabled. This ADR defines P1–P3, not a P4. See the [current review](/sample-autonomous-cloud-coding-agents/architecture/645-p3-readiness-review) and [P3 implementation plan](/sample-autonomous-cloud-coding-agents/architecture/645-p3-implementation-plan). +> **Implementation status (2026-09-15):** P1 ([#689](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/689)) and P2 ([#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733)) are merged. The takeover branch now has a [clean deployment](/sample-autonomous-cloud-coding-agents/architecture/645-p2-clean-deployment-20260913) with bootstrap 1.7.0 and a [verified managed-image update to 2.0](/sample-autonomous-cloud-coding-agents/architecture/645-microvm-image-rebuild-20260914), plus [live coding, PR iteration and cancellation evidence](/sample-autonomous-cloud-coding-agents/architecture/645-p2-live-task-20260914), without manual IAM workarounds. Remaining P2 failure/recovery and permission/network checks prevent full acceptance. Local P3 foundations include suspend/resume commands, explicit VM observations, guarded persistent intent, a policy helper, approval deadlines that survive a frozen clock, the [guest pause controller](/sample-autonomous-cloud-coding-agents/architecture/645-p3-guest-barrier), and [scoped Claude credentials with retained-client refresh](/sample-autonomous-cloud-coding-agents/architecture/645-p3-credentials). The [production HTTP hooks](/sample-autonomous-cloud-coding-agents/architecture/645-p3-lifecycle-hooks) now connect the guest barrier to atomic checkpoints and credential/gate reconciliation locally. [Per-worker image capability](/sample-autonomous-cloud-coding-agents/architecture/645-p3-image-capability) now declares the six hooks and retains/verifies the actual launched image version locally. Supervisor/approval-handler integration remains unfinished; automatic suspension is not enabled. This ADR defines P1–P3, not a P4. See the [current review](/sample-autonomous-cloud-coding-agents/architecture/645-p3-readiness-review) and [P3 implementation plan](/sample-autonomous-cloud-coding-agents/architecture/645-p3-implementation-plan). **Status:** proposed **Date:** 2026-07-29 @@ -64,7 +64,7 @@ Adopt **AWS Lambda MicroVMs as a third, opt-in `ComputeStrategy` backend** named ### 1. Strategy shape: extend the interface with mandatory suspend/resume -`ComputeType` widens to `'agentcore' | 'ecs' | 'lambda-microvm'` (mirrored in `cli/src/types.ts` and the CLI's inline unions). `SessionHandle` gains a `{ strategyType: 'lambda-microvm', microvmId, endpoint }` variant — `microvmId` because every lifecycle API (`suspend-microvm`, `resume-microvm`, `terminate-microvm`, `get-microvm`, and `create-microvm-auth-token` — the latter not called in P1–P3, see sub-decision 3) takes only the MicroVM identifier, and `endpoint` because it is per-session (minted by `RunMicrovm`) and required for any orchestrator→agent HTTP interaction. Note the naming seam: the handle field is `microvmId` (matching `RunMicrovmResponse.microvmId`), while the request key on every lifecycle command is `microvmIdentifier` — the strategy is the only place that translates between the two. The image ARN is deliberately **not** in the handle: like the ECS task definition ARN, it is deployment-time configuration consumed by `startSession` (from construct-injected environment) and recorded in the session-start log entry for diagnostics, not per-session lifecycle state. +`ComputeType` widens to `'agentcore' | 'ecs' | 'lambda-microvm'` (mirrored in `cli/src/types.ts` and the CLI's inline unions). `SessionHandle` gains a `{ strategyType: 'lambda-microvm', microvmId, endpoint }` variant — `microvmId` because every lifecycle API (`suspend-microvm`, `resume-microvm`, `terminate-microvm`, `get-microvm`, and `create-microvm-auth-token` — the latter not called in P1–P3, see sub-decision 3) takes only the MicroVM identifier, and `endpoint` because it is per-session (minted by `RunMicrovm`) and required for any orchestrator→agent HTTP interaction. Note the naming seam: the handle field is `microvmId` (matching `RunMicrovmResponse.microvmId`), while the request key on every lifecycle command is `microvmIdentifier` — the strategy is the only place that translates between the two. **P3 refinement (2026-09-15):** the handle also retains the actual `imageArn` and `imageVersion` returned by Run, plus `lifecycleProtocol` after exact-version verification. The requested image remains deployment configuration, but the version that launched an existing worker is per-session evidence. Save the known handle before optional discovery; check that exact version with `GetMicrovmImageVersion`, then conditionally persist support in the receipt and `compute_metadata`. Current deployment settings cannot substitute for missing legacy evidence. **A second, sharper naming seam: `imageIdentifier` must be an ARN.** The name suggests a bare image name is acceptable — `create-microvm-image --name` takes one, and this ADR originally assumed `run-microvm --image-identifier` would too. It does not: a bare name is rejected with `ValidationException: Malformed ARN - doesn't start with 'arn:'`, and so is `list-microvm-image-builds --image-identifier ` (`Invalid ARN format`). The construct therefore resolves an operator-supplied name to its exact `arn:${Partition}:lambda:${Region}:${Account}:microvm-image:${Name}` ARN **once** — the same value it scopes the lifecycle IAM grant to — and injects THAT as `MICROVM_IMAGE_IDENTIFIER`. One derivation, two consumers, so a request field and an IAM resource can never disagree. The strategy validates the invariant and fails fast with the remedy, because the service's own error names neither the env var nor the fix. @@ -104,7 +104,7 @@ Normative requirements (EARS, per [ADR-020](/sample-autonomous-cloud-coding-agen - When `startSession` is invoked, the strategy shall call `RunMicrovm` with `maximumDurationInSeconds` set to 28 800 (the service maximum, matching AgentCore's 8-hour session cap and sitting inside the orchestrator's ~8.5 h safety-net poll window). - When `startSession` is invoked, the strategy shall pass a fully-qualified MicroVM image ARN as `imageIdentifier`. - If the configured image identifier is not an ARN, then the strategy shall fail the session start with an error naming the environment variable and the redeploy remedy, before performing any AWS call. -- When `startSession` returns, the orchestrator shall persist the MicroVM handle (`microvmId`, `endpoint`) in the task row's `compute_metadata` (the field `cancel-task.ts` already reads ECS handles from). +- When `startSession` returns, the orchestrator shall persist the MicroVM handle (`microvmId`, `endpoint`, actual `imageArn`/`imageVersion` when returned, and verified `lifecycleProtocol` when supported) in the task row's `compute_metadata` (the field `cancel-task.ts` already reads ECS handles from). - The strategy shall omit `idlePolicy` on every `RunMicrovm` call, in every phase. - The orchestrator shall be the sole initiator of suspension, via `suspendSession`. - When `pollSession` observes MicroVM state `SUSPENDED` or `SUSPENDING`, the strategy shall report `suspended` without interpreting task state. @@ -177,10 +177,10 @@ The phasing is therefore: | `/ready` | **P1** (construct enables `hooks.microvmImageHooks.ready`) | **P1** | MANDATORY, not a quality nicety — see above. A 200 proves uvicorn is bound and `server` imported cleanly (pulling in `pipeline` → `runner` → the policy engine), so a missing policy file fails the BUILD instead of the first task. **Since P2-F5 it also WARMS the snapshot** — the hook's 200 is what the service waits for before capturing the snapshot, making this the only place a warm page can be created, and the 225 MiB `claude` binary was cold in it (see the P2-F5 correction below). A required warm-up failure answers 503, so a snapshot that cannot exec the agent's own CLI fails the image build instead of every task. Still makes ZERO AWS calls, logging included (a `--version` exec is neither an AWS call nor a network call). | | `/run` | **P1** (construct sets `hooks.microvmHooks.run`) | **P1** | The payload-delivery channel. Must be served in P1 because `/ready` forces hooks to exist at all, and a hook-less image cannot accept `runHookPayload`. Since P2 it is also the **platform-configuration** channel (see "Platform configuration delivery" below). | | `/validate` | **P2** (construct sets `hooks.microvmImageHooks.validate`) | **P2** | An **image** (build-time) hook, and a **shallow self-check only**: server alive, hook routes registered, interpreter + contract sanity. It runs under the BUILD role, which deliberately holds no Bedrock / Secrets Manager / DynamoDB grants, so it must make **zero AWS API calls** and must not touch credential resolution — the "deeper warm-up assertions (Bedrock reachability, Memory access, tool availability)" this ADR originally assigned here are **not implementable**: every one of them would `AccessDenied` and fail every image build. They belong to the first task's own error handling. 200 when the checks pass, 503 while still initialising. | -| `/terminate` | **P2** (construct sets `hooks.microvmHooks.terminate`) | **P2** | Best-effort final flush: a final structured log line, then 200 — always, inside the hook budget, even with nothing running. It must **not** write terminal task status (the orchestrator finalizes the task and *then* calls `TerminateMicrovm`, so a terminate hook that wrote a status would race that finalization and could clobber the real outcome) and must not join the pipeline thread. `ProgressWriter` writes synchronously but drops failed events, so this hook cannot guarantee that every progress write succeeded. P3 needs an acknowledged durability barrier for suspension. "Always 200" also covers the BODY: the handler reads the raw request rather than a typed model, because a typed body is validated before the handler runs and would answer 422 to malformed JSON — a reported hook failure on a successful teardown. Safe to declare because `TerminateMicrovm` removes the VM with or without in-guest cooperation. **Correction (P2-F8):** the service sends `microvmId: ""` on this hook, unlike `/run` where it is populated, so an empty id is expected-normal and this hook cannot join the guest record to the control-plane one — `/run`'s accepted line carries that correlation instead. | -| `/suspend`, `/resume` | **P3** | **P3** | Declaring a runtime hook the agent does not serve fails the corresponding lifecycle transition, so each is declared only in the phase that implements it. P1 termination is the orchestrator's `TerminateMicrovm`, which needs no in-guest cooperation. | +| `/terminate` | **P2** (construct sets `hooks.microvmHooks.terminate`) | **P2** | Best-effort final flush: a final structured log line, then 200 — always, inside the hook budget, even with nothing running. It must **not** write terminal task status (the orchestrator finalizes the task and *then* calls `TerminateMicrovm`, so a terminate hook that wrote a status would race that finalization and could clobber the real outcome) and must not join the pipeline thread. `ProgressWriter` writes synchronously but drops failed events, so this hook cannot guarantee that every progress write succeeded. P3 supplies a separate acknowledged checkpoint for suspension. "Always 200" also covers the BODY: the handler reads the raw request rather than a typed model, because a typed body is validated before the handler runs and would answer 422 to malformed JSON — a reported hook failure on a successful teardown. Safe to declare because `TerminateMicrovm` removes the VM with or without in-guest cooperation. **Correction (P2-F8):** the service sends `microvmId: ""` on this hook, unlike `/run` where it is populated, so an empty id is expected-normal and this hook cannot join the guest record to the control-plane one — `/run`'s accepted line carries that correlation instead. | +| `/suspend`, `/resume` | **P3**, implemented locally | **P3**, implemented locally | Managed images declare the served checkpoint/credential/gate hooks with the shared 30-second service timeout and protocol marker. The coordinator verifies the actual launched version before allowing new suspension. Supervisor integration and live acceptance remain open. | -Consequence to state plainly, replacing the original "a P1-built MicroVM image is not runnable end to end": **a P1 image is creatable, launchable and payload-deliverable, but carries no smoke-parity guarantee.** P1 delivers the strategy, the construct, the roles/buckets/connectors, the image resource, the packaging script, and the `/ready` + `/run` endpoints — so a `lambda-microvm` task can start a MicroVM and hand it a payload. What P1 has **not** established is anything P2 owns: AgentCore Memory grants and `MEMORY_ID` delivery, the agent's non-secret env parity inside the snapshot, egress specifics from a running MicroVM, and heartbeat/progress behaviour end to end. At the P1 handoff, no clone → change → PR run had happened. P2 subsequently demonstrated that path with an IAM workaround; clean unattended verification is still pending. The construct and the packaging script both surface exactly this at synth/run time (`abca:microvm-image-p1-smoke-unverified`) so an operator cannot mistake a launchable substrate for a verified one. +Consequence to state plainly, replacing the original "a P1-built MicroVM image is not runnable end to end": **a P1 image is creatable, launchable and payload-deliverable, but carries no smoke-parity guarantee.** P1 delivers the strategy, the construct, the roles/buckets/connectors, the image resource, the packaging script, and the `/ready` + `/run` endpoints — so a `lambda-microvm` task can start a MicroVM and hand it a payload. What P1 has **not** established is anything P2 owns: AgentCore Memory grants and `MEMORY_ID` delivery, the agent's non-secret env parity inside the snapshot, egress specifics from a running MicroVM, and heartbeat/progress behaviour end to end. At the P1 handoff, no clone → change → PR run had happened. P2 initially demonstrated that path with an IAM workaround; the 2026-09-14 clean deployment and coding/iteration/cancellation runs later passed without manual IAM changes. The broader failure/recovery, effective IAM and networking matrix remains open. The construct and the packaging script surface that remaining scope at synth/run time (`abca:microvm-image-p1-smoke-unverified`) so an operator cannot mistake a launchable substrate for a verified one. **The `AWS::Lambda::MicrovmImage` L1 enforces the API's enums (P2-F2, live 2026-08-06).** This closes the one item P1 left explicitly open, and it closes it against the construct's own stated reasoning. CloudFormation's generated types make `cpuConfigurations[].architecture` and all four `hooks.*` fields plain strings and document no allowed values, from which P1 concluded that the CloudFormation surface takes a *hook path* while the API takes an `ENABLED`/`DISABLED` flag, and that both were correct for their own surface. CloudFormation refused the change set at **early validation** — the stack was never touched, so there was no rollback and no runtime symptom to trace back — on five values: diff --git a/docs/verification/645-lifecycle-intent.md b/docs/verification/645-lifecycle-intent.md index 715aa7b93..e8ea15322 100644 --- a/docs/verification/645-lifecycle-intent.md +++ b/docs/verification/645-lifecycle-intent.md @@ -52,7 +52,7 @@ These are local policy choices to validate with live measurements: | Poll during a transition/unconfirmed observation | At most 5 seconds | | Database request/read-sequence budget | 5 seconds; write plus recovery can use two budgets | -A caller must supply valid times, session expiry, its normal poll interval, an enable switch and an explicit compatible-image flag. Both flags are required for a new suspend. Capability must describe the image/version that launched **this VM**, backed by coordinator-owned launch metadata or a verified full drain; current deployment settings alone cannot authenticate an older VM's capability. Unknown/legacy capability keeps new suspends off. Disabling new suspends still permits wake and termination. Long poll intervals are shortened to the next grace/wake/session deadline. +A caller supplies valid times, session expiry, its normal poll interval and an enable switch. New suspension also requires compatible-image evidence from the snapshot's saved worker handle. The [image capability implementation](./645-p3-image-capability.md) verifies the actual launched image ARN/version and persists the supported protocol in coordinator-owned metadata. Both policy and store reject unknown/legacy capability; the suspend transaction rejects image metadata changes after the read. Current deployment settings cannot authenticate an older worker. Disabling new suspends still permits wake and termination. Long poll intervals are shortened to the next grace/wake/session deadline. A PENDING approval alone is not a wake condition. An intended suspended VM can wait while there is sufficient time. APPROVED, DENIED, TIMED_OUT, STRANDED, deadline proximity, missing/invalid data or unintended suspension require wake/recovery. While the service reports SUSPENDING, save desired resume but return `requestReady: false`; issue ResumeMicrovm only after observing SUSPENDED. A wake acknowledgement followed by a delayed old suspend remains repairable because wake intent is retained. diff --git a/docs/verification/645-p3-credentials.md b/docs/verification/645-p3-credentials.md index 9a876460d..6f2f3af7a 100644 --- a/docs/verification/645-p3-credentials.md +++ b/docs/verification/645-p3-credentials.md @@ -1,5 +1,7 @@ # #645 P3: scoped credentials across sleep +**Follow-up (2026-09-15):** [per-worker image capability](./645-p3-image-capability.md) now declares the served hooks and verifies the actual launched image version locally. The milestone below records its original scope; supervisor integration and live sleep/wake acceptance remain open. + Date: 2026-09-14. Local implementation and actual pinned Claude process probes. No AWS deployment or suspension was performed for this milestone. Deployed image **2.0** remains unchanged; P3 is not complete. diff --git a/docs/verification/645-p3-guest-barrier.md b/docs/verification/645-p3-guest-barrier.md index 356ef6f98..01db61e98 100644 --- a/docs/verification/645-p3-guest-barrier.md +++ b/docs/verification/645-p3-guest-barrier.md @@ -1,5 +1,7 @@ # #645 P3 guest pause controller +**Follow-up (2026-09-15):** [per-worker image capability](./645-p3-image-capability.md) now declares the served hooks and verifies the actual launched image version locally. The milestone below records its original scope; supervisor integration and live sleep/wake acceptance remain open. + Date: 2026-09-14. Local implementation; deployed image `2.0` still has only ready, validate, run and terminate hooks. Automatic sleeping remains disabled. diff --git a/docs/verification/645-p3-image-capability.md b/docs/verification/645-p3-image-capability.md new file mode 100644 index 000000000..f289a5c68 --- /dev/null +++ b/docs/verification/645-p3-image-capability.md @@ -0,0 +1,92 @@ +# ADR-021 P3: per-worker image capability + +Date: 2026-09-15. Local implementation on `fix/645-microvm-readiness`, following +the [guest hook milestone](./645-p3-lifecycle-hooks.md). This has not been deployed; +the last verified AWS image is still `2.0`. Automatic suspension remains disabled. + +## What this adds + +An image is the saved starting computer. Its version identifies which saved copy a +worker actually started from. Updating the image later does not update an existing +worker. The supervisor therefore needs the worker's own version, rather than the +deployment's newest image, before deciding whether it can safely sleep. + +Managed images and the out-of-band packaging helper now declare all six served +hooks. Suspend/resume use the shared 30-second service timeout, leaving headroom +above the guest's 20-second handler budget. The image also stores the non-secret +`ABCA_MICROVM_LIFECYCLE_PROTOCOL=1` marker. `/validate` rejects a supplied +incompatible marker without contacting AWS. A missing marker remains acceptable +for ordinary legacy/AgentCore startup, but cannot establish sleep support. + +After `RunMicrovm`, the coordinator: + +1. Saves the known worker ID/endpoint and actual returned `imageArn`/`imageVersion` + before any optional image query. +2. Calls `GetMicrovmImageVersion` for that exact ARN/version, with a three-second + request budget. A foreign or missing image identity cannot borrow deployment + configuration. +3. Checks the returned identity, shared marker and port, all six enabled hooks, + and suspend/resume budgets. +4. Conditionally adds `lifecycleProtocol` to both the saved start handle and + `compute_metadata`, only while both records still identify the same launch. + This preserves task status, including cancellation. + +Failed lookup, unsupported hooks/marker or a rejected capability write keeps the +saved worker available for ordinary tasks and cleanup, with new suspension +disabled. Error logs contain the error type, not returned image environment data +or exception text. Saved handles are replayed without reinterpreting them through +the current deployment configuration. + +Both the policy and the lifecycle intent store require persisted capability for +new suspend requests. The transaction also compares image identity and protocol +with the original read, so a stale decision cannot admit sleep. Resume remains +available for legacy/unknown images that need recovery. + +## AWS contract and permission scope + +The pinned SDK exposes actual `imageArn`/`imageVersion` in `RunMicrovmResponse`. +`GetMicrovmImageVersionOutput` exposes that version's hooks and environment. +`UpdateMicrovmImageVersionRequest` accepts identity plus `status`, not replacement +hooks, environment or code. Changing a version to `INACTIVE` stops new launches; +it does not erase an existing worker's support. + +The [official AWS Lambda service reference](https://servicereference.us-east-1.amazonaws.com/v1/lambda/lambda.json) +classifies `GetMicrovmImageVersion` as a read action on `microvmImage`. +The coordinator adds this action to its existing configured-image ARN scope. +It still has no Suspend/Resume grant or automatic suspension caller at this +milestone. Bootstrap policy files did not change. + +## Verification + +- Capability tests cover exact versus requested/current identity, missing and + conflicting markers/hooks/budgets, failed lookups/writes and saved-handle replay. +- DynamoDB Local passes 39 lifecycle tests, including 16 new cases for conditional + capability enrichment, concurrent cancellation, each changed identity field, + lost committed replies, stale suspend admission and legacy wake recovery. +- Construct tests compare the helper's actual hook/environment JSON with the + synthesized image and reject overrides of the reserved protocol marker. +- Contract mutation tests reject unsafe values and literal redeclarations. +- Guest validation tests cover absent, supported, empty and incompatible markers, + with AWS calls forbidden during image validation. + +Local evidence is under `/tmp/abca-645-p2-clean-20260913/p3-image-*20260915.log`. +The focused image suite passed **444 tests**, the orchestrator composition suite +passed **31**, and agent quality passed **1,940** with 11 opt-in database cases +skipped and **86.28%** branch-inclusive coverage. + +The full root build completed compile, lint, synth, docs, contract checks, +928 CLI tests, 11 Forge tests and agent quality. Its CDK run passed **4,847** +tests and failed three old assertions in two suites: the prior action list and +two expectations that omitted image metadata. Those assertions were updated; +the affected stack grant and all six start-session composition tests then passed. +The full build command itself exited nonzero; the corrections were verified by +focused reruns. There was no production-code failure in that run. + +## Remaining completion gates + +Connect durable supervisor recovery and approval-triggered wake, then deploy and +test the matching coordinator/image together. Verify effective image-read and +lifecycle permissions, actual service hooks, expired credentials, approval and +cancellation races, failures and cleanup in AWS. The +[implementation plan](./645-p3-implementation-plan.md) retains these gates and the +remaining P2 checks. Declared hooks and local tests alone do not complete P3. diff --git a/docs/verification/645-p3-implementation-plan.md b/docs/verification/645-p3-implementation-plan.md index 0f8a79b48..bb3093020 100644 --- a/docs/verification/645-p3-implementation-plan.md +++ b/docs/verification/645-p3-implementation-plan.md @@ -6,15 +6,24 @@ Prepared 2026-09-13 from `main` `5e10038c7e28179b302ac4de78b709795aeba3ce`. Read Prerequisite work is tracked here on `fix/645-microvm-readiness`. “Completed” means implemented and checked locally; AWS deployment and live verification have separate completion gates below. -**Latest local P3 milestone (2026-09-14):** production +**Guest hook milestone (2026-09-14):** production [worker suspend/resume hooks](./645-p3-lifecycle-hooks.md) now connect the guest barrier to atomic checkpoint writes and retained-credential refresh followed by task/gate reconciliation. Duplicate acknowledgments stay within one approval generation; a new gate cannot reuse an old wake result. Shared handler/service -budgets are 20/30 seconds. Image declaration/capability, durable supervisor -integration, approval-triggered wake and live sleep/wake verification remain open. +budgets are 20/30 seconds. At this milestone, image capability and supervisor +integration remained open. No deployment or automatic suspension was enabled in this milestone. +**Latest local P3 milestone (2026-09-15):** [per-worker image capability](./645-p3-image-capability.md) +now declares all six hooks and the shared protocol marker. The coordinator saves +the worker handle first, verifies the exact returned image ARN/version, then +conditionally records support in both start receipt and compute metadata. Missing +or unreadable support permits normal coding and disables new suspension. Database +race tests reject changed identities and recover a committed capability after a +lost reply. Durable supervisor integration, approval-triggered wake and live +sleep/wake verification remain open; no deployment or automatic sleep was enabled. + **Live infrastructure and image deployed (2026-09-14):** the [clean P2 deployment record](./645-p2-clean-deployment-20260913.md) tracks the new Oregon environment, four deployment fixes and actual verification results. @@ -83,7 +92,7 @@ is superseded by these records. - [x] Save gate/VM-bound lifecycle intent with stale-writer protection; add explicit VM observations and a tested policy helper. - [x] Add the guest pause controller, original-gate registration, parallel-tool tracking, progress acknowledgment tracking, heartbeat/read drain and generation-guarded wake completion; reseed the application PRNG at run and controller resume. - [x] Add production agent hooks with acknowledged checkpoints, retained ambient/tenant credential renewal, a sole scoped Claude provider and atomic task/gate reconciliation; verify duplicates, original deadlines, timeout and teardown behavior locally. -- [ ] Declare compatible image hooks using shared budgets and bind lifecycle capability to the actual image/version used by each worker; keep automatic sleep disabled until integration/live acceptance. +- [x] Declare compatible image hooks using shared budgets and bind lifecycle capability to the actual image/version used by each worker; keep automatic sleep disabled until integration/live acceptance. - [ ] Persist bounded poll/recovery counters and connect lifecycle policy to the supervisor. - [ ] Connect supervisor and approval handlers, then verify the complete P3 sleep/wake lifecycle in AWS. @@ -422,7 +431,7 @@ The production HTTP resume callback now invokes this refresh before any AWS read 5. **Implemented locally:** `/run` and successful HTTP/controller resume reseed the application PRNG from fresh OS entropy. Do not seed it with task IDs, timestamps or an image-fixed value. Continue using cryptographic randomness for secrets. Test resumed/sibling snapshot uniqueness where meaningful; do not claim that `random` becomes cryptographically safe. 6. **Implemented locally:** `_ApprovalDeadline` captures the original recorded UTC expiry and a monotonic cap before database writes. Remaining time is `min(monotonic_deadline - monotonic_now, created_at + timeout_s - wall_now)`, clamped at zero, including each sleep bound. Resume verifies the original recorded creation time/timeout and coordinator deadline, then releases the same approval loop with this exact object. Expired wake enters the existing timeout/late-decision path; it never creates a fresh window. 7. **Preserved/tested locally:** conditional TIMED_OUT write, strongly consistent reread when that write loses, and late-decision winner behavior. Forward/backward clocks, frozen monotonic time, slow writes/reads, missing rows and cancellation have regression coverage. **Still required:** exercise these through the actual resume barrier and live AWS lifecycle. TTL is asynchronous garbage collection, not a precise alarm clock. -8. **Shared budgets and served-route checks implemented locally; image declaration still pending.** Declare `/suspend` and `/resume` as enabled image hooks only when the same source version serves them, using the shared 30-second service timeout. `/ready` and `/validate` remain AWS-silent. An old image without the new hooks must not be eligible for automatic suspension. Bind capability to the image/version that **actually launched each VM**, using coordinator-owned launch metadata; current deployment settings alone cannot prove an older VM supports the hooks. Unknown/legacy capability keeps new suspends off. A verified full drain can establish a clean boundary, but must not be assumed. +8. **Implemented locally:** managed images and the packaging helper declare all six served hooks with shared budgets and the source's non-secret protocol marker. `/ready` and `/validate` remain AWS-silent; validation rejects a supplied incompatible marker. The coordinator first saves the known worker handle, then verifies the exact Run-returned image ARN/version and conditionally persists support. Both policy and store require this evidence for new suspend; unknown/legacy workers keep new suspends off. See [image capability verification](./645-p3-image-capability.md). Matching artifact/coordinator deployment and live acceptance remain required. ## 6. Wire the supervisor and human decisions diff --git a/docs/verification/645-p3-lifecycle-hooks.md b/docs/verification/645-p3-lifecycle-hooks.md index 4797bcc2d..aff1f03c5 100644 --- a/docs/verification/645-p3-lifecycle-hooks.md +++ b/docs/verification/645-p3-lifecycle-hooks.md @@ -1,5 +1,7 @@ # #645 P3: worker suspend/resume hooks +**Follow-up (2026-09-15):** [per-worker image capability](./645-p3-image-capability.md) now declares the served hooks and verifies the actual launched image version locally. The milestone below records its original scope; supervisor integration and live sleep/wake acceptance remain open. + Date: 2026-09-14. Local implementation and DynamoDB Local verification. Nothing in this milestone was deployed. Managed image **2.0** and automatic suspension remain unchanged. P3 still requires image capability, supervisor diff --git a/docs/verification/645-p3-readiness-review.md b/docs/verification/645-p3-readiness-review.md index 1b8fa10d5..8dba3934b 100644 --- a/docs/verification/645-p3-readiness-review.md +++ b/docs/verification/645-p3-readiness-review.md @@ -22,7 +22,7 @@ tests with fake AWS responses prove renewal before the next request and safe failure without borrowing the parent's keys. The subsequent [HTTP hook milestone](./645-p3-lifecycle-hooks.md) connects pause to an atomic checkpoint and wake to credential renewal plus task/gate reconciliation. -Image capability, supervisor wiring and real AWS sleep/wake verification remain open. +The [image capability milestone](./645-p3-image-capability.md) (2026-09-15) now declares the six hooks and checks/persists support for the actual launched image version. Supervisor wiring and real AWS sleep/wake verification remain open. The reservation review found a separate trust boundary: the old agent role could write/replace/delete its task row, including coordinator metadata. A subsequent local fix restricts main-task writes to reporting/approval attributes, removes replacement/deletion permissions and removes unused worker counter grants. It also removes unused Python submission/session-info helpers and corrects overstated tenant-isolation comments. [Metadata verification](./645-coordinator-metadata.md) records the tests and pending AWS gate. Status reports still come from the agent, and the compute role chooses session tags; this is not complete hostile-worker isolation. @@ -44,7 +44,7 @@ An **IAM role** is a permission badge. A **trust policy** says who may wear that |---|---|---| | P1 | Build the computer, start it, deliver a task, check it and stop it | Merged in [#689](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/689). Strategy, infrastructure, bootstrap permissions, packaging, types, `/ready` and `/run` exist. | | P2 | Make a real coding task work with configuration, permissions, logs and progress | Merged in [#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733). `/validate`, `/terminate`, warm-up, runtime grants and heartbeat support exist. Two recorded tasks completed and opened PRs on August 7, using a manual IAM workaround. Permanent fixes still need a clean rerun. | -| P3 | Sleep during a human approval wait, wake correctly, and keep deadlines/credentials safe | Local foundations include intent/policy, guest pause control, scoped credential renewal and production HTTP checkpoint/wake hooks. Image capability, supervisor integration and live acceptance remain open. | +| P3 | Sleep during a human approval wait, wake correctly, and keep deadlines/credentials safe | Local foundations include intent/policy, guest pause control, scoped credential renewal, production HTTP checkpoint/wake hooks and per-worker image capability. Supervisor integration and live acceptance remain open. | | P4 | — | ADR-021 defines no P4. Verification runbooks have their own numbered phases; those are not extra ADR milestones. | The old unchecked checklist and the word “proposed” do not erase the merged work. Conversely, merged code is not proof that the final deployment path works unattended. diff --git a/scripts/check-constants-sync.ts b/scripts/check-constants-sync.ts index fe53707c3..bd91caaeb 100644 --- a/scripts/check-constants-sync.ts +++ b/scripts/check-constants-sync.ts @@ -56,7 +56,8 @@ const PYTHON_CONSUMERS = [ POLICY_PY, JIRA_REACTIONS_PY, SERVER_PY, CONFIG_PY, PAYLOAD_BOOTSTRAP_PY, MICROVM_HTTP_PY, ]; const MICROVM_COMPUTE_TS = path.join(REPO_ROOT, 'cdk/src/constructs/lambda-microvm-compute.ts'); -const TS_CONSUMERS = [MICROVM_COMPUTE_TS]; +const MICROVM_IMAGE_CAPABILITY_TS = path.join(REPO_ROOT, 'cdk/src/handlers/shared/microvm-image-capability.ts'); +const TS_CONSUMERS = [MICROVM_COMPUTE_TS, MICROVM_IMAGE_CAPABILITY_TS]; /** Env var names must be UPPER_SNAKE — they are installed into a process env. */ const ENV_NAME_PATTERN = /^[A-Z][A-Z0-9_]*$/; @@ -144,6 +145,22 @@ const OWNED_PYTHON_PATTERNS: ReadonlyArray<{ name: string; regex: RegExp }> = [ * the same way the Python ones are. */ const OWNED_TS_PATTERNS: ReadonlyArray<{ name: string; regex: RegExp }> = [ + { + name: 'LIFECYCLE_HOOK_TIMEOUT_SECONDS', + regex: /^\s*(?:export\s+)?const\s+LIFECYCLE_HOOK_TIMEOUT_SECONDS\s*(?::\s*number)?\s*=\s*-?\d+\b/m, + }, + { + name: 'AGENT_HOOK_PORT', + regex: /^\s*(?:export\s+)?const\s+AGENT_HOOK_PORT\s*(?::\s*number)?\s*=\s*-?\d+\b/m, + }, + { + name: 'MICROVM_LIFECYCLE_PROTOCOL', + regex: /^\s*(?:export\s+)?const\s+MICROVM_LIFECYCLE_PROTOCOL\s*(?::\s*string)?\s*=\s*(?:String\(\s*)?["'\d]/m, + }, + { + name: 'MICROVM_IMAGE_PROTOCOL_ENV', + regex: /^\s*(?:export\s+)?const\s+MICROVM_IMAGE_PROTOCOL_ENV\s*(?::\s*string)?\s*=\s*["']/m, + }, { name: 'READY_HOOK_TIMEOUT_SECONDS', regex: /^\s*(?:export\s+)?const\s+READY_HOOK_TIMEOUT_SECONDS\s*(?::\s*number)?\s*=\s*-?\d+\b/m, @@ -200,6 +217,7 @@ function main(): number { lifecycle_hook_timeout_seconds: number; lifecycle_handler_budget_seconds: number; }; + microvm_lifecycle?: { protocol_version: number; image_protocol_env: string; hook_port: number }; payload_bootstrap?: { version: number; manifest_prefix: string; @@ -358,6 +376,12 @@ function main(): number { // for a runtime failure becomes a build failure. Both halves live here precisely // so the relationship is checkable; this is the check. const mhb = json.microvm_hook_budgets; + const lifecycle = json.microvm_lifecycle; + if (!lifecycle || !Number.isSafeInteger(lifecycle.protocol_version) || lifecycle.protocol_version <= 0 + || !Number.isInteger(lifecycle.hook_port) || lifecycle.hook_port < 1 || lifecycle.hook_port > 65535 + || typeof lifecycle.image_protocol_env !== 'string' || !/^ABCA_MICROVM_[A-Z0-9_]+$/.test(lifecycle.image_protocol_env)) { + invariantErrors.push('microvm_lifecycle requires a positive protocol version, valid hook port and ABCA_MICROVM_ marker name'); + } const BUDGET_FIELDS = [ 'ready_hook_timeout_seconds', 'warmup_total_budget_seconds', From dc98a3eb2663b587d7936065bc95af55b57f7521 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Tue, 15 Sep 2026 13:46:19 -0400 Subject: [PATCH 037/149] feat(microvm): supervise approval suspension and recovery --- agent/src/server.py | 6 +- cdk/bootstrap/BOOTSTRAP_HASH | 2 +- cdk/bootstrap/BOOTSTRAP_VERSION | 2 +- cdk/bootstrap/bootstrap-template.yaml | 10 +- .../policies/compute-lambda-microvm.json | 13 + cdk/package.json | 1 + .../policies/compute-lambda-microvm.ts | 15 + cdk/src/bootstrap/version.ts | 5 +- cdk/src/constructs/task-api.ts | 25 +- cdk/src/constructs/task-orchestrator.ts | 46 +- cdk/src/constructs/task-status.ts | 4 +- cdk/src/handlers/approve-task.ts | 51 +- cdk/src/handlers/deny-task.ts | 47 +- cdk/src/handlers/orchestrate-task.ts | 144 ++--- cdk/src/handlers/shared/agent-heartbeat.ts | 36 ++ cdk/src/handlers/shared/compute-strategy.ts | 28 +- .../handlers/shared/microvm-approval-wake.ts | 155 +++++ cdk/src/handlers/shared/microvm-control.ts | 32 + cdk/src/handlers/shared/microvm-lifecycle.ts | 31 +- cdk/src/handlers/shared/microvm-supervisor.ts | 400 +++++++++++++ .../handlers/shared/microvm-suspend-config.ts | 50 ++ cdk/src/handlers/shared/orchestrator.ts | 336 +++-------- .../shared/strategies/agentcore-strategy.ts | 10 +- .../shared/strategies/ecs-strategy.ts | 12 +- .../strategies/lambda-microvm-strategy.ts | 77 ++- cdk/src/stacks/agent.ts | 11 +- .../__snapshots__/version.test.ts.snap | 2 +- cdk/test/bootstrap/policies.test.ts | 14 +- .../constructs/task-api-microvm-wake.test.ts | 81 +++ cdk/test/constructs/task-orchestrator.test.ts | 56 +- cdk/test/handlers/approve-task.test.ts | 60 ++ cdk/test/handlers/deny-task.test.ts | 60 ++ .../handlers/orchestrate-task-microvm.test.ts | 251 ++++---- cdk/test/handlers/orchestrate-task.test.ts | 75 ++- .../handlers/shared/error-classifier.test.ts | 7 +- .../shared/microvm-approval-wake.test.ts | 211 +++++++ .../shared/microvm-lifecycle-local.test.ts | 107 ++++ .../handlers/shared/microvm-lifecycle.test.ts | 28 + .../shared/microvm-start-recovery.test.ts | 4 +- .../shared/microvm-supervisor.test.ts | 564 ++++++++++++++++++ .../shared/microvm-suspend-config.test.ts | 94 +++ cdk/test/handlers/shared/orchestrator.test.ts | 462 +++----------- .../handlers/shared/session-lifecycle.test.ts | 85 +++ .../lambda-microvm-strategy.test.ts | 7 +- cdk/test/stacks/agent.test.ts | 11 +- ...ADR-021-lambda-microvms-compute-backend.md | 26 +- docs/design/CEDAR_HITL_GATES.md | 4 +- docs/design/COMPUTE.md | 4 +- docs/design/DEPLOYMENT_ROLES.md | 20 + docs/design/ORCHESTRATOR.md | 22 +- .../docs/architecture/Cedar-hitl-gates.md | 4 +- docs/src/content/docs/architecture/Compute.md | 4 +- .../docs/architecture/Deployment-roles.md | 20 + .../content/docs/architecture/Orchestrator.md | 22 +- ...Adr-021-lambda-microvms-compute-backend.md | 26 +- docs/verification/645-lifecycle-intent.md | 8 +- docs/verification/645-p3-image-capability.md | 2 +- .../645-p3-implementation-plan.md | 35 +- docs/verification/645-p3-readiness-review.md | 10 +- docs/verification/645-p3-supervisor.md | 191 ++++++ yarn.lock | 162 +++++ 61 files changed, 3273 insertions(+), 1015 deletions(-) create mode 100644 cdk/src/handlers/shared/agent-heartbeat.ts create mode 100644 cdk/src/handlers/shared/microvm-approval-wake.ts create mode 100644 cdk/src/handlers/shared/microvm-control.ts create mode 100644 cdk/src/handlers/shared/microvm-supervisor.ts create mode 100644 cdk/src/handlers/shared/microvm-suspend-config.ts create mode 100644 cdk/test/constructs/task-api-microvm-wake.test.ts create mode 100644 cdk/test/handlers/shared/microvm-approval-wake.test.ts create mode 100644 cdk/test/handlers/shared/microvm-supervisor.test.ts create mode 100644 cdk/test/handlers/shared/microvm-suspend-config.test.ts create mode 100644 docs/verification/645-p3-supervisor.md diff --git a/agent/src/server.py b/agent/src/server.py index 65da4920d..9077d5a09 100644 --- a/agent/src/server.py +++ b/agent/src/server.py @@ -842,7 +842,7 @@ async def invoke_agent(request: Request, body: InvocationRequest): # -------------------------------------------------------------------------- -# AWS Lambda MicroVMs lifecycle hooks (ADR-021 P1 + P2) +# AWS Lambda MicroVMs lifecycle hooks (ADR-021 P1 through P3) # -------------------------------------------------------------------------- # The MicroVM backend has NO orchestrator→agent HTTP path: the task payload # arrives as the ``/run`` hook's request body and nothing else dials in. The @@ -850,8 +850,8 @@ async def invoke_agent(request: Request, body: InvocationRequest): # (8080 — the same uvicorn process that serves /invocations and /ping), so the # hooks live here rather than in a sidecar. # -# Six hooks are served; image declaration/capability and supervisor wiring for -# ``/suspend`` + ``/resume`` remain a separate P3 rollout step: +# Six hooks are served and declared in managed images. The supervisor checks the +# launched version's capability and rollout settings before requesting suspension: # * ``/ready`` (build, P1) is MANDATORY. ``CreateMicrovmImage`` refuses an image # that enables ANY lifecycle hook without it ("The ready (/ready) MicroVM # image hook must be enabled when any MicroVM lifecycle hook … is enabled"), diff --git a/cdk/bootstrap/BOOTSTRAP_HASH b/cdk/bootstrap/BOOTSTRAP_HASH index eb86d4a5d..49d71c9a7 100644 --- a/cdk/bootstrap/BOOTSTRAP_HASH +++ b/cdk/bootstrap/BOOTSTRAP_HASH @@ -1 +1 @@ -db4602a57169676471a0aa9581fdd4b34826ae1bd642dad7491529a37cfc3429 +0f557449b9b55426b3ae54e28ca76807e2eb74ead7472ace5f035f590b651bd0 diff --git a/cdk/bootstrap/BOOTSTRAP_VERSION b/cdk/bootstrap/BOOTSTRAP_VERSION index bd8bf882d..27f9cd322 100644 --- a/cdk/bootstrap/BOOTSTRAP_VERSION +++ b/cdk/bootstrap/BOOTSTRAP_VERSION @@ -1 +1 @@ -1.7.0 +1.8.0 diff --git a/cdk/bootstrap/bootstrap-template.yaml b/cdk/bootstrap/bootstrap-template.yaml index 4477e703a..265a486bf 100644 --- a/cdk/bootstrap/bootstrap-template.yaml +++ b/cdk/bootstrap/bootstrap-template.yaml @@ -1,7 +1,7 @@ # GENERATED FILE - DO NOT EDIT DIRECTLY # This template is generated by: npx tsx scripts/generate-bootstrap-template.ts -# ABCA Bootstrap Policy Version: 1.7.0 -# ABCA Bootstrap Policy Hash: db4602a57169676471a0aa9581fdd4b34826ae1bd642dad7491529a37cfc3429 +# ABCA Bootstrap Policy Version: 1.8.0 +# ABCA Bootstrap Policy Hash: 0f557449b9b55426b3ae54e28ca76807e2eb74ead7472ace5f035f590b651bd0 # # Based on the default CDK bootstrap template with the following modifications: # - BootstrapVariant set to "ABCA: Least-Privilege Bootstrap" @@ -868,7 +868,7 @@ Resources: ManagedPolicyName: Fn::Sub: cdk-${Qualifier}-IaCRole-ABCA-Compute-LambdaMicrovms-${AWS::AccountId}-${AWS::Region} PolicyDocument: >- - {"Statement":[{"Action":["lambda:CreateMicrovmImage","lambda:GetMicrovmImage","lambda:UpdateMicrovmImage","lambda:DeleteMicrovmImage","lambda:ListMicrovmImages","lambda:GetMicrovmImageVersion","lambda:UpdateMicrovmImageVersion","lambda:DeleteMicrovmImageVersion","lambda:ListMicrovmImageVersions","lambda:GetMicrovmImageBuild","lambda:ListMicrovmImageBuilds","lambda:ListManagedMicrovmImages","lambda:ListManagedMicrovmImageVersions","lambda:CreateNetworkConnector","lambda:GetNetworkConnector","lambda:UpdateNetworkConnector","lambda:DeleteNetworkConnector","lambda:ListNetworkConnectors","lambda:PassNetworkConnector"],"Effect":"Allow","Resource":"*","Sid":"LambdaMicrovms"},{"Action":"iam:PassRole","Effect":"Allow","Resource":["arn:aws:iam::*:role/backgroundagent-dev-LambdaMicrovmComputeBuild*","arn:aws:iam::*:role/backgroundagent-dev-LambdaMicrovmComputeConnector*"],"Sid":"MicrovmPassRoles"}],"Version":"2012-10-17"} + {"Statement":[{"Action":["lambda:CreateMicrovmImage","lambda:GetMicrovmImage","lambda:UpdateMicrovmImage","lambda:DeleteMicrovmImage","lambda:ListMicrovmImages","lambda:GetMicrovmImageVersion","lambda:UpdateMicrovmImageVersion","lambda:DeleteMicrovmImageVersion","lambda:ListMicrovmImageVersions","lambda:GetMicrovmImageBuild","lambda:ListMicrovmImageBuilds","lambda:ListManagedMicrovmImages","lambda:ListManagedMicrovmImageVersions","lambda:CreateNetworkConnector","lambda:GetNetworkConnector","lambda:UpdateNetworkConnector","lambda:DeleteNetworkConnector","lambda:ListNetworkConnectors","lambda:PassNetworkConnector"],"Effect":"Allow","Resource":"*","Sid":"LambdaMicrovms"},{"Action":"iam:PassRole","Effect":"Allow","Resource":["arn:aws:iam::*:role/backgroundagent-dev-LambdaMicrovmComputeBuild*","arn:aws:iam::*:role/backgroundagent-dev-LambdaMicrovmComputeConnector*"],"Sid":"MicrovmPassRoles"},{"Action":["ssm:GetParameters","ssm:PutParameter","ssm:DeleteParameter","ssm:AddTagsToResource","ssm:RemoveTagsFromResource","ssm:ListTagsForResource"],"Effect":"Allow","Resource":"arn:aws:ssm:*:*:parameter/backgroundagent-*/microvm-approval-suspend-enabled","Sid":"MicrovmSuspendConfiguration"}],"Version":"2012-10-17"} Description: 'ABCA Bootstrap: IaCRole-ABCA-Compute-LambdaMicrovms permissions for CloudFormation execution role' Condition: IncludeComputeLambdaMicrovms Outputs: @@ -899,10 +899,10 @@ Outputs: Value: '32' BootstrapPolicyVersion: Description: The version of the ABCA bootstrap policy bundle - Value: 1.7.0 + Value: 1.8.0 BootstrapPolicyHash: Description: SHA-256 hash of the ABCA bootstrap policy bundle for drift detection - Value: db4602a57169676471a0aa9581fdd4b34826ae1bd642dad7491529a37cfc3429 + Value: 0f557449b9b55426b3ae54e28ca76807e2eb74ead7472ace5f035f590b651bd0 BootstrapPolicySet: Description: Comma-separated list of active ABCA bootstrap policy names Value: diff --git a/cdk/bootstrap/policies/compute-lambda-microvm.json b/cdk/bootstrap/policies/compute-lambda-microvm.json index dd7656ad6..3f09b0b0f 100644 --- a/cdk/bootstrap/policies/compute-lambda-microvm.json +++ b/cdk/bootstrap/policies/compute-lambda-microvm.json @@ -34,6 +34,19 @@ "arn:aws:iam::*:role/backgroundagent-dev-LambdaMicrovmComputeConnector*" ], "Sid": "MicrovmPassRoles" + }, + { + "Action": [ + "ssm:GetParameters", + "ssm:PutParameter", + "ssm:DeleteParameter", + "ssm:AddTagsToResource", + "ssm:RemoveTagsFromResource", + "ssm:ListTagsForResource" + ], + "Effect": "Allow", + "Resource": "arn:aws:ssm:*:*:parameter/backgroundagent-*/microvm-approval-suspend-enabled", + "Sid": "MicrovmSuspendConfiguration" } ], "Version": "2012-10-17" diff --git a/cdk/package.json b/cdk/package.json index bccf3df3e..538826403 100644 --- a/cdk/package.json +++ b/cdk/package.json @@ -29,6 +29,7 @@ "@aws-sdk/client-s3": "^3.1078.0", "@aws-sdk/client-secrets-manager": "^3.1078.0", "@aws-sdk/client-sns": "^3.1078.0", + "@aws-sdk/client-ssm": "^3.1078.0", "@aws-sdk/client-sts": "^3.1078.0", "@aws-sdk/credential-provider-node": "^3.972.61", "@aws-sdk/lib-dynamodb": "^3.1078.0", diff --git a/cdk/src/bootstrap/policies/compute-lambda-microvm.ts b/cdk/src/bootstrap/policies/compute-lambda-microvm.ts index 122de1fa4..e3f0f9f9c 100644 --- a/cdk/src/bootstrap/policies/compute-lambda-microvm.ts +++ b/cdk/src/bootstrap/policies/compute-lambda-microvm.ts @@ -157,6 +157,21 @@ export function computeLambdaMicrovmPolicy(): iam.PolicyDocument { 'arn:aws:iam::*:role/backgroundagent-dev-LambdaMicrovmComputeConnector*', ], }), + // A shared value is required because durable executions retain their + // original Lambda version/environment. This grants only deploy-time + // management of the MicroVM switch, outside CDK's bootstrap namespace. + new iam.PolicyStatement({ + sid: 'MicrovmSuspendConfiguration', + actions: [ + 'ssm:GetParameters', + 'ssm:PutParameter', + 'ssm:DeleteParameter', + 'ssm:AddTagsToResource', + 'ssm:RemoveTagsFromResource', + 'ssm:ListTagsForResource', + ], + resources: ['arn:aws:ssm:*:*:parameter/backgroundagent-*/microvm-approval-suspend-enabled'], + }), ], }); } diff --git a/cdk/src/bootstrap/version.ts b/cdk/src/bootstrap/version.ts index 553f49d52..387a458f5 100644 --- a/cdk/src/bootstrap/version.ts +++ b/cdk/src/bootstrap/version.ts @@ -55,8 +55,11 @@ import { allPolicies } from './policies'; * for nested stacks (#645 clean deployment). The policy hash now includes that * inline policy and every nested JSON field; the old root-key replacer omitted * Action/Resource/Condition changes from its input. + * + * 1.7.0 → 1.8.0 adds scoped SSM parameter lifecycle/tag permissions for the P3 + * MicroVM suspension switch. Re-bootstrap before deploying the live parameter. */ -export const BOOTSTRAP_VERSION = '1.7.0'; +export const BOOTSTRAP_VERSION = '1.8.0'; function canonicalize(value: unknown): unknown { if (Array.isArray(value)) return value.map(canonicalize); diff --git a/cdk/src/constructs/task-api.ts b/cdk/src/constructs/task-api.ts index e52cd1b0a..b0959f57d 100644 --- a/cdk/src/constructs/task-api.ts +++ b/cdk/src/constructs/task-api.ts @@ -200,10 +200,8 @@ export interface TaskApiProps { /** * IAM resource ARN of the MicroVM image this deployment provisioned (ADR-021 - * sub-decision 4). When provided, the cancel Lambda gets - * `lambda:TerminateMicrovm` **scoped to that one image**, so a cancelled - * MicroVM-backed task actually stops billing — mirroring the conditional - * AgentCore `RUNTIME_ARN` / `ecsClusterArn` wiring above. + * sub-decision 4). When provided, cancel gets TerminateMicrovm and the approval + * decision handlers get GetMicrovm/ResumeMicrovm, scoped to that one image. * * An ARN rather than an on/off boolean so the grant is exactly scoped. `TaskApi` * is constructed before `LambdaMicrovmCompute` (the cancel Lambda's ARN is @@ -733,15 +731,14 @@ export class TaskApi extends Construct { // ADR-021: cancelling a `lambda-microvm` task must actively terminate the // MicroVM — leaving it to the 8-hour `maximumDurationInSeconds` cap would - // keep billing 16 vCPU and would hold account memory quota that gates - // admission of new tasks. Conditional for the same reason the AgentCore and - // ECS grants above are: a deployment without the backend gets no grant. + // keep running compute or retained snapshots billable. Conditional for the + // same reason the AgentCore and ECS grants above are: a deployment without + // the backend gets no grant. // // ONLY `lambda:TerminateMicrovm`. `cancel-task.ts` sends // `TerminateMicrovmCommand` and nothing else — it does not read MicroVM // state first — so `lambda:GetMicrovm` would be a permission with no caller. - // (The approve/deny Lambdas get `ResumeMicrovm` + `GetMicrovm` in P3, where - // a state read is genuinely needed for the resume reconciliation.) + // The approve/deny Lambdas separately get ResumeMicrovm + GetMicrovm below. // // Resource is the MicroVM *image*, not the running instance: every MicroVM // lifecycle action authorizes against `microvm-image:` (Service @@ -1029,6 +1026,14 @@ export class TaskApi extends Construct { props.taskTable.grantReadWriteData(denyTaskFn); props.taskApprovalsTable.grantReadWriteData(denyTaskFn); props.taskEventsTable.grantReadWriteData(denyTaskFn); + if (props.lambdaMicrovmImageArn) { + for (const decisionFn of [approveTaskFn, denyTaskFn]) { + decisionFn.addToRolePolicy(new iam.PolicyStatement({ + actions: ['lambda:GetMicrovm', 'lambda:ResumeMicrovm'], + resources: [props.lambdaMicrovmImageArn, `${props.lambdaMicrovmImageArn}:*`], + })); + } + } // GetPendingFn — GET /pending const getPendingFn = new lambda.NodejsFunction(this, 'GetPendingFn', { @@ -1428,7 +1433,7 @@ export class TaskApi extends Construct { }, { id: 'AwsSolutions-IAM5', - reason: 'DynamoDB index/* wildcards generated by CDK grantReadWriteData/grantReadData for GSI access; ecs:StopTask is conditioned on the cluster ARN; lambda:TerminateMicrovm is scoped to the single platform MicroVM image ARN plus a :* version-suffix sibling (delivered as a Lazy.string because TaskApi is built before the MicroVM construct) — ADR-021', + reason: 'DynamoDB index/* wildcards generated by CDK grantReadWriteData/grantReadData for GSI access; ecs:StopTask is conditioned on the cluster ARN; Lambda MicroVM cancel/wake actions (TerminateMicrovm/GetMicrovm/ResumeMicrovm) are scoped to the single platform MicroVM image ARN plus a :* version-suffix sibling (delivered as a Lazy.string because TaskApi is built before the MicroVM construct) — ADR-021', }, ], true); } diff --git a/cdk/src/constructs/task-orchestrator.ts b/cdk/src/constructs/task-orchestrator.ts index eeaff9fc9..65466ebc0 100644 --- a/cdk/src/constructs/task-orchestrator.ts +++ b/cdk/src/constructs/task-orchestrator.ts @@ -26,6 +26,7 @@ import { Runtime, Architecture } from 'aws-cdk-lib/aws-lambda'; import * as lambda from 'aws-cdk-lib/aws-lambda-nodejs'; import * as s3 from 'aws-cdk-lib/aws-s3'; import * as secretsmanager from 'aws-cdk-lib/aws-secretsmanager'; +import * as ssm from 'aws-cdk-lib/aws-ssm'; import { NagSuppressions } from 'cdk-nag'; import { Construct } from 'constructs'; @@ -210,11 +211,9 @@ export interface TaskOrchestratorProps { * ## Names, ARNs — and NO grants * * Every field is an identifier, never a secret value, and NONE of them adds an - * IAM grant to the orchestrator role: it forwards these strings and never calls - * the resources they name (the agent does, through its own execution role / - * SessionRole). The approvals and nudges tables in particular stay ungranted to - * the orchestrator, which is asserted by a unit test — a "while I'm here" grant - * would hand the orchestration plane tenant-data access it has never needed. + * IAM grant to the orchestrator role. P3's separate `microvmConfig` grants + * explicit approval reads/condition checks for lifecycle supervision. Forwarding + * these names alone still grants no access to approvals, nudges or tenant roles. * * ## All-or-nothing, and wired unconditionally * @@ -283,8 +282,8 @@ export interface TaskOrchestratorProps { * * `ingressConnectorArns` is required for a different reason — it is a security * control whose absence has a *wider* meaning than "off" (see the field). Only - * `imageVersion` is genuinely optional, and its absent state ("let the service - * resolve the latest ACTIVE version") is a real, intended configuration. + * `imageVersion` may be omitted to resolve the latest ACTIVE version. + * Approval suspension defaults off while wake/cleanup stay available. */ readonly microvmConfig?: { /** @@ -315,6 +314,10 @@ export interface TaskOrchestratorProps { * active version, which is what a rebuild-in-place flow wants. */ readonly imageVersion?: string; + /** Coordinator reads and condition-checks the current gate before sleeping. */ + readonly approvalsTable: dynamodb.ITable; + /** Static opt-in and live Parameter Store value for new suspends; default false. */ + readonly approvalSuspendEnabled?: boolean; /** Role the MicroVM assumes at runtime; passed on `RunMicrovm`. */ readonly executionRoleArn: string; /** Egress network connectors; comma-joined into the env var. */ @@ -386,6 +389,12 @@ export class TaskOrchestrator extends Construct { const handlersDir = path.join(__dirname, '..', 'handlers'); const maxConcurrent = props.maxConcurrentTasksPerUser ?? 10; + const suspendParameter = props.microvmConfig ? new ssm.StringParameter(this, 'MicrovmApprovalSuspendEnabled', { + parameterName: `/${Stack.of(this).stackName}/microvm-approval-suspend-enabled`, + stringValue: String(props.microvmConfig.approvalSuspendEnabled ?? false), + description: 'Allow new approval suspensions; existing durable executions reread before suspending.', + allowedPattern: '^(true|false)$', + }) : undefined; // Hydration pulls in bedrock-agentcore (bundled), durable-execution, and // attachment screening (URL resolution). pdf-parse is needed for PDF text @@ -476,6 +485,9 @@ export class TaskOrchestrator extends Construct { // unconditional; there is no "no ingress configured" state to express. MICROVM_INGRESS_CONNECTOR_ARNS: props.microvmConfig.ingressConnectorArns.join(','), MICROVM_PAYLOAD_BUCKET: props.microvmConfig.payloadBucket.bucketName, + TASK_APPROVALS_TABLE_NAME: props.microvmConfig.approvalsTable.tableName, + MICROVM_APPROVAL_SUSPEND_ENABLED: String(props.microvmConfig.approvalSuspendEnabled ?? false), + MICROVM_APPROVAL_SUSPEND_PARAMETER_NAME: suspendParameter!.parameterName, ...(props.microvmConfig.imageVersion && { MICROVM_IMAGE_VERSION: props.microvmConfig.imageVersion, }), @@ -489,7 +501,7 @@ export class TaskOrchestrator extends Construct { // PLATFORM_CONFIG_ENV_VARS map verbatim — one stack value, one name, three // backends. NO IAM grant accompanies any of these (see the prop docs). ...(props.agentPlatformConfig && { - TASK_APPROVALS_TABLE_NAME: props.agentPlatformConfig.taskApprovalsTableName, + TASK_APPROVALS_TABLE_NAME: props.microvmConfig?.approvalsTable.tableName ?? props.agentPlatformConfig.taskApprovalsTableName, NUDGES_TABLE_NAME: props.agentPlatformConfig.nudgesTableName, LOG_GROUP_NAME: props.agentPlatformConfig.logGroupName, ARTIFACTS_BUCKET_NAME: props.agentPlatformConfig.artifactsBucketName, @@ -616,13 +628,11 @@ export class TaskOrchestrator extends Construct { // GetMicrovm — pollSession // GetMicrovmImageVersion — attest the actual launched snapshot's lifecycle hooks // TerminateMicrovm — stopSession / finalize (the active cleanup path) + // SuspendMicrovm / ResumeMicrovm — durable approval-wait supervision // PassNetworkConnector — required to attach egress connectors, even the // AWS-managed ones // // NOT granted, deliberately: - // - lambda:SuspendMicrovm / lambda:ResumeMicrovm — the ADR's grant list - // names them and strategy methods exist, but no supervisor caller is - // wired yet. Grant them when that integration adds actual calls. // - lambda:CreateMicrovmAuthToken — granted to no role in any phase; no // JWE consumer exists (ADR-021 sub-decision 3). if (props.microvmConfig) { @@ -657,9 +667,21 @@ export class TaskOrchestrator extends Construct { 'lambda:GetMicrovm', 'lambda:GetMicrovmImageVersion', 'lambda:TerminateMicrovm', + 'lambda:SuspendMicrovm', + 'lambda:ResumeMicrovm', ], resources: microvmImageResources, })); + this.fn.addToRolePolicy(new iam.PolicyStatement({ + sid: 'MicrovmApprovalObservation', + actions: ['dynamodb:GetItem', 'dynamodb:ConditionCheckItem'], + resources: [props.microvmConfig.approvalsTable.tableArn], + })); + this.fn.addToRolePolicy(new iam.PolicyStatement({ + sid: 'MicrovmSuspendConfiguration', + actions: ['ssm:GetParameter'], + resources: [suspendParameter!.parameterArn], + })); // `lambda:PassNetworkConnector` supports NO resource-level permissions // (the Service Authorization Reference lists no resource type for it), so @@ -786,7 +808,7 @@ export class TaskOrchestrator extends Construct { }, { id: 'AwsSolutions-IAM5', - reason: 'DynamoDB index/* wildcards generated by CDK grantReadWriteData; AgentCore runtime/* required for sub-resource invocation; Secrets Manager wildcards generated by CDK grantRead; AgentCore Memory wildcards generated by CDK grantRead/grantWrite; ECS RunTask/DescribeTasks/StopTask conditioned on cluster ARN; iam:PassRole scoped to ECS task/execution roles and conditioned on ecs-tasks.amazonaws.com; S3 writes restricted to bootstrap manifests and task payload/launch objects; GetObject and DeleteObject restricted to */payload.json and */launch.json for signing, replay and cleanup; ListBucket is scoped to each payload bucket so absent launch records return NoSuchKey; MicroVM launch/state/cleanup and image-capability actions (RunMicrovm/GetMicrovm/TerminateMicrovm/GetMicrovmImageVersion) are scoped to the single platform MicroVM image ARN plus a :* version-suffix sibling (every one of them authorizes against the image resource, not the per-session instance; no account-wide wildcard is used); lambda:PassNetworkConnector requires Resource:* because the action supports no resource-level permissions and the AWS-managed connectors live outside this account; iam:PassRole is scoped to the exact MicroVM execution role without iam:PassedToService (ADR-021 P2r2-F10); Agent Registry read scoped to the wired registry ARN, with a record/* suffix wildcard because record ids are server-assigned and unknown at synth (#246)', + reason: 'DynamoDB index/* wildcards generated by CDK grantReadWriteData; AgentCore runtime/* required for sub-resource invocation; Secrets Manager wildcards generated by CDK grantRead; AgentCore Memory wildcards generated by CDK grantRead/grantWrite; ECS RunTask/DescribeTasks/StopTask conditioned on cluster ARN; iam:PassRole scoped to ECS task/execution roles and conditioned on ecs-tasks.amazonaws.com; S3 writes restricted to bootstrap manifests and task payload/launch objects; GetObject and DeleteObject restricted to */payload.json and */launch.json for signing, replay and cleanup; ListBucket is scoped to each payload bucket so absent launch records return NoSuchKey; MicroVM launch/state/sleep/wake/cleanup and image-capability actions (RunMicrovm/GetMicrovm/SuspendMicrovm/ResumeMicrovm/TerminateMicrovm/GetMicrovmImageVersion) are scoped to the single platform MicroVM image ARN plus a :* version-suffix sibling (every one of them authorizes against the image resource, not the per-session instance; no account-wide wildcard is used); lambda:PassNetworkConnector requires Resource:* because the action supports no resource-level permissions and the AWS-managed connectors live outside this account; iam:PassRole is scoped to the exact MicroVM execution role without iam:PassedToService (ADR-021 P2r2-F10); Agent Registry read scoped to the wired registry ARN, with a record/* suffix wildcard because record ids are server-assigned and unknown at synth (#246)', }, ], true); } diff --git a/cdk/src/constructs/task-status.ts b/cdk/src/constructs/task-status.ts index cda9ab789..960083691 100644 --- a/cdk/src/constructs/task-status.ts +++ b/cdk/src/constructs/task-status.ts @@ -130,8 +130,8 @@ export const VALID_TRANSITIONS: Readonly { +export async function handler( + event: APIGatewayProxyEvent, context?: Pick, +): Promise { + const invocationStartedMs = Date.now(); const requestId = ulid(); try { @@ -209,11 +214,14 @@ export async function handler(event: APIGatewayProxyEvent): Promise { + options.abortSignal?.throwIfAborted(); + await ddb.send(new PutCommand({ + TableName: EVENTS_TABLE_NAME, + Item: { + task_id: taskId, + user_id: callerUserId, + event_id: ulid(), + event_type: eventType, + timestamp: new Date().toISOString(), + ttl: nowEpoch + AUDIT_EVENT_RETENTION_DAYS * 86400, + metadata, + }, + }), options); + }, + }); + } catch (wakeError) { + logger.warn('MicroVM wake helper failed after decision commit', { + task_id: taskId, request_id, ...microvmErrorIdentity(wakeError), }); } diff --git a/cdk/src/handlers/deny-task.ts b/cdk/src/handlers/deny-task.ts index 6758d9886..27ac48ea1 100644 --- a/cdk/src/handlers/deny-task.ts +++ b/cdk/src/handlers/deny-task.ts @@ -19,11 +19,13 @@ import { TransactionCanceledException } from '@aws-sdk/client-dynamodb'; import { PutCommand, TransactWriteCommand, UpdateCommand } from '@aws-sdk/lib-dynamodb'; -import type { APIGatewayProxyEvent, APIGatewayProxyResult } from 'aws-lambda'; +import type { APIGatewayProxyEvent, APIGatewayProxyResult, Context } from 'aws-lambda'; import { ulid } from 'ulid'; import { scanDenyReason } from './shared/deny-reason-scanner'; import { extractUserId } from './shared/gateway'; import { logger } from './shared/logger'; +import { APPROVAL_AUDIT_TIMEOUT_MS, approvalPostCommitOptions, wakeMicrovmAfterApproval } from './shared/microvm-approval-wake'; +import { microvmErrorIdentity } from './shared/microvm-control'; import { formatMinuteBucket, RATE_LIMIT_ROW_TTL_SECONDS } from './shared/rate-limit'; import { ErrorCode, errorResponse, successResponse } from './shared/response'; import { DENY_REASON_MAX_LENGTH, type DenyRequest, type DenyResponse } from './shared/types'; @@ -62,7 +64,10 @@ const AUDIT_EVENT_RETENTION_DAYS = Number(process.env.TASK_RETENTION_DAYS ?? '90 * @param event - API Gateway proxy event. * @returns API Gateway proxy result. */ -export async function handler(event: APIGatewayProxyEvent): Promise { +export async function handler( + event: APIGatewayProxyEvent, context?: Pick, +): Promise { + const invocationStartedMs = Date.now(); const requestId = ulid(); try { @@ -186,8 +191,11 @@ export async function handler(event: APIGatewayProxyEvent): Promise { + options.abortSignal?.throwIfAborted(); + await ddb.send(new PutCommand({ + TableName: EVENTS_TABLE_NAME, + Item: { + task_id: taskId, + user_id: callerUserId, + event_id: ulid(), + event_type: eventType, + timestamp: new Date().toISOString(), + ttl: nowEpoch + AUDIT_EVENT_RETENTION_DAYS * 86400, + metadata, + }, + }), options); + }, + }); + } catch (wakeError) { + logger.warn('MicroVM wake helper failed after decision commit', { + task_id: taskId, request_id, ...microvmErrorIdentity(wakeError), }); } diff --git a/cdk/src/handlers/orchestrate-task.ts b/cdk/src/handlers/orchestrate-task.ts index a408ac0c6..c73dfb9d0 100644 --- a/cdk/src/handlers/orchestrate-task.ts +++ b/cdk/src/handlers/orchestrate-task.ts @@ -20,10 +20,11 @@ import { withDurableExecution, type DurableExecutionHandler } from '@aws/durable-execution-sdk-js'; import { TaskStatus, TERMINAL_STATUSES } from '../constructs/task-status'; import { resolveComputeStrategy } from './shared/compute-strategy'; -import { MicrovmStartUncertainError } from './shared/error-classifier'; +import { formatMicrovmTerminalFailure, MicrovmStartUncertainError } from './shared/error-classifier'; import { reportIssueFailure as reportJiraIssueFailure } from './shared/jira-feedback'; import { reportIssueFailure } from './shared/linear-feedback'; import { logger } from './shared/logger'; +import { stopMicrovmWithDiagnostics, superviseMicrovm } from './shared/microvm-supervisor'; import { admissionControl, buildComputeMetadata, @@ -36,7 +37,6 @@ import { loadTask, pollTaskStatus, queueTask, - reconcileMicrovmSubstrateState, transitionTask, type PollState, } from './shared/orchestrator'; @@ -375,10 +375,7 @@ const durableHandler: DurableExecutionHandler = asyn ? resolveComputeStrategy(blueprintConfig) : undefined; - // Kept as a SEPARATE local rather than widening `computeStrategy`'s condition: - // the ECS cross-check below is gated on `computeStrategy` truthiness, so - // reusing that local for lambda-microvm would route MicroVM polls through the - // ECS exit-code/patience logic. Two locals keep the ECS path byte-identical. + // ECS crash checks and MicroVM lifecycle supervision have separate policies. const microvmStrategy = blueprintConfig.compute_type === 'lambda-microvm' ? resolveComputeStrategy(blueprintConfig) : undefined; @@ -388,18 +385,40 @@ const durableHandler: DurableExecutionHandler = asyn // While RUNNING, the runtime updates `agent_heartbeat_at`; if that timestamp // goes stale, `pollTaskStatus` sets `sessionUnhealthy` so we fail fast instead // of waiting the full MAX_POLL_ATTEMPTS window (~8.5h) after a silent crash. - // HYDRATING without transition to RUNNING is still bounded by MAX_NON_RUNNING_POLLS (~5min). + // ECS/AgentCore retain their poll-count startup/total bounds. MicroVM uses + // persisted wall-clock deadlines so faster transition polls cannot shorten a session. const finalPollState = await context.waitForCondition( 'await-agent-completion', async (state) => { + if (microvmStrategy && sessionHandle.strategyType === 'lambda-microvm') { + const supervised = await superviseMicrovm({ + taskId, + userId: task.user_id, + handle: sessionHandle, + strategy: microvmStrategy, + previous: state.microvmSupervisor, + pollIntervalMs: blueprintConfig.poll_interval_ms ?? DEFAULT_POLL_INTERVAL_SECONDS * 1000, + suspendEnabled: process.env.MICROVM_APPROVAL_SUSPEND_ENABLED === 'true', + emitEvent: (type, metadata, options) => emitTaskEvent(taskId, type, metadata, correlation, options), + }); + const failure = supervised.kind === 'failure' ? supervised.reason + : supervised.kind === 'substrate-terminal' ? 'substrate-terminal' : undefined; + return { + attempts: state.attempts + 1, + lastStatus: supervised.snapshot?.status ?? state.lastStatus, + sessionUnhealthy: supervised.heartbeatUnhealthy, + microvmSupervisor: supervised.state, + microvmFailureReason: failure, + microvmFailureMessage: supervised.kind === 'substrate-terminal' ? formatMicrovmTerminalFailure( + `substrate state ${supervised.substrate!.status}`, supervised.substrate!.reason, + ) : undefined, + microvmOwnershipLost: supervised.kind === 'ownership-lost', + }; + } const ddbState = await pollTaskStatus(taskId, state, blueprintConfig.compute_type); let consecutiveEcsPollFailures = 0; let consecutiveEcsCompletedPolls = 0; - // Carried forward by default: an unrelated poll (or a MicroVM poll that - // threw) must not silently re-arm the once-per-episode anomaly event. - let microvmSuspendAnomalyReported = state.microvmSuspendAnomalyReported ?? false; - // ECS compute-level crash detection: if DDB is not terminal, check ECS task status if ( ddbState.lastStatus && @@ -448,52 +467,7 @@ const durableHandler: DurableExecutionHandler = asyn } } - // Lambda MicroVMs substrate cross-check (ADR-021 sub-decision 1). Same - // division of labour as the ECS block above — the strategy reports raw - // substrate state and `reconcileMicrovmSubstrateState` interprets it - // against the DDB status — but the rules differ: a `suspended` VM is - // healthy during an approval wait, an anomaly (not a failure) otherwise, - // and a terminal VM with a non-terminal task row is a substrate failure. - if ( - ddbState.lastStatus - && !TERMINAL_STATUSES.includes(ddbState.lastStatus) - && microvmStrategy - && sessionHandle.strategyType === 'lambda-microvm' - ) { - try { - const substrateStatus = await microvmStrategy.pollSession(sessionHandle); - const { taskFailed, suspendAnomalyReported } = await reconcileMicrovmSubstrateState({ - taskId, - ddbStatus: ddbState.lastStatus, - substrate: substrateStatus, - microvmId: sessionHandle.microvmId, - userId: task.user_id, - correlation, - log, - repo: task.repo, - // Threaded so `microvm_suspend_anomaly` is emitted once per anomaly - // EPISODE rather than on every ~30 s poll; a non-anomalous observation - // re-arms it (see reconcileMicrovmSubstrateState). - suspendAnomalyReported: microvmSuspendAnomalyReported, - }); - microvmSuspendAnomalyReported = suspendAnomalyReported; - if (taskFailed) { - return { attempts: ddbState.attempts, lastStatus: TaskStatus.FAILED }; - } - } catch (err) { - // Non-fatal: a GetMicrovm hiccup must not abort the durable poll step. - // The task stays bounded by MAX_POLL_ATTEMPTS (~8.5 h) and, once the - // agent is RUNNING, by its own terminal write. A repeated-failure - // escalation counter (the ECS `MAX_CONSECUTIVE_ECS_POLL_FAILURES` - // analogue) is deliberately deferred — it would add PollState fields - // that P3's suspend policy will need to reshape anyway. - log.warn('MicroVM pollSession check failed (non-fatal)', { - error: err instanceof Error ? err.message : String(err), - }); - } - } - - return { ...ddbState, consecutiveEcsPollFailures, consecutiveEcsCompletedPolls, microvmSuspendAnomalyReported }; + return { ...ddbState, consecutiveEcsPollFailures, consecutiveEcsCompletedPolls }; }, { initialState: { attempts: 0 }, @@ -504,6 +478,13 @@ const durableHandler: DurableExecutionHandler = asyn if (state.sessionUnhealthy) { return { shouldContinue: false }; } + if (state.microvmSupervisor) { + if (state.microvmFailureReason || state.microvmOwnershipLost) return { shouldContinue: false }; + return { + shouldContinue: true, + delay: { seconds: Math.max(1, Math.ceil(state.microvmSupervisor.nextPollInMs / 1000)) }, + }; + } if (state.attempts >= MAX_POLL_ATTEMPTS) { return { shouldContinue: false }; } @@ -523,40 +504,31 @@ const durableHandler: DurableExecutionHandler = asyn // Step 6: Finalize — update terminal status, emit events, release concurrency await context.step('finalize', async () => { - await finalizeTask(taskId, finalPollState, task.user_id); - // The task is terminal — the substrate has long since read its payload, so - // delete the ephemeral S3 payload object now. Best-effort (both deleters - // swallow errors) and a no-op for AgentCore tasks / deployments without a - // payload bucket; the bucket's 1-day lifecycle rule is the backstop if this - // delete or the whole step never runs. - // - // Both payload-carrying backends get this, and the MicroVM one is NOT - // optional polish: its execution role holds `grantRead` on the WHOLE payload - // bucket (the guest must read its object before any tenant identity exists), - // keys are `/payload.json`, and the guest runs untrusted repo code — - // so a TTL-only reaper left every finished task's hydrated prompt readable by - // any concurrently running MicroVM until asynchronous lifecycle deletion. - // Task-scoped payload reads are a separate improvement (#700); deleting - // completed payloads does not isolate other tasks that are still active. + let finalized = false; + try { + if (!finalPollState.microvmOwnershipLost) { + finalized = (await finalizeTask(taskId, finalPollState, task.user_id)) !== false; + } + } finally { + // Even a database finalization failure must not lose cleanup of this handle. + // A replacement worker, if any, is never followed or terminated here. + if (microvmStrategy && sessionHandle.strategyType === 'lambda-microvm') { + await stopMicrovmWithDiagnostics({ + taskId, + handle: sessionHandle, + strategy: microvmStrategy, + emitEvent: (type, metadata, options) => emitTaskEvent(taskId, type, metadata, correlation, options), + }); + } + } + if (!finalized) return; + // Delete task instructions and their saved signed download capability after + // finalization. Shared manifests remain; bucket lifecycle is a backstop. if (blueprintConfig.compute_type === 'ecs') { await deleteEcsPayload(taskId); } else if (blueprintConfig.compute_type === 'lambda-microvm') { await deleteMicrovmPayload(taskId); } - // ADR-021: "When the orchestrator finalizes a `lambda-microvm` task, the - // orchestrator shall call terminate-microvm (termination shall not rely on - // any substrate timeout)." Without this the VM lingers until - // `maximumDurationInSeconds` (8 h) expires — with `idlePolicy` omitted there - // is no tighter substrate bound. A leaked running VM can keep billing until - // that cap; suspended VMs retain snapshot charges. Whether suspended VMs - // consume the account memory quota is still unverified (ADR-021). - // - // `stopSession` is internally best-effort (it swallows and level-differentiates - // every failure), so this cannot fail the finalize step or strand the task in - // a non-terminal state. - if (microvmStrategy && sessionHandle.strategyType === 'lambda-microvm') { - await microvmStrategy.stopSession(sessionHandle); - } }); }; diff --git a/cdk/src/handlers/shared/agent-heartbeat.ts b/cdk/src/handlers/shared/agent-heartbeat.ts new file mode 100644 index 000000000..7e8f4cbae --- /dev/null +++ b/cdk/src/handlers/shared/agent-heartbeat.ts @@ -0,0 +1,36 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +// SPDX-License-Identifier: MIT-0 + +/** Shared by the two runtimes that run server.py's periodic heartbeat worker. */ +export const AGENT_HEARTBEAT_GRACE_SEC = 120; +export const AGENT_HEARTBEAT_STALE_SEC = 240; + +export function evaluateAgentHeartbeat( + startedAtMs: number | undefined, heartbeatAtMs: number | undefined, nowMs: number, +): 'stale' | 'missing' | undefined { + if (startedAtMs === undefined || !Number.isFinite(startedAtMs)) return undefined; + const age = (nowMs - startedAtMs) / 1000; + if (heartbeatAtMs !== undefined) { + return age > AGENT_HEARTBEAT_GRACE_SEC + && (nowMs - heartbeatAtMs) / 1000 > AGENT_HEARTBEAT_STALE_SEC ? 'stale' : undefined; + } + return age > AGENT_HEARTBEAT_GRACE_SEC + AGENT_HEARTBEAT_STALE_SEC ? 'missing' : undefined; +} diff --git a/cdk/src/handlers/shared/compute-strategy.ts b/cdk/src/handlers/shared/compute-strategy.ts index 25208878d..87841a59a 100644 --- a/cdk/src/handlers/shared/compute-strategy.ts +++ b/cdk/src/handlers/shared/compute-strategy.ts @@ -77,10 +77,9 @@ export type SessionHandle = * to choose a stable failure code. Consumers classify that code, so arbitrary * words in the reason cannot change the category or user-facing retry advice. * - * Declared on all four variants for UNIFORMITY, though only ``completed`` and - * ``failed`` are read today (``reconcileMicrovmSubstrateState`` returns early for - * the other two). The wide union is deliberate rather than dead weight: - * ``suspended.reason`` can explain an observation in P3 diagnostics. The policy + * Declared on all four variants for uniform diagnostics. Terminal failure + * formatting consumes ``completed`` and ``failed``; ``suspended.reason`` can + * explain an observation without deciding whether it is healthy. The policy * must distinguish intended suspension through durable orchestrator intent and * task/approval state, not by parsing this service-provided text. Keeping the * field on every variant preserves the same diagnostic shape. @@ -96,8 +95,16 @@ export type SessionStatus = ( ) & { /** Explicit MicroVM observation; coarse `running` also covers pending/unknown. */ readonly microvmState?: MicrovmObservedState; + /** Service observations used to retain the original lifetime across durable replay. */ + readonly microvmStartedAtMs?: number; + readonly microvmMaximumDurationSeconds?: number; }; +/** A caller may impose a shorter total budget across several control operations. */ +export interface SessionControlOptions { + readonly abortSignal?: AbortSignal; +} + /** * `supported: true` means the lifecycle command was acknowledged, not that the * target state has been reached. Callers must observe/reconcile the session. @@ -107,6 +114,11 @@ export type SessionLifecycleResult = | { readonly supported: false } | { readonly supported: true }; +/** Optional evidence from best-effort cleanup. A request is not confirmed teardown. */ +export type SessionStopResult = + | { readonly outcome: 'requested' | 'not-found' } + | { readonly outcome: 'unconfirmed'; readonly error_type: string; readonly aws_request_id?: string }; + export interface ComputeStrategy { readonly type: ComputeType; startSession(input: { @@ -136,10 +148,10 @@ export interface ComputeStrategy { */ readOnly?: boolean; }): Promise; - pollSession(handle: SessionHandle): Promise; - stopSession(handle: SessionHandle): Promise; - suspendSession(handle: SessionHandle): Promise; - resumeSession(handle: SessionHandle): Promise; + pollSession(handle: SessionHandle, options?: SessionControlOptions): Promise; + stopSession(handle: SessionHandle, options?: SessionControlOptions): Promise; + suspendSession(handle: SessionHandle, options?: SessionControlOptions): Promise; + resumeSession(handle: SessionHandle, options?: SessionControlOptions): Promise; } export function resolveComputeStrategy(blueprintConfig: BlueprintConfig): ComputeStrategy { diff --git a/cdk/src/handlers/shared/microvm-approval-wake.ts b/cdk/src/handlers/shared/microvm-approval-wake.ts new file mode 100644 index 000000000..49ec2a432 --- /dev/null +++ b/cdk/src/handlers/shared/microvm-approval-wake.ts @@ -0,0 +1,155 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +// SPDX-License-Identifier: MIT-0 + +import { GetMicrovmCommand, LambdaMicrovmsClient, ResumeMicrovmCommand } from '@aws-sdk/client-lambda-microvms'; +import type { Context } from 'aws-lambda'; +import type { SessionControlOptions } from './compute-strategy'; +import { logger } from './logger'; +import { microvmErrorIdentity } from './microvm-control'; +import { readMicrovmLifecycleSnapshot, saveMicrovmLifecycleIntent, type MicrovmLifecycleSnapshot } from './microvm-lifecycle'; +import { makeClient } from './ua'; +import { TaskStatus } from '../../constructs/task-status'; + +const API_TIMEOUT_MS = 15_000; +const RESPONSE_RESERVE_MS = 1_000; +export const APPROVAL_POST_COMMIT_TIMEOUT_MS = 8_000; +export const APPROVAL_AUDIT_TIMEOUT_MS = 2_000; +let client: LambdaMicrovmsClient | undefined; + +/** Reserve time to send 202 after a committed decision, including slow prior work. */ +export function approvalPostCommitOptions( + invocationStartedMs: number, context?: Pick, +): SessionControlOptions { + const remaining = context?.getRemainingTimeInMillis() ?? API_TIMEOUT_MS - (Date.now() - invocationStartedMs); + const budget = Math.floor(Math.min(APPROVAL_POST_COMMIT_TIMEOUT_MS, remaining - RESPONSE_RESERVE_MS)); + return { + abortSignal: Number.isSafeInteger(budget) && budget > 0 + ? AbortSignal.timeout(budget) : AbortSignal.abort(new Error('Approval post-commit budget exhausted')), + }; +} + +interface ApprovalWakeInput { + readonly taskId: string; + readonly userId: string; + readonly requestId: string; + readonly decision: 'APPROVED' | 'DENIED'; + readonly options: SessionControlOptions; + readonly emitEvent: (eventType: string, metadata: Record, options: SessionControlOptions) => Promise; +} + +function relevant(snapshot: MicrovmLifecycleSnapshot, input: ApprovalWakeInput): boolean { + if (snapshot.status === TaskStatus.AWAITING_APPROVAL) { + return snapshot.requestId === input.requestId + && (snapshot.approval.kind !== 'present' || snapshot.approval.status === input.decision); + } + // The guest may already have consumed this decision while an earlier Suspend + // is still in flight. Fence only this decision's old intent, never a new gate. + return snapshot.status === TaskStatus.RUNNING && snapshot.requestId === null + && snapshot.intent?.request_id === input.requestId; +} + +/** + * Optional latency improvement after the approval transaction commits. The + * durable supervisor remains responsible for retries and observed RUNNING. + * This path cannot suspend, terminate, alter task status or rewrite a decision. + */ +export async function wakeMicrovmAfterApproval(input: ApprovalWakeInput): Promise { + const options = { abortSignal: input.options.abortSignal ?? AbortSignal.timeout(APPROVAL_POST_COMMIT_TIMEOUT_MS) }; + let stage = 'task-read'; + let microvmId: string | undefined; + const report = async (reason: string, error?: unknown) => { + const metadata = { + task_id: input.taskId, + request_id: input.requestId, + ...(microvmId && { microvm_id: microvmId }), + stage, + reason, + ...(error !== undefined && microvmErrorIdentity(error)), + }; + logger.warn('MicroVM approval wake deferred to durable supervisor', metadata); + try { + const abortSignal = AbortSignal.any([options.abortSignal, AbortSignal.timeout(APPROVAL_AUDIT_TIMEOUT_MS)]); + abortSignal.throwIfAborted(); + await input.emitEvent('microvm_resume_orphan', metadata, { abortSignal }); + } catch (auditError) { + logger.warn('MicroVM resume audit failed after decision commit', { ...metadata, ...microvmErrorIdentity(auditError) }); + } + }; + try { + options.abortSignal?.throwIfAborted(); + const snapshot = await readMicrovmLifecycleSnapshot(input.taskId, input.userId, options); + options.abortSignal.throwIfAborted(); + if (!snapshot) return; + microvmId = snapshot.handle.microvmId; + if (!relevant(snapshot, input)) { + await report('task-or-gate-changed'); + return; + } + stage = 'wake-intent'; + const saved = await saveMicrovmLifecycleIntent(snapshot, 'resume', Date.now(), options); + if (saved.status !== 'saved') { + await report(`intent-${saved.status}`); + return; + } + + // Save even if AWS still reports RUNNING/SUSPENDING. A delayed Suspend must + // not strand the worker after the decision handler has returned. + stage = 'substrate-read'; + options.abortSignal?.throwIfAborted(); + client ??= makeClient(LambdaMicrovmsClient); + const observed = await client.send(new GetMicrovmCommand({ microvmIdentifier: microvmId }), options); + options.abortSignal?.throwIfAborted(); + if (observed.state !== 'SUSPENDED') { + if (observed.state !== 'RUNNING' && observed.state !== 'SUSPENDING') await report('state-not-resumable'); + return; + } + + stage = 'pre-resume-read'; + const current = await readMicrovmLifecycleSnapshot(input.taskId, input.userId, options); + options.abortSignal?.throwIfAborted(); + if (!current || current.handle.microvmId !== microvmId || current.handle.endpoint !== snapshot.handle.endpoint + || current.handle.sessionId !== snapshot.handle.sessionId + || current.requestId !== snapshot.requestId || current.status !== snapshot.status + || current.intent?.generation !== saved.intent.generation + || (current.status === TaskStatus.AWAITING_APPROVAL && !relevant(current, input))) { + await report('worker-or-gate-changed-before-resume'); + return; + } + + stage = 'resume-request'; + try { + await client.send(new ResumeMicrovmCommand({ microvmIdentifier: microvmId }), options); + logger.info('MicroVM wake requested after approval decision', { + task_id: input.taskId, request_id: input.requestId, microvm_id: microvmId, + }); + } catch (error) { + await report('resume-request-failed', error); + } finally { + // Check after both acknowledged and uncertain outcomes. No next action + // follows a cancellation, changed gate or ownership loss in this handler. + stage = 'post-resume-read'; + options.abortSignal?.throwIfAborted(); + await readMicrovmLifecycleSnapshot(input.taskId, input.userId, options); + } + } catch (error) { + await report('wake-reconciliation-failed', error); + } +} diff --git a/cdk/src/handlers/shared/microvm-control.ts b/cdk/src/handlers/shared/microvm-control.ts new file mode 100644 index 000000000..a9fa72a83 --- /dev/null +++ b/cdk/src/handlers/shared/microvm-control.ts @@ -0,0 +1,32 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +// SPDX-License-Identifier: MIT-0 + +/** Control-plane diagnostics must not copy SDK messages, payloads or credentials. */ +export function microvmErrorIdentity(error: unknown): { error_type: string; aws_request_id?: string } { + const outer = error as { name?: unknown; cause?: unknown; $metadata?: { requestId?: unknown } } | undefined; + const cause = outer?.cause as typeof outer; + const name = cause?.name ?? outer?.name; + const requestId = cause?.$metadata?.requestId ?? outer?.$metadata?.requestId; + return { + error_type: typeof name === 'string' && /^[A-Za-z0-9_]{1,100}$/.test(name) ? name : 'Error', + ...(typeof requestId === 'string' && /^[A-Za-z0-9-]{1,128}$/.test(requestId) && { aws_request_id: requestId }), + }; +} diff --git a/cdk/src/handlers/shared/microvm-lifecycle.ts b/cdk/src/handlers/shared/microvm-lifecycle.ts index b62d428f4..e34d56e06 100644 --- a/cdk/src/handlers/shared/microvm-lifecycle.ts +++ b/cdk/src/handlers/shared/microvm-lifecycle.ts @@ -19,7 +19,7 @@ import { randomUUID } from 'node:crypto'; import { GetCommand, TransactWriteCommand } from '@aws-sdk/lib-dynamodb'; -import type { SessionHandle } from './compute-strategy'; +import type { SessionControlOptions, SessionHandle } from './compute-strategy'; import { readMicrovmImageMetadata, supportsMicrovmLifecycle } from './microvm-image-capability'; import type { ApprovalStatus } from './types'; import { makeDocClient } from './ua'; @@ -60,6 +60,9 @@ export interface MicrovmLifecycleSnapshot { readonly requestId: string | null; readonly intent?: MicrovmLifecycleIntent; readonly approval: LifecycleApproval; + /** Latest persisted guest heartbeat; used only to bound recovery liveness grace. */ + readonly heartbeatAtMs?: number; + readonly taskStartedAtMs?: number; } export type SaveLifecycleResult = @@ -75,6 +78,13 @@ const APPROVALS_TABLE = process.env.TASK_APPROVALS_TABLE_NAME!; const APPROVAL_STATUSES: readonly ApprovalStatus[] = ['PENDING', 'APPROVED', 'DENIED', 'TIMED_OUT', 'STRANDED']; const LIVE_TASK_STATUSES: readonly TaskStatusType[] = [TaskStatus.HYDRATING, TaskStatus.RUNNING, TaskStatus.AWAITING_APPROVAL]; +function storeSignal(options?: SessionControlOptions): AbortSignal { + const limit = AbortSignal.timeout(MICROVM_LIFECYCLE_STORE_TIMEOUT_MS); + const signal = options?.abortSignal ? AbortSignal.any([limit, options.abortSignal]) : limit; + signal.throwIfAborted(); + return signal; +} + function nonblank(value: unknown): value is string { return typeof value === 'string' && value.trim().length > 0; } @@ -112,9 +122,12 @@ function parseApproval(row: Record | undefined, taskId: string, } /** Missing/non-MicroVM tasks are inapplicable. Invalid identity/state fails visibly. */ -export async function readMicrovmLifecycleSnapshot(taskId: string, userId: string): Promise { - const abortSignal = AbortSignal.timeout(MICROVM_LIFECYCLE_STORE_TIMEOUT_MS); +export async function readMicrovmLifecycleSnapshot( + taskId: string, userId: string, options?: SessionControlOptions, +): Promise { + const abortSignal = storeSignal(options); const result = await ddb.send(new GetCommand({ TableName: TASK_TABLE, Key: { task_id: taskId }, ConsistentRead: true }), { abortSignal }); + abortSignal.throwIfAborted(); const task = result.Item; if (!task) return undefined; if (task.user_id !== userId || task.task_id !== taskId) throw new Error('MicroVM lifecycle task identity mismatch'); @@ -141,6 +154,7 @@ export async function readMicrovmLifecycleSnapshot(taskId: string, userId: strin Key: { task_id: taskId, request_id: requestId }, ConsistentRead: true, }), { abortSignal }); + abortSignal.throwIfAborted(); approval = parseApproval(response.Item, taskId, userId, requestId); } catch (error) { // This explicit observation forbids suspend and permits conservative wake. @@ -156,6 +170,10 @@ export async function readMicrovmLifecycleSnapshot(taskId: string, userId: strin requestId, approval, intent: task.microvm_lifecycle, + ...(typeof task.agent_heartbeat_at === 'string' && Number.isSafeInteger(Date.parse(task.agent_heartbeat_at)) + && { heartbeatAtMs: Date.parse(task.agent_heartbeat_at) }), + ...(typeof task.started_at === 'string' && Number.isSafeInteger(Date.parse(task.started_at)) + && { taskStartedAtMs: Date.parse(task.started_at) }), handle: { strategyType: 'lambda-microvm', sessionId: task.session_id, @@ -188,6 +206,7 @@ function eligible(snapshot: MicrovmLifecycleSnapshot, action: LifecycleAction, n */ export async function saveMicrovmLifecycleIntent( snapshot: MicrovmLifecycleSnapshot, action: LifecycleAction, nowMs = Date.now(), + options?: SessionControlOptions, ): Promise { if (!timestamp(nowMs)) throw new Error('MicroVM lifecycle time must be epoch milliseconds'); if (!eligible(snapshot, action, nowMs)) return { status: 'ineligible' }; @@ -262,7 +281,9 @@ export async function saveMicrovmLifecycleIntent( ], }); try { - await ddb.send(command, { abortSignal: AbortSignal.timeout(MICROVM_LIFECYCLE_STORE_TIMEOUT_MS) }); + const abortSignal = storeSignal(options); + await ddb.send(command, { abortSignal }); + abortSignal.throwIfAborted(); return { status: 'saved', intent }; } catch (error) { const failure = error as { name?: string; CancellationReasons?: { Code?: string }[] }; @@ -270,7 +291,7 @@ export async function saveMicrovmLifecycleIntent( && failure.CancellationReasons?.some(reason => reason.Code === 'ConditionalCheckFailed')) return { status: 'stale' }; // Lost committed reply: observe exactly our generation and unchanged task // identity before reporting success. Unknown/unreadable outcomes stay errors. - const current = await readMicrovmLifecycleSnapshot(snapshot.taskId, snapshot.userId); + const current = await readMicrovmLifecycleSnapshot(snapshot.taskId, snapshot.userId, options); if (current?.intent?.generation === intent.generation) { if (current.status === snapshot.status && current.requestId === snapshot.requestId && current.handle.microvmId === snapshot.handle.microvmId && current.handle.endpoint === snapshot.handle.endpoint diff --git a/cdk/src/handlers/shared/microvm-supervisor.ts b/cdk/src/handlers/shared/microvm-supervisor.ts new file mode 100644 index 000000000..9f1a6758e --- /dev/null +++ b/cdk/src/handlers/shared/microvm-supervisor.ts @@ -0,0 +1,400 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +// SPDX-License-Identifier: MIT-0 + +import { evaluateAgentHeartbeat } from './agent-heartbeat'; +import type { ComputeStrategy, SessionControlOptions, SessionHandle, SessionStatus } from './compute-strategy'; +import { logger } from './logger'; +import { microvmErrorIdentity } from './microvm-control'; +import { + intentMatchesGate, readMicrovmLifecycleSnapshot, saveMicrovmLifecycleIntent, + type MicrovmLifecycleSnapshot, +} from './microvm-lifecycle'; +import { decideMicrovmLifecycle, MICROVM_TRANSITION_POLL_MS } from './microvm-lifecycle-policy'; +import { readMicrovmSuspendEnabled } from './microvm-suspend-config'; +import { MICROVM_MAX_DURATION_SECONDS } from './strategies/lambda-microvm-strategy'; +import { TaskStatus, TERMINAL_STATUSES, type TaskStatusType } from '../../constructs/task-status'; + +type MicrovmHandle = Extract; +export const MICROVM_SUPERVISOR_CYCLE_MS = 45_000; +export const MICROVM_MAX_POLL_FAILURES = 3; +export const MICROVM_RECOVERY_TIMEOUT_MS = 120_000; +export const MICROVM_STARTUP_TIMEOUT_MS = 300_000; +export const MICROVM_CLEANUP_TIMEOUT_MS = 25_000; + +interface Recovery { + readonly kind: 'starting' | 'unconfirmed' | 'suspend' | 'wake'; + readonly sinceMs: number; +} + +/** JSON-only state retained by waitForCondition; a fresh Lambda must reuse it. */ +export interface MicrovmSupervisorState { + readonly version: 1; + readonly microvmId: string; + readonly firstObservedAtMs: number; + readonly sessionDeadlineMs: number; + readonly lifetimeVerified: boolean; + readonly consecutivePollFailures: number; + readonly consecutiveResumeFailures: number; + readonly recovery?: Recovery; + readonly anomalyReported: boolean; + readonly nextPollInMs: number; +} + +export interface MicrovmSupervisorInput { + readonly taskId: string; + readonly userId: string; + readonly handle: MicrovmHandle; + readonly strategy: ComputeStrategy; + readonly previous?: MicrovmSupervisorState; + readonly pollIntervalMs: number; + readonly suspendEnabled: boolean; + /** Implementations must respect the supplied signal; event failure is best-effort. */ + readonly emitEvent?: (eventType: string, metadata: Record, options: SessionControlOptions) => Promise; +} + +type SupervisorOutcome = + | { readonly kind: 'continue' } + | { readonly kind: 'closed'; readonly status: TaskStatusType } + | { readonly kind: 'substrate-terminal' } + | { readonly kind: 'failure' | 'ownership-lost'; readonly reason: string }; + +export type MicrovmSupervisorResult = { + readonly state: MicrovmSupervisorState; + readonly snapshot?: MicrovmLifecycleSnapshot; + readonly substrate?: SessionStatus; + readonly deferHeartbeat: boolean; + readonly heartbeatUnhealthy: boolean; +} & SupervisorOutcome; + +function permanent(error: unknown): boolean { + return ['AccessDeniedException', 'ValidationException', 'UnrecognizedClientException', 'InvalidSignatureException'] + .includes(microvmErrorIdentity(error).error_type); +} + +function closed(status: TaskStatusType): boolean { + return TERMINAL_STATUSES.includes(status) || status === TaskStatus.FINALIZING; +} + +function sameWorker(snapshot: MicrovmLifecycleSnapshot | undefined, input: MicrovmSupervisorInput): snapshot is MicrovmLifecycleSnapshot { + return snapshot?.handle.microvmId === input.handle.microvmId + && snapshot.handle.sessionId === input.handle.sessionId && snapshot.handle.endpoint === input.handle.endpoint; +} + +/** + * One bounded poll cycle. Stores intent before control calls and rechecks after + * every outcome. Task finalization and compute cleanup remain the durable caller's + * responsibility; no task status or capacity reservation is changed here. + */ +export async function superviseMicrovm(input: MicrovmSupervisorInput): Promise { + const now = Date.now(); + if (!Number.isSafeInteger(input.pollIntervalMs) || input.pollIntervalMs <= 0) { + throw new Error('MicroVM supervision requires a positive poll interval'); + } + if (input.previous && (input.previous.version !== 1 || input.previous.microvmId !== input.handle.microvmId)) { + throw new Error('MicroVM supervisor state belongs to another worker or protocol'); + } + const prior = input.previous; + let state: MicrovmSupervisorState = prior ?? { + version: 1, + microvmId: input.handle.microvmId, + firstObservedAtMs: now, + sessionDeadlineMs: now + MICROVM_MAX_DURATION_SECONDS * 1000, + lifetimeVerified: false, + consecutivePollFailures: 0, + consecutiveResumeFailures: 0, + anomalyReported: false, + nextPollInMs: input.pollIntervalMs, + }; + const options: SessionControlOptions = { abortSignal: AbortSignal.timeout(MICROVM_SUPERVISOR_CYCLE_MS) }; + let snapshot: MicrovmLifecycleSnapshot | undefined; + let substrate: SessionStatus | undefined; + let stage = 'task-read'; + let readFailed = false; + let suspendRequested = false; + const result = (outcome: SupervisorOutcome): MicrovmSupervisorResult => { + if (outcome.kind === 'continue') { + if (!readFailed) {state = { ...state, consecutivePollFailures: 0 };} else if (state.consecutivePollFailures >= MICROVM_MAX_POLL_FAILURES) { + outcome = { kind: 'failure', reason: `${stage}-failed-repeatedly` }; + } + } + const deferHeartbeat = snapshot?.status === TaskStatus.AWAITING_APPROVAL || state.recovery !== undefined; + return { + ...outcome, + state, + snapshot, + substrate, + deferHeartbeat, + heartbeatUnhealthy: !deferHeartbeat && snapshot?.status === TaskStatus.RUNNING + && evaluateAgentHeartbeat(snapshot.taskStartedAtMs, snapshot.heartbeatAtMs, Date.now()) !== undefined, + }; + }; + const retry = () => { + state = { ...state, nextPollInMs: Math.min(MICROVM_TRANSITION_POLL_MS, input.pollIntervalMs) }; + return result({ kind: 'continue' }); + }; + const report = async (eventType: string, detail: Record) => { + const metadata = { + task_id: input.taskId, + microvm_id: input.handle.microvmId, + ...(snapshot && { request_id: snapshot.requestId }), + ...detail, + }; + logger.warn(eventType, metadata); + if (input.emitEvent) { + try { await input.emitEvent(eventType, metadata, options); } catch (error) { + logger.warn('MicroVM lifecycle audit failed', { ...metadata, ...microvmErrorIdentity(error) }); + } + } + }; + const fresh = async () => { + options.abortSignal!.throwIfAborted(); + const current = await readMicrovmLifecycleSnapshot(input.taskId, input.userId, options); + options.abortSignal!.throwIfAborted(); + return current; + }; + const beginRecovery = (kind: Recovery['kind'], sinceMs = Date.now()) => { + state = { + ...state, + recovery: state.recovery?.kind === kind ? state.recovery : { kind, sinceMs }, + }; + }; + const recoveryExpired = () => state.recovery !== undefined + && Date.now() - state.recovery.sinceMs >= (state.recovery.kind === 'starting' + ? MICROVM_STARTUP_TIMEOUT_MS : MICROVM_RECOVERY_TIMEOUT_MS); + + try { + snapshot = await fresh(); + if (!sameWorker(snapshot, input)) return result({ kind: 'ownership-lost', reason: 'worker-record-changed-or-missing' }); + if (closed(snapshot.status)) { + if (snapshot.status === TaskStatus.FINALIZING && Date.now() >= state.sessionDeadlineMs) { + return result({ kind: 'failure', reason: 'session-deadline' }); + } + state = { ...state, nextPollInMs: Math.max(1, Math.min(input.pollIntervalMs, state.sessionDeadlineMs - Date.now())) }; + return result({ kind: 'closed', status: snapshot.status }); + } + stage = 'substrate-read'; + substrate = await input.strategy.pollSession(input.handle, options); + options.abortSignal!.throwIfAborted(); + const startedAt = substrate.microvmStartedAtMs; + const maximum = substrate.microvmMaximumDurationSeconds; + if (Number.isSafeInteger(startedAt) && startedAt! >= 0 + && Number.isSafeInteger(maximum) && maximum! > 0) { + const observedDeadline = startedAt! + Math.min(maximum!, MICROVM_MAX_DURATION_SECONDS) * 1000; + if (Number.isSafeInteger(observedDeadline)) { + state = { ...state, lifetimeVerified: true, sessionDeadlineMs: Math.min(state.sessionDeadlineMs, observedDeadline) }; + } + } + const approvalFailure = snapshot!.approval.kind === 'unavailable'; + readFailed = approvalFailure; + state = { + ...state, + consecutivePollFailures: approvalFailure ? state.consecutivePollFailures + 1 : state.consecutivePollFailures, + }; + const observed = substrate.microvmState ?? 'UNKNOWN'; + const wakeRepairWanted = state.recovery?.kind === 'wake' + && (!intentMatchesGate(snapshot!) || snapshot!.intent?.action !== 'resume'); + const awaitingDecisionConsumption = snapshot.status === TaskStatus.AWAITING_APPROVAL + && (snapshot.approval.kind !== 'present' || snapshot.approval.status !== 'PENDING' + || Date.now() >= snapshot.approval.deadlineMs); + if (observed === 'RUNNING') { + // AWS RUNNING does not prove the guest consumed a decided/expired gate. + // Recovery ends after fresh guest liveness, or an intentional early wake + // where the original human decision is still pending. + if (state.recovery?.kind !== 'wake' || (!wakeRepairWanted && ( + (snapshot.status === TaskStatus.RUNNING && (snapshot.heartbeatAtMs ?? -1) >= state.recovery.sinceMs) + || (snapshot.status === TaskStatus.AWAITING_APPROVAL && !awaitingDecisionConsumption) + ))) { + state = { ...state, recovery: undefined, consecutiveResumeFailures: 0 }; + } + } else if (observed === 'PENDING' || observed === 'UNKNOWN') { + // An uncertain observation cannot reset an in-flight wake/suspend clock. + if (!state.recovery) { + beginRecovery(observed === 'PENDING' ? 'starting' : 'unconfirmed', + observed === 'PENDING' ? state.firstObservedAtMs : Date.now()); + } + } + + let suspendEnabled = input.suspendEnabled && state.lifetimeVerified; + const policy = () => decideMicrovmLifecycle({ + snapshot: snapshot!, + substrate: substrate!, + nowMs: Date.now(), + sessionDeadlineMs: state.sessionDeadlineMs, + pollIntervalMs: input.pollIntervalMs, + suspendEnabled, + }); + let decision = policy(); + if ((wakeRepairWanted || (state.recovery?.kind === 'wake' + && (observed === 'SUSPENDING' || observed === 'SUSPENDED'))) + && (decision.action === 'wait' || decision.action === 'suspend') + && (observed === 'RUNNING' || observed === 'SUSPENDING' || observed === 'SUSPENDED')) { + // A previous cycle may have lost the wake-intent write. Its durable recovery + // state still forbids leaving the worker asleep while that write is retried. + decision = { + action: 'resume', + requestReady: observed === 'SUSPENDED', + reason: 'wake-recovery', + nextPollInMs: MICROVM_TRANSITION_POLL_MS, + }; + } + if (decision.action === 'suspend') { + suspendEnabled = await readMicrovmSuspendEnabled(options); + decision = policy(); + } + state = { ...state, nextPollInMs: decision.nextPollInMs }; + if (decision.action === 'reconcile-terminal') return result({ kind: 'substrate-terminal' }); + if (decision.action === 'terminate') return result({ kind: 'failure', reason: decision.reason }); + if (snapshot.status === TaskStatus.HYDRATING && Date.now() - state.firstObservedAtMs >= MICROVM_STARTUP_TIMEOUT_MS) { + return result({ kind: 'failure', reason: 'startup-deadline' }); + } + + const anomaly = (observed === 'SUSPENDING' || observed === 'SUSPENDED') + && state.recovery?.kind !== 'wake' + && !(snapshot.status === TaskStatus.AWAITING_APPROVAL && intentMatchesGate(snapshot) && snapshot.intent?.action === 'resume') + && (snapshot.status !== TaskStatus.AWAITING_APPROVAL || !intentMatchesGate(snapshot)); + if (anomaly && !state.anomalyReported) await report('microvm_suspend_anomaly', { reason: decision.reason }); + state = { ...state, anomalyReported: anomaly }; + + if (decision.action === 'wait') { + if (observed === 'SUSPENDING') beginRecovery('suspend', snapshot!.intent?.requested_at_ms ?? Date.now()); + if (observed === 'SUSPENDED') state = { ...state, recovery: undefined }; + if (recoveryExpired()) return result({ kind: 'failure', reason: 'recovery-deadline' }); + if (approvalFailure && state.consecutivePollFailures >= MICROVM_MAX_POLL_FAILURES) { + return result({ kind: 'failure', reason: 'approval-read-failed-repeatedly' }); + } + return result({ kind: 'continue' }); + } + + stage = `${decision.action}-intent`; + if (decision.action === 'resume' + && (observed === 'SUSPENDING' || observed === 'SUSPENDED' || awaitingDecisionConsumption || wakeRepairWanted)) { + beginRecovery('wake'); + } + const saved = await saveMicrovmLifecycleIntent(snapshot!, decision.action, Date.now(), options); + if (saved.status !== 'saved') return retry(); + snapshot = { ...snapshot!, intent: saved.intent }; + + if (decision.action === 'resume') { + if (observed === 'SUSPENDING' || observed === 'SUSPENDED') beginRecovery('wake'); + if (recoveryExpired()) return result({ kind: 'failure', reason: 'wake-deadline' }); + if (!decision.requestReady) return retry(); + } else { + beginRecovery('suspend', saved.intent.requested_at_ms); + // An uncompleted attempt while still RUNNING is a lost saving opportunity. + // Fence it with wake intent, including a delayed service-side suspension. + if (recoveryExpired()) { + beginRecovery('wake'); + await saveMicrovmLifecycleIntent(snapshot, 'resume', Date.now(), options); + return retry(); + } + } + + // Recheck the live switch after saving intent, then refresh the gate. An + // immutable Lambda environment alone cannot disable an existing execution. + if (decision.action === 'suspend') suspendEnabled = await readMicrovmSuspendEnabled(options); + // Database success is not a lock over the next AWS request. + stage = 'pre-command-read'; + snapshot = await fresh(); + if (!sameWorker(snapshot, input)) return result({ kind: 'ownership-lost', reason: 'worker-record-changed-or-missing' }); + if (closed(snapshot!.status)) return result({ kind: 'closed', status: snapshot!.status }); + if (snapshot!.intent?.generation !== saved.intent.generation || snapshot.requestId !== saved.intent.request_id) return retry(); + if (decision.action === 'suspend' && policy().action !== 'suspend') { + // Approval/deadline/disable can win after intent was saved but before the call. + beginRecovery('wake'); + await saveMicrovmLifecycleIntent(snapshot!, 'resume', Date.now(), options); + return retry(); + } + + stage = `${decision.action}-request`; + suspendRequested = decision.action === 'suspend'; + let commandError: unknown; + try { + const acknowledgement = decision.action === 'suspend' + ? await input.strategy.suspendSession(input.handle, options) + : await input.strategy.resumeSession(input.handle, options); + if (!acknowledgement.supported) commandError = new Error('Lifecycle request is unsupported'); + } catch (error) { commandError = error; } + if (commandError) { + await report(`microvm_${decision.action}_request_failed`, { stage, ...microvmErrorIdentity(commandError) }); + } + + stage = 'post-command-read'; + snapshot = await fresh(); + if (!sameWorker(snapshot, input)) return result({ kind: 'ownership-lost', reason: 'worker-record-changed-or-missing' }); + if (closed(snapshot!.status)) return result({ kind: 'closed', status: snapshot!.status }); + if (decision.action === 'suspend' && (commandError || policy().action !== 'suspend')) { + // Even a failed command can have committed. Retain wake until fresh AWS + // observations confirm recovery; never erase it after an acknowledgment. + beginRecovery('wake'); + await saveMicrovmLifecycleIntent(snapshot!, 'resume', Date.now(), options); + } + if (decision.action === 'resume') { + state = { ...state, consecutiveResumeFailures: commandError ? state.consecutiveResumeFailures + 1 : 0 }; + if (commandError && (permanent(commandError) || state.consecutiveResumeFailures >= MICROVM_MAX_POLL_FAILURES)) { + return result({ kind: 'failure', reason: 'resume-request-failed-repeatedly' }); + } + } + return retry(); + } catch (error) { + // A lost post-command read/write cannot prove the worker stayed awake. + // Persist this recovery obligation even when the wake-intent write failed. + if (suspendRequested) beginRecovery('wake'); + state = { ...state, consecutivePollFailures: state.consecutivePollFailures + (readFailed ? 0 : 1) }; + readFailed = true; + await report('microvm_supervisor_request_failed', { + stage, consecutive_failures: state.consecutivePollFailures, ...microvmErrorIdentity(error), + }); + if (permanent(error) || state.consecutivePollFailures >= MICROVM_MAX_POLL_FAILURES + || recoveryExpired() || Date.now() >= state.sessionDeadlineMs) { + return result({ kind: 'failure', reason: `${stage}-failed` }); + } + return retry(); + } +} + +/** Bounded best-effort cleanup; an unconfirmed outcome stays visible with its handle. */ +export async function stopMicrovmWithDiagnostics( + input: Pick, +): Promise { + const options: SessionControlOptions = { abortSignal: AbortSignal.timeout(MICROVM_CLEANUP_TIMEOUT_MS) }; + let failure = { error_type: 'NoStopEvidence' } as ReturnType; + for (let attempt = 0; attempt < 2; attempt++) { + try { + options.abortSignal!.throwIfAborted(); + const stopped = await input.strategy.stopSession(input.handle, options); + if (stopped && stopped.outcome !== 'unconfirmed') return; + if (stopped?.outcome === 'unconfirmed') failure = stopped; + } catch (error) { failure = microvmErrorIdentity(error); } + if (['AccessDeniedException', 'ValidationException'].includes(failure.error_type)) break; + } + const metadata = { + task_id: input.taskId, + microvm_id: input.handle.microvmId, + error_type: failure.error_type, + ...(failure.aws_request_id && { aws_request_id: failure.aws_request_id }), + }; + logger.error('MicroVM cleanup unconfirmed; retained handle requires recovery', metadata); + try { + await input.emitEvent?.('microvm_cleanup_unconfirmed', metadata, options); + } catch (error) { + logger.warn('MicroVM cleanup audit failed', { ...metadata, ...microvmErrorIdentity(error) }); + } +} diff --git a/cdk/src/handlers/shared/microvm-suspend-config.ts b/cdk/src/handlers/shared/microvm-suspend-config.ts new file mode 100644 index 000000000..bb828817e --- /dev/null +++ b/cdk/src/handlers/shared/microvm-suspend-config.ts @@ -0,0 +1,50 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import { GetParameterCommand, SSMClient } from '@aws-sdk/client-ssm'; +import type { SessionControlOptions } from './compute-strategy'; +import { logger } from './logger'; +import { microvmErrorIdentity } from './microvm-control'; +import { makeClient } from './ua'; + +export const MICROVM_SUSPEND_CONFIG_TIMEOUT_MS = 3_000; +let client: SSMClient | undefined; + +/** + * Durable executions pin their function version, including its environment. + * Read a stable shared parameter without caching so existing executions can + * observe disable. An unavailable setting loses savings, not a healthy worker. + */ +export async function readMicrovmSuspendEnabled(options: SessionControlOptions): Promise { + const name = process.env.MICROVM_APPROVAL_SUSPEND_PARAMETER_NAME; + if (!name) return false; + const localSignal = AbortSignal.timeout(MICROVM_SUSPEND_CONFIG_TIMEOUT_MS); + const abortSignal = options.abortSignal + ? AbortSignal.any([options.abortSignal, localSignal]) : localSignal; + try { + abortSignal.throwIfAborted(); + client ??= makeClient(SSMClient, { maxAttempts: 1 }); + const response = await client.send(new GetParameterCommand({ Name: name }), { abortSignal }); + abortSignal.throwIfAborted(); + return response.Parameter?.Name === name && response.Parameter.Value === 'true'; + } catch (error) { + logger.warn('MicroVM suspension setting unavailable; new suspension disabled', microvmErrorIdentity(error)); + return false; + } +} diff --git a/cdk/src/handlers/shared/orchestrator.ts b/cdk/src/handlers/shared/orchestrator.ts index f41f8804e..e3efe35b3 100644 --- a/cdk/src/handlers/shared/orchestrator.ts +++ b/cdk/src/handlers/shared/orchestrator.ts @@ -20,12 +20,13 @@ import { S3Client } from '@aws-sdk/client-s3'; import { GetCommand, PutCommand, UpdateCommand } from '@aws-sdk/lib-dynamodb'; import { ulid } from 'ulid'; -import type { SessionHandle, SessionStatus } from './compute-strategy'; +import { evaluateAgentHeartbeat } from './agent-heartbeat'; +import type { SessionControlOptions, SessionHandle } from './compute-strategy'; import { AttachmentBudgetExceededError, AttachmentConfigurationError, AttachmentResolutionError, hydrateContext, resolveGitHubToken } from './context-hydration'; -import { formatMicrovmTerminalFailure } from './error-classifier'; import { logger, type Logger } from './logger'; import { writeMinimalEpisode } from './memory'; import { readMicrovmImageMetadata } from './microvm-image-capability'; +import type { MicrovmSupervisorState } from './microvm-supervisor'; import { coerceNumericOrNull } from './numeric'; import { computePromptVersion } from './prompt-version'; import { makeRegistryClient } from './registry/factory'; @@ -64,25 +65,13 @@ export interface PollState { readonly consecutiveEcsPollFailures?: number; /** Consecutive polls where ECS reports completed but DDB is not terminal — escalated after 5. */ readonly consecutiveEcsCompletedPolls?: number; - /** - * True once `microvm_suspend_anomaly` has been emitted for the CURRENT anomaly - * episode, so the event fires once per episode instead of on every ~30 s poll - * (an 8-hour suspended task would otherwise write ~960 identical events). - * - * Re-armed (set back to false) by any non-anomalous observation — see - * {@link reconcileMicrovmSubstrateState}. Kept as a plain boolean rather than a - * counter/timestamp on purpose: P3's suspend policy will reshape this area - * anyway, and one flag is the smallest thing that fixes the duplication without - * pre-committing to a shape that work will have to undo. - */ - readonly microvmSuspendAnomalyReported?: boolean; + readonly microvmSupervisor?: MicrovmSupervisorState; + readonly microvmFailureReason?: string; + readonly microvmFailureMessage?: string; + /** A stale execution must clean up only its own handle, leaving the replacement alone. */ + readonly microvmOwnershipLost?: boolean; } -/** After RUNNING this long, we expect `agent_heartbeat_at` from the agent (if ever set). */ -const AGENT_HEARTBEAT_GRACE_SEC = 120; -/** If `agent_heartbeat_at` exists and is older than this, the session is treated as lost. */ -const AGENT_HEARTBEAT_STALE_SEC = 240; - /** * Whether a backend's liveness is (partly) inferred from `agent_heartbeat_at`. * @@ -209,6 +198,7 @@ export async function transitionTask( fromStatus: TaskStatusType, toStatus: TaskStatusType, extraAttrs?: Record, + expectedMicrovmId?: string, ): Promise { const validTargets = VALID_TRANSITIONS[fromStatus]; if (!validTargets.includes(toStatus)) { @@ -245,13 +235,19 @@ export async function transitionTask( } } + let condition = fromStatus === TaskStatus.SUBMITTED && toStatus === TaskStatus.QUEUED + ? '#status = :fromStatus AND attribute_not_exists(concurrency_slot)' + : '#status = :fromStatus'; + if (expectedMicrovmId) { + condition += ' AND compute_type = :microvmType AND session_id = :microvmId AND compute_metadata.microvmId = :microvmId'; + expressionValues[':microvmType'] = 'lambda-microvm'; + expressionValues[':microvmId'] = expectedMicrovmId; + } await ddb.send(new UpdateCommand({ TableName: TABLE_NAME, Key: { task_id: taskId }, UpdateExpression: updateExpression, - ConditionExpression: fromStatus === TaskStatus.SUBMITTED && toStatus === TaskStatus.QUEUED - ? '#status = :fromStatus AND attribute_not_exists(concurrency_slot)' - : '#status = :fromStatus', + ConditionExpression: condition, ExpressionAttributeNames: expressionNames, ExpressionAttributeValues: expressionValues, })); @@ -296,7 +292,9 @@ export async function emitTaskEvent( eventType: string, metadata?: Record, correlation?: EventCorrelation, + options?: SessionControlOptions, ): Promise { + options?.abortSignal?.throwIfAborted(); await ddb.send(new PutCommand({ TableName: EVENTS_TABLE_NAME, Item: { @@ -309,7 +307,7 @@ export async function emitTaskEvent( ...(correlation?.repo && { repo: correlation.repo }), ...(metadata && { metadata }), }, - })); + }), options); } /** Minimum allowed poll interval (5 seconds). */ @@ -349,182 +347,6 @@ export function buildComputeMetadata(handle: SessionHandle): Record { - const { - taskId, ddbStatus, substrate, microvmId, userId, correlation, log, repo, - suspendAnomalyReported = false, - } = args; - - if (substrate.status === 'running') { - // Healthy — and it also ENDS any anomaly episode, so the next one reports. - return { taskFailed: false, suspendAnomalyReported: false }; - } - - if (substrate.status === 'suspended') { - if (ddbStatus === TaskStatus.AWAITING_APPROVAL) { - // Orchestrator-intended suspend during an approval wait — the whole point - // of this backend. Nothing to report, and the anomaly is re-armed: if the - // task later leaves AWAITING_APPROVAL while still suspended, that is a new - // and genuinely reportable episode. - return { taskFailed: false, suspendAnomalyReported: false }; - } - // Suspended outside an approval wait. Nothing in ABCA suspends a MicroVM - // except the orchestrator's (P3) approval-wait policy, so this means either - // an out-of-band SuspendMicrovm call or a substrate-side suspend we did not - // ask for. Surface it — do NOT fail-fast (ADR-021: "an anomaly to surface, - // not fail-fast"); the VM's state is intact and resumable. - log.warn('MicroVM is suspended while the task is not awaiting approval', { - microvm_id: microvmId, - task_status: ddbStatus, - anomaly_already_reported: suspendAnomalyReported, - }); - if (!suspendAnomalyReported) { - await emitTaskEvent(taskId, 'microvm_suspend_anomaly', { - microvm_id: microvmId, - task_status: ddbStatus, - reason: 'suspended_outside_approval_wait', - }, correlation); - } - return { taskFailed: false, suspendAnomalyReported: true }; - } - - // Terminal substrate report (`completed` or `failed`). `pollSession` reports - // TERMINATING/TERMINATED/NotFound as `completed` because it cannot see an exit - // code; `failed` only reaches here if a future mapping adds one. - // - // `substrate.reason` is `GetMicrovm`'s `stateReason`, carried through verbatim. - // Appending it is what makes this string true on the dominant failure: without - // it a `/run` hook 4xx (which self-terminates the VM in ~12 s) rendered as the - // bare "substrate state completed", and the classifier's remedy then named a - // session duration cap, a host fault, or an external terminate — none of which - // happened. With it the operator gets "substrate state completed (Run lifecycle - // hook returned HTTP status 400…)", which points at the guest logs where the - // agent's own structured 4xx body already is. - const detail = substrate.status === 'failed' - ? substrate.error - : `substrate state ${substrate.status}`; - const failureReason = formatMicrovmTerminalFailure(detail, substrate.reason); - - const reread = await loadTask(taskId); - if (TERMINAL_STATUSES.includes(reread.status)) { - // The agent wrote its terminal status between this cycle's status read and - // now — the normal shutdown ordering. Not a failure. - log.info('MicroVM terminated after the agent wrote a terminal status', { - microvm_id: microvmId, - task_status: reread.status, - }); - // Terminal either way, so the flag no longer matters; carried through - // unchanged rather than reset so the value never lies about what happened. - return { taskFailed: false, suspendAnomalyReported }; - } - - log.error('MicroVM reached a terminal state before the agent wrote a terminal status', { - microvm_id: microvmId, - task_status: reread.status, - detail: failureReason, - }); - // `releaseConcurrency: false` — the finalize step sees the now-terminal task - // and decrements, matching the ECS substrate-failure branch in orchestrate-task. - await failTask( - taskId, - reread.status, - failureReason, - userId, - false, - repo, - ); - return { taskFailed: true, suspendAnomalyReported }; -} - /** * Load blueprint configuration for a task's repository and merge with platform defaults. * @param task - the task record (needs task.repo). @@ -1087,36 +909,24 @@ export async function pollTaskStatus( && item?.session_id && typeof item.started_at === 'string' ) { - const startedMs = Date.parse(item.started_at); const now = Date.now(); - if (!Number.isNaN(startedMs)) { - const runningAgeSec = (now - startedMs) / 1000; - - if (typeof item.agent_heartbeat_at === 'string') { - // Agent has sent at least one heartbeat — check staleness - const hbMs = Date.parse(item.agent_heartbeat_at); - if (!Number.isNaN(hbMs)) { - const hbAgeSec = (now - hbMs) / 1000; - if (runningAgeSec > AGENT_HEARTBEAT_GRACE_SEC && hbAgeSec > AGENT_HEARTBEAT_STALE_SEC) { - sessionUnhealthy = true; - logger.warn('Agent heartbeat stale while task RUNNING', { - task_id: taskId, - compute_type: computeType, - agent_heartbeat_at: item.agent_heartbeat_at, - heartbeat_age_sec: Math.round(hbAgeSec), - }); - } - } - } else if (runningAgeSec > AGENT_HEARTBEAT_GRACE_SEC + AGENT_HEARTBEAT_STALE_SEC) { - // Agent never sent a heartbeat and task has been RUNNING well past - // the grace period — likely early crash before pipeline started. - sessionUnhealthy = true; - logger.warn('Agent never sent heartbeat while task RUNNING past grace period', { - task_id: taskId, - compute_type: computeType, - running_age_sec: Math.round(runningAgeSec), - }); - } + const startedMs = Date.parse(item.started_at); + const heartbeatMs = typeof item.agent_heartbeat_at === 'string' ? Date.parse(item.agent_heartbeat_at) : undefined; + const health = evaluateAgentHeartbeat(startedMs, heartbeatMs, now); + sessionUnhealthy = health !== undefined; + if (health === 'stale') { + logger.warn('Agent heartbeat stale while task RUNNING', { + task_id: taskId, + compute_type: computeType, + agent_heartbeat_at: item.agent_heartbeat_at, + heartbeat_age_sec: Math.round((now - heartbeatMs!) / 1000), + }); + } else if (health === 'missing') { + logger.warn('Agent never sent heartbeat while task RUNNING past grace period', { + task_id: taskId, + compute_type: computeType, + running_age_sec: Math.round((now - startedMs) / 1000), + }); } } @@ -1137,9 +947,17 @@ export async function finalizeTask( taskId: string, pollState: PollState, userId: string, -): Promise { +): Promise { + const current = await loadTask(taskId, true); + const expectedId = pollState.microvmSupervisor?.microvmId; + if (expectedId && (current.user_id !== userId || current.compute_type !== 'lambda-microvm' + || current.session_id !== expectedId || current.compute_metadata?.microvmId !== expectedId)) { + logger.warn('MicroVM finalization skipped after worker ownership changed', { task_id: taskId, microvm_id: expectedId }); + return false; + } try { - await finalizeTaskOutcome(taskId, pollState); + await finalizeTaskOutcome(taskId, pollState, current); + return true; } finally { // The marker makes this safe after a crash, an event failure, or a competing // cleaner. A still-active task keeps its reservation. @@ -1147,28 +965,44 @@ export async function finalizeTask( } } -async function finalizeTaskOutcome(taskId: string, pollState: PollState): Promise { +async function finalizeTaskOutcome(taskId: string, pollState: PollState, task: TaskRecord): Promise { // Finalization can immediately follow a committed start failure/cancellation. // A stale active state would emit the wrong terminal event. - const task = await loadTask(taskId, true); const currentStatus = task.status; // Correlation envelope on this function's own log lines too, not just the // events it emits — admission→terminal logs must join by {user_id, repo}. const { log, correlation } = envelopeFor(task); - // Lost session: RUNNING but agent heartbeats stopped (crash/OOM) — fail fast. - // - // FINALIZING is in the guard DEFENSIVELY, and is currently unreachable: the - // only writer of `sessionUnhealthy` is `pollTaskStatus`, which computes it - // under `currentStatus === TaskStatus.RUNNING`, so a FINALIZING task can never - // arrive here with the flag set. It is kept rather than removed because the - // reachability depends on a predicate in ANOTHER function: the day - // `pollTaskStatus` widens its own status gate (P3's suspend policy already has - // to revisit that block), a heartbeat-stale FINALIZING task must fail rather - // than fall through to the normal terminal path and be reported as a success. - // Dropping the arm would make that a silent behaviour change instead of a - // no-op. Do NOT "simplify" it away without also pinning `pollTaskStatus`'s - // RUNNING-only gate with a test. + if (pollState.microvmFailureReason && !TERMINAL_STATUSES.includes(currentStatus)) { + // Approval/HYDRATING permit FAILED, not TIMED_OUT. This is infrastructure + // failure and must not invent a TIMED_OUT decision on the approval row. + const timeout = pollState.microvmFailureReason === 'session-deadline' + && (currentStatus === TaskStatus.RUNNING || currentStatus === TaskStatus.FINALIZING); + const terminal = timeout ? TaskStatus.TIMED_OUT : TaskStatus.FAILED; + try { + await transitionTask(taskId, currentStatus, terminal, { + completed_at: new Date().toISOString(), + error_message: pollState.microvmFailureMessage ?? `MicroVM supervisor: ${pollState.microvmFailureReason}`, + }, pollState.microvmSupervisor?.microvmId); + } catch (error) { + const winner = await loadTask(taskId, true); + if (!TERMINAL_STATUSES.includes(winner.status)) throw error; + await emitTaskEvent(taskId, `task_${winner.status.toLowerCase()}`, { + final_status: winner.status, poll_attempts: pollState.attempts, + }, correlation); + return; + } + await emitTaskEvent(taskId, timeout ? 'task_timed_out' : 'task_failed', { + reason: 'microvm_supervisor', + detail: pollState.microvmFailureReason, + microvm_id: pollState.microvmSupervisor?.microvmId, + poll_attempts: pollState.attempts, + }, correlation); + return; + } + + // A heartbeat failure is detected while RUNNING. The strong read above may + // already observe FINALIZING; neither active status is a successful outcome. if ( pollState.sessionUnhealthy && (currentStatus === TaskStatus.RUNNING || currentStatus === TaskStatus.FINALIZING) @@ -1180,7 +1014,7 @@ async function finalizeTaskOutcome(taskId: string, pollState: PollState): Promis error_message: 'Agent session lost: no recent heartbeat from the agent ' + `(${substrateNoun(task.compute_type)} may have crashed, been OOM-killed, or stopped)`, - }); + }, pollState.microvmSupervisor?.microvmId); transitioned = true; } catch (err) { // Task may have transitioned concurrently (e.g. agent wrote terminal status). @@ -1277,16 +1111,14 @@ async function finalizeTaskOutcome(taskId: string, pollState: PollState): Promis } // If still RUNNING / FINALIZING / AWAITING_APPROVAL after the poll - // window closes, transition to TIMED_OUT. AWAITING_APPROVAL uses the - // same transition — the stranded-approval reconciler is a secondary - // safety net with a longer timeout for tasks the orchestrator already - // lost track of. + // window closes, terminate the task through an allowed transition. Approval + // waits permit FAILED, not TIMED_OUT; their approval row remains agent-owned. if ( currentStatus === TaskStatus.RUNNING || currentStatus === TaskStatus.FINALIZING || currentStatus === TaskStatus.AWAITING_APPROVAL ) { - const terminalStatus = TaskStatus.TIMED_OUT; + const terminalStatus = currentStatus === TaskStatus.AWAITING_APPROVAL ? TaskStatus.FAILED : TaskStatus.TIMED_OUT; try { await transitionTask(taskId, currentStatus, terminalStatus, { completed_at: new Date().toISOString(), @@ -1304,7 +1136,7 @@ async function finalizeTaskOutcome(taskId: string, pollState: PollState): Promis }, correlation); return; } - await emitTaskEvent(taskId, 'task_timed_out', { + await emitTaskEvent(taskId, terminalStatus === TaskStatus.FAILED ? 'task_failed' : 'task_timed_out', { reason: currentStatus === TaskStatus.AWAITING_APPROVAL ? 'approval_poll_timeout' : 'poll_timeout', diff --git a/cdk/src/handlers/shared/strategies/agentcore-strategy.ts b/cdk/src/handlers/shared/strategies/agentcore-strategy.ts index 3e814958e..c4ef04e50 100644 --- a/cdk/src/handlers/shared/strategies/agentcore-strategy.ts +++ b/cdk/src/handlers/shared/strategies/agentcore-strategy.ts @@ -19,7 +19,7 @@ import { randomUUID } from 'crypto'; import { BedrockAgentCoreClient, InvokeAgentRuntimeCommand, StopRuntimeSessionCommand } from '@aws-sdk/client-bedrock-agentcore'; -import type { ComputeStrategy, SessionHandle, SessionLifecycleResult, SessionStatus } from '../compute-strategy'; +import type { ComputeStrategy, SessionControlOptions, SessionHandle, SessionLifecycleResult, SessionStatus } from '../compute-strategy'; import { logger } from '../logger'; import type { BlueprintConfig } from '../repo-config'; import { makeClient } from '../ua'; @@ -79,7 +79,8 @@ export class AgentCoreComputeStrategy implements ComputeStrategy { }; } - async pollSession(_handle: SessionHandle): Promise { + async pollSession(_handle: SessionHandle, options?: SessionControlOptions): Promise { + options?.abortSignal?.throwIfAborted(); return { status: 'running' }; } @@ -93,17 +94,18 @@ export class AgentCoreComputeStrategy implements ComputeStrategy { return { supported: false }; } - async stopSession(handle: SessionHandle): Promise { + async stopSession(handle: SessionHandle, options?: SessionControlOptions): Promise { if (handle.strategyType !== 'agentcore') { throw new Error('stopSession called with non-agentcore handle'); } const { runtimeArn } = handle; try { + options?.abortSignal?.throwIfAborted(); await getClient().send(new StopRuntimeSessionCommand({ agentRuntimeArn: runtimeArn, runtimeSessionId: handle.sessionId, - })); + }), options); logger.info('AgentCore session stopped', { session_id: handle.sessionId }); } catch (err) { const errName = err instanceof Error ? err.name : undefined; diff --git a/cdk/src/handlers/shared/strategies/ecs-strategy.ts b/cdk/src/handlers/shared/strategies/ecs-strategy.ts index 44689eb14..e03ea5dd9 100644 --- a/cdk/src/handlers/shared/strategies/ecs-strategy.ts +++ b/cdk/src/handlers/shared/strategies/ecs-strategy.ts @@ -18,7 +18,7 @@ */ import { ECSClient, RunTaskCommand, DescribeTasksCommand, StopTaskCommand } from '@aws-sdk/client-ecs'; -import type { ComputeStrategy, SessionHandle, SessionLifecycleResult, SessionStatus } from '../compute-strategy'; +import type { ComputeStrategy, SessionControlOptions, SessionHandle, SessionLifecycleResult, SessionStatus } from '../compute-strategy'; import { logger } from '../logger'; import { deletePayloadReference, preparePayloadReference, redactPayloadUrls } from '../payload-bootstrap'; import type { BlueprintConfig } from '../repo-config'; @@ -250,16 +250,17 @@ export class EcsComputeStrategy implements ComputeStrategy { }; } - async pollSession(handle: SessionHandle): Promise { + async pollSession(handle: SessionHandle, options?: SessionControlOptions): Promise { if (handle.strategyType !== 'ecs') { throw new Error('pollSession called with non-ecs handle'); } const { clusterArn, taskArn } = handle; + options?.abortSignal?.throwIfAborted(); const result = await getClient().send(new DescribeTasksCommand({ cluster: clusterArn, tasks: [taskArn], - })); + }), options); const ecsTask = result.tasks?.[0]; if (!ecsTask) { @@ -286,18 +287,19 @@ export class EcsComputeStrategy implements ComputeStrategy { return { status: 'running' }; } - async stopSession(handle: SessionHandle): Promise { + async stopSession(handle: SessionHandle, options?: SessionControlOptions): Promise { if (handle.strategyType !== 'ecs') { throw new Error('stopSession called with non-ecs handle'); } const { clusterArn, taskArn } = handle; try { + options?.abortSignal?.throwIfAborted(); await getClient().send(new StopTaskCommand({ cluster: clusterArn, task: taskArn, reason: 'Stopped by orchestrator', - })); + }), options); logger.info('ECS task stopped', { task_arn: taskArn }); } catch (err) { const errName = err instanceof Error ? err.name : undefined; diff --git a/cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts b/cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts index 702976fcf..69ef1bbbb 100644 --- a/cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts +++ b/cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts @@ -31,9 +31,10 @@ import { // producer and `agent/src/server.py`'s `/run` consumer. Imported (not copied) so // `tsc` fails on a renamed field — see `contracts/constants.md`. import sharedConstants from '../../../../../contracts/constants.json'; -import type { ComputeStrategy, SessionHandle, SessionLifecycleResult, SessionStatus } from '../compute-strategy'; +import type { ComputeStrategy, SessionControlOptions, SessionHandle, SessionLifecycleResult, SessionStatus, SessionStopResult } from '../compute-strategy'; import { MicrovmStartUncertainError } from '../error-classifier'; import { logger } from '../logger'; +import { microvmErrorIdentity } from '../microvm-control'; import { MICROVM_IMAGE_CAPABILITY_REQUEST_TIMEOUT_MS, MICROVM_LIFECYCLE_PROTOCOL, readMicrovmImageMetadata, verifyMicrovmImageLifecycle, @@ -54,6 +55,13 @@ function getClient(): LambdaMicrovmsClient { /** Bound a control request, not the transition itself. A timeout needs reconciliation. */ export const MICROVM_LIFECYCLE_REQUEST_TIMEOUT_MS = 10_000; +function controlSignal(options?: SessionControlOptions): AbortSignal { + const limit = AbortSignal.timeout(MICROVM_LIFECYCLE_REQUEST_TIMEOUT_MS); + const signal = options?.abortSignal ? AbortSignal.any([limit, options.abortSignal]) : limit; + signal.throwIfAborted(); + return signal; +} + /** * Fully-qualified MicroVM image **ARN** passed as `imageIdentifier` on every * `RunMicrovm`. @@ -368,6 +376,8 @@ function wrapMicrovmError(operation: string, err: unknown): Error { : message; const safeCause = new Error(message); safeCause.name = name ?? 'Error'; + const requestId = microvmErrorIdentity(err).aws_request_id; + if (requestId) Object.assign(safeCause, { $metadata: { requestId } }); return new Error(`${MICROVM_ERROR_MARKER} ${operation} failed: ${detail}`, { cause: safeCause }); } @@ -682,7 +692,7 @@ export class LambdaMicrovmComputeStrategy implements ComputeStrategy { * Report the substrate's view of the session — MECHANICALLY. No task-state * interpretation happens here (ADR-021 sub-decision 1): this method sees only * the handle, so the health rules that need the task's DynamoDB status live in - * the orchestrator (``reconcileMicrovmSubstrateState``). + * the durable supervisor (``superviseMicrovm``). * * State mapping: * - ``PENDING`` / ``RUNNING`` → ``running`` (PENDING is still booting, the @@ -718,7 +728,7 @@ export class LambdaMicrovmComputeStrategy implements ComputeStrategy { * its remedy suggested. Reporting the reason keeps this method mechanical (no * branch reads it) while giving the orchestrator something true to say. */ - async pollSession(handle: SessionHandle): Promise { + async pollSession(handle: SessionHandle, options?: SessionControlOptions): Promise { if (handle.strategyType !== 'lambda-microvm') { throw new Error('pollSession called with non-lambda-microvm handle'); } @@ -726,11 +736,20 @@ export class LambdaMicrovmComputeStrategy implements ComputeStrategy { let state: string | undefined; let stateReason: string | undefined; + let lifetime: Pick = {}; try { const result = await getClient().send(new GetMicrovmCommand({ microvmIdentifier: microvmId, - })); + }), { abortSignal: controlSignal(options) }); state = result.state; + const startedAtMs = result.startedAt instanceof Date ? result.startedAt.getTime() : NaN; + if (Number.isSafeInteger(startedAtMs) && startedAtMs >= 0 + && Number.isSafeInteger(result.maximumDurationInSeconds) && result.maximumDurationInSeconds! > 0) { + lifetime = { + microvmStartedAtMs: startedAtMs, + microvmMaximumDurationSeconds: result.maximumDurationInSeconds, + }; + } // `Success.` is the service's own "nothing to report" value on a clean // termination — carrying it would append noise to every healthy task's // detail string, so it is normalized away here rather than filtered at @@ -773,10 +792,10 @@ export class LambdaMicrovmComputeStrategy implements ComputeStrategy { switch (state) { case MicrovmState.PENDING: case MicrovmState.RUNNING: - return { status: 'running', microvmState: state, ...(stateReason && { reason: stateReason }) }; + return { status: 'running', microvmState: state, ...lifetime, ...(stateReason && { reason: stateReason }) }; case MicrovmState.SUSPENDING: case MicrovmState.SUSPENDED: - return { status: 'suspended', microvmState: state, ...(stateReason && { reason: stateReason }) }; + return { status: 'suspended', microvmState: state, ...lifetime, ...(stateReason && { reason: stateReason }) }; case MicrovmState.TERMINATING: case MicrovmState.TERMINATED: if (stateReason) { @@ -790,14 +809,14 @@ export class LambdaMicrovmComputeStrategy implements ComputeStrategy { state_reason: stateReason, }); } - return { status: 'completed', microvmState: state, ...(stateReason && { reason: stateReason }) }; + return { status: 'completed', microvmState: state, ...lifetime, ...(stateReason && { reason: stateReason }) }; default: logger.warn('Unrecognized MicroVM state — reporting running', { microvm_id: microvmId, state, ...(stateReason && { state_reason: stateReason }), }); - return { status: 'running', microvmState: 'UNKNOWN', ...(stateReason && { reason: stateReason }) }; + return { status: 'running', microvmState: 'UNKNOWN', ...lifetime, ...(stateReason && { reason: stateReason }) }; } } @@ -816,26 +835,27 @@ export class LambdaMicrovmComputeStrategy implements ComputeStrategy { * ``stateReason`` — nothing self-terminates, so nothing cleans up if the * orchestrator does not. */ - async stopSession(handle: SessionHandle): Promise { + async stopSession(handle: SessionHandle, options?: SessionControlOptions): Promise { if (handle.strategyType !== 'lambda-microvm') { throw new Error('stopSession called with non-lambda-microvm handle'); } - await this.terminateBestEffort(handle.microvmId, 'session stop'); + return this.terminateBestEffort(handle.microvmId, 'session stop', options); } /** Submit a suspend request; the caller owns gate checks and state reconciliation. */ - async suspendSession(handle: SessionHandle): Promise { - return this.requestLifecycle('suspendSession', handle); + async suspendSession(handle: SessionHandle, options?: SessionControlOptions): Promise { + return this.requestLifecycle('suspendSession', handle, options); } /** Submit a resume request; acknowledgement alone does not establish RUNNING. */ - async resumeSession(handle: SessionHandle): Promise { - return this.requestLifecycle('resumeSession', handle); + async resumeSession(handle: SessionHandle, options?: SessionControlOptions): Promise { + return this.requestLifecycle('resumeSession', handle, options); } private async requestLifecycle( operation: 'suspendSession' | 'resumeSession', handle: SessionHandle, + options?: SessionControlOptions, ): Promise { if (handle.strategyType !== 'lambda-microvm') { throw new Error(`${operation} called with non-lambda-microvm handle`); @@ -848,7 +868,7 @@ export class LambdaMicrovmComputeStrategy implements ComputeStrategy { try { await getClient().send( suspend ? new SuspendMicrovmCommand(request) : new ResumeMicrovmCommand(request), - { abortSignal: AbortSignal.timeout(MICROVM_LIFECYCLE_REQUEST_TIMEOUT_MS) }, + { abortSignal: controlSignal(options) }, ); } catch (error) { // Includes Conflict/NotFound: neither proves the desired state was reached. @@ -872,40 +892,41 @@ export class LambdaMicrovmComputeStrategy implements ComputeStrategy { * @param reason - why we are terminating, for the log line (the orphan-reap and * the ordinary finalize path are worth telling apart in CloudWatch). */ - private async terminateBestEffort(microvmId: string, reason: string): Promise { + private async terminateBestEffort(microvmId: string, reason: string, options?: SessionControlOptions): Promise { try { await getClient().send(new TerminateMicrovmCommand({ microvmIdentifier: microvmId, - })); - logger.info('Lambda MicroVM terminated', { microvm_id: microvmId, reason }); + }), { abortSignal: controlSignal(options) }); + logger.info('Lambda MicroVM termination requested', { microvm_id: microvmId, reason }); + return { outcome: 'requested' }; } catch (err) { - const errName = err instanceof Error ? err.name : undefined; - if (errName === 'ResourceNotFoundException' || errName === 'ConflictException') { - // Already terminated (reaped) or already TERMINATING — the desired end - // state either way. ConflictException joins the info branch because a - // concurrent terminate (orchestrator finalize racing a user cancel) is - // routine here, and warning on it would train operators to ignore warns. - logger.info('MicroVM already terminated or terminating', { + const identity = microvmErrorIdentity(err); + const errName = identity.error_type; + if (errName === 'ResourceNotFoundException') { + logger.info('MicroVM no longer found during termination', { microvm_id: microvmId, reason, error_type: errName, }); + return { outcome: 'not-found' }; } else if (errName === 'ThrottlingException' || errName === 'AccessDeniedException') { // A throttle or a missing lambda:TerminateMicrovm grant means the VM is // probably STILL RUNNING and billing — escalate. logger.error('Failed to terminate MicroVM', { microvm_id: microvmId, reason, - error_type: errName, - error: redactPayloadUrls(err instanceof Error ? err.message : String(err)), + ...identity, }); } else { logger.warn('Failed to terminate MicroVM (best-effort)', { microvm_id: microvmId, reason, - error: redactPayloadUrls(err instanceof Error ? err.message : String(err)), + ...identity, }); } + // Conflict can mean another lifecycle operation is in flight. It does not + // prove termination; retain that uncertainty for caller orphan reporting. + return { outcome: 'unconfirmed', ...identity }; } } } diff --git a/cdk/src/stacks/agent.ts b/cdk/src/stacks/agent.ts index 40a8120c4..c2db0236e 100644 --- a/cdk/src/stacks/agent.ts +++ b/cdk/src/stacks/agent.ts @@ -346,6 +346,11 @@ export class AgentStack extends Stack { // MicroVM termination grant (ADR-021 sub-decision 4). const computeType = this.node.tryGetContext('compute_type') ?? 'agentcore'; const lambdaMicrovmEnabled = computeType === 'lambda-microvm'; + const suspendContext = this.node.tryGetContext('microvm_approval_suspend_enabled'); + if (suspendContext !== undefined && ![true, false, 'true', 'false'].includes(suspendContext)) { + throw new Error('microvm_approval_suspend_enabled must be true or false'); + } + const microvmApprovalSuspendEnabled = suspendContext === true || suspendContext === 'true'; // --- Tool-federation Gateway deploy gate (ADR-019 P1) --- // Whether to provision the AgentCore Gateway that federates the agent's MCP @@ -408,7 +413,7 @@ export class AgentStack extends Stack { && isLambdaMicrovmImageConfigured(microvmImageInputs); // MicroVM image ARN placeholder — the image is created AFTER TaskApi, but the - // cancel Lambda's grant must be scoped to it. Same Lazy.string cycle-break as + // cancel and decision-handler grants must be scoped to it. Same Lazy.string cycle-break as // the runtime / orchestrator / SessionRole ARNs below. let microvmImageArnHolder: string | undefined; const lazyMicrovmImageArn = Lazy.string({ @@ -1121,7 +1126,7 @@ export class AgentStack extends Stack { }) : undefined; - // Resolve the Lazy TaskApi's cancel grant is scoped by. The invariant the + // Resolve the image ARN used by TaskApi's cancel and wake grants. The invariant the // Lazy's `produce` guards: `microvmImageConfigured` (computed from the same // inputs, via the same predicate) is true exactly when the construct sets // `imageArn`, so a configured deployment always has an ARN to resolve and an @@ -1285,6 +1290,8 @@ export class AgentStack extends Stack { imageIdentifier: lambdaMicrovm.imageIdentifier, imageArn: lambdaMicrovm.imageArn, imageVersion: lambdaMicrovm.imageVersion, + approvalsTable: taskApprovalsTable.table, + approvalSuspendEnabled: microvmApprovalSuspendEnabled, executionRoleArn: lambdaMicrovm.executionRole.roleArn, egressConnectorArns: lambdaMicrovm.egressConnectorArns, // Explicit NO_INGRESS, not an omission: RunMicrovm attaches a PUBLIC diff --git a/cdk/test/bootstrap/__snapshots__/version.test.ts.snap b/cdk/test/bootstrap/__snapshots__/version.test.ts.snap index acd755005..bb6744d54 100644 --- a/cdk/test/bootstrap/__snapshots__/version.test.ts.snap +++ b/cdk/test/bootstrap/__snapshots__/version.test.ts.snap @@ -1,3 +1,3 @@ // Jest Snapshot v1, https://jestjs.io/docs/snapshot-testing -exports[`bootstrap version module hash is stable 1`] = `"db4602a57169676471a0aa9581fdd4b34826ae1bd642dad7491529a37cfc3429"`; +exports[`bootstrap version module hash is stable 1`] = `"0f557449b9b55426b3ae54e28ca76807e2eb74ead7472ace5f035f590b651bd0"`; diff --git a/cdk/test/bootstrap/policies.test.ts b/cdk/test/bootstrap/policies.test.ts index 3ea8184b0..2abf0558e 100644 --- a/cdk/test/bootstrap/policies.test.ts +++ b/cdk/test/bootstrap/policies.test.ts @@ -500,7 +500,7 @@ describe('computeLambdaMicrovmPolicy', () => { const resolvedDoc = stack.resolve(doc); const statements = resolvedDoc.Statement as Array<{ Sid: string }>; - expect(statements.map((s) => s.Sid)).toEqual(['LambdaMicrovms', 'MicrovmPassRoles']); + expect(statements.map((s) => s.Sid)).toEqual(['LambdaMicrovms', 'MicrovmPassRoles', 'MicrovmSuspendConfiguration']); }); it('covers the expected service prefixes', () => { @@ -511,8 +511,16 @@ describe('computeLambdaMicrovmPolicy', () => { ); const prefixes = new Set(allActions.map((a) => a.split(':')[0])); - // `iam` joins `lambda` as of the MicrovmPassRoles statement (ADR-021 P2r2-F9). - expect(prefixes).toEqual(new Set(['lambda', 'iam'])); + expect(prefixes).toEqual(new Set(['lambda', 'iam', 'ssm'])); + }); + + it('limits parameter lifecycle and tagging to the ABCA MicroVM suspension switch', () => { + const statement = stack.resolve(doc).Statement.find((s: { Sid: string }) => s.Sid === 'MicrovmSuspendConfiguration'); + expect(statement.Resource).toBe('arn:aws:ssm:*:*:parameter/backgroundagent-*/microvm-approval-suspend-enabled'); + expect(statement.Action).toEqual([ + 'ssm:GetParameters', 'ssm:PutParameter', 'ssm:DeleteParameter', + 'ssm:AddTagsToResource', 'ssm:RemoveTagsFromResource', 'ssm:ListTagsForResource', + ]); }); describe('MicrovmPassRoles (ADR-021 P2r2-F9)', () => { diff --git a/cdk/test/constructs/task-api-microvm-wake.test.ts b/cdk/test/constructs/task-api-microvm-wake.test.ts new file mode 100644 index 000000000..ce745fb3f --- /dev/null +++ b/cdk/test/constructs/task-api-microvm-wake.test.ts @@ -0,0 +1,81 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +// SPDX-License-Identifier: MIT-0 + +import { App, Stack } from 'aws-cdk-lib'; +import { Template } from 'aws-cdk-lib/assertions'; +import * as dynamodb from 'aws-cdk-lib/aws-dynamodb'; +import { TaskApi } from '../../src/constructs/task-api'; + +const IMAGE_ARN = 'arn:aws:lambda:us-east-1:123456789012:microvm-image:agent'; +let configured: Template; +let unconfigured: Template; +function fixture(imageArn?: string): Template { + const stack = new Stack(new App(), 'ApiTest'); + const table = (id: string, sortKey?: string) => new dynamodb.Table(stack, id, { + partitionKey: { name: 'task_id', type: dynamodb.AttributeType.STRING }, + ...(sortKey && { sortKey: { name: sortKey, type: dynamodb.AttributeType.STRING } }), + }); + new TaskApi(stack, 'Api', { + taskTable: table('Tasks'), + taskEventsTable: table('Events', 'event_id'), + taskApprovalsTable: table('Approvals', 'request_id'), + lambdaMicrovmImageArn: imageArn, + }); + return Template.fromStack(stack); +} +type Statement = { Action: string | string[]; Resource: unknown }; +function grants(template: Template, functionName: string): Statement[] { + return Object.entries(template.findResources('AWS::IAM::Policy')) + .filter(([id]) => id.includes(functionName)) + .flatMap(([, policy]) => policy.Properties.PolicyDocument.Statement as Statement[]) + .filter(statement => JSON.stringify(statement.Action).includes('Microvm')); +} +beforeAll(() => { + configured = fixture(IMAGE_ARN); + unconfigured = fixture(); +}); + +test.each(['ApproveTaskFn', 'DenyTaskFn'])('%s gets only Get/Resume against this image', name => { + expect(grants(configured, name)).toEqual([expect.objectContaining({ + Action: ['lambda:GetMicrovm', 'lambda:ResumeMicrovm'], + Resource: [IMAGE_ARN, `${IMAGE_ARN}:*`], + })]); +}); + +test('cancel keeps only Terminate against the same image', () => { + expect(grants(configured, 'CancelTaskFn')).toEqual([expect.objectContaining({ + Action: 'lambda:TerminateMicrovm', Resource: [IMAGE_ARN, `${IMAGE_ARN}:*`], + })]); +}); + +test.each(['ApproveTaskFn', 'DenyTaskFn', 'CancelTaskFn'])('%s gets no lifecycle grant without an image', name => { + expect(grants(unconfigured, name)).toEqual([]); +}); + +test('the decision APIs keep their15s Lambda budget and read worker IDs from saved task metadata', () => { + const functions = Object.entries(configured.findResources('AWS::Lambda::Function')) + .filter(([id]) => id.includes('ApproveTaskFn') || id.includes('DenyTaskFn')); + expect(functions).toHaveLength(2); + for (const [, resource] of functions) { + expect(resource.Properties.Timeout).toBe(15); + expect(resource.Properties.Environment.Variables.MICROVM_IMAGE_IDENTIFIER).toBeUndefined(); + } +}); diff --git a/cdk/test/constructs/task-orchestrator.test.ts b/cdk/test/constructs/task-orchestrator.test.ts index 4a523fae4..4c5cd709e 100644 --- a/cdk/test/constructs/task-orchestrator.test.ts +++ b/cdk/test/constructs/task-orchestrator.test.ts @@ -608,6 +608,7 @@ describe('TaskOrchestrator with the Lambda MicroVMs backend (ADR-021)', () => { imageIdentifier?: string; imageArn?: string; imageVersion?: string; + approvalSuspendEnabled?: boolean; ingressConnectorArns?: string[]; }): { template: Template } { const app = new App(); @@ -627,6 +628,8 @@ describe('TaskOrchestrator with the Lambda MicroVMs backend (ADR-021)', () => { }), runtimeArn: 'arn:aws:bedrock-agentcore:us-east-1:123456789012:runtime/test-runtime', microvmConfig: { + approvalsTable: mkTable('MicrovmApprovalsTable', 'request_id'), + approvalSuspendEnabled: config?.approvalSuspendEnabled, imageIdentifier: config?.imageIdentifier ?? IMAGE_ARN, imageArn: config?.imageArn ?? IMAGE_ARN, imageVersion: config?.imageVersion, @@ -672,12 +675,13 @@ describe('TaskOrchestrator with the Lambda MicroVMs backend (ADR-021)', () => { template = createMicrovmStack().template; pinnedTemplate = createMicrovmStack({ imageVersion: '4', + approvalSuspendEnabled: true, ingressConnectorArns: ['arn:aws:lambda:us-east-1:aws:network-connector:x', 'arn:y'], }).template; // What LambdaMicrovmCompute passes when the operator gave a bare image NAME: // the exact ARN it derived, never a wildcard. nameDerivedTemplate = createMicrovmStack({ - imageIdentifier: 'abca-agent', + imageIdentifier: NAME_DERIVED_IMAGE_ARN, imageArn: NAME_DERIVED_IMAGE_ARN, }).template; noMicrovmTemplate = createStack().template; @@ -692,6 +696,34 @@ describe('TaskOrchestrator with the Lambda MicroVMs backend (ADR-021)', () => { expect(env.MICROVM_PAYLOAD_BUCKET).toBeDefined(); }); + test('new sleep defaults off while explicit opt-in enables it', () => { + expect(orchestratorEnv(template).MICROVM_APPROVAL_SUSPEND_ENABLED).toBe('false'); + expect(orchestratorEnv(pinnedTemplate).MICROVM_APPROVAL_SUSPEND_ENABLED).toBe('true'); + template.hasResourceProperties('AWS::SSM::Parameter', { + Name: '/TestStack/microvm-approval-suspend-enabled', Type: 'String', Value: 'false', + }); + pinnedTemplate.hasResourceProperties('AWS::SSM::Parameter', { + Name: '/TestStack/microvm-approval-suspend-enabled', Type: 'String', Value: 'true', + }); + }); + + test('existing executions receive a stable parameter name with exact read-only permission', () => { + const [parameterId] = Object.keys(template.findResources('AWS::SSM::Parameter')); + expect(orchestratorEnv(template).MICROVM_APPROVAL_SUSPEND_PARAMETER_NAME).toEqual({ Ref: parameterId }); + const statement = microvmStatements(template).find(s => s.Sid === 'MicrovmSuspendConfiguration')!; + expect(statement.Action).toBe('ssm:GetParameter'); + expect(statement.Resource).toEqual({ + 'Fn::Join': ['', ['arn:', { Ref: 'AWS::Partition' }, ':ssm:us-east-1:123456789012:parameter', { Ref: parameterId }]], + }); + }); + + test('approval observation grants read and transaction condition checks, never approval writes', () => { + const statement = microvmStatements(template).find(s => s.Sid === 'MicrovmApprovalObservation')!; + expect(statement.Action).toEqual(['dynamodb:GetItem', 'dynamodb:ConditionCheckItem']); + expect(JSON.stringify(statement.Resource)).toContain('MicrovmApprovalsTable'); + expect(orchestratorEnv(template).TASK_APPROVALS_TABLE_NAME).toEqual({ Ref: expect.stringMatching(/^MicrovmApprovalsTable/) }); + }); + test('ALWAYS injects the ingress var, carrying the NO_INGRESS control', () => { // OUTCOME assertion, not an omission assertion. This var used to be injected // only when non-empty, and the test asserted it was `undefined` — which is @@ -715,6 +747,8 @@ describe('TaskOrchestrator with the Lambda MicroVMs backend (ADR-021)', () => { // ingress included, is present whenever microvmConfig is. const env = orchestratorEnv(template); expect(Object.keys(env).filter((k) => k.startsWith('MICROVM_')).sort()).toEqual([ + 'MICROVM_APPROVAL_SUSPEND_ENABLED', + 'MICROVM_APPROVAL_SUSPEND_PARAMETER_NAME', 'MICROVM_EGRESS_CONNECTOR_ARNS', 'MICROVM_EXECUTION_ROLE_ARN', 'MICROVM_IMAGE_IDENTIFIER', @@ -733,7 +767,7 @@ describe('TaskOrchestrator with the Lambda MicroVMs backend (ADR-021)', () => { expect(env.MICROVM_INGRESS_CONNECTOR_ARNS).not.toContain('NO_INGRESS'); }); - test('grants lifecycle calls and image-version discovery without enabling pause/wake callers yet', () => { + test('grants the lifecycle actions used by the supervisor and image discovery', () => { const actions = microvmStatements(template) .flatMap(s => Array.isArray(s.Action) ? s.Action : [s.Action]) .filter(a => a.startsWith('lambda:')); @@ -741,7 +775,9 @@ describe('TaskOrchestrator with the Lambda MicroVMs backend (ADR-021)', () => { 'lambda:GetMicrovm', 'lambda:GetMicrovmImageVersion', 'lambda:PassNetworkConnector', + 'lambda:ResumeMicrovm', 'lambda:RunMicrovm', + 'lambda:SuspendMicrovm', 'lambda:TerminateMicrovm', ]); }); @@ -754,11 +790,9 @@ describe('TaskOrchestrator with the Lambda MicroVMs backend (ADR-021)', () => { }); test('a name-derived image ARN is scoped to that exact name, never a wildcard', () => { - // An out-of-band image referenced by bare NAME is a valid RunMicrovm - // identifier but not an IAM resource. LambdaMicrovmCompute resolves it to the - // exact `microvmImage` ARN, so ADR-021's "scoped to platform-created images" - // holds here too — the account/Region-wide `microvm-image:*` widening this - // test previously accepted is a compliance violation, not a fallback. + // LambdaMicrovmCompute resolves a configured image name to an exact ARN + // for both RunMicrovm and IAM. Neither path may fall back to an account-wide + // image wildcard; the service does not accept a bare RunMicrovm image name. const lifecycle = microvmStatements(nameDerivedTemplate).find(s => s.Sid === 'MicrovmLifecycle')!; expect(lifecycle.Resource).toEqual([ NAME_DERIVED_IMAGE_ARN, @@ -792,14 +826,14 @@ describe('TaskOrchestrator with the Lambda MicroVMs backend (ADR-021)', () => { expect(JSON.stringify(passRole.Resource)).not.toContain('*'); }); - test('grants NO suspend/resume (P3) and NO auth-token minting (never)', () => { + test('grants supervisor control but no auth-token minting', () => { const actions = new Set( Object.values(template.findResources('AWS::IAM::Policy')) .flatMap(p => p.Properties.PolicyDocument.Statement as Array<{ Action: string | string[] }>) .flatMap(s => Array.isArray(s.Action) ? s.Action : [s.Action]), ); - expect(actions.has('lambda:SuspendMicrovm')).toBe(false); - expect(actions.has('lambda:ResumeMicrovm')).toBe(false); + expect(actions.has('lambda:SuspendMicrovm')).toBe(true); + expect(actions.has('lambda:ResumeMicrovm')).toBe(true); expect(actions.has('lambda:CreateMicrovmAuthToken')).toBe(false); expect(actions.has('lambda:CreateMicrovmShellAuthToken')).toBe(false); }); @@ -831,6 +865,8 @@ describe('TaskOrchestrator with the Lambda MicroVMs backend (ADR-021)', () => { test('adds no MicroVM statements when microvmConfig is omitted', () => { expect(microvmStatements(noMicrovmTemplate)).toEqual([]); expect(orchestratorEnv(noMicrovmTemplate).MICROVM_IMAGE_IDENTIFIER).toBeUndefined(); + expect(orchestratorEnv(noMicrovmTemplate).MICROVM_APPROVAL_SUSPEND_PARAMETER_NAME).toBeUndefined(); + noMicrovmTemplate.resourceCountIs('AWS::SSM::Parameter', 0); }); }); diff --git a/cdk/test/handlers/approve-task.test.ts b/cdk/test/handlers/approve-task.test.ts index f6067059b..b94dacbaf 100644 --- a/cdk/test/handlers/approve-task.test.ts +++ b/cdk/test/handlers/approve-task.test.ts @@ -21,6 +21,11 @@ import type { APIGatewayProxyEvent } from 'aws-lambda'; // --- Mocks --- const mockSend = jest.fn(); +const mockWake = jest.fn(); +jest.mock('../../src/handlers/shared/microvm-approval-wake', () => ({ + ...jest.requireActual('../../src/handlers/shared/microvm-approval-wake'), + wakeMicrovmAfterApproval: (...args: unknown[]) => mockWake(...args), +})); // Construct a stub TransactionCanceledException that has the // `err.name` + `CancellationReasons` the handler reads, plus makes @@ -92,6 +97,7 @@ function makeEvent(overrides: Partial = {}): APIGatewayPro beforeEach(() => { mockSend.mockReset(); + mockWake.mockReset().mockResolvedValue(undefined); ulidCounter = 0; }); @@ -235,3 +241,57 @@ describe('approve-task — error classification', () => { expect(res.statusCode).toBe(202); }); }); + +describe('postcommit MicroVM wake', () => { + test('wake diagnostics are task-bound and keep the supplied remaining-time signal', async () => { + mockSend.mockResolvedValue({}); + mockWake.mockImplementationOnce(async input => { + await input.emitEvent('microvm_resume_orphan', { stage: 'resume-request' }, input.options); + }); + expect((await handler(makeEvent())).statusCode).toBe(202); + const emitted = mockSend.mock.calls.find(([command]) => command.input.Item?.event_type === 'microvm_resume_orphan'); + expect(emitted?.[0].input.Item).toMatchObject({ + task_id: 'task-1', user_id: 'user-alice', metadata: { stage: 'resume-request' }, + }); + expect(emitted?.[1]).toBe(mockWake.mock.calls[0][0].options); + }); + + test('wake runs after the committed transaction and keeps its decision identity', async () => { + mockSend.mockResolvedValue({}); + const response = await handler(makeEvent()); + expect(response.statusCode).toBe(202); + const transaction = mockSend.mock.calls.findIndex(([command]) => command._type === 'TransactWrite'); + expect(mockSend.mock.invocationCallOrder[transaction]).toBeLessThan(mockWake.mock.invocationCallOrder[0]); + expect(mockWake.mock.calls[0][0]).toMatchObject({ + taskId: 'task-1', + userId: 'user-alice', + decision: 'APPROVED', + options: { abortSignal: expect.any(AbortSignal) }, + }); + }); + test('an unexpected wake failure cannot change the committed202 response', async () => { + mockSend.mockResolvedValue({}); + mockWake.mockRejectedValue(new Error('private wake failure')); + const response = await handler(makeEvent()); + expect(response.statusCode).toBe(202); + expect(response.body).not.toContain('private wake failure'); + expect(mockSend.mock.calls.filter(([command]) => command._type === 'TransactWrite')).toHaveLength(1); + }); + test('transaction failure cannot wake a worker', async () => { + mockSend.mockResolvedValueOnce({}).mockRejectedValueOnce(new Error('transaction failed')); + expect((await handler(makeEvent())).statusCode).toBe(500); + expect(mockWake).not.toHaveBeenCalled(); + }); + test('near Lambda timeout, optional postcommit work gets an already-expired budget', async () => { + mockSend.mockResolvedValue({}); + const response = await handler(makeEvent(), { getRemainingTimeInMillis: () => 500 }); + expect(response.statusCode).toBe(202); + expect(mockSend.mock.calls.map(([command]) => command._type)).toEqual(['Update', 'TransactWrite']); + expect(mockWake.mock.calls[0][0].options.abortSignal.aborted).toBe(true); + }); + test('audit failure still permits wake and does not fail the decision', async () => { + mockSend.mockResolvedValueOnce({}).mockResolvedValueOnce({}).mockRejectedValueOnce(new Error('audit failed')); + expect((await handler(makeEvent())).statusCode).toBe(202); + expect(mockWake).toHaveBeenCalledTimes(1); + }); +}); diff --git a/cdk/test/handlers/deny-task.test.ts b/cdk/test/handlers/deny-task.test.ts index 4fabae92e..ca0f23604 100644 --- a/cdk/test/handlers/deny-task.test.ts +++ b/cdk/test/handlers/deny-task.test.ts @@ -20,6 +20,11 @@ import type { APIGatewayProxyEvent } from 'aws-lambda'; const mockSend = jest.fn(); +const mockWake = jest.fn(); +jest.mock('../../src/handlers/shared/microvm-approval-wake', () => ({ + ...jest.requireActual('../../src/handlers/shared/microvm-approval-wake'), + wakeMicrovmAfterApproval: (...args: unknown[]) => mockWake(...args), +})); class MockTransactionCanceledException extends Error { name = 'TransactionCanceledException'; @@ -90,6 +95,7 @@ function makeEvent(overrides: Partial = {}): APIGatewayPro beforeEach(() => { mockSend.mockReset(); + mockWake.mockReset().mockResolvedValue(undefined); ulidCounter = 0; }); @@ -204,3 +210,57 @@ describe('deny-task — error classification', () => { expect(body.data.request_id).toBe('01KREQ'); }); }); + +describe('postcommit MicroVM wake', () => { + test('wake diagnostics are task-bound and keep the supplied remaining-time signal', async () => { + mockSend.mockResolvedValue({}); + mockWake.mockImplementationOnce(async input => { + await input.emitEvent('microvm_resume_orphan', { stage: 'resume-request' }, input.options); + }); + expect((await handler(makeEvent())).statusCode).toBe(202); + const emitted = mockSend.mock.calls.find(([command]) => command.input.Item?.event_type === 'microvm_resume_orphan'); + expect(emitted?.[0].input.Item).toMatchObject({ + task_id: 'task-1', user_id: 'user-alice', metadata: { stage: 'resume-request' }, + }); + expect(emitted?.[1]).toBe(mockWake.mock.calls[0][0].options); + }); + + test('wake runs after the committed transaction and keeps its decision identity', async () => { + mockSend.mockResolvedValue({}); + const response = await handler(makeEvent()); + expect(response.statusCode).toBe(202); + const transaction = mockSend.mock.calls.findIndex(([command]) => command._type === 'TransactWrite'); + expect(mockSend.mock.invocationCallOrder[transaction]).toBeLessThan(mockWake.mock.invocationCallOrder[0]); + expect(mockWake.mock.calls[0][0]).toMatchObject({ + taskId: 'task-1', + userId: 'user-alice', + decision: 'DENIED', + options: { abortSignal: expect.any(AbortSignal) }, + }); + }); + test('an unexpected wake failure cannot change the committed202 response', async () => { + mockSend.mockResolvedValue({}); + mockWake.mockRejectedValue(new Error('private wake failure')); + const response = await handler(makeEvent()); + expect(response.statusCode).toBe(202); + expect(response.body).not.toContain('private wake failure'); + expect(mockSend.mock.calls.filter(([command]) => command._type === 'TransactWrite')).toHaveLength(1); + }); + test('transaction failure cannot wake a worker', async () => { + mockSend.mockResolvedValueOnce({}).mockRejectedValueOnce(new Error('transaction failed')); + expect((await handler(makeEvent())).statusCode).toBe(500); + expect(mockWake).not.toHaveBeenCalled(); + }); + test('near Lambda timeout, optional postcommit work gets an already-expired budget', async () => { + mockSend.mockResolvedValue({}); + const response = await handler(makeEvent(), { getRemainingTimeInMillis: () => 500 }); + expect(response.statusCode).toBe(202); + expect(mockSend.mock.calls.map(([command]) => command._type)).toEqual(['Update', 'TransactWrite']); + expect(mockWake.mock.calls[0][0].options.abortSignal.aborted).toBe(true); + }); + test('audit failure still permits wake and does not fail the decision', async () => { + mockSend.mockResolvedValueOnce({}).mockResolvedValueOnce({}).mockRejectedValueOnce(new Error('audit failed')); + expect((await handler(makeEvent())).statusCode).toBe(202); + expect(mockWake).toHaveBeenCalledTimes(1); + }); +}); diff --git a/cdk/test/handlers/orchestrate-task-microvm.test.ts b/cdk/test/handlers/orchestrate-task-microvm.test.ts index 3a4cdc43e..131713a38 100644 --- a/cdk/test/handlers/orchestrate-task-microvm.test.ts +++ b/cdk/test/handlers/orchestrate-task-microvm.test.ts @@ -39,10 +39,17 @@ const MICROVM_ID = 'mvm-0123456789abcdef'; const ENDPOINT = 'https://mvm-0123456789abcdef.microvm.lambda.us-east-1.amazonaws.com'; const mockMicrovmSend = jest.fn(); +const mockSsmSend = jest.fn(); +jest.mock('@aws-sdk/client-ssm', () => ({ + SSMClient: jest.fn(() => ({ send: mockSsmSend })), + GetParameterCommand: jest.fn((input: unknown) => ({ input })), +})); jest.mock('@aws-sdk/client-lambda-microvms', () => ({ LambdaMicrovmsClient: jest.fn(() => ({ send: mockMicrovmSend })), RunMicrovmCommand: jest.fn((input: unknown) => ({ _type: 'RunMicrovm', input })), GetMicrovmCommand: jest.fn((input: unknown) => ({ _type: 'GetMicrovm', input })), + SuspendMicrovmCommand: jest.fn((input: unknown) => ({ _type: 'SuspendMicrovm', input })), + ResumeMicrovmCommand: jest.fn((input: unknown) => ({ _type: 'ResumeMicrovm', input })), TerminateMicrovmCommand: jest.fn((input: unknown) => ({ _type: 'TerminateMicrovm', input })), MicrovmState: { PENDING: 'PENDING', @@ -96,7 +103,13 @@ const mockTransitionTask = jest.fn(); const mockEmitTaskEvent = jest.fn(); const mockFinalizeTask = jest.fn(); const mockPollTaskStatus = jest.fn(); -const mockReconcile = jest.fn(); +const mockReadLifecycle = jest.fn(); +const mockSaveIntent = jest.fn(); +jest.mock('../../src/handlers/shared/microvm-lifecycle', () => ({ + ...jest.requireActual('../../src/handlers/shared/microvm-lifecycle'), + readMicrovmLifecycleSnapshot: (...args: unknown[]) => mockReadLifecycle(...args), + saveMicrovmLifecycleIntent: (...args: unknown[]) => mockSaveIntent(...args), +})); const mockFailTask = jest.fn(); const mockLoadTask = jest.fn(); const mockClaimStart = jest.fn(); @@ -121,7 +134,6 @@ jest.mock('../../src/handlers/shared/orchestrator', () => ({ loadBlueprintConfig: (...a: unknown[]) => mockLoadBlueprint(...a), loadTask: (...args: unknown[]) => mockLoadTask(...args), pollTaskStatus: (...a: unknown[]) => mockPollTaskStatus(...a), - reconcileMicrovmSubstrateState: (...a: unknown[]) => mockReconcile(...a), transitionTask: (...a: unknown[]) => mockTransitionTask(...a), buildComputeMetadata: realOrchestrator.buildComputeMetadata, })); @@ -157,6 +169,7 @@ process.env.AGENT_SESSION_ROLE_ARN = 'arn:aws:iam::123456789012:role/AbcaAgentSe import { TaskStatus } from '../../src/constructs/task-status'; import { handler } from '../../src/handlers/orchestrate-task'; +import type { MicrovmLifecycleSnapshot } from '../../src/handlers/shared/microvm-lifecycle'; import { LambdaMicrovmComputeStrategy, MICROVM_RUN_HOOK_PAYLOAD_LIMIT_BYTES } from '../../src/handlers/shared/strategies/lambda-microvm-strategy'; /** @@ -255,6 +268,19 @@ function failedTransition() { return mockTransitionTask.mock.calls.find(([, , to]) => to === TaskStatus.FAILED)!; } +function lifecycle(status: MicrovmLifecycleSnapshot['status'] = 'RUNNING'): MicrovmLifecycleSnapshot { + return { + taskId: 'TASK001', + userId: 'user-1', + status, + requestId: null, + approval: { kind: 'none' }, + taskStartedAtMs: Date.now() - 600_000, + heartbeatAtMs: Date.now(), + handle: { strategyType: 'lambda-microvm', sessionId: MICROVM_ID, microvmId: MICROVM_ID, endpoint: ENDPOINT }, + }; +} + beforeEach(() => { jest.clearAllMocks(); mockDdbSend.mockReset().mockResolvedValue({}); @@ -285,11 +311,67 @@ beforeEach(() => { return {}; }); mockMicrovmSend.mockReset(); + mockSsmSend.mockReset(); mockPollTaskStatus.mockResolvedValue({ attempts: 1, lastStatus: TaskStatus.COMPLETED }); - mockReconcile.mockResolvedValue({ taskFailed: false }); + mockReadLifecycle.mockReset().mockResolvedValue(lifecycle('COMPLETED')); + mockSaveIntent.mockReset().mockImplementation(async (snapshot, action) => { + const intent = { + version: 1, + generation: 'generation', + microvm_id: MICROVM_ID, + request_id: snapshot.requestId, + action, + requested_at_ms: Date.now(), + deadline_ms: snapshot.approval.kind === 'present' ? snapshot.approval.deadlineMs : null, + }; + mockReadLifecycle.mockResolvedValue({ ...snapshot, intent }); + return { status: 'saved', intent }; + }); + delete process.env.MICROVM_APPROVAL_SUSPEND_ENABLED; + delete process.env.MICROVM_APPROVAL_SUSPEND_PARAMETER_NAME; }); describe('orchestrate-task for a lambda-microvm task', () => { + test.each(['true', 'false', 'unavailable'])('real supervisor checks live setting %s before Suspend', async value => { + runMicrovmOk(); + const now = Date.now(); + const snapshot = lifecycle('AWAITING_APPROVAL'); + mockReadLifecycle.mockResolvedValue({ + ...snapshot, + requestId: 'gate', + handle: { + ...snapshot.handle, + imageArn: 'arn:aws:lambda:us-east-1:123456789012:microvm-image:agent', + imageVersion: '3.0', + lifecycleProtocol: '1', + }, + approval: { + kind: 'present', + status: 'PENDING', + created_at: new Date(now - 45_000).toISOString(), + timeout_s: 600, + createdAtMs: now - 45_000, + deadlineMs: now + 555_000, + }, + }); + const parameterName = '/backgroundagent-dev/microvm-approval-suspend-enabled'; + process.env.MICROVM_APPROVAL_SUSPEND_ENABLED = 'true'; + process.env.MICROVM_APPROVAL_SUSPEND_PARAMETER_NAME = parameterName; + if (value === 'unavailable') mockSsmSend.mockRejectedValue(new Error('unavailable')); + else mockSsmSend.mockResolvedValue({ Parameter: { Name: parameterName, Value: value } }); + mockMicrovmSend.mockResolvedValueOnce({ + microvmId: MICROVM_ID, + state: 'RUNNING', + startedAt: new Date(now - 60_000), + maximumDurationInSeconds: 28_800, + }).mockResolvedValue({}); + await handler({ task_id: 'TASK001' }, fakeContext().ctx as never); + expect(mockSsmSend).toHaveBeenCalledWith({ input: { Name: parameterName } }, { abortSignal: expect.any(AbortSignal) }); + expect(commandsOfType('SuspendMicrovm')).toHaveLength(value === 'true' ? 1 : 0); + expect(mockFinalizeTask.mock.calls[0][1].microvmFailureReason).toBeUndefined(); + expect(commandsOfType('TerminateMicrovm')).toHaveLength(1); + }); + test('registry assets alone exceed the hook cap and survive real hydration and v2 S3 delivery (#818)', async () => { const runtime = { transport: 'http', @@ -591,149 +673,110 @@ describe('orchestrate-task for a lambda-microvm task', () => { // billed. `fakeContext` runs one iteration and ignores `waitStrategy`, so this // needed `loopingContext` to be assertable at all. - test('a stale agent heartbeat stops the poll AND reclaims the MicroVM', async () => { + test('a stale agent heartbeat stops the real supervisor and reclaims the MicroVM', async () => { runMicrovmOk(); - // The substrate stays healthy throughout — this is the blind spot. mockMicrovmSend.mockResolvedValue({ microvmId: MICROVM_ID, state: 'RUNNING' }); - mockReconcile.mockResolvedValue({ taskFailed: false, suspendAnomalyReported: false }); - mockPollTaskStatus.mockResolvedValue({ - attempts: 1, - lastStatus: TaskStatus.RUNNING, - sessionUnhealthy: true, - }); - + mockReadLifecycle.mockResolvedValue({ ...lifecycle(), heartbeatAtMs: Date.now() - 300_000 }); const looping = loopingContext(); await handler({ task_id: 'TASK001' }, looping.ctx as never); - - // Exited on the FIRST unhealthy observation — not after burning the 8.5 h window. expect(looping.iterations).toHaveLength(1); - // finalize ran, and it saw the unhealthy flag (that is what writes FAILED). - expect(mockFinalizeTask).toHaveBeenCalledTimes(1); expect(mockFinalizeTask.mock.calls[0][1]).toMatchObject({ sessionUnhealthy: true }); - // ...and the reservation was actually reclaimed. This is the billing outcome. expect(commandsOfType('TerminateMicrovm')).toHaveLength(1); - expect(commandsOfType('TerminateMicrovm')[0].input) - .toEqual({ microvmIdentifier: MICROVM_ID }); + expect(mockPollTaskStatus).not.toHaveBeenCalled(); }); test('a healthy heartbeat keeps polling until a terminal task status', async () => { - // The other side of the same predicate: without this, a `sessionUnhealthy: true` - // hard-coded into the poll would pass the test above. runMicrovmOk(); mockMicrovmSend.mockResolvedValue({ microvmId: MICROVM_ID, state: 'RUNNING' }); - mockReconcile.mockResolvedValue({ taskFailed: false, suspendAnomalyReported: false }); - mockPollTaskStatus - .mockResolvedValueOnce({ attempts: 1, lastStatus: TaskStatus.RUNNING, sessionUnhealthy: false }) - .mockResolvedValueOnce({ attempts: 2, lastStatus: TaskStatus.RUNNING, sessionUnhealthy: false }) - .mockResolvedValue({ attempts: 3, lastStatus: TaskStatus.COMPLETED, sessionUnhealthy: false }); - + mockReadLifecycle.mockResolvedValueOnce(lifecycle()).mockResolvedValueOnce(lifecycle()); const looping = loopingContext(); await handler({ task_id: 'TASK001' }, looping.ctx as never); - expect(looping.iterations).toHaveLength(3); expect(mockFinalizeTask.mock.calls[0][1]).toMatchObject({ lastStatus: TaskStatus.COMPLETED }); - // Terminate happens on every finalize, healthy or not — the VM does not - // self-terminate on this substrate. expect(commandsOfType('TerminateMicrovm')).toHaveLength(1); }); - test('cross-checks the substrate through reconcileMicrovmSubstrateState while non-terminal', async () => { + test('unexpected suspension uses durable wake intent and requests Resume', async () => { runMicrovmOk(); - mockPollTaskStatus.mockResolvedValue({ attempts: 1, lastStatus: TaskStatus.RUNNING }); - // GetMicrovm during the poll, then TerminateMicrovm on finalize. + mockReadLifecycle.mockResolvedValue(lifecycle()); mockMicrovmSend.mockResolvedValueOnce({ microvmId: MICROVM_ID, state: 'SUSPENDED' }); - await handler({ task_id: 'TASK001' }, fakeContext().ctx as never); - expect(commandsOfType('GetMicrovm')).toHaveLength(1); - expect(mockReconcile).toHaveBeenCalledTimes(1); - const args = mockReconcile.mock.calls[0][0]; - expect(args.microvmId).toBe(MICROVM_ID); - expect(args.ddbStatus).toBe(TaskStatus.RUNNING); - // The strategy's mechanical mapping is what the orchestrator interprets. - expect(args.substrate).toEqual({ status: 'suspended', microvmState: 'SUSPENDED' }); + expect(mockSaveIntent.mock.calls[0][1]).toBe('resume'); + expect(commandsOfType('ResumeMicrovm')).toHaveLength(1); + expect(mockFinalizeTask.mock.calls[0][1].microvmSupervisor.recovery.kind).toBe('wake'); + expect(mockEmitTaskEvent.mock.calls.filter(c => c[1] === 'microvm_suspend_anomaly')).toHaveLength(1); }); - test('returns a failed poll state when reconciliation fails the task', async () => { + test('terminal substrate carries its classified reason into strong finalization', async () => { runMicrovmOk(); - mockPollTaskStatus.mockResolvedValue({ attempts: 3, lastStatus: TaskStatus.RUNNING }); - mockMicrovmSend.mockResolvedValueOnce({ microvmId: MICROVM_ID, state: 'TERMINATED' }); - mockReconcile.mockResolvedValue({ taskFailed: true }); - - await handler({ task_id: 'TASK001' }, fakeContext().ctx as never); - - expect(mockFinalizeTask).toHaveBeenCalledWith( - 'TASK001', - { attempts: 3, lastStatus: TaskStatus.FAILED }, - 'user-1', - ); + mockReadLifecycle.mockResolvedValue(lifecycle()); + mockMicrovmSend.mockResolvedValueOnce({ microvmId: MICROVM_ID, state: 'TERMINATED', stateReason: 'Run lifecycle hook returned HTTP status 400.' }); + const looping = loopingContext(); + await handler({ task_id: 'TASK001' }, looping.ctx as never); + expect(looping.iterations).toHaveLength(1); + expect(mockFinalizeTask.mock.calls[0][1]).toMatchObject({ + microvmFailureReason: 'substrate-terminal', + microvmFailureMessage: expect.stringMatching(/^MICROVM_RUN_HOOK_REJECTED:/), + }); }); - test('skips the substrate cross-check once the DDB status is terminal', async () => { + test('a terminal task skips Get but still terminates its original handle', async () => { runMicrovmOk(); - mockPollTaskStatus.mockResolvedValue({ attempts: 1, lastStatus: TaskStatus.COMPLETED }); - await handler({ task_id: 'TASK001' }, fakeContext().ctx as never); - expect(commandsOfType('GetMicrovm')).toHaveLength(0); - expect(mockReconcile).not.toHaveBeenCalled(); - // Finalize still terminates. - expect(commandsOfType('TerminateMicrovm')).toHaveLength(1); - }); - - test('a GetMicrovm poll failure is non-fatal and finalize still terminates', async () => { - runMicrovmOk(); - mockPollTaskStatus.mockResolvedValue({ attempts: 1, lastStatus: TaskStatus.RUNNING }); - mockMicrovmSend.mockRejectedValueOnce(new Error('transient')); - - await expect(handler({ task_id: 'TASK001' }, fakeContext().ctx as never)).resolves.toBeUndefined(); - - expect(mockReconcile).not.toHaveBeenCalled(); expect(commandsOfType('TerminateMicrovm')).toHaveLength(1); }); - test('threads suspendAnomalyReported into the reconcile call so the event fires once', async () => { + test('three Get failures stop the real durable loop and reclaim the VM', async () => { runMicrovmOk(); - mockPollTaskStatus.mockResolvedValue({ attempts: 1, lastStatus: TaskStatus.RUNNING }); - mockMicrovmSend.mockResolvedValueOnce({ microvmId: MICROVM_ID, state: 'SUSPENDED' }); - mockReconcile.mockResolvedValue({ taskFailed: false, suspendAnomalyReported: true }); - - await handler({ task_id: 'TASK001' }, fakeContext().ctx as never); - - // First poll of the task: nothing reported yet. - expect(mockReconcile.mock.calls[0][0].suspendAnomalyReported).toBe(false); - // ...and the reconciler's answer is carried into the state the next poll reads. - expect(mockFinalizeTask).toHaveBeenCalledWith( - 'TASK001', - expect.objectContaining({ microvmSuspendAnomalyReported: true }), - 'user-1', - ); + mockReadLifecycle.mockResolvedValue(lifecycle()); + mockMicrovmSend.mockRejectedValue(new Error('transient')); + const looping = loopingContext(); + await handler({ task_id: 'TASK001' }, looping.ctx as never); + expect(looping.iterations).toHaveLength(3); + expect(mockFinalizeTask.mock.calls[0][1]).toMatchObject({ + microvmFailureReason: 'substrate-read-failed', microvmSupervisor: { consecutivePollFailures: 3 }, + }); + expect(commandsOfType('TerminateMicrovm')).toHaveLength(2); + expect(mockEmitTaskEvent.mock.calls.filter(c => c[1] === 'microvm_cleanup_unconfirmed')).toHaveLength(1); }); - test('a MicroVM poll failure carries the anomaly flag forward rather than re-arming it', async () => { - // A GetMicrovm hiccup is not evidence that the anomaly ended, so it must not - // silently re-arm the event and produce a duplicate on the next cycle. + test('fast polls cannot exhaust the old attempt cap while the session deadline remains', async () => { runMicrovmOk(); - mockPollTaskStatus.mockResolvedValue({ attempts: 1, lastStatus: TaskStatus.RUNNING }); - mockMicrovmSend.mockRejectedValueOnce(new Error('transient')); - + mockReadLifecycle.mockResolvedValue(lifecycle()); + mockMicrovmSend.mockResolvedValue({ microvmId: MICROVM_ID, state: 'RUNNING' }); const { ctx } = fakeContext(); - // Seed the poll state as if a previous cycle had already reported. - const seededCtx = { + let continued: boolean | undefined; + await handler({ task_id: 'TASK001' }, { ...ctx, - waitForCondition: async ( - _name: string, - fn: (state: Record) => Promise, - ) => fn({ attempts: 1, microvmSuspendAnomalyReported: true }), - }; + waitForCondition: async (_name: string, fn: (state: Record) => Promise>, cfg: any) => { + const state = JSON.parse(JSON.stringify(await fn({ attempts: 2000 }))); + continued = cfg.waitStrategy(state).shouldContinue; + return state; + }, + } as never); + expect(continued).toBe(true); + }); - await handler({ task_id: 'TASK001' }, seededCtx as never); + test('lost worker ownership skips task finalization and payload deletion, but reaps only the old VM', async () => { + runMicrovmOk(); + mockReadLifecycle.mockResolvedValue({ ...lifecycle(), handle: { ...lifecycle().handle, microvmId: 'other' } }); + const looping = loopingContext(); + await handler({ task_id: 'TASK001' }, looping.ctx as never); + expect(looping.iterations).toHaveLength(1); + expect(mockFinalizeTask).not.toHaveBeenCalled(); + expect(s3CommandsOfType('DeleteObject')).toHaveLength(0); + expect(commandsOfType('GetMicrovm')).toHaveLength(0); + expect(commandsOfType('TerminateMicrovm')[0].input.microvmIdentifier).toBe(MICROVM_ID); + }); - expect(mockFinalizeTask).toHaveBeenCalledWith( - 'TASK001', - expect.objectContaining({ microvmSuspendAnomalyReported: true }), - 'user-1', - ); + test('database finalization failure still requests termination and propagates for durable retry', async () => { + runMicrovmOk(); + mockFinalizeTask.mockRejectedValue(new Error('database unavailable')); + await expect(handler({ task_id: 'TASK001' }, fakeContext().ctx as never)).rejects.toThrow('database unavailable'); + expect(commandsOfType('TerminateMicrovm')).toHaveLength(1); + expect(s3CommandsOfType('DeleteObject')).toHaveLength(0); }); describe('orphan reap when session start fails AFTER RunMicrovm succeeded', () => { diff --git a/cdk/test/handlers/orchestrate-task.test.ts b/cdk/test/handlers/orchestrate-task.test.ts index e2f934f47..81aef8e35 100644 --- a/cdk/test/handlers/orchestrate-task.test.ts +++ b/cdk/test/handlers/orchestrate-task.test.ts @@ -1226,6 +1226,79 @@ describe('hydrateAndTransition — registry asset resolution (#246)', () => { }); describe('finalizeTask', () => { + const supervisor = { + version: 1 as const, + microvmId: 'vm', + firstObservedAtMs: 1, + sessionDeadlineMs: 28_800_001, + lifetimeVerified: true, + consecutivePollFailures: 3, + consecutiveResumeFailures: 0, + anomalyReported: false, + nextPollInMs: 5_000, + }; + const vmTask = () => ({ + ...baseTask, + status: 'AWAITING_APPROVAL', + compute_type: 'lambda-microvm', + session_id: 'vm', + compute_metadata: { microvmId: 'vm', endpoint: 'https://vm.example' }, + }); + + test.each(['AWAITING_APPROVAL', 'RUNNING', 'HYDRATING'])('supervisor failure strongly reads %s and conditionally fails only its worker', async status => { + mockDdbSend.mockResolvedValueOnce({ Item: { ...vmTask(), status } }).mockResolvedValue({}); + await finalizeTask('TASK001', { + attempts: 3, microvmSupervisor: supervisor, microvmFailureReason: 'substrate-read-failed', + }, 'user-123'); + const read = mockDdbSend.mock.calls[0][0]; + expect(read.input.ConsistentRead).toBe(true); + const write = mockDdbSend.mock.calls[1][0]; + expect(write.input.ConditionExpression).toContain('compute_metadata.microvmId = :microvmId'); + expect(write.input.ExpressionAttributeValues).toMatchObject({ + ':fromStatus': status, + ':toStatus': 'FAILED', + ':microvmId': 'vm', + ':attr_error_message': 'MicroVM supervisor: substrate-read-failed', + }); + expect(mockReleaseTaskSlot).toHaveBeenCalledTimes(1); + }); + + test('supervisor finalization preserves a committed cancellation', async () => { + mockDdbSend.mockResolvedValueOnce({ Item: { ...vmTask(), status: 'CANCELLED', memory_written: true } }) + .mockResolvedValue({}); + await finalizeTask('TASK001', { + attempts: 3, microvmSupervisor: supervisor, microvmFailureReason: 'resume-request-failed-repeatedly', + }, 'user-123'); + expect(mockDdbSend.mock.calls.some(([command]) => command.input.ExpressionAttributeValues?.[':toStatus'])).toBe(false); + expect(mockDdbSend.mock.calls.find(([command]) => command._type === 'Put')?.[0].input.Item.event_type).toBe('task_cancelled'); + }); + + test('changed worker ownership skips both task mutation and reservation release', async () => { + mockDdbSend.mockResolvedValueOnce({ Item: { ...vmTask(), session_id: 'replacement' } }); + await expect(finalizeTask('TASK001', { + attempts: 3, microvmSupervisor: supervisor, microvmFailureReason: 'substrate-read-failed', + }, 'user-123')).resolves.toBe(false); + expect(mockDdbSend).toHaveBeenCalledTimes(1); + expect(mockReleaseTaskSlot).not.toHaveBeenCalled(); + }); + + test.each([['RUNNING', 'TIMED_OUT'], ['AWAITING_APPROVAL', 'FAILED'], ['HYDRATING', 'FAILED']])( + 'absolute session expiry in %s uses the allowed %s task outcome', async (status, expected) => { + mockDdbSend.mockResolvedValueOnce({ Item: { ...vmTask(), status } }).mockResolvedValue({}); + await finalizeTask('TASK001', { + attempts: 2000, microvmSupervisor: supervisor, microvmFailureReason: 'session-deadline', + }, 'user-123'); + expect(mockDdbSend.mock.calls[1][0].input.ExpressionAttributeValues[':toStatus']).toBe(expected); + }, + ); + + test('the legacy approval poll timeout also uses FAILED without changing the approval decision', async () => { + mockDdbSend.mockResolvedValueOnce({ Item: vmTask() }).mockResolvedValue({}); + await finalizeTask('TASK001', { attempts: 1020 }, 'user-123'); + expect(mockDdbSend.mock.calls[1][0].input.ExpressionAttributeValues[':toStatus']).toBe('FAILED'); + expect(mockDdbSend.mock.calls[2][0].input.Item.event_type).toBe('task_failed'); + }); + test.each([ ['FAILED', 'HYDRATING'], ['CANCELLED', 'RUNNING'], @@ -1393,7 +1466,7 @@ describe('finalizeTask', () => { mockDdbSend.mockResolvedValueOnce({ Item: { ...baseTask, status: 'COMPLETED', memory_written: true } }) .mockResolvedValue({}); mockReleaseTaskSlot.mockResolvedValue(false); - await expect(finalizeTask('TASK001', { attempts: 10 }, 'user-123')).resolves.toBeUndefined(); + await expect(finalizeTask('TASK001', { attempts: 10 }, 'user-123')).resolves.toBe(true); }); test('a reservation release outage propagates for retry', async () => { diff --git a/cdk/test/handlers/shared/error-classifier.test.ts b/cdk/test/handlers/shared/error-classifier.test.ts index d9c385147..56662ee2e 100644 --- a/cdk/test/handlers/shared/error-classifier.test.ts +++ b/cdk/test/handlers/shared/error-classifier.test.ts @@ -585,8 +585,7 @@ describe('classifyError', () => { }); test('classifies a MicroVM substrate-failure reason written by the orchestrator', () => { - // Must stay in lockstep with the reason string - // `reconcileMicrovmSubstrateState` persists. + // Retain classification of legacy messages persisted by the P2 reconciler. const result = classifyError( 'MicroVM substrate terminated before the agent wrote a terminal status: substrate state completed', )!; @@ -599,7 +598,7 @@ describe('classifyError', () => { // --- lifecycle-hook 4xx: non-retryable, and it must OUTRANK the generic entry --- // // The service reports a guest 4xx in `stateReason`, which - // `reconcileMicrovmSubstrateState` appends to the persisted message — so BOTH + // the P2 reconciler appended to its persisted message — so BOTH // the generic `MicroVM substrate terminated` string and the hook-status string // are present in one `error_message` and ORDER decides the answer. These tests // exist because the wrong order is invisible to a message-only assertion. @@ -608,7 +607,7 @@ describe('classifyError', () => { const hookReason = (status: number) => `Run lifecycle hook returned HTTP status ${status}. ` + 'Please check your hook endpoint and application logs for more details.'; - /** ...as `reconcileMicrovmSubstrateState` persists it. */ + /** Legacy persisted form; current finalization also supplies stable MICROVM_* codes. */ const reconciled = (reason: string) => 'MicroVM substrate terminated before the agent wrote a terminal status: ' + `substrate state completed (${reason})`; diff --git a/cdk/test/handlers/shared/microvm-approval-wake.test.ts b/cdk/test/handlers/shared/microvm-approval-wake.test.ts new file mode 100644 index 000000000..2bb71089b --- /dev/null +++ b/cdk/test/handlers/shared/microvm-approval-wake.test.ts @@ -0,0 +1,211 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +// SPDX-License-Identifier: MIT-0 + +const mockRead = jest.fn(); +const mockSave = jest.fn(); +const mockSend = jest.fn(); +const mockEmit = jest.fn(); +const mockLogger = { warn: jest.fn(), info: jest.fn() }; +jest.mock('../../../src/handlers/shared/microvm-lifecycle', () => ({ + readMicrovmLifecycleSnapshot: (...args: unknown[]) => mockRead(...args), + saveMicrovmLifecycleIntent: (...args: unknown[]) => mockSave(...args), +})); +jest.mock('../../../src/handlers/shared/logger', () => ({ logger: mockLogger })); +jest.mock('@aws-sdk/client-lambda-microvms', () => ({ + LambdaMicrovmsClient: jest.fn(() => ({ send: mockSend })), + GetMicrovmCommand: jest.fn(input => ({ type: 'get', input })), + ResumeMicrovmCommand: jest.fn(input => ({ type: 'resume', input })), +})); + +import { approvalPostCommitOptions, wakeMicrovmAfterApproval } from '../../../src/handlers/shared/microvm-approval-wake'; +import type { MicrovmLifecycleSnapshot } from '../../../src/handlers/shared/microvm-lifecycle'; + +const NOW = Date.parse('2026-09-15T12:00:00Z'); +let row: MicrovmLifecycleSnapshot; +let controller: AbortController; +function wake(decision: 'APPROVED' | 'DENIED' = 'APPROVED') { + return wakeMicrovmAfterApproval({ + taskId: 'task', + userId: 'user', + requestId: 'gate', + decision, + options: { abortSignal: controller.signal }, + emitEvent: mockEmit, + }); +} +beforeEach(() => { + jest.clearAllMocks(); + jest.spyOn(Date, 'now').mockReturnValue(NOW); + controller = new AbortController(); + row = { + taskId: 'task', + userId: 'user', + status: 'AWAITING_APPROVAL', + requestId: 'gate', + handle: { strategyType: 'lambda-microvm', microvmId: 'vm', sessionId: 'vm', endpoint: 'https://vm.example' }, + approval: { + kind: 'present', + status: 'APPROVED', + created_at: new Date(NOW - 60_000).toISOString(), + timeout_s: 600, + createdAtMs: NOW - 60_000, + deadlineMs: NOW + 540_000, + }, + }; + mockRead.mockReset().mockImplementation(async () => structuredClone(row)); + mockSave.mockReset().mockImplementation(async (snapshot, action) => { + row = { + ...snapshot, + intent: { + version: 1, + microvm_id: 'vm', + request_id: snapshot.requestId, + generation: 'wake-generation', + action, + requested_at_ms: NOW, + deadline_ms: NOW + 540_000, + }, + }; + return { status: 'saved', intent: row.intent }; + }); + mockSend.mockReset().mockResolvedValue({ state: 'SUSPENDED' }); + mockEmit.mockReset().mockResolvedValue(undefined); +}); +afterEach(() => jest.restoreAllMocks()); + +test.each(['APPROVED', 'DENIED'] as const)('%s saves wake before Get, rechecks ownership, then requests Resume and reads again', async decision => { + if (row.approval.kind === 'present') row = { ...row, approval: { ...row.approval, status: decision } }; + await wake(decision); + expect(mockSend.mock.calls.map(([command]) => command.type)).toEqual(['get', 'resume']); + expect(mockSave.mock.calls[0][1]).toBe('resume'); + expect(mockSave.mock.invocationCallOrder[0]).toBeLessThan(mockSend.mock.invocationCallOrder[0]); + expect(mockRead).toHaveBeenCalledTimes(3); + expect(mockRead.mock.invocationCallOrder[1]).toBeLessThan(mockSend.mock.invocationCallOrder[1]); + expect(mockSend.mock.invocationCallOrder[1]).toBeLessThan(mockRead.mock.invocationCallOrder[2]); + expect(mockSend.mock.calls[1][0].input).toEqual({ microvmIdentifier: 'vm' }); + expect(row.intent?.action).toBe('resume'); +}); + +test.each(['RUNNING', 'SUSPENDING', 'PENDING', 'UNKNOWN', 'TERMINATED'])('%s retains wake intent without premature Resume', async state => { + mockSend.mockResolvedValue({ state }); + await wake(); + expect(row.intent?.action).toBe('resume'); + expect(mockSend.mock.calls.map(([command]) => command.type)).toEqual(['get']); +}); + +test.each(['missing', 'cancelled', 'new-gate', 'pending'])('%s snapshot cannot trigger compute control', async kind => { + if (kind === 'missing') mockRead.mockResolvedValue(undefined); + if (kind === 'cancelled') row = { ...row, status: 'CANCELLED' }; + if (kind === 'new-gate') row = { ...row, requestId: 'new-gate' }; + if (kind === 'pending' && row.approval.kind === 'present') row = { ...row, approval: { ...row.approval, status: 'PENDING' } }; + await wake(); + expect(mockSave).not.toHaveBeenCalled(); + expect(mockSend).not.toHaveBeenCalled(); +}); + +test.each(['stale', 'ineligible'])('a %s intent write cannot authorize Get or Resume', async status => { + mockSave.mockResolvedValue({ status }); + await wake(); + expect(mockSend).not.toHaveBeenCalled(); +}); + +test.each(['cancel', 'gate', 'worker', 'generation'])('%s winning after Get blocks Resume', async change => { + mockSend.mockImplementationOnce(async () => { + if (change === 'cancel') row = { ...row, status: 'CANCELLED' }; + if (change === 'gate') row = { ...row, requestId: 'new-gate' }; + if (change === 'worker') row = { ...row, handle: { ...row.handle, microvmId: 'other' } }; + if (change === 'generation') row = { ...row, intent: { ...row.intent!, generation: 'other' } }; + return { state: 'SUSPENDED' }; + }); + await wake(); + expect(mockSend).toHaveBeenCalledTimes(1); +}); + +test('a consumed decision can fence its old suspend while task RUNNING', async () => { + row = { + ...row, + status: 'RUNNING', + requestId: null, + approval: { kind: 'none' }, + intent: { + version: 1, + microvm_id: 'vm', + request_id: 'gate', + action: 'suspend', + generation: 'old', + requested_at_ms: NOW - 5_000, + deadline_ms: NOW + 540_000, + }, + }; + await wake(); + expect(row.intent?.request_id).toBeNull(); + expect(row.intent?.action).toBe('resume'); + expect(mockSend.mock.calls.map(([command]) => command.type)).toEqual(['get', 'resume']); +}); + +test('a failed Resume still reads again, retains wake and only logs safe identifiers', async () => { + mockSend.mockResolvedValueOnce({ state: 'SUSPENDED' }).mockRejectedValueOnce(Object.assign( + new Error('private SDK details'), { name: 'AccessDeniedException', $metadata: { requestId: 'safe-123' } }, + )); + await expect(wake()).resolves.toBeUndefined(); + expect(mockRead).toHaveBeenCalledTimes(3); + expect(row.intent?.action).toBe('resume'); + expect(mockLogger.warn.mock.calls[0][1]).toMatchObject({ error_type: 'AccessDeniedException', aws_request_id: 'safe-123' }); + expect(mockEmit).toHaveBeenCalledWith('microvm_resume_orphan', expect.objectContaining({ + task_id: 'task', + request_id: 'gate', + microvm_id: 'vm', + stage: 'resume-request', + reason: 'resume-request-failed', + error_type: 'AccessDeniedException', + aws_request_id: 'safe-123', + }), expect.objectContaining({ abortSignal: expect.any(AbortSignal) })); + expect(JSON.stringify(mockLogger.warn.mock.calls)).not.toContain('private SDK details'); +}); + +test('an orphan-event failure remains best-effort and never discards the wake intent', async () => { + mockSend.mockResolvedValueOnce({ state: 'SUSPENDED' }).mockRejectedValueOnce(new Error('request failed')); + mockEmit.mockRejectedValue(new Error('private audit error')); + await expect(wake()).resolves.toBeUndefined(); + expect(row.intent?.action).toBe('resume'); + expect(mockRead).toHaveBeenCalledTimes(3); + expect(JSON.stringify(mockLogger.warn.mock.calls)).not.toContain('private audit error'); +}); + +test.each(['before-read', 'after-read', 'after-get'])('expired parent budget %s blocks subsequent work', async stage => { + if (stage === 'before-read') controller.abort(); + if (stage === 'after-read') mockRead.mockImplementationOnce(async () => { controller.abort(); return row; }); + if (stage === 'after-get') mockSend.mockImplementationOnce(async () => { controller.abort(); return { state: 'SUSPENDED' }; }); + await wake(); + if (stage === 'before-read') expect(mockRead).not.toHaveBeenCalled(); + if (stage !== 'after-get') expect(mockSave).not.toHaveBeenCalled(); + expect(mockSend.mock.calls.filter(([command]) => command.type === 'resume')).toHaveLength(0); +}); + +test('postcommit budget reserves response time and does not restart an expired invocation', () => { + const timeout = jest.spyOn(AbortSignal, 'timeout'); + approvalPostCommitOptions(NOW, { getRemainingTimeInMillis: () => 14_000 }); + expect(timeout).toHaveBeenLastCalledWith(8_000); + approvalPostCommitOptions(NOW, { getRemainingTimeInMillis: () => 2_000 }); + expect(timeout).toHaveBeenLastCalledWith(1_000); + expect(approvalPostCommitOptions(NOW, { getRemainingTimeInMillis: () => 900 }).abortSignal?.aborted).toBe(true); + expect(approvalPostCommitOptions(NOW - 14_500).abortSignal?.aborted).toBe(true); +}); diff --git a/cdk/test/handlers/shared/microvm-lifecycle-local.test.ts b/cdk/test/handlers/shared/microvm-lifecycle-local.test.ts index 21a82c959..1ae7bd5e0 100644 --- a/cdk/test/handlers/shared/microvm-lifecycle-local.test.ts +++ b/cdk/test/handlers/shared/microvm-lifecycle-local.test.ts @@ -29,6 +29,9 @@ if (endpoint && (new URL(endpoint).hostname !== '127.0.0.1' || new URL(endpoint) const mockBeforeSend = jest.fn(); const mockAfterSend = jest.fn(); const mockClients: DynamoDBDocumentClient[] = []; +jest.mock('../../../src/handlers/shared/microvm-suspend-config', () => ({ + readMicrovmSuspendEnabled: async () => true, +})); jest.mock('../../../src/handlers/shared/ua', () => { const actual = jest.requireActual('../../../src/handlers/shared/ua'); return { @@ -57,6 +60,7 @@ const approvals = `lifecycle-approvals-${suffix}`; Object.assign(process.env, { TASK_TABLE_NAME: tasks, TASK_APPROVALS_TABLE_NAME: approvals }); import { readMicrovmLifecycleSnapshot, saveMicrovmLifecycleIntent } from '../../../src/handlers/shared/microvm-lifecycle'; import { claimMicrovmStart, saveMicrovmImageCapability, saveMicrovmStartHandle } from '../../../src/handlers/shared/microvm-start'; +import { superviseMicrovm, type MicrovmSupervisorState } from '../../../src/handlers/shared/microvm-supervisor'; const raw = new DynamoDBClient({ endpoint: endpoint ?? 'http://127.0.0.1:1', @@ -136,6 +140,109 @@ local('MicroVM lifecycle against DynamoDB Local', () => { if (!value) throw new Error('Expected a MicroVM task'); return value; } + + test('real supervisor and store preserve approval-during-suspend wake across serialized polls', async () => { + const handle = (await current()).handle; + let observed = 'RUNNING'; + const strategy = { + type: 'lambda-microvm' as const, + startSession: jest.fn(), + pollSession: jest.fn(async () => ({ + status: 'running' as const, + microvmState: observed as 'RUNNING' | 'SUSPENDING' | 'SUSPENDED', + microvmStartedAtMs: Date.now() - 60_000, + microvmMaximumDurationSeconds: 28_800, + })), + stopSession: jest.fn(), + suspendSession: jest.fn(async () => { + expect((await current()).intent?.action).toBe('suspend'); + await admin.send(new UpdateCommand({ + TableName: approvals, + Key: { task_id: 'task', request_id: 'gate' }, + UpdateExpression: 'SET #s = :s', + ExpressionAttributeNames: { '#s': 'status' }, + ExpressionAttributeValues: { ':s': 'APPROVED' }, + })); + return { supported: true as const }; + }), + resumeSession: jest.fn(async () => ({ supported: true as const })), + }; + const cycle = (previous?: MicrovmSupervisorState) => superviseMicrovm({ + taskId: 'task', + userId: 'user', + handle, + strategy, + suspendEnabled: true, + pollIntervalMs: 30_000, + previous: previous ? JSON.parse(JSON.stringify(previous)) : undefined, + }); + const first = await cycle(); + expect(first.kind).toBe('continue'); + const savedWake = (await current()).intent; + expect(savedWake?.action).toBe('resume'); + observed = 'SUSPENDING'; + const second = await cycle(first.state); + expect(strategy.resumeSession.mock.calls).toHaveLength(0); + observed = 'SUSPENDED'; + const third = await cycle(second.state); + expect(strategy.resumeSession.mock.calls).toHaveLength(1); + expect((await current()).intent).toEqual(savedWake); + await admin.send(new UpdateCommand({ + TableName: tasks, + Key: { task_id: 'task' }, + UpdateExpression: 'SET #s = :s, agent_heartbeat_at = :now REMOVE awaiting_approval_request_id', + ExpressionAttributeNames: { '#s': 'status' }, + ExpressionAttributeValues: { ':s': 'RUNNING', ':now': new Date().toISOString() }, + })); + observed = 'RUNNING'; + const restored = await cycle(third.state); + expect((await current()).intent).toMatchObject({ action: 'resume', request_id: null }); + expect((await cycle(restored.state)).state.recovery).toBeUndefined(); + expect(strategy.suspendSession.mock.calls).toHaveLength(1); + }); + + test('real pre-command read prevents Suspend when approval wins just after intent commit', async () => { + mockAfterSend.mockImplementation(async command => { + const intent = command instanceof TransactWriteCommand + ? command.input.TransactItems?.[0].Update?.ExpressionAttributeValues?.[':intent'] : undefined; + if (intent?.action === 'suspend') { + await admin.send(new UpdateCommand({ + TableName: approvals, + Key: { task_id: 'task', request_id: 'gate' }, + UpdateExpression: 'SET #s = :s', + ExpressionAttributeNames: { '#s': 'status' }, + ExpressionAttributeValues: { ':s': 'APPROVED' }, + })); + } + }); + const strategy = { + type: 'lambda-microvm' as const, + startSession: jest.fn(), + stopSession: jest.fn(), + pollSession: jest.fn(async () => ({ + status: 'running' as const, + microvmState: 'RUNNING' as const, + microvmStartedAtMs: Date.now() - 60_000, + microvmMaximumDurationSeconds: 28_800, + })), + suspendSession: jest.fn(), + resumeSession: jest.fn(), + }; + const result = await superviseMicrovm({ + taskId: 'task', + userId: 'user', + handle: (await current()).handle, + strategy, + suspendEnabled: true, + pollIntervalMs: 30_000, + }); + expect(result.kind).toBe('continue'); + expect(strategy.suspendSession.mock.calls).toHaveLength(0); + expect(strategy.resumeSession.mock.calls).toHaveLength(0); + expect(await current()).toMatchObject({ + status: 'AWAITING_APPROVAL', approval: { status: 'APPROVED' }, intent: { action: 'resume' }, + }); + }); async function taskRow() { return (await admin.send(new GetCommand({ TableName: tasks, Key: { task_id: 'task' }, ConsistentRead: true }))).Item!; } diff --git a/cdk/test/handlers/shared/microvm-lifecycle.test.ts b/cdk/test/handlers/shared/microvm-lifecycle.test.ts index 2888a41ba..6a5ebbcdd 100644 --- a/cdk/test/handlers/shared/microvm-lifecycle.test.ts +++ b/cdk/test/handlers/shared/microvm-lifecycle.test.ts @@ -169,6 +169,34 @@ describe('MicroVM lifecycle policy', () => { }); describe('MicroVM lifecycle store', () => { + test('an expired caller budget prevents reads, writes and lost-reply recovery', async () => { + const controller = new AbortController(); + controller.abort(new Error('caller deadline')); + const options = { abortSignal: controller.signal }; + await expect(readMicrovmLifecycleSnapshot('task', 'user', options)).rejects.toThrow('caller deadline'); + await expect(saveMicrovmLifecycleIntent(snapshot(), 'resume', NOW, options)).rejects.toThrow('caller deadline'); + expect(mockSend).not.toHaveBeenCalled(); + }); + test('a task read finishing after the caller budget cannot start the approval read', async () => { + const controller = new AbortController(); + mockSend.mockImplementationOnce(async () => { + controller.abort(new Error('caller deadline')); + return { Item: task }; + }); + await expect(readMicrovmLifecycleSnapshot('task', 'user', { abortSignal: controller.signal })) + .rejects.toThrow('caller deadline'); + expect(mockSend).toHaveBeenCalledTimes(1); + }); + test('a write finishing after the caller budget cannot start another recovery budget', async () => { + const controller = new AbortController(); + mockSend.mockImplementationOnce(async () => { + controller.abort(new Error('caller deadline')); + return {}; + }); + await expect(saveMicrovmLifecycleIntent(snapshot(), 'resume', NOW, { abortSignal: controller.signal })) + .rejects.toThrow('caller deadline'); + expect(mockSend).toHaveBeenCalledTimes(1); + }); test('a cancelled task may retain its gate pointer and needs no approval read', async () => { mockSend.mockResolvedValueOnce({ Item: { ...task, status: 'CANCELLED' } }); const closed = await readMicrovmLifecycleSnapshot('task', 'user'); diff --git a/cdk/test/handlers/shared/microvm-start-recovery.test.ts b/cdk/test/handlers/shared/microvm-start-recovery.test.ts index 882bcae71..7e2192682 100644 --- a/cdk/test/handlers/shared/microvm-start-recovery.test.ts +++ b/cdk/test/handlers/shared/microvm-start-recovery.test.ts @@ -256,7 +256,7 @@ test('a cancelled task with a saved handle reaps that computer instead of starti expect(runCalls()).toHaveLength(1); expect(mockMicrovmSend).toHaveBeenLastCalledWith({ kind: 'terminate', input: { microvmIdentifier: handle.microvmId }, - }); + }, { abortSignal: expect.any(AbortSignal) }); }); test('failure to save a returned handle terminates the known computer', async () => { @@ -269,7 +269,7 @@ test('failure to save a returned handle terminates the known computer', async () .rejects.toThrow('MICROVM_START_RECEIPT_SAVE_FAILED'); expect(mockMicrovmSend).toHaveBeenLastCalledWith({ kind: 'terminate', input: { microvmIdentifier: handle.microvmId }, - }); + }, { abortSignal: expect.any(AbortSignal) }); }); test('a lost receipt-write response recovers the committed handle without termination', async () => { diff --git a/cdk/test/handlers/shared/microvm-supervisor.test.ts b/cdk/test/handlers/shared/microvm-supervisor.test.ts new file mode 100644 index 000000000..6b4d58b10 --- /dev/null +++ b/cdk/test/handlers/shared/microvm-supervisor.test.ts @@ -0,0 +1,564 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +// SPDX-License-Identifier: MIT-0 + +import type { ComputeStrategy, SessionStatus } from '../../../src/handlers/shared/compute-strategy'; +import type { MicrovmLifecycleSnapshot } from '../../../src/handlers/shared/microvm-lifecycle'; + +const mockRead = jest.fn(); +const mockSave = jest.fn(); +const mockSuspendEnabled = jest.fn(); +jest.mock('../../../src/handlers/shared/microvm-suspend-config', () => ({ + readMicrovmSuspendEnabled: (...args: unknown[]) => mockSuspendEnabled(...args), +})); +const mockLogger = { warn: jest.fn(), info: jest.fn(), error: jest.fn() }; +jest.mock('../../../src/handlers/shared/microvm-lifecycle', () => ({ + ...jest.requireActual('../../../src/handlers/shared/microvm-lifecycle'), + readMicrovmLifecycleSnapshot: (...args: unknown[]) => mockRead(...args), + saveMicrovmLifecycleIntent: (...args: unknown[]) => mockSave(...args), +})); +jest.mock('../../../src/handlers/shared/logger', () => ({ logger: mockLogger })); + +import { + stopMicrovmWithDiagnostics, superviseMicrovm, MICROVM_RECOVERY_TIMEOUT_MS, + type MicrovmSupervisorInput, type MicrovmSupervisorState, +} from '../../../src/handlers/shared/microvm-supervisor'; + +const NOW = Date.parse('2026-09-15T10:00:00Z'); +const handle = { + strategyType: 'lambda-microvm' as const, + sessionId: 'vm', + microvmId: 'vm', + endpoint: 'https://vm.example', + imageArn: 'arn:aws:lambda:us-east-1:123456789012:microvm-image:agent', + imageVersion: '3.0', + lifecycleProtocol: '1', +}; +let time: number; +let row: MicrovmLifecycleSnapshot; +let strategy: { + type: ComputeStrategy['type']; + startSession: jest.Mock; + pollSession: jest.Mock; + stopSession: jest.Mock; + suspendSession: jest.Mock; + resumeSession: jest.Mock; +}; +let emitEvent: jest.Mock; +let generation: number; + +function observation(state: string = 'RUNNING'): SessionStatus { + return { + status: state === 'TERMINATED' ? 'completed' : state.startsWith('SUSPEND') ? 'suspended' : 'running', + microvmState: state as SessionStatus['microvmState'], + microvmStartedAtMs: NOW - 60_000, + microvmMaximumDurationSeconds: 28_800, + }; +} +function intent(action: 'suspend' | 'resume', requestedAt = NOW - 10_000) { + row = { + ...row, + intent: { + version: 1, + generation: `generation-${++generation}`, + microvm_id: 'vm', + request_id: row.requestId, + action, + requested_at_ms: requestedAt, + deadline_ms: row.approval.kind === 'present' ? row.approval.deadlineMs : null, + }, + }; +} +function working() { + row = { ...row, status: 'RUNNING', requestId: null, approval: { kind: 'none' } }; +} +function approve() { + if (row.approval.kind !== 'present') throw new Error('fixture has no approval'); + row = { ...row, approval: { ...row.approval, status: 'APPROVED' } }; +} +function run(previous?: MicrovmSupervisorState, change: Partial = {}) { + return superviseMicrovm({ + taskId: 'task', + userId: 'user', + handle, + strategy, + pollIntervalMs: 30_000, + suspendEnabled: true, + previous: previous ? JSON.parse(JSON.stringify(previous)) : undefined, + emitEvent, + ...change, + }); +} + +beforeEach(() => { + jest.clearAllMocks(); + time = NOW; + generation = 0; + jest.spyOn(Date, 'now').mockImplementation(() => time); + row = { + taskId: 'task', + userId: 'user', + status: 'AWAITING_APPROVAL', + handle, + requestId: 'gate', + approval: { + kind: 'present', + status: 'PENDING', + created_at: new Date(NOW - 45_000).toISOString(), + timeout_s: 600, + createdAtMs: NOW - 45_000, + deadlineMs: NOW + 555_000, + }, + }; + mockRead.mockReset().mockImplementation(async () => structuredClone(row)); + mockSuspendEnabled.mockReset().mockResolvedValue(true); + mockSave.mockReset().mockImplementation(async (_snapshot, action) => { + if (row.intent?.action !== action || row.intent.request_id !== row.requestId) intent(action, time); + return { status: 'saved', intent: row.intent }; + }); + strategy = { + type: 'lambda-microvm', + startSession: jest.fn(), + pollSession: jest.fn().mockResolvedValue(observation()), + stopSession: jest.fn().mockResolvedValue(undefined), + suspendSession: jest.fn().mockResolvedValue({ supported: true }), + resumeSession: jest.fn().mockResolvedValue({ supported: true }), + }; + emitEvent = jest.fn().mockResolvedValue(undefined); +}); +afterEach(() => jest.restoreAllMocks()); + +test('saves intent, rechecks the gate, requests suspend and rechecks the outcome', async () => { + const result = await run(); + expect(result.kind).toBe('continue'); + expect(result.state.recovery?.kind).toBe('suspend'); + expect(strategy.suspendSession).toHaveBeenCalledTimes(1); + expect(mockRead).toHaveBeenCalledTimes(3); + expect(mockSave.mock.invocationCallOrder[0]).toBeLessThan(mockRead.mock.invocationCallOrder[1]); + expect(mockRead.mock.invocationCallOrder[1]).toBeLessThan(strategy.suspendSession.mock.invocationCallOrder[0]); + expect(strategy.suspendSession.mock.invocationCallOrder[0]).toBeLessThan(mockRead.mock.invocationCallOrder[2]); + expect(strategy.stopSession).not.toHaveBeenCalled(); +}); + +test.each(['disabled', 'legacy', 'unverified-lifetime', 'short-window'])('%s keeps a working VM awake', async kind => { + if (kind === 'legacy') row = { ...row, handle: { ...handle, lifecycleProtocol: undefined } }; + if (kind === 'unverified-lifetime') strategy.pollSession.mockResolvedValue({ status: 'running', microvmState: 'RUNNING' }); + if (kind === 'short-window') time = NOW + 480_000; + expect((await run(undefined, { suspendEnabled: kind !== 'disabled' })).kind).toBe('continue'); + expect(strategy.suspendSession).not.toHaveBeenCalled(); + expect(strategy.resumeSession).not.toHaveBeenCalled(); + expect(mockSuspendEnabled).not.toHaveBeenCalled(); +}); + +test('an existing opt-in execution observes live disable without failing healthy compute', async () => { + working(); + const first = await run(); + row = { + ...row, + status: 'AWAITING_APPROVAL', + requestId: 'next-gate', + approval: { + kind: 'present', + status: 'PENDING', + created_at: new Date(NOW - 45_000).toISOString(), + timeout_s: 600, + createdAtMs: NOW - 45_000, + deadlineMs: NOW + 555_000, + }, + }; + mockSuspendEnabled.mockResolvedValue(false); + for (let attempt = 0; attempt < 4; attempt++) { + const result = await run(first.state); + expect(result.kind).toBe('continue'); + expect(result.state.consecutivePollFailures).toBe(0); + } + expect(strategy.suspendSession).not.toHaveBeenCalled(); + expect(mockSave).not.toHaveBeenCalled(); +}); + +test('disable after intent commit fences wake before any Suspend request', async () => { + mockSuspendEnabled.mockResolvedValueOnce(true).mockResolvedValue(false); + const result = await run(); + expect(result.state.recovery?.kind).toBe('wake'); + expect(row.intent?.action).toBe('resume'); + expect(strategy.suspendSession).not.toHaveBeenCalled(); + expect(mockSuspendEnabled).toHaveBeenCalledTimes(2); +}); + +test('wake recovery never depends on reading the suspension setting', async () => { + const first = await run(); + approve(); + strategy.pollSession.mockResolvedValue(observation('SUSPENDED')); + mockSuspendEnabled.mockClear().mockResolvedValue(false); + expect((await run(first.state)).kind).toBe('continue'); + expect(strategy.resumeSession).toHaveBeenCalledTimes(1); + expect(mockSuspendEnabled).not.toHaveBeenCalled(); +}); + +test('later polls and serialized restart cannot extend the original service deadline', async () => { + working(); + const first = await run(); + const deadline = NOW - 60_000 + 28_800_000; + expect(first.state.sessionDeadlineMs).toBe(deadline); + time += 60_000; + strategy.pollSession.mockResolvedValue({ ...observation(), microvmStartedAtMs: time, microvmMaximumDurationSeconds: 99_999 }); + const later = await run(first.state); + expect(later.state.sessionDeadlineMs).toBe(deadline); + time = deadline; + expect(await run(later.state)).toMatchObject({ kind: 'failure', reason: 'session-deadline' }); +}); + +test('unknown state has a fixed recovery window across serialized polls', async () => { + working(); + strategy.pollSession.mockResolvedValue(observation('UNKNOWN')); + const first = await run(); + time += MICROVM_RECOVERY_TIMEOUT_MS; + const expired = await run(first.state); + expect(expired).toMatchObject({ kind: 'failure', reason: 'recovery-deadline' }); + expect(expired.state.recovery?.sinceMs).toBe(first.state.recovery?.sinceMs); +}); + +test('repeated Get errors retain their count even when task reads succeed', async () => { + strategy.pollSession.mockRejectedValue(Object.assign(new Error('network'), { name: 'TimeoutError' })); + const first = await run(); + const second = await run(first.state); + const third = await run(second.state); + expect(first.kind).toBe('continue'); + expect(second.state.consecutivePollFailures).toBe(2); + expect(third).toMatchObject({ kind: 'failure', state: { consecutivePollFailures: 3 } }); +}); + +test('a complete successful observation clears prior poll failures', async () => { + working(); + strategy.pollSession.mockRejectedValueOnce(new Error('network')); + const failed = await run(); + expect(failed.state.consecutivePollFailures).toBe(1); + expect((await run(failed.state)).state.consecutivePollFailures).toBe(0); +}); + +test('permanent Get denial escalates without waiting for more requests', async () => { + strategy.pollSession.mockRejectedValue(Object.assign(new Error('private-detail'), { name: 'AccessDeniedException' })); + expect((await run()).kind).toBe('failure'); + expect(JSON.stringify(emitEvent.mock.calls)).not.toContain('private-detail'); +}); + +test.each(['missing', 'replaced'])('%s worker ownership is never followed to another VM', async kind => { + if (kind === 'missing') mockRead.mockResolvedValue(undefined); + else row = { ...row, handle: { ...handle, microvmId: 'other-vm', sessionId: 'other-vm' } }; + expect((await run()).kind).toBe('ownership-lost'); + expect(strategy.pollSession).not.toHaveBeenCalled(); + expect(strategy.suspendSession).not.toHaveBeenCalled(); +}); + +test('a cancelled task is returned for cleanup without any wake or status change', async () => { + row = { ...row, status: 'CANCELLED' }; + expect(await run()).toMatchObject({ kind: 'closed', status: 'CANCELLED' }); + expect(strategy.pollSession).not.toHaveBeenCalled(); + expect(mockSave).not.toHaveBeenCalled(); +}); + +test('terminal substrate observations are handed back for strong task reconciliation', async () => { + strategy.pollSession.mockResolvedValue(observation('TERMINATED')); + expect((await run()).kind).toBe('substrate-terminal'); + expect(mockSave).not.toHaveBeenCalled(); +}); + +test('an intended long sleep stays asleep and does not fail heartbeat liveness', async () => { + intent('suspend'); + strategy.pollSession.mockResolvedValue(observation('SUSPENDED')); + const result = await run(); + expect(result).toMatchObject({ kind: 'continue', deferHeartbeat: true }); + expect(result.state.recovery).toBeUndefined(); + expect(strategy.resumeSession).not.toHaveBeenCalled(); +}); + +test('a service stuck in SUSPENDING has a bounded transition window', async () => { + intent('suspend', NOW - MICROVM_RECOVERY_TIMEOUT_MS); + strategy.pollSession.mockResolvedValue(observation('SUSPENDING')); + expect((await run()).kind).toBe('failure'); +}); + +test('approval between intent save and the pre-command read prevents suspend', async () => { + mockRead.mockImplementationOnce(async () => structuredClone(row)) + .mockImplementationOnce(async () => { approve(); return structuredClone(row); }); + expect((await run()).kind).toBe('continue'); + expect(strategy.suspendSession).not.toHaveBeenCalled(); + expect(row.intent?.action).toBe('resume'); +}); + +test('approval during suspend saves sticky wake, then waits for SUSPENDED before Resume', async () => { + strategy.suspendSession.mockImplementationOnce(async () => { approve(); return { supported: true }; }); + const first = await run(); + expect(row.intent?.action).toBe('resume'); + expect(strategy.resumeSession).not.toHaveBeenCalled(); + strategy.pollSession.mockResolvedValue(observation('SUSPENDING')); + const suspending = await run(first.state); + expect(strategy.resumeSession).not.toHaveBeenCalled(); + strategy.pollSession.mockResolvedValue(observation('SUSPENDED')); + const waking = await run(suspending.state); + expect(strategy.resumeSession).toHaveBeenCalledTimes(1); + expect(waking.state.recovery?.kind).toBe('wake'); + strategy.pollSession.mockResolvedValue(observation()); + const awake = await run(waking.state); + expect(row.intent?.action).toBe('resume'); + expect(awake.state.recovery?.kind).toBe('wake'); + working(); + row = { ...row, heartbeatAtMs: time }; + const restored = await run(awake.state); + expect(row.intent).toMatchObject({ action: 'resume', request_id: null }); + expect((await run(restored.state)).state.recovery).toBeUndefined(); +}); + +test('cancellation during a command cannot trigger a post-command resume', async () => { + strategy.suspendSession.mockImplementationOnce(async () => { + row = { ...row, status: 'CANCELLED' }; + return { supported: true }; + }); + expect(await run()).toMatchObject({ kind: 'closed', status: 'CANCELLED' }); + expect(strategy.resumeSession).not.toHaveBeenCalled(); +}); + +test('an uncertain suspend failure leaves coding recoverable and fences late suspension', async () => { + strategy.suspendSession.mockRejectedValueOnce(Object.assign(new Error('private-detail'), { name: 'TimeoutError' })); + const result = await run(); + expect(result.kind).toBe('continue'); + expect(row.intent?.action).toBe('resume'); + expect(result.state.recovery?.kind).toBe('wake'); + expect(JSON.stringify(emitEvent.mock.calls)).not.toContain('private-detail'); +}); + +test('repeated resume failures escalate despite successful state reads', async () => { + approve(); + intent('suspend'); + strategy.pollSession.mockResolvedValue(observation('SUSPENDED')); + strategy.resumeSession.mockRejectedValue(Object.assign(new Error('retry'), { name: 'TimeoutError' })); + const first = await run(); + const second = await run(first.state); + expect(await run(second.state)).toMatchObject({ + kind: 'failure', reason: 'resume-request-failed-repeatedly', state: { consecutiveResumeFailures: 3 }, + }); +}); + +test('uncertain observations cannot reset a wake recovery clock', async () => { + approve(); + intent('suspend'); + strategy.pollSession.mockResolvedValue(observation('SUSPENDED')); + const first = await run(); + time += 60_000; + strategy.pollSession.mockResolvedValue(observation('UNKNOWN')); + const unknown = await run(first.state); + expect(unknown.state.recovery).toEqual(first.state.recovery); + time += 60_000; + strategy.pollSession.mockResolvedValue(observation('SUSPENDED')); + expect((await run(unknown.state)).kind).toBe('failure'); + expect(strategy.resumeSession).toHaveBeenCalledTimes(1); +}); + +test('durable wake recovery repairs a missing wake write even after a RUNNING observation', async () => { + intent('suspend'); + const initial = await run(undefined, { suspendEnabled: false }); + const previous = { ...initial.state, recovery: { kind: 'wake' as const, sinceMs: NOW } }; + const result = await run(previous); + expect(result.kind).toBe('continue'); + expect(row.intent?.action).toBe('resume'); + expect(strategy.suspendSession).not.toHaveBeenCalled(); +}); + +test('recovered RUNNING gets heartbeat grace only until a fresh heartbeat', async () => { + working(); + row = { ...row, heartbeatAtMs: NOW - 300_000 }; + strategy.pollSession.mockResolvedValue(observation('SUSPENDED')); + const waking = await run(); + strategy.pollSession.mockResolvedValue(observation()); + time += 5_000; + const waiting = await run(waking.state); + expect(waiting.deferHeartbeat).toBe(true); + time += 45_000; + row = { ...row, heartbeatAtMs: time }; + const healthy = await run(waiting.state); + expect(healthy.deferHeartbeat).toBe(false); + expect(healthy.state.recovery).toBeUndefined(); + time += 300_000; + expect((await run(healthy.state)).deferHeartbeat).toBe(false); +}); + +test('a resumed RUNNING worker with no fresh heartbeat cannot retain grace forever', async () => { + working(); + row = { ...row, heartbeatAtMs: NOW - 300_000 }; + strategy.pollSession.mockResolvedValue(observation('SUSPENDED')); + const waking = await run(); + strategy.pollSession.mockResolvedValue(observation()); + time += MICROVM_RECOVERY_TIMEOUT_MS; + expect((await run(waking.state)).kind).toBe('failure'); +}); + +test('intent storage failures remain bounded when all observations succeed', async () => { + approve(); + strategy.pollSession.mockResolvedValue(observation('SUSPENDED')); + mockSave.mockRejectedValue(new Error('store timeout')); + const first = await run(); + const second = await run(first.state); + expect((await run(second.state)).kind).toBe('failure'); + expect(strategy.resumeSession).not.toHaveBeenCalled(); +}); + +test('an audit failure does not discard the wake recovery outcome', async () => { + working(); + strategy.pollSession.mockResolvedValue(observation('SUSPENDED')); + emitEvent.mockRejectedValue(new Error('private audit detail')); + expect((await run()).kind).toBe('continue'); + expect(strategy.resumeSession).toHaveBeenCalledTimes(1); + expect(JSON.stringify(mockLogger.warn.mock.calls)).not.toContain('private audit detail'); +}); + +test.each(['post-read', 'wake-write'])('lost %s after Suspend retains the wake obligation across restart', async failure => { + if (failure === 'post-read') { + mockRead.mockResolvedValueOnce(structuredClone(row)) + .mockImplementationOnce(async () => structuredClone(row)) + .mockRejectedValueOnce(new Error('lost read')); + } else { + strategy.suspendSession.mockImplementationOnce(async () => { + approve(); + mockSave.mockRejectedValueOnce(new Error('lost compensating write')); + return { supported: true }; + }); + } + const uncertain = await run(); + expect(uncertain.state.recovery?.kind).toBe('wake'); + const repaired = await run(uncertain.state); + expect(repaired.kind).toBe('continue'); + expect(row.intent?.action).toBe('resume'); + expect(strategy.suspendSession).toHaveBeenCalledTimes(1); +}); + +test('a normal wake in progress is not reported as an unintended suspension', async () => { + approve(); + intent('resume'); + strategy.pollSession.mockResolvedValue(observation('SUSPENDING')); + await run(); + expect(emitEvent).not.toHaveBeenCalled(); +}); + +test('HYDRATING startup is bounded even when AWS reports RUNNING', async () => { + working(); + row = { ...row, status: 'HYDRATING' }; + const starting = await run(); + time += 300_000; + expect(await run(starting.state)).toMatchObject({ kind: 'failure', reason: 'startup-deadline' }); +}); + +test('FINALIZING can finish normally but cannot exceed the original lifetime', async () => { + working(); + const active = await run(); + row = { ...row, status: 'FINALIZING' }; + expect((await run(active.state)).kind).toBe('closed'); + time = active.state.sessionDeadlineMs; + expect(await run(active.state)).toMatchObject({ kind: 'failure', reason: 'session-deadline' }); +}); + +test('ordinary RUNNING heartbeat failures remain visible after recovery ends', async () => { + working(); + row = { ...row, taskStartedAtMs: NOW - 600_000, heartbeatAtMs: NOW - 300_000 }; + expect((await run()).heartbeatUnhealthy).toBe(true); + row = { ...row, heartbeatAtMs: NOW }; + expect((await run()).heartbeatUnhealthy).toBe(false); + row = { ...row, heartbeatAtMs: undefined }; + expect((await run()).heartbeatUnhealthy).toBe(true); +}); + +test('AWS RUNNING cannot hide a worker that never consumes a committed decision', async () => { + approve(); + const first = await run(); + expect(first.state.recovery?.kind).toBe('wake'); + time += MICROVM_RECOVERY_TIMEOUT_MS; + expect(await run(first.state)).toMatchObject({ kind: 'failure', reason: 'wake-deadline' }); +}); + +test('an expired gate stuck PENDING is bounded even when AWS stays RUNNING', async () => { + time += 600_000; + const first = await run(); + expect(first.state.recovery?.kind).toBe('wake'); + time += MICROVM_RECOVERY_TIMEOUT_MS; + expect((await run(first.state)).kind).toBe('failure'); +}); + +test('multiple failed requests within one cycle count as one failed cycle', async () => { + row = { ...row, approval: { kind: 'unavailable', errorType: 'TimeoutError' } }; + strategy.pollSession.mockResolvedValue(observation('SUSPENDED')); + mockSave.mockRejectedValue(new Error('write timeout')); + const first = await run(); + expect(first.state.consecutivePollFailures).toBe(1); + const second = await run(first.state); + expect(second.state.consecutivePollFailures).toBe(2); + expect((await run(second.state)).kind).toBe('failure'); +}); + +test('a new gate between wake intent and dispatch prevents that stale Resume', async () => { + approve(); + strategy.pollSession.mockResolvedValue(observation('SUSPENDED')); + mockRead.mockImplementationOnce(async () => structuredClone(row)) + .mockImplementationOnce(async () => { row = { ...row, requestId: 'next-gate' }; return structuredClone(row); }); + await run(); + expect(strategy.resumeSession).not.toHaveBeenCalled(); +}); + +test('anomaly reporting survives a failed poll and rearms after a fresh recovered heartbeat', async () => { + working(); + row = { ...row, heartbeatAtMs: time }; + strategy.pollSession.mockResolvedValue(observation('SUSPENDED')); + const first = await run(); + expect(emitEvent.mock.calls.filter(([type]) => type === 'microvm_suspend_anomaly')).toHaveLength(1); + strategy.pollSession.mockRejectedValueOnce(new Error('transient')); + const failed = await run(first.state); + expect(failed.state.anomalyReported).toBe(true); + const recovering = await run(failed.state); + expect(emitEvent.mock.calls.filter(([type]) => type === 'microvm_suspend_anomaly')).toHaveLength(1); + strategy.pollSession.mockResolvedValue(observation()); + const recovered = await run(recovering.state); + expect(recovered.state.recovery).toBeUndefined(); + strategy.pollSession.mockResolvedValue(observation('SUSPENDED')); + await run(recovered.state); + expect(emitEvent.mock.calls.filter(([type]) => type === 'microvm_suspend_anomaly')).toHaveLength(2); +}); + +describe('cleanup evidence', () => { + test.each(['requested', 'not-found'] as const)('%s ends cleanup without claiming more evidence', async outcome => { + strategy.stopSession.mockResolvedValue({ outcome }); + await stopMicrovmWithDiagnostics({ taskId: 'task', handle, strategy, emitEvent }); + expect(strategy.stopSession).toHaveBeenCalledTimes(1); + expect(emitEvent).not.toHaveBeenCalled(); + }); + test('a conflict gets one bounded retry, then reports uncertainty with the retained handle', async () => { + strategy.stopSession.mockResolvedValue({ outcome: 'unconfirmed', error_type: 'ConflictException', aws_request_id: 'safe-123' }); + await stopMicrovmWithDiagnostics({ taskId: 'task', handle, strategy, emitEvent }); + expect(strategy.stopSession).toHaveBeenCalledTimes(2); + expect(strategy.stopSession.mock.calls[1][1]).toBe(strategy.stopSession.mock.calls[0][1]); + expect(emitEvent).toHaveBeenCalledWith('microvm_cleanup_unconfirmed', { + task_id: 'task', microvm_id: 'vm', error_type: 'ConflictException', aws_request_id: 'safe-123', + }, expect.objectContaining({ abortSignal: expect.any(AbortSignal) })); + }); + test('denied cleanup does not keep retrying and failed audit never masks finalization', async () => { + strategy.stopSession.mockResolvedValue({ outcome: 'unconfirmed', error_type: 'AccessDeniedException' }); + emitEvent.mockRejectedValue(new Error('private audit details')); + await expect(stopMicrovmWithDiagnostics({ taskId: 'task', handle, strategy, emitEvent })).resolves.toBeUndefined(); + expect(strategy.stopSession).toHaveBeenCalledTimes(1); + expect(JSON.stringify(mockLogger.warn.mock.calls)).not.toContain('private audit details'); + }); +}); diff --git a/cdk/test/handlers/shared/microvm-suspend-config.test.ts b/cdk/test/handlers/shared/microvm-suspend-config.test.ts new file mode 100644 index 000000000..1dd37bb39 --- /dev/null +++ b/cdk/test/handlers/shared/microvm-suspend-config.test.ts @@ -0,0 +1,94 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +const mockSend = jest.fn(); +const mockWarn = jest.fn(); +jest.mock('@aws-sdk/client-ssm', () => ({ + SSMClient: jest.fn(() => ({ send: mockSend })), + GetParameterCommand: jest.fn((input: unknown) => ({ input })), +})); +jest.mock('../../../src/handlers/shared/logger', () => ({ logger: { warn: mockWarn } })); + +import { readMicrovmSuspendEnabled } from '../../../src/handlers/shared/microvm-suspend-config'; + +const parameterName = '/backgroundagent-dev/microvm-approval-suspend-enabled'; +const originalName = process.env.MICROVM_APPROVAL_SUSPEND_PARAMETER_NAME; +beforeEach(() => { + process.env.MICROVM_APPROVAL_SUSPEND_PARAMETER_NAME = parameterName; + mockSend.mockReset().mockResolvedValue({ Parameter: { Name: parameterName, Value: 'true' } }); + mockWarn.mockClear(); +}); +afterEach(() => { + jest.restoreAllMocks(); + if (originalName === undefined) delete process.env.MICROVM_APPROVAL_SUSPEND_PARAMETER_NAME; + else process.env.MICROVM_APPROVAL_SUSPEND_PARAMETER_NAME = originalName; +}); + +test('rereads the exact parameter and observes disable without an environment change', async () => { + expect(await readMicrovmSuspendEnabled({})).toBe(true); + mockSend.mockResolvedValue({ Parameter: { Name: parameterName, Value: 'false' } }); + expect(await readMicrovmSuspendEnabled({})).toBe(false); + expect(mockSend).toHaveBeenCalledTimes(2); + expect(mockSend).toHaveBeenLastCalledWith({ input: { Name: parameterName } }, { abortSignal: expect.any(AbortSignal) }); +}); + +test.each(['false', 'TRUE', ' true ', '', undefined])('does not enable on value %s', async value => { + mockSend.mockResolvedValue({ Parameter: { Name: parameterName, Value: value } }); + expect(await readMicrovmSuspendEnabled({})).toBe(false); +}); + +test.each([{}, { Parameter: { Name: '/other', Value: 'true' } }])('requires the requested parameter identity', async response => { + mockSend.mockResolvedValue(response); + expect(await readMicrovmSuspendEnabled({})).toBe(false); +}); + +test('missing configuration never starts a request', async () => { + delete process.env.MICROVM_APPROVAL_SUSPEND_PARAMETER_NAME; + expect(await readMicrovmSuspendEnabled({})).toBe(false); + expect(mockSend).not.toHaveBeenCalled(); +}); + +test('read failure disables savings and logs only safe error identifiers', async () => { + mockSend.mockRejectedValue(Object.assign(new Error('signed-url-secret'), { + name: 'AccessDeniedException', $metadata: { requestId: 'request-123' }, + })); + expect(await readMicrovmSuspendEnabled({})).toBe(false); + expect(JSON.stringify(mockWarn.mock.calls)).not.toContain('signed-url-secret'); + expect(mockWarn).toHaveBeenCalledWith(expect.any(String), { + error_type: 'AccessDeniedException', aws_request_id: 'request-123', + }); +}); + +test('an exhausted parent budget starts no request', async () => { + expect(await readMicrovmSuspendEnabled({ abortSignal: AbortSignal.abort() })).toBe(false); + expect(mockSend).not.toHaveBeenCalled(); +}); + +test.each(['parent', 'local'])('%s expiry aborts the actual SDK request and rejects a late true response', async source => { + const parent = new AbortController(); + const local = new AbortController(); + const timeout = jest.spyOn(AbortSignal, 'timeout').mockReturnValue(local.signal); + mockSend.mockImplementation(async (_command, options) => { + (source === 'parent' ? parent : local).abort(); + expect(options.abortSignal.aborted).toBe(true); + return { Parameter: { Name: parameterName, Value: 'true' } }; + }); + expect(await readMicrovmSuspendEnabled({ abortSignal: parent.signal })).toBe(false); + expect(timeout).toHaveBeenCalledWith(3_000); +}); diff --git a/cdk/test/handlers/shared/orchestrator.test.ts b/cdk/test/handlers/shared/orchestrator.test.ts index 590b40655..a358aa4fc 100644 --- a/cdk/test/handlers/shared/orchestrator.test.ts +++ b/cdk/test/handlers/shared/orchestrator.test.ts @@ -42,9 +42,9 @@ import { TaskStatus } from '../../../src/constructs/task-status'; import type { SessionHandle, SessionStatus } from '../../../src/handlers/shared/compute-strategy'; // The real classifier: the reason-append must not break the anchor the substrate // -failure classification keys on. -import { classifyError } from '../../../src/handlers/shared/error-classifier'; +import { classifyError, formatMicrovmTerminalFailure } from '../../../src/handlers/shared/error-classifier'; import { renderFailureReply, renderPanelFailureReason } from '../../../src/handlers/shared/failure-reply'; -import { buildComputeMetadata, reconcileMicrovmSubstrateState } from '../../../src/handlers/shared/orchestrator'; +import { buildComputeMetadata, finalizeTask } from '../../../src/handlers/shared/orchestrator'; import { toTaskDetail, type TaskRecord } from '../../../src/handlers/shared/types'; const MICROVM_ID = 'mvm-0123456789abcdef'; @@ -59,40 +59,51 @@ function commandsOfType(type: string): Array<{ _type: string; input: Record c._type === type); } -/** - * Prime the mocked doc client: the FIRST Get returns a task row with - * ``rereadStatus``; every Put/Update resolves empty. Mirrors the single re-read - * `reconcileMicrovmSubstrateState` performs before failing a task. - */ -function primeReread(rereadStatus: string): void { - mockDdbSend.mockImplementation((cmd: { _type: string }) => { - if (cmd._type === 'Get') { - return Promise.resolve({ - Item: { task_id: 'TASK001', user_id: 'user-1', repo: 'org/repo', status: rereadStatus }, - }); - } - return Promise.resolve({}); - }); +/** Strong finalization observes this committed task before choosing an outcome. */ +function primeReread(status: string): void { + mockDdbSend.mockImplementation((cmd: { _type: string }) => Promise.resolve(cmd._type === 'Get' ? { + Item: { + task_id: 'TASK001', + user_id: 'user-1', + repo: 'org/repo', + status, + memory_written: true, + compute_type: 'lambda-microvm', + session_id: MICROVM_ID, + compute_metadata: { microvmId: MICROVM_ID, endpoint: ENDPOINT }, + }, + } : {})); } - -const CORRELATION = { user_id: 'user-1', repo: 'org/repo' }; - -function reconcile(substrate: SessionStatus, ddbStatus: string, suspendAnomalyReported?: boolean) { - return reconcileMicrovmSubstrateState({ - taskId: 'TASK001', - ddbStatus: ddbStatus as never, - substrate, - microvmId: MICROVM_ID, - userId: 'user-1', - correlation: CORRELATION, - log: mockLogger, - repo: 'org/repo', - ...(suspendAnomalyReported !== undefined && { suspendAnomalyReported }), - }); +const mockRelease = jest.fn(); +jest.mock('../../../src/handlers/shared/task-concurrency', () => ({ + acquireTaskSlot: jest.fn(), releaseTaskSlot: (...args: unknown[]) => mockRelease(...args), +})); +function finish(substrate: SessionStatus, polledStatus = TaskStatus.RUNNING) { + return finalizeTask('TASK001', { + attempts: 1, + lastStatus: polledStatus, + microvmFailureReason: 'substrate-terminal', + microvmFailureMessage: formatMicrovmTerminalFailure( + substrate.status === 'failed' ? substrate.error : `substrate state ${substrate.status}`, substrate.reason, + ), + microvmSupervisor: { + version: 1, + microvmId: MICROVM_ID, + firstObservedAtMs: 1, + sessionDeadlineMs: 28_800_001, + lifetimeVerified: true, + consecutivePollFailures: 0, + consecutiveResumeFailures: 0, + anomalyReported: false, + nextPollInMs: 5_000, + }, + }, 'user-1'); } beforeEach(() => { jest.clearAllMocks(); + mockLogger.child.mockReturnValue(mockLogger); + mockRelease.mockResolvedValue(false); mockDdbSend.mockReset(); mockDdbSend.mockResolvedValue({}); }); @@ -172,347 +183,58 @@ describe('buildComputeMetadata', () => { }); }); -describe('reconcileMicrovmSubstrateState', () => { - describe('running substrate', () => { - test('is a no-op: no DDB reads, no events, task not failed', async () => { - const result = await reconcile({ status: 'running' }, TaskStatus.RUNNING); - - // `suspendAnomalyReported: false` RE-ARMS the once-per-episode event: a VM - // that resumed and is later suspended again earns a fresh anomaly event. - expect(result).toEqual({ taskFailed: false, suspendAnomalyReported: false }); - expect(mockDdbSend).not.toHaveBeenCalled(); - }); - }); - - describe('suspended substrate', () => { - test('is healthy while the task is AWAITING_APPROVAL — no event, no failure', async () => { - const result = await reconcile({ status: 'suspended' }, TaskStatus.AWAITING_APPROVAL); - - // The orchestrator-intended suspend during an approval wait is the whole - // economic point of the backend: it must be silent — and it re-arms the - // anomaly event, because leaving AWAITING_APPROVAL while still suspended - // would be a new, genuinely reportable episode. - expect(result).toEqual({ taskFailed: false, suspendAnomalyReported: false }); - expect(mockDdbSend).not.toHaveBeenCalled(); - expect(mockLogger.warn).not.toHaveBeenCalled(); - }); - - test('writes an anomaly event and does NOT fail the task when the status is RUNNING', async () => { - const result = await reconcile({ status: 'suspended' }, TaskStatus.RUNNING); - - expect(result).toEqual({ taskFailed: false, suspendAnomalyReported: true }); - - const puts = commandsOfType('Put'); - expect(puts).toHaveLength(1); - expect(puts[0].input.TableName).toBe('TaskEvents'); - const item = puts[0].input.Item as Record; - expect(item.event_type).toBe('microvm_suspend_anomaly'); - expect(item.task_id).toBe('TASK001'); - // Correlation envelope (#245) stamped as top-level fields. - expect(item.user_id).toBe('user-1'); - expect(item.repo).toBe('org/repo'); - expect(item.metadata).toEqual({ - microvm_id: MICROVM_ID, - task_status: TaskStatus.RUNNING, - reason: 'suspended_outside_approval_wait', - }); - - // Crucially: no status transition — a suspended VM is resumable, so - // failing the task would destroy recoverable work. - expect(commandsOfType('Update')).toHaveLength(0); - expect(mockLogger.warn).toHaveBeenCalled(); - }); - - test.each([ - TaskStatus.HYDRATING, - TaskStatus.RUNNING, - TaskStatus.FINALIZING, - ])('treats suspended + %s as an anomaly rather than a failure', async (status) => { - const result = await reconcile({ status: 'suspended' }, status); - - expect(result).toEqual({ taskFailed: false, suspendAnomalyReported: true }); - expect(commandsOfType('Put')[0].input.Item).toMatchObject({ - event_type: 'microvm_suspend_anomaly', - metadata: { task_status: status }, - }); - }); - - test('emits the anomaly event ONCE across repeated polls of the same episode', async () => { - // The poll runs every ~30 s for up to 8.5 h; without the flag an - // out-of-band suspend would write ~960 identical TaskEvents, burying the - // first informative one. The caller threads the returned flag back in. - let reported: boolean | undefined; - for (let poll = 0; poll < 5; poll += 1) { - const result = await reconcile({ status: 'suspended' }, TaskStatus.RUNNING, reported); - reported = result.suspendAnomalyReported; - // The no-fail-fast behaviour is unchanged on EVERY poll — that is the - // property the suppression must not break. - expect(result.taskFailed).toBe(false); - expect(result.suspendAnomalyReported).toBe(true); - } - - expect(commandsOfType('Put')).toHaveLength(1); - expect((commandsOfType('Put')[0].input.Item as Record).event_type) - .toBe('microvm_suspend_anomaly'); - // The WARN log is deliberately NOT suppressed: per-poll evidence is what a - // timeline investigation needs, and CloudWatch is not a user-facing surface. - expect(mockLogger.warn).toHaveBeenCalledTimes(5); - }); - - test('the repeat-suppressed polls record that the event was already reported', async () => { - await reconcile({ status: 'suspended' }, TaskStatus.RUNNING, true); - - expect(commandsOfType('Put')).toHaveLength(0); - expect(mockLogger.warn).toHaveBeenCalledWith( - expect.stringContaining('suspended while the task is not awaiting approval'), - expect.objectContaining({ anomaly_already_reported: true }), - ); - }); - - test('RE-ARMS after the VM resumes, so a second episode emits again', async () => { - // Recovery genuinely re-arms (documented decision): a flapping suspend loop - // is the pathology an operator most needs to see, and latching forever - // would hide it after the first occurrence. - const first = await reconcile({ status: 'suspended' }, TaskStatus.RUNNING, false); - expect(first.suspendAnomalyReported).toBe(true); - - const recovered = await reconcile({ status: 'running' }, TaskStatus.RUNNING, first.suspendAnomalyReported); - expect(recovered.suspendAnomalyReported).toBe(false); - - const second = await reconcile({ status: 'suspended' }, TaskStatus.RUNNING, recovered.suspendAnomalyReported); - expect(second.suspendAnomalyReported).toBe(true); - - // Two episodes → two events. - expect(commandsOfType('Put')).toHaveLength(2); - }); - - test('RE-ARMS when the task enters AWAITING_APPROVAL, so a later out-of-band suspend reports', async () => { - const first = await reconcile({ status: 'suspended' }, TaskStatus.RUNNING, false); - expect(first.suspendAnomalyReported).toBe(true); - - // The gate opened: this suspend is now the intended one. - const intended = await reconcile( - { status: 'suspended' }, TaskStatus.AWAITING_APPROVAL, first.suspendAnomalyReported); - expect(intended.suspendAnomalyReported).toBe(false); - expect(commandsOfType('Put')).toHaveLength(1); - - // The gate closed but the VM is still suspended — a new anomaly. - const third = await reconcile( - { status: 'suspended' }, TaskStatus.RUNNING, intended.suspendAnomalyReported); - expect(third.suspendAnomalyReported).toBe(true); - expect(commandsOfType('Put')).toHaveLength(2); - }); - - test('defaults to NOT-yet-reported when the caller omits the flag', async () => { - // Back-compat for any caller (and the first poll of every task) that has no - // prior state: the event must fire, not be suppressed by an undefined flag. - const result = await reconcile({ status: 'suspended' }, TaskStatus.RUNNING); - expect(result.suspendAnomalyReported).toBe(true); - expect(commandsOfType('Put')).toHaveLength(1); - }); +describe('MicroVM terminal finalization', () => { + test.each([ + ['MicroVM host unavailable.', 'MICROVM_SUBSTRATE_TERMINATED', 'compute', true], + ['capacity unavailable in this Availability Zone.', 'MICROVM_SUBSTRATE_TERMINATED', 'compute', true], + ['MicroVM unavailable in this region.', 'MICROVM_SUBSTRATE_TERMINATED', 'compute', true], + ['INSUFFICIENT_GITHUB_REPO_PERMISSIONS', 'MICROVM_SUBSTRATE_TERMINATED', 'compute', true], + ['BLOCKED[missing_secret]: diagnostic text', 'MICROVM_SUBSTRATE_TERMINATED', 'compute', true], + ["agent_status='success', build_ok=False", 'MICROVM_SUBSTRATE_TERMINATED', 'compute', true], + ["agent_status='success', build_ok=timeout [auto-retried]", 'MICROVM_SUBSTRATE_TERMINATED', 'compute', true], + ['Run lifecycle hook returned HTTP status 400.', 'MICROVM_RUN_HOOK_REJECTED', 'config', false], + ['Run lifecycle hook returned HTTP status 500.', 'MICROVM_SUBSTRATE_TERMINATED', 'compute', true], + ])('persists stable classification and consistent user guidance for %s', async (reason, code, category, retryable) => { + primeReread(TaskStatus.RUNNING); + await finish({ status: 'completed', reason }); + const values = commandsOfType('Update')[0].input.ExpressionAttributeValues as Record; + const errorMessage = String(values[':attr_error_message']); + expect(errorMessage).toMatch(new RegExp(`^${code}: `)); + expect(errorMessage).toContain(reason); + expect(toTaskDetail({ + task_id: 'TASK001', status: TaskStatus.FAILED, error_message: errorMessage, + } as TaskRecord).error_classification).toMatchObject({ category, retryable }); + const input = { status: TaskStatus.FAILED, errorMessage, taskId: 'TASK001' }; + for (const reply of [renderFailureReply(input), renderPanelFailureReason(input)]) { + expect(reply).toMatch(retryable ? /reply here to try again/i : /needs your ABCA admin/i); + expect(reply).not.toContain('Lambda MicroVMs is not available in this Region'); + expect(reply).not.toContain('I automatically tried again'); + } }); - describe('terminal substrate', () => { - test.each([ - ['MicroVM host unavailable.', 'MICROVM_SUBSTRATE_TERMINATED', 'compute', true], - ['capacity unavailable in this Availability Zone.', 'MICROVM_SUBSTRATE_TERMINATED', 'compute', true], - ['MicroVM unavailable in this region.', 'MICROVM_SUBSTRATE_TERMINATED', 'compute', true], - ['INSUFFICIENT_GITHUB_REPO_PERMISSIONS', 'MICROVM_SUBSTRATE_TERMINATED', 'compute', true], - ['BLOCKED[missing_secret]: diagnostic text', 'MICROVM_SUBSTRATE_TERMINATED', 'compute', true], - ["agent_status='success', build_ok=False", 'MICROVM_SUBSTRATE_TERMINATED', 'compute', true], - ["agent_status='success', build_ok=timeout [auto-retried]", 'MICROVM_SUBSTRATE_TERMINATED', 'compute', true], - ['Run lifecycle hook returned HTTP status 400.', 'MICROVM_RUN_HOOK_REJECTED', 'config', false], - ['Run lifecycle hook returned HTTP status 500.', 'MICROVM_SUBSTRATE_TERMINATED', 'compute', true], - ])('persists a stable failure code and consistent user guidance for %s', async (reason, code, category, retryable) => { - primeReread(TaskStatus.RUNNING); - await reconcile({ status: 'completed', reason }, TaskStatus.RUNNING); - const values = commandsOfType('Update')[0].input.ExpressionAttributeValues as Record; - const errorMessage = String(values[':attr_error_message']); - expect(errorMessage).toMatch(new RegExp(`^${code}: `)); - expect(errorMessage).toContain(reason); - expect(toTaskDetail({ - task_id: 'TASK001', status: TaskStatus.FAILED, error_message: errorMessage, - } as TaskRecord).error_classification).toMatchObject({ category, retryable }); - - const input = { status: TaskStatus.FAILED, errorMessage, taskId: 'TASK001' }; - for (const reply of [renderFailureReply(input), renderPanelFailureReason(input)]) { - expect(reply).toMatch(retryable ? /reply here to try again/i : /needs your ABCA admin/i); - expect(reply).not.toContain('Lambda MicroVMs is not available in this Region'); - expect(reply).not.toContain('I automatically tried again'); - } - }); - - test('fails the task when the re-read status is still non-terminal', async () => { - primeReread(TaskStatus.RUNNING); - - const result = await reconcile({ status: 'completed' }, TaskStatus.RUNNING); - - expect(result).toEqual({ taskFailed: true, suspendAnomalyReported: false }); - - // Re-read before acting (guards the "agent wrote terminal, VM torn down" - // race), then the FAILED transition. - expect(commandsOfType('Get')).toHaveLength(1); - const updates = commandsOfType('Update'); - expect(updates).toHaveLength(1); - expect(updates[0].input.TableName).toBe('Tasks'); - const values = updates[0].input.ExpressionAttributeValues as Record; - expect(values[':toStatus']).toBe(TaskStatus.FAILED); - expect(values[':fromStatus']).toBe(TaskStatus.RUNNING); - // The reason string is what error-classifier keys the substrate-failure - // classification on — keep the two in lockstep. - expect(values[':attr_error_message']).toBe( - 'MICROVM_SUBSTRATE_TERMINATED: MicroVM substrate terminated before the agent wrote a terminal status: substrate state completed', - ); - - // Plus the task_failed audit event. - const puts = commandsOfType('Put'); - expect(puts).toHaveLength(1); - expect((puts[0].input.Item as Record).event_type).toBe('task_failed'); - }); - - test('does NOT fail the task when the re-read shows the agent already wrote a terminal status', async () => { - primeReread(TaskStatus.COMPLETED); - - const result = await reconcile({ status: 'completed' }, TaskStatus.RUNNING); - - // Normal shutdown ordering: agent writes COMPLETED, exits, VM terminates. - // Without the re-read this would have failed a successful task. - expect(result).toEqual({ taskFailed: false, suspendAnomalyReported: false }); - expect(commandsOfType('Update')).toHaveLength(0); - expect(commandsOfType('Put')).toHaveLength(0); - }); - - test.each([ - TaskStatus.COMPLETED, - TaskStatus.FAILED, - TaskStatus.CANCELLED, - TaskStatus.TIMED_OUT, - ])('accepts a re-read terminal status of %s without failing the task', async (status) => { + test.each([TaskStatus.COMPLETED, TaskStatus.FAILED, TaskStatus.CANCELLED, TaskStatus.TIMED_OUT])( + 'preserves a committed %s winner after a stale active poll', async status => { primeReread(status); - - const result = await reconcile({ status: 'completed' }, TaskStatus.RUNNING); - - expect(result).toEqual({ taskFailed: false, suspendAnomalyReported: false }); - expect(commandsOfType('Update')).toHaveLength(0); - }); - - test('carries the substrate error detail into the failure reason', async () => { - primeReread(TaskStatus.RUNNING); - - const result = await reconcile({ status: 'failed', error: 'host fault' }, TaskStatus.RUNNING); - - expect(result).toEqual({ taskFailed: true, suspendAnomalyReported: false }); - const values = commandsOfType('Update')[0].input.ExpressionAttributeValues as Record; - expect(values[':attr_error_message']).toBe( - 'MICROVM_SUBSTRATE_TERMINATED: MicroVM substrate terminated before the agent wrote a terminal status: host fault', - ); - }); - - // --- stateReason in the detail (review B1) --- - - test('appends the substrate reason so the DOMINANT failure names its real cause', async () => { - // The exact live shape: a /run hook 4xx self-terminates the VM in ~12 s - // (645-p2-smoke-runbook.md §6.1). Without the reason this read "substrate - // state completed" and the classifier's remedy named a session duration cap, a - // host fault, or an external terminate — none of which happened. - primeReread(TaskStatus.RUNNING); - const reason = 'Run lifecycle hook returned HTTP status 400. Please check your hook endpoint ' - + 'and application logs for more details.'; - - await reconcile({ status: 'completed', reason }, TaskStatus.RUNNING); - - const values = commandsOfType('Update')[0].input.ExpressionAttributeValues as Record; - expect(values[':attr_error_message']).toBe( - 'MICROVM_RUN_HOOK_REJECTED: MicroVM substrate terminated before the agent wrote a terminal status: ' - + `substrate state completed (${reason})`, - ); - }); - - test('appends the reason to a failed substrate report too, without losing the error', async () => { - primeReread(TaskStatus.RUNNING); - - await reconcile( - { status: 'failed', error: 'host fault', reason: 'hypervisor evicted the guest' }, - TaskStatus.RUNNING, - ); - - const values = commandsOfType('Update')[0].input.ExpressionAttributeValues as Record; - expect(values[':attr_error_message']).toBe( - 'MICROVM_SUBSTRATE_TERMINATED: MicroVM substrate terminated before the agent wrote a terminal status: ' - + 'host fault (hypervisor evicted the guest)', - ); - }); - - test('keeps a stable code when the substrate supplies no reason', async () => { - primeReread(TaskStatus.RUNNING); - - await reconcile({ status: 'completed' }, TaskStatus.RUNNING); - - const values = commandsOfType('Update')[0].input.ExpressionAttributeValues as Record; - expect(values[':attr_error_message']).toBe( - 'MICROVM_SUBSTRATE_TERMINATED: MicroVM substrate terminated before the agent wrote a terminal status: substrate state completed', - ); - }); - - test('a hook 4xx selects the non-retryable failure code', async () => { - primeReread(TaskStatus.RUNNING); - await reconcile({ status: 'completed', reason: 'Run lifecycle hook returned HTTP status 400.' }, TaskStatus.RUNNING); - const values = commandsOfType('Update')[0].input.ExpressionAttributeValues as Record; - - const classification = classifyError(String(values[':attr_error_message'])); - - expect(classification!.title).toBe('The MicroVM rejected its own run payload'); - expect(classification!.retryable).toBe(false); - }); - - test('a NON-hook reason keeps the generic retryable substrate-failure entry', async () => { - // An ordinary service reason must not select the hook-rejection code. - primeReread(TaskStatus.RUNNING); - await reconcile( - { status: 'completed', reason: 'host fault (hypervisor evicted the guest)' }, - TaskStatus.RUNNING, - ); - const values = commandsOfType('Update')[0].input.ExpressionAttributeValues as Record; - - const classification = classifyError(String(values[':attr_error_message'])); - - expect(classification!.title).toBe('The MicroVM stopped before the agent reported a result'); - expect(classification!.retryable).toBe(true); - }); - - test('fails from AWAITING_APPROVAL too — a terminated VM cannot resume the gate', async () => { - primeReread(TaskStatus.AWAITING_APPROVAL); - - const result = await reconcile({ status: 'completed' }, TaskStatus.AWAITING_APPROVAL); - - expect(result).toEqual({ taskFailed: true, suspendAnomalyReported: false }); - const values = commandsOfType('Update')[0].input.ExpressionAttributeValues as Record; - expect(values[':fromStatus']).toBe(TaskStatus.AWAITING_APPROVAL); - expect(values[':toStatus']).toBe(TaskStatus.FAILED); - }); - - test('transitions from the RE-READ status, not the stale polled status', async () => { - // Task moved HYDRATING → RUNNING between the poll read and the re-read; the - // conditional transition must use the fresh value or it fails its own - // ConditionExpression and the task is left stuck. - primeReread(TaskStatus.RUNNING); - - await reconcile({ status: 'completed' }, TaskStatus.HYDRATING); - - const values = commandsOfType('Update')[0].input.ExpressionAttributeValues as Record; - expect(values[':fromStatus']).toBe(TaskStatus.RUNNING); - }); - - test('does not decrement concurrency — the finalize step owns the release', async () => { - primeReread(TaskStatus.RUNNING); - - await reconcile({ status: 'completed' }, TaskStatus.RUNNING); - - // Matches the ECS substrate-failure branch: failTask(..., releaseConcurrency=false). - const concurrencyWrites = commandsOfType('Update').filter( - c => c.input.TableName === 'Concurrency', - ); - expect(concurrencyWrites).toHaveLength(0); - }); + await finish({ status: 'completed' }); + expect(commandsOfType('Get')[0].input.ConsistentRead).toBe(true); + for (const command of commandsOfType('Update')) { + expect(command.input.ConditionExpression).toBeUndefined(); // terminal TTL stamp only + } + expect((commandsOfType('Put')[0].input.Item as Record).event_type).toBe(`task_${status.toLowerCase()}`); + expect(mockRelease).toHaveBeenCalledTimes(1); + }, + ); + + test('transitions from the strong current status, retaining the original service error', async () => { + primeReread(TaskStatus.AWAITING_APPROVAL); + await finish({ status: 'failed', error: 'host fault', reason: 'hypervisor evicted the guest' }); + const values = commandsOfType('Update')[0].input.ExpressionAttributeValues as Record; + expect(values[':fromStatus']).toBe(TaskStatus.AWAITING_APPROVAL); + expect(values[':toStatus']).toBe(TaskStatus.FAILED); + expect(values[':attr_error_message']).toBe( + 'MICROVM_SUBSTRATE_TERMINATED: MicroVM substrate terminated before the agent wrote a terminal status: host fault (hypervisor evicted the guest)', + ); + expect(classifyError(String(values[':attr_error_message']))?.retryable).toBe(true); + expect(mockRelease).toHaveBeenCalledWith('TASK001', 'user-1'); }); }); diff --git a/cdk/test/handlers/shared/session-lifecycle.test.ts b/cdk/test/handlers/shared/session-lifecycle.test.ts index edda851fa..c550f0c59 100644 --- a/cdk/test/handlers/shared/session-lifecycle.test.ts +++ b/cdk/test/handlers/shared/session-lifecycle.test.ts @@ -49,6 +49,91 @@ const microvm = handles[2] as Extract { jest.clearAllMocks(); mockMicrovmSend.mockReset().mockResolvedValue({}); + mockEcsSend.mockReset().mockResolvedValue({}); + mockAgentcoreSend.mockReset().mockResolvedValue({}); +}); + +describe.each(handles.filter(handle => handle.strategyType !== 'lambda-microvm'))( + '$strategyType caller budgets', handle => { + test('an expired budget prevents poll/stop requests', async () => { + const controller = new AbortController(); + controller.abort(new Error('caller deadline')); + const options = { abortSignal: controller.signal }; + const strategy = resolveComputeStrategy({ compute_type: handle.strategyType, runtime_arn: '' }); + await expect(strategy.pollSession(handle, options)).rejects.toThrow('caller deadline'); + await expect(strategy.stopSession(handle, options)).resolves.toBeUndefined(); + expect(mockEcsSend).not.toHaveBeenCalled(); + expect(mockAgentcoreSend).not.toHaveBeenCalled(); + }); + test('stop propagates caller cancellation into its pending SDK request', async () => { + const controller = new AbortController(); + const send = handle.strategyType === 'ecs' ? mockEcsSend : mockAgentcoreSend; + send.mockImplementationOnce((_command, options) => new Promise((_resolve, reject) => { + options.abortSignal.addEventListener('abort', () => reject(new Error('caller deadline'))); + })); + const strategy = resolveComputeStrategy({ compute_type: handle.strategyType, runtime_arn: '' }); + const result = strategy.stopSession(handle, { abortSignal: controller.signal }); + controller.abort(); + await expect(result).resolves.toBeUndefined(); + expect(send).toHaveBeenCalledTimes(1); + }); + }, +); + +describe.each(['pollSession', 'suspendSession', 'resumeSession', 'stopSession'] as const)( + '%s composed budget', operation => { + test('does not send a control request after its caller deadline', async () => { + const controller = new AbortController(); + controller.abort(new Error('caller deadline')); + const strategy = resolveComputeStrategy({ compute_type: 'lambda-microvm', runtime_arn: '' }); + const result = strategy[operation](microvm, { abortSignal: controller.signal }); + if (operation === 'stopSession') await expect(result).resolves.toMatchObject({ outcome: 'unconfirmed' }); + else await expect(result).rejects.toThrow('caller deadline'); + expect(mockMicrovmSend).not.toHaveBeenCalled(); + }); + test('a caller can end the in-flight request before the default limit', async () => { + const controller = new AbortController(); + mockMicrovmSend.mockImplementationOnce((_command, options) => new Promise((_resolve, reject) => { + options.abortSignal.addEventListener('abort', () => reject(new Error('caller deadline'))); + })); + const strategy = resolveComputeStrategy({ compute_type: 'lambda-microvm', runtime_arn: '' }); + const result = strategy[operation](microvm, { abortSignal: controller.signal }); + const assertion = operation === 'stopSession' + ? expect(result).resolves.toMatchObject({ outcome: 'unconfirmed' }) + : expect(result).rejects.toThrow('caller deadline'); + controller.abort(); + await assertion; + expect(mockMicrovmSend).toHaveBeenCalledTimes(1); + expect(mockMicrovmSend.mock.calls[0][1].abortSignal.aborted).toBe(true); + }); + }, +); + +describe('MicroVM lifetime observations', () => { + const startedAt = new Date('2026-09-15T10:00:00Z'); + test.each(['RUNNING', 'SUSPENDING', 'SUSPENDED', 'TERMINATED', 'UNKNOWN'])( + 'retains the original service lifetime in %s as durable JSON data', async state => { + mockMicrovmSend.mockResolvedValue({ state, startedAt, maximumDurationInSeconds: 28_800 }); + const strategy = resolveComputeStrategy({ compute_type: 'lambda-microvm', runtime_arn: '' }); + const observation = await strategy.pollSession(microvm); + expect(JSON.parse(JSON.stringify(observation))).toMatchObject({ + microvmStartedAtMs: startedAt.getTime(), microvmMaximumDurationSeconds: 28_800, + }); + }, + ); + test.each([ + { startedAt: new Date('invalid'), maximumDurationInSeconds: 28_800 }, + { startedAt, maximumDurationInSeconds: 0 }, + { startedAt, maximumDurationInSeconds: -1 }, + { startedAt, maximumDurationInSeconds: 1.5 }, + { maximumDurationInSeconds: 28_800 }, + ])('does not invent a lifetime from incomplete service data: %j', async lifetime => { + mockMicrovmSend.mockResolvedValue({ state: 'RUNNING', ...lifetime }); + const strategy = resolveComputeStrategy({ compute_type: 'lambda-microvm', runtime_arn: '' }); + const observation = await strategy.pollSession(microvm); + expect(observation.microvmStartedAtMs).toBeUndefined(); + expect(observation.microvmMaximumDurationSeconds).toBeUndefined(); + }); }); describe.each(['suspendSession', 'resumeSession'] as const)('%s contract', operation => { diff --git a/cdk/test/handlers/shared/strategies/lambda-microvm-strategy.test.ts b/cdk/test/handlers/shared/strategies/lambda-microvm-strategy.test.ts index 383dc47b5..a3d76da3c 100644 --- a/cdk/test/handlers/shared/strategies/lambda-microvm-strategy.test.ts +++ b/cdk/test/handlers/shared/strategies/lambda-microvm-strategy.test.ts @@ -814,7 +814,7 @@ describe('LambdaMicrovmComputeStrategy', () => { test.each([ ['ResourceNotFoundException', 'info'], - ['ConflictException', 'info'], + ['ConflictException', 'warn'], ['ThrottlingException', 'error'], ['AccessDeniedException', 'error'], ['InternalServerException', 'warn'], @@ -823,7 +823,10 @@ describe('LambdaMicrovmComputeStrategy', () => { err.name = errName; mockSend.mockRejectedValueOnce(err); - await expect(new LambdaMicrovmComputeStrategy().stopSession(makeHandle())).resolves.toBeUndefined(); + await expect(new LambdaMicrovmComputeStrategy().stopSession(makeHandle())).resolves.toEqual( + errName === 'ResourceNotFoundException' ? { outcome: 'not-found' } + : { outcome: 'unconfirmed', error_type: errName }, + ); const byLevel: Record = { info: mockLogger.info, diff --git a/cdk/test/stacks/agent.test.ts b/cdk/test/stacks/agent.test.ts index 962d334d5..bedb13c5c 100644 --- a/cdk/test/stacks/agent.test.ts +++ b/cdk/test/stacks/agent.test.ts @@ -1149,6 +1149,8 @@ describe('AgentStack with the Lambda MicroVMs substrate gate (--context compute_ const env = orchestrator.Properties.Environment.Variables as Record; expect(Object.keys(env).filter(k => k.startsWith('MICROVM_')).sort()).toEqual([ + 'MICROVM_APPROVAL_SUSPEND_ENABLED', + 'MICROVM_APPROVAL_SUSPEND_PARAMETER_NAME', 'MICROVM_EGRESS_CONNECTOR_ARNS', 'MICROVM_EXECUTION_ROLE_ARN', 'MICROVM_IMAGE_IDENTIFIER', @@ -1157,6 +1159,7 @@ describe('AgentStack with the Lambda MicroVMs substrate gate (--context compute_ ]); // Image version is deliberately unpinned. expect(env.MICROVM_IMAGE_VERSION).toBeUndefined(); + expect(env.MICROVM_APPROVAL_SUSPEND_ENABLED).toBe('false'); // Ingress is NOT empty and NOT omitted: RunMicrovm attaches a PUBLIC // HTTP_INGRESS connector (with a public endpoint) when the field is absent, // so "no inbound" is an explicit control on every launch. @@ -1191,6 +1194,8 @@ describe('AgentStack with the Lambda MicroVMs substrate gate (--context compute_ 'lambda:GetMicrovm', 'lambda:GetMicrovmImageVersion', 'lambda:TerminateMicrovm', + 'lambda:SuspendMicrovm', + 'lambda:ResumeMicrovm', ]); // Every MicroVM lifecycle action authorizes against the *image* resource, // which is why "scoped to platform-created images" is achievable at all. @@ -1208,10 +1213,10 @@ describe('AgentStack with the Lambda MicroVMs substrate gate (--context compute_ expect(passRole.Action).toBe('iam:PassRole'); }); - test('does NOT grant suspend/resume (P3) or auth-token minting (never)', () => { + test('grants supervisor sleep/wake without auth-token minting or shell access', () => { const rendered = JSON.stringify(template.toJSON()); - expect(rendered).not.toContain('lambda:SuspendMicrovm'); - expect(rendered).not.toContain('lambda:ResumeMicrovm'); + expect(rendered).toContain('lambda:SuspendMicrovm'); + expect(rendered).toContain('lambda:ResumeMicrovm'); expect(rendered).not.toContain('lambda:CreateMicrovmAuthToken'); expect(rendered).not.toContain('lambda:CreateMicrovmShellAuthToken'); expect(rendered).not.toContain('lambda:ConnectMicrovm'); diff --git a/docs/decisions/ADR-021-lambda-microvms-compute-backend.md b/docs/decisions/ADR-021-lambda-microvms-compute-backend.md index 690e5e046..d73fe49c1 100644 --- a/docs/decisions/ADR-021-lambda-microvms-compute-backend.md +++ b/docs/decisions/ADR-021-lambda-microvms-compute-backend.md @@ -1,6 +1,6 @@ # ADR-021: AWS Lambda MicroVMs as a third ComputeStrategy backend -> **Implementation status (2026-09-15):** P1 ([#689](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/689)) and P2 ([#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733)) are merged. The takeover branch now has a [clean deployment](../verification/645-p2-clean-deployment-20260913.md) with bootstrap 1.7.0 and a [verified managed-image update to 2.0](../verification/645-microvm-image-rebuild-20260914.md), plus [live coding, PR iteration and cancellation evidence](../verification/645-p2-live-task-20260914.md), without manual IAM workarounds. Remaining P2 failure/recovery and permission/network checks prevent full acceptance. Local P3 foundations include suspend/resume commands, explicit VM observations, guarded persistent intent, a policy helper, approval deadlines that survive a frozen clock, the [guest pause controller](../verification/645-p3-guest-barrier.md), and [scoped Claude credentials with retained-client refresh](../verification/645-p3-credentials.md). The [production HTTP hooks](../verification/645-p3-lifecycle-hooks.md) now connect the guest barrier to atomic checkpoints and credential/gate reconciliation locally. [Per-worker image capability](../verification/645-p3-image-capability.md) now declares the six hooks and retains/verifies the actual launched image version locally. Supervisor/approval-handler integration remains unfinished; automatic suspension is not enabled. This ADR defines P1–P3, not a P4. See the [current review](../verification/645-p3-readiness-review.md) and [P3 implementation plan](../verification/645-p3-implementation-plan.md). +> **Implementation status (2026-09-15):** P1 ([#689](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/689)) and P2 ([#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733)) are merged. The takeover branch now has a [clean deployment](../verification/645-p2-clean-deployment-20260913.md) with bootstrap 1.7.0 and a [verified managed-image update to 2.0](../verification/645-microvm-image-rebuild-20260914.md), plus [live coding, PR iteration and cancellation evidence](../verification/645-p2-live-task-20260914.md), without manual IAM workarounds. Remaining P2 failure/recovery and permission/network checks prevent full acceptance. Local P3 foundations include suspend/resume commands, explicit VM observations, guarded persistent intent, a policy helper, approval deadlines that survive a frozen clock, the [guest pause controller](../verification/645-p3-guest-barrier.md), and [scoped Claude credentials with retained-client refresh](../verification/645-p3-credentials.md). The [production HTTP hooks](../verification/645-p3-lifecycle-hooks.md) now connect the guest barrier to atomic checkpoints and credential/gate reconciliation locally. [Per-worker image capability](../verification/645-p3-image-capability.md) now declares the six hooks and retains/verifies the actual launched image version locally. [Supervisor/approval-handler integration](../verification/645-p3-supervisor.md) now adds durable recovery, bounded post-commit wake and scoped IAM locally. Live acceptance remains open; automatic suspension defaults off. This ADR defines P1–P3, not a P4. See the [current review](../verification/645-p3-readiness-review.md) and [P3 implementation plan](../verification/645-p3-implementation-plan.md). **Status:** proposed **Date:** 2026-07-29 @@ -60,19 +60,19 @@ Adopt **AWS Lambda MicroVMs as a third, opt-in `ComputeStrategy` backend** named ### 1. Strategy shape: extend the interface with mandatory suspend/resume -`ComputeType` widens to `'agentcore' | 'ecs' | 'lambda-microvm'` (mirrored in `cli/src/types.ts` and the CLI's inline unions). `SessionHandle` gains a `{ strategyType: 'lambda-microvm', microvmId, endpoint }` variant — `microvmId` because every lifecycle API (`suspend-microvm`, `resume-microvm`, `terminate-microvm`, `get-microvm`, and `create-microvm-auth-token` — the latter not called in P1–P3, see sub-decision 3) takes only the MicroVM identifier, and `endpoint` because it is per-session (minted by `RunMicrovm`) and required for any orchestrator→agent HTTP interaction. Note the naming seam: the handle field is `microvmId` (matching `RunMicrovmResponse.microvmId`), while the request key on every lifecycle command is `microvmIdentifier` — the strategy is the only place that translates between the two. **P3 refinement (2026-09-15):** the handle also retains the actual `imageArn` and `imageVersion` returned by Run, plus `lifecycleProtocol` after exact-version verification. The requested image remains deployment configuration, but the version that launched an existing worker is per-session evidence. Save the known handle before optional discovery; check that exact version with `GetMicrovmImageVersion`, then conditionally persist support in the receipt and `compute_metadata`. Current deployment settings cannot substitute for missing legacy evidence. +`ComputeType` widens to `'agentcore' | 'ecs' | 'lambda-microvm'` (mirrored in `cli/src/types.ts` and the CLI's inline unions). `SessionHandle` gains a `{ strategyType: 'lambda-microvm', microvmId, endpoint }` variant — `microvmId` because every lifecycle API (`suspend-microvm`, `resume-microvm`, `terminate-microvm`, `get-microvm`, and `create-microvm-auth-token` — the latter not called in P1–P3, see sub-decision 3) takes only the MicroVM identifier, and `endpoint` because it is per-session (minted by `RunMicrovm`) and required for any orchestrator→agent HTTP interaction. Note the naming seam: the handle field is `microvmId` (matching `RunMicrovmResponse.microvmId`), while the request key on every lifecycle command is `microvmIdentifier` — the strategy and inline approval-wake helper translate between the two at their control calls. **P3 refinement (2026-09-15):** the handle also retains the actual `imageArn` and `imageVersion` returned by Run, plus `lifecycleProtocol` after exact-version verification. The requested image remains deployment configuration, but the version that launched an existing worker is per-session evidence. Save the known handle before optional discovery; check that exact version with `GetMicrovmImageVersion`, then conditionally persist support in the receipt and `compute_metadata`. Current deployment settings cannot substitute for missing legacy evidence. **A second, sharper naming seam: `imageIdentifier` must be an ARN.** The name suggests a bare image name is acceptable — `create-microvm-image --name` takes one, and this ADR originally assumed `run-microvm --image-identifier` would too. It does not: a bare name is rejected with `ValidationException: Malformed ARN - doesn't start with 'arn:'`, and so is `list-microvm-image-builds --image-identifier ` (`Invalid ARN format`). The construct therefore resolves an operator-supplied name to its exact `arn:${Partition}:lambda:${Region}:${Account}:microvm-image:${Name}` ARN **once** — the same value it scopes the lifecycle IAM grant to — and injects THAT as `MICROVM_IMAGE_IDENTIFIER`. One derivation, two consumers, so a request field and an IAM resource can never disagree. The strategy validates the invariant and fails fast with the remedy, because the service's own error names neither the env var nor the fix. -The `ComputeStrategy` interface now has **mandatory** `suspendSession(handle)` / `resumeSession(handle)` methods returning `SessionLifecycleResult`, defined as `{ supported: false } | { supported: true }`. This local P3 foundation changes all three strategies together: AgentCore and ECS explicitly return false without calling AWS; MicroVM submits a command with a 10-second request bound. A future backend must implement both methods. The planned supervisor policy will use this explicit capability result. +The `ComputeStrategy` interface now has **mandatory** `suspendSession(handle)` / `resumeSession(handle)` methods returning `SessionLifecycleResult`, defined as `{ supported: false } | { supported: true }`. This local P3 foundation changes all three strategies together: AgentCore and ECS explicitly return false without calling AWS; MicroVM submits a command with a 10-second request bound. A future backend must implement both methods. The durable supervisor uses this explicit result without treating acknowledgment as final state. -`supported: true` means AWS acknowledged the command, not that the target state was reached. The SDK's suspend/resume response bodies are empty. All operational failures, including conflicts and not-found responses, propagate through sanitized errors; no generic conflict is normalized into success without observed-state evidence. A timed-out command may still have committed and requires reconciliation. The installed SDK has no `RESUMING` state. `SessionStatus.microvmState` now supplies explicit observations because the coarse `running` result also includes PENDING/unknown. The local policy/store uses retained generations and sticky wake intent to handle a delayed suspend after approval; the supervisor still needs to connect that policy, bound recovery and confirm wake state. See `docs/verification/645-lifecycle-intent.md`. +`supported: true` means AWS acknowledged the command, not that the target state was reached. The SDK's suspend/resume response bodies are empty. All operational failures, including conflicts and not-found responses, propagate through sanitized errors; no generic conflict is normalized into success without observed-state evidence. A timed-out command may still have committed and requires reconciliation. The installed SDK has no `RESUMING` state. `SessionStatus.microvmState` now supplies explicit observations because the coarse `running` result also includes PENDING/unknown. The local policy/store uses retained generations and sticky wake intent to handle a delayed suspend after approval; the durable supervisor now connects that policy, retains fixed recovery/session deadlines and waits for the required compute and guest observations. See `docs/verification/645-lifecycle-intent.md`. **Poll semantics — the strategy reports, the orchestrator interprets.** `pollSession(handle)` receives only the session handle and cannot see task state, so the health rules must live where the DynamoDB status lives. `SessionStatus` gains a `'suspended'` variant; the strategy maps `GetMicrovm` state mechanically and the **orchestrator** cross-references against the task row — the same division of labor `finalPollState` already uses for ECS (substrate stopped + non-terminal DynamoDB status → failed) and `pollTaskStatus` uses for agentcore heartbeats: substrate `suspended` + task `AWAITING_APPROVAL` is healthy (orchestrator-intended suspend); `suspended` with any other task status is an anomaly to surface, not fail-fast; substrate terminal + non-terminal task status → classify failed. **Liveness on this backend combines substrate state and the agent heartbeat.** `GetMicrovm` reports whether the VM is running or terminal. The independent `_heartbeat_worker` in `agent/src/server.py` refreshes the task timestamp every 45 seconds while the task is `RUNNING`; a stale timestamp detects loss of that writer, including a process crash or inability to write DynamoDB. It does **not** prove pipeline progress: a hung coding thread can coexist with a healthy heartbeat thread. A failed `/run` hook can trigger service teardown; after an accepted hook, ABCA still needs explicit termination and the maximum-duration backstop. A general progress watchdog remains separate work. -AgentCore and MicroVM tasks start the same periodic heartbeat worker, so they use the same 120-second grace and 240-second stale thresholds. ECS starts the pipeline directly, bypassing `server.py`; its pipeline writes a timestamp once but does not start this worker. Enabling the same stale check for ECS would therefore fail healthy tasks after roughly four minutes. The check applies only to `RUNNING`, not `AWAITING_APPROVAL`. The transition back to `RUNNING` refreshes `agent_heartbeat_at` in the same conditional transaction that clears the matching approval request. A poll immediately after a long gate therefore sees a fresh heartbeat before the next worker tick. P3 still needs the separate suspend/resume lifecycle and recovery policy described below. +AgentCore and MicroVM tasks start the same periodic heartbeat worker, so they use the same 120-second grace and 240-second stale thresholds. ECS starts the pipeline directly, bypassing `server.py`; its pipeline writes a timestamp once but does not start this worker. Enabling the same stale check for ECS would therefore fail healthy tasks after roughly four minutes. The check applies only to `RUNNING`, not `AWAITING_APPROVAL`. The transition back to `RUNNING` refreshes `agent_heartbeat_at` in the same conditional transaction that clears the matching approval request. A poll immediately after a long gate therefore sees a fresh heartbeat before the next worker tick. The local P3 supervisor additionally bounds recovery through suspension and decision consumption; live verification remains required. The service's `MicrovmState` enum has **six** members, not three, so the mapping is stated exhaustively (one line of rationale each, mirrored in the strategy's doc comment): @@ -122,10 +122,10 @@ Normative requirements (EARS, per [ADR-020](./ADR-020-ears-requirements-syntax.m The headline economic win is suspend during **HITL approval waits** (Cedar approval gates, [CEDAR_HITL_GATES.md](../design/CEDAR_HITL_GATES.md)): while a task waits on a human decision, the MicroVM is suspended (compute charges stop; memory/disk state — cloned repo, warm build caches — is preserved) and resumed when the decision lands. Under Cedar decision #6 the approval window is bounded (default 300 s, ceiling 1 h, timeout → deny), so the saving per gate is bounded at ~1 h of compute — real at 16 vCPU, and it makes any future extension of gate ceilings (the off-hours posture §14.8 deliberately defers) cheap on this backend. -The handshake must respect the existing approval mechanics: the agent **discovers decisions itself** by polling DynamoDB (`_poll_for_decision`, monotonic timeout), the approve/deny Lambda writes only the decision rows, and `AWAITING_APPROVAL` holds the concurrency slot (Cedar decision #7). Nothing "delivers" an approval to the agent, and suspension freezes the agent's monotonic clock — so the design is: +The handshake must respect the existing approval mechanics: the agent **discovers decisions itself** by polling DynamoDB (`_poll_for_decision`, monotonic timeout), the approve/deny Lambda commits the decision transaction before writing separate coordinator wake intent, and `AWAITING_APPROVAL` holds the concurrency slot (Cedar decision #7). Nothing "delivers" an approval to the agent, and suspension freezes the agent's monotonic clock — so the design is: - **Suspend — orchestrator-owned.** The orchestrator's durable poll observes `AWAITING_APPROVAL` on a `lambda-microvm` task and calls `suspendSession` after a grace period, and only when the gate's remaining window exceeds grace + resume overhead (suspending a 30 s gate is pure loss). Suspend is a policy decision on a poll observation, not a user action. -- **Resume — inline in the approve/deny Lambdas, orchestrator poll as backstop.** After the transactional decision write commits, `ApproveTaskFn`/`DenyTaskFn` load the MicroVM handle from the task row's `compute_metadata` (persisted at session start — the same field `cancel-task.ts` reads ECS handles from) via a post-commit strongly-consistent `GetItem`, then call `resumeSession` **best-effort**: on failure they log a warning and write a resume-orphan task event; the decision response never fails on a compute error (the decision row is already durable). The orchestrator poll reconciles: current approval row APPROVED/DENIED + MicroVM still `SUSPENDED` → retry resume; a PENDING row alone must not trigger wake-up. +- **Resume — inline in the approve/deny Lambdas, orchestrator poll as backstop.** After the transactional decision write commits, `ApproveTaskFn`/`DenyTaskFn` load the MicroVM handle from the task row's `compute_metadata` (persisted at session start — the same field `cancel-task.ts` reads ECS handles from) via a post-commit strongly-consistent `GetItem`, then save wake intent, observe the VM and request `ResumeMicrovm` **best-effort** only when SUSPENDED: on failure they log a warning and write a resume-orphan task event; the decision response never fails on a compute error (the decision row is already durable). The orchestrator poll reconciles: current approval row APPROVED/DENIED + MicroVM still `SUSPENDED` → retry resume; a PENDING row alone must not trigger wake-up. *Why inline rather than poll-only — codebase precedent:* resume-on-approve is structurally identical to task cancellation — a user-initiated, latency-sensitive action whose purpose is an immediate compute-lifecycle side effect. `cancel-task.ts` already resolves this exact tension: the API-plane handler invokes ECS `StopTask` / AgentCore `StopRuntimeSession` inline, best-effort (a failed stop logs a warning and the state transition stands; a `task_cancel_compute_orphan` event is written when no stoppable compute handle exists, `reason: missing_runtime_handle`) — with the conditional IAM wired in `task-api.ts`. The resume path goes one step further than the precedent by also writing the orphan event on *failed* resume calls, because a failed resume strands a suspended VM awaiting a decision — a stronger liveness consequence than a failed stop of an already-cancelled task. The alternative (orchestrator-poll-only resume) preserves single-owner lifecycle purity but pays up to a full poll interval (~30 s) of latency on every approval, and the purity argument was already litigated and declined for cancel. `approve-task.ts` is deliberately minimal today (security-critical ownership comparison, Cedar finding #6); the resume call is therefore added *after* the transaction commits, cannot alter the decision outcome, and carries one conditional `lambda:ResumeMicrovm` grant — the same blast-radius trade the cancel handler accepted in review. - **Timeout under freeze — the agent re-bases on the wall clock it already owns.** The agent's monotonic gate timer freezes while suspended, so resuming near the deadline is not enough: the frozen timer would still hold its remaining budget and fire the deny minutes *after* the user-visible window — colliding with the approval row's TTL (`created_at + timeout_s + 120s`) and triggering the "row reaped → stranded" fallback on a healthy gate. Instead, the gate expires at **`min(monotonic budget, created_at + timeout_s)`**, evaluated on each poll iteration and on `/resume`. This is not a new principle: Cedar decision #6 is already "min wins" for timeouts, the wall-clock deadline is already durable in the approval row the agent itself writes (`created_at` is recorded by the agent; clock changes across suspend still need testing), and §13.12's late-approval race fix already establishes that the durable row is authoritative over the agent's local timer. Deny authority stays agent-side (the conditional `TIMED_OUT` write + ConsistentRead re-read race protection is untouched); the orchestrator's resume at `deadline − margin` is purely the wake-up mechanism, with no correctness role. @@ -136,7 +136,7 @@ The handshake must respect the existing approval mechanics: the agent **discover It does **not** relieve the orchestrator of anything, because the two cases are disjoint. The service reaps a hook *result* it did not like; it has no view of the guest once the hook returned 200. So a task that starts normally — the overwhelming majority — has no service-side reaper at all, and a VM whose pipeline finished, crashed after `/run`, or hung is reaped by nobody but `TerminateMicrovm`. A leaked handle therefore remains a cost incident that bills until the 8 h cap; only the "the guest rejected its own payload" corner now cleans itself up. - **Concurrency slot stays held** during suspend. Cedar decision #7's rationale ("container alive, consuming memory") weakens under suspend, and the harder replacement rationale — "AWS counts `SUSPENDED` MicroVMs toward the account memory quota, so releasing ABCA's slot would not free real capacity" — is **undischarged**: the suspended VM stayed in `list-microvms` at every checkpoint, but that only proves *listed*. `L-CD1C0CC4` (1024 GB, account-scoped) exposes no `UsageMetric`, `AWS/Usage` carries only `CallCount` per API, and no MicroVM memory metric exists in any namespace, so consumption is **not observable safely** — proving it would need a large concurrent fleet. The conclusion (hold the slot) stands as the conservative choice, not as a verified fact. Size the arithmetic against the 32 GiB **peak** rather than the 8 GiB baseline: a busy fleet scales up, so peak is what actually competes for the account quota. -In P3, the agent's `/suspend` hook must acknowledge durable progress before returning 200 within its hook budget; `/resume` must refresh credentials and reseed application PRNG state from fresh OS entropy. These hooks are not implemented yet. A non-cryptographic PRNG such as Python `random` remains unsuitable for secrets even after reseeding; security-sensitive randomness must use an OS-backed cryptographic source. +In P3, the agent's `/suspend` hook must acknowledge durable progress before returning 200 within its hook budget; `/resume` must refresh credentials and reseed application PRNG state from fresh OS entropy. These hooks are implemented locally with atomic checkpoints, retained credential refresh and original-gate reconciliation; matching deployment and live acceptance remain open. A non-cryptographic PRNG such as Python `random` remains unsuitable for secrets even after reseeding; security-sensitive randomness must use an OS-backed cryptographic source. **Amended (P2 review): Application PRNG reseeding is P3 scope, and the P2 exposure is negligible.** The original EARS requirement below implied a `/run`-time reseed had to land with P2; it did not, and shipping P2 with the requirement unmet-but-asserted was itself the defect. Measured exposure, which is what changed the scoping: @@ -148,8 +148,8 @@ So the reseed moves to P3 alongside `/suspend` + `/resume`, where a *resumed* VM Normative requirements (EARS): -- While a `lambda-microvm` task is in `AWAITING_APPROVAL` and the gate's remaining window exceeds the configured grace period plus resume overhead, the orchestrator shall call `suspendSession` after the grace period. -- When the approve or deny Lambda commits a decision for a `lambda-microvm` task, the Lambda shall load the MicroVM handle from `compute_metadata` and call `resumeSession` best-effort. +- While approval suspension is enabled for a verified compatible worker, a `lambda-microvm` task is in `AWAITING_APPROVAL`, and the gate's remaining window exceeds the configured grace period plus resume overhead, the orchestrator shall call `suspendSession` after the grace period. +- When the approve or deny Lambda commits a decision for a `lambda-microvm` task, the Lambda shall load the MicroVM handle from `compute_metadata`, persist wake intent for the current identity, and request `ResumeMicrovm` best-effort after observing SUSPENDED and rechecking the gate. - If the inline resume fails, then the Lambda shall record a resume-orphan task event and shall still return the decision outcome. - While the current approval row is APPROVED or DENIED and the MicroVM remains `SUSPENDED`, the orchestrator shall retry `resumeSession`. A PENDING row alone is not a wake-up condition. - While a `lambda-microvm` task waits on an approval gate, the agent shall evaluate gate expiry as the earlier of its monotonic budget and the row's wall-clock deadline (`created_at + timeout_s`), on each poll iteration and on `/resume`. @@ -173,8 +173,8 @@ The phasing is therefore: | `/ready` | **P1** (construct enables `hooks.microvmImageHooks.ready`) | **P1** | MANDATORY, not a quality nicety — see above. A 200 proves uvicorn is bound and `server` imported cleanly (pulling in `pipeline` → `runner` → the policy engine), so a missing policy file fails the BUILD instead of the first task. **Since P2-F5 it also WARMS the snapshot** — the hook's 200 is what the service waits for before capturing the snapshot, making this the only place a warm page can be created, and the 225 MiB `claude` binary was cold in it (see the P2-F5 correction below). A required warm-up failure answers 503, so a snapshot that cannot exec the agent's own CLI fails the image build instead of every task. Still makes ZERO AWS calls, logging included (a `--version` exec is neither an AWS call nor a network call). | | `/run` | **P1** (construct sets `hooks.microvmHooks.run`) | **P1** | The payload-delivery channel. Must be served in P1 because `/ready` forces hooks to exist at all, and a hook-less image cannot accept `runHookPayload`. Since P2 it is also the **platform-configuration** channel (see "Platform configuration delivery" below). | | `/validate` | **P2** (construct sets `hooks.microvmImageHooks.validate`) | **P2** | An **image** (build-time) hook, and a **shallow self-check only**: server alive, hook routes registered, interpreter + contract sanity. It runs under the BUILD role, which deliberately holds no Bedrock / Secrets Manager / DynamoDB grants, so it must make **zero AWS API calls** and must not touch credential resolution — the "deeper warm-up assertions (Bedrock reachability, Memory access, tool availability)" this ADR originally assigned here are **not implementable**: every one of them would `AccessDenied` and fail every image build. They belong to the first task's own error handling. 200 when the checks pass, 503 while still initialising. | -| `/terminate` | **P2** (construct sets `hooks.microvmHooks.terminate`) | **P2** | Best-effort final flush: a final structured log line, then 200 — always, inside the hook budget, even with nothing running. It must **not** write terminal task status (the orchestrator finalizes the task and *then* calls `TerminateMicrovm`, so a terminate hook that wrote a status would race that finalization and could clobber the real outcome) and must not join the pipeline thread. `ProgressWriter` writes synchronously but drops failed events, so this hook cannot guarantee that every progress write succeeded. P3 supplies a separate acknowledged checkpoint for suspension. "Always 200" also covers the BODY: the handler reads the raw request rather than a typed model, because a typed body is validated before the handler runs and would answer 422 to malformed JSON — a reported hook failure on a successful teardown. Safe to declare because `TerminateMicrovm` removes the VM with or without in-guest cooperation. **Correction (P2-F8):** the service sends `microvmId: ""` on this hook, unlike `/run` where it is populated, so an empty id is expected-normal and this hook cannot join the guest record to the control-plane one — `/run`'s accepted line carries that correlation instead. | -| `/suspend`, `/resume` | **P3**, implemented locally | **P3**, implemented locally | Managed images declare the served checkpoint/credential/gate hooks with the shared 30-second service timeout and protocol marker. The coordinator verifies the actual launched version before allowing new suspension. Supervisor integration and live acceptance remain open. | +| `/terminate` | **P2** (construct sets `hooks.microvmHooks.terminate`) | **P2** | Best-effort final flush: a final structured log line, then 200 — always, inside the hook budget, even with nothing running. It must **not** write terminal task status (the orchestrator normally finalizes before `TerminateMicrovm`, and also attempts cleanup if database finalization fails, so a terminate hook that wrote a status would race that finalization and could clobber the real outcome) and must not join the pipeline thread. `ProgressWriter` writes synchronously but drops failed events, so this hook cannot guarantee that every progress write succeeded. P3 supplies a separate acknowledged checkpoint for suspension. "Always 200" also covers the BODY: the handler reads the raw request rather than a typed model, because a typed body is validated before the handler runs and would answer 422 to malformed JSON — a reported hook failure on a successful teardown. Safe to declare because `TerminateMicrovm` removes the VM with or without in-guest cooperation. **Correction (P2-F8):** the service sends `microvmId: ""` on this hook, unlike `/run` where it is populated, so an empty id is expected-normal and this hook cannot join the guest record to the control-plane one — `/run`'s accepted line carries that correlation instead. | +| `/suspend`, `/resume` | **P3**, implemented locally | **P3**, implemented locally | Managed images declare the served checkpoint/credential/gate hooks with the shared 30-second service timeout and protocol marker. The coordinator verifies the actual launched version before allowing new suspension. Supervisor integration is implemented locally with a default-off rollout flag; live acceptance remains open. | Consequence to state plainly, replacing the original "a P1-built MicroVM image is not runnable end to end": **a P1 image is creatable, launchable and payload-deliverable, but carries no smoke-parity guarantee.** P1 delivers the strategy, the construct, the roles/buckets/connectors, the image resource, the packaging script, and the `/ready` + `/run` endpoints — so a `lambda-microvm` task can start a MicroVM and hand it a payload. What P1 has **not** established is anything P2 owns: AgentCore Memory grants and `MEMORY_ID` delivery, the agent's non-secret env parity inside the snapshot, egress specifics from a running MicroVM, and heartbeat/progress behaviour end to end. At the P1 handoff, no clone → change → PR run had happened. P2 initially demonstrated that path with an IAM workaround; the 2026-09-14 clean deployment and coding/iteration/cancellation runs later passed without manual IAM changes. The broader failure/recovery, effective IAM and networking matrix remains open. The construct and the packaging script surface that remaining scope at synth/run time (`abca:microvm-image-p1-smoke-unverified`) so an operator cannot mistake a launchable substrate for a verified one. @@ -402,7 +402,7 @@ Lambda MicroVMs launched in 5 regions (us-east-1/2, us-west-2, eu-west-1, ap-nor - Unit tests for the strategy: start/poll/stop mapping (including `SessionStatus` `'suspended'` reported mechanically, without task-state interpretation), v2 reference-size bounds (4,096 bytes), immutable signed-reference replay and scoped payload transport, the image-identifier-must-be-an-ARN guard, the explicit `NO_INGRESS` argument (including the blank-env-var fallback, which must never omit the field), error classification (`ServiceQuotaExceededException`, `ThrottlingException`, `ResourceNotFoundException`, regional-unavailability), the omit-`idlePolicy` invariant, and `maximumDurationInSeconds` fixed at 28 800. - Agent tests: `/ready` returns 200 after required local warm-up succeeds, without starting a task or calling AWS; `/run` accepts authenticated v2 references and rejects old unsigned envelopes, starts the pipeline asynchronously through the same mapper `/invocations` uses, returns before the pipeline finishes, and rejects every unusable envelope with a named code before spawning; `/suspend` and `/resume` are NOT served (`/validate` and `/terminate` joined the served set in P2). - Orchestrator tests: substrate-terminal + non-terminal task status → failed classification; `suspended` + non-`AWAITING_APPROVAL` status → anomaly event, no fail-fast; `compute_metadata` persisted with `microvmId`/`endpoint` after `startSession`. -- CDK assertions: MicroVM resources present only when `ComputeTypes` includes the backend; synth failure for unsupported regions (plus the context-flag escape hatch); memory size validated against the accepted list at synth; the connector operator role and its trust; two connectors with the build-only one carrying port 80 and the runtime one not; the `/ready` + `/run` hook declaration and the absence of the others; IAM actions scoped as specified (orchestrator lifecycle set; no `CreateMicrovmAuthToken` anywhere); payload-bucket grants (execution role read-only); backend cost-allocation tags; types-sync check covers the widened `ComputeType`. +- CDK assertions: MicroVM resources present only when `ComputeTypes` includes the backend; synth failure for unsupported regions (plus the context-flag escape hatch); memory size validated against the accepted list at synth; the connector operator role and its trust; two connectors with the build-only one carrying port 80 and the runtime one not; the current six-hook declaration, shared protocol marker and per-worker capability admission; IAM actions scoped as specified (orchestrator lifecycle set; no `CreateMicrovmAuthToken` anywhere); payload-bucket grants (execution role read-only); backend cost-allocation tags; types-sync check covers the widened `ComputeType`. - CLI tests: onboarding rejection with remedy when the availability probe fails; doctor check present when a blueprint selects the backend. - P1 verification items (external service facts) — **executed 2026-07-31, us-east-1**; see `docs/verification/645-p1-lambda-microvm-runbook.md` for the full evidence. Discharged: `runHookPayload` limit (**4 096**, not 16 KB), the accepted baseline memory sizes (`[512…8192]` MiB — note the developer guide, not the probe, is what establishes that this is a BASELINE with a 32 GiB peak), image-identifier ARN requirement, IAM action names and the observed image-ARN shape, region probe behaviour, manual suspend/resume without `idlePolicy`, terminate timing and the `TERMINATED`-persists-≥10-min finding, the default public `HTTP_INGRESS`, and the `/ready` requirement. **Not** discharged: account-quota treatment of `SUSPENDED` MicroVMs (not observable safely), suspended TTL beyond 1 h (truncated), the vertical-scaling behaviour itself (no workload here approached the baseline, so the 4× peak is documented rather than observed), and the `AWS::Lambda::MicrovmImage` CloudFormation value shapes (never exercised — the run used the out-of-band script path; **discharged, and REFUTED, by the P2 run — see P2-F2 in sub-decision 3**). Record the closed answers in COMPUTE.md. diff --git a/docs/design/CEDAR_HITL_GATES.md b/docs/design/CEDAR_HITL_GATES.md index b41c329fe..90c8008f6 100644 --- a/docs/design/CEDAR_HITL_GATES.md +++ b/docs/design/CEDAR_HITL_GATES.md @@ -121,7 +121,7 @@ Settled during the 2026-04-23 design discussion and extended after the 2026-04-2 | 4 | **Scope allowlist: in-process, seeded from persisted `initial_approvals`** | Runtime escalation lives in the `PolicyEngine` instance. Submit-time `--pre-approve` flags persist on TaskTable and seed the allowlist at container startup. Lost on restart (rare; reconciler fails stranded tasks). | | 5 | **CLI UX: standalone `bgagent approve/deny` + `--pre-approve ` + `bgagent policies list` + `bgagent pending`** | No inline interactive prompt in the streaming CLI for v1. Discovery + listing commands solve the request_id/rule_id copy problem. | | 6 | **Timeouts: per-task default + per-rule Cedar annotation override, min wins, bounded floor + ceiling, fail-closed** | Per-task default: **300s** (5 min), overridable via `--approval-timeout` on submit and bounded by `[30, min(3600, maxLifetime - 300)]`. Floor: 30s (engine-enforced on both task default and rule annotations). Ceiling: `min(1h, maxLifetime_remaining - cleanup_margin)` — sized so the TTL on the approval row always covers the decision window. On timeout → deny (never auto-approve). See §14.8 for the off-hours trade-off this posture deliberately accepts. | -| 7 | **Concurrency slots: AWAITING_APPROVAL holds the slot** | Matches PAUSED semantics. Container is alive, consuming memory. | +| 7 | **Concurrency slots: AWAITING_APPROVAL holds the slot** | Bounds unfinished sessions and their eventual resume demand, including a suspended MicroVM. This is an ABCA admission policy; suspended AWS memory-quota consumption remains unverified. | | 8 | **Hard-deny is absolute** | No `--pre-approve` scope, and no blueprint `disable:` directive, can bypass it. CreateTaskFn validates and rejects `rule:`; blueprint loader rejects `disable:` entries that name built-in hard-deny rules. | | 9 | **Submit-time scope cap: 20 entries, ≤128 chars each** | Keeps audit trail legible, bounds allowlist check cost, limits abuse-vector damage. | | 10 | **Cedar annotations (verified working)** | `@rule_id(...)`, `@tier(...)`, `@approval_timeout_s(...)`, `@severity(...)`, `@category(...)`. Recoverable via `cedarpy.policies_to_json_str()` → JSON. Multi-match merging: min timeout wins (clamped by floor), max severity wins. | @@ -1307,7 +1307,7 @@ stateDiagram-v2 **AWAITING_APPROVAL holds the user's concurrency slot.** -Rationale: the Docker container is alive. Memory allocated. The AgentCore microVM pool is committed. Releasing the slot while the resource is still held lies to accounting and opens a resource-exhaustion vector. +Rationale: the task still owns an unfinished compute session and may resume work. Retaining its ABCA reservation prevents an unbounded collection of parked tasks from bypassing admission control. The rule also applies to P3 Lambda MicroVM suspension; suspended AWS memory-quota consumption remains unverified and is not the basis for claiming quota usage. Resume and terminal cleanup use the existing task-owned reservation protocol. Concrete behavior: diff --git a/docs/design/COMPUTE.md b/docs/design/COMPUTE.md index d50b215bf..6dacd81a1 100644 --- a/docs/design/COMPUTE.md +++ b/docs/design/COMPUTE.md @@ -83,9 +83,9 @@ For managed images, run `cdk/scripts/package-microvm-artifact.sh --stack-name **Operators must re-bootstrap for this.** The statement ships in bootstrap policy bundle **1.6.0**; a CDKToolkit stack bootstrapped at 1.5.0 or earlier will fail the CDK-managed MicroVM image deploy with a caller-side `iam:PassRole` AccessDenied on the build role. Check `CDKToolkit`'s `BootstrapPolicyVersion` output, and re-run `mise //cdk:bootstrap` (with `ComputeTypes` including `lambda-microvm`) if it is behind. +P3 additionally requires **bundle 1.8.0** for `MicrovmSuspendConfiguration`. +This statement lets CloudFormation manage and tag the live suspension setting +at `//microvm-approval-suspend-enabled`. The +coordinator gets only `GetParameter` on its exact parameter. Existing durable +executions retain their Lambda version and reread this setting before new +suspension, so disable can reach executions already running. + ```json { "Statement": [ @@ -881,6 +888,19 @@ The second statement, `MicrovmPassRoles`, is the one exception to the rule that "arn:aws:iam::*:role/backgroundagent-dev-LambdaMicrovmComputeConnector*" ], "Sid": "MicrovmPassRoles" + }, + { + "Action": [ + "ssm:GetParameters", + "ssm:PutParameter", + "ssm:DeleteParameter", + "ssm:AddTagsToResource", + "ssm:RemoveTagsFromResource", + "ssm:ListTagsForResource" + ], + "Effect": "Allow", + "Resource": "arn:aws:ssm:*:*:parameter/backgroundagent-*/microvm-approval-suspend-enabled", + "Sid": "MicrovmSuspendConfiguration" } ], "Version": "2012-10-17" diff --git a/docs/design/ORCHESTRATOR.md b/docs/design/ORCHESTRATOR.md index 80325b8b9..a1ede8f3d 100644 --- a/docs/design/ORCHESTRATOR.md +++ b/docs/design/ORCHESTRATOR.md @@ -96,7 +96,7 @@ stateDiagram-v2 AWAITING_APPROVAL --> RUNNING : Approved or denied (resume) AWAITING_APPROVAL --> CANCELLED : User cancels mid-approval - AWAITING_APPROVAL --> FAILED : Stranded-approval reconciler + AWAITING_APPROVAL --> FAILED : Infrastructure loss or stranded wait FINALIZING --> COMPLETED : PR or commits found FINALIZING --> FAILED : No useful work @@ -121,11 +121,11 @@ stateDiagram-v2 | `HYDRATING` | `FAILED` | Hydration error | GitHub API failure, guardrail blocks content, Bedrock unavailable | | `RUNNING` | `AWAITING_APPROVAL` | Cedar soft-deny gate fires | Tool call triggers a soft-deny policy rule during execution | | `RUNNING` | `FINALIZING` | Session ends | Response received or session terminated | -| `RUNNING` | `TIMED_OUT` | Max duration exceeded | AgentCore and Lambda MicroVMs have an 8h substrate cap; the orchestrator's own safety-net poll window is `MAX_POLL_ATTEMPTS` (1020) × 30s ≈ 8.5h, after which a still-`RUNNING` task is driven to `TIMED_OUT` | +| `RUNNING` | `TIMED_OUT` | Max duration exceeded | AgentCore and Lambda MicroVMs have an 8h substrate cap; MicroVM supervision retains the original service deadline across replay; other backends retain the 1,020-attempt safety window (about 8.5h at 30s) | | `RUNNING` | `FAILED` | Session crash | Heartbeat or substrate liveness lost (see Liveness monitoring) | | `AWAITING_APPROVAL` | `RUNNING` | Approved or denied | Human decision received; agent resumes | | `AWAITING_APPROVAL` | `CANCELLED` | User cancels | Explicit cancel while awaiting approval | -| `AWAITING_APPROVAL` | `FAILED` | Stranded reconciler | Approval request orphaned (agent died mid-wait) | +| `AWAITING_APPROVAL` | `FAILED` | Infrastructure failure or stranded wait | Lost compute, exhausted supervisor recovery/window, or an orphaned approval; the approval decision is not rewritten | | `FINALIZING` | `COMPLETED` | Success inferred | PR exists or commits on branch | | `FINALIZING` | `FAILED` | Failure inferred | No commits, no PR, or agent reported error | @@ -152,7 +152,7 @@ Multiple timeout mechanisms work together to prevent runaway tasks. Substrate ti | Type | Default | Effect | |---|---|---| -| Max session duration | 8 hours | AgentCore caps a session at 8h; Lambda MicroVMs use `maximumDurationInSeconds: 28,800`, including suspended time. The orchestrator's safety-net poll loop runs up to `MAX_POLL_ATTEMPTS` (1020) × 30s ≈ 8.5h; a task still `RUNNING` when that window is exhausted is driven to `TIMED_OUT`. | +| Max session duration | 8 hours | AgentCore caps a session at 8h; Lambda MicroVMs use `maximumDurationInSeconds: 28,800`, including suspended time. MicroVM uses the saved absolute service deadline, so fast transition polling cannot shorten the session. Other backends retain the 1,020-attempt safety window. An exhausted approval wait uses FAILED, its allowed infrastructure-failure transition. | | Idle timeout | Backend-specific | AgentCore has an idle timeout. Lambda MicroVMs omit `idlePolicy` because inbound-traffic idleness would suspend an outbound-only agent while it is working. See Liveness monitoring. | | Max turns | 100 (range 1-500) | Agent stops after N model invocations. Configurable per task or per repo. | | Max cost budget | $0.01-$100 | Agent stops when budget is reached. Per-task or per-repo via Blueprint. | @@ -223,7 +223,7 @@ The orchestrator polls for completion using `waitForCondition` from the Durable | ECS | `DescribeTasks`, including container exit status and exit code | | Lambda MicroVMs | `GetMicrovm` state plus agent heartbeat | -While waiting between polls, the durable orchestrator suspends without compute charges. If the session is terminated externally (crash, timeout, cancellation), the poll detects it and the orchestrator proceeds to finalization using GitHub-based result inference as fallback. +While waiting between polls, the durable orchestrator suspends without compute charges. If the session is terminated externally (crash, timeout, cancellation), the poll detects it and the orchestrator proceeds to finalization after a strongly consistent task read; it preserves an already committed terminal result. ### Step 6: Finalization @@ -292,10 +292,12 @@ When the session is unhealthy, the task transitions to `FAILED` with "Agent sess **Lambda MicroVM state polling.** Liveness is a dual signal. The strategy maps `GetMicrovm` mechanically: `PENDING`/`RUNNING` report `running`, `SUSPENDING`/`SUSPENDED` report `suspended`, and `TERMINATING`/`TERMINATED` report terminal completion. The orchestrator supplies the health interpretation: -- `suspended` is healthy only while the task is `AWAITING_APPROVAL`; in any other task state it emits an anomaly and keeps polling rather than failing recoverable work. -- A terminal substrate report paired with a non-terminal task is a failure, but the orchestrator first re-reads the task row to confirm the agent did not write a terminal result between the original read and VM termination. +- Intentional suspension requires the matching pending gate and saved suspend intent. Unexpected suspension emits one anomaly per episode and starts bounded wake recovery, preserving recoverable work. +- A terminal substrate report paired with a non-terminal task is a failure, but finalization first strongly re-reads the task row to confirm the agent did not write a terminal result between the original read and VM termination. - Substrate state detects a dead VM; heartbeat staleness detects loss of the heartbeat writer inside a VM that still reports `RUNNING`. The independent heartbeat thread can continue during a pipeline hang, so a fresh timestamp is not proof of progress. +The P3 supervisor saves intent before control calls and rechecks the gate before and after them. Its durable state retains an absolute service lifetime, consecutive failures, recovery start time and next delay. Three failed cycles or 120 seconds of unconfirmed wake cannot become an indefinite wait. AWS RUNNING does not end recovery while the guest remains stuck on a decided/expired approval; fresh guest liveness is required. API approve/deny commit first, then attempt a bounded wake without changing the decision response. Automatic suspension defaults off via `microvm_approval_suspend_enabled`; disabling new sleep preserves wake and cleanup. See the [supervisor runbook](../verification/645-p3-supervisor.md). + `TERMINATED` is the normal terminal signal and remains observable for at least 10 minutes. `ResourceNotFoundException` maps to completion only as a late fallback after the control-plane record is eventually reaped; polling does not wait for `NotFound`. **`/ping` health endpoint (AgentCore only).** The agent's FastAPI server responds to AgentCore's `/ping` calls while the coding task runs in a separate thread. AgentCore sees `HealthyBusy` and keeps the session alive. @@ -322,7 +324,7 @@ Long-running distributed systems fail. The orchestrator is designed so that ever | Hydration | Guardrail API unavailable | Fail the task (fail-closed: unscreened content never reaches agent) | | Session start | Selected compute service throttled | Exponential backoff. Fail after retries exhausted. | | Session start | Session crashes immediately | AgentCore: heartbeat never set, detected after 360s grace window. ECS: `DescribeTasks` reports failure. Lambda MicroVMs: `GetMicrovm` reports terminal state or the heartbeat never appears. | -| Running | Agent crashes mid-task | AgentCore: heartbeat goes stale. ECS: `DescribeTasks` reports stopped task. Lambda MicroVMs: `GetMicrovm` detects VM death and heartbeat staleness detects loss of the in-guest writer. Finalization inspects GitHub for partial work. | +| Running | Agent crashes mid-task | AgentCore: heartbeat goes stale. ECS: `DescribeTasks` reports stopped task. Lambda MicroVMs: `GetMicrovm` detects VM death and heartbeat staleness detects loss of the in-guest writer. Finalization preserves committed task results and records a specific failure for an active lost session. | | Running | Agent hits turn or budget limit | Session ends normally. Finalize based on what was produced. | | Running | Idle for 15 min | AgentCore kills session. Task transitions to `TIMED_OUT`. | | Finalization | GitHub API down | Retry 3x. If still failing, mark `FAILED` with infrastructure reason. | @@ -338,7 +340,7 @@ Long-running distributed systems fail. The orchestrator is designed so that ever ## Concurrency and scaling -Each task runs in an isolated compute session. The orchestrator reserves capacity per user before starting compute; AWS separately enforces backend quotas. Approval waits keep their reservation, including when a future P3 implementation suspends the MicroVM. +Each task runs in an isolated compute session. The orchestrator reserves capacity per user before starting compute; AWS separately enforces backend quotas. Approval waits keep their reservation, including when P3 suspends the MicroVM. This bounds unfinished sessions and their eventual resume demand; AWS memory-quota use while suspended remains unverified. ### Capacity limits @@ -433,7 +435,7 @@ Three DynamoDB tables back the orchestrator: one for task state, one for the aud | `branch_name` | String | `bgagent/{task_id}/{slug}` for new tasks; PR's `head_ref` for PR tasks | | `session_id` | String? | Backend session identifier (AgentCore session ID, ECS task ARN, or MicroVM ID) | | `compute_type` | String? | Selected backend: `agentcore`, `ecs`, or `lambda-microvm` | -| `compute_metadata` | Map? | Backend lifecycle handle; Lambda MicroVMs persist `microvmId` and `endpoint` | +| `compute_metadata` | Map? | Backend lifecycle handle; Lambda MicroVMs persist `microvmId`, `endpoint`, actual image identity and verified lifecycle protocol when available | | `concurrency_slot` | Map? | Internal reservation `{state, acquired_at, released_at?}`; excluded from public task responses | | `execution_id` | String? | Durable execution ID | | `pr_url` | String? | PR URL (set during finalization) | diff --git a/docs/src/content/docs/architecture/Cedar-hitl-gates.md b/docs/src/content/docs/architecture/Cedar-hitl-gates.md index e47f209cf..1539a5a2f 100644 --- a/docs/src/content/docs/architecture/Cedar-hitl-gates.md +++ b/docs/src/content/docs/architecture/Cedar-hitl-gates.md @@ -125,7 +125,7 @@ Settled during the 2026-04-23 design discussion and extended after the 2026-04-2 | 4 | **Scope allowlist: in-process, seeded from persisted `initial_approvals`** | Runtime escalation lives in the `PolicyEngine` instance. Submit-time `--pre-approve` flags persist on TaskTable and seed the allowlist at container startup. Lost on restart (rare; reconciler fails stranded tasks). | | 5 | **CLI UX: standalone `bgagent approve/deny` + `--pre-approve ` + `bgagent policies list` + `bgagent pending`** | No inline interactive prompt in the streaming CLI for v1. Discovery + listing commands solve the request_id/rule_id copy problem. | | 6 | **Timeouts: per-task default + per-rule Cedar annotation override, min wins, bounded floor + ceiling, fail-closed** | Per-task default: **300s** (5 min), overridable via `--approval-timeout` on submit and bounded by `[30, min(3600, maxLifetime - 300)]`. Floor: 30s (engine-enforced on both task default and rule annotations). Ceiling: `min(1h, maxLifetime_remaining - cleanup_margin)` — sized so the TTL on the approval row always covers the decision window. On timeout → deny (never auto-approve). See §14.8 for the off-hours trade-off this posture deliberately accepts. | -| 7 | **Concurrency slots: AWAITING_APPROVAL holds the slot** | Matches PAUSED semantics. Container is alive, consuming memory. | +| 7 | **Concurrency slots: AWAITING_APPROVAL holds the slot** | Bounds unfinished sessions and their eventual resume demand, including a suspended MicroVM. This is an ABCA admission policy; suspended AWS memory-quota consumption remains unverified. | | 8 | **Hard-deny is absolute** | No `--pre-approve` scope, and no blueprint `disable:` directive, can bypass it. CreateTaskFn validates and rejects `rule:`; blueprint loader rejects `disable:` entries that name built-in hard-deny rules. | | 9 | **Submit-time scope cap: 20 entries, ≤128 chars each** | Keeps audit trail legible, bounds allowlist check cost, limits abuse-vector damage. | | 10 | **Cedar annotations (verified working)** | `@rule_id(...)`, `@tier(...)`, `@approval_timeout_s(...)`, `@severity(...)`, `@category(...)`. Recoverable via `cedarpy.policies_to_json_str()` → JSON. Multi-match merging: min timeout wins (clamped by floor), max severity wins. | @@ -1311,7 +1311,7 @@ stateDiagram-v2 **AWAITING_APPROVAL holds the user's concurrency slot.** -Rationale: the Docker container is alive. Memory allocated. The AgentCore microVM pool is committed. Releasing the slot while the resource is still held lies to accounting and opens a resource-exhaustion vector. +Rationale: the task still owns an unfinished compute session and may resume work. Retaining its ABCA reservation prevents an unbounded collection of parked tasks from bypassing admission control. The rule also applies to P3 Lambda MicroVM suspension; suspended AWS memory-quota consumption remains unverified and is not the basis for claiming quota usage. Resume and terminal cleanup use the existing task-owned reservation protocol. Concrete behavior: diff --git a/docs/src/content/docs/architecture/Compute.md b/docs/src/content/docs/architecture/Compute.md index 0e2580e43..f95a0d394 100644 --- a/docs/src/content/docs/architecture/Compute.md +++ b/docs/src/content/docs/architecture/Compute.md @@ -87,9 +87,9 @@ For managed images, run `cdk/scripts/package-microvm-artifact.sh --stack-name **Operators must re-bootstrap for this.** The statement ships in bootstrap policy bundle **1.6.0**; a CDKToolkit stack bootstrapped at 1.5.0 or earlier will fail the CDK-managed MicroVM image deploy with a caller-side `iam:PassRole` AccessDenied on the build role. Check `CDKToolkit`'s `BootstrapPolicyVersion` output, and re-run `mise //cdk:bootstrap` (with `ComputeTypes` including `lambda-microvm`) if it is behind. +P3 additionally requires **bundle 1.8.0** for `MicrovmSuspendConfiguration`. +This statement lets CloudFormation manage and tag the live suspension setting +at `//microvm-approval-suspend-enabled`. The +coordinator gets only `GetParameter` on its exact parameter. Existing durable +executions retain their Lambda version and reread this setting before new +suspension, so disable can reach executions already running. + ```json { "Statement": [ @@ -885,6 +892,19 @@ The second statement, `MicrovmPassRoles`, is the one exception to the rule that "arn:aws:iam::*:role/backgroundagent-dev-LambdaMicrovmComputeConnector*" ], "Sid": "MicrovmPassRoles" + }, + { + "Action": [ + "ssm:GetParameters", + "ssm:PutParameter", + "ssm:DeleteParameter", + "ssm:AddTagsToResource", + "ssm:RemoveTagsFromResource", + "ssm:ListTagsForResource" + ], + "Effect": "Allow", + "Resource": "arn:aws:ssm:*:*:parameter/backgroundagent-*/microvm-approval-suspend-enabled", + "Sid": "MicrovmSuspendConfiguration" } ], "Version": "2012-10-17" diff --git a/docs/src/content/docs/architecture/Orchestrator.md b/docs/src/content/docs/architecture/Orchestrator.md index 8f0f612cf..4a72a1022 100644 --- a/docs/src/content/docs/architecture/Orchestrator.md +++ b/docs/src/content/docs/architecture/Orchestrator.md @@ -100,7 +100,7 @@ stateDiagram-v2 AWAITING_APPROVAL --> RUNNING : Approved or denied (resume) AWAITING_APPROVAL --> CANCELLED : User cancels mid-approval - AWAITING_APPROVAL --> FAILED : Stranded-approval reconciler + AWAITING_APPROVAL --> FAILED : Infrastructure loss or stranded wait FINALIZING --> COMPLETED : PR or commits found FINALIZING --> FAILED : No useful work @@ -125,11 +125,11 @@ stateDiagram-v2 | `HYDRATING` | `FAILED` | Hydration error | GitHub API failure, guardrail blocks content, Bedrock unavailable | | `RUNNING` | `AWAITING_APPROVAL` | Cedar soft-deny gate fires | Tool call triggers a soft-deny policy rule during execution | | `RUNNING` | `FINALIZING` | Session ends | Response received or session terminated | -| `RUNNING` | `TIMED_OUT` | Max duration exceeded | AgentCore and Lambda MicroVMs have an 8h substrate cap; the orchestrator's own safety-net poll window is `MAX_POLL_ATTEMPTS` (1020) × 30s ≈ 8.5h, after which a still-`RUNNING` task is driven to `TIMED_OUT` | +| `RUNNING` | `TIMED_OUT` | Max duration exceeded | AgentCore and Lambda MicroVMs have an 8h substrate cap; MicroVM supervision retains the original service deadline across replay; other backends retain the 1,020-attempt safety window (about 8.5h at 30s) | | `RUNNING` | `FAILED` | Session crash | Heartbeat or substrate liveness lost (see Liveness monitoring) | | `AWAITING_APPROVAL` | `RUNNING` | Approved or denied | Human decision received; agent resumes | | `AWAITING_APPROVAL` | `CANCELLED` | User cancels | Explicit cancel while awaiting approval | -| `AWAITING_APPROVAL` | `FAILED` | Stranded reconciler | Approval request orphaned (agent died mid-wait) | +| `AWAITING_APPROVAL` | `FAILED` | Infrastructure failure or stranded wait | Lost compute, exhausted supervisor recovery/window, or an orphaned approval; the approval decision is not rewritten | | `FINALIZING` | `COMPLETED` | Success inferred | PR exists or commits on branch | | `FINALIZING` | `FAILED` | Failure inferred | No commits, no PR, or agent reported error | @@ -156,7 +156,7 @@ Multiple timeout mechanisms work together to prevent runaway tasks. Substrate ti | Type | Default | Effect | |---|---|---| -| Max session duration | 8 hours | AgentCore caps a session at 8h; Lambda MicroVMs use `maximumDurationInSeconds: 28,800`, including suspended time. The orchestrator's safety-net poll loop runs up to `MAX_POLL_ATTEMPTS` (1020) × 30s ≈ 8.5h; a task still `RUNNING` when that window is exhausted is driven to `TIMED_OUT`. | +| Max session duration | 8 hours | AgentCore caps a session at 8h; Lambda MicroVMs use `maximumDurationInSeconds: 28,800`, including suspended time. MicroVM uses the saved absolute service deadline, so fast transition polling cannot shorten the session. Other backends retain the 1,020-attempt safety window. An exhausted approval wait uses FAILED, its allowed infrastructure-failure transition. | | Idle timeout | Backend-specific | AgentCore has an idle timeout. Lambda MicroVMs omit `idlePolicy` because inbound-traffic idleness would suspend an outbound-only agent while it is working. See Liveness monitoring. | | Max turns | 100 (range 1-500) | Agent stops after N model invocations. Configurable per task or per repo. | | Max cost budget | $0.01-$100 | Agent stops when budget is reached. Per-task or per-repo via Blueprint. | @@ -227,7 +227,7 @@ The orchestrator polls for completion using `waitForCondition` from the Durable | ECS | `DescribeTasks`, including container exit status and exit code | | Lambda MicroVMs | `GetMicrovm` state plus agent heartbeat | -While waiting between polls, the durable orchestrator suspends without compute charges. If the session is terminated externally (crash, timeout, cancellation), the poll detects it and the orchestrator proceeds to finalization using GitHub-based result inference as fallback. +While waiting between polls, the durable orchestrator suspends without compute charges. If the session is terminated externally (crash, timeout, cancellation), the poll detects it and the orchestrator proceeds to finalization after a strongly consistent task read; it preserves an already committed terminal result. ### Step 6: Finalization @@ -296,10 +296,12 @@ When the session is unhealthy, the task transitions to `FAILED` with "Agent sess **Lambda MicroVM state polling.** Liveness is a dual signal. The strategy maps `GetMicrovm` mechanically: `PENDING`/`RUNNING` report `running`, `SUSPENDING`/`SUSPENDED` report `suspended`, and `TERMINATING`/`TERMINATED` report terminal completion. The orchestrator supplies the health interpretation: -- `suspended` is healthy only while the task is `AWAITING_APPROVAL`; in any other task state it emits an anomaly and keeps polling rather than failing recoverable work. -- A terminal substrate report paired with a non-terminal task is a failure, but the orchestrator first re-reads the task row to confirm the agent did not write a terminal result between the original read and VM termination. +- Intentional suspension requires the matching pending gate and saved suspend intent. Unexpected suspension emits one anomaly per episode and starts bounded wake recovery, preserving recoverable work. +- A terminal substrate report paired with a non-terminal task is a failure, but finalization first strongly re-reads the task row to confirm the agent did not write a terminal result between the original read and VM termination. - Substrate state detects a dead VM; heartbeat staleness detects loss of the heartbeat writer inside a VM that still reports `RUNNING`. The independent heartbeat thread can continue during a pipeline hang, so a fresh timestamp is not proof of progress. +The P3 supervisor saves intent before control calls and rechecks the gate before and after them. Its durable state retains an absolute service lifetime, consecutive failures, recovery start time and next delay. Three failed cycles or 120 seconds of unconfirmed wake cannot become an indefinite wait. AWS RUNNING does not end recovery while the guest remains stuck on a decided/expired approval; fresh guest liveness is required. API approve/deny commit first, then attempt a bounded wake without changing the decision response. Automatic suspension defaults off via `microvm_approval_suspend_enabled`; disabling new sleep preserves wake and cleanup. See the [supervisor runbook](/sample-autonomous-cloud-coding-agents/architecture/645-p3-supervisor). + `TERMINATED` is the normal terminal signal and remains observable for at least 10 minutes. `ResourceNotFoundException` maps to completion only as a late fallback after the control-plane record is eventually reaped; polling does not wait for `NotFound`. **`/ping` health endpoint (AgentCore only).** The agent's FastAPI server responds to AgentCore's `/ping` calls while the coding task runs in a separate thread. AgentCore sees `HealthyBusy` and keeps the session alive. @@ -326,7 +328,7 @@ Long-running distributed systems fail. The orchestrator is designed so that ever | Hydration | Guardrail API unavailable | Fail the task (fail-closed: unscreened content never reaches agent) | | Session start | Selected compute service throttled | Exponential backoff. Fail after retries exhausted. | | Session start | Session crashes immediately | AgentCore: heartbeat never set, detected after 360s grace window. ECS: `DescribeTasks` reports failure. Lambda MicroVMs: `GetMicrovm` reports terminal state or the heartbeat never appears. | -| Running | Agent crashes mid-task | AgentCore: heartbeat goes stale. ECS: `DescribeTasks` reports stopped task. Lambda MicroVMs: `GetMicrovm` detects VM death and heartbeat staleness detects loss of the in-guest writer. Finalization inspects GitHub for partial work. | +| Running | Agent crashes mid-task | AgentCore: heartbeat goes stale. ECS: `DescribeTasks` reports stopped task. Lambda MicroVMs: `GetMicrovm` detects VM death and heartbeat staleness detects loss of the in-guest writer. Finalization preserves committed task results and records a specific failure for an active lost session. | | Running | Agent hits turn or budget limit | Session ends normally. Finalize based on what was produced. | | Running | Idle for 15 min | AgentCore kills session. Task transitions to `TIMED_OUT`. | | Finalization | GitHub API down | Retry 3x. If still failing, mark `FAILED` with infrastructure reason. | @@ -342,7 +344,7 @@ Long-running distributed systems fail. The orchestrator is designed so that ever ## Concurrency and scaling -Each task runs in an isolated compute session. The orchestrator reserves capacity per user before starting compute; AWS separately enforces backend quotas. Approval waits keep their reservation, including when a future P3 implementation suspends the MicroVM. +Each task runs in an isolated compute session. The orchestrator reserves capacity per user before starting compute; AWS separately enforces backend quotas. Approval waits keep their reservation, including when P3 suspends the MicroVM. This bounds unfinished sessions and their eventual resume demand; AWS memory-quota use while suspended remains unverified. ### Capacity limits @@ -437,7 +439,7 @@ Three DynamoDB tables back the orchestrator: one for task state, one for the aud | `branch_name` | String | `bgagent/{task_id}/{slug}` for new tasks; PR's `head_ref` for PR tasks | | `session_id` | String? | Backend session identifier (AgentCore session ID, ECS task ARN, or MicroVM ID) | | `compute_type` | String? | Selected backend: `agentcore`, `ecs`, or `lambda-microvm` | -| `compute_metadata` | Map? | Backend lifecycle handle; Lambda MicroVMs persist `microvmId` and `endpoint` | +| `compute_metadata` | Map? | Backend lifecycle handle; Lambda MicroVMs persist `microvmId`, `endpoint`, actual image identity and verified lifecycle protocol when available | | `concurrency_slot` | Map? | Internal reservation `{state, acquired_at, released_at?}`; excluded from public task responses | | `execution_id` | String? | Durable execution ID | | `pr_url` | String? | PR URL (set during finalization) | diff --git a/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md b/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md index 4bf9fad74..f671983e7 100644 --- a/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md +++ b/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md @@ -4,7 +4,7 @@ title: Adr 021 lambda microvms compute backend # ADR-021: AWS Lambda MicroVMs as a third ComputeStrategy backend -> **Implementation status (2026-09-15):** P1 ([#689](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/689)) and P2 ([#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733)) are merged. The takeover branch now has a [clean deployment](/sample-autonomous-cloud-coding-agents/architecture/645-p2-clean-deployment-20260913) with bootstrap 1.7.0 and a [verified managed-image update to 2.0](/sample-autonomous-cloud-coding-agents/architecture/645-microvm-image-rebuild-20260914), plus [live coding, PR iteration and cancellation evidence](/sample-autonomous-cloud-coding-agents/architecture/645-p2-live-task-20260914), without manual IAM workarounds. Remaining P2 failure/recovery and permission/network checks prevent full acceptance. Local P3 foundations include suspend/resume commands, explicit VM observations, guarded persistent intent, a policy helper, approval deadlines that survive a frozen clock, the [guest pause controller](/sample-autonomous-cloud-coding-agents/architecture/645-p3-guest-barrier), and [scoped Claude credentials with retained-client refresh](/sample-autonomous-cloud-coding-agents/architecture/645-p3-credentials). The [production HTTP hooks](/sample-autonomous-cloud-coding-agents/architecture/645-p3-lifecycle-hooks) now connect the guest barrier to atomic checkpoints and credential/gate reconciliation locally. [Per-worker image capability](/sample-autonomous-cloud-coding-agents/architecture/645-p3-image-capability) now declares the six hooks and retains/verifies the actual launched image version locally. Supervisor/approval-handler integration remains unfinished; automatic suspension is not enabled. This ADR defines P1–P3, not a P4. See the [current review](/sample-autonomous-cloud-coding-agents/architecture/645-p3-readiness-review) and [P3 implementation plan](/sample-autonomous-cloud-coding-agents/architecture/645-p3-implementation-plan). +> **Implementation status (2026-09-15):** P1 ([#689](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/689)) and P2 ([#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733)) are merged. The takeover branch now has a [clean deployment](/sample-autonomous-cloud-coding-agents/architecture/645-p2-clean-deployment-20260913) with bootstrap 1.7.0 and a [verified managed-image update to 2.0](/sample-autonomous-cloud-coding-agents/architecture/645-microvm-image-rebuild-20260914), plus [live coding, PR iteration and cancellation evidence](/sample-autonomous-cloud-coding-agents/architecture/645-p2-live-task-20260914), without manual IAM workarounds. Remaining P2 failure/recovery and permission/network checks prevent full acceptance. Local P3 foundations include suspend/resume commands, explicit VM observations, guarded persistent intent, a policy helper, approval deadlines that survive a frozen clock, the [guest pause controller](/sample-autonomous-cloud-coding-agents/architecture/645-p3-guest-barrier), and [scoped Claude credentials with retained-client refresh](/sample-autonomous-cloud-coding-agents/architecture/645-p3-credentials). The [production HTTP hooks](/sample-autonomous-cloud-coding-agents/architecture/645-p3-lifecycle-hooks) now connect the guest barrier to atomic checkpoints and credential/gate reconciliation locally. [Per-worker image capability](/sample-autonomous-cloud-coding-agents/architecture/645-p3-image-capability) now declares the six hooks and retains/verifies the actual launched image version locally. [Supervisor/approval-handler integration](/sample-autonomous-cloud-coding-agents/architecture/645-p3-supervisor) now adds durable recovery, bounded post-commit wake and scoped IAM locally. Live acceptance remains open; automatic suspension defaults off. This ADR defines P1–P3, not a P4. See the [current review](/sample-autonomous-cloud-coding-agents/architecture/645-p3-readiness-review) and [P3 implementation plan](/sample-autonomous-cloud-coding-agents/architecture/645-p3-implementation-plan). **Status:** proposed **Date:** 2026-07-29 @@ -64,19 +64,19 @@ Adopt **AWS Lambda MicroVMs as a third, opt-in `ComputeStrategy` backend** named ### 1. Strategy shape: extend the interface with mandatory suspend/resume -`ComputeType` widens to `'agentcore' | 'ecs' | 'lambda-microvm'` (mirrored in `cli/src/types.ts` and the CLI's inline unions). `SessionHandle` gains a `{ strategyType: 'lambda-microvm', microvmId, endpoint }` variant — `microvmId` because every lifecycle API (`suspend-microvm`, `resume-microvm`, `terminate-microvm`, `get-microvm`, and `create-microvm-auth-token` — the latter not called in P1–P3, see sub-decision 3) takes only the MicroVM identifier, and `endpoint` because it is per-session (minted by `RunMicrovm`) and required for any orchestrator→agent HTTP interaction. Note the naming seam: the handle field is `microvmId` (matching `RunMicrovmResponse.microvmId`), while the request key on every lifecycle command is `microvmIdentifier` — the strategy is the only place that translates between the two. **P3 refinement (2026-09-15):** the handle also retains the actual `imageArn` and `imageVersion` returned by Run, plus `lifecycleProtocol` after exact-version verification. The requested image remains deployment configuration, but the version that launched an existing worker is per-session evidence. Save the known handle before optional discovery; check that exact version with `GetMicrovmImageVersion`, then conditionally persist support in the receipt and `compute_metadata`. Current deployment settings cannot substitute for missing legacy evidence. +`ComputeType` widens to `'agentcore' | 'ecs' | 'lambda-microvm'` (mirrored in `cli/src/types.ts` and the CLI's inline unions). `SessionHandle` gains a `{ strategyType: 'lambda-microvm', microvmId, endpoint }` variant — `microvmId` because every lifecycle API (`suspend-microvm`, `resume-microvm`, `terminate-microvm`, `get-microvm`, and `create-microvm-auth-token` — the latter not called in P1–P3, see sub-decision 3) takes only the MicroVM identifier, and `endpoint` because it is per-session (minted by `RunMicrovm`) and required for any orchestrator→agent HTTP interaction. Note the naming seam: the handle field is `microvmId` (matching `RunMicrovmResponse.microvmId`), while the request key on every lifecycle command is `microvmIdentifier` — the strategy and inline approval-wake helper translate between the two at their control calls. **P3 refinement (2026-09-15):** the handle also retains the actual `imageArn` and `imageVersion` returned by Run, plus `lifecycleProtocol` after exact-version verification. The requested image remains deployment configuration, but the version that launched an existing worker is per-session evidence. Save the known handle before optional discovery; check that exact version with `GetMicrovmImageVersion`, then conditionally persist support in the receipt and `compute_metadata`. Current deployment settings cannot substitute for missing legacy evidence. **A second, sharper naming seam: `imageIdentifier` must be an ARN.** The name suggests a bare image name is acceptable — `create-microvm-image --name` takes one, and this ADR originally assumed `run-microvm --image-identifier` would too. It does not: a bare name is rejected with `ValidationException: Malformed ARN - doesn't start with 'arn:'`, and so is `list-microvm-image-builds --image-identifier ` (`Invalid ARN format`). The construct therefore resolves an operator-supplied name to its exact `arn:${Partition}:lambda:${Region}:${Account}:microvm-image:${Name}` ARN **once** — the same value it scopes the lifecycle IAM grant to — and injects THAT as `MICROVM_IMAGE_IDENTIFIER`. One derivation, two consumers, so a request field and an IAM resource can never disagree. The strategy validates the invariant and fails fast with the remedy, because the service's own error names neither the env var nor the fix. -The `ComputeStrategy` interface now has **mandatory** `suspendSession(handle)` / `resumeSession(handle)` methods returning `SessionLifecycleResult`, defined as `{ supported: false } | { supported: true }`. This local P3 foundation changes all three strategies together: AgentCore and ECS explicitly return false without calling AWS; MicroVM submits a command with a 10-second request bound. A future backend must implement both methods. The planned supervisor policy will use this explicit capability result. +The `ComputeStrategy` interface now has **mandatory** `suspendSession(handle)` / `resumeSession(handle)` methods returning `SessionLifecycleResult`, defined as `{ supported: false } | { supported: true }`. This local P3 foundation changes all three strategies together: AgentCore and ECS explicitly return false without calling AWS; MicroVM submits a command with a 10-second request bound. A future backend must implement both methods. The durable supervisor uses this explicit result without treating acknowledgment as final state. -`supported: true` means AWS acknowledged the command, not that the target state was reached. The SDK's suspend/resume response bodies are empty. All operational failures, including conflicts and not-found responses, propagate through sanitized errors; no generic conflict is normalized into success without observed-state evidence. A timed-out command may still have committed and requires reconciliation. The installed SDK has no `RESUMING` state. `SessionStatus.microvmState` now supplies explicit observations because the coarse `running` result also includes PENDING/unknown. The local policy/store uses retained generations and sticky wake intent to handle a delayed suspend after approval; the supervisor still needs to connect that policy, bound recovery and confirm wake state. See `docs/verification/645-lifecycle-intent.md`. +`supported: true` means AWS acknowledged the command, not that the target state was reached. The SDK's suspend/resume response bodies are empty. All operational failures, including conflicts and not-found responses, propagate through sanitized errors; no generic conflict is normalized into success without observed-state evidence. A timed-out command may still have committed and requires reconciliation. The installed SDK has no `RESUMING` state. `SessionStatus.microvmState` now supplies explicit observations because the coarse `running` result also includes PENDING/unknown. The local policy/store uses retained generations and sticky wake intent to handle a delayed suspend after approval; the durable supervisor now connects that policy, retains fixed recovery/session deadlines and waits for the required compute and guest observations. See `docs/verification/645-lifecycle-intent.md`. **Poll semantics — the strategy reports, the orchestrator interprets.** `pollSession(handle)` receives only the session handle and cannot see task state, so the health rules must live where the DynamoDB status lives. `SessionStatus` gains a `'suspended'` variant; the strategy maps `GetMicrovm` state mechanically and the **orchestrator** cross-references against the task row — the same division of labor `finalPollState` already uses for ECS (substrate stopped + non-terminal DynamoDB status → failed) and `pollTaskStatus` uses for agentcore heartbeats: substrate `suspended` + task `AWAITING_APPROVAL` is healthy (orchestrator-intended suspend); `suspended` with any other task status is an anomaly to surface, not fail-fast; substrate terminal + non-terminal task status → classify failed. **Liveness on this backend combines substrate state and the agent heartbeat.** `GetMicrovm` reports whether the VM is running or terminal. The independent `_heartbeat_worker` in `agent/src/server.py` refreshes the task timestamp every 45 seconds while the task is `RUNNING`; a stale timestamp detects loss of that writer, including a process crash or inability to write DynamoDB. It does **not** prove pipeline progress: a hung coding thread can coexist with a healthy heartbeat thread. A failed `/run` hook can trigger service teardown; after an accepted hook, ABCA still needs explicit termination and the maximum-duration backstop. A general progress watchdog remains separate work. -AgentCore and MicroVM tasks start the same periodic heartbeat worker, so they use the same 120-second grace and 240-second stale thresholds. ECS starts the pipeline directly, bypassing `server.py`; its pipeline writes a timestamp once but does not start this worker. Enabling the same stale check for ECS would therefore fail healthy tasks after roughly four minutes. The check applies only to `RUNNING`, not `AWAITING_APPROVAL`. The transition back to `RUNNING` refreshes `agent_heartbeat_at` in the same conditional transaction that clears the matching approval request. A poll immediately after a long gate therefore sees a fresh heartbeat before the next worker tick. P3 still needs the separate suspend/resume lifecycle and recovery policy described below. +AgentCore and MicroVM tasks start the same periodic heartbeat worker, so they use the same 120-second grace and 240-second stale thresholds. ECS starts the pipeline directly, bypassing `server.py`; its pipeline writes a timestamp once but does not start this worker. Enabling the same stale check for ECS would therefore fail healthy tasks after roughly four minutes. The check applies only to `RUNNING`, not `AWAITING_APPROVAL`. The transition back to `RUNNING` refreshes `agent_heartbeat_at` in the same conditional transaction that clears the matching approval request. A poll immediately after a long gate therefore sees a fresh heartbeat before the next worker tick. The local P3 supervisor additionally bounds recovery through suspension and decision consumption; live verification remains required. The service's `MicrovmState` enum has **six** members, not three, so the mapping is stated exhaustively (one line of rationale each, mirrored in the strategy's doc comment): @@ -126,10 +126,10 @@ Normative requirements (EARS, per [ADR-020](/sample-autonomous-cloud-coding-agen The headline economic win is suspend during **HITL approval waits** (Cedar approval gates, [CEDAR_HITL_GATES.md](/sample-autonomous-cloud-coding-agents/architecture/cedar-hitl-gates)): while a task waits on a human decision, the MicroVM is suspended (compute charges stop; memory/disk state — cloned repo, warm build caches — is preserved) and resumed when the decision lands. Under Cedar decision #6 the approval window is bounded (default 300 s, ceiling 1 h, timeout → deny), so the saving per gate is bounded at ~1 h of compute — real at 16 vCPU, and it makes any future extension of gate ceilings (the off-hours posture §14.8 deliberately defers) cheap on this backend. -The handshake must respect the existing approval mechanics: the agent **discovers decisions itself** by polling DynamoDB (`_poll_for_decision`, monotonic timeout), the approve/deny Lambda writes only the decision rows, and `AWAITING_APPROVAL` holds the concurrency slot (Cedar decision #7). Nothing "delivers" an approval to the agent, and suspension freezes the agent's monotonic clock — so the design is: +The handshake must respect the existing approval mechanics: the agent **discovers decisions itself** by polling DynamoDB (`_poll_for_decision`, monotonic timeout), the approve/deny Lambda commits the decision transaction before writing separate coordinator wake intent, and `AWAITING_APPROVAL` holds the concurrency slot (Cedar decision #7). Nothing "delivers" an approval to the agent, and suspension freezes the agent's monotonic clock — so the design is: - **Suspend — orchestrator-owned.** The orchestrator's durable poll observes `AWAITING_APPROVAL` on a `lambda-microvm` task and calls `suspendSession` after a grace period, and only when the gate's remaining window exceeds grace + resume overhead (suspending a 30 s gate is pure loss). Suspend is a policy decision on a poll observation, not a user action. -- **Resume — inline in the approve/deny Lambdas, orchestrator poll as backstop.** After the transactional decision write commits, `ApproveTaskFn`/`DenyTaskFn` load the MicroVM handle from the task row's `compute_metadata` (persisted at session start — the same field `cancel-task.ts` reads ECS handles from) via a post-commit strongly-consistent `GetItem`, then call `resumeSession` **best-effort**: on failure they log a warning and write a resume-orphan task event; the decision response never fails on a compute error (the decision row is already durable). The orchestrator poll reconciles: current approval row APPROVED/DENIED + MicroVM still `SUSPENDED` → retry resume; a PENDING row alone must not trigger wake-up. +- **Resume — inline in the approve/deny Lambdas, orchestrator poll as backstop.** After the transactional decision write commits, `ApproveTaskFn`/`DenyTaskFn` load the MicroVM handle from the task row's `compute_metadata` (persisted at session start — the same field `cancel-task.ts` reads ECS handles from) via a post-commit strongly-consistent `GetItem`, then save wake intent, observe the VM and request `ResumeMicrovm` **best-effort** only when SUSPENDED: on failure they log a warning and write a resume-orphan task event; the decision response never fails on a compute error (the decision row is already durable). The orchestrator poll reconciles: current approval row APPROVED/DENIED + MicroVM still `SUSPENDED` → retry resume; a PENDING row alone must not trigger wake-up. *Why inline rather than poll-only — codebase precedent:* resume-on-approve is structurally identical to task cancellation — a user-initiated, latency-sensitive action whose purpose is an immediate compute-lifecycle side effect. `cancel-task.ts` already resolves this exact tension: the API-plane handler invokes ECS `StopTask` / AgentCore `StopRuntimeSession` inline, best-effort (a failed stop logs a warning and the state transition stands; a `task_cancel_compute_orphan` event is written when no stoppable compute handle exists, `reason: missing_runtime_handle`) — with the conditional IAM wired in `task-api.ts`. The resume path goes one step further than the precedent by also writing the orphan event on *failed* resume calls, because a failed resume strands a suspended VM awaiting a decision — a stronger liveness consequence than a failed stop of an already-cancelled task. The alternative (orchestrator-poll-only resume) preserves single-owner lifecycle purity but pays up to a full poll interval (~30 s) of latency on every approval, and the purity argument was already litigated and declined for cancel. `approve-task.ts` is deliberately minimal today (security-critical ownership comparison, Cedar finding #6); the resume call is therefore added *after* the transaction commits, cannot alter the decision outcome, and carries one conditional `lambda:ResumeMicrovm` grant — the same blast-radius trade the cancel handler accepted in review. - **Timeout under freeze — the agent re-bases on the wall clock it already owns.** The agent's monotonic gate timer freezes while suspended, so resuming near the deadline is not enough: the frozen timer would still hold its remaining budget and fire the deny minutes *after* the user-visible window — colliding with the approval row's TTL (`created_at + timeout_s + 120s`) and triggering the "row reaped → stranded" fallback on a healthy gate. Instead, the gate expires at **`min(monotonic budget, created_at + timeout_s)`**, evaluated on each poll iteration and on `/resume`. This is not a new principle: Cedar decision #6 is already "min wins" for timeouts, the wall-clock deadline is already durable in the approval row the agent itself writes (`created_at` is recorded by the agent; clock changes across suspend still need testing), and §13.12's late-approval race fix already establishes that the durable row is authoritative over the agent's local timer. Deny authority stays agent-side (the conditional `TIMED_OUT` write + ConsistentRead re-read race protection is untouched); the orchestrator's resume at `deadline − margin` is purely the wake-up mechanism, with no correctness role. @@ -140,7 +140,7 @@ The handshake must respect the existing approval mechanics: the agent **discover It does **not** relieve the orchestrator of anything, because the two cases are disjoint. The service reaps a hook *result* it did not like; it has no view of the guest once the hook returned 200. So a task that starts normally — the overwhelming majority — has no service-side reaper at all, and a VM whose pipeline finished, crashed after `/run`, or hung is reaped by nobody but `TerminateMicrovm`. A leaked handle therefore remains a cost incident that bills until the 8 h cap; only the "the guest rejected its own payload" corner now cleans itself up. - **Concurrency slot stays held** during suspend. Cedar decision #7's rationale ("container alive, consuming memory") weakens under suspend, and the harder replacement rationale — "AWS counts `SUSPENDED` MicroVMs toward the account memory quota, so releasing ABCA's slot would not free real capacity" — is **undischarged**: the suspended VM stayed in `list-microvms` at every checkpoint, but that only proves *listed*. `L-CD1C0CC4` (1024 GB, account-scoped) exposes no `UsageMetric`, `AWS/Usage` carries only `CallCount` per API, and no MicroVM memory metric exists in any namespace, so consumption is **not observable safely** — proving it would need a large concurrent fleet. The conclusion (hold the slot) stands as the conservative choice, not as a verified fact. Size the arithmetic against the 32 GiB **peak** rather than the 8 GiB baseline: a busy fleet scales up, so peak is what actually competes for the account quota. -In P3, the agent's `/suspend` hook must acknowledge durable progress before returning 200 within its hook budget; `/resume` must refresh credentials and reseed application PRNG state from fresh OS entropy. These hooks are not implemented yet. A non-cryptographic PRNG such as Python `random` remains unsuitable for secrets even after reseeding; security-sensitive randomness must use an OS-backed cryptographic source. +In P3, the agent's `/suspend` hook must acknowledge durable progress before returning 200 within its hook budget; `/resume` must refresh credentials and reseed application PRNG state from fresh OS entropy. These hooks are implemented locally with atomic checkpoints, retained credential refresh and original-gate reconciliation; matching deployment and live acceptance remain open. A non-cryptographic PRNG such as Python `random` remains unsuitable for secrets even after reseeding; security-sensitive randomness must use an OS-backed cryptographic source. **Amended (P2 review): Application PRNG reseeding is P3 scope, and the P2 exposure is negligible.** The original EARS requirement below implied a `/run`-time reseed had to land with P2; it did not, and shipping P2 with the requirement unmet-but-asserted was itself the defect. Measured exposure, which is what changed the scoping: @@ -152,8 +152,8 @@ So the reseed moves to P3 alongside `/suspend` + `/resume`, where a *resumed* VM Normative requirements (EARS): -- While a `lambda-microvm` task is in `AWAITING_APPROVAL` and the gate's remaining window exceeds the configured grace period plus resume overhead, the orchestrator shall call `suspendSession` after the grace period. -- When the approve or deny Lambda commits a decision for a `lambda-microvm` task, the Lambda shall load the MicroVM handle from `compute_metadata` and call `resumeSession` best-effort. +- While approval suspension is enabled for a verified compatible worker, a `lambda-microvm` task is in `AWAITING_APPROVAL`, and the gate's remaining window exceeds the configured grace period plus resume overhead, the orchestrator shall call `suspendSession` after the grace period. +- When the approve or deny Lambda commits a decision for a `lambda-microvm` task, the Lambda shall load the MicroVM handle from `compute_metadata`, persist wake intent for the current identity, and request `ResumeMicrovm` best-effort after observing SUSPENDED and rechecking the gate. - If the inline resume fails, then the Lambda shall record a resume-orphan task event and shall still return the decision outcome. - While the current approval row is APPROVED or DENIED and the MicroVM remains `SUSPENDED`, the orchestrator shall retry `resumeSession`. A PENDING row alone is not a wake-up condition. - While a `lambda-microvm` task waits on an approval gate, the agent shall evaluate gate expiry as the earlier of its monotonic budget and the row's wall-clock deadline (`created_at + timeout_s`), on each poll iteration and on `/resume`. @@ -177,8 +177,8 @@ The phasing is therefore: | `/ready` | **P1** (construct enables `hooks.microvmImageHooks.ready`) | **P1** | MANDATORY, not a quality nicety — see above. A 200 proves uvicorn is bound and `server` imported cleanly (pulling in `pipeline` → `runner` → the policy engine), so a missing policy file fails the BUILD instead of the first task. **Since P2-F5 it also WARMS the snapshot** — the hook's 200 is what the service waits for before capturing the snapshot, making this the only place a warm page can be created, and the 225 MiB `claude` binary was cold in it (see the P2-F5 correction below). A required warm-up failure answers 503, so a snapshot that cannot exec the agent's own CLI fails the image build instead of every task. Still makes ZERO AWS calls, logging included (a `--version` exec is neither an AWS call nor a network call). | | `/run` | **P1** (construct sets `hooks.microvmHooks.run`) | **P1** | The payload-delivery channel. Must be served in P1 because `/ready` forces hooks to exist at all, and a hook-less image cannot accept `runHookPayload`. Since P2 it is also the **platform-configuration** channel (see "Platform configuration delivery" below). | | `/validate` | **P2** (construct sets `hooks.microvmImageHooks.validate`) | **P2** | An **image** (build-time) hook, and a **shallow self-check only**: server alive, hook routes registered, interpreter + contract sanity. It runs under the BUILD role, which deliberately holds no Bedrock / Secrets Manager / DynamoDB grants, so it must make **zero AWS API calls** and must not touch credential resolution — the "deeper warm-up assertions (Bedrock reachability, Memory access, tool availability)" this ADR originally assigned here are **not implementable**: every one of them would `AccessDenied` and fail every image build. They belong to the first task's own error handling. 200 when the checks pass, 503 while still initialising. | -| `/terminate` | **P2** (construct sets `hooks.microvmHooks.terminate`) | **P2** | Best-effort final flush: a final structured log line, then 200 — always, inside the hook budget, even with nothing running. It must **not** write terminal task status (the orchestrator finalizes the task and *then* calls `TerminateMicrovm`, so a terminate hook that wrote a status would race that finalization and could clobber the real outcome) and must not join the pipeline thread. `ProgressWriter` writes synchronously but drops failed events, so this hook cannot guarantee that every progress write succeeded. P3 supplies a separate acknowledged checkpoint for suspension. "Always 200" also covers the BODY: the handler reads the raw request rather than a typed model, because a typed body is validated before the handler runs and would answer 422 to malformed JSON — a reported hook failure on a successful teardown. Safe to declare because `TerminateMicrovm` removes the VM with or without in-guest cooperation. **Correction (P2-F8):** the service sends `microvmId: ""` on this hook, unlike `/run` where it is populated, so an empty id is expected-normal and this hook cannot join the guest record to the control-plane one — `/run`'s accepted line carries that correlation instead. | -| `/suspend`, `/resume` | **P3**, implemented locally | **P3**, implemented locally | Managed images declare the served checkpoint/credential/gate hooks with the shared 30-second service timeout and protocol marker. The coordinator verifies the actual launched version before allowing new suspension. Supervisor integration and live acceptance remain open. | +| `/terminate` | **P2** (construct sets `hooks.microvmHooks.terminate`) | **P2** | Best-effort final flush: a final structured log line, then 200 — always, inside the hook budget, even with nothing running. It must **not** write terminal task status (the orchestrator normally finalizes before `TerminateMicrovm`, and also attempts cleanup if database finalization fails, so a terminate hook that wrote a status would race that finalization and could clobber the real outcome) and must not join the pipeline thread. `ProgressWriter` writes synchronously but drops failed events, so this hook cannot guarantee that every progress write succeeded. P3 supplies a separate acknowledged checkpoint for suspension. "Always 200" also covers the BODY: the handler reads the raw request rather than a typed model, because a typed body is validated before the handler runs and would answer 422 to malformed JSON — a reported hook failure on a successful teardown. Safe to declare because `TerminateMicrovm` removes the VM with or without in-guest cooperation. **Correction (P2-F8):** the service sends `microvmId: ""` on this hook, unlike `/run` where it is populated, so an empty id is expected-normal and this hook cannot join the guest record to the control-plane one — `/run`'s accepted line carries that correlation instead. | +| `/suspend`, `/resume` | **P3**, implemented locally | **P3**, implemented locally | Managed images declare the served checkpoint/credential/gate hooks with the shared 30-second service timeout and protocol marker. The coordinator verifies the actual launched version before allowing new suspension. Supervisor integration is implemented locally with a default-off rollout flag; live acceptance remains open. | Consequence to state plainly, replacing the original "a P1-built MicroVM image is not runnable end to end": **a P1 image is creatable, launchable and payload-deliverable, but carries no smoke-parity guarantee.** P1 delivers the strategy, the construct, the roles/buckets/connectors, the image resource, the packaging script, and the `/ready` + `/run` endpoints — so a `lambda-microvm` task can start a MicroVM and hand it a payload. What P1 has **not** established is anything P2 owns: AgentCore Memory grants and `MEMORY_ID` delivery, the agent's non-secret env parity inside the snapshot, egress specifics from a running MicroVM, and heartbeat/progress behaviour end to end. At the P1 handoff, no clone → change → PR run had happened. P2 initially demonstrated that path with an IAM workaround; the 2026-09-14 clean deployment and coding/iteration/cancellation runs later passed without manual IAM changes. The broader failure/recovery, effective IAM and networking matrix remains open. The construct and the packaging script surface that remaining scope at synth/run time (`abca:microvm-image-p1-smoke-unverified`) so an operator cannot mistake a launchable substrate for a verified one. @@ -406,7 +406,7 @@ Lambda MicroVMs launched in 5 regions (us-east-1/2, us-west-2, eu-west-1, ap-nor - Unit tests for the strategy: start/poll/stop mapping (including `SessionStatus` `'suspended'` reported mechanically, without task-state interpretation), v2 reference-size bounds (4,096 bytes), immutable signed-reference replay and scoped payload transport, the image-identifier-must-be-an-ARN guard, the explicit `NO_INGRESS` argument (including the blank-env-var fallback, which must never omit the field), error classification (`ServiceQuotaExceededException`, `ThrottlingException`, `ResourceNotFoundException`, regional-unavailability), the omit-`idlePolicy` invariant, and `maximumDurationInSeconds` fixed at 28 800. - Agent tests: `/ready` returns 200 after required local warm-up succeeds, without starting a task or calling AWS; `/run` accepts authenticated v2 references and rejects old unsigned envelopes, starts the pipeline asynchronously through the same mapper `/invocations` uses, returns before the pipeline finishes, and rejects every unusable envelope with a named code before spawning; `/suspend` and `/resume` are NOT served (`/validate` and `/terminate` joined the served set in P2). - Orchestrator tests: substrate-terminal + non-terminal task status → failed classification; `suspended` + non-`AWAITING_APPROVAL` status → anomaly event, no fail-fast; `compute_metadata` persisted with `microvmId`/`endpoint` after `startSession`. -- CDK assertions: MicroVM resources present only when `ComputeTypes` includes the backend; synth failure for unsupported regions (plus the context-flag escape hatch); memory size validated against the accepted list at synth; the connector operator role and its trust; two connectors with the build-only one carrying port 80 and the runtime one not; the `/ready` + `/run` hook declaration and the absence of the others; IAM actions scoped as specified (orchestrator lifecycle set; no `CreateMicrovmAuthToken` anywhere); payload-bucket grants (execution role read-only); backend cost-allocation tags; types-sync check covers the widened `ComputeType`. +- CDK assertions: MicroVM resources present only when `ComputeTypes` includes the backend; synth failure for unsupported regions (plus the context-flag escape hatch); memory size validated against the accepted list at synth; the connector operator role and its trust; two connectors with the build-only one carrying port 80 and the runtime one not; the current six-hook declaration, shared protocol marker and per-worker capability admission; IAM actions scoped as specified (orchestrator lifecycle set; no `CreateMicrovmAuthToken` anywhere); payload-bucket grants (execution role read-only); backend cost-allocation tags; types-sync check covers the widened `ComputeType`. - CLI tests: onboarding rejection with remedy when the availability probe fails; doctor check present when a blueprint selects the backend. - P1 verification items (external service facts) — **executed 2026-07-31, us-east-1**; see `docs/verification/645-p1-lambda-microvm-runbook.md` for the full evidence. Discharged: `runHookPayload` limit (**4 096**, not 16 KB), the accepted baseline memory sizes (`[512…8192]` MiB — note the developer guide, not the probe, is what establishes that this is a BASELINE with a 32 GiB peak), image-identifier ARN requirement, IAM action names and the observed image-ARN shape, region probe behaviour, manual suspend/resume without `idlePolicy`, terminate timing and the `TERMINATED`-persists-≥10-min finding, the default public `HTTP_INGRESS`, and the `/ready` requirement. **Not** discharged: account-quota treatment of `SUSPENDED` MicroVMs (not observable safely), suspended TTL beyond 1 h (truncated), the vertical-scaling behaviour itself (no workload here approached the baseline, so the 4× peak is documented rather than observed), and the `AWS::Lambda::MicrovmImage` CloudFormation value shapes (never exercised — the run used the out-of-band script path; **discharged, and REFUTED, by the P2 run — see P2-F2 in sub-decision 3**). Record the closed answers in COMPUTE.md. diff --git a/docs/verification/645-lifecycle-intent.md b/docs/verification/645-lifecycle-intent.md index e8ea15322..19d85652d 100644 --- a/docs/verification/645-lifecycle-intent.md +++ b/docs/verification/645-lifecycle-intent.md @@ -1,6 +1,6 @@ # MicroVM lifecycle intent: local implementation and verification -Status: implemented locally on 2026-09-13, **not connected to automatic suspension or deployed**. This is the next foundation after `f3e684d4`. The [P3 checklist](./645-p3-implementation-plan.md) tracks the remaining integration and live gates. +Status: foundation implemented locally on 2026-09-13; [production supervisor/API integration](./645-p3-supervisor.md) added on 2026-09-15. **Not deployed; new suspension defaults off.** This is the next foundation after `f3e684d4`. The [P3 checklist](./645-p3-implementation-plan.md) tracks the remaining integration and live gates. ## In plain language @@ -11,7 +11,7 @@ Each instruction has a unique revision stamp. A supervisor holding an older stam ## What the code does - `cdk/src/handlers/shared/microvm-lifecycle.ts` reads the task and its current approval consistently, validates identity, and saves intent with database conditions. -- `cdk/src/handlers/shared/microvm-lifecycle-policy.ts` chooses an action from observations. It makes no AWS calls or human decisions. Its returned action is a recommendation for the future supervisor integration. +- `cdk/src/handlers/shared/microvm-lifecycle-policy.ts` chooses an action from observations. It makes no AWS calls or human decisions. Its returned action is reconciled by the durable supervisor. - `SessionStatus.microvmState` carries explicit MicroVM state alongside the existing coarse status. Known SDK states are preserved; absent/future states become local `UNKNOWN`, and a not-found response becomes local `NOT_FOUND`. These last two are observations defined by this application, not service states. Existing status/reason behavior is preserved. The internal `microvm_lifecycle` task attribute contains: @@ -56,7 +56,7 @@ A caller supplies valid times, session expiry, its normal poll interval and an e A PENDING approval alone is not a wake condition. An intended suspended VM can wait while there is sufficient time. APPROVED, DENIED, TIMED_OUT, STRANDED, deadline proximity, missing/invalid data or unintended suspension require wake/recovery. While the service reports SUSPENDING, save desired resume but return `requestReady: false`; issue ResumeMicrovm only after observing SUSPENDED. A wake acknowledgement followed by a delayed old suspend remains repairable because wake intent is retained. -Terminal/cancelled/finalizing tasks cannot resume. Terminal VM observations go through existing task reconciliation. PENDING/UNKNOWN or a legacy coarse `running` result cannot authorize suspend or confirm wake. Persistent failures/unconfirmed states still need bounded escalation in the supervisor. +Terminal/cancelled/finalizing tasks cannot resume. Terminal VM observations go through existing task reconciliation. PENDING/UNKNOWN or a legacy coarse `running` result cannot authorize suspend or confirm wake. The supervisor now bounds persistent failures and unconfirmed recovery across serialized polls. ## Verification and deployment gates @@ -76,7 +76,7 @@ Before enabling automatic sleep: 1. Complete a clean P2 deployment/rerun, including the earlier bootstrap, metadata, capacity and managed-image gates. Follow the coordinated v2 drain/rollout procedure. 2. Implement guest lifecycle context, acknowledged progress durability, credential refresh preserving task identity, resume barriers and snapshot randomness handling. Reuse the original approval deadline. -3. Connect policy/store to durable supervisor polling, preserving counters, backoff, anomaly episodes and a bounded wake-recovery clock. Handle stale results by observing again. Never reset the recovery clock on repeated saves. +3. **Implemented locally:** policy/store are connected to durable supervisor polling, preserving counters, next-poll delay, anomaly episodes and a bounded wake-recovery clock. Handle stale results by observing again. Never reset the recovery clock on repeated saves. 4. Connect approve/deny after the decision commits, with bounded best-effort wake and repair diagnostics. Preserve current decision responses on wake failure. 5. Add and verify the coordinator's required task/approval transaction permissions and scoped MicroVM lifecycle grants. Grant no lifecycle action or intent-write permission to workers. Check the total handler time budget, not only each individual call. 6. Deploy compatible hooks/image and coordinator together with automatic suspension initially disabled. Validate actual transition/conflict/timeout behavior and then the full P3 acceptance matrix in an isolated development deployment. diff --git a/docs/verification/645-p3-image-capability.md b/docs/verification/645-p3-image-capability.md index f289a5c68..b2194e42e 100644 --- a/docs/verification/645-p3-image-capability.md +++ b/docs/verification/645-p3-image-capability.md @@ -84,7 +84,7 @@ focused reruns. There was no production-code failure in that run. ## Remaining completion gates -Connect durable supervisor recovery and approval-triggered wake, then deploy and +Durable supervisor recovery and approval-triggered wake are now [connected locally](./645-p3-supervisor.md). Next deploy and test the matching coordinator/image together. Verify effective image-read and lifecycle permissions, actual service hooks, expired credentials, approval and cancellation races, failures and cleanup in AWS. The diff --git a/docs/verification/645-p3-implementation-plan.md b/docs/verification/645-p3-implementation-plan.md index bb3093020..ce4ba65fb 100644 --- a/docs/verification/645-p3-implementation-plan.md +++ b/docs/verification/645-p3-implementation-plan.md @@ -1,6 +1,6 @@ # ADR-021 P3 implementation and completion plan -Prepared 2026-09-13 from `main` `5e10038c7e28179b302ac4de78b709795aeba3ce`. Read the [review](./645-p3-readiness-review.md) for evidence and the beginner introduction. This document proposes work; it does not mark P3 as implemented. +Prepared 2026-09-13 from `main` `5e10038c7e28179b302ac4de78b709795aeba3ce`. Read the [review](./645-p3-readiness-review.md) for evidence and the beginner introduction. This document tracks implementation and validation; local completion does not mark P3 live acceptance complete. ## Implementation progress @@ -15,14 +15,15 @@ budgets are 20/30 seconds. At this milestone, image capability and supervisor integration remained open. No deployment or automatic suspension was enabled in this milestone. -**Latest local P3 milestone (2026-09-15):** [per-worker image capability](./645-p3-image-capability.md) +**Image milestone (2026-09-15):** [per-worker image capability](./645-p3-image-capability.md) now declares all six hooks and the shared protocol marker. The coordinator saves the worker handle first, verifies the exact returned image ARN/version, then conditionally records support in both start receipt and compute metadata. Missing or unreadable support permits normal coding and disables new suspension. Database race tests reject changed identities and recover a committed capability after a -lost reply. Durable supervisor integration, approval-triggered wake and live -sleep/wake verification remain open; no deployment or automatic sleep was enabled. +lost reply. No deployment or automatic sleep was enabled in that milestone. + +**Supervisor milestone (2026-09-15):** [production supervision and approval wake](./645-p3-supervisor.md) now connect the policy/store to durable polling and post-commit approve/deny handlers. Recovery clocks and the original service lifetime survive replay; failed wake and cleanup remain visible. The rollout flag defaults off and uses a live Parameter Store switch for existing durable executions. Full repository validation passes (5,028 CDK, 1,951 Python and 928 CLI tests); all live P3 gates remain open. **Live infrastructure and image deployed (2026-09-14):** the [clean P2 deployment record](./645-p2-clean-deployment-20260913.md) tracks the new @@ -93,8 +94,10 @@ is superseded by these records. - [x] Add the guest pause controller, original-gate registration, parallel-tool tracking, progress acknowledgment tracking, heartbeat/read drain and generation-guarded wake completion; reseed the application PRNG at run and controller resume. - [x] Add production agent hooks with acknowledged checkpoints, retained ambient/tenant credential renewal, a sole scoped Claude provider and atomic task/gate reconciliation; verify duplicates, original deadlines, timeout and teardown behavior locally. - [x] Declare compatible image hooks using shared budgets and bind lifecycle capability to the actual image/version used by each worker; keep automatic sleep disabled until integration/live acceptance. -- [ ] Persist bounded poll/recovery counters and connect lifecycle policy to the supervisor. -- [ ] Connect supervisor and approval handlers, then verify the complete P3 sleep/wake lifecycle in AWS. +- [x] Persist bounded poll/recovery counters and connect lifecycle policy to the supervisor. +- [x] Connect post-commit approval wake, bounded diagnostics/cleanup, the default-off rollout flag and scoped IAM. +- [x] Complete full repository validation with the P3 supervisor and live switch. +- [ ] Deploy and verify the complete P3 sleep/wake lifecycle in AWS. First prerequisite batch completed locally on 2026-09-13: @@ -172,7 +175,7 @@ Second P3 foundation batch implemented locally (2026-09-13): - Added explicit `microvmState` observations without changing existing coarse status/reason semantics, plus a policy helper for grace, useful sleep, pre-deadline wake, image/enable guards, missing/unreadable data, cancellation and delayed suspend-after-wake races. - DynamoDB Local verifies actual transaction conditions, rollback, competing writers, changed identities, lost committed replies, cancellation/decision during recovery and a fresh module/client loading saved intent. See the [lifecycle runbook](./645-lifecycle-intent.md) for protocol and remaining integration/deployment gates. - CDK lint/compilation passed. The broad handler/session-role run passed **158 suites / 3,738 tests**, including **23 lifecycle** and **15 existing capacity** DynamoDB Local tests. Five relevant suites passed **218 overlapping tests** and exited normally. The broad run exited successfully after a delay (about 72 seconds total versus 17.5 seconds reported test execution), with no open-handle trace; its cause is not established. Documentation sync, the **77-page** build and link checks pass. No Python source changed. The temporary local database was removed. -- No production caller uses this policy/store yet. No IAM grants, image hooks or automatic suspension were enabled. Durable poll failure/recovery tracking, guest barriers, supervisor/decision-handler wiring and live AWS gates remain unfinished. +- At that foundation milestone, no production caller used this policy/store. No IAM grants, image hooks or automatic suspension were enabled. Durable poll failure/recovery tracking, guest barriers, supervisor/decision-handler wiring and live AWS gates remain unfinished. ## The result we want @@ -334,7 +337,7 @@ The writer inventory found no production callers of Python `write_submitted` or ## 3. Re-prove P2 on the final infrastructure -Use a supported Region and an isolated development repository/account deployment. Record the actual deployed bootstrap bundle (at least 1.7.0 for this source, including exact-self CloudFormation PassRole, or a newer required bundle). Compare effective policies as well as the displayed version. Update bootstrap deliberately when required; a command that skips an already bootstrapped stack is not evidence of refresh. +Use a supported Region and an isolated development repository/account deployment. Record the actual deployed bootstrap bundle (at least 1.8.0 for this source, including exact-self CloudFormation PassRole and the scoped live-suspension parameter permissions, or a newer required bundle). Compare effective policies as well as the displayed version. Update bootstrap deliberately when required; a command that skips an already bootstrapped stack is not evidence of refresh. The [2026-09-13–14 clean deployment](./645-p2-clean-deployment-20260913.md) completed the infrastructure, managed-image creation and build-hook checks in @@ -378,11 +381,11 @@ The installed SDK returns empty suspend/resume responses. Its observed states ar **Implemented locally:** `microvm-lifecycle.ts` stores a typed optional record on the existing task row, separate from `compute_metadata`. It includes format version, generation, VM/gate identity, desired action, original timestamp and deadline. Conditions guard owner/status/gate/handle/generation; suspend atomically checks the same PENDING approval row and unchanged deadline inputs. A wake cannot become a sleep for that gate, and retaining the record prevents an older absent-record snapshot from recreating a sleep intent. A new gate may establish a new generation. No credentials or bearer URLs are stored. -**Still required:** persist failure counters/backoff, anomaly episodes and bounded wake recovery in durable poll-loop state so Lambda replay does not reset them. Repeated intent saves already retain the original timestamp/generation. The record is internal and has no public task API field. Database success is not a lock over a later AWS command: reread before suspend and reconcile after every command outcome. +**Implemented locally:** the durable supervisor persists failure counters, anomaly episodes, next-poll delay and fixed recovery/session deadlines. Lambda replay does not reset them. Repeated intent saves already retain the original timestamp/generation. The record is internal and has no public task API field. Database success is not a lock over a later AWS command: reread before suspend and reconcile after every command outcome. -**Implemented as an unwired helper:** the policy combines task status, the **specific current approval row's status**, desired action, explicit VM state and current time. PENDING alone is not a reason to resume. All terminal approval states, deadline proximity, missing/unreadable data or unintended suspension can require wake. SUSPENDING records desired wake but returns `requestReady: false` until SUSPENDED is observed. +**Connected to the durable supervisor:** the policy combines task status, the **specific current approval row's status**, desired action, explicit VM state and current time. PENDING alone is not a reason to resume. All terminal approval states, deadline proximity, missing/unreadable data or unintended suspension can require wake. SUSPENDING records desired wake but returns `requestReady: false` until SUSPENDED is observed. -Initial local policy values: 30-second suspend grace, 60-second pre-deadline wake margin, 30-second minimum useful sleep and at most 5-second transition polling. Long intervals are clamped to the relevant grace/wake/session deadline. These are tunable choices requiring live measurement, not AWS facts. Three consecutive poll failures before escalation remains proposed supervisor work. The future integration must bound its whole operation, including the store's 5-second request/read-sequence budgets and any lost-reply recovery. +Initial local policy values: 30-second suspend grace, 60-second pre-deadline wake margin, 30-second minimum useful sleep and at most 5-second transition polling. Long intervals are clamped to the relevant grace/wake/session deadline. These are tunable choices requiring live measurement, not AWS facts. The supervisor now escalates after three consecutive failed cycles and bounds the entire cycle to 45 seconds, including the store's 5-second request/read-sequence budgets and lost-reply recovery. Wake/unknown recovery is bounded to 120 seconds and startup to 300 seconds; see the supervisor runbook for the complete budgets. ### State/action table @@ -437,27 +440,27 @@ The production HTTP resume callback now invokes this refresh before any AWS read ### Orchestrator -Implement the state/action table in a small testable policy/reconciliation helper called by the durable poll loop. Read current gate identity/status/deadline consistently. Record intent before requesting suspension; reread after uncertain outcomes and after suspend success to catch an approval that won concurrently. API acknowledgement is not final VM state: poll it. +**Implemented locally; live acceptance pending.** The production path implements the state/action table in a small testable policy/reconciliation helper called by the durable poll loop. Read current gate identity/status/deadline consistently. Record intent before requesting suspension; reread after uncertain outcomes and after suspend success to catch an approval that won concurrently. API acknowledgement is not final VM state: poll it. An approval can arrive before a pending suspend finishes. Even if an inline resume sees “already running,” the orchestrator must later notice that the machine became suspended and wake it. Do not clear durable wake intent merely because one API call appeared successful. When `ResumeMicrovm` is acknowledged but a subsequent observation does not yet confirm `RUNNING`, keep reconciling within a bounded recovery interval. Do not invent a `RESUMING` service state or treat an unknown/coarse state as confirmation. Defer the pre-suspend stale-heartbeat check only within that bound. When the agent restores task RUNNING, use the fresh timestamp from prerequisite 1C. Never exempt genuine crashed RUNNING tasks indefinitely. -Add consecutive MicroVM poll-error tracking. Reset on successful observations; classify permanent failures separately from transient ones. At the chosen threshold, perform a final consistent task read, record an explicit infrastructure failure and finalize/terminate through the existing single-owner path. Emit recovery/orphan diagnostics when termination itself fails; do not silently lose the handle. Keep suspend failures distinguishable from lost compute: failure to save money can leave a task safely awake, whereas failure to wake threatens correctness and needs bounded escalation. +Consecutive MicroVM poll-error tracking resets after a complete successful observation cycle; classify permanent failures separately from transient ones. At the chosen threshold, perform a final consistent task read, record an explicit infrastructure failure and finalize/terminate through the existing single-owner path. Emit recovery/orphan diagnostics when termination itself fails; do not silently lose the handle. Keep suspend failures distinguishable from lost compute: failure to save money can leave a task safely awake, whereas failure to wake threatens correctness and needs bounded escalation. ### Approve and deny handlers -After the existing authorization checks and decision transaction **commit**, use a shared helper to load `compute_metadata` with a strongly consistent task read. Validate compute type, complete handle and current task/gate identity. For MicroVM, request resume with a short bound. No HTTP call to the guest is necessary. +**Implemented locally; live acceptance pending.** After the existing authorization checks and decision transaction **commit**, use a shared helper to load `compute_metadata` with a strongly consistent task read. Validate compute type, complete handle and current task/gate identity. For MicroVM, request resume with a short bound. No HTTP call to the guest is necessary. Missing handle, read failure, wrong/terminal state or resume failure must produce a warning and a structured resume-orphan event (include task ID, gate ID, VM ID when known, stage, reason and safe AWS request ID). Audit-event failure is also best-effort. **None of these post-commit failures may turn a successful decision into a 500 or undo the transaction.** Preserve the current response/status and ownership/already-decided/wrong-gate protections. The poll loop is the repair path. The current API has no independent wall-clock expiry check: the agent owns TIMED_OUT, and the first committed decision wins. Strict API expiry would be a separate behavior change. ### IAM and deployment -Grant orchestrator SuspendMicrovm/ResumeMicrovm on the exact configured image ARN and required version suffix, alongside its existing lifecycle actions. Grant approve/deny ResumeMicrovm, and GetMicrovm only if the shared wake helper uses it, with the same image scope. Do not grant these actions to the agent execution role. Add no token-minting, broad role-passing or network ingress permission. +**Implemented in CDK; effective AWS permissions pending.** Grant orchestrator SuspendMicrovm/ResumeMicrovm on the exact configured image ARN and required version suffix, alongside its existing lifecycle actions. Grant approve/deny ResumeMicrovm, and GetMicrovm only if the shared wake helper uses it, with the same image scope. Do not grant these actions to the agent execution role. Add no token-minting, broad role-passing or network ingress permission. Verify the lifecycle store's DynamoDB permissions too: task GetItem/UpdateItem and approval GetItem/ConditionCheckItem for the supervisor's cross-table suspend transaction. Confirm environment wiring for both table names and test effective permissions. Worker writes to `microvm_lifecycle` must remain excluded. -Check `task-api.ts`'s lazy image-ARN wiring and no-image branch, bootstrap deployment-role coverage, tests/suppressions and CloudFormation resource counts. Image-hook changes and runtime hook serving must deploy together; automatic suspension remains off until the compatible image is ready. Provide an operational disable switch that stops **new suspends while still allowing resume, timeout handling and termination** for already-sleeping tasks. +Check `task-api.ts`'s lazy image-ARN wiring and no-image branch, bootstrap deployment-role coverage, tests/suppressions and CloudFormation resource counts. Image-hook changes and runtime hook serving must deploy together; automatic suspension remains off until the compatible image is ready. The `microvm_approval_suspend_enabled` context defaults false and sets both the static opt-in and a live SSM parameter. Durable executions pin their original environment, so rollback must verify the live parameter is false to stop new suspends in existing executions. Resume, timeout handling and termination remain available. See the [supervisor runbook](./645-p3-supervisor.md#deployment-configuration-and-permissions) for drift/rollback details. Deploy matching source/image with false before controlled opt-in. ## 7. Acceptance matrix and completion gates diff --git a/docs/verification/645-p3-readiness-review.md b/docs/verification/645-p3-readiness-review.md index 8dba3934b..5cf2b17f4 100644 --- a/docs/verification/645-p3-readiness-review.md +++ b/docs/verification/645-p3-readiness-review.md @@ -10,9 +10,9 @@ This review covers the existing MicroVM implementation, related P2 follow-ups, c Further local work adds durable MicroVM start receipts, stable request tokens and bounded handle recovery. AWS token-retention/conflict behavior and unknown-ID cleanup remain live verification gates. The shared finalizer's repeated-decrement risk was subsequently reproduced and fixed locally with task-owned reservation transactions, unified counter writers and revision-guarded reconciliation, including approval waits. DynamoDB Local exercises the real transaction conditions; deployed IAM, scan scale and upgrade/drain behavior still need AWS verification. -The first local P3 foundation adds the supervisor's pause/wake command methods and fixes approval timing. The timer now keeps both an elapsed-time stopwatch and the original clock deadline, using whichever runs out first. A sleeping VM therefore does not get a fresh approval window. Automatic sleeping, guest safety hooks and supervisor repair logic remain unfinished. The installed AWS SDK does not expose a `RESUMING` state; receiving a wake acknowledgement alone does not prove that the VM is awake. +The first local P3 foundation adds the supervisor's pause/wake command methods and fixes approval timing. The timer now keeps both an elapsed-time stopwatch and the original clock deadline, using whichever runs out first. A sleeping VM therefore does not get a fresh approval window. At that foundation milestone, guest hooks and supervisor repair remained unfinished; subsequent milestones below supersede those gaps. The installed AWS SDK does not expose a `RESUMING` state; receiving a wake acknowledgement alone does not prove that the VM is awake. -A second foundation adds saved sleep/wake instructions with revision stamps, so an old supervisor cannot overwrite a newer wake request. A policy helper now handles the observed state and current approval deadline. Real DynamoDB Local transaction tests cover competing writers and restart/readback cases; this is a local database test, not an AWS deployment. See the [lifecycle runbook](./645-lifecycle-intent.md). Guest hooks, durable supervisor integration and live verification remain required. +A second foundation adds saved sleep/wake instructions with revision stamps, so an old supervisor cannot overwrite a newer wake request. A policy helper now handles the observed state and current approval deadline. Real DynamoDB Local transaction tests cover competing writers and restart/readback cases; this is a local database test, not an AWS deployment. See the [lifecycle runbook](./645-lifecycle-intent.md). At that foundation milestone, guest hooks, supervisor integration and live verification remained open. **Further P3 work (2026-09-14):** the [guest pause controller](./645-p3-guest-barrier.md) now keeps coding behind a controlled door while approval work is paused. The @@ -22,7 +22,7 @@ tests with fake AWS responses prove renewal before the next request and safe failure without borrowing the parent's keys. The subsequent [HTTP hook milestone](./645-p3-lifecycle-hooks.md) connects pause to an atomic checkpoint and wake to credential renewal plus task/gate reconciliation. -The [image capability milestone](./645-p3-image-capability.md) (2026-09-15) now declares the six hooks and checks/persists support for the actual launched image version. Supervisor wiring and real AWS sleep/wake verification remain open. +The [image capability milestone](./645-p3-image-capability.md) (2026-09-15) now declares the six hooks and checks/persists support for the actual launched image version. The [supervisor milestone](./645-p3-supervisor.md) now adds durable recovery, post-commit approval wake, bounded cleanup and a default-off rollout flag locally. Real AWS sleep/wake acceptance remains open. The reservation review found a separate trust boundary: the old agent role could write/replace/delete its task row, including coordinator metadata. A subsequent local fix restricts main-task writes to reporting/approval attributes, removes replacement/deletion permissions and removes unused worker counter grants. It also removes unused Python submission/session-info helpers and corrects overstated tenant-isolation comments. [Metadata verification](./645-coordinator-metadata.md) records the tests and pending AWS gate. Status reports still come from the agent, and the compute role chooses session tags; this is not complete hostile-worker isolation. @@ -43,8 +43,8 @@ An **IAM role** is a permission badge. A **trust policy** says who may wear that | Phase | Purpose | Current state | |---|---|---| | P1 | Build the computer, start it, deliver a task, check it and stop it | Merged in [#689](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/689). Strategy, infrastructure, bootstrap permissions, packaging, types, `/ready` and `/run` exist. | -| P2 | Make a real coding task work with configuration, permissions, logs and progress | Merged in [#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733). `/validate`, `/terminate`, warm-up, runtime grants and heartbeat support exist. Two recorded tasks completed and opened PRs on August 7, using a manual IAM workaround. Permanent fixes still need a clean rerun. | -| P3 | Sleep during a human approval wait, wake correctly, and keep deadlines/credentials safe | Local foundations include intent/policy, guest pause control, scoped credential renewal, production HTTP checkpoint/wake hooks and per-worker image capability. Supervisor integration and live acceptance remain open. | +| P2 | Make a real coding task work with configuration, permissions, logs and progress | Merged in [#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733). `/validate`, `/terminate`, warm-up, runtime grants and heartbeat support exist. The September 14 clean rerun passed coding, iteration and cancellation without manual IAM workarounds, followed by image 2.0 payload/start checks. The broader deployed recovery, effective IAM and network matrix remains open. | +| P3 | Sleep during a human approval wait, wake correctly, and keep deadlines/credentials safe | Local foundations include intent/policy, guest pause control, scoped credential renewal, production HTTP checkpoint/wake hooks and per-worker image capability. Durable supervision and post-commit approval wake are now connected locally with bounded recovery and scoped IAM. Live acceptance remains open; automatic sleep defaults off. | | P4 | — | ADR-021 defines no P4. Verification runbooks have their own numbered phases; those are not extra ADR milestones. | The old unchecked checklist and the word “proposed” do not erase the merged work. Conversely, merged code is not proof that the final deployment path works unattended. diff --git a/docs/verification/645-p3-supervisor.md b/docs/verification/645-p3-supervisor.md new file mode 100644 index 000000000..40cc6250f --- /dev/null +++ b/docs/verification/645-p3-supervisor.md @@ -0,0 +1,191 @@ +# ADR-021 P3: supervisor and approval wake + +Status (2026-09-15): production integration and full repository validation pass +locally. This milestone has not been deployed. Automatic suspension defaults +off. The [P3 plan](./645-p3-implementation-plan.md) retains the live acceptance +gates and remaining P2 checks. + +## What this does + +The supervisor is the part of ABCA that watches the rented computer. It now +remembers both its instructions and its deadlines when the supervising Lambda +restarts. “AWS accepted my wake request” and “the agent continued its work” are +different observations. + +For a long approval wait, the supervisor checks the live suspension setting, +saves an instruction to sleep, rereads the setting, checks +the same approval again, requests suspension, and checks again afterward. The +guest's checkpoint hook must complete before AWS freezes it. New sleep requires +verified support from the worker's actual image version. + +An approval can arrive while suspension is still happening. Both the decision +API and supervisor save “wake” immediately. They request Resume only after AWS +reports SUSPENDED. That instruction remains even after an acknowledgment or a +RUNNING observation, so a delayed suspension cannot silently strand the worker. +When the task leaves its gate, the supervisor can re-scope wake to the new +no-gate identity before ending recovery on the following observation. + +AWS RUNNING alone cannot hide an agent stuck on an already decided or expired +gate. Recovery remains bounded until the agent moves forward. An intentional +early wake may remain AWAITING_APPROVAL while its original decision is pending. +An ordinary RUNNING worker still gets the shared heartbeat checks. + +## Bounds and state + +These are application choices requiring live timing verification, except the +configured eight-hour service maximum. + +| Bound | Value | +|---|---| +| Entire supervisor cycle | 45 seconds | +| Individual control request, including Get/Terminate | At most 10 seconds, shortened by caller deadline | +| Lifecycle database read sequence/write | At most 5 seconds, shortened by caller deadline | +| Live suspension setting read | At most 3 seconds, shortened by caller deadline; failure disables new sleep | +| Transition poll | At most 5 seconds | +| Suspend/wake/unknown recovery | 120 seconds; repeated polls do not reset it | +| Startup/HYDRATING recovery | 300 seconds | +| Consecutive failed cycles or Resume requests | 3; permanent control denials can fail earlier | +| Session deadline | Earlier saved deadline or AWS start time + service duration, capped at 28,800 seconds | +| API work after a committed decision | At most 8 seconds, retaining 1 second of remaining Lambda time for the response | +| Each API audit attempt | At most 2 seconds inside that shared budget | +| Final cleanup | At most 25 seconds, with at most two termination attempts | + +The durable poll state contains only JSON values: VM identity, original +observation/deadline, failure counters, recovery kind/start time, anomaly state +and next delay. Before verified service timing is available, a fixed deadline +from the first durable observation bounds supervision and new sleep is disabled. +Faster polls do not consume the older 1,020-attempt limit used by other backends. + +Control calls and database recovery reads share the caller's AbortSignal, a +cancellation notice. An expired operation cannot start another request with a +fresh independent budget. + +## Decisions, failures and cleanup + +Approve/deny keep the existing authorization and conditional transaction. +Optional wake runs only after commit. A failed read, Resume or audit cannot undo +the decision or turn the accepted response into HTTP 500. The guest still owns +approval expiry; this change adds no independent API expiry rule. + +Before a supervisor failure becomes a task outcome, finalization strongly reads +the task and preserves a committed completion/cancellation. Conditional failure +writes also check the original worker ID. A replacement worker is not followed. +Normal capacity release remains task-owned and happens once. + +Infrastructure exhaustion during AWAITING_APPROVAL/HYDRATING uses FAILED, since +those statuses do not allow TIMED_OUT. RUNNING/FINALIZING session expiry uses +TIMED_OUT. The approval row's decision is not rewritten. This also fixes the old +approval-wait finalizer's forbidden AWAITING_APPROVAL → TIMED_OUT transition. + +Termination is attempted even if database finalization fails. Its result +distinguishes requested, not-found and unconfirmed. ConflictException is +unconfirmed: a conflicting lifecycle operation does not prove termination. +An acknowledgment is not reported as observed teardown. + +Diagnostics: + +- `microvm_suspend_anomaly`: once per unexpected suspension episode; a new + episode can be reported after recovery. +- `microvm_supervisor_request_failed` and command-failure events: safe error + identifiers and the stage that failed. +- `microvm_resume_orphan`: an inline wake failed or became ineligible; includes + task/gate, VM when known, stage, reason and validated AWS request ID when + available. A recoverable race is not proof of a permanent orphan. +- `microvm_cleanup_unconfirmed`: bounded cleanup could not establish even a + successful termination request; retain the original handle for recovery. + +Audit writes are best-effort within the existing budget. Structured logs remain +the fallback when that budget is exhausted or the event store fails. No SDK +message, credential or signed payload URL is copied into these diagnostics. + +## Deployment configuration and permissions + +`microvm_approval_suspend_enabled` accepts true/false and defaults false. +It sets both `MICROVM_APPROVAL_SUSPEND_ENABLED` on a configured coordinator and +the String parameter `//microvm-approval-suspend-enabled`. +The stable parameter name is passed as `MICROVM_APPROVAL_SUSPEND_PARAMETER_NAME`. +Both must allow sleep. Compatible per-worker image evidence remains an +independent admission check. + +Durable executions retain their original Lambda version and environment. +An environment-only redeploy therefore cannot disable an existing execution. +The supervisor rereads Parameter Store without caching before saving new +suspend intent and again before the pre-command gate read. Missing, invalid or +unavailable values disable new sleep without counting as compute failure. +Disable after intent commit records compensating wake and skips Suspend. +Waking, timeout and cleanup never depend on reading this setting. + +| Role | Added permissions | +|---|---| +| Coordinator | SuspendMicrovm/ResumeMicrovm on the configured image ARN and its version suffix | +| Coordinator | GetItem/ConditionCheckItem on the existing approvals table | +| Coordinator | GetParameter on this deployment's exact suspension parameter | +| Approve/deny API | GetMicrovm/ResumeMicrovm on that same image scope | +| CloudFormation execution role | Parameter lifecycle/tagging only under `backgroundagent-*/microvm-approval-suspend-enabled`, in the conditional MicroVM bootstrap policy | + +Cancel retains its Terminate-only grant. Worker roles gain no lifecycle actions +or approval-control authority. No token minting or ingress permissions are added. +The table supplied in `microvmConfig` is the source for both the coordinator's +approval-table environment variable and its read/condition-check grant. + +Refresh bootstrap to **1.8.0** before deploying the parameter. The old +`/cdk-bootstrap/*` SSM grant does not cover application settings. +Deploy the matching coordinator, API roles and six-hook image with the flag +false first. Verify the actual image version/capability and effective role +permissions. Then enable it for controlled development acceptance cases. + +For operational rollback, set the live parameter to `false` and verify its +stored value. Also deploy context `false` so the declared configuration agrees. +A direct Parameter Store update is immediate configuration drift until the +declaration is reconciled. CloudFormation need not rewrite a parameter whose +declared value has not changed, so always verify the live value. A suspension +already dispatched can still finish; existing sleepers retain normal wake and +deadline recovery. Do not remove Resume/Terminate permissions while workers can +still be suspended. A function version originally deployed with static false +stays opted out even if the live parameter is later enabled. + +## Validation record + +The initial combined run passed 673 tests across 15 suites, including 41 real +DynamoDB Local cases. Two new database cases exercise approval during Suspend +and approval just after the intent transaction. They use the real supervisor, +store and conditional expressions, with mocked compute only. + +The configured full-stack fixture initially passed 17 MicroVM checks. After the +live switch was added, 340 tests across 13 suites passed, covering uncached reads, +disable after intent commit, real handler/config composition, exact IAM and the +bootstrap policy/golden/size checks. The final root run includes all stack fixtures. + +Full root `mise run build` passed, exit 0 in 640.39 seconds: + +- CDK: 230 suites, 5,028 tests and one snapshot. +- Agent: 1,951 Python tests. +- CLI: 62 suites and 928 tests; Forge: 11 tests. +- Compilation, lint, formatting, types/drift checks, bundled synthesis, + documentation build and links all pass. +- DynamoDB Local was enabled: 41 lifecycle and 15 capacity cases ran. + +The full run caught two old start-recovery assertions omitting the newly bounded +Terminate request's AbortSignal; they now assert it. A subsequent comment lint +failure was corrected before the final passing run. Final evidence: +`p3-supervisor-root-build-r4-20260915.log`. + +Local evidence does not establish AWS timing, IAM effectiveness, actual frozen +credential renewal or deployed durable replay. + +Evidence directory: `/tmp/abca-645-p2-clean-20260913/`. + +## Relevant code + +- `cdk/src/handlers/shared/microvm-supervisor.ts`: durable reconciliation and + bounded cleanup diagnostics. +- `cdk/src/handlers/shared/microvm-approval-wake.ts`: bounded post-commit wake. +- `cdk/src/handlers/shared/microvm-lifecycle.ts`: strong snapshots and conditional intent. +- `cdk/src/handlers/shared/agent-heartbeat.ts`: shared liveness thresholds. +- `cdk/src/handlers/shared/microvm-control.ts`: safe diagnostic identifiers. +- `cdk/src/handlers/shared/microvm-suspend-config.ts`: uncached bounded live switch. +- `cdk/src/handlers/orchestrate-task.ts` and `shared/orchestrator.ts`: poll/finalize ownership. +- `cdk/src/handlers/approve-task.ts` and `deny-task.ts`: accepted decisions and + best-effort wake/event callbacks. +- `cdk/src/constructs/task-orchestrator.ts`, `task-api.ts` and `stacks/agent.ts`: + flag, table and scoped IAM wiring. diff --git a/yarn.lock b/yarn.lock index 7abfc6483..02f75ecc5 100644 --- a/yarn.lock +++ b/yarn.lock @@ -663,6 +663,20 @@ "@smithy/types" "^4.17.2" tslib "^2.6.2" +"@aws-sdk/client-ssm@^3.1078.0": + version "3.1132.0" + resolved "https://registry.yarnpkg.com/@aws-sdk/client-ssm/-/client-ssm-3.1132.0.tgz#b2c731fa3708366e52b647e1ae16057239c9061a" + integrity sha512-PLFPGor3lEvrhpgt9QOfFY1KBqIaoR+aSUgZ0EtzfVJ8KamNX5Lg63iXc+PBMJJpDTI7pQ2Ko1VIbcOgGr3K0A== + dependencies: + "@aws-sdk/core" "^3.978.0" + "@aws-sdk/credential-provider-node" "^3.972.83" + "@aws-sdk/types" "^3.974.5" + "@smithy/core" "^3.33.3" + "@smithy/fetch-http-handler" "^5.7.2" + "@smithy/node-http-handler" "^4.11.3" + "@smithy/types" "^4.17.2" + tslib "^2.6.2" + "@aws-sdk/client-sts@^3.1078.0": version "3.1119.0" resolved "https://registry.yarnpkg.com/@aws-sdk/client-sts/-/client-sts-3.1119.0.tgz#2b080c257fc5f4549f2cb743546cbb5ae9309034" @@ -720,6 +734,20 @@ bowser "^2.11.0" tslib "^2.6.2" +"@aws-sdk/core@^3.978.0": + version "3.978.0" + resolved "https://registry.yarnpkg.com/@aws-sdk/core/-/core-3.978.0.tgz#115e5d2e03edd380fe270d4617db0f4cc75869d6" + integrity sha512-2yX9LUmxPklVjSGTb8dfnWRJSiFQ3TeH2nn7G1mdKHTfnabzF0+gfrS8rYfLWmZrQ8A3mEcxMJjRc51dL5KWaA== + dependencies: + "@aws-sdk/types" "^3.974.5" + "@aws-sdk/xml-builder" "^3.972.40" + "@aws/lambda-invoke-store" "^0.3.0" + "@smithy/core" "^3.33.3" + "@smithy/signature-v4" "^5.6.12" + "@smithy/types" "^4.17.2" + bowser "^2.11.0" + tslib "^2.6.2" + "@aws-sdk/credential-provider-env@^3.972.55": version "3.972.55" resolved "https://registry.yarnpkg.com/@aws-sdk/credential-provider-env/-/credential-provider-env-3.972.55.tgz#1129acf8860db362a30a02a531387fd399448268" @@ -753,6 +781,17 @@ "@smithy/types" "^4.17.2" tslib "^2.6.2" +"@aws-sdk/credential-provider-env@^3.972.71": + version "3.972.71" + resolved "https://registry.yarnpkg.com/@aws-sdk/credential-provider-env/-/credential-provider-env-3.972.71.tgz#98261bde85bd0f2e2d4a7ba97dce679072e16aff" + integrity sha512-JN+JHruYZw3GUZB8YGAlDk4wTDPOEAEEdEzj5nS0xodWR4smzHsN7PnK2j6IeOsDIj2aqua5DSbhXl9Gtf90FQ== + dependencies: + "@aws-sdk/core" "^3.978.0" + "@aws-sdk/types" "^3.974.5" + "@smithy/core" "^3.33.3" + "@smithy/types" "^4.17.2" + tslib "^2.6.2" + "@aws-sdk/credential-provider-http@^3.972.57": version "3.972.57" resolved "https://registry.yarnpkg.com/@aws-sdk/credential-provider-http/-/credential-provider-http-3.972.57.tgz#4f73bb2f0e03525ab11e001ac18c39cb120dd14c" @@ -792,6 +831,19 @@ "@smithy/types" "^4.17.2" tslib "^2.6.2" +"@aws-sdk/credential-provider-http@^3.972.73": + version "3.972.73" + resolved "https://registry.yarnpkg.com/@aws-sdk/credential-provider-http/-/credential-provider-http-3.972.73.tgz#879466b23dcaec4bee635abf1103e1f5f35d122d" + integrity sha512-uyYYnJOnlis8uQzaYGPd7N1JoioCoNpXgnkXYixsWJXHXgXyYi8WXJSDfofxJeWfQIGWLe2Nwyq60Uc7MZdVOg== + dependencies: + "@aws-sdk/core" "^3.978.0" + "@aws-sdk/types" "^3.974.5" + "@smithy/core" "^3.33.3" + "@smithy/fetch-http-handler" "^5.7.2" + "@smithy/node-http-handler" "^4.11.3" + "@smithy/types" "^4.17.2" + tslib "^2.6.2" + "@aws-sdk/credential-provider-ini@^3.972.62": version "3.972.62" resolved "https://registry.yarnpkg.com/@aws-sdk/credential-provider-ini/-/credential-provider-ini-3.972.62.tgz#a2ca7fc7a1899c0a4a38e501baa666cf1449f2e9" @@ -849,6 +901,25 @@ "@smithy/types" "^4.17.2" tslib "^2.6.2" +"@aws-sdk/credential-provider-ini@^3.973.16": + version "3.973.16" + resolved "https://registry.yarnpkg.com/@aws-sdk/credential-provider-ini/-/credential-provider-ini-3.973.16.tgz#b51db2069b915d563e05969cac845f624aed1dec" + integrity sha512-i++ly+0Uxa+u3ebSSyr0S/3CFhFJDxCXT3+Zj+mW2bXenEx5bKGCdTIKFu39SgXBNhWDjex/8cXUx9MUTMCrTw== + dependencies: + "@aws-sdk/core" "^3.978.0" + "@aws-sdk/credential-provider-env" "^3.972.71" + "@aws-sdk/credential-provider-http" "^3.972.73" + "@aws-sdk/credential-provider-login" "^3.972.78" + "@aws-sdk/credential-provider-process" "^3.972.71" + "@aws-sdk/credential-provider-sso" "^3.973.15" + "@aws-sdk/credential-provider-web-identity" "^3.972.77" + "@aws-sdk/nested-clients" "^3.997.45" + "@aws-sdk/types" "^3.974.5" + "@smithy/core" "^3.33.3" + "@smithy/credential-provider-imds" "^4.4.16" + "@smithy/types" "^4.17.2" + tslib "^2.6.2" + "@aws-sdk/credential-provider-login@^3.972.61": version "3.972.61" resolved "https://registry.yarnpkg.com/@aws-sdk/credential-provider-login/-/credential-provider-login-3.972.61.tgz#f5ba696bae9af3505ae5f3e2a70d7e216d0d37f6" @@ -885,6 +956,18 @@ "@smithy/types" "^4.17.2" tslib "^2.6.2" +"@aws-sdk/credential-provider-login@^3.972.78": + version "3.972.78" + resolved "https://registry.yarnpkg.com/@aws-sdk/credential-provider-login/-/credential-provider-login-3.972.78.tgz#c1c7ca497201a7f2bd6ba7683fa3a896c4e218f8" + integrity sha512-eUtswnXu0+Ii9ieRK+0L7aPFV3Z/dnW2VntJzjBP9xs8s+8p5nBNuymIXtXwZ+5r5+XJP3e32nMkuZ/r0HozEA== + dependencies: + "@aws-sdk/core" "^3.978.0" + "@aws-sdk/nested-clients" "^3.997.45" + "@aws-sdk/types" "^3.974.5" + "@smithy/core" "^3.33.3" + "@smithy/types" "^4.17.2" + tslib "^2.6.2" + "@aws-sdk/credential-provider-node@^3.972.61", "@aws-sdk/credential-provider-node@^3.972.64": version "3.972.64" resolved "https://registry.yarnpkg.com/@aws-sdk/credential-provider-node/-/credential-provider-node-3.972.64.tgz#282456c24bad616faefe2a6c68701913f8289c10" @@ -953,6 +1036,23 @@ "@smithy/types" "^4.17.2" tslib "^2.6.2" +"@aws-sdk/credential-provider-node@^3.972.83": + version "3.972.83" + resolved "https://registry.yarnpkg.com/@aws-sdk/credential-provider-node/-/credential-provider-node-3.972.83.tgz#f4e16b3a1d5d0237460c1b532de37eb6505985dc" + integrity sha512-jdso7ejzfRnatxMUZK4S/U6KbaDPCvfIV4XL+IQAPFDBt5rj5Fq595euqlK8Le4lNCMFR9oUpt+1l0aMgaayOQ== + dependencies: + "@aws-sdk/credential-provider-env" "^3.972.71" + "@aws-sdk/credential-provider-http" "^3.972.73" + "@aws-sdk/credential-provider-ini" "^3.973.16" + "@aws-sdk/credential-provider-process" "^3.972.71" + "@aws-sdk/credential-provider-sso" "^3.973.15" + "@aws-sdk/credential-provider-web-identity" "^3.972.77" + "@aws-sdk/types" "^3.974.5" + "@smithy/core" "^3.33.3" + "@smithy/credential-provider-imds" "^4.4.16" + "@smithy/types" "^4.17.2" + tslib "^2.6.2" + "@aws-sdk/credential-provider-process@^3.972.55": version "3.972.55" resolved "https://registry.yarnpkg.com/@aws-sdk/credential-provider-process/-/credential-provider-process-3.972.55.tgz#bebd30d0065ca34f64e14a870f54a09f23cf11e2" @@ -986,6 +1086,17 @@ "@smithy/types" "^4.17.2" tslib "^2.6.2" +"@aws-sdk/credential-provider-process@^3.972.71": + version "3.972.71" + resolved "https://registry.yarnpkg.com/@aws-sdk/credential-provider-process/-/credential-provider-process-3.972.71.tgz#5a2ff72bcc2d7a658ca6472193ea33437c8b5a4e" + integrity sha512-lYmXJa4gvq4xN1lrT5NiP5vIYYKcGWAdj8y+8o6dlcateB5eF3Dn8DtmjjHKfMBrTPAMr2pebIiX/UOj8c1/UA== + dependencies: + "@aws-sdk/core" "^3.978.0" + "@aws-sdk/types" "^3.974.5" + "@smithy/core" "^3.33.3" + "@smithy/types" "^4.17.2" + tslib "^2.6.2" + "@aws-sdk/credential-provider-sso@^3.972.61": version "3.972.61" resolved "https://registry.yarnpkg.com/@aws-sdk/credential-provider-sso/-/credential-provider-sso-3.972.61.tgz#8c15846c65d2f03bfca3347626e1fcd1a0f40726" @@ -1025,6 +1136,19 @@ "@smithy/types" "^4.17.2" tslib "^2.6.2" +"@aws-sdk/credential-provider-sso@^3.973.15": + version "3.973.15" + resolved "https://registry.yarnpkg.com/@aws-sdk/credential-provider-sso/-/credential-provider-sso-3.973.15.tgz#6a1391ca239f6d505dcb55ea6655b8c14fad5af7" + integrity sha512-6Jhcf4v0pSFdjk1EW2kvzuEBKD+UZ2uNcHUIglKKLndD20YhvkL2kdmDOV5/j4mYuWWwe/a1FQ1aomU86/Cg5Q== + dependencies: + "@aws-sdk/core" "^3.978.0" + "@aws-sdk/nested-clients" "^3.997.45" + "@aws-sdk/token-providers" "3.1129.0" + "@aws-sdk/types" "^3.974.5" + "@smithy/core" "^3.33.3" + "@smithy/types" "^4.17.2" + tslib "^2.6.2" + "@aws-sdk/credential-provider-web-identity@^3.972.61": version "3.972.61" resolved "https://registry.yarnpkg.com/@aws-sdk/credential-provider-web-identity/-/credential-provider-web-identity-3.972.61.tgz#1938cc425e6156acd673484f10b437c5c4c85158" @@ -1061,6 +1185,18 @@ "@smithy/types" "^4.17.2" tslib "^2.6.2" +"@aws-sdk/credential-provider-web-identity@^3.972.77": + version "3.972.77" + resolved "https://registry.yarnpkg.com/@aws-sdk/credential-provider-web-identity/-/credential-provider-web-identity-3.972.77.tgz#5e288111937fcb6f9f02236fa153095d6e24d22f" + integrity sha512-uylIQSUWpfLuH2LovxEEfwzJGM/SabLOfLMg6YXu/E8jJEKUdpdILCVCQCdFvHyu/7dLJOHPMfrSwduxO56NkQ== + dependencies: + "@aws-sdk/core" "^3.978.0" + "@aws-sdk/nested-clients" "^3.997.45" + "@aws-sdk/types" "^3.974.5" + "@smithy/core" "^3.33.3" + "@smithy/types" "^4.17.2" + tslib "^2.6.2" + "@aws-sdk/dynamodb-codec@^3.973.26", "@aws-sdk/dynamodb-codec@^3.973.29": version "3.973.29" resolved "https://registry.yarnpkg.com/@aws-sdk/dynamodb-codec/-/dynamodb-codec-3.973.29.tgz#ea72996b2e1f2c687254c69c357fa4278e42f1da" @@ -1211,6 +1347,20 @@ "@smithy/types" "^4.17.2" tslib "^2.6.2" +"@aws-sdk/nested-clients@^3.997.45": + version "3.997.45" + resolved "https://registry.yarnpkg.com/@aws-sdk/nested-clients/-/nested-clients-3.997.45.tgz#3ed0e5fcd25c889025b2379a9434594fe7056a4a" + integrity sha512-mooq9Q+jLa18VoM7HouczmslZU60iiB0aKc/Ztnq/luIL1ud0z4DnYprLR/ZO1gp331S9tJctM1HZr7u6YKBXQ== + dependencies: + "@aws-sdk/core" "^3.978.0" + "@aws-sdk/signature-v4-multi-region" "^3.996.46" + "@aws-sdk/types" "^3.974.5" + "@smithy/core" "^3.33.3" + "@smithy/fetch-http-handler" "^5.7.2" + "@smithy/node-http-handler" "^4.11.3" + "@smithy/types" "^4.17.2" + tslib "^2.6.2" + "@aws-sdk/s3-presigned-post@^3.1078.0": version "3.1081.0" resolved "https://registry.yarnpkg.com/@aws-sdk/s3-presigned-post/-/s3-presigned-post-3.1081.0.tgz#ed27961af8e772e23745e7ff34f34083bc69c291" @@ -1314,6 +1464,18 @@ "@smithy/types" "^4.17.2" tslib "^2.6.2" +"@aws-sdk/token-providers@3.1129.0": + version "3.1129.0" + resolved "https://registry.yarnpkg.com/@aws-sdk/token-providers/-/token-providers-3.1129.0.tgz#a4a11a789d7259874ccf3733e538fc70f8a289e4" + integrity sha512-Sbl3rpzQdsG4ZK2zh0JWUYyZPKKorJlVOddA2T0DVbKJFrsW8J6wgnslxxUH04+WaBMr4A1HzJZvZX0xUvkniA== + dependencies: + "@aws-sdk/core" "^3.978.0" + "@aws-sdk/nested-clients" "^3.997.45" + "@aws-sdk/types" "^3.974.5" + "@smithy/core" "^3.33.3" + "@smithy/types" "^4.17.2" + tslib "^2.6.2" + "@aws-sdk/types@^3.222.0", "@aws-sdk/types@^3.973.15": version "3.973.15" resolved "https://registry.yarnpkg.com/@aws-sdk/types/-/types-3.973.15.tgz#98a4860bed33c32c7088924d0ab52f9eabbdf7c3" From 2a5787266ab3e1c025ed81954d881cba352ad654 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Tue, 15 Sep 2026 13:52:12 -0400 Subject: [PATCH 038/149] docs(microvm): align image warnings with P3 supervisor status --- cdk/scripts/package-microvm-artifact.sh | 4 +++- cdk/src/constructs/lambda-microvm-compute.ts | 7 ++++--- cdk/test/constructs/lambda-microvm-compute.test.ts | 4 +++- 3 files changed, 10 insertions(+), 5 deletions(-) diff --git a/cdk/scripts/package-microvm-artifact.sh b/cdk/scripts/package-microvm-artifact.sh index 48b89486c..8d7d1b4d0 100755 --- a/cdk/scripts/package-microvm-artifact.sh +++ b/cdk/scripts/package-microvm-artifact.sh @@ -266,7 +266,9 @@ REMINDER (ADR-021 P2): clean deployment and coding, iteration and cancellation Ready/validate hooks, heartbeat, logging, Memory writes and cleanup have live evidence. Full P2 acceptance still needs the failure/recovery, effective IAM and networking matrix in docs/verification/645-p3-implementation-plan.md. - /suspend and /resume are declared; supervisor integration and live P3 acceptance remain open. + /suspend and /resume are declared; supervisor integration is implemented. + P3 requires bootstrap bundle 1.8.0 and defaults new suspension off. + Live P3 acceptance remains open. CDK retains warning ID abca:microvm-image-p1-smoke-unverified for compatibility; its text describes the current verification gaps. EOF diff --git a/cdk/src/constructs/lambda-microvm-compute.ts b/cdk/src/constructs/lambda-microvm-compute.ts index a9c635769..a477d7c2c 100644 --- a/cdk/src/constructs/lambda-microvm-compute.ts +++ b/cdk/src/constructs/lambda-microvm-compute.ts @@ -733,8 +733,8 @@ export interface LambdaMicrovmComputeProps extends LambdaMicrovmImageInputs { * `abca:microvm-image-p1-smoke-unverified` warning below records that scope. * P3 also declares the served `/suspend` and `/resume` hooks and bakes a non-secret * protocol marker into the image. The coordinator verifies the actual launched - * version before allowing suspension; supervisor integration and live acceptance - * remain separate rollout gates. + * version before allowing suspension. Supervisor integration is implemented; + * live acceptance remains a separate rollout gate and automatic sleep defaults off. * * ## Deliberately NOT here * @@ -1426,7 +1426,8 @@ export class LambdaMicrovmCompute extends Construct { + 'Heartbeat, logs, Memory writes and cleanup have live evidence. Full P2 acceptance ' + 'still needs the failure/recovery, effective IAM and networking matrix in ' + 'docs/verification/645-p3-implementation-plan.md. P3 checks the actual launched image version; ' - + 'supervisor integration and live sleep/wake acceptance remain open. The warning ID is retained ' + + 'P3 requires bootstrap bundle 1.8.0 and defaults new suspension off; supervisor integration is implemented. ' + + 'Live sleep/wake acceptance remains open. The warning ID is retained ' + 'across phases for existing operator filters.', ); } diff --git a/cdk/test/constructs/lambda-microvm-compute.test.ts b/cdk/test/constructs/lambda-microvm-compute.test.ts index a0549d3c3..ae3c85331 100644 --- a/cdk/test/constructs/lambda-microvm-compute.test.ts +++ b/cdk/test/constructs/lambda-microvm-compute.test.ts @@ -1009,7 +1009,9 @@ describe('LambdaMicrovmCompute — image provisioned from a managed base image', for (const hook of ['/ready', '/validate', '/run', '/terminate', '/suspend', '/resume']) { expect(message).toContain(hook); } - expect(message).toContain('supervisor integration and live sleep/wake acceptance remain open'); + expect(message).toContain('supervisor integration is implemented'); + expect(message).toContain('P3 requires bootstrap bundle 1.8.0 and defaults new suspension off'); + expect(message).toContain('Live sleep/wake acceptance remains open'); }); test('enables every hook the agent serves, and only those (rendered form)', () => { From 2c51c9e284adf3e67d6ccaba9c96e6ae03e27012 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Tue, 15 Sep 2026 14:50:12 -0400 Subject: [PATCH 039/149] fix(orchestrator): retain versions needed by durable executions --- cdk/src/constructs/task-orchestrator.ts | 5 ++++- cdk/src/stacks/agent.ts | 7 +++++++ cdk/test/constructs/task-orchestrator.test.ts | 8 ++++++++ cdk/test/stacks/agent.test.ts | 8 ++++++++ docs/verification/645-p3-credentials.md | 9 +++++---- docs/verification/645-p3-guest-barrier.md | 9 +++++---- docs/verification/645-p3-lifecycle-hooks.md | 6 +++--- docs/verification/645-p3-supervisor.md | 15 +++++++++++++++ 8 files changed, 55 insertions(+), 12 deletions(-) diff --git a/cdk/src/constructs/task-orchestrator.ts b/cdk/src/constructs/task-orchestrator.ts index 65466ebc0..5a7eb5db9 100644 --- a/cdk/src/constructs/task-orchestrator.ts +++ b/cdk/src/constructs/task-orchestrator.ts @@ -18,7 +18,7 @@ */ import * as path from 'path'; -import { ArnFormat, Duration, Stack } from 'aws-cdk-lib'; +import { ArnFormat, Duration, RemovalPolicy, Stack } from 'aws-cdk-lib'; import * as cloudwatch from 'aws-cdk-lib/aws-cloudwatch'; import * as dynamodb from 'aws-cdk-lib/aws-dynamodb'; import * as iam from 'aws-cdk-lib/aws-iam'; @@ -437,6 +437,9 @@ export class TaskOrchestrator extends Construct { executionTimeout: Duration.hours(DURABLE_EXECUTION_TIMEOUT_HOURS), retentionPeriod: Duration.days(DURABLE_RETENTION_DAYS), }, + // Durable executions replay their original code and environment after a + // deployment. Keep published versions until no execution can resume them. + currentVersionOptions: { removalPolicy: RemovalPolicy.RETAIN }, environment: { // Solution-attribution component label (#319): orchestration plane. ABCA_COMPONENT: 'orchestr', diff --git a/cdk/src/stacks/agent.ts b/cdk/src/stacks/agent.ts index c2db0236e..d60ed598f 100644 --- a/cdk/src/stacks/agent.ts +++ b/cdk/src/stacks/agent.ts @@ -478,6 +478,13 @@ export class AgentStack extends Stack { }); inputGuardrail.createVersion('Initial version'); + // A retained durable Lambda version also pins GUARDRAIL_VERSION. Preserve + // that immutable dependency across updates, even when CDK creates a new one. + for (const child of inputGuardrail.node.findAll()) { + if (child instanceof CfnResource && child.cfnResourceType === 'AWS::Bedrock::GuardrailVersion') { + child.applyRemovalPolicy(RemovalPolicy.RETAIN); + } + } // --- TaskApi is constructed before the orchestrator (which it needs the // ARN of) and before the Runtime (which it needs the ARN of, for the diff --git a/cdk/test/constructs/task-orchestrator.test.ts b/cdk/test/constructs/task-orchestrator.test.ts index 4c5cd709e..63d92a6b0 100644 --- a/cdk/test/constructs/task-orchestrator.test.ts +++ b/cdk/test/constructs/task-orchestrator.test.ts @@ -187,6 +187,14 @@ describe('TaskOrchestrator construct', () => { }); }); + test('retains published versions so in-flight durable executions can replay after deployment', () => { + baseTemplate.resourceCountIs('AWS::Lambda::Version', 1); + baseTemplate.hasResource('AWS::Lambda::Version', { + DeletionPolicy: 'Retain', + UpdateReplacePolicy: 'Retain', + }); + }); + test('grants AgentCore runtime invocation permissions with wildcard sub-resource', () => { baseTemplate.hasResourceProperties('AWS::IAM::Policy', { PolicyDocument: { diff --git a/cdk/test/stacks/agent.test.ts b/cdk/test/stacks/agent.test.ts index bedb13c5c..4da420417 100644 --- a/cdk/test/stacks/agent.test.ts +++ b/cdk/test/stacks/agent.test.ts @@ -45,6 +45,14 @@ describe('AgentStack', () => { expect(template).toBeDefined(); }); + test('retains the guardrail version referenced by pinned durable environments', () => { + template.resourceCountIs('AWS::Bedrock::GuardrailVersion', 1); + template.hasResource('AWS::Bedrock::GuardrailVersion', { + DeletionPolicy: 'Retain', + UpdateReplacePolicy: 'Retain', + }); + }); + test('AgentCore runtime has no direct DynamoDB grant, including capacity counters', () => { const roles = Object.entries(template.findResources('AWS::IAM::Role')); const runtimeRoleIds = roles.filter(([, role]) => diff --git a/docs/verification/645-p3-credentials.md b/docs/verification/645-p3-credentials.md index 6f2f3af7a..4294cd2ca 100644 --- a/docs/verification/645-p3-credentials.md +++ b/docs/verification/645-p3-credentials.md @@ -1,10 +1,10 @@ # #645 P3: scoped credentials across sleep -**Follow-up (2026-09-15):** [per-worker image capability](./645-p3-image-capability.md) now declares the served hooks and verifies the actual launched image version locally. The milestone below records its original scope; supervisor integration and live sleep/wake acceptance remain open. +**Follow-up (2026-09-15):** [per-worker image capability](./645-p3-image-capability.md) declares the served hooks and verifies the actual launched image version. The [supervisor milestone](./645-p3-supervisor.md) implements durable recovery and approval wake; the full repository build passes. The record below preserves this milestone's original scope. Live sleep/wake acceptance remains open. Date: 2026-09-14. Local implementation and actual pinned Claude process probes. No AWS deployment or suspension was performed for this milestone. Deployed -image **2.0** remains unchanged; P3 is not complete. +image **2.0** was unchanged at that milestone; P3 was not complete. ## What changed @@ -123,5 +123,6 @@ Before automatic suspension can ship: subprocess behavior and long-expiry sleep in AWS. The local CLI probe uses an explicit settings file containing the production helper command; it does not install `/etc/claude-code/managed-settings.json` on the developer machine. -4. Finish image capability, durable supervisor recovery and approval-triggered wake, - then complete the [P3 plan](./645-p3-implementation-plan.md), including its P2 gates. +4. **Implemented locally in the later image/supervisor milestones:** image capability, + durable supervisor recovery and approval-triggered wake. Complete their live + verification and the [P3 plan](./645-p3-implementation-plan.md), including its P2 gates. diff --git a/docs/verification/645-p3-guest-barrier.md b/docs/verification/645-p3-guest-barrier.md index 01db61e98..30ad56699 100644 --- a/docs/verification/645-p3-guest-barrier.md +++ b/docs/verification/645-p3-guest-barrier.md @@ -1,9 +1,9 @@ # #645 P3 guest pause controller -**Follow-up (2026-09-15):** [per-worker image capability](./645-p3-image-capability.md) now declares the served hooks and verifies the actual launched image version locally. The milestone below records its original scope; supervisor integration and live sleep/wake acceptance remain open. +**Follow-up (2026-09-15):** [per-worker image capability](./645-p3-image-capability.md) declares the served hooks and verifies the actual launched image version. The [supervisor milestone](./645-p3-supervisor.md) implements durable recovery and approval wake; the full repository build passes. The record below preserves this milestone's original scope. Live sleep/wake acceptance remains open. -Date: 2026-09-14. Local implementation; deployed image `2.0` still has only -ready, validate, run and terminate hooks. Automatic sleeping remains disabled. +Date: 2026-09-14. Local implementation; at this milestone, deployed image `2.0` +had only ready, validate, run and terminate hooks. Automatic sleeping was disabled. ## What is implemented @@ -46,7 +46,8 @@ adds cached successful acknowledgments and clears an old wake result for each ne The later [HTTP hook milestone](./645-p3-lifecycle-hooks.md) supplies production callbacks for atomic checkpoint writes, credential renewal and gate reconciliation. -Image capability, supervisor integration and live service verification remain open. +At this milestone, image capability, supervisor integration and live service +verification remained open; the follow-up above records later implementation. The worker has several independent credential consumers: diff --git a/docs/verification/645-p3-lifecycle-hooks.md b/docs/verification/645-p3-lifecycle-hooks.md index aff1f03c5..636c412ae 100644 --- a/docs/verification/645-p3-lifecycle-hooks.md +++ b/docs/verification/645-p3-lifecycle-hooks.md @@ -1,11 +1,11 @@ # #645 P3: worker suspend/resume hooks -**Follow-up (2026-09-15):** [per-worker image capability](./645-p3-image-capability.md) now declares the served hooks and verifies the actual launched image version locally. The milestone below records its original scope; supervisor integration and live sleep/wake acceptance remain open. +**Follow-up (2026-09-15):** [per-worker image capability](./645-p3-image-capability.md) declares the served hooks and verifies the actual launched image version. The [supervisor milestone](./645-p3-supervisor.md) implements durable recovery and approval wake; the full repository build passes. The record below preserves this milestone's original scope. Live sleep/wake acceptance remains open. Date: 2026-09-14. Local implementation and DynamoDB Local verification. Nothing in this milestone was deployed. Managed image **2.0** and automatic -suspension remain unchanged. P3 still requires image capability, supervisor -integration and the live acceptance matrix in the [plan](./645-p3-implementation-plan.md). +suspension were unchanged. At that point, P3 still required image capability, +supervisor integration and the live acceptance matrix in the [plan](./645-p3-implementation-plan.md). ## Behavior diff --git a/docs/verification/645-p3-supervisor.md b/docs/verification/645-p3-supervisor.md index 40cc6250f..e18a6f50d 100644 --- a/docs/verification/645-p3-supervisor.md +++ b/docs/verification/645-p3-supervisor.md @@ -109,6 +109,15 @@ independent admission check. Durable executions retain their original Lambda version and environment. An environment-only redeploy therefore cannot disable an existing execution. +The stack retains published coordinator versions and their immutable guardrail +versions across updates. Otherwise, a rollout could delete the code or guardrail +that an older execution still references. The live alias advances to the new +coordinator version. Retained versions require operator cleanup only after no +execution can resume or retry them; changing the live alias is not that proof. +For the first upgrade from an unretained deployment, drain existing executions +before removing the old versions, or retain those existing resources in a +separate update before replacing them. + The supervisor rereads Parameter Store without caching before saving new suspend intent and again before the pre-command gate read. Missing, invalid or unavailable values disable new sleep without counting as compute failure. @@ -170,6 +179,12 @@ Terminate request's AbortSignal; they now assert it. A subsequent comment lint failure was corrected before the final passing run. Final evidence: `p3-supervisor-root-build-r4-20260915.log`. +The deployment change-set review then exposed missing retention for published +coordinator and guardrail versions. The fix passed 192 focused infrastructure +tests and another complete root build in 657.59 seconds: 5,030 CDK tests, +1,951 Python tests, 928 CLI tests and all other configured checks. DynamoDB Local +remained enabled. Evidence: `p3-retention-root-build-20260915.log`. + Local evidence does not establish AWS timing, IAM effectiveness, actual frozen credential renewal or deployed durable replay. From f19429d652f2a42026b3ce2d2039d82e60aea60d Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Tue, 15 Sep 2026 16:19:39 -0400 Subject: [PATCH 040/149] fix(microvm): stop approval workers on cancellation --- cdk/src/handlers/cancel-task.ts | 12 ++++++--- cdk/test/handlers/cancel-task.test.ts | 36 +++++++++++++++++++-------- 2 files changed, 35 insertions(+), 13 deletions(-) diff --git a/cdk/src/handlers/cancel-task.ts b/cdk/src/handlers/cancel-task.ts index 9963f1f9f..b8802aade 100644 --- a/cdk/src/handlers/cancel-task.ts +++ b/cdk/src/handlers/cancel-task.ts @@ -81,7 +81,13 @@ export async function handler(event: APIGatewayProxyEvent): Promise { }); describe('cancel-task handler', () => { - test('cancels a running task successfully', async () => { + test.each(['HYDRATING', 'RUNNING', 'AWAITING_APPROVAL', 'FINALIZING'])('stops an AgentCore session when cancelling %s', async (status) => { + mockSend.mockReset(); + mockSend + .mockResolvedValueOnce({ Item: { ...RUNNING_TASK, status } }) + .mockResolvedValueOnce({}) + .mockResolvedValueOnce({}); const result = await handler(makeEvent()); expect(result.statusCode).toBe(200); @@ -187,11 +192,14 @@ describe('cancel-task handler', () => { expect(result.statusCode).toBe(409); expect(JSON.parse(result.body).error.code).toBe('TASK_ALREADY_TERMINAL'); + expect(mockAgentCoreSend).not.toHaveBeenCalled(); + expect(mockEcsSend).not.toHaveBeenCalled(); + expect(mockMicrovmSend).not.toHaveBeenCalled(); }); - test('returns 409 on ConditionalCheckFailedException (race condition)', async () => { + test.each(['RUNNING', 'AWAITING_APPROVAL'])('does not stop %s compute when cancellation loses a terminal race', async (status) => { mockSend.mockReset(); - mockSend.mockResolvedValueOnce({ Item: RUNNING_TASK }); + mockSend.mockResolvedValueOnce({ Item: { ...RUNNING_TASK, status, compute_type: 'lambda-microvm' } }); const condError = new Error('Condition not met'); condError.name = 'ConditionalCheckFailedException'; mockSend.mockRejectedValueOnce(condError); @@ -200,6 +208,9 @@ describe('cancel-task handler', () => { expect(result.statusCode).toBe(409); expect(JSON.parse(result.body).error.code).toBe('TASK_ALREADY_TERMINAL'); + expect(mockAgentCoreSend).not.toHaveBeenCalled(); + expect(mockEcsSend).not.toHaveBeenCalled(); + expect(mockMicrovmSend).not.toHaveBeenCalled(); }); test('returns 500 on unexpected DynamoDB error', async () => { @@ -232,21 +243,23 @@ describe('cancel-task handler', () => { expect(typeof eventCall.input.Item.ttl).toBe('number'); }); - test('can cancel tasks in SUBMITTED state', async () => { + test.each(['SUBMITTED', 'QUEUED', 'PENDING_UPLOADS'])('cancels %s without stopping compute', async (status) => { mockSend.mockReset(); mockSend - .mockResolvedValueOnce({ Item: { ...RUNNING_TASK, status: 'SUBMITTED' } }) + .mockResolvedValueOnce({ Item: { ...RUNNING_TASK, status } }) .mockResolvedValueOnce({}) .mockResolvedValueOnce({}); const result = await handler(makeEvent()); expect(result.statusCode).toBe(200); expect(mockAgentCoreSend).not.toHaveBeenCalled(); + expect(mockEcsSend).not.toHaveBeenCalled(); + expect(mockMicrovmSend).not.toHaveBeenCalled(); }); - test('does not call StopRuntimeSession when RUNNING but session_id is missing', async () => { + test.each(['RUNNING', 'AWAITING_APPROVAL'])('does not call StopRuntimeSession when %s has no session_id', async (status) => { mockSend.mockReset(); - const noSession = { ...RUNNING_TASK }; + const noSession = { ...RUNNING_TASK, status }; delete (noSession as { session_id?: string }).session_id; mockSend .mockResolvedValueOnce({ Item: noSession }) @@ -287,10 +300,11 @@ describe('cancel-task handler', () => { expect(mockAgentCoreSend).toHaveBeenCalled(); }); - test('cancels ECS-backed running task via StopTask', async () => { + test.each(['HYDRATING', 'RUNNING', 'AWAITING_APPROVAL', 'FINALIZING'])('stops an ECS session when cancelling %s', async (status) => { mockSend.mockReset(); const ecsTask = { ...RUNNING_TASK, + status, compute_type: 'ecs', compute_metadata: { clusterArn: 'arn:aws:ecs:us-east-1:123456789012:cluster/agent-cluster', @@ -332,7 +346,7 @@ describe('cancel-task handler', () => { expect(mockAgentCoreSend).not.toHaveBeenCalled(); }); - test('cancels a lambda-microvm task via TerminateMicrovm, not AgentCore or ECS', async () => { + test.each(['HYDRATING', 'RUNNING', 'AWAITING_APPROVAL', 'FINALIZING'])('terminates only the MicroVM session when cancelling %s', async (status) => { mockSend.mockReset(); // NOTE: RUNNING_TASK carries `agent_runtime_arn`, and RUNTIME_ARN is also set // in this suite's env — exactly the mixed-deployment shape that would send a @@ -340,6 +354,7 @@ describe('cancel-task handler', () => { // first. This test is the regression guard for that branch ordering. const microvmTask = { ...RUNNING_TASK, + status, compute_type: 'lambda-microvm', session_id: 'mvm-0123456789abcdef', compute_metadata: { @@ -378,10 +393,11 @@ describe('cancel-task handler', () => { expect(mockAgentCoreSend).not.toHaveBeenCalled(); }); - test('a TerminateMicrovm failure still returns 200 (the CANCELLED write stands)', async () => { + test.each(['RUNNING', 'AWAITING_APPROVAL'])('keeps cancellation committed when TerminateMicrovm fails for %s', async (status) => { mockSend.mockReset(); const microvmTask = { ...RUNNING_TASK, + status, compute_type: 'lambda-microvm', compute_metadata: { microvmId: 'mvm-0123456789abcdef', endpoint: 'https://x' }, }; From 61e1176c54a5e078b785c6a2e2fd698e30461816 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Tue, 15 Sep 2026 16:55:56 -0400 Subject: [PATCH 041/149] docs(microvm): record P3 deployment and six live guest checks --- agent/workflows/default/agent-v1.yaml | 6 +- docs/verification/645-p3-image-capability.md | 8 +- .../645-p3-implementation-plan.md | 20 +- .../645-p3-live-deployment-20260915.md | 244 ++++++++++++++++++ docs/verification/645-p3-readiness-review.md | 2 +- docs/verification/645-p3-supervisor.md | 9 +- 6 files changed, 275 insertions(+), 14 deletions(-) create mode 100644 docs/verification/645-p3-live-deployment-20260915.md diff --git a/agent/workflows/default/agent-v1.yaml b/agent/workflows/default/agent-v1.yaml index f0910879b..5cbfb5033 100644 --- a/agent/workflows/default/agent-v1.yaml +++ b/agent/workflows/default/agent-v1.yaml @@ -6,7 +6,7 @@ # hatch — so its behavior is auditable and overridable per-repo via the Blueprint. # # Deliberately conservative because it runs when *nothing* was specified: -# requires_repo:false (no clone), a read-leaning tool set (no Bash/Write/Edit), +# requires_repo:false (repository optional), a read-leaning tool set (no Bash/Write/Edit), # and delivery to S3 + a comment milestone. A caller who wants coding selects # (or maps to) a coding workflow. soft_deny is still mandatory (read_only:false) # so any future tool addition stays HITL-gated. @@ -23,7 +23,9 @@ domain: hybrid description: >- Run the user's request through the agent and deliver the result. Minimal default — no repo, build, or PR assumptions. -# No clone; if a repo is supplied it is hydrated as context, not scaffolded. +# Without a repo, pipeline.py skips cloning and delivers an S3 artifact. When a +# repo is supplied, the current shared pipeline clones it and runs its repo-bound +# build/PR post-hooks. This flag alone does not disable that delivery path. requires_repo: false read_only: false prompt: diff --git a/docs/verification/645-p3-image-capability.md b/docs/verification/645-p3-image-capability.md index b2194e42e..d1216e4be 100644 --- a/docs/verification/645-p3-image-capability.md +++ b/docs/verification/645-p3-image-capability.md @@ -1,8 +1,12 @@ # ADR-021 P3: per-worker image capability Date: 2026-09-15. Local implementation on `fix/645-microvm-readiness`, following -the [guest hook milestone](./645-p3-lifecycle-hooks.md). This has not been deployed; -the last verified AWS image is still `2.0`. Automatic suspension remains disabled. +the [guest hook milestone](./645-p3-lifecycle-hooks.md). At this milestone, the +last verified AWS image was `2.0` and automatic suspension was disabled. + +**Deployment follow-up:** the [P3 live record](./645-p3-live-deployment-20260915.md) +verifies active image `3.0`, protocol `1`, and initial isolated guest checks. +Automatic suspension remains disabled; full live acceptance is still open. ## What this adds diff --git a/docs/verification/645-p3-implementation-plan.md b/docs/verification/645-p3-implementation-plan.md index ce4ba65fb..52f189536 100644 --- a/docs/verification/645-p3-implementation-plan.md +++ b/docs/verification/645-p3-implementation-plan.md @@ -23,7 +23,16 @@ or unreadable support permits normal coding and disables new suspension. Databas race tests reject changed identities and recover a committed capability after a lost reply. No deployment or automatic sleep was enabled in that milestone. -**Supervisor milestone (2026-09-15):** [production supervision and approval wake](./645-p3-supervisor.md) now connect the policy/store to durable polling and post-commit approve/deny handlers. Recovery clocks and the original service lifetime survive replay; failed wake and cleanup remain visible. The rollout flag defaults off and uses a live Parameter Store switch for existing durable executions. Full repository validation passes (5,028 CDK, 1,951 Python and 928 CLI tests); all live P3 gates remain open. +**Supervisor milestone (2026-09-15):** [production supervision and approval wake](./645-p3-supervisor.md) connect the policy/store to durable polling and post-commit approve/deny handlers. Recovery clocks and the original service lifetime survive replay; failed wake and cleanup remain visible. The rollout flag defaults off and uses a live Parameter Store switch for existing durable executions. That milestone passed full repository validation (5,028 CDK, 1,951 Python and 928 CLI tests). + +**P3 deployment follow-up (2026-09-15):** the [live record](./645-p3-live-deployment-20260915.md) +verifies bootstrap `1.8.0`, source `9a5f4606`, six-hook image `3.0`, 475 root resources +and both suspension settings off. Rollout review exposed missing retention for +pinned coordinator/guardrail versions; the fix passed another full build +(5,030 CDK tests), and existing versions were protected before replacement. +Initial isolated guest cases pass. These use a local production +supervisor and manual guarded Suspend; deployed durable entrypoint, automatic +suspension admission and the rest of the acceptance matrix remain open. **Live infrastructure and image deployed (2026-09-14):** the [clean P2 deployment record](./645-p2-clean-deployment-20260913.md) tracks the new @@ -34,8 +43,8 @@ validate hooks passed, and authenticated API reads passed after the update. The subsequent [image rebuild fix](./645-microvm-image-rebuild-20260914.md) deployed `e1d5debe` through a normal reviewed update and activated version `2.0`. Its ready/validate hooks passed; repeated packaging reused the artifact, and -redeploying the same cloud assembly reported no changes. The root still has -474 resources. +redeploying the same cloud assembly reported no changes. The root had +474 resources at that milestone. The [live task verification](./645-p2-live-task-20260914.md) on image `1.0` passed normal coding, PR iteration and cancellation on `isadeks/vercel-abca-linear`, including heartbeat, npm checks, Memory writes and automatic cleanup. The @@ -57,7 +66,7 @@ with local process/reply faults. Saved-handle recovery, changed-input refusal, cancellation and the actual 120-second cutoff passed. These tests use operator credentials and do not interrupt the deployed durable Lambda. Deployed-coordinator recovery and the wider IAM/network matrix still remain. -Full P2 acceptance and all P3 live gates remain open. The batch notes below +Full P2 acceptance and the remaining P3 live gates remain open. The batch notes below record what was verified at their original completion; their deployment status is superseded by these records. @@ -87,7 +96,7 @@ is superseded by these records. - [x] Give managed image builds immutable, checksum-verified artifacts and require their digest in deployment context; packaging, construct, stack and CDK-nag regressions and the full build pass. - [x] Verify a normal CloudFormation update builds and activates image `2.0` from the changed artifact URI; repeat packaging reuses the verified object and a same-assembly redeploy reports no changes. - Deferred at user request: publish mise tasks in the target repository and verify its default commands. The CLI addition and temporary overrides were withdrawn; this repository configuration work is outside the current P3 implementation. -- Optional: production nesting remains unimplemented; the clean deployment uses 474 of the root stack's 500 resource slots. P3 does not inherently require nesting. Recheck the count for supported feature combinations and validate the split/migration if adopted. +- Optional: production nesting remains unimplemented; the P3 deployment uses 475 of the root stack's 500 resource slots. P3 does not inherently require nesting. Recheck the count for supported feature combinations and validate the split/migration if adopted. - [x] Add mandatory pause/wake command methods across all three compute strategies, with explicit unsupported results and bounded MicroVM requests. - [x] Keep the original approval deadline through database writes and polling, including frozen/backward clocks; preserve decision races and cancellation. - [x] Save gate/VM-bound lifecycle intent with stale-writer protection; add explicit VM observations and a tested policy helper. @@ -97,6 +106,7 @@ is superseded by these records. - [x] Persist bounded poll/recovery counters and connect lifecycle policy to the supervisor. - [x] Connect post-commit approval wake, bounded diagnostics/cleanup, the default-off rollout flag and scoped IAM. - [x] Complete full repository validation with the P3 supervisor and live switch. +- [x] Deploy the supervisor and six-hook image with suspension disabled; verify six isolated guest cases, repair the discovered cancellation stop omission, and prove API termination before test cleanup. - [ ] Deploy and verify the complete P3 sleep/wake lifecycle in AWS. First prerequisite batch completed locally on 2026-09-13: diff --git a/docs/verification/645-p3-live-deployment-20260915.md b/docs/verification/645-p3-live-deployment-20260915.md new file mode 100644 index 000000000..d42308a8e --- /dev/null +++ b/docs/verification/645-p3-live-deployment-20260915.md @@ -0,0 +1,244 @@ +# ADR-021 P3 development deployment and guest checks + +Date: 2026-09-15. The development stack is deployed with the P3 supervisor and +six-hook image. Automatic suspension remains disabled. Six isolated guest +cases pass; suspended cancellation exposed a stop-path bug that was repaired, +deployed and verified with a stronger probe below. +The full [P3 acceptance matrix](./645-p3-implementation-plan.md#7-acceptance-matrix-and-completion-gates) +is not complete. + +## Deployed configuration + +| Item | Verified value | +|---|---| +| Stack | `backgroundagent-dev`, account ``, `us-west-2` | +| Source | `9a5f4606` on `fix/645-microvm-readiness` | +| Cancellation follow-up | `aff5e637`; cancellation Lambda code updated separately | +| Bootstrap | `1.8.0` | +| Stack status | `UPDATE_COMPLETE` | +| Root resources | 475 | +| MicroVM image | `backgroundagent-dev-abca-agent`, version `3.0`, `ACTIVE` / `SUCCESSFUL` | +| Guest protocol | `ABCA_MICROVM_LIFECYCLE_PROTOCOL=1` | +| Coordinator `live` alias | Lambda version `3` | +| Static suspension flag | `false` | +| Live SSM parameter | `/backgroundagent-dev/microvm-approval-suspend-enabled`, String `false`, version 1 | + +The Lambda version identifies the supervisor's saved code. The MicroVM image +version identifies the worker's saved starting computer. They are separate +version sequences. + +The image has ready, validate, run, suspend, resume and terminate hooks. Suspend +and resume each have a 30-second hook timeout. The verified image artifact is: + +```text +SHA256: 6c695b782f108064c9d5ad461e67f2071d15d1cae71e7aef47afb6dfaba7f3c3 +110 files; 483188 bytes +``` + +The coordinator's deployed code SHA256 is +`r9JIjsu84GQOfFYAepGVHKZrdps4q/bd2PeOEMCS8PY=`. + +## Rollout review and retention repair + +The initial application change set exposed a missing prerequisite: it removed +the previous coordinator and guardrail versions without retaining them. A +durable execution keeps its original Lambda version and environment, including +its guardrail version. Deleting either dependency can break later recovery. + +The fix in `9a5f4606` retains both kinds of version across updates. It passed +192 focused infrastructure tests and the full root build in 657.59 seconds: +5,030 CDK tests, one snapshot, 1,951 Python tests, 928 CLI tests and the other +configured build checks. The 56 DynamoDB Local lifecycle/capacity cases ran. + +Before replacing the existing unretained versions, a separate update adopted +retention for them. Its resource properties were identical to the previous +deployment. The only direct changes were `DeletionPolicy` and +`UpdateReplacePolicy` on those two versions; CloudFormation also listed +dependent references for reevaluation. Both live policies were verified as +`Retain` after that update. + +The final application change set contained 164 modifications, three additions +and two removals. Both removals had `PolicyAction: Retain`. There were no +table, bucket, user-pool or secret changes, no required resource replacements, +and the coordinator alias and MicroVM image retained their identities. +The stack update started at 19:11:06 UTC and was observed complete at 19:19:02 UTC. + +The combined coordinator policy comparison found no removed permissions. Added +permissions were image-scoped GetMicrovmImageVersion/Suspend/Resume, approval +GetItem/ConditionCheckItem, and exact-parameter GetParameter. Template/policy +comparison is distinct from exercising every permission against the service. + +The IAM simulator evaluated the deployed coordinator, API and worker roles +against the configured image and an unrelated image. The coordinator's tested +lifecycle actions were allowed only on the configured image; approve/deny had +Get/Resume, cancellation had Terminate, and the execution role had none of those +actions. The coordinator could read its exact live-switch parameter and could +not read the unrelated test parameter. No unrelated-image action was allowed. +The tenant-session simulation also had no allowed lifecycle action, but reported +missing DynamoDB/tag context; it is not the broader cross-task IAM acceptance +test. Actual approve/deny calls separately verified the positive Resume path. + +### Preserve the original deployment template + +In this environment, CloudFormation `GetTemplate` returned question marks in +place of Unicode characters through both the CLI and JavaScript SDK. Reusing +that response produced unrelated changes, including a Cedar layer replacement. +That change set was discarded without execution. + +The original template downloaded from the deployment's S3 asset matched the +saved assembly and preserved those characters. The retention adoption used that +artifact. Minifying the JSON also kept it within the 1,000,000-byte S3-template +limit; pretty-printing it exceeded that limit. + +## Isolated guest acceptance + +The private probe creates one owned task with no repository or notification +destination. It uses the production launch strategy, image-capability checks and +local production supervisor. For long waits it saves and rechecks a valid +suspend intent before manually requesting Suspend. Decisions invoke the +deployed approve/deny/cancel handlers with the fixture's owner identity. + +This covers actual guest hooks and decision-handler IAM. It does not exercise +API Gateway authentication, admission/capacity reservation or the deployed +durable coordinator entrypoint. The global suspension settings remain false. + +Each task asks for exactly one `Read` of `/etc/os-release`. Tool-result events +distinguish a successful read from a denied attempt. Cleanup confirms worker +termination before deleting launch payloads, then verifies the task's S3 prefix +is empty. The strengthened cancellation check requires API-driven termination +before this independent cleanup runs. + +| Case | Observed result | +|---|---| +| Ordinary, no gate | `COMPLETED`; one Read call and successful result; 17 watcher cycles, no scheduling gaps or recovery failures; service lifetime verified; VM terminated and payload deletion requested | +| Long gate + approve | Observed `SUSPENDED`; deployed approve returned 202; same task completed; one successful Read; one approval gate; original approval clock unchanged; VM terminated and payload deletion requested | +| Long gate + deny | Frozen worker woke; deployed deny returned 202; one Read attempt returned an authoritative denial, with no retry; task completed and VM terminated | +| Original-deadline timeout | Worker was RUNNING about 54 seconds before expiry; original request became TIMED_OUT; tool error arrived 140 ms after the original deadline; no retry; task completed and VM terminated | +| Decision during grace | Enabled pure policy returned `wait` / `suspend-grace`; deployed approval committed within seven seconds of gate creation; no Suspend request; one successful Read; task completed | +| Suspended cancellation | After repairing the first run's stop omission, the API logged TerminateMicrovm and AWS reported TERMINATED before fixture cleanup; task stayed CANCELLED; one gated Read attempt and no tool result | + +Ordinary task: `01M2K9H3J4V23EST0SWNE228WR`, worker +`microvm-a724cc8d-f31e-3e2d-8866-3b1ef47beaab`. + +Approve task: `01M2K9QKBSP5NQ3Q8BNNB058MH`, worker +`microvm-2f927077-c8fb-3ae5-a177-e92818409da2`. Its request +`01M2K9SMP624BEKKN4ME2K3P6R` was created at 19:48:10 UTC with a 300-second timeout. +Suspend was requested at 19:48:44.782; AWS reported SUSPENDED at 19:48:50.056. +Approval committed at 19:48:50.782 and returned 202 at 19:48:52.035. Completion +was observed at 19:48:57.119. The request's creation time and timeout were +unchanged. The worker completed between polls, so AWS RUNNING after wake was not +separately sampled. + +Deny task: `01M2KA3MFW5RAX7TT9GRHPN2WQ`, worker +`microvm-59ae9e19-4722-3143-97a9-725af4381a64`. The request was created at +19:54:23 UTC with a 300-second timeout. AWS reported SUSPENDED at 19:55:03.289; +the denial committed at 19:55:03.941 and returned 202 at 19:55:05.267. +RUNNING was observed at 19:55:10.465 and completion at 19:55:15.549. +The single Read result had `is_error=true` and an authoritative human denial. +The approval row retained its original creation time and timeout. + +Timeout task: `01M2KA8QAVWAZ61VQH12KAZS0X`, worker +`microvm-b669614a-3613-3e1f-9747-91c0c83d2f42`. Request +`01M2KAACH57GRDTMNK3M1Z5W6F` was created at 19:57:19 UTC with a 180-second timeout. +The worker was observed SUSPENDED at 19:57:55.132, gained a resume intent at +19:59:20.009 and was RUNNING at 19:59:25.281 while the approval remained pending. +Its one Read attempt returned `User timed_out` at 20:00:19.140, 140 ms after the +original 20:00:19 deadline. Completion was observed at 20:00:23.076. The original +approval row became TIMED_OUT without changing its creation time or timeout. + +Short-gate task: `01M2KAPN555FJJ0C5779GQT1RA`, worker +`microvm-9d71d590-524d-33a3-85c4-ab64e27f74f3`. Its request was created at +20:04:43 UTC with a 300-second timeout. At 20:04:48.970 the policy still required +the grace period. Approval committed at 20:04:49.706, the one Read succeeded at +20:04:52.118, and completion was observed at 20:04:55.835. The original clock +was unchanged. No Suspend request was issued. + +Read-only S3 checks confirmed zero objects under all eight completed task +prefixes, including the initial ordinary attempt below. Shared bootstrap +manifests retain their normal lifecycle policy. Task/trace evidence is retained. + +An earlier ordinary probe's guest also completed, but the local Mac entered +Maintenance Sleep for 900 seconds immediately after the first poll. The watcher +then exceeded its own deadline. `pmset` timestamps confirmed the cause. That run +does not establish supervisor timing; subsequent probes use command-scoped +`caffeinate -is` and record scheduling gaps. + +### Cancellation defect found by the audit + +The first suspended-cancellation task was `01M2KB0SX2M4JW413E8BWN4ZY2`, worker +`microvm-b7cd04f8-ee57-3341-b5ec-9f607570e310`. AWS reported SUSPENDED at +20:10:54.053; the deployed cancel handler returned 200 at 20:10:55.759. +The task became CANCELLED and the pending Read produced no tool result. + +However, the handler logged no TerminateMicrovm call. Source review confirmed +that its `status === RUNNING` guard excluded AWAITING_APPROVAL, the state used +by a sleeping approval worker. The fixture's independent cleanup subsequently +terminated the VM, masking the API's omission in the original exit-code check. +This run does **not** pass API-termination acceptance. + +The repair attempts to stop a saved session during HYDRATING, RUNNING, +AWAITING_APPROVAL or FINALIZING across all three compute substrates. It still +commits cancellation first, preserves a successful response if stopping fails, +and makes no stop call when the conditional cancellation loses a terminal race. +Pre-session states continue to skip stopping compute. + +Ten new regression assertions failed before the fix; all 34 cancellation tests +passed afterward. The related approval/supervisor/cancellation run passed 149 +tests in five suites, and compilation/lint passed. The strengthened live probe +polls for TERMINATED/not-found with a 60-second deadline after calling the API, +before any independent cleanup. + +The repair is committed as `aff5e637`. Fresh full synthesis also changed +unrelated runtime asset references and guardrail version IDs. The Bedrock alpha +construct derives the inner version ID from an unresolved UpdatedAt token; +identical guardrail settings did not produce identical version IDs in these +syntheses. Investigating that churn remains a follow-up. + +The repair's cloud assembly therefore uses the exact previously deployed +template with only the cancellation Lambda's Code/Metadata replaced by the +new CDK bundle. It publishes two file assets and no Docker assets. AWS's reviewed +change set contained that one direct modification and two unchanged API ARN +references for reevaluation. There were no image, guardrail, coordinator, IAM or +suspension-setting changes. The update was executed at 20:28:39 UTC and reached +UPDATE_COMPLETE. The live function checksum matches the published ZIP: +`L1sQUO2flAB3sO2nA+7xcuR058eJH2haBUIvhgyvhPg=`. + +The strict retry task was `01M2KC7632R97JN3YZ17RSN9GT`, worker +`microvm-0641d6d3-afd4-3d73-88a2-06a0d304c86e`. AWS reported SUSPENDED at +20:31:46.938. The deployed API logged TerminateMicrovm at 20:31:48.532 and +returned 200 at 20:31:48.576. AWS reported TERMINATED at 20:31:51.047, before +fixture cleanup. The task stayed CANCELLED, the one gated Read had no tool +result, and its payload prefix was empty. The original approval clock stayed +unchanged; cancellation leaves the approval row PENDING under a terminal task. +This passes API-driven suspended termination, but does not establish slot +accounting or cancellation races during suspend/resume transitions. + +## Evidence and remaining gates + +Private evidence directory: `/tmp/abca-645-p2-clean-20260913/`. + +- `p3-retention-root-build-20260915.log`: final full validation. +- `p3-retention-adoption-verified-20260915.json`: existing version protection. +- `p3-supervisor-reviewed-changeset-r2-20260915.json`: executed application review. +- `p3-live-configuration-verified-20260915.json`: live image, alias and flags. +- `p3-live-iam-simulation-summary-20260915.json`: per-resource lifecycle evaluations. +- `p3-live-ssm-iam-simulation-20260915.json`: exact switch versus unrelated setting. +- `p3-guest-ordinary-r2-20260915/acceptance-review.json`: ordinary task and tool evidence. +- `p3-guest-approve-20260915/acceptance-review.json`: freeze, decision, tool and cleanup evidence. +- `p3-guest-deny-20260915/acceptance-review.json`: denial prevented execution. +- `p3-guest-timeout-20260915/acceptance-review.json`: original-deadline comparison. +- `p3-guest-short-20260915/acceptance-review.json`: early decision and successful read. +- `p3-guest-cancel-suspended-20260915/acceptance-review.json`: original cancellation defect. +- `p3-guest-cancel-suspended-r2-20260915/acceptance-review.json`: API termination before cleanup. +- `p3-cancel-regression-before-20260915.log`: ten failing assertions before repair. +- `p3-cancel-regression-after-20260915.log`: all 34 cancellation tests pass after repair. +- `p3-cancel-related-tests-20260915.log`: 149 related tests pass. +- `p3-cancel-reviewed-changeset-20260915.json`: narrow CloudFormation repair. +- `p3-cancel-deployment-verified-20260915.json`: deployed code checksum. +- `p3-guest-payload-cleanup-verified-20260915.json`: empty task-owned S3 prefixes. + +Remaining acceptance includes deployed durable +replay/recovery, live suspension admission/rollback, credentials expiring while +frozen, lifecycle fault/race injection, and the residual P2 effective-IAM, +network and capacity/migration checks. These remain open in the +[implementation plan](./645-p3-implementation-plan.md). diff --git a/docs/verification/645-p3-readiness-review.md b/docs/verification/645-p3-readiness-review.md index 5cf2b17f4..a6e1f005a 100644 --- a/docs/verification/645-p3-readiness-review.md +++ b/docs/verification/645-p3-readiness-review.md @@ -22,7 +22,7 @@ tests with fake AWS responses prove renewal before the next request and safe failure without borrowing the parent's keys. The subsequent [HTTP hook milestone](./645-p3-lifecycle-hooks.md) connects pause to an atomic checkpoint and wake to credential renewal plus task/gate reconciliation. -The [image capability milestone](./645-p3-image-capability.md) (2026-09-15) now declares the six hooks and checks/persists support for the actual launched image version. The [supervisor milestone](./645-p3-supervisor.md) now adds durable recovery, post-commit approval wake, bounded cleanup and a default-off rollout flag locally. Real AWS sleep/wake acceptance remains open. +The [image capability milestone](./645-p3-image-capability.md) (2026-09-15) declares the six hooks and checks/persists support for the actual launched image version. The [supervisor milestone](./645-p3-supervisor.md) adds durable recovery, post-commit approval wake, bounded cleanup and default-off suspension settings. The [development deployment](./645-p3-live-deployment-20260915.md) verifies image `3.0` and six isolated guest cases, including API-driven suspended cancellation after repairing its stop-path defect. Full AWS durable-workflow and sleep/wake acceptance remains open. The reservation review found a separate trust boundary: the old agent role could write/replace/delete its task row, including coordinator metadata. A subsequent local fix restricts main-task writes to reporting/approval attributes, removes replacement/deletion permissions and removes unused worker counter grants. It also removes unused Python submission/session-info helpers and corrects overstated tenant-isolation comments. [Metadata verification](./645-coordinator-metadata.md) records the tests and pending AWS gate. Status reports still come from the agent, and the compute role chooses session tags; this is not complete hostile-worker isolation. diff --git a/docs/verification/645-p3-supervisor.md b/docs/verification/645-p3-supervisor.md index e18a6f50d..c22287b75 100644 --- a/docs/verification/645-p3-supervisor.md +++ b/docs/verification/645-p3-supervisor.md @@ -1,9 +1,10 @@ # ADR-021 P3: supervisor and approval wake -Status (2026-09-15): production integration and full repository validation pass -locally. This milestone has not been deployed. Automatic suspension defaults -off. The [P3 plan](./645-p3-implementation-plan.md) retains the live acceptance -gates and remaining P2 checks. +Status (2026-09-15): production integration and full repository validation pass. +The [development deployment](./645-p3-live-deployment-20260915.md) now runs the +supervisor and six-hook image `3.0`; initial isolated guest checks +pass. Automatic suspension remains off. The [P3 plan](./645-p3-implementation-plan.md) +retains the remaining live acceptance gates and P2 checks. ## What this does From 92e1f777e8de43ef8ea96412553daeaa41415276 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Tue, 15 Sep 2026 19:39:15 -0400 Subject: [PATCH 042/149] fix: preserve approval callbacks through MicroVM suspension (#645) --- agent/scripts/verify_approval_hook_timeout.py | 281 +++++++++++++++++ agent/src/hooks.py | 30 +- agent/tests/test_hooks.py | 25 ++ .../strategies/lambda-microvm-strategy.ts | 10 +- cdk/test/scripts/check-constants-sync.test.ts | 4 + contracts/constants.json | 3 +- contracts/constants.md | 13 +- docs/verification/645-coordinator-metadata.md | 6 +- .../645-effective-iam-20260915.md | 109 +++++++ docs/verification/645-p3-callback-timeout.md | 80 +++++ .../645-p3-durable-live-20260915.md | 295 ++++++++++++++++++ .../645-p3-implementation-plan.md | 32 +- .../645-p3-live-deployment-20260915.md | 13 +- docs/verification/645-p3-readiness-review.md | 10 + scripts/check-constants-sync.ts | 18 +- 15 files changed, 903 insertions(+), 26 deletions(-) create mode 100644 agent/scripts/verify_approval_hook_timeout.py create mode 100644 docs/verification/645-effective-iam-20260915.md create mode 100644 docs/verification/645-p3-callback-timeout.md create mode 100644 docs/verification/645-p3-durable-live-20260915.md diff --git a/agent/scripts/verify_approval_hook_timeout.py b/agent/scripts/verify_approval_hook_timeout.py new file mode 100644 index 000000000..57823a866 --- /dev/null +++ b/agent/scripts/verify_approval_hook_timeout.py @@ -0,0 +1,281 @@ +#!/usr/bin/env python3 +# Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. +# SPDX-License-Identifier: MIT-0 + +"""Opt-in pinned CLI approval callback probe; loopback model and synthetic keys. + +Compare --configured microvm --delay 650 with --delay 650 for the CLI default. +An explicit --timeout 1 --delay 3 provides a short failure control. This is +outside the unit suite; it tests actual SDK/CLI transport, not VM freezing. +""" + +import argparse +import asyncio +import importlib.metadata +import json +import os +import subprocess +import sys +import tempfile +import threading +import time +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer +from pathlib import Path + +from verify_microvm_credentials import MODEL, event_frame, model_response + +ROOT = Path(__file__).resolve().parents[2] +MAX_PROBE_DELAY_S = 900 + + +def require(condition: bool, detail: str) -> None: + if not condition: + raise RuntimeError(detail) + + +async def probe(timeout, delay, configured): + import claude_agent_sdk + from claude_agent_sdk import ClaudeAgentOptions, ClaudeSDKClient, ResultMessage + from claude_agent_sdk.types import HookMatcher + + cli = Path(claude_agent_sdk.__file__).parent / "_bundled/claude" + version = subprocess.run( + [str(cli), "--version"], capture_output=True, text=True, check=True, timeout=10 + ).stdout.strip() + require(version == "2.1.191 (Claude Code)", "Probe requires reviewed CLI 2.1.191") + require( + importlib.metadata.version("claude-agent-sdk") == "0.2.110", + "Probe requires reviewed SDK 0.2.110", + ) + audit = [] + lifecycle = None + pre_matcher = HookMatcher(timeout=timeout) + if configured: + sys.path.insert(0, str(ROOT / "agent/src")) + from hooks import build_hook_matchers + from microvm_lifecycle import register_task + from policy import PolicyEngine + + if configured == "microvm": + lifecycle = register_task("local-hook-probe", "local-probe-vm") + # PolicyEngine logs with os.write(1), bypassing sys.stdout. Redirect + # only synchronous setup, before the probe starts any server threads. + saved_stdout = os.dup(1) + try: + os.dup2(2, 1) + pre_matcher = build_hook_matchers( + engine=PolicyEngine(task_type="new_task", repo="probe/owned"), + task_id="local-hook-probe", + )["PreToolUse"][0] + finally: + os.dup2(saved_stdout, 1) + os.close(saved_stdout) + with tempfile.TemporaryDirectory(prefix="abca-645-hook-timeout-") as directory: + temp = Path(directory) + target = temp / "owned-marker.txt" + target.write_text("OWNED_READ_MARKER\n") + settings = temp / "settings.json" + settings.write_text("{}") + aws_config = temp / "aws-config" + aws_config.write_text("[default]\nregion = us-west-2\n") + aws_creds = temp / "aws-credentials" + aws_creds.write_text("") + config = temp / "claude-config" + config.mkdir() + started = time.monotonic() + + def record(kind, **data): + audit.append({"kind": kind, "elapsed_s": time.monotonic() - started, **data}) + + def tool_response(): + events = [ + { + "type": "message_start", + "message": { + "id": "msg_tool", + "type": "message", + "role": "assistant", + "model": MODEL, + "content": [], + "stop_reason": None, + "stop_sequence": None, + "usage": {"input_tokens": 1, "output_tokens": 0}, + }, + }, + { + "type": "content_block_start", + "index": 0, + "content_block": { + "type": "tool_use", + "id": "toolu_owned", + "name": "Read", + "input": {}, + }, + }, + { + "type": "content_block_delta", + "index": 0, + "delta": { + "type": "input_json_delta", + "partial_json": json.dumps({"file_path": str(target)}), + }, + }, + {"type": "content_block_stop", "index": 0}, + { + "type": "message_delta", + "delta": { + "stop_reason": "tool_use", + "stop_sequence": None, + }, + "usage": {"output_tokens": 1}, + }, + {"type": "message_stop"}, + ] + return b"".join(event_frame(event) for event in events) + + class Handler(BaseHTTPRequestHandler): + def log_message(self, format, *args): + del format, args + + def do_POST(self): + body = json.loads(self.rfile.read(int(self.headers["Content-Length"]))) + prior = [ + item + for message in body.get("messages", []) + for item in message.get("content", []) + if isinstance(item, dict) and item.get("type") == "tool_result" + ] + record("model-request", results=prior) + stream = model_response() if prior else tool_response() + self.send_response(200) + self.send_header("Content-Type", "application/vnd.amazon.eventstream") + self.send_header("Content-Length", str(len(stream))) + self.end_headers() + self.wfile.write(stream) + + server = ThreadingHTTPServer(("127.0.0.1", 0), Handler) + thread = threading.Thread(target=server.serve_forever, daemon=True) + thread.start() + endpoint = f"http://127.0.0.1:{server.server_port}" + + async def pre(data, tool_id, ctx): + require(data["tool_name"] == "Read", "Unexpected tool") + require(data["tool_input"] == {"file_path": str(target)}, "Unexpected target") + record("pre-start", tool_id=tool_id) + try: + await asyncio.sleep(delay) + record("pre-allow") + return { + "hookSpecificOutput": { + "hookEventName": "PreToolUse", + "permissionDecision": "allow", + "permissionDecisionReason": "Owned local test gate released", + } + } + except BaseException as error: + record("pre-cancelled", error_type=type(error).__name__) + raise + + async def post(data, tool_id, ctx): + record("post", tool_name=data["tool_name"]) + return {} + + env = { + "CLAUDE_CONFIG_DIR": str(config), + "CLAUDE_CODE_USE_BEDROCK": "1", + "CLAUDE_CODE_DISABLE_NONESSENTIAL_TRAFFIC": "1", + "CLAUDE_CODE_MAX_RETRIES": "0", + "DISABLE_TELEMETRY": "1", + "DISABLE_ERROR_REPORTING": "1", + "DISABLE_AUTOUPDATER": "1", + "ANTHROPIC_BEDROCK_BASE_URL": endpoint, + "AWS_ENDPOINT_URL": endpoint, + "AWS_REGION": "us-west-2", + "AWS_DEFAULT_REGION": "us-west-2", + "AWS_CONFIG_FILE": str(aws_config), + "AWS_SHARED_CREDENTIALS_FILE": str(aws_creds), + "AWS_ACCESS_KEY_ID": "SYNTHETIC_NOT_VALID", + "AWS_SECRET_ACCESS_KEY": "synthetic-not-valid-in-aws", + "AWS_SESSION_TOKEN": "synthetic-not-valid-in-aws", + "AWS_EC2_METADATA_DISABLED": "true", + } + errors = [] + # Retain production matcher settings, replacing only its callback with + # a controlled wait so the transport's cancellation is observable. + pre_matcher.hooks = [pre] + try: + options = ClaudeAgentOptions( + model=MODEL, + max_turns=2, + cwd=directory, + tools=["Read"], + permission_mode="bypassPermissions", + setting_sources=[], + settings=str(settings), + env=env, + stderr=errors.append, + hooks={ + "PreToolUse": [pre_matcher], + "PostToolUse": [HookMatcher(hooks=[post])], + }, + ) + async with ClaudeSDKClient(options=options) as client: + await client.query("Read the owned marker once, then stop.") + async for message in client.receive_response(): + if isinstance(message, ResultMessage): + record("result", is_error=message.is_error) + return { + "sdk_version": "0.2.110", + "cli_version": version, + "timeout_s": pre_matcher.timeout, + "delay_s": delay, + "configured": configured, + "audit": audit, + "stderr": errors, + } + finally: + server.shutdown() + server.server_close() + thread.join(timeout=2) + if lifecycle: + from microvm_lifecycle import unregister_task + + unregister_task(lifecycle) + + +if __name__ == "__main__": + parser = argparse.ArgumentParser() + parser.add_argument("--timeout", type=float) + parser.add_argument("--delay", type=float, default=3) + parser.add_argument("--configured", choices=["microvm", "standard"]) + args = parser.parse_args() + if args.configured and args.timeout is not None: + parser.error("--configured uses production settings; omit --timeout") + if args.delay <= 0 or args.delay > MAX_PROBE_DELAY_S: + parser.error("--delay must be within (0, 900] seconds") + for key in tuple(os.environ): + if key.startswith(("AWS_", "ANTHROPIC_", "CLAUDE_", "OTEL_", "BEDROCK_")): + del os.environ[key] + result = asyncio.run( + asyncio.wait_for(probe(args.timeout, args.delay, args.configured), max(args.delay + 30, 45)) + ) + calls = [event for event in result["audit"] if event["kind"] == "pre-start"] + posts = [event for event in result["audit"] if event["kind"] == "post"] + outcomes = [ + item + for event in result["audit"] + if event["kind"] == "model-request" + for item in event["results"] + ] + require(len(calls) == 1 and len(outcomes) == 1, "Expected one tool call and one result") + if args.configured: + require(len(posts) == 1 and not outcomes[0].get("is_error"), "Read did not complete") + require("OWNED_READ_MARKER" in str(outcomes[0]["content"]), "Owned marker not read") + elif args.timeout is not None and args.timeout < args.delay: + require(not posts and outcomes[0].get("is_error"), "Expired hook permitted the tool") + require( + any(event["kind"] == "pre-cancelled" for event in result["audit"]), + "Expected callback cancellation", + ) + result["verified"] = True + print(json.dumps(result, indent=2)) diff --git a/agent/src/hooks.py b/agent/src/hooks.py index e4b035fcc..27f6e509e 100644 --- a/agent/src/hooks.py +++ b/agent/src/hooks.py @@ -36,6 +36,7 @@ from output_scanner import scan_tool_output from policy import APPROVAL_RATE_LIMIT, FLOOR_TIMEOUT_S, Outcome from progress_writer import _generate_ulid +from shared_constants import SHARED_CONSTANTS from shell import log, log_error_cw from stuck_guard import StuckGuard @@ -1790,8 +1791,9 @@ def build_hook_matchers( Returns a dict mapping HookEvent strings to lists of HookMatcher instances, ready to pass as ``hooks=...`` to ClaudeAgentOptions. - The SDK expects ``dict[HookEvent, list[HookMatcher]]`` where HookMatcher - has ``matcher: str | None`` and ``hooks: list[HookCallback]``. + The SDK expects ``dict[HookEvent, list[HookMatcher]]``. PreToolUse also needs + an explicit callback timeout: the CLI can otherwise cancel a live approval + wait and return a generic denial without completing its approval record. ``progress`` is forwarded to both the PreToolUse hook (approval gate milestones) and the Stop hook (nudge/denial acks). ``user_id`` is @@ -1821,12 +1823,13 @@ async def _pre( hook_input: HookInput, tool_use_id: str | None, ctx: HookContext ) -> HookJSONOutput: # Fail-closed wrapper (mirrors _post and _stop). If the inner hook - # or its dispatch path raises an unexpected exception (asyncio - # cancellation, TypeError from a malformed payload, etc.), the + # or its dispatch path raises an unexpected exception (for example, + # TypeError from a malformed payload), the # SDK's default behaviour for an unhandled hook exception is # undefined — we MUST NOT trust it to fail closed. Mapping every # uncaught exception to a DENY here makes the security posture - # explicit at the SDK boundary. + # explicit at the SDK boundary. SDK cancellation propagates and still + # releases the tool registration in finally. allowed = False try: if lifecycle: @@ -1935,8 +1938,23 @@ async def _stop( # Empty dict == allow stop. SyncHookJSONOutput(**{}) is fine. return SyncHookJSONOutput(**result) + # The callback transport must outlive the approval loop. A frozen MicroVM + # may wake after the gate deadline during supervisor recovery, so keep that + # callback alive for the bounded VM lifetime. The original gate deadline + # still decides permission; this does not give the user more approval time. + callback_window_s = ( + SHARED_CONSTANTS["microvm_lifecycle"]["maximum_duration_seconds"] + if lifecycle + else SHARED_CONSTANTS["approval_timeout_s"]["max"] + ) matchers = { - "PreToolUse": [HookMatcher(matcher=None, hooks=[_pre])], + "PreToolUse": [ + HookMatcher( + matcher=None, + hooks=[_pre], + timeout=callback_window_s + CLEANUP_MARGIN_120S, + ) + ], "PostToolUse": [HookMatcher(matcher=None, hooks=[_post])], "Stop": [HookMatcher(matcher=None, hooks=[_stop])], } diff --git a/agent/tests/test_hooks.py b/agent/tests/test_hooks.py index 3e02708c0..297079fa4 100644 --- a/agent/tests/test_hooks.py +++ b/agent/tests/test_hooks.py @@ -726,6 +726,31 @@ def test_post_hook_matcher_structure(self): assert post_matcher.matcher is None assert len(post_matcher.hooks) == 1 + @pytest.mark.parametrize("microvm", [False, True]) + def test_callback_budget_outlives_approval_without_changing_its_deadline(self, microvm): + from hooks import _ApprovalDeadline + from microvm_lifecycle import register_task, unregister_task + from shared_constants import SHARED_CONSTANTS + + context = register_task("callback-budget", "owned-vm") if microvm else None + engine = PolicyEngine(task_type="new_task", repo="owner/repo", task_default_timeout_s=30) + original = _ApprovalDeadline.from_recorded("2026-09-15T23:00:00Z", 30) + try: + matchers = build_hook_matchers(engine=engine, task_id="callback-budget") + protected_window = ( + SHARED_CONSTANTS["microvm_lifecycle"]["maximum_duration_seconds"] + if microvm + else SHARED_CONSTANTS["approval_timeout_s"]["max"] + ) + assert matchers["PreToolUse"][0].timeout > protected_window + assert engine.task_default_timeout_s == 30 + assert original.wall_deadline == 1789513230 + assert matchers["PostToolUse"][0].timeout is None + assert matchers["Stop"][0].timeout is None + finally: + if context: + unregister_task(context) + def test_matchers_with_trajectory(self): engine = PolicyEngine(task_type="new_task", repo="owner/repo") # Pass None for trajectory — should still work diff --git a/cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts b/cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts index 69ef1bbbb..87fd7bc8c 100644 --- a/cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts +++ b/cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts @@ -107,7 +107,7 @@ const HTTP_REQUEST_TIMEOUT = 408; * so a Blueprint override would be policy without a driver; add one only if a * real need appears. */ -export const MICROVM_MAX_DURATION_SECONDS = 28_800; +export const MICROVM_MAX_DURATION_SECONDS = sharedConstants.microvm_lifecycle.maximum_duration_seconds; /** * Hard service cap on ``runHookPayload`` (bytes), measured live rather than read @@ -460,10 +460,10 @@ function assertImageArn(identifier: string): void { * control-plane state machine the orchestrator can observe through * {@link LambdaMicrovmComputeStrategy.pollSession}. * - * P3 command primitives implement mandatory suspend/resume alongside the - * explicit unsupported results in the other two strategies. The supervisor - * must still supply gate policy, durable intent and state reconciliation - * before automatic suspension can be enabled with compatible agent hooks. + * Suspend/resume submit service commands; the other two strategies return + * explicit unsupported results. microvm-supervisor supplies gate policy, + * durable intent and state reconciliation. Automatic suspension additionally + * requires compatible image hooks and enabled static/live rollout settings. */ export class LambdaMicrovmComputeStrategy implements ComputeStrategy { readonly type = 'lambda-microvm'; diff --git a/cdk/test/scripts/check-constants-sync.test.ts b/cdk/test/scripts/check-constants-sync.test.ts index e232e4a33..0b16bc921 100644 --- a/cdk/test/scripts/check-constants-sync.test.ts +++ b/cdk/test/scripts/check-constants-sync.test.ts @@ -67,6 +67,7 @@ const FIXTURE_FILES = [ 'cdk/src/handlers/shared/payload-bootstrap.ts', 'cdk/src/constructs/lambda-microvm-compute.ts', 'cdk/src/handlers/shared/microvm-image-capability.ts', + 'cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts', ]; interface RunResult { @@ -134,6 +135,8 @@ describe('check-constants-sync', () => { test.each([ ['protocol_version', 0], ['protocol_version', 1.5], ['protocol_version', '1'], ['hook_port', 0], ['hook_port', 65536], ['hook_port', 8080.5], + ['maximum_duration_seconds', 0], ['maximum_duration_seconds', 28801], + ['maximum_duration_seconds', 28800.5], ['maximum_duration_seconds', '28800'], ['image_protocol_env', 'AWS_ACCESS_KEY_ID'], ['image_protocol_env', 'ABCA_MICROVM_bad'], ])('rejects invalid %s=%s', (key, value) => { const result = runInMutatedRepo(root => patchContract(root, json => { @@ -144,6 +147,7 @@ describe('check-constants-sync', () => { }); test.each([ ['cdk/src/constructs/lambda-microvm-compute.ts', 'AGENT_HOOK_PORT', '8080'], + ['cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts', 'MICROVM_MAX_DURATION_SECONDS', '28_800'], ['cdk/src/constructs/lambda-microvm-compute.ts', 'LIFECYCLE_HOOK_TIMEOUT_SECONDS', '30'], ['cdk/src/handlers/shared/microvm-image-capability.ts', 'MICROVM_LIFECYCLE_PROTOCOL', '"1"'], ['cdk/src/handlers/shared/microvm-image-capability.ts', 'MICROVM_LIFECYCLE_PROTOCOL', 'String(1)'], diff --git a/contracts/constants.json b/contracts/constants.json index 49a6064df..0bc985d28 100644 --- a/contracts/constants.json +++ b/contracts/constants.json @@ -80,7 +80,8 @@ "microvm_lifecycle": { "protocol_version": 1, "image_protocol_env": "ABCA_MICROVM_LIFECYCLE_PROTOCOL", - "hook_port": 8080 + "hook_port": 8080, + "maximum_duration_seconds": 28800 }, "linear_vault": { "_why": "The vault caches a grant keyed by the WHOLE token request, customParameters included. A one-token divergence between any two of these copies makes every resolve a cache miss, which post-#812 is reported as consent-required and can latch a healthy workspace revoked.", diff --git a/contracts/constants.md b/contracts/constants.md index e03bd5783..db4e8724d 100644 --- a/contracts/constants.md +++ b/contracts/constants.md @@ -25,6 +25,7 @@ the contract. This is the neutral location both runtimes read. | `cdk/src/handlers/shared/payload-bootstrap.ts`, `cdk/src/constructs/payload-bootstrap-permissions.ts` | `payload_bootstrap` | import-time | | `agent/src/server.py` | `SHARED_CONSTANTS["microvm_platform_config"]`, `SHARED_CONSTANTS["microvm_hook_budgets"]`, `SHARED_CONSTANTS["microvm_lifecycle"]` | import-time | | `agent/src/microvm_http.py` | `SHARED_CONSTANTS["microvm_hook_budgets"]` | import-time | +| `agent/src/hooks.py` | `microvm_lifecycle.maximum_duration_seconds`, `approval_timeout_s.max` | SDK matcher construction | | `cdk/src/handlers/shared/types.ts`, `jira-app-actor.ts` | `../../../../contracts/constants.json` | synth-time `import` | | `cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts` | `microvm_platform_config` | synth-time `import`, read per session start | | `cdk/src/constructs/lambda-microvm-compute.ts` | `microvm_hook_budgets`, `microvm_lifecycle` | synth-time `import` | @@ -100,7 +101,8 @@ JSON at TypeScript compile time via `resolveJsonModule`. "microvm_lifecycle": { "protocol_version": 1, "image_protocol_env": "ABCA_MICROVM_LIFECYCLE_PROTOCOL", - "hook_port": 8080 + "hook_port": 8080, + "maximum_duration_seconds": 28800 } } ``` @@ -226,7 +228,14 @@ The image declares suspend/resume using this service timeout. These values do no enable automatic suspension. `microvm_lifecycle` owns protocol version `1`, marker name -`ABCA_MICROVM_LIFECYCLE_PROTOCOL` and hook port `8080`. The marker is baked into +`ABCA_MICROVM_LIFECYCLE_PROTOCOL`, hook port `8080`, and the 28,800-second +maximum VM lifetime. The strategy sends that lifetime to `RunMicrovm`; the +agent sizes its PreToolUse SDK callback timeout to outlive the same bound by +120 seconds. This callback timeout lets an expired approval finish reconciliation +after a delayed wake; the original approval deadline still controls permission. +Other backends use the maximum approval interval plus the same margin. + +The marker is baked into the immutable image; it is neither a credential nor task/deployment configuration. `/validate` rejects a supplied unsupported marker. The coordinator checks the exact image ARN/version returned by Run, including all six enabled hooks and lifecycle diff --git a/docs/verification/645-coordinator-metadata.md b/docs/verification/645-coordinator-metadata.md index 1784ec45b..33911eca0 100644 --- a/docs/verification/645-coordinator-metadata.md +++ b/docs/verification/645-coordinator-metadata.md @@ -39,7 +39,11 @@ The regression first failed against the old policy because the main-table grant Python contract tests run the real task writers against recording clients, including all terminal-result fields and both approval transactions. They compare the requested attributes against the same JSON list used by CDK and require review when a new writer is added. -These checks inspect generated policies and request compatibility. They do **not** execute AWS authorization. DynamoDB Local, used for the capacity tests, also does not implement IAM. Keep the following live gate open. +These local checks inspect generated policies and request compatibility; they do +not execute AWS authorization. DynamoDB Local also does not implement IAM. +The subsequent [37-case AWS matrix](./645-effective-iam-20260915.md) passed using +the unchanged deployed MicroVM role and real tagged sessions from a temporary +Lambda. Other backend ambient roles and migration/scale remain open. ## AWS acceptance and rollout gate diff --git a/docs/verification/645-effective-iam-20260915.md b/docs/verification/645-effective-iam-20260915.md new file mode 100644 index 000000000..befc530c7 --- /dev/null +++ b/docs/verification/645-effective-iam-20260915.md @@ -0,0 +1,109 @@ +# ADR-021: effective AWS metadata and payload permissions + +Date: 2026-09-15. These checks complement the +[metadata contract](./645-coordinator-metadata.md), +[payload bootstrap](./645-payload-bootstrap.md), and +[durable lifecycle checks](./645-p3-durable-live-20260915.md). + +## Scope + +Temporary Lambda functions used the unchanged deployed MicroVM execution role +in `backgroundagent-dev`, `us-west-2`. Its existing trust permits Lambda. Their +code was limited to fixed disposable records and objects. They assumed real +task-tagged STS sessions. No existing role or bucket policy was modified. + +This tests actual effective AWS authorization. It does not test the MicroVM +network connector, other backends' ambient roles, or ECS rollout. Compute roles +still choose their session tags; hostile-worker identity binding remains a limit. + +## Metadata: 37 checks passed + +Two disposable terminal task rows carried protected owner, worker, start-receipt +and reservation fields. No worker or capacity reservation was acquired for them. + +| Requests | Observed result | +|---|---| +| Each tagged session reads its own task | Allowed | +| Cross-task reads and reads without a task tag | `AccessDeniedException` | +| Own heartbeat and complete reporting attribute set | Allowed | +| Foreign/missing-task-tag updates | `AccessDeniedException` | +| Owner, TTL, session ID, compute type and compute metadata changes | `AccessDeniedException` | +| Aliased/nested start/reservation updates and removals | `AccessDeniedException` | +| ADD to protected TTL; DELETE from a protected set | `AccessDeniedException` | +| Put, Delete, BatchWrite put/delete and PartiQL insert/update/delete | `AccessDeniedException` | +| Approval Put plus reporting Update transaction | Allowed | +| Forbidden task Update/Put/Delete plus valid approval Put | Entire transaction denied; no approval created | +| Ambient worker task/counter reads and updates | `AccessDeniedException` | +| Scoped task session counter reads and updates | `AccessDeniedException` | + +The entire task record was unchanged afterward. False conditions protected +negative single-item writes where supported. Only disposable rows could be +affected if deletion unexpectedly succeeded. These are authorization checks of +request shapes, not a normal approval workflow on the synthetic terminal rows. + +Verifier `backgroundagent-dev-p3-iam-20260915` version 2 had ZIP SHA-256 +`IzStMi0xeUIUNRQGhvNwrhGy5gs2XFGClArQc4U79f4=`. Version 1 passed a 31-case subset. +All versions, both task rows and the owned approval were removed; cleanup was +verified at 22:20:28.279Z. Existing roles and shared logs were retained. + +## S3: 10 checks and public control passed + +Three harmless owned markers occupied a bootstrap manifest key, task payload key +and private launch key. They contained no credentials or real task instructions. + +| Request | Observed result | +|---|---| +| Ambient worker reads its own bootstrap prefix | Allowed | +| Ambient worker reads payload/private launch or lists payload bucket | `AccessDenied` | +| Ambient worker replaces/deletes manifest | `AccessDenied` | +| Scoped session SDK reads payload, launch or manifest | `AccessDenied` | +| Worker-signed SDK read of a public NOAA object | `AccessDenied` | + +From the same Lambda invocation, an anonymous one-byte request to +`https://noaa-goes16.s3.amazonaws.com/index.html` returned HTTP 206, while the +worker-signed request to the same object was denied. No public resources or +public-access settings were created or changed. + +The deployed worker policy explicitly denies `s3:GetObject*` outside its own +`bootstrap/*` prefix and denies `s3:List*` on the payload bucket. This public +control uses an existing NOAA object, not a crafted public manifest. Earlier +guest probes separately verified private foreign-manifest rejection. + +Version 1 incorrectly required the public error text to name an explicit deny. +AWS provides enhanced reasons mainly for +[same-account/organization requests](https://docs.aws.amazon.com/AmazonS3/latest/userguide/troubleshoot-403-errors.html). +Version 2 pairs same-function anonymous success with signed denial and retains +the deployed policy separately. + +Verifier `backgroundagent-dev-p3-s3-20260915` version 2 had ZIP SHA-256 +`ZN6YKfbf5XqqYiTQ/3BtdyOYSvBl3LuFNHsP3Tfem8g=`. All versions were deleted and +absence verified at 22:31:12.242Z. The three object bodies remained unchanged. + +## Signing credentials expire: passed + +A temporary signer role trusts only the operator's existing IAM principal and +can read exactly the harmless owned payload. Its STS session lasts 900 seconds; +the download URL has a nominal 3,600-second lifetime. Initial HTTP 200 returned +the expected body. + +The credentials expired at 22:41:22Z. At 22:41:27.332Z, the same URL returned +HTTP 400 `ExpiredToken`, with about 45 minutes of nominal URL lifetime remaining. +S3's response date was 22:41:27Z, request ID `30CHHS5HK249BHPF`. Neither the object +nor the role policy changed during the test. A signed link therefore cannot +outlive the temporary credentials that signed it. + +Cleanup was verified at 22:42:55.386Z: all three owned objects were deleted, +the payload prefix was empty, manifest HEAD returned 404, and the temporary +signer role and inline policy were absent. The private mode-0600 URL file was +deleted. No signing keys or URL appear in logs or this record. + +## Evidence and remaining gates + +Private code, policies, per-operation results, hashes and cleanup ledgers: + +- `/tmp/abca-645-p2-clean-20260913/p3-effective-iam-20260915` +- `/tmp/abca-645-p2-clean-20260913/p3-effective-s3-20260915` + +Migration/drain, reconciler scale, other ambient roles, runtime network negatives +and coordinated ECS rollout remain separate gates. No credential values appear +in the result documents. diff --git a/docs/verification/645-p3-callback-timeout.md b/docs/verification/645-p3-callback-timeout.md new file mode 100644 index 000000000..f0502500b --- /dev/null +++ b/docs/verification/645-p3-callback-timeout.md @@ -0,0 +1,80 @@ +# ADR-021 P3: preserve the SDK approval callback + +Date: 2026-09-15. This follows the +[real long-sleep failure](./645-p3-durable-live-20260915.md#expired-credentials-passed-long-approval-semantics-failed). +The change below is local and has not been deployed. + +## Problem + +The approval timer and the coding program's callback timer are separate. +The first controls how long a person may approve a tool. The second controls how +long Claude waits for our Python hook to answer. A suspended VM must preserve +that unanswered callback until its approval can be reconciled. + +The live hour-long case renewed expired AWS credentials successfully, but +Claude abandoned its pending Read callback when the VM woke. It produced a +generic denial while the approval record remained `PENDING`. The task then +completed before the original approval deadline. This is a failed lifecycle +case even though the tool did not run and cleanup succeeded. + +`build_hook_matchers` supplied no explicit `HookMatcher.timeout`. A local probe +with the actual pinned SDK `0.2.110` and Claude `2.1.191` reproduced the same +generic denial and Python `CancelledError` when a one-second callback budget +expired during a three-second wait. A ten-second budget allowed that wait and +one successful read. The SDK's documented 60-second default is not reliable for +this binary: an unset timeout allowed a 65-second wait. The longer default probe +then cancelled at 600.034 seconds and returned the same generic denial. The +actual default for this pinned binary is ten minutes. + +## Local change + +PreToolUse matchers now receive an explicit timeout: + +| Worker | Callback budget | +|---|---| +| ECS / AgentCore | Maximum approval window, 3,600 seconds, plus 120 seconds | +| Registered MicroVM | Maximum VM lifetime, 28,800 seconds, plus 120 seconds | + +The larger MicroVM budget covers a supervisor that recovers after the approval +deadline. The existing approval loop still uses its original deadline and denies +an expired gate. PostToolUse and Stop callback settings are unchanged. + +The existing eight-hour `RunMicrovm` limit now comes from +`contracts/constants.json`, shared with the Python callback calculation. The +service duration is unchanged. The drift checker rejects invalid durations and +a new hardcoded strategy copy. An outdated exception comment was also corrected: +Python cancellation propagates through `finally`; it is not caught by +`except Exception`. + +## Reproduction and validation + +The opt-in [probe](../../agent/scripts/verify_approval_hook_timeout.py) runs the +actual pinned coding program against a loopback fake Bedrock stream, synthetic +AWS keys and one disposable marker file. It makes no paid model call or AWS +request. It takes production matcher settings and replaces only the callback +with a controlled wait. + +```bash +agent/.venv/bin/python agent/scripts/verify_approval_hook_timeout.py --timeout 1 --delay 3 +agent/.venv/bin/python agent/scripts/verify_approval_hook_timeout.py --configured microvm --delay 650 +agent/.venv/bin/python agent/scripts/verify_approval_hook_timeout.py --configured standard --delay 650 +``` + +The full root build passed in 364.12 seconds: 4,993 CDK tests passed with 56 +optional DynamoDB Local tests skipped; 1,942 Python tests passed with 11 skipped; +all 928 CLI tests passed. Compilation, lint, drift checks, synthesis and docs +build also passed. The focused CDK run passed all 201 assertions, though its +partial coverage could not satisfy the global full-suite threshold; the later +full build passed that threshold. + +Both production-configured comparisons passed a 650-second wait: MicroVM +callback budget 28,920 seconds, standard budget 3,720 seconds. Each produced +one successful marker read and one post-tool callback. The unset default +cancelled at 600.034 seconds; the explicit one-second failure control cancelled +without reading the marker. A final short configured probe also verified that +diagnostics go to stderr and stdout remains valid JSON. + +This probe does not simulate a real VM snapshot. AWS acceptance still requires +an updated image and a fresh long-sleep run with the original approval deadline, +the expected tool result and coordinator cleanup. The separate service-reported +resume-hook connection refusal remains unresolved. diff --git a/docs/verification/645-p3-durable-live-20260915.md b/docs/verification/645-p3-durable-live-20260915.md new file mode 100644 index 000000000..b51be4708 --- /dev/null +++ b/docs/verification/645-p3-durable-live-20260915.md @@ -0,0 +1,295 @@ +# ADR-021 P3: real AWS durable supervision + +Date: 2026-09-15. This continues the +[deployment and isolated guest checks](./645-p3-live-deployment-20260915.md). +Production automatic suspension remains disabled. The full +[acceptance matrix](./645-p3-implementation-plan.md#7-acceptance-matrix-and-completion-gates) +is not complete. + +## Scope and isolation + +Temporary Lambda `backgroundagent-dev-p3-durable-20260915` runs the compiled +production durable handler, supervisor, hydration, strategy, admission and +finalization code from `42c0bde9`, using development image `3.0` in `us-west-2`. +Its wrapper permits fixed synthetic task IDs and owners, selects a repository-free +MicroVM blueprint, uses five-second polling, and optionally crashes immediately +after a saved worker-start receipt. Each task allows six turns and a one-dollar +model budget. Its requested tool action is `Read("/etc/os-release")`; the long +credential-expiry case first runs a foreground `sleep 180` to age its credentials. + +The fixture has its own role and suspension parameter. DynamoDB permissions name +the fixture task/owner keys; payload writes name their object paths. It has no +public endpoint or event trigger. Optional global Memory is disabled. The real +deployed approve, deny and cancel handlers receive synthetic owned API events. +No repository, PR, issue or notification destination is configured. + +This proves real AWS durable execution and worker lifecycle behavior under those +fixture settings. It does not prove API Gateway authorization, normal repository +configuration routing, Memory, publication, or the entire effective-policy matrix. +The temporary role is not the production coordinator role. + +AWS sends a durable-execution envelope to Lambda. The production SDK unwraps it; +the fixture checks its task allowlist inside `loadTask`, before a database read. +It must not treat the outer envelope as the ordinary `{task_id}` input. + +| Fixture version | Purpose | ZIP SHA-256, base64 | +|---|---|---| +| 1 | Unused preparation version; no executions | `yQgkDHEjsqt+xS8whUn4g6dHqYxcF5xirSHPo78aybI=` | +| 2 | Preserve the production SDK envelope | `eh9wh2+/wnoURXkFGa46rlH9FGGKRC8UF9xetYnZGIE=` | +| 3 | Add command timing and AWS request IDs | `WBoCDV9SXZ8dP4/9FgVkD6bCyqJhAdmDtdq/komlzik=` | +| 4 | Add fresh approval retry ID; correct diagnostic worker-ID field | `fPxIeB6wRlyv2xaPDUdqWU6HBDw7s1iy2e+B4GO6cLc=` | +| 5 | Add credential-expiry fixture; 90-minute execution bound | `LYsZfWmwZuGUiu87kr+AnGd4sNbpS8bwqxlcwkquKZQ=` | +| 6 | Add fixed approval-fault and late-deadline identities | `NTxEsRcA2QK7TYa5tQGJcfkZdcMj6O7hXKIsZTsbfj4=` | + +Versions 3–4 change only private fixture diagnostics/allowlisting, not production +lifecycle behavior. Diagnostics omit credentials and launch bodies. + +## Acceptance evidence + +A case passes only after the coordinator itself has produced the expected +terminal task, terminated its worker, deleted its task payload prefix, and +released its saved capacity reservation with the owner's counter at zero. +Independent fixture cleanup runs afterward or after failure; it cannot make a +failed coordinator check pass. + +| Case | Task ID | Result | +|---|---|---| +| Ordinary, no approval | `01M2KENV5CKDZVM1HAT70AKRV9` | Passed on version 2 | +| Automatic sleep, approve | `01M2KENV5EMWVJ7FT312YWXJAD` | Failed during wake; diagnosis open | +| Automatic sleep, deny | `01M2KENV5EMWVJ7FT312YWXJAE` | Passed on version 3 | +| Cancel automatically suspended worker | `01M2KENV5EMWVJ7FT312YWXJAF` | Passed on version 3 | +| No decision; original deadline | `01M2KENV5EMWVJ7FT312YWXJAG` | Passed on version 4 | +| Crash after saved start, then approve | `01M2KENV5EMWVJ7FT312YWXJAH` | Start recovery passed; later wake failed | +| Cancel during start-crash recovery | `01M2KENV5EMWVJ7FT312YWXJAJ` | Passed on version 4 | +| Worker dies while suspended | `01M2KENV5EMWVJ7FT312YWXJAK` | Expected failure and cleanup passed on version 4 | +| Disable suspension during grace | `01M2KENV5EMWVJ7FT312YWXJAM` | Passed on version 4 | +| Fresh approval retry | `01M2KGTJFZDTT7WJVR70B2GMEN` | Passed on version 4 | +| Credentials expire while frozen | `01M2KHNBGNGV55MZY0M4F90C1A` | Credential renewal passed; approval callback was lost | +| Immediate state read fails after approval | `01M2KKXYPDQXFA1RB2M1H9D7MA` | Passed on version 6 | +| Immediate Resume submission fails after approval | `01M2KKXYPD2DQTZS4MMYQ6PCNH` | Approval survived; supervisor wake failed | +| State read and audit writes fail after approval | `01M2KKXYPDEWYWBE76XMZRG708` | Passed on version 6 | +| Supervisor unavailable beyond approval deadline | `01M2KM4RC576WFTA9YW6HT65H0` | Passed in isolated late-deadline function | + +### Ordinary durable execution + +Worker `microvm-2e68dad5-0553-3c05-94ff-94b2a9451481` ran one successful Read. +Execution `dd5da645-4cb6-35a6-a146-b075db6db4c6` ran from +21:27:36.828Z to 21:28:12.055Z. Its history contains seven distinct +`InvocationCompleted` request IDs and seven poll callbacks. Load, admission, +preflight, hydration, start and finalization each started once. + +All seven saved supervisor states retain `firstObservedAtMs=1789507660909` +and `sessionDeadlineMs=1789536460730`, with verified service lifetime. Replay +did not reset either clock. The slot was acquired at 21:27:40.081Z and released at +21:28:11.836Z. Worker termination, payload absence and counter zero were verified +before private cleanup. + +### Approval wake failure: unresolved + +Worker `microvm-c1d17307-e488-3149-b2c9-5d7f25f6278e` created its approval at +21:31:10Z with a 600-second timeout. The guest wrote its suspend checkpoint at +21:31:41.230Z and returned HTTP 200 from `/suspend`. AWS was observed +`SUSPENDED` at 21:31:44.124Z. Approval committed at 21:31:44.844Z and the deployed +API returned 202 at 21:31:46.234Z. + +AWS terminated the worker at 21:31:46.958Z with: + +> Resume lifecycle hook connection was refused. Please check your hook endpoint +> and application logs for more details. + +The guest stream contains no subsequent `/resume` access log or tool result. +The supervisor recorded `MICROVM_SUBSTRATE_TERMINATED`, left the task `FAILED`, +and released its slot. The durable execution itself reported `SUCCEEDED` +because the handler completed failure reconciliation; that does **not** make +the coding task or this acceptance case successful. + +The start-crash case below encountered the same refusal. Its telemetry proves +that overlapping API/supervisor Resume requests are not required for this failure. +The root cause remains unknown. No product workaround has been applied. A fresh +approval retry subsequently passed without a product change; that does not erase +the failed wakes. The inline Resume-failure case below reproduced the refusal +with only the supervisor submitting an actual Resume request. + +### Denial after automatic sleep + +Worker `microvm-bafa9cab-58e0-38da-b36f-b75ee699df59` received one automatic +Suspend request at 21:49:15.533Z. AWS was observed `SUSPENDED` at +21:49:18.625Z; denial committed at 21:49:19.417Z. The supervisor also sent Resume +at 21:49:20.758Z while the decision API was finishing its wake request. +The worker was observed `RUNNING` at 21:49:26.113Z. + +TaskEvents contain one Read attempt and one authoritative-denial result, with no +successful tool execution. Original creation time 21:48:45Z and the 600-second +timeout are unchanged. At 21:49:36.664Z the task was `COMPLETED`, the durable +execution `SUCCEEDED`, the VM `TERMINATED`, and the slot released. Payload absence +and counter zero were verified before private cleanup. + +### Cancellation after automatic sleep + +Worker `microvm-263ecb51-d404-3862-96e2-4e5d2b3698a1` was observed +`SUSPENDED` at 21:51:12.152Z. The repaired deployed cancel handler returned 200 +at 21:51:13.907Z. At 21:51:19.174Z the task remained `CANCELLED`, its execution +had succeeded, its worker was terminated, and its slot released. Counter zero +and payload absence were verified before private cleanup. + +The original approval remains `PENDING`, with unchanged creation time and timeout. +There is one gated tool attempt and no tool result. API and coordinator each +record a cancellation event; the saved capacity reservation changes to released +once, at 21:51:17.273Z. + +### Original deadline and credential renewal + +Worker `microvm-be8e4e0a-85b1-3cb2-96f6-f3a749d56113` created its 180-second +approval at 21:52:22Z. It was observed suspended at 21:52:53.939Z and running at +21:54:25.043Z, about 57 seconds before the original deadline. There was one +supervisor Resume request at 21:54:22.525Z and no decision API call. + +The guest recorded timeout at 21:55:22.126Z and one Read timeout result at +21:55:22.169Z. Completion and cleanup passed by 21:55:34.891Z. All 42 durable +invocations retained the same supervisor clocks and original approval deadline. + +CloudTrail records initial tagged `AssumeRole` at 21:52:09Z, expiring at +22:52:09Z, and renewal during resume at 21:54:23Z, expiring at 22:54:23Z. +Role, user tag and task tag were unchanged. Only timestamps, tags and request IDs +were retained. This proves renewal before expiry; the longer case tests expired keys. + +### Real process crash and cancellation race + +The first crash fixture saved worker +`microvm-9e6bdb3f-bc10-3921-9c50-287bbe20b574`, then called `process.exit(42)`. +AWS recorded the failed invocation at 21:56:05.731Z, started another at +21:56:06.763Z, and completed the saved start step at 21:56:08.983Z. The same worker +continued. Later its `/suspend` returned 200 at 21:56:48.531Z; approval returned +202 at 21:56:55.364Z. AWS terminated it at 21:56:56.279Z with the same hook +connection refusal. Telemetry shows one Suspend and no supervisor Resume before +termination. Overall acceptance failed; start recovery and failure cleanup worked. + +The cancellation race used worker +`microvm-6cba9cf9-ebf2-3d95-9c8a-c3d628454c58`. Its invocation exited at +21:58:32.834Z. The watcher saw `HYDRATING` and the saved worker at 21:58:33.199Z; +cancel returned 200 at 21:58:34.875Z. The replayed start step returned `null` at +21:58:36.182Z and `finalize-before-session` finished at 21:58:36.424Z. +Cancellation, termination, payload absence and counter zero passed. There were +two real invocations, one admission and no tool result. + +### Fresh approval, worker death and live rollback + +Fresh approval worker `microvm-94aa4204-435e-36f5-9ebf-ee6686f18c20` was observed +suspended at 22:00:37.831Z. Approval returned 202 at 22:00:38.750Z and one Read +succeeded at 22:00:39.437Z. Completion and cleanup passed by 22:00:49.713Z. +Eleven invocations retained the original 22:00:07Z/600-second approval clock. + +The worker-death fixture terminated +`microvm-0b21daf9-f196-3e6f-a318-1991663a10b2` at 22:02:21.578Z after automatic +suspension. The coordinator marked the task `FAILED`, released its reservation, +deleted the payload and completed its durable execution. Acceptance passed by +22:02:32.693Z; no tool result was produced. + +The rollback fixture disabled its own live switch at 22:04:16.566Z, during grace +for an approval created at 22:04:11Z. Worker +`microvm-909268eb-2ea0-321f-853d-782322215fdc` stayed running for over 60 seconds +afterward. Approval and cleanup passed by 22:05:32.733Z. The existing durable +execution honored the switch despite its immutable static flag still being true. + +### Approval survives immediate wake and audit faults + +Temporary `backgroundagent-dev-p3-approve-fault-20260915:1` wrapped the compiled +production approval handler with fixed task/owner checks and injected SDK +failures. It used the unchanged deployed approval role and configuration. +Its ZIP SHA-256 was `WG4JXZvqDmQlpwHss4W/tcIzGm6EY0esThypYcxRvwY=`. + +The state-read case injected `GetMicrovm` `TimeoutError` at 22:51:55.332Z. +The handler persisted a `microvm_resume_orphan` warning event and returned 202. +The durable supervisor woke worker +`microvm-8f4f231a-e997-3aee-9d8c-2b8e5193722e`; one Read succeeded, and completion +and cleanup passed by 22:52:06.537Z. + +The Resume-submission case injected `ResumeMicrovm` `TimeoutError` at +22:54:42.026Z. The approval stayed committed and the API returned 202. The +supervisor submitted the real wake request, but AWS terminated worker +`microvm-dca019ee-7d19-389e-b90b-bde129f77329` at 22:54:45.753Z with the same +resume-hook connection refusal. The task became `FAILED`, the durable execution +`SUCCEEDED`, and the reservation was released. Overall acceptance failed; the +runner then performed independent cleanup. This case cannot count as a +successful wake-recovery test. + +The audit-failure case injected both the state-read failure and all TaskEvents +writes from the decision handler. Logs confirm failures of both +`approval_decision_recorded` and the orphan warning write. The API still returned +202 at 23:02:05.851Z. Worker +`microvm-0f74d6a8-057e-36b5-9c8b-e3d8ab2fd079` performed one successful Read at +23:02:06.757Z. Thirteen durable invocations retained the original clocks; +completion and coordinator cleanup passed without independent repair. + +All fault-handler versions and its own log group were removed after log +collection; function absence was verified at 23:03:09Z. Its production role +was left unchanged. + +### Supervisor outage beyond the original deadline + +`backgroundagent-dev-p3-late-20260915:1` used the exact version-6 bundle and the +temporary fixture role, with a separate function concurrency limit. The main +supervisor and concurrent long-expiry test were not paused. + +Worker `microvm-f1594855-b72c-3952-8670-57700e870ba5` created its approval at +23:02:48Z with a 180-second deadline, 23:05:48Z. After observing it suspended, +the operator set only the isolated function's reserved concurrency to zero at +23:03:21.847Z. The worker remained suspended. The limit was removed at +23:06:07.845Z, after the original deadline. + +The supervisor submitted Resume at 23:06:27.476Z, request ID +`e070a904-9720-4e79-8937-7098293048f0`. The guest returned `/resume` HTTP 200 at +23:06:28.010Z and the Read timeout at 23:06:28.146Z. It did not execute the Read +or create a new approval window. Twenty-one recorded invocations and thirteen +poll callbacks retained the original supervisor clocks. Coordinator cleanup +passed at 23:06:40.234Z with no independent repair. + +### Expired credentials passed; long approval semantics failed + +The shared approval maximum is 3,600 seconds. This fixture first runs harmless +foreground `sleep 180`, then requests the gated Read. It changes no production +limits and never asks the agent to read credentials. + +CloudTrail records original task credentials at 22:07:17Z, expiring at +23:07:17Z. Approval was created at 22:10:32Z with a 3,600-second timeout; +suspension was observed at 22:11:08.410Z. An independent check at 23:08:55.158Z +still found the VM suspended after those credentials expired. + +The worker woke at 23:09:33Z. CloudTrail records a new `AssumeRole` at +23:09:33Z, 136 seconds after the original expiration, with the same role and +task/user tags. New expiration is 00:09:33Z on September 16; request ID is +`f9b97857-4c4f-4f7f-af94-21cc760aa147`. Subsequent model completion, task writes +and trace upload worked. This proves actual scoped credential renewal after +expiry. It does not cover Gateway signing or other backends. + +The approval behavior failed independently. The guest logged a CLI-generated +generic denial at 23:09:33.528Z, before `/resume` returned 200 at 23:09:33.908Z. +The model reported that the read was not permitted and finished. The original +approval remained `PENDING`, with no Read-result TaskEvent, although its deadline +was still 23:10:32Z. The runner's terminal/cleanup checks alone passed; the +independent semantic audit correctly rejected overall acceptance. + +All 733 durable invocations retained the same supervisor clocks. Coordinator +cleanup finished by 23:09:45.465Z, with counter zero, worker terminated and no task +payload. The [callback timeout follow-up](./645-p3-callback-timeout.md) investigates +the lost wait and records the local fix and remaining deployment check. + +## Evidence and outstanding work + +Private evidence is under +`/tmp/abca-645-p2-clean-20260913/p3-durable-fixture-20260915`. +Each case has observations, durable history and TaskEvents. Full execution-data +history is retained privately when needed to inspect persisted clocks. +`deployment-ledger.json` records the exact temporary resources and versions. + +Eleven complete durable cases passed. Three cases failed during approval wake, +and the long-sleep case passed credential renewal but failed approval semantics. +The intermittent refusal and the callback-timeout fix's AWS validation remain +open, along with the remaining lifecycle faults, networking and migration. +See the [effective IAM checks](./645-effective-iam-20260915.md). + +All temporary durable/fault/late-deadline Lambda functions and versions, owned +log groups, the durable fixture role and its suspension parameter were removed. +Main cleanup completed at 23:16:39Z after verifying all 14 main executions were +terminal and all 23 workers listed for the development image were terminated. +Completed fixture task records and traces remain audit data. diff --git a/docs/verification/645-p3-implementation-plan.md b/docs/verification/645-p3-implementation-plan.md index 52f189536..1632edbb3 100644 --- a/docs/verification/645-p3-implementation-plan.md +++ b/docs/verification/645-p3-implementation-plan.md @@ -34,6 +34,24 @@ Initial isolated guest cases pass. These use a local production supervisor and manual guarded Suspend; deployed durable entrypoint, automatic suspension admission and the rest of the acceptance matrix remain open. +**AWS durable follow-up (2026-09-15):** a +[temporary isolated durable supervisor](./645-p3-durable-live-20260915.md) +now exercises the production handler and automatic lifecycle policy in AWS. +Eleven complete cases pass, including original-deadline timeout, cancellation +during real process-crash recovery, worker death, live-switch rollback and +wake after an actual supervisor outage past the approval deadline. +Three approval wakes failed with a service-reported connection-refused error; +a fresh retry passed, but the cause remains open. The long case renewed actual +expired credentials, but its pending approval callback was abandoned before the +original deadline, so overall acceptance failed. Temporary AWS fixtures have been +removed. The [callback-timeout follow-up](./645-p3-callback-timeout.md) records the +reproduction and local correction; its updated image still needs AWS validation. +The [effective IAM follow-up](./645-effective-iam-20260915.md) adds +37 actual AWS metadata checks, 10 S3 checks, and failure after real signer +credential expiry. Public-object checks pair anonymous success with worker-signed +denial. These results supersede corresponding untested items +above without completing the full acceptance matrix. + **Live infrastructure and image deployed (2026-09-14):** the [clean P2 deployment record](./645-p2-clean-deployment-20260913.md) tracks the new Oregon environment, four deployment fixes and actual verification results. @@ -79,16 +97,18 @@ is superseded by these records. - [x] Require new ARN fields to participate in validation; pin contract fields and anchor (#817). - [x] Bind configuration to IAM-authenticated deployment manifests and use single-object payload links for ECS/MicroVM (#817 / #700). - [x] Verify MicroVM manifest/download transport, malformed or mismatched inputs, URL expiry/revocation, a foreign private-bucket denial and >1 MiB transport in AWS; verify concurrent/repeated/conflicting S3 preparation with operator credentials. -- [ ] Complete v2 effective-role/public-bucket tests, expired signer credentials, coordinator recovery and the ECS/coordinated-rollout matrix in AWS. +- [x] Verify deployed MicroVM-role metadata/S3 permissions, public-object denial and actual signer-credential expiry in AWS; see the [effective IAM evidence](./645-effective-iam-20260915.md) for scope. +- [ ] Complete other-backend roles, runtime network paths and the ECS/coordinated-rollout matrix in AWS. - [x] Implement saved MicroVM start receipts, stable tokens, input fingerprints and handle recovery. - [x] Verify immediate identical `RunMicrovm` replay returns the same worker ID in the live payload probes. - [x] Verify simultaneous identical Run calls, changed-parameter rejection, and replay after termination through roughly five minutes against AWS; distinguish cached Run responses from fresh VM state. - [x] Verify production start/receipt/payload code against AWS with lost replies and local process death, saved-handle recovery, changed-input/cancellation refusal and actual receipt expiry. -- [ ] Verify deployed durable-Lambda checkpoint/registration recovery and races, AWS behavior after token retention expires, and operator cleanup of genuinely unknown worker IDs. +- [x] Verify real AWS durable replay after a saved worker receipt and process exit, including cancellation during recovery, in the isolated production-handler fixture. +- [ ] Complete remaining durable registration races, AWS behavior after token retention expires, and operator cleanup of genuinely unknown worker IDs. - [x] Make capacity acquisition/release atomic per task across crash replay; unify counter writers and repair. - [ ] Verify the capacity protocol's upgrade/drain procedure, deployed IAM and scan scale in AWS. - [x] Restrict agent task updates to reporting fields; remove replacement/deletion and worker counter grants. -- [ ] Verify metadata restrictions with real AWS sessions/transactions; retain status/tag trust limits. +- [x] Verify metadata restrictions with real AWS task-tagged sessions and mixed transactions under the deployed MicroVM role; retain status/tag trust limits and the separate other-role gate. - [x] Replace unused logging-failure bookkeeping with structured stdout diagnostics (#810); document shared runtime networking and verify large registry payload delivery (#818). - [x] Deploy a fresh bootstrap, application and managed image from current source; verify build hooks and API reads. - [x] Verify normal coding, PR iteration, Memory writes, live logging and successful/canceled-task cleanup in AWS; observe cancellation preserve another task's capacity. @@ -108,6 +128,8 @@ is superseded by these records. - [x] Complete full repository validation with the P3 supervisor and live switch. - [x] Deploy the supervisor and six-hook image with suspension disabled; verify six isolated guest cases, repair the discovered cancellation stop omission, and prove API termination before test cleanup. - [ ] Deploy and verify the complete P3 sleep/wake lifecycle in AWS. +- [ ] Deploy the explicit SDK callback-timeout fix and repeat long sleep, late wake, approval, denial and cancellation; require final approval/tool evidence as well as cleanup. +- [ ] Resolve the intermittent resume-hook connection refusal; successful retries do not discharge the three failed wakes. First prerequisite batch completed locally on 2026-09-13: @@ -428,7 +450,9 @@ controller and `/run` reseed `random` using fresh OS entropy. The [guest barrier review](./645-p3-guest-barrier.md) records that controller milestone. The subsequent [HTTP hook implementation](./645-p3-lifecycle-hooks.md) now supplies production checkpoint and refresh/reconciliation callbacks. No image -capability, new IAM grants or automatic sleep have been enabled. +capability, new IAM grants or automatic sleep were enabled at that guest-barrier +milestone. The later image, supervisor and live-verification milestones above +record their implementation and deployment. **Credential implementation added locally (2026-09-14):** the [credential verification](./645-p3-credentials.md) records actual pinned-CLI diff --git a/docs/verification/645-p3-live-deployment-20260915.md b/docs/verification/645-p3-live-deployment-20260915.md index d42308a8e..720b49c44 100644 --- a/docs/verification/645-p3-live-deployment-20260915.md +++ b/docs/verification/645-p3-live-deployment-20260915.md @@ -237,8 +237,13 @@ Private evidence directory: `/tmp/abca-645-p2-clean-20260913/`. - `p3-cancel-deployment-verified-20260915.json`: deployed code checksum. - `p3-guest-payload-cleanup-verified-20260915.json`: empty task-owned S3 prefixes. -Remaining acceptance includes deployed durable -replay/recovery, live suspension admission/rollback, credentials expiring while -frozen, lifecycle fault/race injection, and the residual P2 effective-IAM, -network and capacity/migration checks. These remain open in the +The subsequent [AWS durable record](./645-p3-durable-live-20260915.md) adds +automatic suspension, timeout, rollback, crash/cancellation recovery and cleanup +evidence, plus an unresolved intermittent approval-wake failure. +The [effective IAM record](./645-effective-iam-20260915.md) adds actual metadata +and S3 permission checks plus real signer-credential expiry. The long durable +case also proves scoped credential renewal after expiry, but exposes an +[approval callback timeout](./645-p3-callback-timeout.md). Its updated-image +verification, remaining lifecycle faults/deadline races, network and +capacity/migration checks remain tracked in the [implementation plan](./645-p3-implementation-plan.md). diff --git a/docs/verification/645-p3-readiness-review.md b/docs/verification/645-p3-readiness-review.md index a6e1f005a..0caa01d83 100644 --- a/docs/verification/645-p3-readiness-review.md +++ b/docs/verification/645-p3-readiness-review.md @@ -26,6 +26,16 @@ The [image capability milestone](./645-p3-image-capability.md) (2026-09-15) decl The reservation review found a separate trust boundary: the old agent role could write/replace/delete its task row, including coordinator metadata. A subsequent local fix restricts main-task writes to reporting/approval attributes, removes replacement/deletion permissions and removes unused worker counter grants. It also removes unused Python submission/session-info helpers and corrects overstated tenant-isolation comments. [Metadata verification](./645-coordinator-metadata.md) records the tests and pending AWS gate. Status reports still come from the agent, and the compute role chooses session tags; this is not complete hostile-worker isolation. +The [AWS durable follow-up](./645-p3-durable-live-20260915.md) records eleven +passing cases, including real process-crash/cancellation recovery, automatic +deadline wake, supervisor outage and coordinator-owned cleanup. Three approval +wakes failed with a service-reported connection-refused error. A long frozen +worker renewed expired task credentials but lost its approval callback; the +[local callback fix](./645-p3-callback-timeout.md) still needs an updated image and +fresh AWS acceptance. The [effective permissions record](./645-effective-iam-20260915.md) +adds 37 metadata checks, 10 S3 checks and real signer-credential expiry. +Production automatic suspension remains disabled. + ## Start here: the pieces in plain language A **MicroVM** is a small, isolated computer rented from AWS. **Firecracker** is the technology that keeps these small computers separate. A **backend** is the kind of rented computer ABCA chooses to run a coding task. diff --git a/scripts/check-constants-sync.ts b/scripts/check-constants-sync.ts index bd91caaeb..016051c6f 100644 --- a/scripts/check-constants-sync.ts +++ b/scripts/check-constants-sync.ts @@ -57,7 +57,8 @@ const PYTHON_CONSUMERS = [ ]; const MICROVM_COMPUTE_TS = path.join(REPO_ROOT, 'cdk/src/constructs/lambda-microvm-compute.ts'); const MICROVM_IMAGE_CAPABILITY_TS = path.join(REPO_ROOT, 'cdk/src/handlers/shared/microvm-image-capability.ts'); -const TS_CONSUMERS = [MICROVM_COMPUTE_TS, MICROVM_IMAGE_CAPABILITY_TS]; +const MICROVM_STRATEGY_TS = path.join(REPO_ROOT, 'cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts'); +const TS_CONSUMERS = [MICROVM_COMPUTE_TS, MICROVM_IMAGE_CAPABILITY_TS, MICROVM_STRATEGY_TS]; /** Env var names must be UPPER_SNAKE — they are installed into a process env. */ const ENV_NAME_PATTERN = /^[A-Z][A-Z0-9_]*$/; @@ -145,6 +146,10 @@ const OWNED_PYTHON_PATTERNS: ReadonlyArray<{ name: string; regex: RegExp }> = [ * the same way the Python ones are. */ const OWNED_TS_PATTERNS: ReadonlyArray<{ name: string; regex: RegExp }> = [ + { + name: 'MICROVM_MAX_DURATION_SECONDS', + regex: /^\s*(?:export\s+)?const\s+MICROVM_MAX_DURATION_SECONDS\s*(?::\s*number)?\s*=\s*-?\d[\d_]*\b/m, + }, { name: 'LIFECYCLE_HOOK_TIMEOUT_SECONDS', regex: /^\s*(?:export\s+)?const\s+LIFECYCLE_HOOK_TIMEOUT_SECONDS\s*(?::\s*number)?\s*=\s*-?\d+\b/m, @@ -217,7 +222,12 @@ function main(): number { lifecycle_hook_timeout_seconds: number; lifecycle_handler_budget_seconds: number; }; - microvm_lifecycle?: { protocol_version: number; image_protocol_env: string; hook_port: number }; + microvm_lifecycle?: { + protocol_version: number; + image_protocol_env: string; + hook_port: number; + maximum_duration_seconds: number; + }; payload_bootstrap?: { version: number; manifest_prefix: string; @@ -379,8 +389,10 @@ function main(): number { const lifecycle = json.microvm_lifecycle; if (!lifecycle || !Number.isSafeInteger(lifecycle.protocol_version) || lifecycle.protocol_version <= 0 || !Number.isInteger(lifecycle.hook_port) || lifecycle.hook_port < 1 || lifecycle.hook_port > 65535 + || !Number.isInteger(lifecycle.maximum_duration_seconds) + || lifecycle.maximum_duration_seconds <= 0 || lifecycle.maximum_duration_seconds > 28_800 || typeof lifecycle.image_protocol_env !== 'string' || !/^ABCA_MICROVM_[A-Z0-9_]+$/.test(lifecycle.image_protocol_env)) { - invariantErrors.push('microvm_lifecycle requires a positive protocol version, valid hook port and ABCA_MICROVM_ marker name'); + invariantErrors.push('microvm_lifecycle requires a positive protocol version, valid hook port, duration within 1–28800 seconds and ABCA_MICROVM_ marker name'); } const BUDGET_FIELDS = [ 'ready_hook_timeout_seconds', From 2b61a4b00d1cbf41429e455003618457845d9fbd Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Tue, 15 Sep 2026 21:01:44 -0400 Subject: [PATCH 043/149] docs: record P3 callback acceptance and wake failures (#645) --- cdk/src/constructs/lambda-microvm-compute.ts | 43 ++-- .../strategies/lambda-microvm-strategy.ts | 19 +- .../645-p3-callback-live-20260915.md | 226 ++++++++++++++++++ docs/verification/645-p3-callback-timeout.md | 13 +- docs/verification/645-p3-credentials.md | 15 ++ .../645-p3-implementation-plan.md | 47 +++- .../645-p3-live-deployment-20260915.md | 8 +- docs/verification/645-p3-readiness-review.md | 8 +- .../645-p3-resume-refusal-investigation.md | 110 +++++++++ 9 files changed, 439 insertions(+), 50 deletions(-) create mode 100644 docs/verification/645-p3-callback-live-20260915.md create mode 100644 docs/verification/645-p3-resume-refusal-investigation.md diff --git a/cdk/src/constructs/lambda-microvm-compute.ts b/cdk/src/constructs/lambda-microvm-compute.ts index a477d7c2c..89cc049f0 100644 --- a/cdk/src/constructs/lambda-microvm-compute.ts +++ b/cdk/src/constructs/lambda-microvm-compute.ts @@ -744,7 +744,9 @@ export interface LambdaMicrovmComputeProps extends LambdaMicrovmImageInputs { * grant either) and every table the agent touches is `task_id`-partitioned * SessionRole territory. It also has no UserConcurrencyTable grant — that counter * is orchestrator/reconciler-owned and the agent path never writes it. - * `lambda:SuspendMicrovm` / `lambda:ResumeMicrovm` are absent (P3), and + * The coordinator receives scoped `lambda:SuspendMicrovm` / `lambda:ResumeMicrovm` + * grants, and decision handlers also receive scoped Resume permission. Neither + * action is granted to this construct's build or execution role. * `lambda:CreateMicrovmAuthToken` is granted to no role in any phase — no JWE * consumer exists (sub-decision 3). */ @@ -1512,35 +1514,18 @@ export class LambdaMicrovmCompute extends Construct { * failure mode we have actually hit (ADR-021 P1 4.3: the 443-only SG made the * image unbuildable, and the root cause was only readable from this group). * So the build role keeps it. - * - The EXECUTION role does not get it. Across all three live runs (P1, P2 run - * 1, P2 run 2) the only group ever *named* under this prefix in any log or - * inventory was `/aws/lambda-microvms/`, the one CloudFormation - * pre-creates below, and both build-time and guest-runtime lines landed in it - * (P1 runbook line 1833 records that single group being deleted with the - * stack). A create right the runtime never exercises does not belong on the - * role that runs untrusted repo code. + * - The EXECUTION role does not get it. CloudFormation pre-creates + * `/aws/lambda-microvms/`; build and guest-runtime lines use that + * group. The 2026-09-16 verification enumerated this prefix while an image + * 3.0 worker was RUNNING before and after the query. It found only the + * stack-managed image group. Image 4.0 runtime logs also arrived there. + * The live execution role had only CreateLogStream/PutLogEvents grants + * and no attached managed policies. * - * EVIDENCE STRENGTH, stated honestly, because it is a security narrowing: - * the corroborating inventory is an ABSENCE measured AFTER teardown, not a - * during-run enumeration. `645-p2-smoke-runbook.md` **§8.6** ("Billing - * confirmed stopped") records `/aws/lambda-microvms/*` log groups: **none** - * once the stack was deleted, and **§8.8** item 4 (a separate section — the - * deliberately-retained list) names the only service-vended groups created - * outside CloudFormation as `/aws/bedrock-agentcore/runtimes/…` and - * `/aws/lambda/backgroundagent-dev-…`. That combination is load-bearing - * because a service-created group is NOT a CloudFormation resource and so - * would have survived the stack delete and appeared in §8.6 — but it is - * inference from an absence, not a positive observation that no sub-group was - * ever created mid-run. Treat it as strong-but-indirect. - * - * ⚠️ RE-VERIFY on the pending clean re-run (ADR-021 P2 "the row is not yet - * fully closed"), and make it a DURING-RUN enumeration this time — an - * `aws logs describe-log-groups --log-group-name-prefix /aws/lambda-microvms/` - * taken while a task is `RUNNING` is the positive observation the post-teardown - * absence above only implies. If guest logging ever goes silent on this backend, - * this narrowing is the first thing to re-widen — the symptom would be an - * `AccessDeniedException` naming `logs:CreateLogGroup` in the guest's stdout - * fallback, which the MicroVM group still captures. + * See docs/verification/645-p3-callback-live-20260915.md for the inventory and + * policy evidence. This records the tested service behavior, not a guarantee + * about future log-group naming. Investigate a specific CreateLogGroup denial + * before changing runtime permissions. * * @param role - the role to grant. * @param options - `allowCreateLogGroup` gates the build-role-only half. diff --git a/cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts b/cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts index 87fd7bc8c..55e195c8e 100644 --- a/cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts +++ b/cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts @@ -700,19 +700,20 @@ export class LambdaMicrovmComputeStrategy implements ComputeStrategy { * - ``SUSPENDING`` / ``SUSPENDED`` → ``suspended``. SUSPENDING is folded in * because the VM is already on its way to frozen; reporting ``running`` * would tell the orchestrator compute is still progressing when it is not. - * Both map to a state the orchestrator treats as benign-or-anomalous - * depending on the task status, never as a failure. Earlier probes skipped - * this short transition, but P3 must handle it if observed: save wake intent - * and wait for SUSPENDED before issuing ResumeMicrovm. + * The supervisor checks task/gate state to distinguish an expected wait + * from an anomaly. When a wake is needed, it saves wake intent and waits + * for SUSPENDED before issuing ResumeMicrovm. Unresolved transitions have + * a bounded recovery deadline. * - ``TERMINATING`` / ``TERMINATED`` → ``completed``. Both are terminal or - * terminal-bound and carry no exit code, so "the substrate is gone" is all - * the strategy can honestly say; whether that is success or failure is the - * orchestrator's call (it cross-references the DynamoDB status). This is + * terminal-bound and carry no exit code. TERMINATING confirms shutdown is + * underway, not that it has finished. The orchestrator determines task + * success or failure by cross-referencing DynamoDB status. This is * the load-bearing terminal signal: a terminated MicroVM stays observable * as ``TERMINATED`` for at least ~10 minutes (live-measured), so a poller * that waited for NotFound would spin on a finished VM. - * - anything else (an unrecognized future state) → ``running``, so a service - * enum addition can never fail a healthy task. + * - anything else (an unrecognized future state) → ``running`` with explicit + * UNKNOWN state. The supervisor applies bounded recovery instead of + * immediately declaring the task finished. * ``microvmState`` also reports the explicit observed state (or local UNKNOWN / * NOT_FOUND). P3 uses it to distinguish readiness from the coarse status. * diff --git a/docs/verification/645-p3-callback-live-20260915.md b/docs/verification/645-p3-callback-live-20260915.md new file mode 100644 index 000000000..eebc3ebf7 --- /dev/null +++ b/docs/verification/645-p3-callback-live-20260915.md @@ -0,0 +1,226 @@ +# ADR-021 P3: live verification of the approval callback fix + +Verified 2026-09-15–16 UTC. This records the deployment and live retest of +[the explicit callback timeout](./645-p3-callback-timeout.md), source `58ed7bbf`. +All nine core callback cases passed, including real credential expiry during +suspension. The separate intermittent wake failure remains unresolved; this is +not a P3 completion record. + +## Deployment + +The `backgroundagent-dev` stack in `us-west-2` reached `UPDATE_COMPLETE`. +Change set `p3-callback-image-20260915`, executed at 23:39:52.809Z, modified +exactly two resources without replacement: the managed MicroVM image and its +build role's permission to read the new artifact. The image URI and permission +were changed together to the same immutable object: + +```text +0585a80606aaa66e9ce4dbff35a093451488f72d5a732c6cce6036b88fbcd4ed +``` + +Image `backgroundagent-dev-abca-agent`, version `4.0`, became `ACTIVE` with +build state `SUCCESSFUL`. Ready and validate returned HTTP 200. All six hooks +remain on port 8080 with the same protocol and 30-second service budget. +Bootstrap remains `1.8.0`; the root stack still has 475 resources. + +The main coordinator's code, published version `3`, and alias are unchanged. +Both production suspension settings remain off. The ECS and AgentCore runtime +images have not received this source update. + +The deployment used the exact previously uploaded S3 template, with reviewed +artifact substitutions. It did not use the Unicode-damaged `GetTemplate` +response or an unrelated fresh synthesis. + +## Isolated live checks + +A temporary Lambda uses the compiled production durable coordinator, its +production lifecycle policy and real approval/cancel APIs. Its wrapper allows +only fixed, owned, repository-free tasks and pins image `4.0`. Tasks may read +`/etc/os-release`; the credential-expiry task first runs a foreground +`sleep 180`. No repository publication or notification is part of these checks. + +The fixture has its own IAM role, sleep switch and log group. A separate +temporary Lambda isolates the late-deadline supervisor outage from the long +credential-expiry execution. The long execution stays on fixture version `1`; +new boundary cases use published version `2`. + +Acceptance requires final task, approval and tool evidence; stable durable +supervisor clocks; one admission/start/finalization; a terminated worker; empty +payload prefix; and a released slot with counter zero. An empty independent +cleanup record proves the watcher did not repair the result. + +| Case | Current result | +|---|---| +| Ordinary task | Passed on image `4.0` | +| Approval after sleep, first attempt | Approval/read passed; cleanup acceptance invalidated by watcher timing | +| Fresh approval after sleep | Passed | +| Denial after sleep | Passed; one authoritative denial, no successful read | +| Cancellation while asleep | Passed; no tool result, task stayed canceled | +| Original approval deadline | Passed; timed out at the original deadline | +| Short approval window | Passed; 30-second timeout without suspension | +| Approval during grace period | Passed; quick approval without suspension | +| Wake after original deadline | Passed after an isolated supervisor outage | +| Credentials expire while frozen | Passed; fresh keys after expiry and timeout at the original deadline | +| Newly written files and two approval gates | Passed | +| Missing approval record during freeze | First attempt intercepted by connection refusal; fresh attempt returned expected HTTP 409 | + +### First approval attempt: watcher timing error + +Task `01M2KQ4D59154CQ89ZW1VWNXD0`, worker +`microvm-2532572a-6fe0-3a44-b2a7-e6edf3a4a687`, returned HTTP 200 from +`/suspend` at 23:48:19.210Z and `/resume` at 23:48:24.662Z. Approval became +`APPROVED` without changing its original clock, and exactly one Read succeeded +at 23:48:24.909Z. + +At 23:48:29.945Z the task was `COMPLETED`, the durable execution `SUCCEEDED`, +and its slot released, but AWS still reported `TERMINATING`. The private +watcher immediately required `TERMINATED` and entered independent cleanup. +That makes this attempt unsuitable as proof of untouched coordinator cleanup. + +The watcher now observes `TERMINATING` with read-only polling for up to +60 seconds before deciding whether termination failed. It sends no cleanup +request during that wait. Fresh task `01M2KQV00H6CHJKCC2D184DZBR` passed complete +acceptance, including untouched cleanup; the original evidence is retained. + +### Real credential expiry and the original approval deadline + +Task `01M2KQ4D590P0GSK2VGWXB2987`, worker +`microvm-39599381-c0f2-3298-b9e2-6fcb561daa6a`, stayed on the same image `4.0` +worker throughout this test. It completed one foreground `sleep 180`, then +requested exactly one Read of `/etc/os-release`. + +| UTC time | Evidence | +|---|---| +| Sep 15, 23:51:09 | Initial task-scoped STS credentials issued, expiring Sep 16 at 00:51:09 | +| Sep 15, 23:54:24 | Approval created with a 3,600-second window; original deadline 00:54:24 | +| Sep 15, 23:54:55.242 | Guest `/suspend` returned HTTP 200 | +| Sep 16, 00:51:24.348 | Independent read-only observer confirmed `SUSPENDED`, after actual credential expiry | +| Sep 16, 00:53:25 | Fresh credentials issued for the same role and user/task tags, expiring 01:53:25 | +| Sep 16, 00:53:25.372 | Guest `/resume` returned HTTP 200 | +| Sep 16, 00:54:24.136 | Read returned `User timed_out`; the original approval became `TIMED_OUT` | +| Sep 16, 00:54:29 | A further Bedrock response, persisted task events and S3 trace upload succeeded | +| Sep 16, 00:54:31.065 | Coordinator released the task's capacity reservation | + +CloudTrail request `d87e98e4-676f-4640-9ad8-881f7cb79e22` identifies the initial +STS issuance; `bedf0aaa-f2ad-4989-ace5-b421558a188c` identifies renewal after +expiry. The evidence stores timestamps, role/session identity and tags, without +credential values. + +All 744 durable invocations/poll callbacks preserved one first-observed time and +one session deadline. The task finished `COMPLETED`, its durable execution +`SUCCEEDED`, and the worker `TERMINATED` with reason `Success.` Its payload prefix +was empty, counter zero and independent cleanup record empty. A task completing +after correctly reporting a denied/timed-out tool is expected; the Read itself +did not succeed. + +Unlike the previous long test, the approval callback remained alive until the +original deadline. This passes both real credential renewal and callback +semantics; the earlier failed evidence remains in the historical record. + +### Repeated wakes, file persistence and missing approval + +Six further approval cases passed, three each on images `3.0` and `4.0`. +They used the same coordinator bundle, with the image version pinned separately. +Each retained its worker identity and supervisor clocks, ran one approved Read +and completed cleanup without watcher repair. These successful repeats do not +resolve the intermittent refusal. + +Task `01M2KRWNPPRV7CGJQSRFYY3375`, worker +`microvm-09ebf679-e78c-398d-8023-f1a176d5fb51`, wrote two new marker files in an +owned temporary directory. Its first gate was created at 00:21:33Z on Sep 16, +and its second at 00:22:12Z. Both independently suspended and resumed with HTTP +200, retaining distinct approval IDs and their original 600-second windows. +The two successful Reads returned the original marker bytes after their +respective freezes. The full recorded Bash and Read inputs match the intended +commands and paths. + +The pre-freeze SHA-256 values were: + +```text +b81be383132e7a49275fcf91cb22b0f4b015aa7b26387a3f300f7bc6ed395781 +ba01a509d1b49275a6b7c4a936172e507e51debea893b9b4a34e7e02e532e4af +``` + +All 21 durable invocations retained one pair of supervisor clocks. This proves +mutable temporary-filesystem persistence across two approval generations; it +does not replace a full P3 run against a cloned repository. + +The first missing-approval case, task `01M2KRWNPP4BQCJ80SVG6473Z8`, reproduced +the [connection refusal](./645-p3-resume-refusal-investigation.md) on image `4.0`. +The task failed and cleanup passed, but the expected guest response was absent. +Its acceptance record remains failed. + +One fresh attempt, task `01M2KSTSR1V31NTPQ0CKCJJJMN`, worker +`microvm-a093863f-edf0-39f4-9d79-6893ff08042c`, reached the intended failure: +after conditional deletion of its own pending approval, `/resume` returned HTTP +409 at 00:30:46.418Z. The service reported that HTTP status explicitly. No Read +result was produced; the task was `FAILED`, the worker terminated, its slot +released and payload removed without independent cleanup. This is a passing +negative test: the missing gate never allowed the tool to run. + +## Database checks and comment review + +The optional database suites skipped by the full build were run separately +against the documented pinned DynamoDB Local image, using a loopback endpoint +and dummy credentials. All 56 CDK lifecycle/capacity cases and all 47 Python +checkpoint cases passed. The temporary in-memory container was removed and +its absence verified. + +The subsequent source-comment edits passed TypeScript compilation and focused +ESLint. Python type checking also passed, covering the probe's final output +redirection change. + +Source comments now distinguish asynchronous `TERMINATING` from finished +shutdown, describe bounded recovery for unknown/suspending states, and specify +which roles receive lifecycle permissions. + +### Runtime logging inventory + +At 00:13:54.991Z on 2026-09-16, worker +`microvm-82e58ec5-ee2a-399b-aa8b-38a8469ac4a7` on image `3.0` was `RUNNING` +both before and after enumerating `/aws/lambda-microvms/` log groups. The only +group was `/aws/lambda-microvms/backgroundagent-dev-abca-agent`, with 90-day +retention. CloudFormation confirms that +`LambdaMicrovmComputeMicrovmLogGroup46EC26A7` owns that group. Image `4.0` +runtime and build records also arrived there. + +The deployed execution role has one inline policy and no attached managed +policies. Its logging grants allow `CreateLogStream` and `PutLogEvents`, with +no `CreateLogGroup` permission. No permission changes were needed for these +runs. This replaces the old comment's pending during-run inventory check; +it does not guarantee that future service versions will never need another +group. + +## Final cleanup + +Before infrastructure removal, all 18 main-function durable executions were +terminal; the separate late-deadline execution had already finished. All 19 owned +tasks had terminated workers, released reservations, zero counters and empty +task-payload prefixes. Task statuses were 16 `COMPLETED`, one `CANCELLED` and two +`FAILED`. Those counts include the retained failed/mis-measured attempts and must +not be read as 19 passing acceptance cases. + +The main function's six published versions, the separate late-deadline function, +their two log groups, the shared verification-only role and its policies, and +the private suspension switch were removed. Read-only checks at +2026-09-16T00:59:28.099Z confirmed both functions, the role, parameter and log +groups were absent. The in-memory DynamoDB Local container was also removed. + +Full guest streams, 4,028 main-function log events, durable histories, task events, +redacted credential issuance and resource checks are retained in the private +verification archive. Task/event/trace audit records and shared immutable image +artifacts retain their normal lifecycle; they were not deleted as test payloads. +Production suspension settings remain off. + +## Remaining scope + +The three original image `3.0` resume-hook connection refusals and the new image +`4.0` refusal remain unresolved. The +[investigation record](./645-p3-resume-refusal-investigation.md) contains all +four timelines and the next diagnostic steps. The callback fix changes how long +Claude waits for Python; it does not resolve the separate connection failure. +Successful later wakes do not discharge those failures. + +The [implementation plan](./645-p3-implementation-plan.md) retains the wider +P2/P3 acceptance gates. This completed callback retest and cleanup do not close +the separate wake-failure, full-repository, other-backend or migration gates. diff --git a/docs/verification/645-p3-callback-timeout.md b/docs/verification/645-p3-callback-timeout.md index f0502500b..ab0e97ee1 100644 --- a/docs/verification/645-p3-callback-timeout.md +++ b/docs/verification/645-p3-callback-timeout.md @@ -2,7 +2,8 @@ Date: 2026-09-15. This follows the [real long-sleep failure](./645-p3-durable-live-20260915.md#expired-credentials-passed-long-approval-semantics-failed). -The change below is local and has not been deployed. +The change below is deployed in MicroVM image `4.0`; the +[live retest record](./645-p3-callback-live-20260915.md) tracks AWS acceptance. ## Problem @@ -74,7 +75,9 @@ cancelled at 600.034 seconds; the explicit one-second failure control cancelled without reading the marker. A final short configured probe also verified that diagnostics go to stderr and stdout remains valid JSON. -This probe does not simulate a real VM snapshot. AWS acceptance still requires -an updated image and a fresh long-sleep run with the original approval deadline, -the expected tool result and coordinator cleanup. The separate service-reported -resume-hook connection refusal remains unresolved. +The local probe does not simulate a real VM snapshot. The subsequent +[image `4.0` AWS retest](./645-p3-callback-live-20260915.md#real-credential-expiry-and-the-original-approval-deadline) +passed a real freeze beyond credential expiry, renewed the same task's keys and +preserved the callback until its original approval deadline. Approval, denial, +cancellation, short/grace windows and wake after a supervisor outage also passed. +The separate service-reported resume-hook connection refusal remains unresolved. diff --git a/docs/verification/645-p3-credentials.md b/docs/verification/645-p3-credentials.md index 4294cd2ca..81566cfc0 100644 --- a/docs/verification/645-p3-credentials.md +++ b/docs/verification/645-p3-credentials.md @@ -2,6 +2,14 @@ **Follow-up (2026-09-15):** [per-worker image capability](./645-p3-image-capability.md) declares the served hooks and verifies the actual launched image version. The [supervisor milestone](./645-p3-supervisor.md) implements durable recovery and approval wake; the full repository build passes. The record below preserves this milestone's original scope. Live sleep/wake acceptance remains open. +**Live follow-up (2026-09-16):** the +[image `4.0` long-sleep test](./645-p3-callback-live-20260915.md#real-credential-expiry-and-the-original-approval-deadline) +kept the worker frozen beyond real STS expiry, renewed the same role and task/user +tags, and completed a further model response, database writes and S3 upload. +The approval retained its original deadline. Gateway signing and the wider +lifecycle matrix remain separate gates; this supersedes the untested long-expiry +item below without completing P3. + Date: 2026-09-14. Local implementation and actual pinned Claude process probes. No AWS deployment or suspension was performed for this milestone. Deployed image **2.0** was unchanged at that milestone; P3 was not complete. @@ -25,6 +33,13 @@ attribution files or resolving AWS credentials. That leaves resolution to the scoped container provider and keeps keys out of Claude's stale export cache. AgentCore/ECS/local attribution retains its existing helper behavior. +With pinned Claude `2.1.191`, this deliberately empty export logs +`awsCredentialExport did not return valid AWS STS output structure`. The live +long-sleep case showed that message both before and after suspension, followed +by successful model requests through the scoped provider. It is expected for +this provider-selection design; returning cached keys merely to silence the +message would restore the stale-key path the implementation avoids. + `aws_session.py` exports a coherent key/expiry pair under botocore's refresh lock. The production resume callback forces recorded ambient providers to refresh first, then force the original tenant credential object to renew with the original STS diff --git a/docs/verification/645-p3-implementation-plan.md b/docs/verification/645-p3-implementation-plan.md index 1632edbb3..149a02c25 100644 --- a/docs/verification/645-p3-implementation-plan.md +++ b/docs/verification/645-p3-implementation-plan.md @@ -45,7 +45,12 @@ a fresh retry passed, but the cause remains open. The long case renewed actual expired credentials, but its pending approval callback was abandoned before the original deadline, so overall acceptance failed. Temporary AWS fixtures have been removed. The [callback-timeout follow-up](./645-p3-callback-timeout.md) records the -reproduction and local correction; its updated image still needs AWS validation. +reproduction and correction. Image `4.0` is now deployed; its +[fresh AWS acceptance run](./645-p3-callback-live-20260915.md) passed all nine +core callback cases, including renewal after real credential expiry and the +original approval timeout. Mutable files also survived two approval generations. +A fourth [connection refusal](./645-p3-resume-refusal-investigation.md) occurred +on this image and remains unresolved. The [effective IAM follow-up](./645-effective-iam-20260915.md) adds 37 actual AWS metadata checks, 10 S3 checks, and failure after real signer credential expiry. Public-object checks pair anonymous success with worker-signed @@ -128,8 +133,8 @@ is superseded by these records. - [x] Complete full repository validation with the P3 supervisor and live switch. - [x] Deploy the supervisor and six-hook image with suspension disabled; verify six isolated guest cases, repair the discovered cancellation stop omission, and prove API termination before test cleanup. - [ ] Deploy and verify the complete P3 sleep/wake lifecycle in AWS. -- [ ] Deploy the explicit SDK callback-timeout fix and repeat long sleep, late wake, approval, denial and cancellation; require final approval/tool evidence as well as cleanup. -- [ ] Resolve the intermittent resume-hook connection refusal; successful retries do not discharge the three failed wakes. +- [x] Deploy the explicit SDK callback-timeout fix and repeat long sleep, late wake, approval, denial and cancellation; require final approval/tool evidence as well as cleanup. Image `4.0` passed these checks, including real renewal after credential expiry. +- [ ] Resolve the intermittent resume-hook connection refusal; successful retries do not discharge the four failures across images 3.0 and 4.0. See the [investigation and request IDs](./645-p3-resume-refusal-investigation.md). First prerequisite batch completed locally on 2026-09-13: @@ -209,6 +214,42 @@ Second P3 foundation batch implemented locally (2026-09-13): - CDK lint/compilation passed. The broad handler/session-role run passed **158 suites / 3,738 tests**, including **23 lifecycle** and **15 existing capacity** DynamoDB Local tests. Five relevant suites passed **218 overlapping tests** and exited normally. The broad run exited successfully after a delay (about 72 seconds total versus 17.5 seconds reported test execution), with no open-handle trace; its cause is not established. Documentation sync, the **77-page** build and link checks pass. No Python source changed. The temporary local database was removed. - At that foundation milestone, no production caller used this policy/store. No IAM grants, image hooks or automatic suspension were enabled. Durable poll failure/recovery tracking, guest barriers, supervisor/decision-handler wiring and live AWS gates remain unfinished. +## Remaining work in execution order + +The image `4.0` long-sleep acceptance and verification-infrastructure cleanup are +complete. The original approval timed out correctly, real credentials renewed +after expiry, and coordinator cleanup passed without watcher repair. The +[live record](./645-p3-callback-live-20260915.md) records the evidence and verified +absence of all temporary infrastructure. + +The detailed batches below preserve the implementation history. For the current +handoff, use this order: + +1. Resolve the four [resume-hook connection refusals](./645-p3-resume-refusal-investigation.md). + Service-side connection diagnostics and guest/listener health are still + missing. Passing retries, the callback-timeout fix and successful cleanup + do not close this gate. +2. Complete the remaining live race/fault matrix: cancellation during transitions, + late decision races, repeated polling/credential-refresh failures, durable + registration races, service token-retention expiry and recovery of a worker + whose ID was genuinely lost. Keep each injected failure distinct from an + unrelated service failure. +3. Complete effective permissions and network checks for the other backends, + plus runtime/remote-MCP connectivity. Verify a full cloned-repository P3 + workflow on the final image, including mutable workspace state and normal + P2 behavior. Respect the target repository's publication checks. +4. Exercise the coordinated capacity upgrade/drain and rollback procedure under + deployed writer roles, including realistic scan volume. Retain the verified + local transaction and isolated-live results as evidence for their narrower + scope. +5. Perform the final compatible rollout, including shared runtime changes for + ECS/AgentCore, pinned-version retention and rollback checks. Enable automatic + suspension only after the remaining gates pass, then finish the ADR/runbook + and issue handoff with the actual results. + +Nested CloudFormation stacks remain optional. The current root has 475 +resources; moving existing resources is a separate migration decision. + ## The result we want When the coding agent asks a human for permission, its computer may go to sleep. The human can approve or deny while it sleeps. The computer wakes, reads the saved answer, and continues or refuses the action. If nobody answers, it wakes before the deadline and applies the existing timeout-as-denial rule. Its files, task identity, permissions, progress and deadline remain correct. diff --git a/docs/verification/645-p3-live-deployment-20260915.md b/docs/verification/645-p3-live-deployment-20260915.md index 720b49c44..33452ede5 100644 --- a/docs/verification/645-p3-live-deployment-20260915.md +++ b/docs/verification/645-p3-live-deployment-20260915.md @@ -243,7 +243,11 @@ evidence, plus an unresolved intermittent approval-wake failure. The [effective IAM record](./645-effective-iam-20260915.md) adds actual metadata and S3 permission checks plus real signer-credential expiry. The long durable case also proves scoped credential renewal after expiry, but exposes an -[approval callback timeout](./645-p3-callback-timeout.md). Its updated-image -verification, remaining lifecycle faults/deadline races, network and +[approval callback timeout](./645-p3-callback-timeout.md). Image `4.0` now carries +the correction; the [fresh live record](./645-p3-callback-live-20260915.md) +verifies nine core callback cases, including real expired-key renewal with the +original approval deadline preserved. A fourth +[connection refusal](./645-p3-resume-refusal-investigation.md) occurred on that +image. Remaining lifecycle faults/deadline races, network and capacity/migration checks remain tracked in the [implementation plan](./645-p3-implementation-plan.md). diff --git a/docs/verification/645-p3-readiness-review.md b/docs/verification/645-p3-readiness-review.md index 0caa01d83..9192f5682 100644 --- a/docs/verification/645-p3-readiness-review.md +++ b/docs/verification/645-p3-readiness-review.md @@ -31,8 +31,12 @@ passing cases, including real process-crash/cancellation recovery, automatic deadline wake, supervisor outage and coordinator-owned cleanup. Three approval wakes failed with a service-reported connection-refused error. A long frozen worker renewed expired task credentials but lost its approval callback; the -[local callback fix](./645-p3-callback-timeout.md) still needs an updated image and -fresh AWS acceptance. The [effective permissions record](./645-effective-iam-20260915.md) +[callback fix](./645-p3-callback-timeout.md) is deployed in image `4.0`. +[Fresh AWS acceptance](./645-p3-callback-live-20260915.md) passed all nine core +callback cases, including real expired-key renewal and the original approval +deadline. A fourth [connection refusal](./645-p3-resume-refusal-investigation.md) +occurred on image `4.0`; it remains a separate P3 blocker. +The [effective permissions record](./645-effective-iam-20260915.md) adds 37 metadata checks, 10 S3 checks and real signer-credential expiry. Production automatic suspension remains disabled. diff --git a/docs/verification/645-p3-resume-refusal-investigation.md b/docs/verification/645-p3-resume-refusal-investigation.md new file mode 100644 index 000000000..6c4ddb8ea --- /dev/null +++ b/docs/verification/645-p3-resume-refusal-investigation.md @@ -0,0 +1,110 @@ +# ADR-021 P3: intermittent resume-hook connection refusal + +Updated 2026-09-16 UTC. This is an investigation record and a prepared report; +it has not been submitted to AWS or published as an issue. + +## Observed problem + +An automatically suspended worker sometimes terminates immediately after AWS +acknowledges `ResumeMicrovm`. Its final `stateReason` is: + +> Resume lifecycle hook connection was refused. Please check your hook endpoint +> and application logs for more details. + +The guest previously returned HTTP 200 from `/suspend`. Its retained application +stream has no subsequent `/resume` access log. This does not prove the guest +process stayed healthy: application logs alone cannot distinguish a listener, +process, network-restoration or hook-transport failure. + +The coordinator detects termination, marks the task failed and releases its +capacity reservation. Successful failure cleanup does not make the requested +approval workflow successful. + +## Recorded failures + +All times are UTC, in `us-west-2`, on image `backgroundagent-dev-abca-agent`. +The earlier cases are detailed in the +[durable verification record](./645-p3-durable-live-20260915.md). + +| Case | Image | Worker | Termination | +|---|---|---|---| +| Approval | `3.0` | `microvm-c1d17307-e488-3149-b2c9-5d7f25f6278e` | Sep 15, 21:31:46.958 | +| Approval after coordinator process-crash recovery | `3.0` | `microvm-9e6bdb3f-bc10-3921-9c50-287bbe20b574` | Sep 15, 21:56:56.279 | +| Supervisor repair after an injected inline Resume failure | `3.0` | `microvm-dca019ee-7d19-389e-b90b-bde129f77329` | Sep 15, 22:54:45.753 | +| Supervisor wake after an owned approval record was deleted | `4.0` | `microvm-2da070a7-43cf-3bb3-9c5d-69bc106f2cb1` | Sep 16, 00:25:07.524 | + +### Image 4.0 request timeline + +The last case used source `58ed7bbf`'s image, including the explicit approval +callback timeout. Task ID: `01M2KRWNPP4BQCJ80SVG6473Z8`. + +| Time on Sep 16 | Evidence | +|---|---| +| 00:24:31 | Read approval created with its original 600-second window | +| 00:25:01.476 | Supervisor sends Suspend | +| 00:25:01.534 | Suspend acknowledged; request `e2199854-5ad0-46c9-bc59-248ac68ead82` | +| 00:25:01.657 | Guest `/suspend` returns HTTP 200 | +| 00:25:04.601 | Independent observer sees `SUSPENDED` | +| 00:25:04.767 | Fixture conditionally deletes only its own still-pending approval | +| 00:25:06.696 | Supervisor sends the sole Resume request | +| 00:25:06.766 | Resume acknowledged; request `0acb3fec-8b18-4bed-b457-0271a15ffdd6` | +| 00:25:07.524 | AWS records termination with the connection-refused reason | +| 00:25:11.935 | Coordinator releases the task's reservation | + +No decision API was invoked in this case. No Read result was produced. The +coordinator left an empty payload prefix and counter zero without independent +cleanup. The intended missing-approval check still failed acceptance because +the guest's expected HTTP 409/503 response was not observed. + +A single fresh control, task `01M2KSTSR1V31NTPQ0CKCJJJMN`, worker +`microvm-a093863f-edf0-39f4-9d79-6893ff08042c`, did reach `/resume`, which returned +HTTP 409 at 00:30:46.418Z after the same owned-approval deletion. AWS reported +the HTTP status explicitly. No tool ran and cleanup passed. Thus the missing +record can produce the expected application error, distinct from the refusal. + +## What the comparison establishes + +Six fresh approval tasks used identical production coordinator code, five-second +polling, a 600-second gate and one read-only action: three on image `3.0`, then +three on `4.0`. All six completed automatic suspension, HTTP 200 resume, exactly +one approved Read and untouched coordinator cleanup. + +The image `4.0` refusal above happened afterward. Therefore: + +- The failure is intermittent in the observed runs. +- The explicit callback-timeout fix does not eliminate it. +- Overlapping API/supervisor Resume requests are not required. Both the + process-crash case and the supervisor-only cases exclude that explanation. +- The observed failure is separate from the old ten-minute Claude callback + cancellation: the latest failure happened less than a minute after its gate + was created, with minutes still remaining. +- These runs do not establish a failure rate or identify the responsible + component. + +The [guest lifecycle HTTP handler](../../agent/src/microvm_http.py) and +[pause controller](../../agent/src/microvm_lifecycle.py) were reviewed. +Suspend drains tracked work and checkpoints; it does not intentionally close +the HTTP listener. The server's shutdown path is separate. That source review +does not exclude a process crash or a lower-level restore problem. + +## Next investigation + +1. Use the recorded worker IDs, region, timestamps and Resume request IDs to + inspect service-side lifecycle diagnostics. Determine the actual connection + error and whether the request reached the guest, including any transport retry. +2. Correlate guest process/kernel health and the port-8080 listener at restoration. + Application access logs do not provide that missing evidence. +3. If a transport or application race is identified, make a bounded correction + and test that trigger specifically. Do not hide a failed wake by silently + launching another worker: the approved action and workspace may already have + changed. +4. Repeat approval, denial, original/late deadline, multiple-gate persistence and + expired-credential cases on the corrected image. Keep the previous failures + in the evidence record. +5. Leave production automatic suspension off until this gate and the + [remaining acceptance plan](./645-p3-implementation-plan.md) are satisfied. + +The private evidence archive contains full guest streams, redacted control +request telemetry, durable history and task events. It contains no saved AWS +credential values in the report. Temporary verification resources are tracked +separately in the [callback live record](./645-p3-callback-live-20260915.md). From 9ee321455872c6bf6569e5ee5fbae331342937b5 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Tue, 15 Sep 2026 23:26:48 -0400 Subject: [PATCH 044/149] fix: handle pending state during MicroVM wake (#645) --- .../microvm_lifecycle_listener_probe.py | 214 ++++++++++++++++++ cdk/src/handlers/shared/microvm-supervisor.ts | 12 +- .../shared/microvm-supervisor.test.ts | 68 +++++- .../645-p3-implementation-plan.md | 3 + .../645-p3-listener-probe-20260916.md | 113 +++++++++ docs/verification/645-p3-pending-wake.md | 65 ++++++ .../645-p3-resume-refusal-investigation.md | 15 ++ 7 files changed, 485 insertions(+), 5 deletions(-) create mode 100644 agent/scripts/microvm_lifecycle_listener_probe.py create mode 100644 docs/verification/645-p3-listener-probe-20260916.md create mode 100644 docs/verification/645-p3-pending-wake.md diff --git a/agent/scripts/microvm_lifecycle_listener_probe.py b/agent/scripts/microvm_lifecycle_listener_probe.py new file mode 100644 index 000000000..5c1ac2358 --- /dev/null +++ b/agent/scripts/microvm_lifecycle_listener_probe.py @@ -0,0 +1,214 @@ +#!/usr/bin/env python3 +# Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. +# SPDX-License-Identifier: MIT-0 + +"""Isolated MicroVM transport probe; never import into the application runtime. + +Serve the six lifecycle hooks using only Python's standard library. Logs contain +request boundaries, listener health and a disposable marker's hash, without AWS +calls or credentials. The explicit ``close_listener`` mode is a failure control. +Use only in an owned diagnostic image with no ingress and a bounded VM lifetime. +""" + +from __future__ import annotations + +import argparse +import faulthandler +import hashlib +import json +import os +import signal +import socket +import tempfile +import threading +import time +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer +from pathlib import Path + +PREFIX = "/aws/lambda-microvms/runtime/v1/" +HOOKS = {"ready", "validate", "run", "suspend", "resume", "terminate"} +MAX_BODY_BYTES = 16_384 +MAX_CASE_ID_LENGTH = 100 + + +def emit(event: str, **fields: object) -> None: + """One short stdout write, independent of AWS clients and logging threads.""" + record = { + "event": event, + "wall_s": time.time(), + "monotonic_s": time.monotonic(), + "pid": os.getpid(), + **fields, + } + os.write(1, (json.dumps(record, sort_keys=True) + "\n").encode()) + + +class ProbeServer(ThreadingHTTPServer): + daemon_threads = True + + def __init__(self, host: str, port: int) -> None: + self.lock = threading.RLock() + self.case_id = "" + self.mode = "normal" + self.phase = "image" + self.suspends = 0 + self.resumes = 0 + self.marker_hash = "" + self.marker = Path(tempfile.gettempdir()) / f"abca-listener-probe-{os.getpid()}" + super().__init__((host, port), ProbeHandler) + + def health(self) -> dict[str, object]: + try: + listening: object = bool( + self.socket.getsockopt(socket.SOL_SOCKET, socket.SO_ACCEPTCONN) + ) + except OSError as exc: + listening = type(exc).__name__ + return { + "listening": listening, + "threads": threading.active_count(), + "case_id": self.case_id, + "phase": self.phase, + "suspends": self.suspends, + "resumes": self.resumes, + "marker_hash": self.marker_hash, + } + + +class ProbeHandler(BaseHTTPRequestHandler): + server: ProbeServer + protocol_version = "HTTP/1.1" + + def log_message(self, format: str, *args: object) -> None: + # Request headers and service payloads are deliberately not logged. + return + + def do_POST(self) -> None: + hook = self.path.removeprefix(PREFIX) + if not self.path.startswith(PREFIX) or hook not in HOOKS: + self.respond(404, {"code": "UNKNOWN_HOOK"}) + return + emit("hook_enter", hook=hook, **self.server.health()) + try: + self.connection.settimeout(2) + size = int(self.headers.get("Content-Length", "0")) + if not 0 <= size <= MAX_BODY_BYTES: + raise ValueError("Invalid body length") + body = self.rfile.read(size) + if len(body) != size: + raise ValueError("Truncated body") + envelope = json.loads(body) if body else {} + if not isinstance(envelope, dict): + raise ValueError("Expected object") + self.transition(hook, envelope) + except (OSError, ValueError, TypeError, KeyError) as exc: + emit("hook_error", hook=hook, error_type=type(exc).__name__) + self.respond(409, {"code": "PROBE_REJECTED"}) + + def transition(self, hook: str, envelope: dict) -> None: + server = self.server + close_listener = False + with server.lock: + if hook == "run": + payload = json.loads(envelope["runHookPayload"]) + case_id, mode = payload["case_id"], payload.get("mode", "normal") + if ( + not isinstance(case_id, str) + or not 1 <= len(case_id) <= MAX_CASE_ID_LENGTH + or mode not in {"normal", "close_listener"} + ): + raise ValueError("Invalid probe configuration") + if server.case_id and (server.case_id != case_id or server.mode != mode): + raise ValueError("Conflicting run") + if not server.case_id: + server.case_id, server.mode = case_id, mode + marker = os.urandom(32) + server.marker.write_bytes(marker) + server.marker_hash = hashlib.sha256(marker).hexdigest() + server.phase = "running" + elif hook == "suspend": + if server.phase == "running": + server.phase = "suspended" + server.suspends += 1 + close_listener = server.mode == "close_listener" + elif server.phase != "suspended": + raise ValueError("No running probe") + elif hook == "resume": + if server.phase == "suspended": + if hashlib.sha256(server.marker.read_bytes()).hexdigest() != server.marker_hash: + raise ValueError("Marker changed") + server.resumes += 1 + server.phase = "running" + elif server.phase != "running": + raise ValueError("No suspended probe") + elif hook == "terminate": + server.phase = "terminating" + + if close_listener: + # Close only the listening socket; this accepted /suspend connection + # can still send its response. The process stays alive for diagnostics. + server.shutdown() + server.server_close() + emit("listener_closed_intentionally", **server.health()) + result = {"status": "acknowledged", "hook": hook, **server.health()} + emit("hook_ack", **result) + self.respond(200, result) + + def respond(self, status: int, payload: dict) -> None: + body = json.dumps(payload).encode() + self.send_response(status) + self.send_header("Content-Type", "application/json") + self.send_header("Content-Length", str(len(body))) + # Make every service hook establish a connection to the listener. + self.send_header("Connection", "close") + self.end_headers() + self.wfile.write(body) + self.wfile.flush() + self.close_connection = True + + +def observe(server: ProbeServer) -> None: + previous_wall, previous_monotonic = time.time(), time.monotonic() + tick = 0 + while True: + time.sleep(0.25) + wall, monotonic = time.time(), time.monotonic() + wall_gap, monotonic_gap = wall - previous_wall, monotonic - previous_monotonic + tick += 1 + if wall_gap > 1 or monotonic_gap > 1 or tick % 20 == 0: + emit( + "listener_observation", + wall_gap_s=wall_gap, + monotonic_gap_s=monotonic_gap, + **server.health(), + ) + previous_wall, previous_monotonic = wall, monotonic + + +def main() -> None: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--host", default="127.0.0.1") + parser.add_argument("--port", type=int, default=8080) + args = parser.parse_args() + faulthandler.enable() + + def stop(signum: int, _frame: object) -> None: + emit("process_signal", signal=signum) + raise SystemExit(0) + + signal.signal(signal.SIGTERM, stop) + signal.signal(signal.SIGINT, stop) + server = ProbeServer(args.host, args.port) + threading.Thread(target=observe, args=(server,), daemon=True).start() + emit("listener_started", port=server.server_port, **server.health()) + try: + server.serve_forever(poll_interval=0.05) + # The negative control intentionally stops acceptance, not the process. + threading.Event().wait() + finally: + server.server_close() + server.marker.unlink(missing_ok=True) + + +if __name__ == "__main__": + main() diff --git a/cdk/src/handlers/shared/microvm-supervisor.ts b/cdk/src/handlers/shared/microvm-supervisor.ts index 9f1a6758e..b80a56214 100644 --- a/cdk/src/handlers/shared/microvm-supervisor.ts +++ b/cdk/src/handlers/shared/microvm-supervisor.ts @@ -227,8 +227,16 @@ export async function superviseMicrovm(input: MicrovmSupervisorInput): Promise { +test.each(['PENDING', 'UNKNOWN'])('%s cannot reset a wake recovery clock', async observed => { approve(); intent('suspend'); strategy.pollSession.mockResolvedValue(observation('SUSPENDED')); const first = await run(); time += 60_000; - strategy.pollSession.mockResolvedValue(observation('UNKNOWN')); + strategy.pollSession.mockResolvedValue(observation(observed)); const unknown = await run(first.state); expect(unknown.state.recovery).toEqual(first.state.recovery); time += 60_000; @@ -371,6 +371,67 @@ test('uncertain observations cannot reset a wake recovery clock', async () => { expect(strategy.resumeSession).toHaveBeenCalledTimes(1); }); +test('PENDING after an API wake uses the wake intent instead of an expired startup clock', async () => { + intent('suspend'); + strategy.pollSession.mockResolvedValue(observation('SUSPENDED')); + const asleep = await run(); + expect(asleep.state.recovery).toBeUndefined(); + + time += 360_000; + approve(); + intent('resume', time); + const wakeStartedAt = time; + strategy.pollSession.mockResolvedValue(observation('PENDING')); + const pending = await run(asleep.state); + expect(pending).toMatchObject({ + kind: 'continue', + deferHeartbeat: true, + state: { recovery: { kind: 'wake', sinceMs: wakeStartedAt } }, + }); + expect(pending.state.firstObservedAtMs).toBe(asleep.state.firstObservedAtMs); + expect(pending.state.sessionDeadlineMs).toBe(asleep.state.sessionDeadlineMs); + expect(strategy.resumeSession).not.toHaveBeenCalled(); + + time += 10_000; + const replayed = await run(pending.state); + expect(replayed.state.recovery).toEqual(pending.state.recovery); + strategy.pollSession.mockResolvedValue(observation('RUNNING')); + const consuming = await run(replayed.state); + expect(consuming.state.recovery).toEqual(pending.state.recovery); + working(); + row = { ...row, heartbeatAtMs: time }; + const rebound = await run(consuming.state); + expect(rebound.state.recovery).toEqual(pending.state.recovery); + expect(row.intent?.request_id).toBeNull(); + expect((await run(rebound.state)).state.recovery).toBeUndefined(); +}); + +test.each(['PENDING', 'UNKNOWN'])('%s does not extend an already-expired API wake intent', async observed => { + approve(); + intent('resume', NOW - MICROVM_RECOVERY_TIMEOUT_MS); + strategy.pollSession.mockResolvedValue(observation(observed)); + expect(await run()).toMatchObject({ + kind: 'failure', + state: { recovery: { kind: 'wake', sinceMs: NOW - MICROVM_RECOVERY_TIMEOUT_MS } }, + }); + expect(strategy.resumeSession).not.toHaveBeenCalled(); +}); + +test('unexpected PENDING after coding starts receives one bounded recovery window', async () => { + working(); + row = { ...row, heartbeatAtMs: time }; + const running = await run(); + time += 360_000; + strategy.pollSession.mockResolvedValue(observation('PENDING')); + const pending = await run(running.state); + expect(pending).toMatchObject({ + kind: 'continue', + state: { recovery: { kind: 'unconfirmed', sinceMs: time } }, + }); + time += MICROVM_RECOVERY_TIMEOUT_MS; + expect((await run(pending.state)).kind).toBe('failure'); +}); + test('durable wake recovery repairs a missing wake write even after a RUNNING observation', async () => { intent('suspend'); const initial = await run(undefined, { suspendEnabled: false }); @@ -456,9 +517,10 @@ test('a normal wake in progress is not reported as an unintended suspension', as expect(emitEvent).not.toHaveBeenCalled(); }); -test('HYDRATING startup is bounded even when AWS reports RUNNING', async () => { +test.each(['PENDING', 'RUNNING'])('HYDRATING startup is bounded when AWS reports %s', async observed => { working(); row = { ...row, status: 'HYDRATING' }; + strategy.pollSession.mockResolvedValue(observation(observed)); const starting = await run(); time += 300_000; expect(await run(starting.state)).toMatchObject({ kind: 'failure', reason: 'startup-deadline' }); diff --git a/docs/verification/645-p3-implementation-plan.md b/docs/verification/645-p3-implementation-plan.md index 149a02c25..5f16b9926 100644 --- a/docs/verification/645-p3-implementation-plan.md +++ b/docs/verification/645-p3-implementation-plan.md @@ -229,6 +229,9 @@ handoff, use this order: Service-side connection diagnostics and guest/listener health are still missing. Passing retries, the callback-timeout fix and successful cleanup do not close this gate. + The [minimal listener experiment](./645-p3-listener-probe-20260916.md) also + exposed a separate [pending-wake timer bug](./645-p3-pending-wake.md). + Its local correction passed the full build and needs a live coordinator check. 2. Complete the remaining live race/fault matrix: cancellation during transitions, late decision races, repeated polling/credential-refresh failures, durable registration races, service token-retention expiry and recovery of a worker diff --git a/docs/verification/645-p3-listener-probe-20260916.md b/docs/verification/645-p3-listener-probe-20260916.md new file mode 100644 index 000000000..474f2176d --- /dev/null +++ b/docs/verification/645-p3-listener-probe-20260916.md @@ -0,0 +1,113 @@ +# ADR-021 P3: isolate the resume connection failure + +Verified 2026-09-16 UTC. This diagnostic experiment follows the four +[resume-hook connection refusals](./645-p3-resume-refusal-investigation.md). +The original refusal did not recur in this bounded sample. The deliberate +closed-listener control produced the expected refusal. All temporary resources +were removed; the production image and both suspension switches are unchanged. + +## Question and scope + +Can a small HTTP listener reproduce the refusal without the coding agent? +The [probe](../../agent/scripts/microvm_lifecycle_listener_probe.py) uses Python's +standard library to serve the six lifecycle hooks. It makes no AWS SDK calls, +runs no coding task, and receives no repository or tenant configuration. + +The owned image uses the same managed base `al2023-1`, base version `1.0`, +ARM64 architecture, 8,192 MiB memory and hook budgets as production image `4.0`. +Its Dockerfile uses the same pinned Python base: + +```text +python:3.13-slim@sha256:dc1546eefcbe8caaa1f004f16ab76b204b5e1dbd58ff81b899f21cd40541232f +``` + +It has its own build/runtime roles and log group. The build role can read only +the diagnostic artifact; both roles can write only diagnostic logs. Each worker +uses the existing restricted runtime connector, explicit `NO_INGRESS` and a +300-second maximum lifetime. A standalone operator runner controls only workers +belonging to this uniquely named image. + +This is not the production ASGI server, approval controller or durable +coordinator. Responses explicitly close each accepted HTTP connection so the next +hook must contact the listener again. Success here would not prove the full +application or its keep-alive behavior correct. + +## Evidence collected + +The probe writes structured stdout records for request entry, acknowledgment, +listener health, process signals and clock gaps across suspension. `/run` creates +a random disposable marker; each `/resume` checks its original hash and records +the retained suspend/resume counts. + +The failure control deliberately stops and closes the listening socket during +`/suspend`, then sends HTTP 200 on the already accepted connection. The process +remains alive, and its observer can report that the listener is closed. That +case must not be counted as an unexpected platform failure. + +Locally, three normal cycles retained the same marker. The failure control +acknowledged suspension and a new connection then failed with `ECONNREFUSED`. +The two subprocesses were terminated after verification. Ruff, formatting and +type checks passed. + +## Fixed live matrix + +| Cases | Time from observing `SUSPENDED` to requesting Resume | Cycles per worker | +|---|---|---| +| `delay-0-a`, `delay-0-b` | No additional delay | 3 | +| `delay-2-a`, `delay-2-b` | 2 seconds | 3 | +| `delay-10-a`, `delay-10-b` | 10 seconds | 3 | +| `closed-listener` | 2 seconds; deliberate refusal control | 1 | + +The six normal cases target 18 wakes. Failed workers are not replaced or retried +until a case passes. Request IDs, actual state observations, guest records and +termination evidence are retained for each case. A launch whose response is lost +is recovered by enumerating this owned image's workers for cleanup. + +Before these cases, the private runner incorrectly sent runtime image version +`1`; AWS rejected all seven requests without creating a worker. Their records +are retained separately. The corrected request uses `1.0`, matching the image's +published version. This differs from `CreateMicrovmImage.baseImageVersion`, +which requires the major version string `1` despite returning `1.0` in reads. + +## Interpretation and cleanup + +The completed matrix produced four fully passing normal cases, 13 cycles with +all runner checks, and 14 successful resume acknowledgments in guest logs. +Three cases were interrupted by 15-second AWS request timeouts: + +| Case | Fully checked cycles | Result | +|---|---|---| +| `delay-0-a` | 3 | Passed | +| `delay-0-b` | 0 | Read-side timeout; one successful guest resume is recorded | +| `delay-2-a` | 3 | Passed | +| `delay-2-b` | 3 | Passed | +| `delay-10-a` | 3 | Passed | +| `delay-10-b` | 1 | Second Suspend request timed out; guest logs prove it nevertheless acknowledged suspension | +| `closed-listener` | 0 | Timeout before the runner requested suspension | + +These interruptions remain incomplete tests. An API timeout is not proof that +the operation never occurred. The runner terminated the owned workers without +issuing replacement runs for these cases. + +One fresh control, `closed-listener-r2`, used a bounded 45-second request timeout. +Worker `microvm-2167fc9d-21bf-35f3-a3fc-f340813f03a8` deliberately closed its +listener at 03:16:07.093Z and acknowledged `/suspend` on the existing connection. +Resume request `96147502-07f3-4512-9397-0cf1063cca67` was acknowledged at +03:16:10.143Z. At 03:16:10.420Z the observer still ran as PID 1, reporting a +closed listener. AWS recorded termination at 03:16:10.811Z with exactly the +original connection-refused reason. No `/resume` request entered the handler. +This passes the intentional failure control; it is not a fifth unexplained +refusal. + +The normal cases show non-reproduction in a small, different application. +They do not clear the four original failures or identify their cause. The next +useful evidence is listener/process health in the full agent image. + +The experiment also observed `PENDING` during real restoration. That exposed a +separate [supervisor timer bug](./645-p3-pending-wake.md), reproduced and corrected +locally. + +All eight actual workers were verified `TERMINATED`. The owned image, its build +and runtime roles, exact S3 artifact and log group were removed. Read-only checks +at 03:21:47.813Z verified their absence. The archive retains all per-case records, +208 log events, image settings, request IDs and cleanup evidence. diff --git a/docs/verification/645-p3-pending-wake.md b/docs/verification/645-p3-pending-wake.md new file mode 100644 index 000000000..6331f47ce --- /dev/null +++ b/docs/verification/645-p3-pending-wake.md @@ -0,0 +1,65 @@ +# ADR-021 P3: distinguish a pending wake from initial startup + +Date: 2026-09-16. The correction is local; deployment and live coordinator +verification are pending. + +## Trigger + +The [minimal listener experiment](./645-p3-listener-probe-20260916.md) observed +this real AWS sequence for worker +`microvm-47df351d-efe7-33f9-bf86-1892b66dc203`: + +| UTC time | Observation | +|---|---| +| 03:09:26.768 | `SUSPENDED` | +| 03:09:27.054 | `PENDING`, after Resume was requested | +| 03:09:27.684 | `RUNNING` | + +The service has no separate `RESUMING` enum, but `PENDING` can appear during +restoration. It does not always mean a worker's first startup. + +## Bug and correction + +When a worker is stably asleep, the supervisor has no active recovery timer. +The approval API can save a wake instruction and request Resume between +supervisor polls. If the next poll sees `PENDING`, the old supervisor starts a +startup timer using the worker's original first-observed time. + +For a worker older than five minutes, that timer is already expired. The +supervisor can fail an otherwise healthy wake immediately. + +A regression reproduced this with the production supervisor function: after +six minutes asleep, a committed approval and wake instruction followed by +`PENDING` returned `failure` with recovery kind `starting`. This is a local +reproduction of the coordinator bug, informed by a real AWS state sequence; +the minimal experiment itself does not run the durable coordinator. + +The supervisor now handles `PENDING` and unknown observations as follows: + +- A wake instruction matching the same worker and approval uses its saved + request time for bounded wake recovery. +- A worker still in `HYDRATING` may use the original initial-startup timer. +- An unexpected pending state after coding began receives one bounded + uncertainty window. + +Existing recovery timers retain their start time across replay. The original +worker lifetime and first-observed time do not change. An already-expired wake +instruction still fails, and a future instruction timestamp cannot extend the +window. The change adds no durable-state field, IAM permission or image hook. + +## Validation and remaining scope + +The regression failed before the correction. All 150 tests in the supervisor, +MicroVM orchestrator and lifecycle suites passed afterward. Coverage includes +replay, stale wake instructions, pending/unknown observations, initial startup, +and completion after the guest consumes the decision and reports liveness. + +The full repository build passed in 488.56 seconds: 4,999 CDK tests passed with +56 optional DynamoDB Local tests skipped; 1,942 Python tests passed with 11 +skipped; all 928 CLI tests passed. Compilation, lint, type checks, contract/drift +checks, synthesis and the 77-page documentation build also passed. + +A live check of the updated coordinator remains pending. This correction does +not explain or resolve the four separate +[resume-hook connection refusals](./645-p3-resume-refusal-investigation.md). +Production automatic suspension remains off. diff --git a/docs/verification/645-p3-resume-refusal-investigation.md b/docs/verification/645-p3-resume-refusal-investigation.md index 6c4ddb8ea..e74ca4709 100644 --- a/docs/verification/645-p3-resume-refusal-investigation.md +++ b/docs/verification/645-p3-resume-refusal-investigation.md @@ -89,11 +89,26 @@ does not exclude a process crash or a lower-level restore problem. ## Next investigation +The [minimal listener experiment](./645-p3-listener-probe-20260916.md) completed +four full normal cases and recorded 14 guest resume acknowledgments without an +unexpected refusal. Request timeouts interrupted three cases. A fresh deliberate +closed-listener control produced the exact refusal reason while an independent +observer still ran in the guest. This calibrates the diagnostics; it does not +establish why the full agent loses its listener or connection. + +That experiment also exposed a separate [pending-wake timer bug](./645-p3-pending-wake.md) +in the supervisor. Its correction does not account for these four failures, +whose worker termination reason was the service-reported connection refusal. + 1. Use the recorded worker IDs, region, timestamps and Resume request IDs to inspect service-side lifecycle diagnostics. Determine the actual connection error and whether the request reached the guest, including any transport retry. 2. Correlate guest process/kernel health and the port-8080 listener at restoration. Application access logs do not provide that missing evidence. + A diagnostic full-agent image can add an independent parent-process observer + to record child exit status and listener health without changing approval + decisions or granting public ingress. Preserve the original image and compare + a fixed, bounded set of owned tasks. 3. If a transport or application race is identified, make a bounded correction and test that trigger specifically. Do not hide a failed wake by silently launching another worker: the approved action and workspace may already have From 8ae2bf2c9243d1ddaf15bc295d27e55dacbca238 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Wed, 16 Sep 2026 08:31:45 -0400 Subject: [PATCH 045/149] fix: preserve MicroVM startup allowance (#645) --- agent/scripts/microvm_process_observer.py | 137 ++++++++++++++++++ cdk/src/handlers/shared/microvm-supervisor.ts | 11 +- .../shared/microvm-supervisor.test.ts | 44 ++++++ .../645-p3-implementation-plan.md | 3 +- docs/verification/645-p3-pending-wake.md | 52 ++++++- .../645-p3-process-observer-20260916.md | 104 +++++++++++++ docs/verification/645-p3-supervisor.md | 17 ++- 7 files changed, 358 insertions(+), 10 deletions(-) create mode 100644 agent/scripts/microvm_process_observer.py create mode 100644 docs/verification/645-p3-process-observer-20260916.md diff --git a/agent/scripts/microvm_process_observer.py b/agent/scripts/microvm_process_observer.py new file mode 100644 index 000000000..cd2c85010 --- /dev/null +++ b/agent/scripts/microvm_process_observer.py @@ -0,0 +1,137 @@ +"""Temporary image diagnostic: observe the agent server without opening sockets. + +Run as the parent of the normal image command, after ``--``. This changes the +process tree and is diagnostic instrumentation, not a production entry point. +Only process state, port-8080 listener metadata, and cgroup memory counters are +recorded. Arguments, environment, request bodies and credentials are not logged. +""" + +import argparse +import contextlib +import json +import os +import signal +import subprocess +import time +from pathlib import Path +from typing import Any + +_TCP_FIELDS_THROUGH_INODE = 10 +_REPORT_INTERVAL_S = 5 + + +def emit(event: str, **fields: Any) -> None: + print( + json.dumps( + { + "probe": "microvm-process-observer", + "event": event, + "wall_time_ns": time.time_ns(), + "monotonic_ns": time.monotonic_ns(), + "observer_pid": os.getpid(), + **fields, + } + ), + flush=True, + ) + + +def snapshot(pid: int) -> dict[str, Any]: + result: dict[str, Any] = {"child_pid": pid} + try: + status = Path(f"/proc/{pid}/status").read_text() + allowed = {"State", "Threads", "VmRSS", "SigPnd", "ShdPnd"} + result["child_status"] = { + key: value.strip() + for line in status.splitlines() + for key, _, value in [line.partition(":")] + if key in allowed + } + except OSError as error: + result["child_status_error"] = type(error).__name__ + listeners = [] + for family in ("tcp", "tcp6"): + try: + for line in Path(f"/proc/net/{family}").read_text().splitlines()[1:]: + fields = line.split() + if ( + len(fields) >= _TCP_FIELDS_THROUGH_INODE + and fields[1].endswith(":1F90") + and fields[3] == "0A" + ): + listeners.append({"family": family, "inode": fields[9]}) + except OSError as error: + result[f"{family}_error"] = type(error).__name__ + result["listeners_8080"] = listeners + try: + inodes = { + os.readlink(entry) + for entry in Path(f"/proc/{pid}/fd").iterdir() + if entry.name.isdigit() + } + result["child_owns_listener"] = any( + f"socket:[{listener['inode']}]" in inodes for listener in listeners + ) + except OSError as error: + result["child_fds_error"] = type(error).__name__ + try: + result["memory_events"] = Path("/sys/fs/cgroup/memory.events").read_text().strip() + except OSError as error: + result["memory_events_error"] = type(error).__name__ + return result + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("command", nargs=argparse.REMAINDER) + args = parser.parse_args() + command = args.command + if command[:1] == ["--"]: + command = command[1:] + if not command: + parser.error("a child command is required") + child = subprocess.Popen(command, start_new_session=True) + shutdown_at: float | None = None + + def forward_signal(number: int, _frame: Any) -> None: + nonlocal shutdown_at + emit("observer_signal", signal=number, child_pid=child.pid) + shutdown_at = time.monotonic() + 10 + with contextlib.suppress(ProcessLookupError): + os.killpg(child.pid, number) + + signal.signal(signal.SIGTERM, forward_signal) + signal.signal(signal.SIGINT, forward_signal) + emit("child_started", child_pid=child.pid) + previous: dict[str, Any] | None = None + last_sample = time.monotonic() + last_emit = 0.0 + try: + while True: + now = time.monotonic() + state = snapshot(child.pid) + code = child.poll() + if state != previous or now - last_emit >= _REPORT_INTERVAL_S or now - last_sample > 1: + emit( + "process_observation", returncode=code, sample_gap_s=now - last_sample, **state + ) + previous = state + last_emit = now + last_sample = now + if code is not None: + emit("child_exited", returncode=code, **state) + return code if code >= 0 else 128 - code + if shutdown_at is not None and now >= shutdown_at: + emit("shutdown_deadline", child_pid=child.pid) + with contextlib.suppress(ProcessLookupError): + os.killpg(child.pid, signal.SIGKILL) + time.sleep(0.25) + finally: + if child.poll() is None: + with contextlib.suppress(ProcessLookupError): + os.killpg(child.pid, signal.SIGKILL) + child.wait() + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/cdk/src/handlers/shared/microvm-supervisor.ts b/cdk/src/handlers/shared/microvm-supervisor.ts index b80a56214..5f1641164 100644 --- a/cdk/src/handlers/shared/microvm-supervisor.ts +++ b/cdk/src/handlers/shared/microvm-supervisor.ts @@ -51,6 +51,8 @@ export interface MicrovmSupervisorState { readonly firstObservedAtMs: number; readonly sessionDeadlineMs: number; readonly lifetimeVerified: boolean; + /** False until AWS confirms a post-startup state; absent in older saved state. */ + readonly startupConfirmed?: boolean; readonly consecutivePollFailures: number; readonly consecutiveResumeFailures: number; readonly recovery?: Recovery; @@ -118,6 +120,7 @@ export async function superviseMicrovm(input: MicrovmSupervisorInput): Promise { + working(); // The coordinator marks the task RUNNING before AWS reports readiness. + strategy.pollSession.mockResolvedValue(observation('PENDING')); + const first = await run(); + expect(first).toMatchObject({ + kind: 'continue', state: { recovery: { kind: 'starting', sinceMs: NOW } }, + }); + time += 120_000; + const later = await run(first.state); + expect(later.kind).toBe('continue'); + expect(later.state.recovery).toEqual(first.state.recovery); + time = NOW + 300_000; + expect((await run(later.state)).kind).toBe('failure'); +}); + +test('an initial read failure cannot shorten or restart the pending startup allowance', async () => { + working(); + mockRead.mockRejectedValueOnce(new Error('temporary read failure')); + const failed = await run(); + time += 60_000; + strategy.pollSession.mockResolvedValue(observation('PENDING')); + const pending = await run(failed.state); + expect(pending).toMatchObject({ + kind: 'continue', state: { recovery: { kind: 'starting', sinceMs: NOW } }, + }); + time = NOW + 300_000; + expect((await run(pending.state)).kind).toBe('failure'); +}); + +test.each(['current', 'legacy'])('%s saved state does not restart startup after a confirmed running worker', async version => { + working(); + const running = await run(); + const previous = { ...running.state }; + if (version === 'legacy') delete previous.startupConfirmed; + time += 60_000; + strategy.pollSession.mockResolvedValue(observation('PENDING')); + const pending = await run(previous); + expect(pending).toMatchObject({ + kind: 'continue', state: { recovery: { kind: 'unconfirmed', sinceMs: time } }, + }); + time += MICROVM_RECOVERY_TIMEOUT_MS; + expect((await run(pending.state)).kind).toBe('failure'); +}); + test('repeated Get errors retain their count even when task reads succeed', async () => { strategy.pollSession.mockRejectedValue(Object.assign(new Error('network'), { name: 'TimeoutError' })); const first = await run(); diff --git a/docs/verification/645-p3-implementation-plan.md b/docs/verification/645-p3-implementation-plan.md index 5f16b9926..5c11cd8c8 100644 --- a/docs/verification/645-p3-implementation-plan.md +++ b/docs/verification/645-p3-implementation-plan.md @@ -231,7 +231,8 @@ handoff, use this order: do not close this gate. The [minimal listener experiment](./645-p3-listener-probe-20260916.md) also exposed a separate [pending-wake timer bug](./645-p3-pending-wake.md). - Its local correction passed the full build and needs a live coordinator check. + Its correction passed the full build and is deployed in coordinator version 4; + it needs a live coordinator acceptance check. 2. Complete the remaining live race/fault matrix: cancellation during transitions, late decision races, repeated polling/credential-refresh failures, durable registration races, service token-retention expiry and recovery of a worker diff --git a/docs/verification/645-p3-pending-wake.md b/docs/verification/645-p3-pending-wake.md index 6331f47ce..5c5e13631 100644 --- a/docs/verification/645-p3-pending-wake.md +++ b/docs/verification/645-p3-pending-wake.md @@ -1,7 +1,7 @@ # ADR-021 P3: distinguish a pending wake from initial startup -Date: 2026-09-16. The correction is local; deployment and live coordinator -verification are pending. +Date: 2026-09-16. The correction is deployed in coordinator version 4; live +coordinator acceptance remains pending. ## Trigger @@ -38,14 +38,46 @@ The supervisor now handles `PENDING` and unknown observations as follows: - A wake instruction matching the same worker and approval uses its saved request time for bounded wake recovery. -- A worker still in `HYDRATING` may use the original initial-startup timer. +- In deployed version 4, a worker still in `HYDRATING` may use the original + initial-startup timer. The follow-up below also covers the coordinator's + transition to `RUNNING` before AWS startup completes. - An unexpected pending state after coding began receives one bounded uncertainty window. Existing recovery timers retain their start time across replay. The original worker lifetime and first-observed time do not change. An already-expired wake instruction still fails, and a future instruction timestamp cannot extend the -window. The change adds no durable-state field, IAM permission or image hook. +window. The version-4 correction added no durable-state field, IAM permission +or image hook. + +## Follow-up: task RUNNING does not establish worker readiness + +The [full-agent observer experiment](./645-p3-process-observer-20260916.md) +found another startup case. Task `01M2N176C2WQ21PJ92FMTSZA9A` was already +`RUNNING` while AWS still reported its new worker as `PENDING`. Its actual first +supervisor result at 12:09:28.207 UTC used `unconfirmed`, a two-minute allowance, +instead of the intended five-minute startup allowance. + +The coordinator deliberately transitions the task before observing worker +readiness. The follow-up therefore records an optional internal +`startupConfirmed` flag, initially false and set true after AWS reports +`RUNNING`, `SUSPENDING` or `SUSPENDED`. Initial `PENDING` observations can then +use the original startup clock regardless of task status. Failed initial reads +retain that phase across replay. An older saved state without the flag does not +receive a new startup phase. Wake instructions and existing recovery clocks keep +their previous behavior. + +Two regressions failed before this follow-up and passed after it; current and +legacy confirmed-worker cases also retain bounded uncertainty recovery. +All 154 focused tests passed. The full build passed 4,995 CDK tests but hit eight +disk-space failures in one stack suite. After removing obsolete generated +assemblies, that entire suite passed all 138 tests, covering all eight failures. +Python, CLI, lint, types, contracts, documentation and synthesis passed. The +successful west-region synthesis was repeated after the same disk-space issue. + +Four real first-start `PENDING` polls in the private coordinator also retained +the original startup clock. Production deployment of this follow-up is pending; +it changes no IAM permission or agent hook. ## Validation and remaining scope @@ -59,6 +91,18 @@ The full repository build passed in 488.56 seconds: 4,999 CDK tests passed with skipped; all 928 CLI tests passed. Compilation, lint, type checks, contract/drift checks, synthesis and the 77-page documentation build also passed. +The narrow deployment completed at 11:48:07 UTC in `sphia-dev`, `us-west-2`. +The live alias now selects coordinator version 4, with code SHA-256 +`1aITI4Dx3HdUMuAn9DnVLphZTpX6s4jyn+Odoz/N1Rs=`. Version 3 remains retained. +The reviewed change set modified only coordinator code/version/alias resources. +Resolved environment values, guardrail version 3 and agent image 4.0 are unchanged. +Both the environment and SSM automatic-suspension switches remain `false`. + +The exact published coordinator ZIP and deployment evidence are preserved in +the private persistent archive +`/Users/sphias/.local/share/abca-verification/645-p3-20260916/listener-pending-evidence.tar.gz` +(SHA-256 `221dffad6be148b296e95d89b501f6489db9038416d482cb0753c78465d59994`). + A live check of the updated coordinator remains pending. This correction does not explain or resolve the four separate [resume-hook connection refusals](./645-p3-resume-refusal-investigation.md). diff --git a/docs/verification/645-p3-process-observer-20260916.md b/docs/verification/645-p3-process-observer-20260916.md new file mode 100644 index 000000000..cd2544df6 --- /dev/null +++ b/docs/verification/645-p3-process-observer-20260916.md @@ -0,0 +1,104 @@ +# ADR-021 P3: full-agent process observer + +Date: 2026-09-16. Three fallback-wake cases passed. The direct-wake control and +resource cleanup remain in progress. + +## Question and scope + +The [four intermittent wake failures](./645-p3-resume-refusal-investigation.md) +have a successful `/suspend` response followed by an AWS report that the +`/resume` connection was refused. The +[minimal listener control](./645-p3-listener-probe-20260916.md) reproduced that +message by deliberately closing its listener, but did not establish what happened +inside the full agent. + +This experiment runs the full image-4 agent under an independent parent observer. +The parent records child exit status, received signals, port-8080 listener +inodes/ownership, selected process-state fields and cgroup memory counters. +It samples `/proc` without opening connections. It records no environment, +command arguments, task payloads or credentials. + +The observer changes the process tree: the server becomes a child of the +diagnostic parent. That can affect signal handling and timing. Results narrow +the investigation; they do not prove an unmodified image is reliable. + +## Artifact and isolation + +The source archive is the exact production image-4 artifact, +SHA-256 `0585a80606aaa66e9ce4dbff35a093451488f72d5a732c6cce6036b88fbcd4ed`. +The diagnostic archive changes only the Dockerfile startup wrapper and adds +`agent/scripts/microvm_process_observer.py`; 109 other entries are byte-identical. +The child still runs the original `opentelemetry-instrument uvicorn server:app` +command with its original arguments. + +Diagnostic artifact SHA-256: +`fd0875173bc9f0cc7ec7b321923e355d6d5022e4eae6ea7b958f5e7e0f92bd57`. + +The first diagnostic build failed before any task launch: the added ZIP entry +defaulted to mode `0600`, so the non-root agent could not read the root-owned +observer script. Both validation processes reported `Permission denied`; +the image eventually failed its five-minute readiness timeout. The corrected +archive explicitly uses regular-file mode `100644`. Failed image version `1.0` +and its logs remain part of the evidence; the corrected build is version `2.0`. + +The temporary image is `backgroundagent-dev-p3-observer-20260916`, in account +``, `us-west-2`. It uses the same managed base, architecture, memory, +six hooks and restricted network connectors. It has its own build role, artifact +and log group, with ownership tag `abca:verification=645-p3-observer-20260916`. +It reuses the existing worker execution role. No public ingress is added. + +The private durable fixture uses the current production coordinator compiled +from `c3b5c617`, fixed owned task IDs, one-second polls and a 1,800-second worker +lifetime cap. Its own SSM suspension switch is separate from production. +The fixtures ask only for one approved `Read` of `/etc/os-release`. + +Two ordinary approval cases are followed by a case that remains suspended for +six minutes before approval. The latter checks the +[pending-wake timer correction](./645-p3-pending-wake.md). A fourth fixed task ID +is used for the direct-wake, delayed-hook control below. Its results are recorded +separately from the fallback cases. + +## Observed results and direct-wake control + +The first three cases completed with the original worker and deadlines, +one successful `Read`, released capacity, zero user counters, empty launch +prefixes and observed worker termination. Independent parent observations +confirmed the child owned its listening socket; no child exit was logged. + +| Case | Worker | Durable polls | Owned-listener observations | +|---|---|---|---| +| Short approval A | `microvm-7897f87c-c612-3b0f-9958-3f49f9c1f0f2` | 59 | 41 | +| Short approval B | `microvm-af89965a-5607-3550-980d-900bd36c4a87` | 62 | 41 | +| Six-minute suspension | `microvm-e487702f-48d8-3a7a-93e0-93ba4ea573c9` | 363 | 24 | + +These are **fallback-wake** results. The production approval handler's intended +image-scoped permissions reject `GetMicrovm` on the separate diagnostic image. +It committed the approval and wake intent, returned 202, and the private +coordinator performed Resume. The short cases reused the same client port across +`/suspend` and `/resume`; this is a transport difference from the minimal probe. +Neither result establishes a cause for the four original failures. + +The fourth fixed task uses a private copy of the production approval handler +with a dedicated role restricted to that task and the diagnostic image. Image +version `3.0` adds an explicit five-second delay before entering `/resume`. +This delay exists only in the diagnostic archive, whose SHA-256 is +`295fd2cee914ec56ee0ebef47d44df9fb9fc9f530a3baefc7da64826f791bd85`. +The production image and permissions are unchanged. + +The private coordinator's version `3` includes the startup-confirmation +follow-up. Four real initial `PENDING` polls already confirmed the original +startup clock was retained despite the task being `RUNNING`. The direct-wake +result after six minutes remains pending. + +## Local checks and evidence + +Ruff lint/format and Ty passed. Real subprocess checks preserved exit code 42 and +forwarded SIGTERM to the child, reporting its termination and exiting 143. +Live Linux validation confirmed child/listener visibility. The guest does not +expose `/sys/fs/cgroup/memory.events`, so no OOM-counter evidence is claimed. + +Private inputs, ownership ledgers, request IDs and results are recorded under +`/tmp/abca-645-p2-clean-20260913/p3-process-observer-20260916`. Raw evidence stays +private because the surrounding task history may contain signed launch references. +Every owned worker and temporary resource must be accounted for and removed +before closing the experiment. Production automatic suspension remains off. diff --git a/docs/verification/645-p3-supervisor.md b/docs/verification/645-p3-supervisor.md index c22287b75..b1dbb09d3 100644 --- a/docs/verification/645-p3-supervisor.md +++ b/docs/verification/645-p3-supervisor.md @@ -1,9 +1,10 @@ # ADR-021 P3: supervisor and approval wake -Status (2026-09-15): production integration and full repository validation pass. -The [development deployment](./645-p3-live-deployment-20260915.md) now runs the -supervisor and six-hook image `3.0`; initial isolated guest checks -pass. Automatic suspension remains off. The [P3 plan](./645-p3-implementation-plan.md) +Status (2026-09-16): production integration and full repository validation pass. +The development deployment now runs coordinator version `4`, including the +[pending-wake correction](./645-p3-pending-wake.md), and agent image `4.0`, +including the [approval callback correction](./645-p3-callback-timeout.md). +Automatic suspension remains off. The [P3 plan](./645-p3-implementation-plan.md) retains the remaining live acceptance gates and P2 checks. ## What this does @@ -57,6 +58,14 @@ and next delay. Before verified service timing is available, a fixed deadline from the first durable observation bounds supervision and new sleep is disabled. Faster polls do not consume the older 1,020-attempt limit used by other backends. +AWS can report `PENDING` while restoring a suspended worker. A matching saved +wake instruction starts wake recovery at that instruction's original request +time. The local follow-up also preserves the startup clock until an actual AWS +observation confirms startup, because the coordinator can mark a task `RUNNING` +earlier. Another unexpected pending observation gets bounded uncertainty +recovery. Existing recovery clocks are preserved. See the linked correction +record for the distinction between deployed version 4 and this follow-up. + Control calls and database recovery reads share the caller's AbortSignal, a cancellation notice. An expired operation cannot start another request with a fresh independent budget. From 78c79a6be1cf769781f5b9b57495570bf0186776 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Wed, 16 Sep 2026 09:15:26 -0400 Subject: [PATCH 046/149] docs: record verified MicroVM wake timers (#645) --- cdk/src/handlers/approve-task.ts | 8 +- .../645-p3-implementation-plan.md | 9 +- docs/verification/645-p3-pending-wake.md | 27 ++-- .../645-p3-process-observer-20260916.md | 115 +++++++++++++++--- .../645-p3-resume-refusal-investigation.md | 17 ++- docs/verification/645-p3-supervisor.md | 9 +- 6 files changed, 147 insertions(+), 38 deletions(-) diff --git a/cdk/src/handlers/approve-task.ts b/cdk/src/handlers/approve-task.ts index 9449c3df4..043f3fb68 100644 --- a/cdk/src/handlers/approve-task.ts +++ b/cdk/src/handlers/approve-task.ts @@ -127,10 +127,10 @@ export async function handler( const nowIso = new Date().toISOString(); const nowEpoch = Math.floor(Date.now() / 1000); - // 3. Per-user per-minute rate limit. Uses a synthetic row in the - // approvals table keyed on `RATE##MINUTE#` - // so the existing grantReadWriteData wiring carries forward; TTL - // reaps the counter after ~120s. + // 3. Per-user per-minute rate limit, shared with deny. The approvals-table + // partition key is RATE##APPROVE; the sort key is MINUTE#. + // TTL makes old counters eligible for eventual cleanup, not deletion at + // an exact time. Each minute uses its own counter regardless of that delay. const minuteBucket = formatMinuteBucket(new Date()); try { await ddb.send(new UpdateCommand({ diff --git a/docs/verification/645-p3-implementation-plan.md b/docs/verification/645-p3-implementation-plan.md index 5c11cd8c8..31821daaa 100644 --- a/docs/verification/645-p3-implementation-plan.md +++ b/docs/verification/645-p3-implementation-plan.md @@ -231,8 +231,13 @@ handoff, use this order: do not close this gate. The [minimal listener experiment](./645-p3-listener-probe-20260916.md) also exposed a separate [pending-wake timer bug](./645-p3-pending-wake.md). - Its correction passed the full build and is deployed in coordinator version 4; - it needs a live coordinator acceptance check. + Its correction and startup-confirmation follow-up are deployed in coordinator + version 5. Local checks, real first-start observations and the exact API-issued + old-worker `PENDING` branch pass in the + [full-agent observer experiment](./645-p3-process-observer-20260916.md). + Its seven workers and temporary infrastructure were removed, and the exact + evidence was privately archived. These timer results do not explain the + separate connection refusals. 2. Complete the remaining live race/fault matrix: cancellation during transitions, late decision races, repeated polling/credential-refresh failures, durable registration races, service token-retention expiry and recovery of a worker diff --git a/docs/verification/645-p3-pending-wake.md b/docs/verification/645-p3-pending-wake.md index 5c5e13631..caff724ba 100644 --- a/docs/verification/645-p3-pending-wake.md +++ b/docs/verification/645-p3-pending-wake.md @@ -1,7 +1,7 @@ # ADR-021 P3: distinguish a pending wake from initial startup -Date: 2026-09-16. The correction is deployed in coordinator version 4; live -coordinator acceptance remains pending. +Date: 2026-09-16. The pending-wake correction and startup follow-up are deployed +in coordinator version 5. The exact direct pending-wake branch passed in AWS. ## Trigger @@ -76,8 +76,12 @@ Python, CLI, lint, types, contracts, documentation and synthesis passed. The successful west-region synthesis was repeated after the same disk-space issue. Four real first-start `PENDING` polls in the private coordinator also retained -the original startup clock. Production deployment of this follow-up is pending; -it changes no IAM permission or agent hook. +the original startup clock. The follow-up was deployed from `27521f85` and +verified at 12:39:13 UTC: live coordinator version `5`, code SHA-256 +`Y8SVypD7E959O6rHOY64D2m96Schdw9gb9urhQl3YWg=`. +Version 4 is retained. The environment, agent image 4.0, guardrail version 3 +and both disabled suspension switches are unchanged. This changes no IAM +permission or agent hook. ## Validation and remaining scope @@ -91,8 +95,8 @@ The full repository build passed in 488.56 seconds: 4,999 CDK tests passed with skipped; all 928 CLI tests passed. Compilation, lint, type checks, contract/drift checks, synthesis and the 77-page documentation build also passed. -The narrow deployment completed at 11:48:07 UTC in `sphia-dev`, `us-west-2`. -The live alias now selects coordinator version 4, with code SHA-256 +The first narrow deployment completed at 11:48:07 UTC in `sphia-dev`, `us-west-2`. +It selected coordinator version 4, with code SHA-256 `1aITI4Dx3HdUMuAn9DnVLphZTpX6s4jyn+Odoz/N1Rs=`. Version 3 remains retained. The reviewed change set modified only coordinator code/version/alias resources. Resolved environment values, guardrail version 3 and agent image 4.0 are unchanged. @@ -103,7 +107,14 @@ the private persistent archive `/Users/sphias/.local/share/abca-verification/645-p3-20260916/listener-pending-evidence.tar.gz` (SHA-256 `221dffad6be148b296e95d89b501f6489db9038416d482cb0753c78465d59994`). -A live check of the updated coordinator remains pending. This correction does -not explain or resolve the four separate +The [full-agent observer's final timing control](./645-p3-process-observer-20260916.md) +passed on task `01M2N507M71T8F18ETT32Q51M2`. At 13:11:01.153Z, the production +supervisor observed real `PENDING` with no previous recovery on a worker older +than 300 seconds, and initialized wake recovery from the saved API request time. +The same worker completed, preserving its original lifetime and approval clocks, +then terminated with normal capacity/payload cleanup. Private scheduling controls +made that interleaving observable; stored states and clocks were not fabricated. + +This correction does not explain or resolve the four separate [resume-hook connection refusals](./645-p3-resume-refusal-investigation.md). Production automatic suspension remains off. diff --git a/docs/verification/645-p3-process-observer-20260916.md b/docs/verification/645-p3-process-observer-20260916.md index cd2544df6..1fe48622c 100644 --- a/docs/verification/645-p3-process-observer-20260916.md +++ b/docs/verification/645-p3-process-observer-20260916.md @@ -1,7 +1,8 @@ # ADR-021 P3: full-agent process observer -Date: 2026-09-16. Three fallback-wake cases passed. The direct-wake control and -resource cleanup remain in progress. +Date: 2026-09-16. Six task cases passed, including the exact API-first pending-wake +check. One test-role configuration failure is preserved separately. Temporary +resource cleanup and private evidence archiving are complete. ## Question and scope @@ -47,16 +48,17 @@ six hooks and restricted network connectors. It has its own build role, artifact and log group, with ownership tag `abca:verification=645-p3-observer-20260916`. It reuses the existing worker execution role. No public ingress is added. -The private durable fixture uses the current production coordinator compiled -from `c3b5c617`, fixed owned task IDs, one-second polls and a 1,800-second worker -lifetime cap. Its own SSM suspension switch is separate from production. +The initial private durable fixture used production coordinator source +`c3b5c617`; private versions `3` through `6` use the startup-confirmation fix +in `27521f85`. All use fixed owned task IDs, one-second polls and a 1,800-second +worker lifetime cap. The fixture's SSM suspension switch is separate from production. The fixtures ask only for one approved `Read` of `/etc/os-release`. Two ordinary approval cases are followed by a case that remains suspended for six minutes before approval. The latter checks the -[pending-wake timer correction](./645-p3-pending-wake.md). A fourth fixed task ID -is used for the direct-wake, delayed-hook control below. Its results are recorded -separately from the fallback cases. +[pending-wake timer correction](./645-p3-pending-wake.md). Four further fixed task +IDs cover direct wake, a corrected test-role policy, and API-first timing below. +Their results are recorded separately from the fallback cases. ## Observed results and direct-wake control @@ -78,7 +80,7 @@ coordinator performed Resume. The short cases reused the same client port across `/suspend` and `/resume`; this is a transport difference from the minimal probe. Neither result establishes a cause for the four original failures. -The fourth fixed task uses a private copy of the production approval handler +The direct-wake control uses a private copy of the production approval handler with a dedicated role restricted to that task and the diagnostic image. Image version `3.0` adds an explicit five-second delay before entering `/resume`. This delay exists only in the diagnostic archive, whose SHA-256 is @@ -87,8 +89,68 @@ The production image and permissions are unchanged. The private coordinator's version `3` includes the startup-confirmation follow-up. Four real initial `PENDING` polls already confirmed the original -startup clock was retained despite the task being `RUNNING`. The direct-wake -result after six minutes remains pending. +startup clock was retained despite the task being `RUNNING`. + +The fourth task, `01M2N176C2X2M85VMAA415P4JX`, failed before approval: the private +API role lacked access to its user's synthetic rate-limit counter. It returned +500, and the watcher stopped the execution, cancelled the task, terminated +`microvm-625dca00-bc29-36c6-9a63-d3f348accfcb`, deleted launch files and released +capacity. This is a test-role configuration failure, not another resume refusal. + +The corrected policy grants `UpdateItem` on the exact owned +`RATE##APPROVE` partition. A preflight against a nonexistent approval +returned the expected 404 and verified counter value 1, without creating a task +or approval. A fresh fifth task, `01M2N3H3JVAFCK8JXGN44H6KVA`, uses private +coordinator version `4` and private approval-handler version `2`, with the same +diagnostic image `3.0`. + +The fifth case completed successfully on +`microvm-d996d527-64b5-3cb1-9f88-0a143fac9f05`. The private API acknowledged its +single Resume at 12:46:29.760Z, request +`eb6ea613-7cb6-4b62-8e84-d53828a32a31`. The coordinator first observed `SUSPENDED` +with the resume intent and initialized recovery at 12:46:24.145Z; five subsequent +real `PENDING` observations preserved that recovery clock. The task completed, +the worker terminated, and normal coordinator cleanup passed. + +This verifies direct API wake with an already-started recovery timer. The +stricter check for first observing `PENDING` with **no prior recovery** failed, +so that original failed audit remains in the evidence. Private API logs use +Lambda's text prefix; the audit was corrected to decode that prefix before +counting the actual Resume acknowledgment. + +The sixth task, `01M2N4DVR9VPEARNP0DK85PZZY`, used private coordinator version `5` +and approval-handler version `3`. It delayed completion of lifecycle snapshot +reads for at most three seconds after a matching resume intent existed with no +prior recovery. This still allowed the coordinator to win the intent write +before the delay began. The API correctly deferred with `wake-intent/intent-stale`; +the coordinator resumed the worker and completed normal cleanup. Worker +`microvm-a9d311ac-6f2b-3801-a579-71388cf3a7be` terminated at 13:01:31.203Z. +The audit recorded 382 supervisor polls and 32 owned-listener observations. +This is a successful stale-intent fallback, with the failed exact-branch audit +preserved separately. + +A seventh task, `01M2N507M71T8F18ETT32Q51M2`, uses private coordinator version `6` +and approval-handler version `4`. Its wrapper delays only the first lifecycle +snapshot read per supervisor poll when the real approval is already `APPROVED` +or a matching resume intent exists, with no prior recovery. This places the +delay before the coordinator can write a competing intent. It polls actual +`GetMicrovm` for at most three seconds until `PENDING`, then returns a fresh +database snapshot to the production supervisor. Stored states and clocks are +unchanged. The diagnostic guest still delays `/resume` entry for five seconds. +The seventh case passed the exact branch at 13:11:01.153Z: actual `PENDING`, +no prior recovery, and worker age 418,860 ms, beyond the 300-second startup +allowance. Recovery began at `1789564260293`, exactly the saved API wake-request +time. The sole private API Resume acknowledgment was at 13:11:00.823Z, request +`635ac95d-7e3e-49cf-a5b9-25566f45c3eb`. The guest's real delay lasted just over +five seconds. + +Worker `microvm-75c3deba-6819-35ec-8f33-a3b98f1f05ad` completed the single Read +and terminated at 13:11:13.063Z. The audit counted 381 supervisor polls and +32 owned-listener observations. The initial observation time and original +service deadline remained unchanged through durable replay; capacity was +released, the user counter reached zero, and launch files were absent without +watcher repair. This closes the deployed timer correction's live branch check. +It does not resolve the four original connection refusals. ## Local checks and evidence @@ -97,8 +159,29 @@ forwarded SIGTERM to the child, reporting its termination and exiting 143. Live Linux validation confirmed child/listener visibility. The guest does not expose `/sys/fs/cgroup/memory.events`, so no OOM-counter evidence is claimed. -Private inputs, ownership ledgers, request IDs and results are recorded under -`/tmp/abca-645-p2-clean-20260913/p3-process-observer-20260916`. Raw evidence stays -private because the surrounding task history may contain signed launch references. -Every owned worker and temporary resource must be accounted for and removed -before closing the experiment. Production automatic suspension remains off. +Cleanup was verified at 13:12:44 UTC. All seven workers were terminated; +six tasks completed and the test-role failure was cancelled. All reservations +were released and launch prefixes were empty. The private coordinator and +approval function, every published version, all three diagnostic image versions, +three roles, three log groups, the private suspension parameter and three +diagnostic artifact objects were removed, with absence checks. + +The seven fake users' zero-valued capacity counters had no TTL. Their snapshots +were archived, then each row was deleted conditionally on its unchanged +reservation version and zero count; subsequent reads confirmed absence. +Task history, rate-limit rows and traces retain their normal retention policies. + +Private inputs, ownership ledgers, request IDs and results originated under +`/tmp/abca-645-p2-clean-20260913/p3-process-observer-20260916`. The persistent +archive is +`/Users/sphias/.local/share/abca-verification/645-p3-20260916/observer-startup-evidence.tar.gz`: +172,444,846 bytes, 222 entries, SHA-256 +`d5103c683c16fdc8f63504c2cded5ccec5dc435ed2bf8a8d8d5b37bbdc8be7ec`. +It includes the failed cases, 5,144 image-log events, 8,775 coordinator-log events, +39 private-API log events, durable histories and the exact deployed coordinator +ZIP. Extracting that ZIP reproduced its published code hash. The archive and its +directory are owner-only because raw histories may contain signed launch references. + +The final production check confirmed `UPDATE_COMPLETE`, coordinator version `5`, +active agent image `4.0`, guardrail version `3` and both suspension switches off. +The original resume refusals and the wider P3 acceptance gates remain open. diff --git a/docs/verification/645-p3-resume-refusal-investigation.md b/docs/verification/645-p3-resume-refusal-investigation.md index e74ca4709..ea5629b5a 100644 --- a/docs/verification/645-p3-resume-refusal-investigation.md +++ b/docs/verification/645-p3-resume-refusal-investigation.md @@ -100,15 +100,24 @@ That experiment also exposed a separate [pending-wake timer bug](./645-p3-pendin in the supervisor. Its correction does not account for these four failures, whose worker termination reason was the service-reported connection refusal. +The [full-agent observer experiment](./645-p3-process-observer-20260916.md) +adds the independent parent and listener sampling described below. Its completed +fallback and direct API wakes retained a healthy child-owned listener, without +reproducing the refusal. Short full-agent cases reused the same client port for +`/suspend` and `/resume`, unlike the minimal listener's closed connections. +That is an observed transport difference, not an established cause. None of +the four original failures has independent process/listener evidence at restore, +and the guest does not expose the cgroup OOM counters sampled by the observer. + 1. Use the recorded worker IDs, region, timestamps and Resume request IDs to inspect service-side lifecycle diagnostics. Determine the actual connection error and whether the request reached the guest, including any transport retry. 2. Correlate guest process/kernel health and the port-8080 listener at restoration. Application access logs do not provide that missing evidence. - A diagnostic full-agent image can add an independent parent-process observer - to record child exit status and listener health without changing approval - decisions or granting public ingress. Preserve the original image and compare - a fixed, bounded set of owned tasks. + The diagnostic parent now supplies these observations in successful controls; + a refusal must be captured with those observations to make the comparison. + Preserve the original image and use fixed, bounded owned cases if further + testing is justified by a specific transport or process hypothesis. 3. If a transport or application race is identified, make a bounded correction and test that trigger specifically. Do not hide a failed wake by silently launching another worker: the approved action and workspace may already have diff --git a/docs/verification/645-p3-supervisor.md b/docs/verification/645-p3-supervisor.md index b1dbb09d3..4de06550e 100644 --- a/docs/verification/645-p3-supervisor.md +++ b/docs/verification/645-p3-supervisor.md @@ -1,7 +1,7 @@ # ADR-021 P3: supervisor and approval wake Status (2026-09-16): production integration and full repository validation pass. -The development deployment now runs coordinator version `4`, including the +The development deployment now runs coordinator version `5`, including the [pending-wake correction](./645-p3-pending-wake.md), and agent image `4.0`, including the [approval callback correction](./645-p3-callback-timeout.md). Automatic suspension remains off. The [P3 plan](./645-p3-implementation-plan.md) @@ -60,11 +60,12 @@ Faster polls do not consume the older 1,020-attempt limit used by other backends AWS can report `PENDING` while restoring a suspended worker. A matching saved wake instruction starts wake recovery at that instruction's original request -time. The local follow-up also preserves the startup clock until an actual AWS +time. The startup follow-up also preserves the startup clock until an actual AWS observation confirms startup, because the coordinator can mark a task `RUNNING` earlier. Another unexpected pending observation gets bounded uncertainty -recovery. Existing recovery clocks are preserved. See the linked correction -record for the distinction between deployed version 4 and this follow-up. +recovery. Existing recovery clocks are preserved. Both corrections are deployed; +the exact API-issued pending-wake branch and initial-startup clock have live +evidence in the linked verification record. Control calls and database recovery reads share the caller's AbortSignal, a cancellation notice. An expired operation cannot start another request with a From 3fe740bbb1a4d95434b7338adfabb8c687fb0ab7 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Wed, 16 Sep 2026 09:26:23 -0400 Subject: [PATCH 047/149] docs: distinguish paused HTTP resets from wake refusals (#645) --- .../645-p3-resume-refusal-investigation.md | 6 ++ .../645-p3-transport-control-20260916.md | 81 +++++++++++++++++++ 2 files changed, 87 insertions(+) create mode 100644 docs/verification/645-p3-transport-control-20260916.md diff --git a/docs/verification/645-p3-resume-refusal-investigation.md b/docs/verification/645-p3-resume-refusal-investigation.md index ea5629b5a..f28ee39b5 100644 --- a/docs/verification/645-p3-resume-refusal-investigation.md +++ b/docs/verification/645-p3-resume-refusal-investigation.md @@ -109,6 +109,12 @@ That is an observed transport difference, not an established cause. None of the four original failures has independent process/listener evidence at restore, and the guest does not expose the cgroup OOM counters sampled by the observer. +The subsequent [local transport control](./645-p3-transport-control-20260916.md) +reproduced a reset on an old connection after a six-second process pause, while +all eight fresh-connection checks succeeded and the servers remained alive. +This is not a reproduction of the AWS refusal. It supplies a specific comparison +for the next cloud investigation without establishing a production fix. + 1. Use the recorded worker IDs, region, timestamps and Resume request IDs to inspect service-side lifecycle diagnostics. Determine the actual connection error and whether the request reached the guest, including any transport retry. diff --git a/docs/verification/645-p3-transport-control-20260916.md b/docs/verification/645-p3-transport-control-20260916.md new file mode 100644 index 000000000..6b86307af --- /dev/null +++ b/docs/verification/645-p3-transport-control-20260916.md @@ -0,0 +1,81 @@ +# ADR-021 P3: paused-server HTTP transport control + +Date: 2026-09-16. Local diagnostic completed; the original AWS resume refusal +was **not reproduced**. No application code or AWS deployment changed. + +## Question + +The full-agent hooks sometimes reuse one HTTP connection across suspension. +Could an old connection fail after a pause even though the server still accepts +new connections? This is a narrower question than the four recorded +[AWS resume refusals](./645-p3-resume-refusal-investigation.md). + +The earlier full-agent observer measured a 363.501-second wall-clock gap and +363.507-second monotonic-clock gap across one long suspension, with the child +still owning its listener afterward. Another long case measured 367.621 and +367.620 seconds respectively. Thus elapsed-time timers advanced during those +observed freezes. + +## Isolated experiment + +Eight cases ran in a disposable ARM64 Linux container, with no external network +or AWS credentials. It used Python 3.13.13, Uvicorn 0.50.0, FastAPI 0.139.0, +the asyncio loop and the h11 HTTP implementation. The cached image digest was +`sha256:37a88f276acd700dbd0d6d2a69eff2536180bde11620b6084220414d8d2b63f2`. + +A minimal FastAPI application acknowledged `/suspend` and `/resume`. The +controller stopped the server process with `SIGSTOP` for one or six seconds, +then continued it with `SIGCONT`. These signals pause and restart a process. +They do not perform an AWS MicroVM snapshot or restore its networking. + +Each duration was tested with the wake request queued before continuation or +sent just afterward. The comparison response included `Connection: close`, +which tells an HTTP client to open a new connection for its next request. +Every case also made a separate fresh connection and checked that the server +was still alive. + +| Response behavior | Pause | Request relative to continuation | Wake result | Fresh connection | +|---|---:|---|---|---| +| Default keep-alive | 1 s | After | 200 | 200 | +| Default keep-alive | 1 s | Queued before | 200 | 200 | +| Default keep-alive | 6 s | After | Connection reset, errno 104 | 200 | +| Default keep-alive | 6 s | Queued before | 200 | 200 | +| Connection close | 1 s | After | 200 | 200 | +| Connection close | 1 s | Queued before | 200 | 200 | +| Connection close | 6 s | After | 200 | 200 | +| Connection close | 6 s | Queued before | 200 | 200 | + +All eight server processes remained alive until explicit test cleanup. The +container exited successfully and its absence was verified. + +## Interpretation and next reproduction + +One reused connection reset after exceeding the server's five-second keep-alive +timeout. That is different from a new connection being refused: the listener +remained reachable in every case. This does not establish how AWS classifies +its underlying transport errors. Some original failures followed suspension by +less than five seconds, so simple keep-alive expiration does not explain all +four recorded failures. + +The next bounded AWS comparison should: + +1. Preserve the original server command and its position as the first process + in the guest. The previous parent observer changed that position. +2. Compare the deployed image with a diagnostic variant that changes only + lifecycle response connection handling. Preserve exact artifact hashes. +3. Use fresh owned task IDs for ordinary approval and supervisor-only wake. + Record AWS request IDs, actual worker states, hook entry/response times, + connection identity and available process/listener observations. +4. Stop and preserve evidence on a refusal. Determine whether the process died, + the listening socket disappeared, or a healthy listener was unreachable. +5. Apply a correction only when supported by the failing path, then demonstrate + the same trigger passing and recheck normal approval, denial, deadlines and + credential renewal. A few successful comparison runs alone do not prove a fix. +6. Remove only new owned resources. The deployed image and its existing roles + are comparison inputs and must not be deleted by fixture cleanup. + +This AWS comparison is planned, not executed by this local experiment. +Production automatic suspension remains off. + +Scripts, raw results and the clock comparison are retained privately under +`/Users/sphias/.local/share/abca-verification/645-p3-20260916/transport-control`. From bf6127da9a44d665880f06cdec642c6701504dfe Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Wed, 16 Sep 2026 10:27:19 -0400 Subject: [PATCH 048/149] feat: add correlated MicroVM lifecycle diagnostics (#645) --- agent/src/microvm_checkpoint.py | 48 ++--- agent/src/microvm_diagnostics.py | 139 ++++++++++++++ agent/src/microvm_http.py | 42 ++++- agent/src/microvm_lifecycle.py | 47 +++-- agent/tests/test_microvm_http.py | 162 +++++++++++++++- cdk/src/handlers/shared/compute-strategy.ts | 2 +- cdk/src/handlers/shared/error-classifier.ts | 32 +++- .../handlers/shared/microvm-approval-wake.ts | 26 ++- cdk/src/handlers/shared/microvm-control.ts | 10 +- cdk/src/handlers/shared/microvm-supervisor.ts | 38 ++++ .../strategies/lambda-microvm-strategy.ts | 33 +++- .../handlers/shared/error-classifier.test.ts | 30 ++- .../shared/microvm-approval-wake.test.ts | 18 ++ .../handlers/shared/microvm-control.test.ts | 48 +++++ .../shared/microvm-supervisor.test.ts | 34 ++++ .../handlers/shared/session-lifecycle.test.ts | 24 +++ .../645-p3-implementation-plan.md | 5 + .../645-p3-lifecycle-diagnostics.md | 173 ++++++++++++++++++ .../645-p3-resume-refusal-investigation.md | 6 + 19 files changed, 852 insertions(+), 65 deletions(-) create mode 100644 agent/src/microvm_diagnostics.py create mode 100644 cdk/test/handlers/shared/microvm-control.test.ts create mode 100644 docs/verification/645-p3-lifecycle-diagnostics.md diff --git a/agent/src/microvm_checkpoint.py b/agent/src/microvm_checkpoint.py index 7a42982e9..95e2e06f6 100644 --- a/agent/src/microvm_checkpoint.py +++ b/agent/src/microvm_checkpoint.py @@ -10,6 +10,7 @@ from decimal import Decimal from typing import TYPE_CHECKING, Any, Literal +from microvm_diagnostics import lifecycle_stage from microvm_lifecycle import ApprovalPark, LifecycleUnavailable from progress_writer import _ProgressWriter @@ -202,23 +203,25 @@ def _transaction_checks( def checkpoint_before_suspend(park: ApprovalPark) -> None: """Commit a marker only if the same task, sleep intent and pending gate hold.""" record, _ = _record(park) - client, task_table, approvals_table, intent, deadline_ms = _read(park, "suspend") + with lifecycle_stage("checkpoint-identity-read"): + client, task_table, approvals_table, intent, deadline_ms = _read(park, "suspend") if park.deadline.remaining_s() <= 0: raise LifecycleUnavailable("Approval deadline elapsed before checkpoint") # Never reuse an old transaction client token across HTTP requests: cached # success must not bypass conditions after a concurrent approval/cancellation. - _ProgressWriter( - park.task_id, user_id=record.user_id, repo=record.repo - ).write_microvm_checkpoint( - client=client, - condition_checks=_transaction_checks(park, task_table, approvals_table, intent), - metadata={ - "request_id": park.request_id, - "microvm_id": park.microvm_id, - "generation": intent["generation"], - "approval_deadline_ms": deadline_ms, - }, - ) + with lifecycle_stage("checkpoint-transaction"): + _ProgressWriter( + park.task_id, user_id=record.user_id, repo=record.repo + ).write_microvm_checkpoint( + client=client, + condition_checks=_transaction_checks(park, task_table, approvals_table, intent), + metadata={ + "request_id": park.request_id, + "microvm_id": park.microvm_id, + "generation": intent["generation"], + "approval_deadline_ms": deadline_ms, + }, + ) def refresh_and_reconcile_after_resume(park: ApprovalPark) -> None: @@ -226,17 +229,20 @@ def refresh_and_reconcile_after_resume(park: ApprovalPark) -> None: from aws_session import refresh_microvm_credentials _record(park) - refresh_microvm_credentials(park.task_id) - client, task_table, approvals_table, intent, _ = _read(park, "resume") + with lifecycle_stage("credential-refresh"): + refresh_microvm_credentials(park.task_id) + with lifecycle_stage("resume-identity-read"): + client, task_table, approvals_table, intent, _ = _read(park, "resume") # This transaction writes no task/approval state. It acknowledges that both # identities still hold, including cancellation or intent changes after reads. # A concurrent valid decision is allowed; the original loop observes it. - client.transact_write_items( - TransactItems=[ - {"ConditionCheck": check} - for check in _transaction_checks(park, task_table, approvals_table, intent) - ] - ) + with lifecycle_stage("resume-identity-transaction"): + client.transact_write_items( + TransactItems=[ + {"ConditionCheck": check} + for check in _transaction_checks(park, task_table, approvals_table, intent) + ] + ) # The existing approval loop checks its original stopwatch/UTC cap as soon # as the barrier opens. Expiry enters its timeout/late-decision path; a timely # approval already recorded must still win that conditional race. diff --git a/agent/src/microvm_diagnostics.py b/agent/src/microvm_diagnostics.py new file mode 100644 index 000000000..20aa19ebe --- /dev/null +++ b/agent/src/microvm_diagnostics.py @@ -0,0 +1,139 @@ +# Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. +# SPDX-License-Identifier: MIT-0 + +"""Bounded, payload-free lifecycle diagnostics shared with callback threads.""" + +from __future__ import annotations + +import json +import os +import re +import threading +import time +import uuid +from contextlib import contextmanager +from contextvars import ContextVar +from typing import TYPE_CHECKING, Any + +if TYPE_CHECKING: + from collections.abc import Callable + +_ACTIVE: ContextVar[HookDiagnostics | None] = ContextVar("microvm_hook_diagnostics", default=None) +_IDENTIFIER = re.compile(r"[A-Za-z0-9_-]{1,128}\Z") +_OUTPUT_LOCK = threading.Lock() + + +def _error_identity(error: BaseException | None) -> dict[str, str]: + if error is None: + return {} + name = type(error).__name__ + identity = {"error_type": name if _IDENTIFIER.fullmatch(name) else "Error"} + response = getattr(error, "response", None) + if isinstance(response, dict): + for source, field, target in [ + ("Error", "Code", "aws_error_code"), + ("ResponseMetadata", "RequestId", "aws_request_id"), + ]: + section = response.get(source) + value = section.get(field) if isinstance(section, dict) else None + if isinstance(value, str) and _IDENTIFIER.fullmatch(value): + identity[target] = value + return identity + + +class HookDiagnostics: + """One HTTP invocation; copied thread contexts retain this same correlation id. + + Logging cannot authorize a transition. Callback threads may outlive a timeout; + their subsequent records are marked late, never a successful HTTP acknowledgment. + """ + + def __init__(self, action: str, snapshot: Callable[[], dict[str, Any]]) -> None: + self.action = action + self.hook_id = str(uuid.uuid4()) + self._snapshot = snapshot + self._started = time.monotonic() + self._stage = "body-read" + self._closed = False + self._lock = threading.Lock() + + def emit(self, event: str, **fields: Any) -> None: + # Snapshot only safe local facts. Never include request bodies, SDK messages, + # approval records, tool arguments, credentials, or exception tracebacks. + try: + snapshot = self._snapshot() + safe = { + key: value + for key, value in snapshot.items() + if not isinstance(value, str) or _IDENTIFIER.fullmatch(value) + } + with self._lock: + record = { + **safe, + "event": event, + "level": "WARN" if event.endswith("_failed") else "INFO", + "action": self.action, + "hook_id": self.hook_id, + "pid": os.getpid(), + "timestamp_ms": int(time.time() * 1000), + "elapsed_ms": round((time.monotonic() - self._started) * 1000), + "stage": self._stage, + "late": self._closed, + **fields, + } + # Concurrent hooks and their callback threads must not interleave + # the JSON and newline writes of these records. + with _OUTPUT_LOCK: + print(json.dumps(record), flush=True) + except Exception: # noqa: S110 — failure in the logging sink cannot safely log to itself. + # Diagnostics are best effort; a broken stdout must not change the + # response or bypass the controller's generation/barrier checks. + pass + + def stage(self, name: str) -> None: + with self._lock: + if not self._closed: + self._stage = name + self.emit("microvm_hook_stage", callback_stage=name) + + def finish(self, status: int | None, code: str, error: BaseException | None = None) -> None: + with self._lock: + self._closed = True + self.emit( + "microvm_hook_finished", + http_status=status, + code=code, + late=False, + level="INFO" if code == "acknowledged" else "WARN", + **_error_identity(error), + ) + + +@contextmanager +def hook_diagnostics(action: str, snapshot: Callable[[], dict[str, Any]]): + diagnostics = HookDiagnostics(action, snapshot) + token = _ACTIVE.set(diagnostics) + try: + diagnostics.emit("microvm_hook_started") + yield diagnostics + finally: + _ACTIVE.reset(token) + + +@contextmanager +def lifecycle_stage(name: str): + """Record entry before potentially blocking work, including inside to_thread.""" + diagnostics = _ACTIVE.get() + if diagnostics: + diagnostics.stage(name) + try: + yield + except BaseException as error: + if diagnostics: + diagnostics.emit( + "microvm_hook_stage_failed", callback_stage=name, **_error_identity(error) + ) + raise + else: + if diagnostics: + diagnostics.emit("microvm_hook_stage_finished", callback_stage=name) diff --git a/agent/src/microvm_http.py b/agent/src/microvm_http.py index d51210a65..3ac2c426f 100644 --- a/agent/src/microvm_http.py +++ b/agent/src/microvm_http.py @@ -14,6 +14,7 @@ from fastapi.responses import JSONResponse from microvm_checkpoint import checkpoint_before_suspend, refresh_and_reconcile_after_resume +from microvm_diagnostics import hook_diagnostics, lifecycle_stage from microvm_lifecycle import LifecycleUnavailable, get_registered_context from shared_constants import SHARED_CONSTANTS @@ -51,22 +52,43 @@ async def _microvm_id(request: Request) -> str: async def _transition(request: Request, action: Literal["suspend", "resume"]) -> JSONResponse: + lifecycle = get_registered_context() + with hook_diagnostics( + action, lifecycle.diagnostic_snapshot if lifecycle else dict + ) as diagnostics: + try: + response = await _handle_transition(request, action) + except asyncio.CancelledError as exc: + diagnostics.finish(None, "MICROVM_LIFECYCLE_CANCELLED", exc) + raise + diagnostics.finish( + response.status_code, json.loads(bytes(response.body)).get("code", "acknowledged") + ) + return response + + +async def _handle_transition( + request: Request, action: Literal["suspend", "resume"] +) -> JSONResponse: try: end = time.monotonic() + LIFECYCLE_HANDLER_BUDGET_S async with asyncio.timeout(LIFECYCLE_HANDLER_BUDGET_S): - microvm_id = await _microvm_id(request) - lifecycle = get_registered_context() - if lifecycle is None or (microvm_id and microvm_id != lifecycle.microvm_id): - raise LifecycleUnavailable("No matching MicroVM task is registered") + with lifecycle_stage("body-read"): + microvm_id = await _microvm_id(request) + with lifecycle_stage("identity-check"): + lifecycle = get_registered_context() + if lifecycle is None or (microvm_id and microvm_id != lifecycle.microvm_id): + raise LifecycleUnavailable("No matching MicroVM task is registered") remaining = end - time.monotonic() if remaining <= 0: raise TimeoutError("Lifecycle body consumed its budget") - if action == "suspend": - park = await lifecycle.suspend(checkpoint_before_suspend, budget_s=remaining) - else: - park = await lifecycle.resume( - refresh_and_reconcile_after_resume, budget_s=remaining - ) + with lifecycle_stage("controller"): + if action == "suspend": + park = await lifecycle.suspend(checkpoint_before_suspend, budget_s=remaining) + else: + park = await lifecycle.resume( + refresh_and_reconcile_after_resume, budget_s=remaining + ) return JSONResponse( content={ "status": "acknowledged", diff --git a/agent/src/microvm_lifecycle.py b/agent/src/microvm_lifecycle.py index a0dd40c58..c40447134 100644 --- a/agent/src/microvm_lifecycle.py +++ b/agent/src/microvm_lifecycle.py @@ -26,6 +26,8 @@ from dataclasses import dataclass from typing import TYPE_CHECKING, Protocol +from microvm_diagnostics import lifecycle_stage + if TYPE_CHECKING: from collections.abc import Callable @@ -92,6 +94,22 @@ def _check_open(self) -> bool: raise LifecycleUnavailable("Lifecycle barrier is closed") return self._phase in {"active", "parked"} + def diagnostic_snapshot(self) -> dict: + """Read only local state; never include approval contents or tool inputs.""" + with self._lock: + park = self._park or self._last_resume_park + return { + "task_id": self.task_id, + "microvm_id": self.microvm_id, + "request_id": park.request_id if park else None, + "phase": self._phase, + "local_generation": self._generation, + "active_tools": len(self._tools), + "active_activities": self._activities, + "progress_failed": self._progress_failed, + "suspend_ineligible": self._suspend_ineligible, + } + async def wait_until_open(self) -> None: # A thread-safe local predicate works across the server and pipeline's # separate event loops. No AWS calls or new approval timeout are needed. @@ -246,18 +264,20 @@ async def suspend( self._generation += 1 generation = self._generation try: - while True: - with self._lock: - self._assert_transition(generation, "suspending") - if self._progress_failed: - raise LifecycleUnavailable("Progress was not acknowledged") - drained = self._activities == 0 - if drained: - break - if time.monotonic() >= end: - raise TimeoutError("Progress did not drain within lifecycle budget") - await asyncio.sleep(min(0.02, max(0, end - time.monotonic()))) - await self._run_bounded(checkpoint, park, end) + with lifecycle_stage("activity-drain"): + while True: + with self._lock: + self._assert_transition(generation, "suspending") + if self._progress_failed: + raise LifecycleUnavailable("Progress was not acknowledged") + drained = self._activities == 0 + if drained: + break + if time.monotonic() >= end: + raise TimeoutError("Progress did not drain within lifecycle budget") + await asyncio.sleep(min(0.02, max(0, end - time.monotonic()))) + with lifecycle_stage("checkpoint"): + await self._run_bounded(checkpoint, park, end) with self._lock: self._assert_transition(generation, "suspending") if self._progress_failed or park.deadline.remaining_s() <= 0: @@ -297,7 +317,8 @@ async def resume( self._generation += 1 generation = self._generation try: - await self._run_bounded(refresh_and_reconcile, park, end) + with lifecycle_stage("refresh-and-reconcile"): + await self._run_bounded(refresh_and_reconcile, park, end) reseed_random() with self._lock: self._assert_transition(generation, "resuming") diff --git a/agent/tests/test_microvm_http.py b/agent/tests/test_microvm_http.py index 350a8f154..753cc63bb 100644 --- a/agent/tests/test_microvm_http.py +++ b/agent/tests/test_microvm_http.py @@ -6,6 +6,7 @@ from __future__ import annotations import asyncio +import json import threading import time from dataclasses import dataclass @@ -13,15 +14,25 @@ import httpx import pytest +from botocore.exceptions import ClientError from fastapi import Request import microvm_http import microvm_lifecycle as lifecycle import server +from microvm_diagnostics import lifecycle_stage PREFIX = server.MICROVM_HOOK_PREFIX +def diagnostic_records(capsys): + return [ + json.loads(line) + for line in capsys.readouterr().out.splitlines() + if line.startswith("{") and '"event": "microvm_hook_' in line + ] + + @dataclass class Deadline: remaining: float = 60 @@ -68,6 +79,115 @@ async def park(context): @pytest.mark.anyio class TestLifecycleHttp: + async def test_hooks_have_distinct_correlated_timelines( + self, context, callbacks, client, capsys + ): + await park(context) + for action in ["suspend", "resume"]: + response = await client.post(PREFIX + "/" + action, json={}) + assert response.status_code == 200 + records = diagnostic_records(capsys) + starts = [row for row in records if row["event"] == "microvm_hook_started"] + ends = [row for row in records if row["event"] == "microvm_hook_finished"] + assert len(starts) == len(ends) == 2 + assert starts[0]["hook_id"] != starts[1]["hook_id"] + for start, end in zip(starts, ends, strict=True): + assert start["hook_id"] == end["hook_id"] + assert end["task_id"] == "http-task" + assert end["microvm_id"] == "http-vm" + assert end["request_id"] == "gate" + assert end["pid"] > 0 + assert end["elapsed_ms"] >= 0 + assert end["http_status"] == 200 + assert end["code"] == "acknowledged" + assert end["late"] is False + assert ends[0]["phase"] == "suspend-ready" + assert ends[1]["phase"] == "parked" + + async def test_failed_refresh_logs_stage_and_aws_identity_without_secrets( + self, context, callbacks, client, capsys + ): + await park(context) + assert (await client.post(PREFIX + "/suspend", json={})).status_code == 200 + capsys.readouterr() + + def fail_refresh(_park): + with lifecycle_stage("credential-refresh"): + raise ClientError( + { + "Error": {"Code": "AccessDenied", "Message": "secret-credential"}, + "ResponseMetadata": { + "RequestId": "aws-request-123", + "HTTPHeaders": {"Authorization": "secret-header"}, + }, + }, + "AssumeRole", + ) + + callbacks[1].side_effect = fail_refresh + response = await client.post(PREFIX + "/resume", json={"ignored": "secret-body"}) + records = diagnostic_records(capsys) + serialized = json.dumps(records) + response.text + assert "secret-" not in serialized + failure = next(row for row in records if row["event"] == "microvm_hook_stage_failed") + assert failure["stage"] == "credential-refresh" + assert failure["error_type"] == "ClientError" + assert failure["aws_error_code"] == "AccessDenied" + assert failure["aws_request_id"] == "aws-request-123" + end = records[-1] + assert end["stage"] == "credential-refresh" + assert end["phase"] == "failed" + assert end["http_status"] == response.status_code == 503 + assert all(row["hook_id"] == end["hook_id"] for row in records) + with pytest.raises(lifecycle.LifecycleUnavailable): + await context.wait_until_open() + + async def test_logging_failure_does_not_change_hook_result( + self, context, callbacks, client, monkeypatch + ): + await park(context) + monkeypatch.setattr( + "microvm_diagnostics.print", MagicMock(side_effect=OSError("closed")), raising=False + ) + assert (await client.post(PREFIX + "/suspend", json={})).status_code == 200 + assert (await client.post(PREFIX + "/resume", json={})).status_code == 200 + await context.wait_until_open() + + async def test_timeout_reports_blocked_stage_and_marks_late_thread( + self, context, callbacks, client, monkeypatch, capsys + ): + await park(context) + assert (await client.post(PREFIX + "/suspend", json={})).status_code == 200 + capsys.readouterr() + release, finished = threading.Event(), threading.Event() + + def slow_refresh(_park): + try: + with lifecycle_stage("credential-refresh"): + assert release.wait(2) + finally: + finished.set() + + callbacks[1].side_effect = slow_refresh + monkeypatch.setattr(microvm_http, "LIFECYCLE_HANDLER_BUDGET_S", 0.05) + try: + response = await client.post(PREFIX + "/resume", json={}) + assert response.status_code == 503 + before = diagnostic_records(capsys) + end = before[-1] + assert end["stage"] == "credential-refresh" + assert end["code"] == "MICROVM_LIFECYCLE_TIMEOUT" + assert end["phase"] == "failed" + finally: + release.set() + assert await asyncio.to_thread(finished.wait, 2) + after = diagnostic_records(capsys) + assert after + assert all(row["late"] and row["hook_id"] == end["hook_id"] for row in after) + assert all(row["event"] != "microvm_hook_finished" for row in after) + with pytest.raises(lifecycle.LifecycleUnavailable): + await context.wait_until_open() + async def test_duplicate_hooks_acknowledge_without_repeating_work( self, context, callbacks, client ): @@ -240,7 +360,7 @@ def refresh(_park): await context.wait_until_open() async def test_concurrent_resume_reports_busy_while_first_finishes( - self, context, callbacks, client + self, context, callbacks, client, capsys ): await park(context) assert (await client.post(PREFIX + "/suspend", json={})).status_code == 200 @@ -259,8 +379,18 @@ def refresh(_park): release.set() assert (await request).status_code == 200 callbacks[1].assert_called_once() - - async def test_request_body_read_shares_the_total_budget(self, context, callbacks, monkeypatch): + records = [row for row in diagnostic_records(capsys) if row["action"] == "resume"] + ends = [row for row in records if row["event"] == "microvm_hook_finished"] + assert [row["http_status"] for row in ends] == [409, 200] + assert ends[0]["hook_id"] != ends[1]["hook_id"] + for end in ends: + matching = [row for row in records if row["hook_id"] == end["hook_id"]] + assert matching[0]["event"] == "microvm_hook_started" + assert matching[-1] == end + + async def test_request_body_read_shares_the_total_budget( + self, context, callbacks, monkeypatch, capsys + ): await park(context) monkeypatch.setattr(microvm_http, "LIFECYCLE_HANDLER_BUDGET_S", 0.02) @@ -273,6 +403,32 @@ async def slow_receive(): assert response.status_code == 503 for callback in callbacks: callback.assert_not_called() + end = diagnostic_records(capsys)[-1] + assert end["stage"] == "body-read" + assert end["code"] == "MICROVM_LIFECYCLE_TIMEOUT" + + async def test_cancelled_handler_logs_cancellation_and_still_propagates_it( + self, context, callbacks, capsys + ): + await park(context) + entered = asyncio.Event() + + async def receive(): + entered.set() + await asyncio.Event().wait() + + request = Request({"type": "http", "method": "POST", "headers": []}, receive) + pending = asyncio.create_task(microvm_http.microvm_suspend(request)) + await entered.wait() + pending.cancel() + with pytest.raises(asyncio.CancelledError): + await pending + end = diagnostic_records(capsys)[-1] + assert end["code"] == "MICROVM_LIFECYCLE_CANCELLED" + assert end["http_status"] is None + assert end["stage"] == "body-read" + for callback in callbacks: + callback.assert_not_called() async def test_terminate_closes_barrier_and_answers_even_if_body_stalls( self, context, monkeypatch diff --git a/cdk/src/handlers/shared/compute-strategy.ts b/cdk/src/handlers/shared/compute-strategy.ts index 87841a59a..3a8ffc3ba 100644 --- a/cdk/src/handlers/shared/compute-strategy.ts +++ b/cdk/src/handlers/shared/compute-strategy.ts @@ -73,7 +73,7 @@ export type SessionHandle = * ``"substrate state completed"``. * * It is OPTIONAL and OPAQUE to the strategy. The orchestrator retains it as - * diagnostic detail and recognizes the service's documented run-hook 4xx shape + * diagnostic detail and recognizes known run-rejection and resume-hook failure shapes * to choose a stable failure code. Consumers classify that code, so arbitrary * words in the reason cannot change the category or user-facing retry advice. * diff --git a/cdk/src/handlers/shared/error-classifier.ts b/cdk/src/handlers/shared/error-classifier.ts index 3ee238479..08e02679e 100644 --- a/cdk/src/handlers/shared/error-classifier.ts +++ b/cdk/src/handlers/shared/error-classifier.ts @@ -97,6 +97,18 @@ interface ErrorPattern { /** Stable codes written by the orchestrator; diagnostic text cannot override them. */ const MICROVM_TERMINAL_CLASSIFICATIONS: Readonly> = { + MICROVM_RESUME_HOOK_FAILED: { + category: ErrorCategory.COMPUTE, + title: 'The MicroVM could not wake after being paused', + description: + 'AWS could not complete the worker’s wake hook. An accepted wake request does not mean the worker resumed. The task may already have made changes before it paused.', + remedy: + 'An ABCA admin should inspect the task ID and MicroVM ID in the coordinator and approval API logs, including the AWS request ID, ' + + 'then check microvm_hook_started, microvm_hook_stage_failed and microvm_hook_finished in /aws/lambda-microvms/. ' + + 'A connection refusal can happen before the guest hook logs anything. Check saved task progress and cleanup before starting a replacement task; retrying alone is not a verified fix.', + retryable: false, + errorClass: ErrorClass.SERVICE, + }, MICROVM_RUN_HOOK_REJECTED: { category: ErrorCategory.CONFIG, title: 'The MicroVM rejected its own run payload', @@ -116,7 +128,7 @@ const MICROVM_TERMINAL_CLASSIFICATIONS: Readonly = {}; let lifetime: Pick = {}; try { const result = await getClient().send(new GetMicrovmCommand({ microvmIdentifier: microvmId, }), { abortSignal: controlSignal(options) }); state = result.state; + requestIdentity = microvmRequestIdentity(result); const startedAtMs = result.startedAt instanceof Date ? result.startedAt.getTime() : NaN; if (Number.isSafeInteger(startedAtMs) && startedAtMs >= 0 && Number.isSafeInteger(result.maximumDurationInSeconds) && result.maximumDurationInSeconds! > 0) { @@ -784,6 +786,7 @@ export class LambdaMicrovmComputeStrategy implements ComputeStrategy { if (err instanceof Error && err.name === 'ResourceNotFoundException') { logger.info('MicroVM not found on poll — treating as terminal', { microvm_id: microvmId, + ...microvmErrorIdentity(err), }); return { status: 'completed', microvmState: 'NOT_FOUND' }; } @@ -806,6 +809,9 @@ export class LambdaMicrovmComputeStrategy implements ComputeStrategy { // where the operator happens to be looking. logger.warn('MicroVM reached a terminal state with a substrate reason', { microvm_id: microvmId, + image_arn: handle.imageArn, + image_version: handle.imageVersion, + ...requestIdentity, state, state_reason: stateReason, }); @@ -814,6 +820,7 @@ export class LambdaMicrovmComputeStrategy implements ComputeStrategy { default: logger.warn('Unrecognized MicroVM state — reporting running', { microvm_id: microvmId, + ...requestIdentity, state, ...(stateReason && { state_reason: stateReason }), }); @@ -866,14 +873,29 @@ export class LambdaMicrovmComputeStrategy implements ComputeStrategy { } const suspend = operation === 'suspendSession'; const request = { microvmIdentifier: handle.microvmId }; + const startedAt = Date.now(); + const diagnostic = { + operation: suspend ? 'SuspendMicrovm' : 'ResumeMicrovm', + microvm_id: handle.microvmId, + image_arn: handle.imageArn, + image_version: handle.imageVersion, + }; + logger.info('MicroVM lifecycle request started', diagnostic); try { - await getClient().send( + const response = await getClient().send( suspend ? new SuspendMicrovmCommand(request) : new ResumeMicrovmCommand(request), { abortSignal: controlSignal(options) }, ); + // An accepted command is distinct from the observed transition/guest acknowledgment. + logger.info('MicroVM lifecycle request acknowledged', { + ...diagnostic, elapsed_ms: Date.now() - startedAt, ...microvmRequestIdentity(response), + }); } catch (error) { // Includes Conflict/NotFound: neither proves the desired state was reached. // Even a timeout may have committed; the durable caller must observe again. + logger.warn('MicroVM lifecycle request failed', { + ...diagnostic, elapsed_ms: Date.now() - startedAt, ...microvmErrorIdentity(error), + }); throw wrapMicrovmError(suspend ? 'SuspendMicrovm' : 'ResumeMicrovm', error); } return { supported: true }; @@ -894,11 +916,14 @@ export class LambdaMicrovmComputeStrategy implements ComputeStrategy { * the ordinary finalize path are worth telling apart in CloudWatch). */ private async terminateBestEffort(microvmId: string, reason: string, options?: SessionControlOptions): Promise { + const startedAt = Date.now(); try { - await getClient().send(new TerminateMicrovmCommand({ + const response = await getClient().send(new TerminateMicrovmCommand({ microvmIdentifier: microvmId, }), { abortSignal: controlSignal(options) }); - logger.info('Lambda MicroVM termination requested', { microvm_id: microvmId, reason }); + logger.info('Lambda MicroVM termination requested', { + microvm_id: microvmId, reason, elapsed_ms: Date.now() - startedAt, ...microvmRequestIdentity(response), + }); return { outcome: 'requested' }; } catch (err) { const identity = microvmErrorIdentity(err); diff --git a/cdk/test/handlers/shared/error-classifier.test.ts b/cdk/test/handlers/shared/error-classifier.test.ts index 56662ee2e..628da914c 100644 --- a/cdk/test/handlers/shared/error-classifier.test.ts +++ b/cdk/test/handlers/shared/error-classifier.test.ts @@ -17,7 +17,7 @@ * SOFTWARE. */ -import { classifyError, ErrorCategory, ErrorClass, isTransientError, retryGuidance, type ErrorClassification } from '../../../src/handlers/shared/error-classifier'; +import { classifyError, ErrorCategory, ErrorClass, formatMicrovmTerminalFailure, isTransientError, retryGuidance, type ErrorClassification } from '../../../src/handlers/shared/error-classifier'; import { LAMBDA_MICROVM_SUPPORTED_REGIONS } from '../../../src/handlers/shared/microvm-regions'; import { toTaskDetail, type TaskRecord } from '../../../src/handlers/shared/types'; @@ -612,6 +612,34 @@ describe('classifyError', () => { 'MicroVM substrate terminated before the agent wrote a terminal status: ' + `substrate state completed (${reason})`; + test.each([ + 'Resume lifecycle hook connection was refused. Please check your hook endpoint and application logs for more details.', + 'Resume lifecycle hook returned HTTP status 503.', + 'Resume lifecycle hook returned HTTP status 409.', + ])('gives actionable wake diagnostics for %s', reason => { + const persisted = formatMicrovmTerminalFailure('substrate state completed', reason); + expect(persisted).toMatch(/^MICROVM_RESUME_HOOK_FAILED: /); + expect(persisted).toContain(reason); + for (const message of [persisted, reconciled(reason), reason]) { + const result = classifyError(message)!; + expect(result.title).toBe('The MicroVM could not wake after being paused'); + expect(result.errorClass).toBe(ErrorClass.SERVICE); + expect(result.retryable).toBe(false); + expect(result.remedy).toContain('AWS request ID'); + expect(result.remedy).toContain('before starting a replacement'); + expect(retryGuidance(result)).toMatch(/needs your ABCA admin/i); + } + }); + + test('keeps unknown wording generic and honors persisted codes ahead of diagnostic words', () => { + expect(formatMicrovmTerminalFailure('completed', 'diagnostic: Resume lifecycle hook connection was refused.')) + .toMatch(/^MICROVM_SUBSTRATE_TERMINATED: /); + expect(classifyError(`MICROVM_SUBSTRATE_TERMINATED: ${reconciled('Resume lifecycle hook connection was refused.')}`)!.errorClass) + .toBe(ErrorClass.TRANSIENT); + expect(classifyError('MICROVM_RESUME_HOOK_FAILED: concurrency limit; missing_secret')!.errorClass) + .toBe(ErrorClass.SERVICE); + }); + test.each([ 'MicroVM host unavailable.', 'MicroVM capacity unavailable in this Availability Zone.', diff --git a/cdk/test/handlers/shared/microvm-approval-wake.test.ts b/cdk/test/handlers/shared/microvm-approval-wake.test.ts index 2bb71089b..bab59cd39 100644 --- a/cdk/test/handlers/shared/microvm-approval-wake.test.ts +++ b/cdk/test/handlers/shared/microvm-approval-wake.test.ts @@ -91,6 +91,24 @@ beforeEach(() => { }); afterEach(() => jest.restoreAllMocks()); +test('ties the saved decision generation to the accepted AWS request without logging its body', async () => { + mockSend.mockResolvedValueOnce({ state: 'SUSPENDED' }).mockResolvedValueOnce({ + $metadata: { requestId: 'aws-inline-123' }, private: 'secret-response', + }); + await wake(); + expect(mockLogger.info).toHaveBeenCalledWith('MicroVM wake requested after approval decision', expect.objectContaining({ + task_id: 'task', + request_id: 'gate', + microvm_id: 'vm', + generation: 'wake-generation', + intent_requested_at_ms: NOW, + aws_request_id: 'aws-inline-123', + elapsed_ms: 0, + })); + expect(JSON.stringify(mockLogger.info.mock.calls)).not.toContain('secret-response'); + expect(mockRead).toHaveBeenCalledTimes(3); +}); + test.each(['APPROVED', 'DENIED'] as const)('%s saves wake before Get, rechecks ownership, then requests Resume and reads again', async decision => { if (row.approval.kind === 'present') row = { ...row, approval: { ...row.approval, status: decision } }; await wake(decision); diff --git a/cdk/test/handlers/shared/microvm-control.test.ts b/cdk/test/handlers/shared/microvm-control.test.ts new file mode 100644 index 000000000..0d234e8f5 --- /dev/null +++ b/cdk/test/handlers/shared/microvm-control.test.ts @@ -0,0 +1,48 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +// SPDX-License-Identifier: MIT-0 + +import { microvmErrorIdentity, microvmRequestIdentity } from '../../../src/handlers/shared/microvm-control'; + +test.each([undefined, null, {}, { $metadata: {} }, { $metadata: { requestId: 'bad\nsecret' } }])( + 'omits absent or malformed request metadata: %j', response => { + expect(microvmRequestIdentity(response)).toEqual({}); + }, +); + +test('keeps only the request ID from a successful reply', () => { + expect(microvmRequestIdentity({ + $metadata: { requestId: 'aws-123', headers: { authorization: 'secret' } }, payload: 'secret', + })).toEqual({ aws_request_id: 'aws-123' }); +}); + +test('uses the original SDK identity behind a wrapper without copying messages', () => { + expect(microvmErrorIdentity(Object.assign(new Error('private wrapper'), { + cause: Object.assign(new Error('private SDK detail'), { + name: 'AccessDeniedException', $metadata: { requestId: 'aws-456' }, + }), + }))).toEqual({ error_type: 'AccessDeniedException', aws_request_id: 'aws-456' }); +}); + +test('rejects malformed error names and request IDs', () => { + expect(microvmErrorIdentity({ + name: 'private\ncontent', $metadata: { requestId: 'secret'.repeat(30) }, + })).toEqual({ error_type: 'Error' }); +}); diff --git a/cdk/test/handlers/shared/microvm-supervisor.test.ts b/cdk/test/handlers/shared/microvm-supervisor.test.ts index 7f39ddaa0..fdbbbdbf4 100644 --- a/cdk/test/handlers/shared/microvm-supervisor.test.ts +++ b/cdk/test/handlers/shared/microvm-supervisor.test.ts @@ -145,6 +145,40 @@ beforeEach(() => { }); afterEach(() => jest.restoreAllMocks()); +test('logs changed observations once across durable replay without moving the wake deadline', async () => { + intent('resume', NOW - 10_000); + approve(); + strategy.pollSession.mockResolvedValue(observation('PENDING')); + const first = await run(); + expect(mockLogger.info).toHaveBeenCalledWith('MicroVM supervisor observation changed', expect.objectContaining({ + observed_state: 'PENDING', + task_id: 'task', + microvm_id: 'vm', + request_id: 'gate', + approval_state: 'APPROVED', + recovery_kind: 'wake', + recovery_since_ms: NOW - 10_000, + intent_requested_at_ms: NOW - 10_000, + image_version: '3.0', + })); + const signature = first.state.diagnosticSignature; + mockLogger.info.mockClear(); + time += 1_000; + const second = await run(first.state); + expect(second.state.diagnosticSignature).toBe(signature); + expect(mockLogger.info).not.toHaveBeenCalledWith('MicroVM supervisor observation changed', expect.anything()); + time += MICROVM_RECOVERY_TIMEOUT_MS; + const failed = await run(second.state); + expect(failed.kind).toBe('failure'); + expect(failed.state.recovery?.sinceMs).toBe(NOW - 10_000); + expect(mockLogger.warn).toHaveBeenCalledWith('MicroVM supervisor observation changed', expect.objectContaining({ + outcome: 'failure', + recovery_since_ms: NOW - 10_000, + observed_state: 'PENDING', + session_deadline_ms: first.state.sessionDeadlineMs, + })); +}); + test('saves intent, rechecks the gate, requests suspend and rechecks the outcome', async () => { const result = await run(); expect(result.kind).toBe('continue'); diff --git a/cdk/test/handlers/shared/session-lifecycle.test.ts b/cdk/test/handlers/shared/session-lifecycle.test.ts index c550f0c59..58f420140 100644 --- a/cdk/test/handlers/shared/session-lifecycle.test.ts +++ b/cdk/test/handlers/shared/session-lifecycle.test.ts @@ -22,6 +22,8 @@ import { inspect } from 'node:util'; const mockMicrovmSend = jest.fn(); const mockEcsSend = jest.fn(); const mockAgentcoreSend = jest.fn(); +const mockLifecycleLogger = { info: jest.fn(), warn: jest.fn(), error: jest.fn() }; +jest.mock('../../../src/handlers/shared/logger', () => ({ logger: mockLifecycleLogger })); jest.mock('@aws-sdk/client-lambda-microvms', () => ({ ...jest.requireActual('@aws-sdk/client-lambda-microvms'), LambdaMicrovmsClient: jest.fn(() => ({ send: mockMicrovmSend })), @@ -137,6 +139,28 @@ describe('MicroVM lifetime observations', () => { }); describe.each(['suspendSession', 'resumeSession'] as const)('%s contract', operation => { + test('records AWS acknowledgment identity without response or exception contents', async () => { + const strategy = resolveComputeStrategy({ compute_type: 'lambda-microvm', runtime_arn: '' }); + mockMicrovmSend.mockResolvedValueOnce({ + $metadata: { requestId: 'aws-control-123', headers: { Authorization: 'secret-response' } }, + payload: 'secret-payload', + }); + await expect(strategy[operation](microvm)).resolves.toEqual({ supported: true }); + expect(mockLifecycleLogger.info).toHaveBeenCalledWith( + 'MicroVM lifecycle request acknowledged', + expect.objectContaining({ microvm_id: 'mvm-one', aws_request_id: 'aws-control-123', elapsed_ms: expect.any(Number) }), + ); + mockMicrovmSend.mockRejectedValueOnce(Object.assign(new Error('secret-exception'), { + name: 'ConflictException', $metadata: { requestId: 'aws-control-456' }, + })); + await expect(strategy[operation](microvm)).rejects.toThrow(); + expect(mockLifecycleLogger.warn).toHaveBeenCalledWith( + 'MicroVM lifecycle request failed', + expect.objectContaining({ error_type: 'ConflictException', aws_request_id: 'aws-control-456' }), + ); + expect(JSON.stringify([mockLifecycleLogger.info.mock.calls, mockLifecycleLogger.warn.mock.calls])).not.toContain('secret-'); + }); + test.each(handles.filter(h => h.strategyType !== 'lambda-microvm'))( '$strategyType explicitly reports unsupported without calling AWS', async handle => { const strategy = resolveComputeStrategy({ compute_type: handle.strategyType, runtime_arn: 'arn:runtime' }); diff --git a/docs/verification/645-p3-implementation-plan.md b/docs/verification/645-p3-implementation-plan.md index 31821daaa..3df5fdba8 100644 --- a/docs/verification/645-p3-implementation-plan.md +++ b/docs/verification/645-p3-implementation-plan.md @@ -238,6 +238,11 @@ handoff, use this order: Its seven workers and temporary infrastructure were removed, and the exact evidence was privately archived. These timer results do not explain the separate connection refusals. + The [lifecycle diagnostics guide](./645-p3-lifecycle-diagnostics.md) describes + the new hook-stage, AWS request-ID and durable state-change logging. + Three isolated AWS workflows verified it, including actual API wake and + coordinator recovery; normal-stack rollout remains pending. The original + server remained PID 1. Logging and successful controls do not close the defect. 2. Complete the remaining live race/fault matrix: cancellation during transitions, late decision races, repeated polling/credential-refresh failures, durable registration races, service token-retention expiry and recovery of a worker diff --git a/docs/verification/645-p3-lifecycle-diagnostics.md b/docs/verification/645-p3-lifecycle-diagnostics.md new file mode 100644 index 000000000..08bee7110 --- /dev/null +++ b/docs/verification/645-p3-lifecycle-diagnostics.md @@ -0,0 +1,173 @@ +# ADR-021 P3: lifecycle diagnostics + +Updated 2026-09-16. The logging changes passed local checks and three isolated +AWS workflows. Normal-stack rollout remains pending. They do not resolve the four +[recorded wake refusals](./645-p3-resume-refusal-investigation.md). + +## What the records tell us + +Think of waking a worker as delivering a message, then waiting for it to finish +getting ready. These are separate checkpoints: + +1. The approval API saves the decision and a wake intent. That saved decision can + succeed even if the immediate wake request fails. +2. AWS accepts `ResumeMicrovm`. Its request ID is a receipt for that API call; + it does not prove that the guest received `/resume` or resumed execution. +3. The guest receives `/resume`, refreshes credentials, checks the original task + and approval, and returns its HTTP response. +4. The coordinator observes the resulting worker/task state. It retains the + original wake and approval deadlines across retries. + +## Cloud logs + +Search the deployed coordinator and approval API Lambda log groups by `task_id` +and `microvm_id`. The approval gate is `request_id`; an AWS service receipt is +`aws_request_id`. These identifiers have different meanings. + +| Record/message | Useful fields | +|---|---| +| `MicroVM observed after approval decision` | Observed state, saved intent generation/time, Get request ID | +| `MicroVM wake request started after approval decision` | Task, gate, worker, image, saved generation/time | +| `MicroVM wake requested after approval decision` | Same correlation plus Resume request ID and elapsed milliseconds | +| `MicroVM lifecycle request started/acknowledged/failed` | Suspend or Resume operation, worker/image, elapsed time; AWS request ID when supplied | +| `MicroVM supervisor observation changed` | Task/worker/approval state, intent generation, recovery kind/original start, failure counters, outcome/reason, original deadlines | +| `MicroVM reached a terminal state with a substrate reason` | AWS state reason, worker/image, Get request ID | +| `Lambda MicroVM termination requested` | Worker, cleanup reason, Terminate request ID and elapsed time | + +The supervisor saves its last diagnostic signature in durable state, so an +unchanged poll after replay does not repeat the same record. A state change, +recovery change, failure-count change or terminal outcome does. Timestamps and +elapsed time are not part of that signature. Old saved states without a signature +remain usable and log their first new observation. + +## Guest logs + +Select `/aws/lambda-microvms/`. A typical CloudWatch Logs Insights +query is: + +```text +fields @timestamp, event, action, stage, callback_stage, code, http_status, + hook_id, request_id, pid, phase, elapsed_ms, late, + error_type, aws_error_code, aws_request_id +| filter microvm_id = "REPLACE_WITH_WORKER_ID" +| sort @timestamp asc +| limit 500 +``` + +Each HTTP invocation generates a fresh `hook_id`. All its callback-thread +records share that ID. The registered task, worker and gate identify the work; +the request body cannot override those diagnostic identities. + +| Event | Meaning | +|---|---| +| `microvm_hook_started` | The HTTP handler was entered, before reading its body | +| `microvm_hook_stage` | About to perform `callback_stage`; emitted before potentially blocking work | +| `microvm_hook_stage_finished` | That piece of work returned; this alone is not a hook acknowledgment | +| `microvm_hook_stage_failed` | That piece of work failed; includes exception type and safe AWS code/request ID when available | +| `microvm_hook_finished` | Handler result: HTTP status and stable diagnostic code; cancellation has no HTTP status | + +Stages include body reading, local identity checks, the controller, draining +active writes/reads, checkpoint reads/transaction, credential refresh, and resume +identity reads/transaction. `stage` retains the last entered operation, including +when a callback is still blocked at timeout. Nested stage records also identify +their enclosing operation in `callback_stage`. + +Records include the server PID, local phase/generation, active tool/activity +counts and whether earlier progress failure disabled suspension. These are local +observations; they do not prove the listener remained healthy during a freeze. +`microvm_hook_finished` records the response the handler selected; the service's +state and HTTP access logs establish what AWS subsequently observed. + +`late: true` means a callback logged after the handler had already finished. +It cannot turn a timed-out wake into success. The existing controller still owns +the barrier that prevents tools from continuing after an uncertain wake. +Logging failure is best effort and cannot change the hook result. + +The new diagnostics omit bodies, tool arguments, approval contents, credentials, +SDK response bodies, raw exception messages and tracebacks. AWS error codes and +request IDs are bounded identifiers. Existing terminal `state_reason` retention +continues independently for service diagnosis. + +## Reading a failed wake + +1. Find the saved intent and the actual Resume acknowledgment or failure. Check + its generation and original request time; do not restart the timer while + investigating. +2. Find the corresponding guest hook start. If absent, account for log delivery + delay and retention. Absence alone cannot distinguish a dead process, missing + listener, service transport failure or missing logs. +3. If the hook started, inspect the last stage and final code. For example, + `credential-refresh` plus `AccessDenied` points to credential renewal, while + `resume-identity-read` plus `MICROVM_LIFECYCLE_UNAVAILABLE` points to task/gate + reconciliation. A timeout names the operation that was still outstanding. +4. Check the coordinator's outcome, task progress, worker cleanup and capacity + release. A saved approval and successful cleanup do not mean coding resumed. +5. Retain the task/worker IDs, exact image and coordinator versions, UTC window, + AWS request IDs, service state reason and relevant logs. For connection + refusal before hook entry, independent listener/process or service-side + evidence is still required. + +Known Resume connection-refused and HTTP 4xx/5xx reasons now produce the stable +`MICROVM_RESUME_HOOK_FAILED` task error. It asks an admin to inspect this evidence +and saved progress before starting a replacement. It does not promise that +retrying repairs the fault. Persisted classification codes take priority over +diagnostic words; unrecognized AWS wording keeps the generic terminal code. + +## Verification and next deployment + +Local regressions cover hook correlation, sensitive-text exclusion, safe AWS +identifiers, logging-sink failure, callback timeout/late completion, durable +observation deduplication without moving deadlines, and wake error feedback. +The root `mise run build` exited **0**: **5,019 CDK tests** and **928 CLI tests** +passed, along with lint, types, contracts, synthesis and documentation checks. +After the final concurrent-output/cancellation coverage and guide edits, agent +quality and documentation checks passed again: **1,947 Python tests**, **86.43%** +coverage. **56 CDK** and **11 Python** optional DynamoDB Local cases were skipped; +the database transaction conditions were unchanged. + +The private image `backgroundagent-dev-p3-diagnostics-20260916:1.0` was derived +from the exact image `4.0` source artifact. It replaced three lifecycle Python +modules and added `microvm_diagnostics.py`; the Dockerfile, original server +command, dependencies and hook configuration were unchanged. Its artifact SHA-256 +was `15b8867dfe7f45270246695d9c87d2f1ef42d0b338fd331a88fbe5bfc26d7733`. +Private coordinator and approval Lambda versions 1 and 2 used the checked +production code with fixed verification task identities. + +The strict log audit passed at **2026-09-16 14:21:03 UTC**: + +| Case | Actual wake path | Resume AWS request ID | +|---|---|---| +| `approve-a` | Coordinator won the race; API observed restore `PENDING` and retained its successful approval response | `c6ba7eea-5f18-4ccf-b032-b0aeb0cce15a` | +| `approve-b` | Verification wrapper omitted immediate API wake; coordinator issued the sole Resume | `29ebc13a-cf70-4ed9-84b2-a0f31533406c` | +| `approve-inline` | API issued the sole Resume during a verification-only 8-second coordinator polling delay | `4b7d0ba8-3106-4126-bbaa-7ad89df9fc72` | + +The first case was originally intended to exercise the API's Resume call. Its +strict assertion failed, and the new logs showed the actual successful fallback. +That evidence was retained; the fresh third case established the missing API +path. The timing delay affected only the private test wrapper, after suspend +intent. It changed neither production supervisor logic nor original deadlines. + +Every worker produced **32 guest diagnostic records**, one successful suspend +and resume hook pair, and the expected checkpoint/refresh/reconciliation stage +records. The original server remained **PID 1** across each wake. The coordinator +recorded **9, 8 and 5** meaningful observations respectively. AWS receipts were +present for suspend, the actual resume issuer, and cleanup. No guest failure or +late-callback record appeared in these successful workflows. + +All three tasks completed with their original approval deadlines, released +capacity, zero counters, terminated workers and empty launch-payload prefixes. +Complete retained traces each contain exactly one approved `Read` of +`/etc/os-release`. These short successful wakes establish working diagnostics; +failure/redaction paths were verified by local fault injection. The original +connection-refused defect was not reproduced. + +Raw logs, traces, durable histories, artifacts, source digests, original failed +assertions and corrected audit scripts are retained in private verification +evidence. Cleanup verified removal of both private functions (all versions), +the diagnostic image, three roles, the private SSM switch, three log groups, +the artifact object and three owned zero-valued counters. Task/approval history +and trace objects retain their normal retention. +Normal tasks still use coordinator **5** and image **4.0**; they do not yet carry +this new instrumentation. Roll out the checked code before relying on these +records for normal-stack diagnosis. Automatic suspension remains disabled until +the remaining P3 gates pass. diff --git a/docs/verification/645-p3-resume-refusal-investigation.md b/docs/verification/645-p3-resume-refusal-investigation.md index f28ee39b5..bf0e6d72d 100644 --- a/docs/verification/645-p3-resume-refusal-investigation.md +++ b/docs/verification/645-p3-resume-refusal-investigation.md @@ -20,6 +20,12 @@ The coordinator detects termination, marks the task failed and releases its capacity reservation. Successful failure cleanup does not make the requested approval workflow successful. +The [lifecycle diagnostics guide](./645-p3-lifecycle-diagnostics.md) documents +the new logging and wake-failure feedback. Three isolated AWS workflows verified +the instrumentation with the original server running as PID 1, including actual +API wake and coordinator recovery. None reproduced refusal. Normal-stack rollout +remains pending; the historical failures below predate this instrumentation. + ## Recorded failures All times are UTC, in `us-west-2`, on image `backgroundagent-dev-abca-agent`. From 15514e18917b8942ba671a51aaa90c3ba02dab92 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Wed, 16 Sep 2026 10:32:13 -0400 Subject: [PATCH 049/149] docs: track Lambda MicroVM service feedback and current P3 status (#645) --- ...ADR-021-lambda-microvms-compute-backend.md | 4 +- ...Adr-021-lambda-microvms-compute-backend.md | 4 +- .../645-lambda-microvm-service-feedback.md | 178 ++++++++++++++++++ .../645-p3-implementation-plan.md | 3 + .../645-p3-resume-refusal-investigation.md | 2 + 5 files changed, 189 insertions(+), 2 deletions(-) create mode 100644 docs/verification/645-lambda-microvm-service-feedback.md diff --git a/docs/decisions/ADR-021-lambda-microvms-compute-backend.md b/docs/decisions/ADR-021-lambda-microvms-compute-backend.md index d73fe49c1..d8f57353e 100644 --- a/docs/decisions/ADR-021-lambda-microvms-compute-backend.md +++ b/docs/decisions/ADR-021-lambda-microvms-compute-backend.md @@ -1,6 +1,8 @@ # ADR-021: AWS Lambda MicroVMs as a third ComputeStrategy backend -> **Implementation status (2026-09-15):** P1 ([#689](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/689)) and P2 ([#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733)) are merged. The takeover branch now has a [clean deployment](../verification/645-p2-clean-deployment-20260913.md) with bootstrap 1.7.0 and a [verified managed-image update to 2.0](../verification/645-microvm-image-rebuild-20260914.md), plus [live coding, PR iteration and cancellation evidence](../verification/645-p2-live-task-20260914.md), without manual IAM workarounds. Remaining P2 failure/recovery and permission/network checks prevent full acceptance. Local P3 foundations include suspend/resume commands, explicit VM observations, guarded persistent intent, a policy helper, approval deadlines that survive a frozen clock, the [guest pause controller](../verification/645-p3-guest-barrier.md), and [scoped Claude credentials with retained-client refresh](../verification/645-p3-credentials.md). The [production HTTP hooks](../verification/645-p3-lifecycle-hooks.md) now connect the guest barrier to atomic checkpoints and credential/gate reconciliation locally. [Per-worker image capability](../verification/645-p3-image-capability.md) now declares the six hooks and retains/verifies the actual launched image version locally. [Supervisor/approval-handler integration](../verification/645-p3-supervisor.md) now adds durable recovery, bounded post-commit wake and scoped IAM locally. Live acceptance remains open; automatic suspension defaults off. This ADR defines P1–P3, not a P4. See the [current review](../verification/645-p3-readiness-review.md) and [P3 implementation plan](../verification/645-p3-implementation-plan.md). +> **Implementation status (2026-09-16):** P1 ([#689](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/689)) and P2 ([#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733)) are merged. The takeover has a [clean deployment](../verification/645-p2-clean-deployment-20260913.md), [live repository workflow evidence](../verification/645-p2-live-task-20260914.md), and integrated P3 guest hooks, credentials, durable intent and approval/supervisor handling. The current normal deployment uses agent image **4.0**, coordinator **5** and bootstrap **1.8.0**. [Long-sleep credential/deadline acceptance](../verification/645-p3-callback-live-20260915.md) and the [pending-wake timer corrections](../verification/645-p3-pending-wake.md) passed live checks. +> +> **P3 remains incomplete; automatic suspension is disabled.** Four [resume connection refusals](../verification/645-p3-resume-refusal-investigation.md) remain unexplained, alongside remaining race, permission/network, recovery and rollout checks. New [lifecycle diagnostics and wake feedback](../verification/645-p3-lifecycle-diagnostics.md) passed three isolated AWS workflows; their normal-stack rollout is pending. Follow the [current implementation plan](../verification/645-p3-implementation-plan.md) and [service-team feedback tracker](../verification/645-lambda-microvm-service-feedback.md). This ADR defines P1–P3, not a P4. **Status:** proposed **Date:** 2026-07-29 diff --git a/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md b/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md index f671983e7..473d05bef 100644 --- a/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md +++ b/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md @@ -4,7 +4,9 @@ title: Adr 021 lambda microvms compute backend # ADR-021: AWS Lambda MicroVMs as a third ComputeStrategy backend -> **Implementation status (2026-09-15):** P1 ([#689](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/689)) and P2 ([#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733)) are merged. The takeover branch now has a [clean deployment](/sample-autonomous-cloud-coding-agents/architecture/645-p2-clean-deployment-20260913) with bootstrap 1.7.0 and a [verified managed-image update to 2.0](/sample-autonomous-cloud-coding-agents/architecture/645-microvm-image-rebuild-20260914), plus [live coding, PR iteration and cancellation evidence](/sample-autonomous-cloud-coding-agents/architecture/645-p2-live-task-20260914), without manual IAM workarounds. Remaining P2 failure/recovery and permission/network checks prevent full acceptance. Local P3 foundations include suspend/resume commands, explicit VM observations, guarded persistent intent, a policy helper, approval deadlines that survive a frozen clock, the [guest pause controller](/sample-autonomous-cloud-coding-agents/architecture/645-p3-guest-barrier), and [scoped Claude credentials with retained-client refresh](/sample-autonomous-cloud-coding-agents/architecture/645-p3-credentials). The [production HTTP hooks](/sample-autonomous-cloud-coding-agents/architecture/645-p3-lifecycle-hooks) now connect the guest barrier to atomic checkpoints and credential/gate reconciliation locally. [Per-worker image capability](/sample-autonomous-cloud-coding-agents/architecture/645-p3-image-capability) now declares the six hooks and retains/verifies the actual launched image version locally. [Supervisor/approval-handler integration](/sample-autonomous-cloud-coding-agents/architecture/645-p3-supervisor) now adds durable recovery, bounded post-commit wake and scoped IAM locally. Live acceptance remains open; automatic suspension defaults off. This ADR defines P1–P3, not a P4. See the [current review](/sample-autonomous-cloud-coding-agents/architecture/645-p3-readiness-review) and [P3 implementation plan](/sample-autonomous-cloud-coding-agents/architecture/645-p3-implementation-plan). +> **Implementation status (2026-09-16):** P1 ([#689](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/689)) and P2 ([#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733)) are merged. The takeover has a [clean deployment](/sample-autonomous-cloud-coding-agents/architecture/645-p2-clean-deployment-20260913), [live repository workflow evidence](/sample-autonomous-cloud-coding-agents/architecture/645-p2-live-task-20260914), and integrated P3 guest hooks, credentials, durable intent and approval/supervisor handling. The current normal deployment uses agent image **4.0**, coordinator **5** and bootstrap **1.8.0**. [Long-sleep credential/deadline acceptance](/sample-autonomous-cloud-coding-agents/architecture/645-p3-callback-live-20260915) and the [pending-wake timer corrections](/sample-autonomous-cloud-coding-agents/architecture/645-p3-pending-wake) passed live checks. +> +> **P3 remains incomplete; automatic suspension is disabled.** Four [resume connection refusals](/sample-autonomous-cloud-coding-agents/architecture/645-p3-resume-refusal-investigation) remain unexplained, alongside remaining race, permission/network, recovery and rollout checks. New [lifecycle diagnostics and wake feedback](/sample-autonomous-cloud-coding-agents/architecture/645-p3-lifecycle-diagnostics) passed three isolated AWS workflows; their normal-stack rollout is pending. Follow the [current implementation plan](/sample-autonomous-cloud-coding-agents/architecture/645-p3-implementation-plan) and [service-team feedback tracker](/sample-autonomous-cloud-coding-agents/architecture/645-lambda-microvm-service-feedback). This ADR defines P1–P3, not a P4. **Status:** proposed **Date:** 2026-07-29 diff --git a/docs/verification/645-lambda-microvm-service-feedback.md b/docs/verification/645-lambda-microvm-service-feedback.md new file mode 100644 index 000000000..4abc45a24 --- /dev/null +++ b/docs/verification/645-lambda-microvm-service-feedback.md @@ -0,0 +1,178 @@ +# Lambda MicroVM service-team feedback tracker + +Updated 2026-09-16. Working notes for the ADR-021 takeover. **Not submitted to the +service team.** Keep each item's evidence, question, service response and next +action here as verification continues. + +In plain language: a MicroVM is the worker's little computer. A lifecycle hook +is the doorbell AWS rings to tell it to start, pause or wake. An AWS request ID +is the receipt that lets the service team find a particular call. + +| ID | Priority | Topic | Evidence/status | +|---|---|---|---| +| F01 | P3 blocker | Accepted wake ends in connection refusal | Four recorded failures; responsible component unknown | +| F02 | High | Supported IAM conditions and misleading permission errors | Reproduced in earlier P2 work; current service behavior needs confirmation | +| F03 | Medium | A service-side hook timeline and structured failure details | Diagnostic improvement request based on F01 | +| F04 | Medium | `PENDING` also means restoring an existing worker | Observed live; our timer bug is fixed | +| F05 | Medium | HTTP connection handling across suspend/resume | Contract question; no established cause of F01 | +| F06 | Medium | Conditional operator-role requirement for VPC connectors | Earlier deployment failure; application setup fixed | + +## F01 — Wake request accepted, then the hook connection is refused + +**Impact:** the user approves an action, but the worker stops before continuing. +The coordinator releases capacity correctly; the requested coding workflow +still fails. This prevents enabling automatic suspension for normal tasks. + +**Observed:** four failures on images `3.0` and `4.0` in `us-west-2`, September +15–16. AWS accepted `ResumeMicrovm`, then reported: + +> Resume lifecycle hook connection was refused. Please check your hook endpoint +> and application logs for more details. + +The guest had returned HTTP 200 from `/suspend`; retained logs contain no +subsequent `/resume` access entry. Single-issuer cases exclude overlapping API +and coordinator Resume calls as a necessary cause. + +**Best starting evidence for the service team:** + +- Account ``, region `us-west-2`, September 16, 00:25:01–00:25:12 UTC. +- Worker `microvm-2da070a7-43cf-3bb3-9c5d-69bc106f2cb1`, image + `backgroundagent-dev-abca-agent:4.0`. +- Suspend receipt `e2199854-5ad0-46c9-bc59-248ac68ead82`. +- Resume receipt `0acb3fec-8b18-4bed-b457-0271a15ffdd6`, acknowledged + 00:25:06.766 UTC; terminal refusal observed 00:25:07.524 UTC. +- [Prepared investigation report](./645-p3-resume-refusal-investigation.md) + contains all four worker IDs, comparison runs and the complete example timeline. + +**Ask:** inspect the service's restore and hook-transport records for these +receipts. Was this a TCP refusal from the guest listener, a stale connection, +a process exit, or a networking/restore failure? At what point did the service +consider the guest network and listener ready, and what underlying error did it +map to this reason? + +**Limits:** no root cause is established. The Mac was the test controller; the +failing connection was between AWS and the guest inside AWS. Later successful +controls, including three with the original server as PID 1, do not prove this +defect fixed or establish a failure rate. + +**Next:** retain a fresh failure with independent process/listener evidence, +or obtain the service-side evidence above. Service response: pending. + +## F02 — IAM conditions and errors make correct setup difficult + +IAM roles are permission sets. A trust condition adds a rule about who may use +one. `PassRole` is permission to hand a role to a service. + +**Earlier evidence:** the [ADR infrastructure decision](../decisions/ADR-021-lambda-microvms-compute-backend.md) +records August 6–7 tests in which: + +- Trust policies using `aws:SourceAccount` / `aws:SourceArn` prevented the + MicroVM-facing roles from being assumed. Removing those conditions restored + the tested operations. +- A role-assumption problem surfaced as a caller-side `iam:PassRole` denial, + despite an existing grant and an `allowed` policy simulation. +- A separate clean comparison found exact-role `iam:PassRole` with + `iam:PassedToService: lambda.amazonaws.com` denied, while the same exact-role + grant without that condition succeeded. The earlier contaminated comparison + is explicitly corrected in the ADR. + +**Impact:** a normal attempt to tighten permissions breaks deployment or launch, +and the error sends the operator to the wrong policy. Our integration contains +the verified setup and exact-resource restrictions. + +**Ask:** publish the supported condition keys and values for build, execution, +connector-role assumption and both PassRole paths. Can the usual source/service +conditions be supported? Can errors distinguish missing caller permission from +failed target-role assumption? Is this behavior different in the current service? + +**Limits:** these are earlier reproduced integration findings, not a fresh +September 16 comparison or a demonstrated security exploit. Service response: +pending; recheck the recommended recipe before changing our working policies. + +## F03 — Expose the hook attempt that follows an accepted API request + +**Observed:** a successful Resume response provides an API receipt, but does not +establish that the guest received or completed its hook. In F01, the remaining +service explanation is a human-readable `stateReason`. A hook that is never +reached cannot write its own application diagnostic. + +**Our improvement:** [correlated lifecycle diagnostics](./645-p3-lifecycle-diagnostics.md) +now record hook entry/stage/result, PID, AWS receipts and coordinator state +changes. These passed isolated AWS verification. Normal-stack rollout is pending. + +**Ask:** provide a service-side lifecycle attempt timeline or equivalent +structured fields: originating API receipt, hook kind/attempt ID, start/end +times, connection versus HTTP failure, HTTP status, underlying error code and +whether a retry occurred. Link the attempt to worker/image identity and document +where customers can retrieve it after termination. + +**Impact:** this would show whether the doorbell failed or the worker received it +and failed while getting ready. It would also reduce dependence on parsing +human-readable failure strings. Service response: pending. + +## F04 — Document `PENDING` during restore + +**Observed:** a real worker went `SUSPENDED → PENDING → RUNNING` after Resume. +The installed SDK has no separate `RESUMING` state. The +[pending-wake record](./645-p3-pending-wake.md) includes timestamps and a verified +older-worker case with an API-issued wake. + +**Impact:** a client that treats every `PENDING` as first startup can use the +wrong timer. That was **our coordinator bug**; it is fixed and deployed in +coordinator version 5. It is separate from F01. + +**Ask:** publish a complete lifecycle transition table, including observable +restore states, and consider an explicit restoring state or transition +kind/start timestamp. Clarify which timestamps retain the original worker +lifetime and which describe the current transition. + +Service response: pending. Our next action is documentation/contract alignment, +not reopening the corrected timer bug. + +## F05 — Clarify hook HTTP connections across a freeze + +**Evidence:** some successful full-agent suspend/resume access logs used the +same peer port. An isolated Linux paused-process experiment reproduced a reset +of an old HTTP connection after a six-second pause, while every fresh connection +still succeeded. It did **not** reproduce an AWS connection refusal. Some F01 +failures followed suspension by less than five seconds. + +See the [transport control](./645-p3-transport-control-20260916.md) and +[process-observer record](./645-p3-process-observer-20260916.md). + +**Ask:** does the service reuse hook TCP connections across suspend/resume, +honor `Connection: close`, and retry a failed reused connection on a fresh socket? +How are reset, refused and timeout errors classified? What ordering is guaranteed +between guest unfreeze, network restoration and hook delivery? Which clock +semantics should guest timeout/keep-alive timers expect across suspension? + +**Limits:** matching peer ports suggest reuse but do not establish all transport +behavior. A paused Linux process is not an AWS MicroVM restore. Connection-close +behavior is a proposed comparison, not an established fix. Service response: +pending. + +## F06 — Make the VPC connector role requirement obvious before deployment + +**Earlier evidence:** the generated CloudFormation/CDK property allowed omitting +`operatorRole`, but a `VPC_EGRESS` connector failed with: + +> NetworkConnectorOperatorRole is required for VPC_EGRESS connector type + +The [P1 live runbook](./645-p1-lambda-microvm-runbook.md) records the July 31 +failure and successful operator-role setup. Our construct now supplies that role. + +**Ask:** document or validate this conditional requirement in the schema/CDK +surface, with a complete example of the trust and ENI/tag/private-IP permissions. +Clarify whether a service-linked role is ever an alternative for this connector. + +**Limits:** an optional property can be correct for other connector types. This +is a request for clearer conditional validation and setup guidance, not a claim +that every connector requires the same role. Service response: pending. + +## Updating this tracker + +For each new finding, add the actual trigger, UTC window, region, worker/image +versions, AWS receipt IDs, impact, smallest supported conclusion and a concrete +service question. Link raw evidence through the verification report. Retain +unsuccessful controls and distinguish application fixes from service findings. +Record any service answer and the verification needed before closing the item. diff --git a/docs/verification/645-p3-implementation-plan.md b/docs/verification/645-p3-implementation-plan.md index 3df5fdba8..9c8178963 100644 --- a/docs/verification/645-p3-implementation-plan.md +++ b/docs/verification/645-p3-implementation-plan.md @@ -216,6 +216,9 @@ Second P3 foundation batch implemented locally (2026-09-13): ## Remaining work in execution order +Keep service questions and evidence in the +[Lambda MicroVM service-team feedback tracker](./645-lambda-microvm-service-feedback.md). + The image `4.0` long-sleep acceptance and verification-infrastructure cleanup are complete. The original approval timed out correctly, real credentials renewed after expiry, and coordinator cleanup passed without watcher repair. The diff --git a/docs/verification/645-p3-resume-refusal-investigation.md b/docs/verification/645-p3-resume-refusal-investigation.md index bf0e6d72d..090359468 100644 --- a/docs/verification/645-p3-resume-refusal-investigation.md +++ b/docs/verification/645-p3-resume-refusal-investigation.md @@ -2,6 +2,8 @@ Updated 2026-09-16 UTC. This is an investigation record and a prepared report; it has not been submitted to AWS or published as an issue. +The [service-team feedback tracker](./645-lambda-microvm-service-feedback.md) +keeps this blocker alongside related service questions and earlier P2 findings. ## Observed problem From d86f4039e7bd59483c15b244648885968ce51ff3 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Wed, 16 Sep 2026 11:19:43 -0400 Subject: [PATCH 050/149] docs: record diagnostics rollout and AWS transport checks (#645) --- ...ADR-021-lambda-microvms-compute-backend.md | 4 +- ...Adr-021-lambda-microvms-compute-backend.md | 4 +- .../645-lambda-microvm-service-feedback.md | 24 ++- .../645-p3-diagnostics-rollout-20260916.md | 159 ++++++++++++++++++ .../645-p3-implementation-plan.md | 8 +- .../645-p3-lifecycle-diagnostics.md | 12 +- .../645-p3-resume-refusal-investigation.md | 13 +- 7 files changed, 207 insertions(+), 17 deletions(-) create mode 100644 docs/verification/645-p3-diagnostics-rollout-20260916.md diff --git a/docs/decisions/ADR-021-lambda-microvms-compute-backend.md b/docs/decisions/ADR-021-lambda-microvms-compute-backend.md index d8f57353e..45ba527fd 100644 --- a/docs/decisions/ADR-021-lambda-microvms-compute-backend.md +++ b/docs/decisions/ADR-021-lambda-microvms-compute-backend.md @@ -1,8 +1,8 @@ # ADR-021: AWS Lambda MicroVMs as a third ComputeStrategy backend -> **Implementation status (2026-09-16):** P1 ([#689](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/689)) and P2 ([#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733)) are merged. The takeover has a [clean deployment](../verification/645-p2-clean-deployment-20260913.md), [live repository workflow evidence](../verification/645-p2-live-task-20260914.md), and integrated P3 guest hooks, credentials, durable intent and approval/supervisor handling. The current normal deployment uses agent image **4.0**, coordinator **5** and bootstrap **1.8.0**. [Long-sleep credential/deadline acceptance](../verification/645-p3-callback-live-20260915.md) and the [pending-wake timer corrections](../verification/645-p3-pending-wake.md) passed live checks. +> **Implementation status (2026-09-16):** P1 ([#689](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/689)) and P2 ([#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733)) are merged. The takeover has a [clean deployment](../verification/645-p2-clean-deployment-20260913.md), [live repository workflow evidence](../verification/645-p2-live-task-20260914.md), and integrated P3 guest hooks, credentials, durable intent and approval/supervisor handling. The current normal deployment uses agent image **5.0**, coordinator **6** and bootstrap **1.8.0**. [Long-sleep credential/deadline acceptance](../verification/645-p3-callback-live-20260915.md) and the [pending-wake timer corrections](../verification/645-p3-pending-wake.md) passed live checks. > -> **P3 remains incomplete; automatic suspension is disabled.** Four [resume connection refusals](../verification/645-p3-resume-refusal-investigation.md) remain unexplained, alongside remaining race, permission/network, recovery and rollout checks. New [lifecycle diagnostics and wake feedback](../verification/645-p3-lifecycle-diagnostics.md) passed three isolated AWS workflows; their normal-stack rollout is pending. Follow the [current implementation plan](../verification/645-p3-implementation-plan.md) and [service-team feedback tracker](../verification/645-lambda-microvm-service-feedback.md). This ADR defines P1–P3, not a P4. +> **P3 remains incomplete; automatic suspension is disabled.** Four [resume connection refusals](../verification/645-p3-resume-refusal-investigation.md) remain unexplained, alongside remaining race, permission/network, recovery and rollout checks. New [lifecycle diagnostics and wake feedback](../verification/645-p3-lifecycle-diagnostics.md) passed three isolated AWS workflows and are [deployed to the normal stack](../verification/645-p3-diagnostics-rollout-20260916.md). Follow the [current implementation plan](../verification/645-p3-implementation-plan.md) and [service-team feedback tracker](../verification/645-lambda-microvm-service-feedback.md). This ADR defines P1–P3, not a P4. **Status:** proposed **Date:** 2026-07-29 diff --git a/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md b/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md index 473d05bef..3352bd3fa 100644 --- a/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md +++ b/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md @@ -4,9 +4,9 @@ title: Adr 021 lambda microvms compute backend # ADR-021: AWS Lambda MicroVMs as a third ComputeStrategy backend -> **Implementation status (2026-09-16):** P1 ([#689](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/689)) and P2 ([#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733)) are merged. The takeover has a [clean deployment](/sample-autonomous-cloud-coding-agents/architecture/645-p2-clean-deployment-20260913), [live repository workflow evidence](/sample-autonomous-cloud-coding-agents/architecture/645-p2-live-task-20260914), and integrated P3 guest hooks, credentials, durable intent and approval/supervisor handling. The current normal deployment uses agent image **4.0**, coordinator **5** and bootstrap **1.8.0**. [Long-sleep credential/deadline acceptance](/sample-autonomous-cloud-coding-agents/architecture/645-p3-callback-live-20260915) and the [pending-wake timer corrections](/sample-autonomous-cloud-coding-agents/architecture/645-p3-pending-wake) passed live checks. +> **Implementation status (2026-09-16):** P1 ([#689](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/689)) and P2 ([#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733)) are merged. The takeover has a [clean deployment](/sample-autonomous-cloud-coding-agents/architecture/645-p2-clean-deployment-20260913), [live repository workflow evidence](/sample-autonomous-cloud-coding-agents/architecture/645-p2-live-task-20260914), and integrated P3 guest hooks, credentials, durable intent and approval/supervisor handling. The current normal deployment uses agent image **5.0**, coordinator **6** and bootstrap **1.8.0**. [Long-sleep credential/deadline acceptance](/sample-autonomous-cloud-coding-agents/architecture/645-p3-callback-live-20260915) and the [pending-wake timer corrections](/sample-autonomous-cloud-coding-agents/architecture/645-p3-pending-wake) passed live checks. > -> **P3 remains incomplete; automatic suspension is disabled.** Four [resume connection refusals](/sample-autonomous-cloud-coding-agents/architecture/645-p3-resume-refusal-investigation) remain unexplained, alongside remaining race, permission/network, recovery and rollout checks. New [lifecycle diagnostics and wake feedback](/sample-autonomous-cloud-coding-agents/architecture/645-p3-lifecycle-diagnostics) passed three isolated AWS workflows; their normal-stack rollout is pending. Follow the [current implementation plan](/sample-autonomous-cloud-coding-agents/architecture/645-p3-implementation-plan) and [service-team feedback tracker](/sample-autonomous-cloud-coding-agents/architecture/645-lambda-microvm-service-feedback). This ADR defines P1–P3, not a P4. +> **P3 remains incomplete; automatic suspension is disabled.** Four [resume connection refusals](/sample-autonomous-cloud-coding-agents/architecture/645-p3-resume-refusal-investigation) remain unexplained, alongside remaining race, permission/network, recovery and rollout checks. New [lifecycle diagnostics and wake feedback](/sample-autonomous-cloud-coding-agents/architecture/645-p3-lifecycle-diagnostics) passed three isolated AWS workflows and are [deployed to the normal stack](/sample-autonomous-cloud-coding-agents/architecture/645-p3-diagnostics-rollout-20260916). Follow the [current implementation plan](/sample-autonomous-cloud-coding-agents/architecture/645-p3-implementation-plan) and [service-team feedback tracker](/sample-autonomous-cloud-coding-agents/architecture/645-lambda-microvm-service-feedback). This ADR defines P1–P3, not a P4. **Status:** proposed **Date:** 2026-07-29 diff --git a/docs/verification/645-lambda-microvm-service-feedback.md b/docs/verification/645-lambda-microvm-service-feedback.md index 4abc45a24..a15e544dd 100644 --- a/docs/verification/645-lambda-microvm-service-feedback.md +++ b/docs/verification/645-lambda-microvm-service-feedback.md @@ -98,7 +98,12 @@ reached cannot write its own application diagnostic. **Our improvement:** [correlated lifecycle diagnostics](./645-p3-lifecycle-diagnostics.md) now record hook entry/stage/result, PID, AWS receipts and coordinator state -changes. These passed isolated AWS verification. Normal-stack rollout is pending. +changes. These passed isolated AWS verification and are +[deployed in coordinator 6 / image 5.0](./645-p3-diagnostics-rollout-20260916.md). +Two subsequent controlled missing-approval failures reached the guest, logged +the failed identity-read stage, returned HTTP 409 and surfaced specific failure +guidance through the deployed task API. This distinguishes an application +rejection from F01's missing hook-entry evidence. **Ask:** provide a service-side lifecycle attempt timeline or equivalent structured fields: originating API receipt, hook kind/attempt ID, start/end @@ -140,6 +145,16 @@ failures followed suspension by less than five seconds. See the [transport control](./645-p3-transport-control-20260916.md) and [process-observer record](./645-p3-process-observer-20260916.md). +The [September 16 AWS comparison](./645-p3-diagnostics-rollout-20260916.md) +kept the original server as PID 1 and compared normal image 5.0 with a private +image differing only by `--timeout-keep-alive 0`. Quick approval wakes and +measured 10.205 / 8.697-second suspended holds passed on the respective images. +Both missing-approval controls reached the expected HTTP 409. Normal quick +cases reused the same client port; the longer normal wake and the connection-close +cases used fresh ports. No unexpected refusal or reset appeared. +Two original mistitled long-hold attempts are retained and excluded from that +acceptance; fresh corrected cases supplied the stated durations. + **Ask:** does the service reuse hook TCP connections across suspend/resume, honor `Connection: close`, and retry a failed reused connection on a fresh socket? How are reset, refused and timeout errors classified? What ordering is guaranteed @@ -147,9 +162,10 @@ between guest unfreeze, network restoration and hook delivery? Which clock semantics should guest timeout/keep-alive timers expect across suspension? **Limits:** matching peer ports suggest reuse but do not establish all transport -behavior. A paused Linux process is not an AWS MicroVM restore. Connection-close -behavior is a proposed comparison, not an established fix. Service response: -pending. +behavior. A paused Linux process is not an AWS MicroVM restore. The bounded +AWS comparison establishes working paths with both settings, not an explanation +or fix for F01. Production connection handling remains unchanged. Service response: +pending; retain the service contract questions above. ## F06 — Make the VPC connector role requirement obvious before deployment diff --git a/docs/verification/645-p3-diagnostics-rollout-20260916.md b/docs/verification/645-p3-diagnostics-rollout-20260916.md new file mode 100644 index 000000000..94270ad6e --- /dev/null +++ b/docs/verification/645-p3-diagnostics-rollout-20260916.md @@ -0,0 +1,159 @@ +# ADR-021 P3: diagnostics rollout and connection comparison + +Updated 2026-09-16 UTC. The normal development stack now runs the +[lifecycle diagnostics and wake feedback](./645-p3-lifecycle-diagnostics.md). +Automatic suspension remains disabled. The four earlier +[wake refusals](./645-p3-resume-refusal-investigation.md) remain open. + +## Normal deployment + +| Item | Verified value | +|---|---| +| Account / region | `` / `us-west-2` | +| Stack | `backgroundagent-dev`, `UPDATE_COMPLETE` | +| Source | `dff860c63a7e8c211d493d3630c38907497c157a` | +| Coordinator `live` version | `6`; version `5` remains available | +| Coordinator code SHA-256 | `4EJoQ2SGpuPv/vlT7Za3peYIrL3DBKznzK6IyA+9EiU=` | +| Managed worker image | `backgroundagent-dev-abca-agent:5.0`, `SUCCESSFUL` / `ACTIVE` | +| Worker artifact SHA-256 | `9b3150e9e5991cc9fcc8a4adbb2cfbd8f97399f0c16baa9c2b5fe5b101e7b035` | +| Bootstrap / guardrail version | `1.8.0` / `3` | +| Root resources | 475 | +| Static suspension setting / live SSM switch | Both `false` | + +The deployment was verified at **14:54:31 UTC**. Coordinator versions 5 and 6 +have identical environments. An older durable execution can continue to use its +original retained coordinator version. + +The new worker artifact has 111 files and is 486,428 bytes. Relative to image +4.0, it adds `microvm_diagnostics.py` and changes only the three lifecycle +modules. The Dockerfile, startup command, dependencies and hook settings are +unchanged. + +The reviewed CloudFormation change set had 34 changes: + +- 29 Lambda code updates for consumers of the shared diagnostics/classifier. +- The image artifact URI and its build role's exact artifact permission. +- A new coordinator version, alias update and retained previous version. + +No existing resource required replacement. Function environments, database, +network, guardrail and suspension settings were unchanged. Each of the 29 +deployed functions was `Active` / `Successful`, and its `CodeSha256` matched +the reviewed S3 ZIP bytes. + +The assembly preserves the exact previous S3 deployment template and substitutes +the reviewed code/image changes. This avoids the previously recorded unrelated +asset and guardrail-version churn during full synthesis. The full synthesis, +restricted assembly, AWS change set and per-function hash audit are retained in +private evidence. + +## Controlled connection comparison + +An HTTP connection is a communication line. Uvicorn normally keeps that line +open briefly so the next request can reuse it. The private comparison image adds +`--timeout-keep-alive 0`, which closes it after the response. This tests one +possible explanation for the earlier failures; it is not a production fix. + +The private image is `backgroundagent-dev-p3-no-keepalive-20260916:1.0`. +Its artifact SHA-256 is +`0fb02e8cf6935d62abd4622b22791a0213fbf1cb24c31f209fcef67dffd98922`. +All 111 filenames and every file except that one Dockerfile command match the +normal image artifact. The original server command remains PID 1 in both. + +Six fixed, owned tasks compare an immediate wake, a wake after at least eight +seconds observed suspended, and a missing-approval rejection on each image. +Each task requests one read of `/etc/os-release`, with a $1 / six-turn limit, +600-second original approval window and 1,800-second worker ceiling. +Two fresh replacement tasks were needed after the watcher failed to apply the +longer hold to the original pair; eight workers actually ran. The original +mistitled results and watcher are retained. + +The private durable coordinator uses the compiled production implementation. +Version 1 pins normal image 5.0; version 2 changes only its image identifier and +version to the comparison image. Both have code SHA-256 +`v3IUVmZa7U1gAj4l8ytBhfJlL19hbimNOYisgh7rB/Q=`. +Versions 3 and 4 expand the fixed task whitelist for the two replacements, +with code SHA-256 `E+I/xFbrwlnu1U7tOkfyK0spJRW7WtMl4SjrDhkgg+I=`; +they pin the comparison and normal images respectively. The production +supervisor implementation is identical between these private versions. +One-second verification polling and fixed task/configuration wrappers are +explicit test differences from normal task routing. + +The private approval handler commits the real production decision but omits its +immediate wake call identically for both images. The coordinator therefore issues +the sole Resume. Its role is restricted to the eight task/rate keys and its own +logs; it has no MicroVM API permission. Missing-approval cases conditionally +delete only their own still-pending approval after checking task ownership and +the original gate. They do not call the approval API. + +## Results and limits + +The strict lifecycle audit passed at **15:12:51 UTC**. The independent tool and +deployed API feedback audit also passed. + +| Required case | Measured hold after observing suspended | Suspend / resume client ports | Resume HTTP | AWS Resume request ID | +|---|---:|---|---|---| +| Normal, quick | 1.527 s | 54904 / 54904 | 200 | `0dfd1993-5fba-4758-8d58-16ee6aa52e20` | +| Close connections, quick | 0.554 s | 60656 / 60664 | 200 | `b3581707-7432-46af-80d5-8c1f3597d2b0` | +| Normal, longer pause | 10.205 s | 58984 / 56118 | 200 | `c7e545d9-0342-4805-97e1-3c7eb4c4475f` | +| Close connections, longer pause | 8.697 s | 58984 / 56118 | 200 | `32df71a5-f150-40e0-81a6-69c1466f8a54` | +| Normal, missing approval | 0.236 s | 37364 / 37364 | 409 | `d102840b-2d38-49a7-a448-01d1808dc505` | +| Close connections, missing approval | 0.227 s | 58984 / 56118 | 409 | `424ce2c0-0dbb-4a4b-b5ec-f7948041a75c` | + +These holds measure from the watcher's first suspended observation to the +decision/deletion receipt. They are not exact service freeze durations. +Canonical lifecycle access logs supplied the client ports. The normal quick +cases reused the same port; the longer normal case and all close-connection +cases used different ports. + +Every worker retained server PID 1 and had exactly one coordinator-issued +Suspend and Resume acknowledgment with an AWS request ID. Successful workflows +produced 32 guest diagnostic records. Each missing-approval case produced 30: +credential refresh succeeded, `resume-identity-read` failed with +`LifecycleUnavailable`, and the hook returned HTTP 409 with +`MICROVM_LIFECYCLE_UNAVAILABLE`. AWS independently reported the HTTP 409; +neither rejection was a connection refusal. + +Both failed tasks returned `MICROVM_RESUME_HOOK_FAILED` through the normal +deployed GetTask handler, with the specific wake-failure title, diagnostic steps +and `retryable: false`. This used direct Lambda invocation with the fixture's +identity, not API Gateway authentication. + +All eight tasks released their capacity reservation, returned their counter to +zero, terminated their worker and emptied their launch prefix before independent +cleanup. Six tasks completed, each with exactly one successful Read result in +task events and one Read call in a complete retained trace with zero dropped +records. The two rejected tasks have one gated attempt and no tool-result event. +They did not publish a final trajectory before service termination; their event +records and failed guest barrier are the available evidence. + +The original `default-long` and `closed-long` attempts actually waited only +0.361 and 0.474 seconds. The audit rejected their intended hold requirement. +The watcher was corrected before the missing-approval tests and two fresh +long-pause tasks were run. Both original attempts remain documented successful +quick wakes and **do not count as long-pause acceptance**. + +No unexpected connection refusal or reset was reproduced. These controls show +that both connection settings can work; they do not establish a failure rate, +identify the cause of the four original failures, or justify changing production +connection handling. Service-side restore/transport evidence or a fresh failure +with independent process/listener evidence remains necessary. + +## Cleanup and retained evidence + +Cleanup completed at **15:15:37 UTC** after confirming all eight owned workers +were terminated, reservations released, counters zero, payload prefixes empty +and private durable executions finished. It archived all three private log +groups before removing both private functions and every version, the comparison +image, three roles, private SSM switch, artifact and log groups. The eight +synthetic counters were removed only while their count and reservation version +still matched. Explicit absence checks passed. + +Task/approval history, task events and trace objects retain their normal +retention. Private evidence also contains the exact source/artifacts, original +failed audit, corrected watcher, all task histories, six complete traces, guest +logs, service receipts and cleanup proof. + +The normal image, artifact and log group were excluded from cleanup. A final +read confirmed `UPDATE_COMPLETE`, coordinator `live:6`, image `5.0` active and +the suspension switch still false. These results complete the diagnostic +rollout and bounded comparison, not the remaining P3 acceptance matrix. diff --git a/docs/verification/645-p3-implementation-plan.md b/docs/verification/645-p3-implementation-plan.md index 9c8178963..f177debdc 100644 --- a/docs/verification/645-p3-implementation-plan.md +++ b/docs/verification/645-p3-implementation-plan.md @@ -244,8 +244,14 @@ handoff, use this order: The [lifecycle diagnostics guide](./645-p3-lifecycle-diagnostics.md) describes the new hook-stage, AWS request-ID and durable state-change logging. Three isolated AWS workflows verified it, including actual API wake and - coordinator recovery; normal-stack rollout remains pending. The original + coordinator recovery. The [normal-stack rollout](./645-p3-diagnostics-rollout-20260916.md) + now runs coordinator version 6 and image 5.0 with both suspension switches off. The original server remained PID 1. Logging and successful controls do not close the defect. + The same record covers a bounded AWS comparison with connection reuse disabled: + both quick/long wakes and both missing-approval HTTP 409 controls passed, + with specific guest-stage diagnostics and deployed task-API feedback. + Two mistitled long-hold attempts were excluded and replaced by fresh measured + cases. No unexpected refusal appeared; production connection handling is unchanged. 2. Complete the remaining live race/fault matrix: cancellation during transitions, late decision races, repeated polling/credential-refresh failures, durable registration races, service token-retention expiry and recovery of a worker diff --git a/docs/verification/645-p3-lifecycle-diagnostics.md b/docs/verification/645-p3-lifecycle-diagnostics.md index 08bee7110..07f27117d 100644 --- a/docs/verification/645-p3-lifecycle-diagnostics.md +++ b/docs/verification/645-p3-lifecycle-diagnostics.md @@ -1,7 +1,8 @@ # ADR-021 P3: lifecycle diagnostics Updated 2026-09-16. The logging changes passed local checks and three isolated -AWS workflows. Normal-stack rollout remains pending. They do not resolve the four +AWS workflows. The [normal-stack rollout](./645-p3-diagnostics-rollout-20260916.md) +now runs coordinator version 6 and image 5.0. They do not resolve the four [recorded wake refusals](./645-p3-resume-refusal-investigation.md). ## What the records tell us @@ -113,7 +114,7 @@ and saved progress before starting a replacement. It does not promise that retrying repairs the fault. Persisted classification codes take priority over diagnostic words; unrecognized AWS wording keeps the generic terminal code. -## Verification and next deployment +## Verification and deployment Local regressions cover hook correlation, sensitive-text exclusion, safe AWS identifiers, logging-sink failure, callback timeout/late completion, durable @@ -167,7 +168,6 @@ evidence. Cleanup verified removal of both private functions (all versions), the diagnostic image, three roles, the private SSM switch, three log groups, the artifact object and three owned zero-valued counters. Task/approval history and trace objects retain their normal retention. -Normal tasks still use coordinator **5** and image **4.0**; they do not yet carry -this new instrumentation. Roll out the checked code before relying on these -records for normal-stack diagnosis. Automatic suspension remains disabled until -the remaining P3 gates pass. +The subsequent [normal-stack rollout](./645-p3-diagnostics-rollout-20260916.md) +deployed coordinator **6** and image **5.0**, including this instrumentation. +Automatic suspension remains disabled until the remaining P3 gates pass. diff --git a/docs/verification/645-p3-resume-refusal-investigation.md b/docs/verification/645-p3-resume-refusal-investigation.md index 090359468..b93fa7752 100644 --- a/docs/verification/645-p3-resume-refusal-investigation.md +++ b/docs/verification/645-p3-resume-refusal-investigation.md @@ -25,8 +25,9 @@ approval workflow successful. The [lifecycle diagnostics guide](./645-p3-lifecycle-diagnostics.md) documents the new logging and wake-failure feedback. Three isolated AWS workflows verified the instrumentation with the original server running as PID 1, including actual -API wake and coordinator recovery. None reproduced refusal. Normal-stack rollout -remains pending; the historical failures below predate this instrumentation. +API wake and coordinator recovery. None reproduced refusal. The +[normal-stack rollout](./645-p3-diagnostics-rollout-20260916.md) now runs coordinator +version 6 and image 5.0; the historical failures below predate this instrumentation. ## Recorded failures @@ -123,6 +124,14 @@ all eight fresh-connection checks succeeded and the servers remained alive. This is not a reproduction of the AWS refusal. It supplies a specific comparison for the next cloud investigation without establishing a production fix. +The [AWS connection comparison](./645-p3-diagnostics-rollout-20260916.md) has now +tested normal image 5.0 against a private image with HTTP connection reuse +disabled, keeping the original server as PID 1. Both quick and longer wakes +passed, and both missing-approval controls produced the expected guest HTTP 409 +with a failed identity-read stage. The longer normal wake used a fresh client +port; quick normal wakes reused one. No refusal was reproduced. These results +do not identify F01's cause or justify a production connection-setting change. + 1. Use the recorded worker IDs, region, timestamps and Resume request IDs to inspect service-side lifecycle diagnostics. Determine the actual connection error and whether the request reached the guest, including any transport retry. From e17a4a523d081154935e6d3bf3348e68bbef9a1e Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Wed, 16 Sep 2026 12:02:27 -0400 Subject: [PATCH 051/149] fix: explain MicroVM supervision read failures (#645) --- cdk/src/handlers/cancel-task.ts | 11 +++---- cdk/src/handlers/shared/error-classifier.ts | 16 ++++++++++ .../handlers/shared/error-classifier.test.ts | 30 +++++++++++++++++++ 3 files changed, 50 insertions(+), 7 deletions(-) diff --git a/cdk/src/handlers/cancel-task.ts b/cdk/src/handlers/cancel-task.ts index b8802aade..d0db27f0a 100644 --- a/cdk/src/handlers/cancel-task.ts +++ b/cdk/src/handlers/cancel-task.ts @@ -89,10 +89,8 @@ export async function handler(event: APIGatewayProxyEvent): Promise { expect(result.retryable).toBe(true); }); + test.each(['substrate-read-failed', 'substrate-read-failed-repeatedly'])( + 'gives operator guidance for supervisor %s without promising cleanup succeeded', (reason) => { + const result = classifyError(`MicroVM supervisor: ${reason}`)!; + expect(result).toMatchObject({ + category: ErrorCategory.COMPUTE, + title: 'The MicroVM status could not be checked', + retryable: false, + errorClass: ErrorClass.SERVICE, + }); + expect(result.remedy).toContain('microvm_supervisor_request_failed'); + expect(result.remedy).toContain('AWS request ID'); + expect(result.remedy).toContain('Confirm worker termination'); + }, + ); + + test.each([ + 'AgentCore supervisor: substrate-read-failed', + 'Agent output mentions MicroVM supervisor: substrate-read-failed', + 'MicroVM supervisor: substrate-read-failed-unrecognized', + ])('does not infer a supervisor failure from unrelated text: %s', (message) => { + expect(classifyError(message)!.category).toBe(ErrorCategory.UNKNOWN); + }); + + test('a persisted terminal code still takes precedence over supervisor wording', () => { + const result = classifyError('MICROVM_SUBSTRATE_TERMINATED: MicroVM supervisor: substrate-read-failed')!; + expect(result.title).toBe('The MicroVM stopped before the agent reported a result'); + expect(result.retryable).toBe(true); + }); + test('every new MicroVM classification carries a full, non-empty guidance shape', () => { const messages = [ 'Session start failed: UnknownEndpoint: Inaccessible host: `lambda.eu-central-1.amazonaws.com\'', @@ -742,6 +771,7 @@ describe('classifyError', () => { 'MicroVM RunMicrovm failed: ThrottlingException: Rate exceeded', 'MicroVM RunMicrovm failed: ResourceNotFoundException: image not found', 'MicroVM substrate terminated before the agent wrote a terminal status: substrate state completed', + 'MicroVM supervisor: substrate-read-failed', reconciled(hookReason(400)), ]; for (const msg of messages) { From 2fef9cb9e90d229f740754e543a65873d0cc39ce Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Wed, 16 Sep 2026 12:17:09 -0400 Subject: [PATCH 052/149] docs: record MicroVM race checks and feedback rollout (#645) --- ...ADR-021-lambda-microvms-compute-backend.md | 2 +- ...Adr-021-lambda-microvms-compute-backend.md | 2 +- .../645-lambda-microvm-service-feedback.md | 37 ++++ .../645-p3-command-races-20260916.md | 195 ++++++++++++++++++ .../645-p3-implementation-plan.md | 14 +- 5 files changed, 245 insertions(+), 5 deletions(-) create mode 100644 docs/verification/645-p3-command-races-20260916.md diff --git a/docs/decisions/ADR-021-lambda-microvms-compute-backend.md b/docs/decisions/ADR-021-lambda-microvms-compute-backend.md index 45ba527fd..6c0ce5527 100644 --- a/docs/decisions/ADR-021-lambda-microvms-compute-backend.md +++ b/docs/decisions/ADR-021-lambda-microvms-compute-backend.md @@ -2,7 +2,7 @@ > **Implementation status (2026-09-16):** P1 ([#689](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/689)) and P2 ([#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733)) are merged. The takeover has a [clean deployment](../verification/645-p2-clean-deployment-20260913.md), [live repository workflow evidence](../verification/645-p2-live-task-20260914.md), and integrated P3 guest hooks, credentials, durable intent and approval/supervisor handling. The current normal deployment uses agent image **5.0**, coordinator **6** and bootstrap **1.8.0**. [Long-sleep credential/deadline acceptance](../verification/645-p3-callback-live-20260915.md) and the [pending-wake timer corrections](../verification/645-p3-pending-wake.md) passed live checks. > -> **P3 remains incomplete; automatic suspension is disabled.** Four [resume connection refusals](../verification/645-p3-resume-refusal-investigation.md) remain unexplained, alongside remaining race, permission/network, recovery and rollout checks. New [lifecycle diagnostics and wake feedback](../verification/645-p3-lifecycle-diagnostics.md) passed three isolated AWS workflows and are [deployed to the normal stack](../verification/645-p3-diagnostics-rollout-20260916.md). Follow the [current implementation plan](../verification/645-p3-implementation-plan.md) and [service-team feedback tracker](../verification/645-lambda-microvm-service-feedback.md). This ADR defines P1–P3, not a P4. +> **P3 remains incomplete; automatic suspension is disabled.** Four [resume connection refusals](../verification/645-p3-resume-refusal-investigation.md) remain unexplained, alongside remaining race, permission/network, recovery and rollout checks. New [lifecycle diagnostics and wake feedback](../verification/645-p3-lifecycle-diagnostics.md) passed isolated AWS workflows and are [deployed to the normal stack](../verification/645-p3-diagnostics-rollout-20260916.md). Six further [live command-race cases](../verification/645-p3-command-races-20260916.md) now pass, including cancellation during suspension/restore and bounded polling failure. Follow the [current implementation plan](../verification/645-p3-implementation-plan.md) and [service-team feedback tracker](../verification/645-lambda-microvm-service-feedback.md). This ADR defines P1–P3, not a P4. **Status:** proposed **Date:** 2026-07-29 diff --git a/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md b/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md index 3352bd3fa..fe841a7d4 100644 --- a/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md +++ b/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md @@ -6,7 +6,7 @@ title: Adr 021 lambda microvms compute backend > **Implementation status (2026-09-16):** P1 ([#689](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/689)) and P2 ([#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733)) are merged. The takeover has a [clean deployment](/sample-autonomous-cloud-coding-agents/architecture/645-p2-clean-deployment-20260913), [live repository workflow evidence](/sample-autonomous-cloud-coding-agents/architecture/645-p2-live-task-20260914), and integrated P3 guest hooks, credentials, durable intent and approval/supervisor handling. The current normal deployment uses agent image **5.0**, coordinator **6** and bootstrap **1.8.0**. [Long-sleep credential/deadline acceptance](/sample-autonomous-cloud-coding-agents/architecture/645-p3-callback-live-20260915) and the [pending-wake timer corrections](/sample-autonomous-cloud-coding-agents/architecture/645-p3-pending-wake) passed live checks. > -> **P3 remains incomplete; automatic suspension is disabled.** Four [resume connection refusals](/sample-autonomous-cloud-coding-agents/architecture/645-p3-resume-refusal-investigation) remain unexplained, alongside remaining race, permission/network, recovery and rollout checks. New [lifecycle diagnostics and wake feedback](/sample-autonomous-cloud-coding-agents/architecture/645-p3-lifecycle-diagnostics) passed three isolated AWS workflows and are [deployed to the normal stack](/sample-autonomous-cloud-coding-agents/architecture/645-p3-diagnostics-rollout-20260916). Follow the [current implementation plan](/sample-autonomous-cloud-coding-agents/architecture/645-p3-implementation-plan) and [service-team feedback tracker](/sample-autonomous-cloud-coding-agents/architecture/645-lambda-microvm-service-feedback). This ADR defines P1–P3, not a P4. +> **P3 remains incomplete; automatic suspension is disabled.** Four [resume connection refusals](/sample-autonomous-cloud-coding-agents/architecture/645-p3-resume-refusal-investigation) remain unexplained, alongside remaining race, permission/network, recovery and rollout checks. New [lifecycle diagnostics and wake feedback](/sample-autonomous-cloud-coding-agents/architecture/645-p3-lifecycle-diagnostics) passed isolated AWS workflows and are [deployed to the normal stack](/sample-autonomous-cloud-coding-agents/architecture/645-p3-diagnostics-rollout-20260916). Six further [live command-race cases](/sample-autonomous-cloud-coding-agents/architecture/645-p3-command-races-20260916) now pass, including cancellation during suspension/restore and bounded polling failure. Follow the [current implementation plan](/sample-autonomous-cloud-coding-agents/architecture/645-p3-implementation-plan) and [service-team feedback tracker](/sample-autonomous-cloud-coding-agents/architecture/645-lambda-microvm-service-feedback). This ADR defines P1–P3, not a P4. **Status:** proposed **Date:** 2026-07-29 diff --git a/docs/verification/645-lambda-microvm-service-feedback.md b/docs/verification/645-lambda-microvm-service-feedback.md index a15e544dd..9d380c4b6 100644 --- a/docs/verification/645-lambda-microvm-service-feedback.md +++ b/docs/verification/645-lambda-microvm-service-feedback.md @@ -16,6 +16,7 @@ is the receipt that lets the service team find a particular call. | F04 | Medium | `PENDING` also means restoring an existing worker | Observed live; our timer bug is fixed | | F05 | Medium | HTTP connection handling across suspend/resume | Contract question; no established cause of F01 | | F06 | Medium | Conditional operator-role requirement for VPC connectors | Earlier deployment failure; application setup fixed | +| F07 | P2/P3 acceptance gap | Run token retention and recovery without a worker ID | Replay observed through roughly five minutes; maximum retention and post-expiry behavior unknown | ## F01 — Wake request accepted, then the hook connection is refused @@ -105,6 +106,13 @@ the failed identity-read stage, returned HTTP 409 and surfaced specific failure guidance through the deployed task API. This distinguishes an application rejection from F01's missing hook-entry evidence. +The [command-race follow-up](./645-p3-command-races-20260916.md) additionally +records a checkpoint transaction rejected after cancellation, cancellation during +observed `SUSPENDING` / restore `PENDING`, and three consecutive status-read +failures. The latter now has specific platform-error guidance deployed in +coordinator 7 and verified through the normal task API. These controlled failures +and passing race cases do not explain F01. + **Ask:** provide a service-side lifecycle attempt timeline or equivalent structured fields: originating API receipt, hook kind/attempt ID, start/end times, connection versus HTTP failure, HTTP status, underlying error code and @@ -185,6 +193,35 @@ Clarify whether a service-linked role is ever an alternative for this connector. is a request for clearer conditional validation and setup guidance, not a claim that every connector requires the same role. Service response: pending. +## F07 — Specify Run token retention and recovery after a lost reply + +A client token is an order number: repeating the same launch request with that +number should recover the original worker instead of ordering another one. + +**Evidence:** the [start/recovery probes](./645-p2-start-recovery-live-20260914.md) +verified simultaneous identical calls, changed-request rejection and replay of a +terminated worker through roughly five minutes. A replayed Run response could +still report `PENDING`; a separate Get correctly reported the worker's current +state. The installed SDK documents idempotency but gives no token-retention +duration. ABCA therefore stops automatic recovery without a saved handle after +its own conservative 120-second deadline. + +**Impact:** if AWS created a worker but its response was lost, an operator needs +to find that exact worker. An undocumented retention boundary prevents proving +that a late retry cannot create another one. ABCA must not invent a new token +or automatically submit a replacement task to hide the uncertainty. + +**Ask:** what is the guaranteed token-retention period, including after worker +termination or image-version changes? After expiry, is the same token rejected +or treated as a new launch? Which supported API, event or audit field maps the +original token/API receipt to the worker ID when the client never received it? +How long does that mapping remain available? + +**Limits:** five minutes of successful replay does not establish the maximum +retention period. The 120-second limit is ABCA policy, not an AWS guarantee. +Post-expiry behavior and recovery of a genuinely unknown worker ID remain +unverified. Service response: pending. + ## Updating this tracker For each new finding, add the actual trigger, UTC window, region, worker/image diff --git a/docs/verification/645-p3-command-races-20260916.md b/docs/verification/645-p3-command-races-20260916.md new file mode 100644 index 000000000..031f31c31 --- /dev/null +++ b/docs/verification/645-p3-command-races-20260916.md @@ -0,0 +1,195 @@ +# ADR-021 P3: live cancellation, approval and polling-failure races + +Six required AWS cases passed on September 16. They cover cancellation around +sleep/wake commands, directly observed `SUSPENDING` and restore `PENDING` states, +approval during an accepted suspension, and three consecutive failed status +reads. Nine workers were used: three harness-invalidated attempts were retained, +excluded and replaced. + +**P3 remains incomplete.** These checks do not explain the four original +[resume connection refusals](./645-p3-resume-refusal-investigation.md). +Normal automatic suspension remains disabled. + +## Test boundary + +- Account ``, profile `sphia-dev`, region `us-west-2`. +- Normal worker image `backgroundagent-dev-abca-agent:5.0`, artifact SHA-256 + `9b3150e9e5991cc9fcc8a4adbb2cfbd8f97399f0c16baa9c2b5fe5b101e7b035`. +- Temporary durable coordinator `backgroundagent-dev-p3-races-20260916`, + with an exact task whitelist, dedicated role and private suspension switch. +- Compiled production handler and supervisor, plus a private wrapper that + invokes the normal approve/cancel handlers at specific SDK command boundaries + or deliberately hides selected successful Get responses as `TimeoutError`. +- No guest changes. Observed lifecycle hooks ran as PID 1. Each task allowed + one harmless `Read` of `/etc/os-release`, six turns and a $1 budget; worker + and durable execution lifetimes were capped at 1,800 seconds. +- Original approval windows were 600 seconds, except the restore-cancellation + case's 150 seconds. Its normal pre-deadline wake occurred while approval was + still pending. +- Decision/GetTask handlers received fixed synthetic owner events through + direct Lambda invocation. This checks their deployed code and permissions, + not API Gateway authentication. + +The wrapper records the real service receipt and the decision invocation before +returning control to the supervisor. Thus a test proves which command boundary +was crossed. A separate AWS state read records the state actually observed; +an accepted Suspend request alone does not prove `SUSPENDING`. + +## Required results + +| Case | Task | Observed state at intervention | Result | +|---|---|---|---| +| Cancel after pre-command read, before Suspend submission | `01M2NDHS2AC0569CXMJGA5ZR6Z` | `RUNNING` | Cancel committed; subsequent real Suspend rejected; task stayed canceled | +| Cancel after Suspend acceptance, before supervisor post-command read | `01M2NDHS2BYXTGNKVTW1MMHSRJ` | `RUNNING` | Cancel committed; checkpoint refused the changed task; cleanup passed | +| Cancel after Resume acceptance | `01M2NDHS2B73V9S034FBFFE7RM` | Restore `PENDING` | Task stayed canceled; no tool result | +| Approve after Suspend acceptance | `01M2NE0YFTXKRAVH2XPW702K2N` | `RUNNING`; later observed sleep/restore | Same gate resumed; approved Read succeeded once | +| Cancel during observed suspension | `01M2NE0YG0PT357AS5K2MVD33G` | `SUSPENDING` | Task stayed canceled; no tool result | +| Three failed status reads while asleep | `01M2NENZNP5BE95TCSE96SYCJ3` | `SUSPENDED` for all three hidden responses | Failure count reached three; task failed and cleanup passed | + +All six reached terminal worker state, released their saved capacity reservation, +left a zero user counter and removed their launch/payload objects **before any +independent watcher cleanup**. Durable executions succeeded by completing their +expected task finalization; that does not mean the intentionally failed task +became successful. + +Strong final reads and logs retain the original approval creation time/window, +coordinator first-observation time and non-increasing worker lifetime deadline. +The approved case has one Read result containing `PRETTY_NAME` and a complete +trace with zero dropped records. Each negative case has one gated Read attempt +and no tool result. Those terminated tasks do not all have a final trajectory; +the retained task events and lifecycle/control evidence establish this narrower +claim. + +## Specific race evidence + +**Cancellation before submission.** Worker +`microvm-53ec3d4d-d6cc-3542-a910-9d00cc7c4ae0` entered the wrapper's Suspend +boundary at 15:36:35.245 UTC. Cancel returned 200 at 15:36:36.932, before the real +Suspend returned `ValidationException` at 15:36:38.788, receipt +`edccbf8a-436e-4e06-9fec-f3e287db9afa`. The supervisor's post-command read preserved +cancellation despite that service error. + +**Cancellation during checkpoint.** Worker +`microvm-dd64aab4-23f6-3714-91d7-25ea3d9d20b4` received Suspend acceptance at +15:37:50.256, receipt `f7d03f96-a405-4c39-a81d-68c97f324361`. Cancellation committed +before its checkpoint transaction failed with `TransactionCanceledException`, +receipt `7B2F9PRFP9JSUS95BPI0LNMPEVVV4KQNSO5AEMVJF66Q9ASUAAJG`. +The hook returned HTTP 503 at 15:37:50.431 and marked suspension ineligible. +No resume hook ran. This is an observed application checkpoint rejection after +cancellation, not an unexplained wake-connection failure. + +**Cancellation during restore.** Worker +`microvm-ece908cf-7fd0-385e-aff9-bd3878500bb8` received Resume acceptance at +15:41:51.965, receipt `528dbbed-3b34-43f4-9a9f-08332f78fd0e`. A fresh Get reported +`PENDING` at 15:41:52.245, receipt `3612d0cb-3f28-4676-b171-94849583ffaf`. +Cancel returned 200 at 15:41:52.809, before the wrapper returned at 15:41:52.814. +The 150-second approval remained pending; no approved tool ran. + +**Cancellation during suspension.** Worker +`microvm-1f1f49e0-217e-3c65-a44b-45ffe9fe3fd9` received Suspend acceptance at +15:56:03.561, receipt `4e767b18-89d1-4557-ab76-4437b51ab69f`. Get reported +`SUSPENDING` at 15:56:03.744, receipt `e5ca583d-7ab4-4c4b-aed1-ec114ebe5e81`. +Cancel returned 200 at 15:56:05.344, before the supervisor regained control. + +**Repeated polling failure.** Worker +`microvm-ad53c767-8fae-3f61-8b94-eb6ee0202e57` remained genuinely suspended while +the wrapper deliberately hid three real Get responses. Across consecutive +durable cycles, logs recorded `stage=substrate-read`, `error_type=TimeoutError` +and failure counts 1, 2 and 3. The final reason was +`MicroVM supervisor: substrate-read-failed`. The coordinator sent Terminate; +a fresh worker read confirmed `TERMINATED` at 15:54:25.536. There was no watcher +repair. + +## Rejected attempts and harness corrections + +These attempts are not included in the six-case acceptance count: + +1. Approval task `01M2NDHS2B6087T1925SRRPDSX` completed its Read, but the watcher + demanded `TERMINATED` while AWS still reported `TERMINATING`. Its fallback + cleanup ran. A fresh approval task supplied clean acceptance. +2. Poll task `01M2NDHS2BZYNEMHAQ7JDY41ZP` never received its intended faults: + esbuild renamed `GetMicrovmCommand` to `GetMicrovmCommand2`, defeating a + `constructor.name` comparison. It was explicitly canceled. The injector now + checks SDK command classes with `instanceof`. +3. Poll task `01M2NEBTY1CVS4B56ZJDVKS8CE` received all three intended faults and + the coordinator sent Terminate, but the watcher compared a `SUSPENDED` + sample taken **before** it read the completed durable execution. Fallback + cleanup invalidated that acceptance attempt. + +The final watcher observes a fresh terminal worker state for at most 30 seconds +after durable finalization. Task, worker and execution reads are separate calls, +so their values are not one atomic snapshot. This observation window sends no +repair commands. Original sources, logs, failures and replacement identities +remain in the private evidence. + +## Feedback correction and validation + +The polling test exposed an application feedback gap: the persisted supervisor +reason was classified as `unknown` / `user`, with the title “Unexpected error.” +Commit `d150991715c27dafecbe1eb4ab7ce44e98e02bff` gives those exact known reasons a +compute/platform classification and the title “The MicroVM status could not be +checked.” Guidance names `microvm_supervisor_request_failed`, the task/worker +IDs, stage, error type and AWS receipt, and requires checking termination and +saved progress before replacement. + +The same commit corrects two stale cancel-handler comments: the coordinator +saves the runtime ARN, and sleeping-worker AWS memory-quota consumption remains +unverified. The cancel handler's commentless TypeScript output is unchanged. + +Validation passed: 134 focused classifier tests and the full repository build, +including 5,025 CDK, 928 CLI and 1,947 Python tests; Python coverage was 86.43%. +The optional DynamoDB Local suites were skipped (56 CDK and 11 Python cases); +earlier transaction evidence remains separate. + +## Normal deployment of the feedback fix + +At 16:13:15.851 UTC, `backgroundagent-dev` was verified `UPDATE_COMPLETE`, +running coordinator alias `live → 7`, code SHA-256 +`/JcZuu12fMVSOoO4NwxsLGlLIJ5w9XYjAyx7cuBRA+I=`. Version 6 remains available, +and its environment matches version 7 exactly. Image 5.0, Cedar layer version 2 +and both disabled suspension settings were preserved. + +The reviewed change set contained 29 Lambda `Code` updates plus the retained +old/new coordinator version and alias changes: 32 changes, no replacements. +All 29 deployed code hashes matched the actual reviewed S3 ZIP contents +(409,965,183 bytes hashed). + +A direct invocation of the normal GetTask handler for the real failed poll +task returned the new title, compute/service classification, nonretryable flag +and specific supervisor-log/termination guidance. Existing saved task errors +therefore receive the corrected explanation without rewriting their records. + +One preview was rejected before execution. In this environment, the SDK's +`GetTemplate` result replaced every non-ASCII character with `?`, including `§` +and dash characters in descriptions. Reusing that result would have changed +unrelated descriptions/dashboard text and replaced the Cedar layer. The +replacement preview used the exact previously deployed S3 template, verified +against SHA-256 +`e9e58901d4fef62a9b9a07b24068b61cc2cf51849847c261df1f6ab7597b8698`. +AWS then reported only the intended 32 changes. The rejected preview and the +character-for-character loss comparison remain evidence; the underlying cause +within the retrieval path was not isolated. + +Deployment proof is under +`/tmp/abca-645-p2-clean-20260913/p3-race-feedback-rollout-20260916`. + +## Cleanup, evidence and remaining scope + +Cleanup was verified at 16:04:09.508 UTC. All nine workers were terminated. +The private function and its four versions, role/policies, suspension parameter +and log group were deleted; 1,236 function log records were archived first. +Nine zero counters were removed with revision checks. Task/event/trace audit +records remain. Immediate function-deletion observation lagged; subsequent +read-only checks confirmed all temporary resources absent. + +Private evidence is under +`/tmp/abca-645-p2-clean-20260913/p3-command-races-20260916`, including four exact +function bundles, the task whitelist/policies, run histories, hook/API logs, +strict audit, excluded attempts and cleanup proof. + +Still open: late approval versus timeout races, credential-refresh failure +injection, remaining durable registration/unknown-worker recovery, other-backend +permissions/networking, a final cloned-repository workflow, capacity migration +and compatible rollout checks. The [implementation plan](./645-p3-implementation-plan.md) +tracks their order. No original connection refusal occurred in this matrix; +that does not close its separate investigation. diff --git a/docs/verification/645-p3-implementation-plan.md b/docs/verification/645-p3-implementation-plan.md index f177debdc..0811a127b 100644 --- a/docs/verification/645-p3-implementation-plan.md +++ b/docs/verification/645-p3-implementation-plan.md @@ -252,9 +252,17 @@ handoff, use this order: with specific guest-stage diagnostics and deployed task-API feedback. Two mistitled long-hold attempts were excluded and replaced by fresh measured cases. No unexpected refusal appeared; production connection handling is unchanged. -2. Complete the remaining live race/fault matrix: cancellation during transitions, - late decision races, repeated polling/credential-refresh failures, durable - registration races, service token-retention expiry and recovery of a worker +2. The [command-race checks](./645-p3-command-races-20260916.md) now pass cancellation + before/after Suspend, during observed `SUSPENDING`, and during restore + `PENDING`; approval during an accepted suspension and three consecutive + polling failures also pass. Six required cases used nine workers, with three + harness-invalidated attempts explicitly excluded and replaced. All temporary + resources were removed. The same follow-up deployed clearer status-read + failure guidance in coordinator version 7 and verified the normal task API; + image 5.0 and disabled suspension settings remain in place. + Complete the remaining live race/fault matrix: + late decision races, credential-refresh failures, durable registration races, + service token-retention expiry and recovery of a worker whose ID was genuinely lost. Keep each injected failure distinct from an unrelated service failure. 3. Complete effective permissions and network checks for the other backends, From a2678495f77c7138293b83b7dc190fe467201e45 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Wed, 16 Sep 2026 12:49:47 -0400 Subject: [PATCH 053/149] feat(645): configure per-task MicroVM approval sleep delay --- cdk/src/handlers/shared/create-task-core.ts | 14 ++++++ .../shared/microvm-lifecycle-policy.ts | 13 +++++- cdk/src/handlers/shared/microvm-lifecycle.ts | 3 ++ cdk/src/handlers/shared/microvm-supervisor.ts | 11 ++++- cdk/src/handlers/shared/types.ts | 14 ++++++ .../constructs/agent-session-role.test.ts | 2 +- .../handlers/orchestrate-task-microvm.test.ts | 1 + .../handlers/shared/create-task-core.test.ts | 20 +++++++++ .../shared/microvm-lifecycle-local.test.ts | 1 + .../handlers/shared/microvm-lifecycle.test.ts | 45 +++++++++++++++++++ .../shared/microvm-supervisor.test.ts | 21 +++++++++ cli/src/commands/submit.ts | 19 ++++++++ cli/src/types.ts | 9 ++++ cli/test/commands/submit.test.ts | 22 +++++++++ contracts/constants.json | 5 +++ contracts/constants.md | 5 +++ docs/design/COMPUTE.md | 9 ++++ docs/guides/USER_GUIDE.md | 15 +++++++ docs/src/content/docs/architecture/Compute.md | 9 ++++ .../docs/using/Approval-gates-cedar-hitl.md | 16 ++++++- docs/src/content/docs/using/Using-the-cli.md | 1 + .../645-p3-implementation-plan.md | 4 +- docs/verification/645-p3-supervisor.md | 16 +++++++ 23 files changed, 268 insertions(+), 7 deletions(-) diff --git a/cdk/src/handlers/shared/create-task-core.ts b/cdk/src/handlers/shared/create-task-core.ts index de9a92135..6eccf203d 100644 --- a/cdk/src/handlers/shared/create-task-core.ts +++ b/cdk/src/handlers/shared/create-task-core.ts @@ -48,6 +48,9 @@ import { type CreateTaskRequest, createAttachmentRecord, INITIAL_APPROVALS_MAX_ENTRIES, + MICROVM_SLEEP_AFTER_S_DEFAULT, + MICROVM_SLEEP_AFTER_S_MAX, + MICROVM_SLEEP_AFTER_S_MIN, type InlineAttachment, type PresignedAttachment, type TaskRecord, @@ -336,6 +339,15 @@ export async function createTaskCore( approvalTimeoutS = body.approval_timeout_s; } + const microvmSleepAfterS = body.microvm_sleep_after_s === undefined + ? MICROVM_SLEEP_AFTER_S_DEFAULT : body.microvm_sleep_after_s; + if (typeof microvmSleepAfterS !== 'number' || !Number.isInteger(microvmSleepAfterS) + || microvmSleepAfterS < MICROVM_SLEEP_AFTER_S_MIN || microvmSleepAfterS > MICROVM_SLEEP_AFTER_S_MAX) { + return errorResponse(400, ErrorCode.VALIDATION_ERROR, + `Invalid microvm_sleep_after_s. Must be an integer between ${MICROVM_SLEEP_AFTER_S_MIN} ` + + `and ${MICROVM_SLEEP_AFTER_S_MAX} seconds (0 disables sleep).`, requestId); + } + // Cedar HITL — validate initial_approvals if supplied (§7.3 step 4). let initialApprovals: string[] | undefined; if (body.initial_approvals !== undefined) { @@ -811,6 +823,8 @@ export async function createTaskCore( // payload supplied them; ``approval_timeout_s`` defaults to the // engine default at agent runtime when absent here. ...(approvalTimeoutS !== undefined && { approval_timeout_s: approvalTimeoutS }), + // Capture the default so future deployments cannot change this task's preference. + microvm_sleep_after_s: microvmSleepAfterS, ...(initialApprovals !== undefined && { initial_approvals: initialApprovals }), // Persisted counter the stranded-approval reconciler + agent // counter both read (§13.6). Seeded to 0 at task-create time. diff --git a/cdk/src/handlers/shared/microvm-lifecycle-policy.ts b/cdk/src/handlers/shared/microvm-lifecycle-policy.ts index fafacee87..bdea8716a 100644 --- a/cdk/src/handlers/shared/microvm-lifecycle-policy.ts +++ b/cdk/src/handlers/shared/microvm-lifecycle-policy.ts @@ -20,10 +20,11 @@ import type { SessionStatus } from './compute-strategy'; import { supportsMicrovmLifecycle } from './microvm-image-capability'; import { intentMatchesGate, type MicrovmLifecycleSnapshot } from './microvm-lifecycle'; +import { MICROVM_SLEEP_AFTER_S_DEFAULT, MICROVM_SLEEP_AFTER_S_MAX } from './types'; import { TaskStatus, TERMINAL_STATUSES } from '../../constructs/task-status'; // Initial policy values, not service limits. Live timings must validate them. -export const MICROVM_SUSPEND_GRACE_MS = 30_000; +export const MICROVM_SUSPEND_GRACE_MS = MICROVM_SLEEP_AFTER_S_DEFAULT * 1000; export const MICROVM_WAKE_MARGIN_MS = 60_000; export const MICROVM_MIN_USEFUL_SLEEP_MS = 30_000; export const MICROVM_TRANSITION_POLL_MS = 5_000; @@ -91,6 +92,14 @@ export function decideMicrovmLifecycle(input: MicrovmLifecyclePolicyInput): Micr const wakeAt = Math.min(approval.deadlineMs, sessionDeadlineMs) - MICROVM_WAKE_MARGIN_MS; if (nowMs >= wakeAt) return wake('wake-deadline'); + const sleepAfterSeconds = snapshot.sleepAfterSeconds === undefined + ? MICROVM_SLEEP_AFTER_S_DEFAULT : snapshot.sleepAfterSeconds; + // A malformed stored preference loses savings, never wake or cleanup. + if (!Number.isInteger(sleepAfterSeconds) || sleepAfterSeconds <= 0 || sleepAfterSeconds > MICROVM_SLEEP_AFTER_S_MAX) { + return sleeping || snapshot.intent ? wake('task-sleep-disabled') + : { action: 'wait', reason: 'task-sleep-disabled', nextPollInMs }; + } + if (sleeping) { if (!sameGate || snapshot.intent?.action !== 'suspend' || snapshot.intent.deadline_ms !== approval.deadlineMs) { return wake('unintended-suspension'); @@ -103,7 +112,7 @@ export function decideMicrovmLifecycle(input: MicrovmLifecyclePolicyInput): Micr // A prior gate's in-flight suspend must be resolved conservatively. Persist a // wake for this gate rather than attributing that old sleep request to it. if (snapshot.intent?.action === 'suspend' && !sameGate) return wake('previous-gate-suspend'); - const graceEndsAt = approval.createdAtMs + MICROVM_SUSPEND_GRACE_MS; + const graceEndsAt = approval.createdAtMs + sleepAfterSeconds * 1000; if (nowMs < graceEndsAt) { return { action: 'wait', reason: 'suspend-grace', nextPollInMs: Math.min(nextPollInMs, graceEndsAt - nowMs, wakeAt - nowMs) }; } diff --git a/cdk/src/handlers/shared/microvm-lifecycle.ts b/cdk/src/handlers/shared/microvm-lifecycle.ts index e34d56e06..9c5d8593a 100644 --- a/cdk/src/handlers/shared/microvm-lifecycle.ts +++ b/cdk/src/handlers/shared/microvm-lifecycle.ts @@ -63,6 +63,8 @@ export interface MicrovmLifecycleSnapshot { /** Latest persisted guest heartbeat; used only to bound recovery liveness grace. */ readonly heartbeatAtMs?: number; readonly taskStartedAtMs?: number; + /** Persisted approval-wait preference; absent legacy values use the current default. */ + readonly sleepAfterSeconds?: number; } export type SaveLifecycleResult = @@ -170,6 +172,7 @@ export async function readMicrovmLifecycleSnapshot( requestId, approval, intent: task.microvm_lifecycle, + ...(task.microvm_sleep_after_s !== undefined && { sleepAfterSeconds: task.microvm_sleep_after_s }), ...(typeof task.agent_heartbeat_at === 'string' && Number.isSafeInteger(Date.parse(task.agent_heartbeat_at)) && { heartbeatAtMs: Date.parse(task.agent_heartbeat_at) }), ...(typeof task.started_at === 'string' && Number.isSafeInteger(Date.parse(task.started_at)) diff --git a/cdk/src/handlers/shared/microvm-supervisor.ts b/cdk/src/handlers/shared/microvm-supervisor.ts index 11e32ce1b..919232130 100644 --- a/cdk/src/handlers/shared/microvm-supervisor.ts +++ b/cdk/src/handlers/shared/microvm-supervisor.ts @@ -30,6 +30,7 @@ import { import { decideMicrovmLifecycle, MICROVM_TRANSITION_POLL_MS } from './microvm-lifecycle-policy'; import { readMicrovmSuspendEnabled } from './microvm-suspend-config'; import { MICROVM_MAX_DURATION_SECONDS } from './strategies/lambda-microvm-strategy'; +import { MICROVM_SLEEP_AFTER_S_DEFAULT, MICROVM_SLEEP_AFTER_S_MAX } from './types'; import { TaskStatus, TERMINAL_STATUSES, type TaskStatusType } from '../../constructs/task-status'; type MicrovmHandle = Extract; @@ -134,6 +135,7 @@ export async function superviseMicrovm(input: MicrovmSupervisorInput): Promise { if (outcome.kind === 'continue') { if (!readFailed) {state = { ...state, consecutivePollFailures: 0 };} else if (state.consecutivePollFailures >= MICROVM_MAX_POLL_FAILURES) { @@ -152,6 +154,10 @@ export async function superviseMicrovm(input: MicrovmSupervisorInput): Promise= 0 + && snapshot.sleepAfterSeconds <= MICROVM_SLEEP_AFTER_S_MAX ? snapshot.sleepAfterSeconds : null, consecutive_poll_failures: state.consecutivePollFailures, consecutive_resume_failures: state.consecutiveResumeFailures, }; @@ -314,6 +320,7 @@ export async function superviseMicrovm(input: MicrovmSupervisorInput): Promise; readonly ttl?: number; /** @@ -441,6 +443,8 @@ export interface TaskNotificationsConfig { * Strips internal fields not exposed in the API. */ export interface TaskDetail { + /** Configured MicroVM approval-wait delay; 0 disables sleep. Absent on legacy records. */ + readonly microvm_sleep_after_s?: number; readonly task_id: string; readonly status: TaskStatusType; /** ``null`` for a repo-less workflow (#248 Phase 3). */ @@ -734,6 +738,8 @@ export interface GetTaskEventsQuery { * Keep in sync with ``cli/src/types.ts``. */ export interface CreateTaskRequest { + /** MicroVM approval-wait seconds before sleep (0 = off, omitted = 600). Does not extend approval deadlines. */ + readonly microvm_sleep_after_s?: number; /** Target repository (``owner/repo``). Optional since #248 Phase 3: a * repo-less workflow (``requires_repo: false``) is submitted without it. * Required-ness is enforced conditionally in ``createTaskCore`` based on @@ -965,6 +971,9 @@ export function toTaskDetail( ): TaskDetail { const ctx = { task_id: record.task_id }; return { + ...(typeof record.microvm_sleep_after_s === 'number' && Number.isInteger(record.microvm_sleep_after_s) + && record.microvm_sleep_after_s >= MICROVM_SLEEP_AFTER_S_MIN && record.microvm_sleep_after_s <= MICROVM_SLEEP_AFTER_S_MAX + && { microvm_sleep_after_s: record.microvm_sleep_after_s }), task_id: record.task_id, status: record.status, repo: record.repo ?? null, @@ -1545,6 +1554,11 @@ export const APPROVAL_TIMEOUT_S_MAX = sharedConstants.approval_timeout_s.max; * Sourced from ``contracts/constants.json`` (S9). */ export const APPROVAL_TIMEOUT_S_DEFAULT = sharedConstants.approval_timeout_s.default; +/** Per-task MicroVM sleep delay bounds; zero disables automatic sleep. */ +export const MICROVM_SLEEP_AFTER_S_MIN = sharedConstants.microvm_sleep_after_s.min; +export const MICROVM_SLEEP_AFTER_S_MAX = sharedConstants.microvm_sleep_after_s.max; +export const MICROVM_SLEEP_AFTER_S_DEFAULT = sharedConstants.microvm_sleep_after_s.default; + /** * Cedar HITL: bounds + platform default for the per-task approval-gate cap * (design decision #13, §4 step 5). Blueprints may override via diff --git a/cdk/test/constructs/agent-session-role.test.ts b/cdk/test/constructs/agent-session-role.test.ts index cfd9fd938..fcea6e9b8 100644 --- a/cdk/test/constructs/agent-session-role.test.ts +++ b/cdk/test/constructs/agent-session-role.test.ts @@ -157,7 +157,7 @@ describe('AgentSessionRole construct', () => { expect(attrs).toEqual(taskWriteAttributes); expect(s.Condition.Null['dynamodb:Attributes']).toBe('false'); for (const protectedAttribute of [ - 'microvm_start', 'microvm_lifecycle', 'concurrency_slot', 'user_id', 'created_at', + 'microvm_start', 'microvm_lifecycle', 'microvm_sleep_after_s', 'concurrency_slot', 'user_id', 'created_at', 'session_id', 'compute_type', 'compute_metadata', 'agent_runtime_arn', ]) { expect(attrs).not.toContain(protectedAttribute); diff --git a/cdk/test/handlers/orchestrate-task-microvm.test.ts b/cdk/test/handlers/orchestrate-task-microvm.test.ts index 131713a38..bb0017c58 100644 --- a/cdk/test/handlers/orchestrate-task-microvm.test.ts +++ b/cdk/test/handlers/orchestrate-task-microvm.test.ts @@ -270,6 +270,7 @@ function failedTransition() { function lifecycle(status: MicrovmLifecycleSnapshot['status'] = 'RUNNING'): MicrovmLifecycleSnapshot { return { + sleepAfterSeconds: 30, taskId: 'TASK001', userId: 'user-1', status, diff --git a/cdk/test/handlers/shared/create-task-core.test.ts b/cdk/test/handlers/shared/create-task-core.test.ts index 00de18b9e..954724037 100644 --- a/cdk/test/handlers/shared/create-task-core.test.ts +++ b/cdk/test/handlers/shared/create-task-core.test.ts @@ -854,6 +854,26 @@ describe('createTaskCore', () => { expect(result.body).toContain('pr_number'); }); + test.each([undefined, 0, 1, 600, 3600])('persists and returns the resolved MicroVM sleep delay %s', async value => { + const result = await createTaskCore( + { repo: 'org/repo', task_description: 'wait preference', ...(value !== undefined && { microvm_sleep_after_s: value }) }, + makeContext(), 'req-sleep', + ); + expect(result.statusCode).toBe(201); + expect(getPersistedTaskRecord().microvm_sleep_after_s).toBe(value ?? 600); + expect(JSON.parse(result.body).data.microvm_sleep_after_s).toBe(value ?? 600); + }); + + test.each([-1, 3601, 0.5, '600', null, false, {}, []])('rejects invalid MicroVM sleep delay %j before creating a task', async value => { + const result = await createTaskCore( + { repo: 'org/repo', task_description: 'invalid delay', microvm_sleep_after_s: value } as any, + makeContext(), 'req-sleep-invalid', + ); + expect(result.statusCode).toBe(400); + expect(JSON.parse(result.body).error.message).toContain('microvm_sleep_after_s'); + expect(mockSend.mock.calls.some(([command]) => command._type === 'Put')).toBe(false); + }); + // -- trace flag (design §10.1) -------------------------------------- test('trace: true persists on the task record and surfaces in the response', async () => { diff --git a/cdk/test/handlers/shared/microvm-lifecycle-local.test.ts b/cdk/test/handlers/shared/microvm-lifecycle-local.test.ts index 1ae7bd5e0..6e9230c62 100644 --- a/cdk/test/handlers/shared/microvm-lifecycle-local.test.ts +++ b/cdk/test/handlers/shared/microvm-lifecycle-local.test.ts @@ -103,6 +103,7 @@ local('MicroVM lifecycle against DynamoDB Local', () => { TableName: tasks, Item: { task_id: 'task', + microvm_sleep_after_s: 30, user_id: 'user', status: 'AWAITING_APPROVAL', compute_type: 'lambda-microvm', diff --git a/cdk/test/handlers/shared/microvm-lifecycle.test.ts b/cdk/test/handlers/shared/microvm-lifecycle.test.ts index 6a5ebbcdd..1270ec43c 100644 --- a/cdk/test/handlers/shared/microvm-lifecycle.test.ts +++ b/cdk/test/handlers/shared/microvm-lifecycle.test.ts @@ -39,6 +39,7 @@ const imageMetadata = { lifecycleProtocol: '1', }; const task = { + microvm_sleep_after_s: 30, task_id: 'task', user_id: 'user', status: 'AWAITING_APPROVAL', @@ -58,6 +59,7 @@ const intent = (action: 'suspend' | 'resume' = 'suspend'): MicrovmLifecycleInten deadline_ms: DEADLINE, }); const snapshot = (overrides: Partial = {}): MicrovmLifecycleSnapshot => ({ + sleepAfterSeconds: 30, taskId: 'task', userId: 'user', status: 'AWAITING_APPROVAL', @@ -83,6 +85,49 @@ beforeEach(() => { afterEach(() => jest.restoreAllMocks()); describe('MicroVM lifecycle policy', () => { + test('default waits ten minutes from the original gate creation before sleeping', () => { + const current = snapshot({ + sleepAfterSeconds: undefined, + approval: { + kind: 'present', + status: 'PENDING', + created_at: new Date(NOW).toISOString(), + createdAtMs: NOW, + timeout_s: 1800, + deadlineMs: NOW + 1_800_000, + }, + }); + expect(policy('RUNNING', { snapshot: current, nowMs: NOW + 599_000 })) + .toMatchObject({ action: 'wait', reason: 'suspend-grace', nextPollInMs: 1000 }); + expect(policy('RUNNING', { snapshot: current, nowMs: NOW + 600_000 })).toMatchObject({ action: 'suspend' }); + }); + test('a default five-minute gate stays awake and retains its original deadline', () => { + const current = snapshot({ + sleepAfterSeconds: undefined, + approval: { + kind: 'present', + status: 'PENDING', + created_at: new Date(NOW).toISOString(), + createdAtMs: NOW, + timeout_s: 300, + deadlineMs: NOW + 300_000, + }, + }); + expect(policy('RUNNING', { snapshot: current, nowMs: NOW + 200_000 })) + .toMatchObject({ action: 'wait', reason: 'suspend-grace', nextPollInMs: 30_000 }); + expect(policy('RUNNING', { snapshot: current, nowMs: NOW + 240_000 })) + .toMatchObject({ action: 'resume', reason: 'wake-deadline' }); + expect(current.approval).toMatchObject({ deadlineMs: NOW + 300_000 }); + }); + test.each([0, -1, NaN, Infinity, 0.5, 3601, '30', null])('off or malformed delay %s keeps awake and repairs existing sleep', value => { + const current = snapshot({ sleepAfterSeconds: value as number }); + expect(policy('RUNNING', { snapshot: current })).toMatchObject({ action: 'wait', reason: 'task-sleep-disabled' }); + expect(policy('SUSPENDED', { snapshot: { ...current, intent: intent() } })) + .toMatchObject({ action: 'resume', requestReady: true, reason: 'task-sleep-disabled' }); + expect(policy('SUSPENDING', { snapshot: { ...current, intent: intent() } })) + .toMatchObject({ action: 'resume', requestReady: false, reason: 'task-sleep-disabled' }); + expect(policy('SUSPENDED', { snapshot: { ...current, status: 'CANCELLED' } })).toMatchObject({ action: 'terminate' }); + }); test('long pending gate may suspend only after grace and explicit RUNNING', () => { expect(policy()).toEqual({ action: 'suspend', requestReady: true, reason: 'pending-long-gate', nextPollInMs: 5_000 }); expect(mockSend).not.toHaveBeenCalled(); diff --git a/cdk/test/handlers/shared/microvm-supervisor.test.ts b/cdk/test/handlers/shared/microvm-supervisor.test.ts index fdbbbdbf4..51cb36ad6 100644 --- a/cdk/test/handlers/shared/microvm-supervisor.test.ts +++ b/cdk/test/handlers/shared/microvm-supervisor.test.ts @@ -114,6 +114,7 @@ beforeEach(() => { jest.spyOn(Date, 'now').mockImplementation(() => time); row = { taskId: 'task', + sleepAfterSeconds: 30, userId: 'user', status: 'AWAITING_APPROVAL', handle, @@ -201,6 +202,26 @@ test.each(['disabled', 'legacy', 'unverified-lifetime', 'short-window'])('%s kee expect(mockSuspendEnabled).not.toHaveBeenCalled(); }); +test('per-task off overrides enabled deployment and records the reason', async () => { + row = { ...row, sleepAfterSeconds: 0 }; + expect((await run()).kind).toBe('continue'); + expect(strategy.suspendSession).not.toHaveBeenCalled(); + expect(mockLogger.info).toHaveBeenCalledWith('MicroVM supervisor observation changed', expect.objectContaining({ + sleep_after_s: 0, policy_reason: 'task-sleep-disabled', + })); +}); + +test('fresh task preference after intent commit prevents an in-flight suspend', async () => { + mockSave.mockImplementationOnce(async () => { + intent('suspend', time); + row = { ...row, sleepAfterSeconds: 0 }; + return { status: 'saved', intent: row.intent }; + }); + expect((await run()).kind).toBe('continue'); + expect(strategy.suspendSession).not.toHaveBeenCalled(); + expect(row.intent?.action).toBe('resume'); +}); + test('an existing opt-in execution observes live disable without failing healthy compute', async () => { working(); const first = await run(); diff --git a/cli/src/commands/submit.ts b/cli/src/commands/submit.ts index 8fd3feec1..947b29922 100644 --- a/cli/src/commands/submit.ts +++ b/cli/src/commands/submit.ts @@ -36,6 +36,9 @@ import { INITIAL_APPROVALS_MAX_ENTRY_LENGTH, MAX_BUDGET_USD_MAX, MAX_BUDGET_USD_MIN, + MICROVM_SLEEP_AFTER_S_DEFAULT, + MICROVM_SLEEP_AFTER_S_MAX, + MICROVM_SLEEP_AFTER_S_MIN, } from '../types'; import { exitCodeForStatus, waitForTask } from '../wait'; @@ -88,6 +91,11 @@ export function makeSubmitCommand(): Command { + 'Overrides the platform default of 300s. Per-rule @approval_timeout_s still min-wins at gate-firing.', parseInt, ) + .option( + '--microvm-sleep-after ', + `Sleep while waiting for approval after this many seconds (default ${MICROVM_SLEEP_AFTER_S_DEFAULT}; ` + + 'off keeps the worker awake). MicroVM only; requires platform sleep support and does not extend approval deadlines.', + ) .option( '--pre-approve ', 'Cedar HITL pre-approval scope to seed at task start (repeatable). ' @@ -150,6 +158,16 @@ export function makeSubmitCommand(): Command { ); } } + let microvmSleepAfterS: number | undefined; + if (opts.microvmSleepAfter !== undefined) { + const raw = opts.microvmSleepAfter as string; + microvmSleepAfterS = raw === 'off' ? 0 : /^\d+$/.test(raw) ? Number(raw) : NaN; + if (!Number.isSafeInteger(microvmSleepAfterS) + || microvmSleepAfterS < MICROVM_SLEEP_AFTER_S_MIN || microvmSleepAfterS > MICROVM_SLEEP_AFTER_S_MAX) { + throw new CliError('--microvm-sleep-after must be off or an integer between ' + + `${MICROVM_SLEEP_AFTER_S_MIN} and ${MICROVM_SLEEP_AFTER_S_MAX} seconds.`); + } + } const preApproveRaw = (opts.preApprove ?? []) as readonly string[]; let initialApprovals: readonly ApprovalScope[] | undefined; if (preApproveRaw.length > 0) { @@ -226,6 +244,7 @@ export function makeSubmitCommand(): Command { ...(prNumber !== undefined && { pr_number: prNumber }), ...(opts.trace && { trace: true }), ...(opts.approvalTimeout !== undefined && { approval_timeout_s: opts.approvalTimeout }), + ...(microvmSleepAfterS !== undefined && { microvm_sleep_after_s: microvmSleepAfterS }), ...(initialApprovals !== undefined && { initial_approvals: initialApprovals }), ...(attachments.length > 0 && { attachments }), }; diff --git a/cli/src/types.ts b/cli/src/types.ts index 7655fea7c..8a809722d 100644 --- a/cli/src/types.ts +++ b/cli/src/types.ts @@ -167,6 +167,8 @@ export interface ErrorClassification { /** Task detail returned by GET /v1/tasks/{task_id}. */ export interface TaskDetail { + /** Configured MicroVM approval-wait delay; 0 disables sleep. Absent on legacy records. */ + readonly microvm_sleep_after_s?: number; readonly task_id: string; readonly status: TaskStatusType; /** ``null`` for a repo-less workflow (#248 Phase 3). */ @@ -451,6 +453,8 @@ export interface CreateTaskResponse extends TaskDetail { /** Create task request body for POST /v1/tasks. */ export interface CreateTaskRequest { + /** MicroVM approval-wait seconds before sleep (0 = off, omitted = 600). Does not extend approval deadlines. */ + readonly microvm_sleep_after_s?: number; /** Optional since #248 Phase 3: repo-less workflows submit without it. */ readonly repo?: string; readonly issue_number?: number; @@ -802,6 +806,11 @@ export const APPROVAL_TIMEOUT_S_MAX = 3600; /** Default approval_timeout_s when the submit payload omits it. */ export const APPROVAL_TIMEOUT_S_DEFAULT = 300; +/** Per-task MicroVM sleep delay bounds; zero disables automatic sleep. */ +export const MICROVM_SLEEP_AFTER_S_MIN = 0; +export const MICROVM_SLEEP_AFTER_S_MAX = 3600; +export const MICROVM_SLEEP_AFTER_S_DEFAULT = 600; + /** Minimum allowed max_budget_usd (1 cent). * Sourced from ``contracts/constants.json`` via cdk types.ts (#258). */ export const MAX_BUDGET_USD_MIN = 0.01; diff --git a/cli/test/commands/submit.test.ts b/cli/test/commands/submit.test.ts index 0d6d825c4..d9cc70eec 100644 --- a/cli/test/commands/submit.test.ts +++ b/cli/test/commands/submit.test.ts @@ -422,6 +422,28 @@ describe('submit command', () => { }); describe('Cedar HITL extensions', () => { + test.each([['off', 0], ['0', 0], ['30', 30], ['600', 600], ['3600', 3600]])( + 'forwards MicroVM sleep delay %s without changing approval timeout', async (value, expected) => { + mockCreateTask.mockResolvedValue({ task_id: 't-sleep', status: 'SUBMITTED' }); + await makeSubmitCommand().parseAsync([ + 'node', 'test', '--repo', 'owner/repo', '--task', 'ok', '--microvm-sleep-after', String(value), + ]); + const [body] = mockCreateTask.mock.calls[0]; + expect(body.microvm_sleep_after_s).toBe(expected); + expect(body).not.toHaveProperty('approval_timeout_s'); + }, + ); + test.each(['-1', '3601', '1.5', '30seconds', '1e2', '', ' ', 'NaN'])('rejects malformed MicroVM delay %j', async value => { + await expect(makeSubmitCommand().parseAsync([ + 'node', 'test', '--repo', 'owner/repo', '--task', 'ok', '--microvm-sleep-after', value, + ])).rejects.toThrow(/--microvm-sleep-after must be off or an integer/); + expect(mockCreateTask).not.toHaveBeenCalled(); + }); + test('omits the sleep override so the server captures its default', async () => { + mockCreateTask.mockResolvedValue({ task_id: 't-sleep', status: 'SUBMITTED' }); + await makeSubmitCommand().parseAsync(['node', 'test', '--repo', 'owner/repo', '--task', 'ok']); + expect(mockCreateTask.mock.calls[0][0]).not.toHaveProperty('microvm_sleep_after_s'); + }); // --approval-timeout ------------------------------------------------------ test('forwards --approval-timeout as approval_timeout_s', async () => { diff --git a/contracts/constants.json b/contracts/constants.json index 0bc985d28..e25cb8bd2 100644 --- a/contracts/constants.json +++ b/contracts/constants.json @@ -9,6 +9,11 @@ "max": 3600, "default": 300 }, + "microvm_sleep_after_s": { + "min": 0, + "max": 3600, + "default": 600 + }, "max_budget_usd": { "min": 0.01, "max": 100 diff --git a/contracts/constants.md b/contracts/constants.md index db4e8724d..8fa28deaa 100644 --- a/contracts/constants.md +++ b/contracts/constants.md @@ -124,6 +124,11 @@ JSON at TypeScript compile time via `resolveJsonModule`. (1 hour). - **`approval_timeout_s.default`** — value applied when the submit payload omits `approval_timeout_s`. 300 seconds (5 minutes) per §6 decision #6. +- **`microvm_sleep_after_s`** — per-task delay before sleeping during a + pending human approval: whole seconds from 0 to 3600, default 600 + (10 minutes). Zero disables sleep. Task creation persists the resolved + preference; only the MicroVM supervisor consumes it. It does not extend + approval deadlines or override the deployment's suspension switch. - **`max_budget_usd.min`** — floor for a task's `max_budget_usd` (1 cent). Validated server-side (`validation.ts`) and pre-validated by `bgagent submit --max-budget` (#258). diff --git a/docs/design/COMPUTE.md b/docs/design/COMPUTE.md index 6dacd81a1..0f44807a4 100644 --- a/docs/design/COMPUTE.md +++ b/docs/design/COMPUTE.md @@ -87,6 +87,15 @@ Networking separates image build from execution: the build-only connector permit Local P3 now connects the guest checkpoint/credential hooks, actual image-version capability, durable supervisor and post-commit approval wake. The `microvm_approval_suspend_enabled` deployment context defaults false and controls both a static opt-in and a live Parameter Store switch. Existing durable executions reread the live switch before new suspension because their original Lambda environment is pinned. Recovery and the original service lifetime survive supervisor replay; the API preserves accepted decisions when optional wake fails. AgentCore/ECS retain explicit unsupported pause/wake results. Approval expiry still uses the original UTC/monotonic deadline. See the [supervisor runbook](../verification/645-p3-supervisor.md) for budgets, scoped permissions and remaining live gates. +Tasks can set `microvm_sleep_after_s` (CLI: `--microvm-sleep-after `). +The default is 600 seconds of waiting for each approval; zero disables sleep. +Creation persists the resolved preference, while legacy rows use the default. +The global suspension switch still takes precedence. The five-minute default +approval window therefore stays awake; only longer windows can reach the +ten-minute sleep delay. Neither this setting nor suspension extends a gate's +original deadline. Snapshot storage and save/restore fees mean the delay is a +user preference, not a guarantee of savings for every pause. + ## ECS Fargate task sizing (build vs. planning) When a repo is `compute_type: ecs`, `EcsAgentCluster` provisions **two** Fargate task definitions, and the orchestrator picks between them per task by whether the resolved workflow is **read-only**: diff --git a/docs/guides/USER_GUIDE.md b/docs/guides/USER_GUIDE.md index f8d47407e..e414eb15d 100644 --- a/docs/guides/USER_GUIDE.md +++ b/docs/guides/USER_GUIDE.md @@ -639,6 +639,7 @@ Created: 2026-04-01T00:39:51.271Z | `--idempotency-key` | Idempotency key for deduplication. | | `--trace` | Enable detailed tracing: raises progress preview cap to 4 KB and uploads full NDJSON trajectory to S3 on completion. Download with `bgagent trace download`. | | `--approval-timeout` | Cedar HITL per-task approval timeout in seconds (default 300). A matching rule with its own `@approval_timeout_s` annotation still takes the minimum. See [Approval gates](#approval-gates-cedar-hitl). | +| `--microvm-sleep-after` | Seconds to wait for approval before putting a Lambda MicroVM to sleep (default 600 = 10 minutes; 0–3600 accepted). Use `off` to keep it awake. Requires the deployment's automatic-sleep feature to be enabled; does not change approval deadlines or affect other compute backends. | | `--pre-approve` | Cedar HITL scope to approve up-front (repeatable). Same scope forms as `bgagent approve --scope`. Hard-deny rules are always enforced. | | `--wait` | Poll until the task reaches a terminal status. | | `--output` | Output format: `text` (default) or `json`. | @@ -932,6 +933,20 @@ node lib/bin/bgagent.js submit --repo owner/repo --issue 42 --approval-timeout 6 `--approval-timeout` sets the task-wide default; a rule with its own `@approval_timeout_s` annotation still takes the minimum of the two. +For Lambda MicroVM tasks, `--microvm-sleep-after 600` selects the default +10-minute delay; `--microvm-sleep-after 120` selects two minutes and +`--microvm-sleep-after off` keeps the worker awake. The delay starts when each +approval request is created. Waking for approval, denial, or an approaching +deadline remains automatic. Sleep never starts a new approval timer. + +The default approval timeout is five minutes, so those requests stay awake +with the ten-minute sleep delay. A longer task timeout does not override a +shorter policy-rule timeout. Sleeping saves compute charges but adds snapshot +save/restore charges and wake-up time; short pauses can cost more than staying +awake. The API equivalent is `microvm_sleep_after_s` (zero means off); task +details return the saved setting. Automatic suspension remains disabled by +default pending the [P3 acceptance checks](../verification/645-p3-implementation-plan.md). + ## Webhook integration Webhooks allow external systems (CI pipelines, GitHub Actions, custom automation) to create tasks without Cognito credentials. Each webhook integration has its own HMAC-SHA256 shared secret. diff --git a/docs/src/content/docs/architecture/Compute.md b/docs/src/content/docs/architecture/Compute.md index f95a0d394..52495f52f 100644 --- a/docs/src/content/docs/architecture/Compute.md +++ b/docs/src/content/docs/architecture/Compute.md @@ -91,6 +91,15 @@ Networking separates image build from execution: the build-only connector permit Local P3 now connects the guest checkpoint/credential hooks, actual image-version capability, durable supervisor and post-commit approval wake. The `microvm_approval_suspend_enabled` deployment context defaults false and controls both a static opt-in and a live Parameter Store switch. Existing durable executions reread the live switch before new suspension because their original Lambda environment is pinned. Recovery and the original service lifetime survive supervisor replay; the API preserves accepted decisions when optional wake fails. AgentCore/ECS retain explicit unsupported pause/wake results. Approval expiry still uses the original UTC/monotonic deadline. See the [supervisor runbook](/sample-autonomous-cloud-coding-agents/architecture/645-p3-supervisor) for budgets, scoped permissions and remaining live gates. +Tasks can set `microvm_sleep_after_s` (CLI: `--microvm-sleep-after `). +The default is 600 seconds of waiting for each approval; zero disables sleep. +Creation persists the resolved preference, while legacy rows use the default. +The global suspension switch still takes precedence. The five-minute default +approval window therefore stays awake; only longer windows can reach the +ten-minute sleep delay. Neither this setting nor suspension extends a gate's +original deadline. Snapshot storage and save/restore fees mean the delay is a +user preference, not a guarantee of savings for every pause. + ## ECS Fargate task sizing (build vs. planning) When a repo is `compute_type: ecs`, `EcsAgentCluster` provisions **two** Fargate task definitions, and the orchestrator picks between them per task by whether the resolved workflow is **read-only**: diff --git a/docs/src/content/docs/using/Approval-gates-cedar-hitl.md b/docs/src/content/docs/using/Approval-gates-cedar-hitl.md index b28882b5d..74580144a 100644 --- a/docs/src/content/docs/using/Approval-gates-cedar-hitl.md +++ b/docs/src/content/docs/using/Approval-gates-cedar-hitl.md @@ -104,4 +104,18 @@ node lib/bin/bgagent.js submit --repo owner/repo --issue 42 --approval-timeout 6 `--pre-approve` can be repeated up to the platform limit (see `bgagent submit --help` for the current cap). Valid scope forms are the same as the `approve --scope` table above. Hard-deny rules are still enforced — `--pre-approve` only short-circuits soft-deny rules. -`--approval-timeout` sets the task-wide default; a rule with its own `@approval_timeout_s` annotation still takes the minimum of the two. \ No newline at end of file +`--approval-timeout` sets the task-wide default; a rule with its own `@approval_timeout_s` annotation still takes the minimum of the two. + +For Lambda MicroVM tasks, `--microvm-sleep-after 600` selects the default +10-minute delay; `--microvm-sleep-after 120` selects two minutes and +`--microvm-sleep-after off` keeps the worker awake. The delay starts when each +approval request is created. Waking for approval, denial, or an approaching +deadline remains automatic. Sleep never starts a new approval timer. + +The default approval timeout is five minutes, so those requests stay awake +with the ten-minute sleep delay. A longer task timeout does not override a +shorter policy-rule timeout. Sleeping saves compute charges but adds snapshot +save/restore charges and wake-up time; short pauses can cost more than staying +awake. The API equivalent is `microvm_sleep_after_s` (zero means off); task +details return the saved setting. Automatic suspension remains disabled by +default pending the [P3 acceptance checks](/sample-autonomous-cloud-coding-agents/architecture/645-p3-implementation-plan). \ No newline at end of file diff --git a/docs/src/content/docs/using/Using-the-cli.md b/docs/src/content/docs/using/Using-the-cli.md index 7b6c29a9f..f9785b6b1 100644 --- a/docs/src/content/docs/using/Using-the-cli.md +++ b/docs/src/content/docs/using/Using-the-cli.md @@ -115,6 +115,7 @@ Created: 2026-04-01T00:39:51.271Z | `--idempotency-key` | Idempotency key for deduplication. | | `--trace` | Enable detailed tracing: raises progress preview cap to 4 KB and uploads full NDJSON trajectory to S3 on completion. Download with `bgagent trace download`. | | `--approval-timeout` | Cedar HITL per-task approval timeout in seconds (default 300). A matching rule with its own `@approval_timeout_s` annotation still takes the minimum. See [Approval gates](#approval-gates-cedar-hitl). | +| `--microvm-sleep-after` | Seconds to wait for approval before putting a Lambda MicroVM to sleep (default 600 = 10 minutes; 0–3600 accepted). Use `off` to keep it awake. Requires the deployment's automatic-sleep feature to be enabled; does not change approval deadlines or affect other compute backends. | | `--pre-approve` | Cedar HITL scope to approve up-front (repeatable). Same scope forms as `bgagent approve --scope`. Hard-deny rules are always enforced. | | `--wait` | Poll until the task reaches a terminal status. | | `--output` | Output format: `text` (default) or `json`. | diff --git a/docs/verification/645-p3-implementation-plan.md b/docs/verification/645-p3-implementation-plan.md index 0811a127b..33f65af7a 100644 --- a/docs/verification/645-p3-implementation-plan.md +++ b/docs/verification/645-p3-implementation-plan.md @@ -289,7 +289,7 @@ When the coding agent asks a human for permission, its computer may go to sleep. P3 coordinates the supervisor, approval records and the sleeping computer. -For example, with a five-minute approval window and the suggested settings below: the question appears at **12:00**; after **12:00:30** the VM can sleep. If approval arrives at **12:02**, ABCA wakes it and it reads the saved answer. If nobody answers, ABCA wakes it around **12:04** so the agent can deny at **12:05**. Waking must not start a new five-minute timer. +For example, with a thirty-minute approval window and the default ten-minute sleep delay: the question appears at **12:00**; after **12:10** the VM can sleep. If approval arrives at **12:15**, ABCA wakes it and reads the saved answer. If nobody answers, ABCA wakes it around **12:29** so the agent can deny at **12:30**. Waking must not start a new timer. Default five-minute approval windows stay awake. Users can choose a different delay or disable task sleep. A few implementation words used below: @@ -489,7 +489,7 @@ The installed SDK returns empty suspend/resume responses. Its observed states ar **Connected to the durable supervisor:** the policy combines task status, the **specific current approval row's status**, desired action, explicit VM state and current time. PENDING alone is not a reason to resume. All terminal approval states, deadline proximity, missing/unreadable data or unintended suspension can require wake. SUSPENDING records desired wake but returns `requestReady: false` until SUSPENDED is observed. -Initial local policy values: 30-second suspend grace, 60-second pre-deadline wake margin, 30-second minimum useful sleep and at most 5-second transition polling. Long intervals are clamped to the relevant grace/wake/session deadline. These are tunable choices requiring live measurement, not AWS facts. The supervisor now escalates after three consecutive failed cycles and bounds the entire cycle to 45 seconds, including the store's 5-second request/read-sequence budgets and lost-reply recovery. Wake/unknown recovery is bounded to 120 seconds and startup to 300 seconds; see the supervisor runbook for the complete budgets. +Current local policy values: a per-task suspend delay of 600 seconds by default (`microvm_sleep_after_s`, 0–3600 seconds, zero disables sleep), 60-second pre-deadline wake margin, 30-second minimum available sleep window and at most 5-second transition polling. The earlier 30-second grace remains explicit in historical verification fixtures; it is no longer the application default. Long intervals are clamped to the relevant grace/wake/session deadline. Snapshot costs and actual wait distributions still need measurement; the 30-second available window does not promise financial savings. The supervisor escalates after three consecutive failed cycles and bounds the entire cycle to 45 seconds, including the store's 5-second request/read-sequence budgets and lost-reply recovery. Wake/unknown recovery is bounded to 120 seconds and startup to 300 seconds; see the supervisor runbook for the complete budgets. ### State/action table diff --git a/docs/verification/645-p3-supervisor.md b/docs/verification/645-p3-supervisor.md index 4de06550e..507aaf83b 100644 --- a/docs/verification/645-p3-supervisor.md +++ b/docs/verification/645-p3-supervisor.md @@ -118,6 +118,22 @@ The stable parameter name is passed as `MICROVM_APPROVAL_SUSPEND_PARAMETER_NAME` Both must allow sleep. Compatible per-worker image evidence remains an independent admission check. +Task submission accepts `microvm_sleep_after_s`: an integer from 0 to 3600, +default 600 seconds. Zero disables new sleep for that task. The CLI exposes +`--microvm-sleep-after `. Creation stores the resolved default, +and task details return it. Legacy task rows without the setting use 600; +malformed stored values prohibit new sleep and permit wake/cleanup. +Fresh observations recheck the preference before Suspend. Supervisor logs +include the validated delay and policy reason, without echoing malformed input. +The worker role cannot modify this coordinator-owned setting. + +The delay is measured from each original approval creation time. The +60-second pre-deadline wake margin and 30-second minimum available sleep +window remain safety bounds, not financial break-even promises. Default +five-minute approvals stay awake with a ten-minute sleep delay. The choice +does not change approval deadlines, the eight-hour worker lifetime, or the +global enablement requirements. + Durable executions retain their original Lambda version and environment. An environment-only redeploy therefore cannot disable an existing execution. The stack retains published coordinator versions and their immutable guardrail From bf57112d4591ca645ce57518277b227407d05b9d Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Wed, 16 Sep 2026 13:49:32 -0400 Subject: [PATCH 054/149] docs: record sleep controls and AWS capacity verification --- .../verification/645-capacity-reservations.md | 9 ++ .../645-lambda-microvm-service-feedback.md | 31 ++-- .../645-p3-capacity-scan-20260916.md | 78 ++++++++++ .../645-p3-implementation-plan.md | 39 ++++- .../645-p3-resume-refusal-investigation.md | 51 ++++++- .../645-p3-user-sleep-20260916.md | 139 ++++++++++++++++++ 6 files changed, 329 insertions(+), 18 deletions(-) create mode 100644 docs/verification/645-p3-capacity-scan-20260916.md create mode 100644 docs/verification/645-p3-user-sleep-20260916.md diff --git a/docs/verification/645-capacity-reservations.md b/docs/verification/645-capacity-reservations.md index 2222e283e..64df74d07 100644 --- a/docs/verification/645-capacity-reservations.md +++ b/docs/verification/645-capacity-reservations.md @@ -43,6 +43,15 @@ docker stop abca645-capacity-ddb ## Limits of this evidence +The [September 16 AWS follow-up](./645-p3-capacity-scan-20260916.md) now verifies +600 users across real multi-page task/counter scans, an exact deployed reconciler +artifact with equivalent permissions on isolated tables, interrupted scans with +zero writes, and conservative handling of an older unmarked task. A separate +[normal-role check](./645-p3-user-sleep-20260916.md) repaired an overcount and +terminal reservation while preserving a real waiting worker. The bounded +volume and role checks do not complete the old-writer drain/upgrade/rollback +procedure or establish arbitrary production scale. + Local tests prove the application requests and DynamoDB Local's transaction behavior. They do not establish deployed IAM, AWS scaling, successful rollout or MicroVM sleep/wake behavior. Terminal events may repeat or be lost independently of the atomic seat update. The reservation/start markers share the task row. Subsequent prerequisite work restricts agent updates to reporting/approval attributes and removes whole-row replacement/deletion plus direct worker access to the counter. Public-API omission alone was not protection. See [coordinator metadata verification](./645-coordinator-metadata.md) for the writer inventory, actual policy boundary, remaining status/tag trust limits and required AWS authorization checks. These local transaction tests do not prove that security boundary. diff --git a/docs/verification/645-lambda-microvm-service-feedback.md b/docs/verification/645-lambda-microvm-service-feedback.md index 9d380c4b6..5086eb2f7 100644 --- a/docs/verification/645-lambda-microvm-service-feedback.md +++ b/docs/verification/645-lambda-microvm-service-feedback.md @@ -10,7 +10,7 @@ is the receipt that lets the service team find a particular call. | ID | Priority | Topic | Evidence/status | |---|---|---|---| -| F01 | P3 blocker | Accepted wake ends in connection refusal | Four recorded failures; responsible component unknown | +| F01 | P3 blocker | Accepted wake ends in connection refusal | Five recorded failures, including instrumented image 5.0; responsible component unknown | | F02 | High | Supported IAM conditions and misleading permission errors | Reproduced in earlier P2 work; current service behavior needs confirmation | | F03 | Medium | A service-side hook timeline and structured failure details | Diagnostic improvement request based on F01 | | F04 | Medium | `PENDING` also means restoring an existing worker | Observed live; our timer bug is fixed | @@ -24,7 +24,7 @@ is the receipt that lets the service team find a particular call. The coordinator releases capacity correctly; the requested coding workflow still fails. This prevents enabling automatic suspension for normal tasks. -**Observed:** four failures on images `3.0` and `4.0` in `us-west-2`, September +**Observed:** five failures on images `3.0`, `4.0` and `5.0` in `us-west-2`, September 15–16. AWS accepted `ResumeMicrovm`, then reported: > Resume lifecycle hook connection was refused. Please check your hook endpoint @@ -36,14 +36,18 @@ and coordinator Resume calls as a necessary cause. **Best starting evidence for the service team:** -- Account ``, region `us-west-2`, September 16, 00:25:01–00:25:12 UTC. -- Worker `microvm-2da070a7-43cf-3bb3-9c5d-69bc106f2cb1`, image - `backgroundagent-dev-abca-agent:4.0`. -- Suspend receipt `e2199854-5ad0-46c9-bc59-248ac68ead82`. -- Resume receipt `0acb3fec-8b18-4bed-b457-0271a15ffdd6`, acknowledged - 00:25:06.766 UTC; terminal refusal observed 00:25:07.524 UTC. +- Account ``, region `us-west-2`, September 16, 17:09:39–17:10:42 UTC. +- Worker `microvm-6ff103ab-a41d-348b-8a56-721c9050b623`, image + `backgroundagent-dev-abca-agent:5.0`, original server PID 1. +- Suspend receipt `98a53cab-7737-4675-944e-7a8202781149`; guest checkpoint + succeeded and `/suspend` returned HTTP 200 at 17:09:40.186 UTC. +- Resume receipt `d32f9929-e7cb-4603-a6c1-0bd2a801f5b5`, acknowledged + 17:10:39.404 UTC; service termination timestamp 17:10:40.415 UTC. +- The instrumented guest logged no resume hook entry or subsequent callback + stage. This case used a single supervisor wake before the original deadline, + with no injected failure. - [Prepared investigation report](./645-p3-resume-refusal-investigation.md) - contains all four worker IDs, comparison runs and the complete example timeline. + contains all five worker IDs, comparison runs and the complete example timelines. **Ask:** inspect the service's restore and hook-transport records for these receipts. Was this a TCP refusal from the guest listener, a stale connection, @@ -217,6 +221,15 @@ or treated as a new launch? Which supported API, event or audit field maps the original token/API receipt to the worker ID when the client never received it? How long does that mapping remain available? +**CloudTrail check:** at 17:36 UTC on September 16, event-history lookup for the +known image 5.0 launch at 17:08:54 UTC found no `RunMicrovm` event. The complete +17:08:45–17:09:05 window contained 74 events, including `GetMicrovm`, +`ListMicrovms` and `GetMicrovmImageVersion` from `lambda.amazonaws.com`. +Only event names, sources and window metadata were retained. This observation +does not establish the logging contract or exclude later delivery. Is Run a +management or data event, and what audit configuration is required to retain +the launch token-to-worker mapping? + **Limits:** five minutes of successful replay does not establish the maximum retention period. The 120-second limit is ABCA policy, not an AWS guarantee. Post-expiry behavior and recovery of a genuinely unknown worker ID remain diff --git a/docs/verification/645-p3-capacity-scan-20260916.md b/docs/verification/645-p3-capacity-scan-20260916.md new file mode 100644 index 000000000..7b6dcc213 --- /dev/null +++ b/docs/verification/645-p3-capacity-scan-20260916.md @@ -0,0 +1,78 @@ +# ADR-021 P3: AWS capacity scan and interrupted-read checks + +Verified September 16, 2026, in account ``, `us-west-2`. +All checks passed and all temporary resources were removed. This is bounded +pagination and reconciliation evidence; it does not establish an arbitrary +production retention volume or complete the coordinated writer upgrade/rollback. + +## Deployment and data + +The temporary Lambda used the exact deployed ConcurrencyReconciler artifact, +SHA-256 `ZbDOQoSBUD3t4e0efIyopTKifQwKkQBBF8IA2wBLMRw=`, with the same Node.js 24, +ARM64, 256 MB memory and 300-second timeout. Its DynamoDB action sets matched the +normal reconciler's policy, with resource ARNs substituted to two owned tables. +It had no normal-table access, event schedule or coding workers. + +Each table contained 600 records, with 5,000 bytes of padding per record to +exercise real AWS pagination. Task records covered active held reservations, +one terminal held reservation and an older active task without a reservation +marker. Three counters were deliberately too high or too low. + +| Strongly consistent scan | Rows | Observed pages | Consumed read capacity units | +|---|---|---|---| +| Tasks | 600 | 4 | 779 | +| Counters | 600 | 3 | 761 | + +These page and capacity measurements come from separate observer scans of the +same tables. The deployed handler uses projections and does not itself request +consumed-capacity telemetry. Its successful results required accounting for +all 600 users. + +## Results + +The first deployed invocation: + +- Repaired counters from 3, 0 and 7 to their actual one held reservation. +- Released the terminal task's reservation, changing its counter from 1 to 0. +- Preserved the other 598 active held reservations. +- Left the ambiguous older task's counter at 2 and logged + `CONCURRENCY_RESERVATION_UNKNOWN`. +- Logged `scanned=600`, `corrected=3`, `errors=0`. + +Lambda reported **1,475.42 ms** duration and **112 MB** maximum memory. +The invocation receipt is `2c55ca9e-9d47-4b06-947c-56d4f665e455`. +These numbers describe this fixture, not a throughput or maximum-size guarantee. + +Two interrupted-read cases ran the production handler locally against the +real AWS tables. The SDK boundary discarded the next read by throwing an +explicit fixture error before the second page of either the counter scan or +the task scan. Actual earlier AWS pages and their receipts were retained. +In both cases the handler failed with **zero update/transaction requests**, and +all 600 counters remained unchanged. A deliberate overcount was left in place +before these checks so that a premature repair would have been visible. +These were injected interruptions, not observed AWS outages. + +After the older unmarked task became terminal, a second deployed invocation +repaired its counter from 2 to 0. It also repaired the deliberate overcount +from 9 to 1. It logged `scanned=600`, `corrected=2`, `errors=0`, with +**343.39 ms** duration and **113 MB** maximum memory. +Receipt: `5bba4ca9-014e-47de-abbf-e0d4be98c9ab`. + +The earlier [normal-role live repair](./645-p3-user-sleep-20260916.md) separately +verified correction and terminal release while a real MicroVM waited for +approval. Together, these results add actual deployed writer permissions, +multiple scan pages, partial-read safety and conservative legacy handling to +the existing [transaction evidence and upgrade procedure](./645-capacity-reservations.md). + +## Cleanup and limits + +The private `backgroundagent-dev-p3-capacity-20260916` function, role, log group +and both 600-row tables were deleted after ownership-tag checks. Seventeen +function log events and final task/counter snapshots were saved first. +Read-only absence checks completed at **17:35:25.915 UTC**. The normal +reconciler's artifact and environment remained unchanged. + +Evidence is retained in +`/tmp/abca-645-p2-clean-20260913/p3-capacity-scan-20260916`. +The full old-writer drain/upgrade/rollback procedure and production retention +volume remain separate gates. No workload-based memory change is proposed. diff --git a/docs/verification/645-p3-implementation-plan.md b/docs/verification/645-p3-implementation-plan.md index 33f65af7a..0bb3f56e8 100644 --- a/docs/verification/645-p3-implementation-plan.md +++ b/docs/verification/645-p3-implementation-plan.md @@ -6,6 +6,20 @@ Prepared 2026-09-13 from `main` `5e10038c7e28179b302ac4de78b709795aeba3ce`. Read Prerequisite work is tracked here on `fix/645-microvm-readiness`. “Completed” means implemented and checked locally; AWS deployment and live verification have separate completion gates below. +**Adjustable sleep deployed (2026-09-16):** source `a81c565d` adds +`microvm_sleep_after_s` and CLI `--microvm-sleep-after `, with a +600-second default and zero to stay awake. The full build passed 7,939 tests; +56 real DynamoDB Local transaction tests passed separately. +The [settings and live record](./645-p3-user-sleep-20260916.md) verifies normal +coordinator version 8, retained version 7, unchanged image 5.0 and 475 resources, +both suspension gates off, and seven deployed API checks. Five of six worker +cases passed: default/off/custom timing, late approval winning its decision +race, and safe failure after actual credential-renewal denial. The timeout-wins +case reproduced the fifth resume connection refusal. Its failure remains open; +all six workers and temporary verification infrastructure were cleaned up. +The normal deployed capacity reconciler also passed an owned overcount and +terminal-reservation repair while preserving the real waiting worker. + **Guest hook milestone (2026-09-14):** production [worker suspend/resume hooks](./645-p3-lifecycle-hooks.md) now connect the guest barrier to atomic checkpoint writes and retained-credential refresh followed by @@ -111,7 +125,8 @@ is superseded by these records. - [x] Verify real AWS durable replay after a saved worker receipt and process exit, including cancellation during recovery, in the isolated production-handler fixture. - [ ] Complete remaining durable registration races, AWS behavior after token retention expires, and operator cleanup of genuinely unknown worker IDs. - [x] Make capacity acquisition/release atomic per task across crash replay; unify counter writers and repair. -- [ ] Verify the capacity protocol's upgrade/drain procedure, deployed IAM and scan scale in AWS. +- [x] Verify normal deployed capacity repair while preserving a waiting worker; verify bounded 600-user AWS pagination, partial-scan safety and conservative legacy handling. +- [ ] Complete the capacity protocol's old-writer upgrade/drain and rollback procedure; validate volume against the intended production workload. - [x] Restrict agent task updates to reporting fields; remove replacement/deletion and worker counter grants. - [x] Verify metadata restrictions with real AWS task-tagged sessions and mixed transactions under the deployed MicroVM role; retain status/tag trust limits and the separate other-role gate. - [x] Replace unused logging-failure bookkeeping with structured stdout diagnostics (#810); document shared runtime networking and verify large registry payload delivery (#818). @@ -134,7 +149,7 @@ is superseded by these records. - [x] Deploy the supervisor and six-hook image with suspension disabled; verify six isolated guest cases, repair the discovered cancellation stop omission, and prove API termination before test cleanup. - [ ] Deploy and verify the complete P3 sleep/wake lifecycle in AWS. - [x] Deploy the explicit SDK callback-timeout fix and repeat long sleep, late wake, approval, denial and cancellation; require final approval/tool evidence as well as cleanup. Image `4.0` passed these checks, including real renewal after credential expiry. -- [ ] Resolve the intermittent resume-hook connection refusal; successful retries do not discharge the four failures across images 3.0 and 4.0. See the [investigation and request IDs](./645-p3-resume-refusal-investigation.md). +- [ ] Resolve the intermittent resume-hook connection refusal; successful retries do not discharge the five failures across images 3.0, 4.0 and 5.0. See the [investigation and request IDs](./645-p3-resume-refusal-investigation.md). First prerequisite batch completed locally on 2026-09-13: @@ -228,7 +243,7 @@ absence of all temporary infrastructure. The detailed batches below preserve the implementation history. For the current handoff, use this order: -1. Resolve the four [resume-hook connection refusals](./645-p3-resume-refusal-investigation.md). +1. Resolve the five [resume-hook connection refusals](./645-p3-resume-refusal-investigation.md). Service-side connection diagnostics and guest/listener health are still missing. Passing retries, the callback-timeout fix and successful cleanup do not close this gate. @@ -252,6 +267,11 @@ handoff, use this order: with specific guest-stage diagnostics and deployed task-API feedback. Two mistitled long-hold attempts were excluded and replaced by fresh measured cases. No unexpected refusal appeared; production connection handling is unchanged. + A subsequent image 5.0 timeout-race case reproduced the refusal after a normal + pre-deadline wake. The guest logged a successful checkpoint and suspend HTTP + 200, but no resume hook entry. The prepared service report now includes its + API receipts and precise timeline; the new diagnostics have not established + the cause. 2. The [command-race checks](./645-p3-command-races-20260916.md) now pass cancellation before/after Suspend, during observed `SUSPENDING`, and during restore `PENDING`; approval during an accepted suspension and three consecutive @@ -261,10 +281,15 @@ handoff, use this order: failure guidance in coordinator version 7 and verified the normal task API; image 5.0 and disabled suspension settings remain in place. Complete the remaining live race/fault matrix: - late decision races, credential-refresh failures, durable registration races, + timeout-winning decision races, durable registration races, service token-retention expiry and recovery of a worker whose ID was genuinely lost. Keep each injected failure distinct from an unrelated service failure. + The [adjustable-sleep follow-up](./645-p3-user-sleep-20260916.md) completed the + late-approval winner and actual credential-refresh denial checks, plus default, + off and custom delays. It deployed coordinator version 8 with both gates off. + The timeout-winner case was interrupted by the fifth unexplained wake refusal + and remains unaccepted. 3. Complete effective permissions and network checks for the other backends, plus runtime/remote-MCP connectivity. Verify a full cloned-repository P3 workflow on the final image, including mutable workspace state and normal @@ -273,6 +298,12 @@ handoff, use this order: deployed writer roles, including realistic scan volume. Retain the verified local transaction and isolated-live results as evidence for their narrower scope. + The [AWS scan follow-up](./645-p3-capacity-scan-20260916.md) now passes a + 600-user fixture with real multi-page reads, exact normal Lambda artifact, + equivalent table permissions, zero writes after interrupted scans, and + conservative handling until an older task settles. Its temporary tables and + function were removed. This bounds the verified volume without claiming the + production migration or arbitrary retention scale. 5. Perform the final compatible rollout, including shared runtime changes for ECS/AgentCore, pinned-version retention and rollback checks. Enable automatic suspension only after the remaining gates pass, then finish the ADR/runbook diff --git a/docs/verification/645-p3-resume-refusal-investigation.md b/docs/verification/645-p3-resume-refusal-investigation.md index b93fa7752..9153534c2 100644 --- a/docs/verification/645-p3-resume-refusal-investigation.md +++ b/docs/verification/645-p3-resume-refusal-investigation.md @@ -27,7 +27,8 @@ the new logging and wake-failure feedback. Three isolated AWS workflows verified the instrumentation with the original server running as PID 1, including actual API wake and coordinator recovery. None reproduced refusal. The [normal-stack rollout](./645-p3-diagnostics-rollout-20260916.md) now runs coordinator -version 6 and image 5.0; the historical failures below predate this instrumentation. +version 7 and image 5.0 after the subsequent command-race follow-up. A fifth +failure on image 5.0 now includes the new diagnostics, as recorded below. ## Recorded failures @@ -41,6 +42,46 @@ The earlier cases are detailed in the | Approval after coordinator process-crash recovery | `3.0` | `microvm-9e6bdb3f-bc10-3921-9c50-287bbe20b574` | Sep 15, 21:56:56.279 | | Supervisor repair after an injected inline Resume failure | `3.0` | `microvm-dca019ee-7d19-389e-b90b-bde129f77329` | Sep 15, 22:54:45.753 | | Supervisor wake after an owned approval record was deleted | `4.0` | `microvm-2da070a7-43cf-3bb3-9c5d-69bc106f2cb1` | Sep 16, 00:25:07.524 | +| Supervisor wake before the original approval deadline | `5.0` | `microvm-6ff103ab-a41d-348b-8a56-721c9050b623` | Sep 16, 17:10:40.415 | + +### Image 5.0 diagnostic timeline + +Task `01M2NHJ8YSRTRY3XPT06SHT9QD` used the normal image and original server +as PID 1, with a 150-second approval window and a custom 30-second sleep delay. +The private coordinator ran source `a81c565d`; this case did not inject a +command or guest failure. The intended check was an approval timeout winning +before a subsequent approval API call. The unexpected refusal prevented that +check from reaching its decision race. + +| Time on Sep 16 | Evidence | +|---|---| +| 17:09:09 | Read approval created; original deadline 17:11:39 | +| 17:09:39.879 | Supervisor sends Suspend | +| 17:09:39.940 | AWS accepts; request `98a53cab-7737-4675-944e-7a8202781149` | +| 17:09:39.971 | Guest logs suspend hook entry, PID 1 | +| 17:09:40.185 | Checkpoint transaction and controller finish; hook acknowledges HTTP 200 after 214 ms | +| 17:09:40.186 | HTTP access log records `/suspend` 200 | +| 17:09:41.524 | Observer reads `SUSPENDED` | +| 17:10:39.344 | Supervisor sends its sole Resume, about 60 seconds before the original deadline | +| 17:10:39.404 | AWS accepts; request `d32f9929-e7cb-4603-a6c1-0bd2a801f5b5` | +| 17:10:40.415 | Service records termination with the connection-refused reason | +| 17:10:41.716 | Coordinator writes `FAILED` with `MICROVM_RESUME_HOOK_FAILED` | +| 17:10:41.752 | Coordinator releases the task's reservation | + +The fully paginated guest log window contains no resume hook entry, callback +stage, HTTP access record or later application output. Thus the newly +instrumented credential-refresh and identity-reconciliation callbacks did not +leave evidence of starting. This does not establish that the process or +listener survived restoration. The failed harness subsequently called its +idempotent cleanup helpers; the task's recorded failure and reservation release +predate that fallback, but this case is excluded from successful lifecycle +acceptance. + +Four preceding settings/decision cases completed: the default sleep request +occurred after 600.385 seconds, an explicit off setting never requested sleep, +a custom delay requested sleep after 30.395 seconds, and an approval committed +after the deadline won its conditional decision race. These passing controls +do not explain or discharge this fifth failure. ### Image 4.0 request timeline @@ -85,8 +126,8 @@ The image `4.0` refusal above happened afterward. Therefore: - Overlapping API/supervisor Resume requests are not required. Both the process-crash case and the supervisor-only cases exclude that explanation. - The observed failure is separate from the old ten-minute Claude callback - cancellation: the latest failure happened less than a minute after its gate - was created, with minutes still remaining. + cancellation: the image 4.0 case failed less than a minute after its gate + was created; the image 5.0 case failed before its original deadline. - These runs do not establish a failure rate or identify the responsible component. @@ -106,7 +147,7 @@ observer still ran in the guest. This calibrates the diagnostics; it does not establish why the full agent loses its listener or connection. That experiment also exposed a separate [pending-wake timer bug](./645-p3-pending-wake.md) -in the supervisor. Its correction does not account for these four failures, +in the supervisor. Its correction does not account for these failures, whose worker termination reason was the service-reported connection refusal. The [full-agent observer experiment](./645-p3-process-observer-20260916.md) @@ -115,7 +156,7 @@ fallback and direct API wakes retained a healthy child-owned listener, without reproducing the refusal. Short full-agent cases reused the same client port for `/suspend` and `/resume`, unlike the minimal listener's closed connections. That is an observed transport difference, not an established cause. None of -the four original failures has independent process/listener evidence at restore, +the five recorded failures has independent process/listener evidence at restore, and the guest does not expose the cgroup OOM counters sampled by the observer. The subsequent [local transport control](./645-p3-transport-control-20260916.md) diff --git a/docs/verification/645-p3-user-sleep-20260916.md b/docs/verification/645-p3-user-sleep-20260916.md new file mode 100644 index 000000000..b543041f8 --- /dev/null +++ b/docs/verification/645-p3-user-sleep-20260916.md @@ -0,0 +1,139 @@ +# ADR-021 P3: adjustable approval sleep and live verification + +Verified September 16, 2026, in account ``, `us-west-2`. +Source `a81c565d` is deployed with automatic suspension still disabled. +The settings and five live cases pass. P3 acceptance remains incomplete because +the sixth case reproduced the [resume connection refusal](./645-p3-resume-refusal-investigation.md). + +## User behavior + +`microvm_sleep_after_s` controls how long a MicroVM waits at an approval gate +before requesting sleep. The API accepts whole seconds from 0 through 3600, +persists 600 when omitted, and returns the value in task details. Zero means +stay awake. Existing records without the field use the same 600-second default. + +The CLI exposes `bgagent submit --microvm-sleep-after `. +For example, `--microvm-sleep-after 120` waits two minutes, and +`--microvm-sleep-after off` keeps the worker awake. This setting does not extend +the approval deadline. A five-minute approval window ends before a ten-minute +sleep delay, so that gate stays awake. + +The deployment flag and live Parameter Store switch still control whether +automatic sleep is allowed at all. Approval, denial, timeout and cancellation +continue to trigger the appropriate wake or cleanup behavior. + +## Validation and rollout + +The full build passed 5,050 CDK, 942 CLI and 1,947 Python tests: **7,939 total**. +A separate pinned DynamoDB Local run passed all **56** lifecycle/capacity +transaction tests. Type parity, compilation, lint, documentation generation, +the 77-page documentation build and link checks passed. + +CloudFormation reached `UPDATE_COMPLETE` at **17:21:50.192 UTC**: + +- Normal coordinator alias `live` points to version **8**, code SHA-256 + `lYgbdJRz0bLxPXT3XsW0VGaTqLCM8hB1vVGk9JuTRB0=`. +- Version **7** was retained; CloudFormation recorded `DELETE_SKIPPED`, and + its previous code checksum remains readable. +- All **475** resources retain their physical identity except the expected + coordinator version rotation. Image **5.0** remains `ACTIVE` / `SUCCESSFUL`. +- The coordinator environment is unchanged. Its suspension flag is false; + the normal Parameter Store switch remains false at version 1. +- The coordinator and CreateTask code checksums match the exact reviewed S3 + artifact bytes. + +The exact reviewed template contains 31 Lambda code changes plus the retained +version/alias rotation: 34 changes. `DescribeChangeSet` reported 34 entries with +`IncludePropertyValues=true`, but 87 without it. The additional 53 entries were +unchanged resources with dynamic ARN dependencies. Both their template +properties and final physical identities were checked. The reviewed template +SHA-256 is `4957871b24243f900e16fa9a04b58db63d85b1da5fcf7265d5f403fe5ecd3e3c`. +The exact S3 template bytes, rather than lossy `GetTemplate` text, supplied the +deployment baseline. + +Seven checks exercised the newly deployed normal Lambda API handlers: + +| Request | Verified result | +|---|---| +| Omitted setting | Stored and returned 600 | +| Setting 0 | Stored and returned 0 | +| Setting 120 | Stored and returned 120 | +| -1, 3601, string `"600"`, fraction 0.5 | Each rejected with HTTP 400 and the valid range | + +The three accepted tasks used metadata-only pending attachments. They stayed +`PENDING_UPLOADS`, launched no worker, acquired no reservation, and were canceled +through the normal API. No attachment was uploaded or signed upload instruction +retained. Direct handler invocation with a synthetic trusted identity verifies +deployed validation and persistence; it does not test API Gateway authentication. + +## Six live worker cases + +The private durable coordinator used the production handler, source `a81c565d`, +normal image 5.0 and the original server as PID 1. Each task was repository-free, +limited to one attempted Read of `/etc/os-release`, six model turns, a $1 model +budget and a 1,800-second worker/durable ceiling. Normal suspension stayed off. + +| Case | Task | Result | +|---|---|---| +| Default ten minutes | `01M2NHJ8YCQAM7VPMQFES4N2HJ` | Sleep requested after 600.385 seconds; approval completed | +| Sleep off | `01M2NHJ8YEVH062H60ZX45XKCE` | No Suspend request; approval completed | +| Custom thirty seconds | `01M2NHJ8YF5709VYHPXNMAMC0N` | Sleep requested after 30.395 seconds; approval completed | +| Late approval wins | `01M2NHJ8YQZQ0T1QFTA5B0J8ZB` | Approval committed after the original deadline; the approved Read completed | +| Timeout wins | `01M2NHJ8YSRTRY3XPT06SHT9QD` | **Failed acceptance:** pre-deadline wake ended in connection refusal before the intended timeout race | +| Credential renewal denied | `01M2NHJ8YS7WTCVPWQTABNM5D3` | Expected failure: resume HTTP 503, precise credential-refresh diagnostic, no Read result | + +The four successful coding cases each produced exactly one Read result and a +complete trace with no dropped events. All five passing cases preserved the +original deadline and lifetime observations, terminated their worker, released +capacity and removed payloads without observer repair. + +The late-approval fixture intentionally suppressed coordinator wake delivery +until after the deadline. Those wrapper responses are explicitly marked +synthetic acknowledgments, not AWS receipts. The normal approval API committed +the decision after expiry while the worker was still suspended. This verifies +the existing conditional-write rule: a committed approval can win before the +worker's timeout write. + +For the credential fault, an exact task-tag condition temporarily denied STS +role renewal for this fixture alone. The guest reached `/resume` and logged: + +- `callback_stage=credential-refresh` +- `error_type=ClientError`, `aws_error_code=AccessDenied` +- AWS request ID `149c138b-54e6-4dc8-ba95-1138136d4a14` +- Resume HTTP 503 after 441 ms; no subsequent identity transaction or Read result + +The coordinator surfaced `MICROVM_RESUME_HOOK_FAILED` and cleaned up. The +temporary deny was removed and its absence verified. This application rejection +is distinct from the unexpected case, whose retained guest stream contains no +resume hook entry at all. + +## Capacity and cleanup + +While the default worker waited, a bounded check exercised the normal deployed +ConcurrencyReconciler. An owned counter was deliberately increased from 1 to 3, +representing the real waiting worker, an added terminal task with a held +reservation, and one extra overcount. The reconciler repaired 3 to 2, then released +the terminal reservation to leave 1. The waiting worker and its reservation +were preserved. The auxiliary task was removed without independent counter +repair. Lambda receipt: `8710433a-4624-488d-b123-5cfd0c2c3736`. + +All six worker instances are terminated. Their reservations are released and +payload prefixes empty. The private function and all versions, role, Parameter +Store switch, log group and six zero counters were removed; 2,213 coordinator +log events were saved before deletion. An immediate post-delete function read +still returned a value, so the first absence assertion failed. A subsequent +read-only check confirmed that the function, role, parameter and log group were +absent. Both observations are retained. + +The failed timeout fixture called its fallback cleanup helpers after the +coordinator had already recorded failure and released its reservation. It is +excluded from successful lifecycle acceptance. Cleanup does not turn this +failure into a pass. + +Evidence is retained under +`/tmp/abca-645-p2-clean-20260913/p3-user-sleep-20260916` and +`/tmp/abca-645-p2-clean-20260913/p3-user-sleep-rollout-20260916`. +The [implementation plan](./645-p3-implementation-plan.md) tracks the remaining +acceptance gates. The [service feedback](./645-lambda-microvm-service-feedback.md) +contains the fifth failure's worker ID, timestamps and API receipts; it has not +been submitted externally. From 934c6874a41deba61a5c710667f5f9de6c73a045 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Wed, 16 Sep 2026 13:56:29 -0400 Subject: [PATCH 055/149] docs: record durable registration and orphan recovery checks --- .../645-lambda-microvm-service-feedback.md | 12 +- .../645-p3-implementation-plan.md | 14 ++- .../645-p3-registration-20260916.md | 113 ++++++++++++++++++ 3 files changed, 133 insertions(+), 6 deletions(-) create mode 100644 docs/verification/645-p3-registration-20260916.md diff --git a/docs/verification/645-lambda-microvm-service-feedback.md b/docs/verification/645-lambda-microvm-service-feedback.md index 5086eb2f7..8de9dfac5 100644 --- a/docs/verification/645-lambda-microvm-service-feedback.md +++ b/docs/verification/645-lambda-microvm-service-feedback.md @@ -16,7 +16,7 @@ is the receipt that lets the service team find a particular call. | F04 | Medium | `PENDING` also means restoring an existing worker | Observed live; our timer bug is fixed | | F05 | Medium | HTTP connection handling across suspend/resume | Contract question; no established cause of F01 | | F06 | Medium | Conditional operator-role requirement for VPC connectors | Earlier deployment failure; application setup fixed | -| F07 | P2/P3 acceptance gap | Run token retention and recovery without a worker ID | Replay observed through roughly five minutes; maximum retention and post-expiry behavior unknown | +| F07 | P2/P3 acceptance gap | Run token retention and recovery without a worker ID | Guest-log recovery verified; maximum retention, post-expiry behavior and recovery without identity logs unknown | ## F01 — Wake request accepted, then the hook connection is refused @@ -230,9 +230,17 @@ does not establish the logging contract or exclude later delivery. Is Run a management or data event, and what audit configuration is required to retain the launch token-to-worker mapping? +**Recovery follow-up:** the [September 16 durable check](./645-p3-registration-20260916.md) +discarded both accepted Run responses without saving their worker IDs. After the +task failed, the operator recovered the still-running worker from an exact +task/worker pair in its accepted `/run` guest log, checked its service identity +and terminated it. No Read result occurred and no replacement worker was +launched. This is a verified operator path when identity logs exist; it is not +a service token lookup or automatic recovery. + **Limits:** five minutes of successful replay does not establish the maximum retention period. The 120-second limit is ABCA policy, not an AWS guarantee. -Post-expiry behavior and recovery of a genuinely unknown worker ID remain +Post-expiry behavior and recovery without unambiguous guest identity logs remain unverified. Service response: pending. ## Updating this tracker diff --git a/docs/verification/645-p3-implementation-plan.md b/docs/verification/645-p3-implementation-plan.md index 0bb3f56e8..f79dc2c7b 100644 --- a/docs/verification/645-p3-implementation-plan.md +++ b/docs/verification/645-p3-implementation-plan.md @@ -123,7 +123,8 @@ is superseded by these records. - [x] Verify simultaneous identical Run calls, changed-parameter rejection, and replay after termination through roughly five minutes against AWS; distinguish cached Run responses from fresh VM state. - [x] Verify production start/receipt/payload code against AWS with lost replies and local process death, saved-handle recovery, changed-input/cancellation refusal and actual receipt expiry. - [x] Verify real AWS durable replay after a saved worker receipt and process exit, including cancellation during recovery, in the isolated production-handler fixture. -- [ ] Complete remaining durable registration races, AWS behavior after token retention expires, and operator cleanup of genuinely unknown worker IDs. +- [x] Verify deployed durable recovery after a committed registration reply is lost, cancellation during registration, and operator recovery/termination of a worker whose ID was not saved. +- [ ] Establish AWS behavior after token retention expires and recovery when guest identity logs are missing or ambiguous. - [x] Make capacity acquisition/release atomic per task across crash replay; unify counter writers and repair. - [x] Verify normal deployed capacity repair while preserving a waiting worker; verify bounded 600-user AWS pagination, partial-scan safety and conservative legacy handling. - [ ] Complete the capacity protocol's old-writer upgrade/drain and rollback procedure; validate volume against the intended production workload. @@ -281,15 +282,20 @@ handoff, use this order: failure guidance in coordinator version 7 and verified the normal task API; image 5.0 and disabled suspension settings remain in place. Complete the remaining live race/fault matrix: - timeout-winning decision races, durable registration races, - service token-retention expiry and recovery of a worker - whose ID was genuinely lost. Keep each injected failure distinct from an + timeout-winning decision races, service token-retention expiry and recovery + when guest identity evidence is unavailable. Keep each injected failure distinct from an unrelated service failure. The [adjustable-sleep follow-up](./645-p3-user-sleep-20260916.md) completed the late-approval winner and actual credential-refresh denial checks, plus default, off and custom delays. It deployed coordinator version 8 with both gates off. The timeout-winner case was interrupted by the fifth unexplained wake refusal and remains unaccepted. + The [durable registration follow-up](./645-p3-registration-20260916.md) + passed a lost reply after an actual registration commit, cancellation before + identity registration, and explicit operator recovery/termination of a live + worker whose ID never reached coordinator state. The latter required an + exact task/worker pair in the guest log; it does not establish post-retention + behavior or a recovery path when those logs are unavailable. 3. Complete effective permissions and network checks for the other backends, plus runtime/remote-MCP connectivity. Verify a full cloned-repository P3 workflow on the final image, including mutable workspace state and normal diff --git a/docs/verification/645-p3-registration-20260916.md b/docs/verification/645-p3-registration-20260916.md new file mode 100644 index 000000000..62db3bf76 --- /dev/null +++ b/docs/verification/645-p3-registration-20260916.md @@ -0,0 +1,113 @@ +# ADR-021 P3: durable registration and unknown-worker recovery + +Verified September 16, 2026, in account ``, `us-west-2`. +All three cases passed on normal image 5.0 with sleep disabled. They used the +production durable coordinator from source `a81c565d` with explicit SDK-boundary +faults in a private wrapper. Normal coordinator version 8 was unchanged. + +## What was verified + +| Case | Task | Worker | Result | +|---|---|---|---| +| Reply lost after registration commits | `01M2NMZS79FGKJVA3EP10AZB2W` | `microvm-ca2ceb94-be3f-3bec-a02a-9d4bb4396ee0` | Recovered saved identity; one Run, one approved Read, normal cleanup | +| Cancellation during registration | `01M2NMZS7F6JPKPBF9YVDJBY3J` | `microvm-162d5f5c-ed2c-3a3e-b066-af2685ea199d` | Cancellation committed first; identity retained for automatic termination; zero tools | +| All launch replies lost | `01M2NMZS7FARVEH9VE1ES3DGZB` | `microvm-3e8f25f0-0d9d-34fe-84e4-b27bcfa5e4f9` | Task reported uncertain start; operator recovered and stopped the live worker from an exact guest-log identity match | + +Each repository-free fixture permitted at most one attempted Read of +`/etc/os-release`, with a six-turn/$1 model budget and 1,800-second worker and +durable execution ceilings. No publication or notification was requested. + +### Lost registration reply + +The actual DynamoDB handle-registration write committed at 17:48:07.652 UTC, +request `I5PMH0TGHU49H2U0GSBL82FGQFVV4KQNSO5AEMVJF66Q9ASUAAJG`. +The wrapper then threw a timeout instead of delivering the successful reply. +The production code read the saved identity and continued with that worker. + +Only one Run call occurred, receipt `47b3b36e-1da7-423f-b170-ca9c0ec737d0`. +The task completed after one approved Read. Its complete trace contains exactly +that tool call and no dropped events. The coordinator terminated the worker, +released its reservation and removed payloads without observer repair. + +### Cancellation before identity registration + +The normal cancellation API returned HTTP 200 at 17:48:37.385 UTC, receipt +`aabf2ae3-feaa-4bd5-9070-3bf5e89e1649`. The subsequent identity write committed +at 17:48:37.406 with the task already `CANCELLED`. + +Saving the known worker ID after cancellation is intentional: it gives cleanup +a computer to stop. It does not restore the task to running. The coordinator +observed cancellation and terminated the same worker. No tool call or result +occurred. Reservation release and payload deletion completed automatically. + +### Worker ID absent from coordinator state + +The first Run response was held until the owned guest reached its approval +wait, then discarded. This ensures a live worker exists for the intended +recovery check. The wrapper discarded the retry's response too; neither +returned worker ID nor endpoint was written to the task, wrapper steps or +coordinator logs before operator recovery. + +| Time UTC | Evidence | +|---|---| +| 17:48:42.835 | First Run request uses the task ID as its stable token | +| 17:48:44.867 | Guest logs the accepted `/run` with both task ID and worker ID | +| 17:48:50.039 | Owned task observed waiting for approval, without a saved worker ID | +| 17:48:50.045 | First accepted reply discarded; receipt `60a86e07-2eb9-4fdb-ad66-92f04a4a6529` | +| 17:48:50.258 | Retry reply discarded; receipt `b3d0ab9c-f3e3-4210-b65a-a6c36a0e9433` | +| 17:48:51.604 | Observer reads failed task and completed durable execution | +| 17:48:55.814 | Operator recovers the exact worker from the guest log and verifies it is `RUNNING` | +| 17:48:55.983 | Operator requests termination; receipt `a8f66ec1-d23c-4ae3-b085-ca2bb5a08b94` | +| 17:48:57.662 | Worker observed `TERMINATED` | + +Both Run attempts used the same token. The task and normal task API reported +`MICROVM_START_OUTCOME_UNKNOWN`. A snapshot taken before operator recovery +contains no session ID, saved start handle or compute worker ID. The pending +Read never produced a result. + +The coordinator released capacity and removed payloads. The operator explicitly +terminated the recovered computer; this case does **not** claim automatic +cleanup of an unknown ID. The final inventory contained exactly the three new +workers and no active worker. This check exhausted the normal two start attempts +within seconds; it does not retest the separate 120-second receipt cutoff. + +## Operator procedure when the start result is uncertain + +1. Read the task consistently and inspect its saved start receipt and compute + metadata. If an ID is already saved, use that identity and normal cleanup. + Preserve the original task ID, client token, timestamp and error evidence. +2. If the ID is absent, inspect the configured MicroVM guest log group in the + original region and launch window. In this deployment it is + `/aws/lambda-microvms/backgroundagent-dev-abca-agent`. Require an explicit + accepted `/run` entry containing **both** the exact task ID and worker ID. + A nearby timestamp or membership in the same shared image is insufficient. +3. Require one unambiguous identity. Read `GetMicrovm` and check its image + ARN/version, execution role, start time and current state against the original + launch. Confirm the task is terminal or has been canceled before stopping + a worker that may still be active. +4. Terminate that exact worker, verify its terminal state, and verify task + reservation/counter and payload cleanup. Retain the operation receipt and + identity evidence privately. Do not start another worker to mask the unknown + result. + +This procedure requires retained guest identity logs. A bootstrap failure before +that log entry, missing logs or multiple candidates requires further +investigation; do not guess which shared worker to terminate. The +[service feedback F07](./645-lambda-microvm-service-feedback.md) still asks for a +supported token/receipt-to-worker mapping and guaranteed token-retention behavior. + +## Evidence and cleanup + +The private function was `backgroundagent-dev-p3-registration-20260916`, +version 2, code SHA-256 +`khdHcavtr07wQQDqzWowtMNo/q5XZak+2JodEMhmtXk=`. +Unused version 1 and its original artifacts were retained in the evidence; no +task used it. Version 2's wrapper bundle SHA-256 is +`cc058531418d339cb295817b153353ab0c8e68b7a9ee231def58c533f671d5e6`. + +Evidence under `/tmp/abca-645-p2-clean-20260913/p3-registration-20260916` +includes durable histories, all task events/approvals, complete diagnostic +windows, the pre-recovery task snapshot, exact guest identity entry, inventory, +trace and guarded cleanup ledger. All worker instances are terminated; the +cleanup ledger records removal and absence checks for the private function +versions, role, parameter, log group and zero counters. From bd435b5fe23593005db4a4520d2c2cfb6aa5d342 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Wed, 16 Sep 2026 14:35:16 -0400 Subject: [PATCH 056/149] docs: record AgentCore compatibility and cleanup proof (#645) --- .../shared/strategies/agentcore-strategy.ts | 2 +- .../verification/645-p3-agentcore-20260916.md | 104 ++++++++++++++++++ .../645-p3-capacity-scan-20260916.md | 6 + .../645-p3-implementation-plan.md | 19 +++- .../645-p3-user-sleep-20260916.md | 9 ++ 5 files changed, 138 insertions(+), 2 deletions(-) create mode 100644 docs/verification/645-p3-agentcore-20260916.md diff --git a/cdk/src/handlers/shared/strategies/agentcore-strategy.ts b/cdk/src/handlers/shared/strategies/agentcore-strategy.ts index c4ef04e50..9835b9ec5 100644 --- a/cdk/src/handlers/shared/strategies/agentcore-strategy.ts +++ b/cdk/src/handlers/shared/strategies/agentcore-strategy.ts @@ -49,7 +49,7 @@ export class AgentCoreComputeStrategy implements ComputeStrategy { // injection: when set, AgentCore exchanges the caller's identity for // a workload token and delivers it to the agent container via the // `WorkloadAccessToken` request header (read by - // `BedrockAgentCoreContext.set_workload_access_token` in app.py). + // `BedrockAgentCoreContext.set_workload_access_token` in server.py). // Without it, the agent's `resolve_linear_api_token()` short-circuits // before reaching the Identity SDK call. Requires the orchestrator // role to have `bedrock-agentcore:InvokeAgentRuntimeForUser` in diff --git a/docs/verification/645-p3-agentcore-20260916.md b/docs/verification/645-p3-agentcore-20260916.md new file mode 100644 index 000000000..85ec06e51 --- /dev/null +++ b/docs/verification/645-p3-agentcore-20260916.md @@ -0,0 +1,104 @@ +# ADR-021 P3: AgentCore compatibility verification + +Verified on 2026-09-16 in `backgroundagent-dev`, account ``, +region `us-west-2`. These results cover the shared AgentCore runtime and +approval/cancellation paths. The [MicroVM wake refusal](./645-p3-resume-refusal-investigation.md) +and other unfinished [P3 gates](./645-p3-implementation-plan.md) remain open. + +## Reviewed runtime update + +The normal AgentCore container was still older than the deployed MicroVM code. +The update uses the ARM64 container built from source `a81c565d`: + +- ECR asset tag: `3b50ba93fcff4e8129ee6140a8c65d876b45e47744c92775bc80bd078fd5d486`. +- Image digest: `sha256:8490592823b913558407b7b03d9c4f346c72fde2f15503849a50539fb7a654df`. +- Runtime: `backgroundagentdevRuntimeCC6E3A5A-Eq3mE6Gg2d`. +- Runtime version and `DEFAULT` endpoint advanced from `4` to `5`, both `READY`. + +The actual CloudFormation change set modified only +`Runtime99E3DDFA.AgentRuntimeArtifact.ContainerConfiguration.ContainerUri`, +with no replacement. All 475 resource identities remained unchanged. +Coordinator alias `live` still points to version `8`; MicroVM image `5.0` +remains `ACTIVE`/`SUCCESSFUL`; both normal suspension gates remain off. +The old AgentCore version `4` is still readable and references its original +container. This establishes retention, not a live rollback exercise. + +The reviewed template came from the exact deployed S3 bytes, SHA-256 +`4957871b24243f900e16fa9a04b58db63d85b1da5fcf7265d5f403fe5ecd3e3c`. +The new compact template is 706,153 bytes, SHA-256 +`9741f2d4ea6f43e8dde6deff0cb791780a54efdb933e7046c9851b9e6a2e3f9d`. +An initial preparation failed locally because stack metadata was missing. +A subsequent attempt uploaded the image but failed template validation: +pretty-printing expanded the JSON to 1,281,334 bytes. Restoring compact JSON +fixed that size error without changing resource properties. Neither failed +attempt updated the stack. + +## Bounded live cases + +A private durable Lambda ran the production coordinator with a fixed AgentCore +blueprint and two owned tasks. Its role restricted table operations to those +task/user keys and runtime calls to the normal AgentCore runtime. The tasks had +no repository or notification destination, a six-turn/$1 budget, and a +180-second approval window. Both requested `microvm_sleep_after_s=30`. +Only the private fixture's suspension gates were enabled. + +The fixture refused every MicroVM SDK call. It recorded each generated AgentCore +session ID before the actual invocation, then retained the AWS invocation +receipt. The approval API and cancellation API were the normal deployed +handlers. Their Lambda invocations supplied a synthetic trusted user context; +this did not exercise API Gateway authentication. + +| Case | Task | Session | Result | +|---|---|---|---| +| Approve after the sleep delay | `01M2NQFXGBXSAPV0PKFMXYTVX3` | `11991ca4-f2f5-4a83-89d0-cc161a76f92b` | `COMPLETED`; one successful `Read` | +| Cancel at the approval gate | `01M2NQFXGBQVTBQPDWKTTTJD44` | `3c0398a3-ce01-4444-b370-b7c111b009f2` | `CANCELLED`; no tool result | + +The approval case stayed at the gate for an observed 41.273 seconds before the +API returned HTTP 202. It never acquired MicroVM lifecycle/start state or called +the MicroVM service. The approved action read only `/etc/os-release`; task events +show exactly one call and one successful result. Its stored trace had zero +dropped events and SHA-256 +`453919a3accb63bd93dbec08685c121c011da8573254379ff6308bec224958d6`. +Invocation receipt: `24ca6184-fe2c-47aa-a521-552cc29d3cb2`. +Approval receipt: `0eb437b1-029f-4269-a5ce-6782caf0182e`. + +Cancellation returned HTTP 200, kept the task terminal, and stopped its session. +The blocked `Read` attempt produced no result. Its approval record remained +`PENDING`; cancellation did not approve the tool. No final trace was available +for this interrupted task. Invocation receipt: +`cf1dee27-8ead-4c46-8fb3-3c3cc7165028`. +Cancellation receipt: `5c9ca126-f456-42ba-b7c6-f850fa072045`. + +Both durable executions finished `SUCCEEDED`, both task reservations were +released, and both counters reached zero without independent reservation repair. +The successful session was explicitly stopped by fixture cleanup +(`53369953-9911-4aae-989c-da2cffed0c4a`). +Both sessions subsequently returned `ResourceNotFoundException` to another +owned-session stop request, verifying absence. + +## Excluded attempt and scope limits + +An initial fixture patched the root workspace's AgentCore SDK copy, while the +production strategy imported a separate copy under `cdk/node_modules`. +Its task completed, but the missing pre-invocation audit invalidated the intended +fixture. That attempt, task `01M2NQ6087WM33JR0EZT7WRDNJ`, is excluded from the +acceptance results. Its session was explicitly stopped and its zero counter +removed. The corrected bundle resolved the exact production SDK and used fresh +task identities. Both versions and all raw evidence were retained for audit +before infrastructure cleanup. + +Cleanup completed at **18:30:28.935 UTC**. The private coordinator's two +versions, role, switch, log group and zero counters were removed after ownership +checks. All three durable executions, including the excluded attempt, had +finished. The archive includes 252 coordinator log events and 17 relevant +runtime log events. Read-only checks confirmed the temporary resources were +absent; the normal runtime and its shared logs were retained. + +This verifies AgentCore's shared approval/cancellation behavior and exclusion +from MicroVM sleep. It does not verify the full other-backend IAM/network matrix, +remote MCP connectivity, ECS, capacity migration, or a cloned-repository P3 run. +The normal stack contains no ECS cluster or task definition. + +Evidence directories: +`/tmp/abca-645-p2-clean-20260913/p3-agentcore-compatibility-20260916` and +`/tmp/abca-645-p2-clean-20260913/p3-agentcore-fixtures-20260916`. diff --git a/docs/verification/645-p3-capacity-scan-20260916.md b/docs/verification/645-p3-capacity-scan-20260916.md index 7b6dcc213..824f420fa 100644 --- a/docs/verification/645-p3-capacity-scan-20260916.md +++ b/docs/verification/645-p3-capacity-scan-20260916.md @@ -28,6 +28,12 @@ same tables. The deployed handler uses projections and does not itself request consumed-capacity telemetry. Its successful results required accounting for all 600 users. +A subsequent read-only measurement of the normal development deployment found +86 task rows in one page (58 read capacity units) and 36 counter rows in one page +(4 units). The 600-row fixture therefore exceeded this deployment's current +row count and scan volume. It does not establish capacity for a future production +retention policy or workload. + ## Results The first deployed invocation: diff --git a/docs/verification/645-p3-implementation-plan.md b/docs/verification/645-p3-implementation-plan.md index f79dc2c7b..5be653266 100644 --- a/docs/verification/645-p3-implementation-plan.md +++ b/docs/verification/645-p3-implementation-plan.md @@ -20,6 +20,17 @@ all six workers and temporary verification infrastructure were cleaned up. The normal deployed capacity reconciler also passed an owned overcount and terminal-reservation repair while preserving the real waiting worker. +**AgentCore compatibility follow-up (2026-09-16):** the +[reviewed container update and live checks](./645-p3-agentcore-20260916.md) +advanced the normal AgentCore runtime and its default endpoint to version 5. +The actual change modified only its container URI; all 475 resource identities +and normal MicroVM settings stayed unchanged. Fresh production-handler fixtures +passed approval and cancellation, kept AgentCore awake beyond a requested +MicroVM sleep delay, and released both reservations. Both test sessions were +verified absent. An earlier wrapper-invalidated attempt is explicitly excluded. +ECS, the wider role/network matrix and the unexplained MicroVM wake failures +remain separate gates. + **Guest hook milestone (2026-09-14):** production [worker suspend/resume hooks](./645-p3-lifecycle-hooks.md) now connect the guest barrier to atomic checkpoint writes and retained-credential refresh followed by @@ -300,6 +311,10 @@ handoff, use this order: plus runtime/remote-MCP connectivity. Verify a full cloned-repository P3 workflow on the final image, including mutable workspace state and normal P2 behavior. Respect the target repository's publication checks. + The [AgentCore follow-up](./645-p3-agentcore-20260916.md) now verifies its + current shared container, approval/cancellation, exclusion from MicroVM sleep, + reservation release and owned-session cleanup. The stack has no ECS resources; + those checks require a separate bounded deployment. 4. Exercise the coordinated capacity upgrade/drain and rollback procedure under deployed writer roles, including realistic scan volume. Retain the verified local transaction and isolated-live results as evidence for their narrower @@ -309,7 +324,9 @@ handoff, use this order: equivalent table permissions, zero writes after interrupted scans, and conservative handling until an older task settles. Its temporary tables and function were removed. This bounds the verified volume without claiming the - production migration or arbitrary retention scale. + production migration or arbitrary retention scale. A subsequent measurement + found 86 task rows and 36 counter rows in the normal development deployment, + below the fixture's 600 rows per table. 5. Perform the final compatible rollout, including shared runtime changes for ECS/AgentCore, pinned-version retention and rollback checks. Enable automatic suspension only after the remaining gates pass, then finish the ADR/runbook diff --git a/docs/verification/645-p3-user-sleep-20260916.md b/docs/verification/645-p3-user-sleep-20260916.md index b543041f8..3f8cb7488 100644 --- a/docs/verification/645-p3-user-sleep-20260916.md +++ b/docs/verification/645-p3-user-sleep-20260916.md @@ -137,3 +137,12 @@ The [implementation plan](./645-p3-implementation-plan.md) tracks the remaining acceptance gates. The [service feedback](./645-lambda-microvm-service-feedback.md) contains the fifth failure's worker ID, timestamps and API receipts; it has not been submitted externally. + +The durable private archive is +`~/.local/share/abca-verification/645-p3-20260916/sleep-registration-capacity-evidence.tar.gz` +(208 files, 462,751,059 bytes, mode 0600). Its SHA-256 is +`359dd94c366bd9b9122e4e6ff46e7b717c1ea00830cdf0d63e848e66fcc7c7ae`. +Every archived file was checked against its manifest. It also includes the +[capacity scan](./645-p3-capacity-scan-20260916.md) and +[registration/recovery](./645-p3-registration-20260916.md) evidence, cleanup +ledgers, private wrapper bundles and exact source snapshot `0ddcb4ea`. From a140c772a0592b13cbe1f202fc70058fd89d0fa6 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Wed, 16 Sep 2026 15:11:34 -0400 Subject: [PATCH 057/149] fix(microvm): classify generic wake-hook failures (#645) --- cdk/src/handlers/approve-task.ts | 20 +++++++------------ cdk/src/handlers/shared/error-classifier.ts | 6 +++--- .../handlers/shared/error-classifier.test.ts | 1 + cdk/test/handlers/shared/orchestrator.test.ts | 1 + 4 files changed, 12 insertions(+), 16 deletions(-) diff --git a/cdk/src/handlers/approve-task.ts b/cdk/src/handlers/approve-task.ts index 043f3fb68..981ae09c8 100644 --- a/cdk/src/handlers/approve-task.ts +++ b/cdk/src/handlers/approve-task.ts @@ -308,22 +308,16 @@ export async function handler( * * Per §7.1, the cancellation reasons are per-item (index 0 is the * approvals-row Update, index 1 is the task-row Update). We read them - * to distinguish: - * - approvals item cancelled: - * - via `attribute_exists` failure → row missing → 404 - * - via `user_id = :caller` failure → wrong owner → 404 (no oracle) - * - via `status = :pending` failure → already decided → 409 + * to classify the failed item: + * - approvals item cancelled → 404 for missing, wrong-owner or already-decided rows * - task-row item cancelled only → task not in AWAITING_APPROVAL → 409 * * DDB does not return which sub-clause of the `ConditionExpression` - * failed, so we infer from whichever row was cancelled. If ONLY the - * approvals Update tripped, it could be any of {missing, wrong owner, - * wrong status}; we conservatively return 404 to prevent the existence - * oracle. The more-specific 409 ALREADY_DECIDED path requires - * additional information we do not have from the reason array alone; - * implementations that want the stronger distinction need to do a - * subsequent GetItem, which re-introduces the race the transaction - * eliminates. v1 accepts the less-granular 404 on ownership drift. + * failed. This transaction does not request the old item on condition failure, + * so an approvals-row failure cannot distinguish those three cases. It takes + * precedence even when the task-row condition also fails. Returning 404 for all + * three avoids revealing another user's approval. In particular, a late decision + * after an approval timeout returns REQUEST_NOT_FOUND, not ALREADY_DECIDED. */ function classifyCancel( err: TransactionCanceledException, diff --git a/cdk/src/handlers/shared/error-classifier.ts b/cdk/src/handlers/shared/error-classifier.ts index b64577f94..64dd0d40a 100644 --- a/cdk/src/handlers/shared/error-classifier.ts +++ b/cdk/src/handlers/shared/error-classifier.ts @@ -137,7 +137,7 @@ const MICROVM_TERMINAL_CLASSIFICATIONS: Readonly { + `substrate state completed (${reason})`; test.each([ + 'Resume lifecycle hook failed. Please check your hook endpoint and application logs for more details.', 'Resume lifecycle hook connection was refused. Please check your hook endpoint and application logs for more details.', 'Resume lifecycle hook returned HTTP status 503.', 'Resume lifecycle hook returned HTTP status 409.', diff --git a/cdk/test/handlers/shared/orchestrator.test.ts b/cdk/test/handlers/shared/orchestrator.test.ts index a358aa4fc..74c547d27 100644 --- a/cdk/test/handlers/shared/orchestrator.test.ts +++ b/cdk/test/handlers/shared/orchestrator.test.ts @@ -194,6 +194,7 @@ describe('MicroVM terminal finalization', () => { ["agent_status='success', build_ok=timeout [auto-retried]", 'MICROVM_SUBSTRATE_TERMINATED', 'compute', true], ['Run lifecycle hook returned HTTP status 400.', 'MICROVM_RUN_HOOK_REJECTED', 'config', false], ['Run lifecycle hook returned HTTP status 500.', 'MICROVM_SUBSTRATE_TERMINATED', 'compute', true], + ['Resume lifecycle hook failed. Please check your hook endpoint and application logs for more details.', 'MICROVM_RESUME_HOOK_FAILED', 'compute', false], ])('persists stable classification and consistent user guidance for %s', async (reason, code, category, retryable) => { primeReread(TaskStatus.RUNNING); await finish({ status: 'completed', reason }); From 30d101c2f5720dee7cf5b909f508dcc04a9330e3 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Wed, 16 Sep 2026 15:23:04 -0400 Subject: [PATCH 058/149] docs: record independently observed wake failure (#645) --- .../verification/645-capacity-reservations.md | 10 ++ .../645-lambda-microvm-service-feedback.md | 41 +++++ .../verification/645-p3-agentcore-20260916.md | 6 + .../645-p3-implementation-plan.md | 26 +++- .../645-p3-pid1-observer-20260916.md | 140 ++++++++++++++++++ .../645-p3-resume-refusal-investigation.md | 10 ++ 6 files changed, 230 insertions(+), 3 deletions(-) create mode 100644 docs/verification/645-p3-pid1-observer-20260916.md diff --git a/docs/verification/645-capacity-reservations.md b/docs/verification/645-capacity-reservations.md index 64df74d07..87ddcf175 100644 --- a/docs/verification/645-capacity-reservations.md +++ b/docs/verification/645-capacity-reservations.md @@ -52,6 +52,16 @@ terminal reservation while preserving a real waiting worker. The bounded volume and role checks do not complete the old-writer drain/upgrade/rollback procedure or establish arbitrary production scale. +A subsequent read-only deployment audit checked upload confirmation, coordinator +version 8, counter reconciliation, queue pickup and stranded-task reconciliation. +All five referenced the same code assets as the `a81c565d` assembly, and each +downloaded S3 ZIP matched its deployed Lambda `CodeSha256`. At that observation, +none of the normal coordinator's retained versions 2–8 had a running durable +execution. This establishes artifact alignment and a point-in-time execution +inventory; it does not prove admissions were paused or the upgrade/rollback +sequence was rehearsed. Evidence is in +`/tmp/abca-645-p2-clean-20260913/p3-capacity-writer-inventory-20260916`. + Local tests prove the application requests and DynamoDB Local's transaction behavior. They do not establish deployed IAM, AWS scaling, successful rollout or MicroVM sleep/wake behavior. Terminal events may repeat or be lost independently of the atomic seat update. The reservation/start markers share the task row. Subsequent prerequisite work restricts agent updates to reporting/approval attributes and removes whole-row replacement/deletion plus direct worker access to the counter. Public-API omission alone was not protection. See [coordinator metadata verification](./645-coordinator-metadata.md) for the writer inventory, actual policy boundary, remaining status/tag trust limits and required AWS authorization checks. These local transaction tests do not prove that security boundary. diff --git a/docs/verification/645-lambda-microvm-service-feedback.md b/docs/verification/645-lambda-microvm-service-feedback.md index 8de9dfac5..92bf8a1b3 100644 --- a/docs/verification/645-lambda-microvm-service-feedback.md +++ b/docs/verification/645-lambda-microvm-service-feedback.md @@ -17,6 +17,7 @@ is the receipt that lets the service team find a particular call. | F05 | Medium | HTTP connection handling across suspend/resume | Contract question; no established cause of F01 | | F06 | Medium | Conditional operator-role requirement for VPC connectors | Earlier deployment failure; application setup fixed | | F07 | P2/P3 acceptance gap | Run token retention and recovery without a worker ID | Guest-log recovery verified; maximum retention, post-expiry behavior and recovery without identity logs unknown | +| F08 | P3 blocker | Generic wake-hook failure with an observed listener | PID 1 owned its listener after restore, 519 ms before termination; no resume hook entry | ## F01 — Wake request accepted, then the hook connection is refused @@ -62,6 +63,8 @@ defect fixed or establish a failure rate. **Next:** retain a fresh failure with independent process/listener evidence, or obtain the service-side evidence above. Service response: pending. +F08 now supplies an independently observed failed wake with different service +wording. Its relationship to these five connection refusals remains unknown. ## F02 — IAM conditions and errors make correct setup difficult @@ -117,6 +120,11 @@ failures. The latter now has specific platform-error guidance deployed in coordinator 7 and verified through the normal task API. These controlled failures and passing race cases do not explain F01. +F08 exposed another service message, `Resume lifecycle hook failed.`, without +an HTTP status or underlying connection error. The local classifier correction +recognizes that observed wording and prevents misleading retry advice. It does +not establish why the service failed to complete the hook. + **Ask:** provide a service-side lifecycle attempt timeline or equivalent structured fields: originating API receipt, hook kind/attempt ID, start/end times, connection versus HTTP failure, HTTP status, underlying error code and @@ -243,6 +251,39 @@ retention period. The 120-second limit is ABCA policy, not an AWS guarantee. Post-expiry behavior and recovery without unambiguous guest identity logs remain unverified. Service response: pending. +## F08 — Generic wake-hook failure while PID 1 owns its listener + +**Observed:** September 16, 18:55:13–18:56:16 UTC, account ``, +region `us-west-2`. A diagnostic image kept the original server as PID 1 and +added a separate observer child. Application code, connection keepalive, +8,192 MiB and lifecycle hooks matched normal image 5.0. + +- Worker: `microvm-da3668f9-04a6-393c-939a-33c059e4b2a0`. +- Image: `backgroundagent-dev-p3-pid1-observer-20260916:1.0`. +- Task: `01M2NRA1BNVMXGD5YDVYD4XH4T`. +- Suspend receipt: `802e1264-4569-49ef-aada-70be7c26e71b`; guest checkpoint + and suspend HTTP 200 completed at 18:55:14.013. +- Resume receipt: `11bc8d9b-ada5-440e-aef3-034d8ac04848`, accepted at + 18:56:13.121. +- At 18:56:13.350, observer PID 7 saw PID 1 running and owning its port-8080 + listening socket, inode `481`, which also existed before the freeze. +- AWS terminated the worker at 18:56:13.869 with + `Resume lifecycle hook failed. Please check your hook endpoint and application logs for more details.` + The retained log contains no resume hook entry, stage or access line. + +**Ask:** what exact connection/HTTP error maps to this generic reason? For this +receipt, was the hook attempted on a fresh or reused socket, what bytes/status +were received, and was another connection attempted before termination? +Does the service retain the hook-attempt timeline separately from the API +receipt? Compare it with the five exact connection refusals in F01. + +**Limits:** the observed process and listener existed 519 ms before termination. +This does not prove continuous health or event-loop responsiveness, and it does +not establish whether this failure shares F01's cause. The diagnostic process +can affect scheduling. The [full record](./645-p3-pid1-observer-20260916.md) +retains the passing comparison, excluded fixture assertion, sampled state, +timestamps and failure. Service response: pending. + ## Updating this tracker For each new finding, add the actual trigger, UTC window, region, worker/image diff --git a/docs/verification/645-p3-agentcore-20260916.md b/docs/verification/645-p3-agentcore-20260916.md index 85ec06e51..6b816b64c 100644 --- a/docs/verification/645-p3-agentcore-20260916.md +++ b/docs/verification/645-p3-agentcore-20260916.md @@ -102,3 +102,9 @@ The normal stack contains no ECS cluster or task definition. Evidence directories: `/tmp/abca-645-p2-clean-20260913/p3-agentcore-compatibility-20260916` and `/tmp/abca-645-p2-clean-20260913/p3-agentcore-fixtures-20260916`. + +The durable private archive is +`/Users/sphias/.local/share/abca-verification/645-p3-20260916/agentcore-compatibility-evidence.tar.gz`. +It contains 77 files, including source `9cd74c7e`, and is 38,952,488 bytes with +SHA-256 `0409a18b6ebd73a538f49ec4e41f1a3a9fa20ab8e0edc55179e5eb179fe0653b`. +Every file was verified against its manifest; archive permissions are `0600`. diff --git a/docs/verification/645-p3-implementation-plan.md b/docs/verification/645-p3-implementation-plan.md index 5be653266..7145d8170 100644 --- a/docs/verification/645-p3-implementation-plan.md +++ b/docs/verification/645-p3-implementation-plan.md @@ -31,6 +31,18 @@ verified absent. An earlier wrapper-invalidated attempt is explicitly excluded. ECS, the wider role/network matrix and the unexplained MicroVM wake failures remain separate gates. +**Independent PID 1 follow-up (2026-09-16):** the +[new diagnostic](./645-p3-pid1-observer-20260916.md) kept the original server as +PID 1 and observed it from a child process. One corrected timeout-winner case +passed with automatic cleanup. Another wake failed with the distinct generic +message `Resume lifecycle hook failed.` The observer saw PID 1 owning its +listener after restoration, 519 ms before AWS terminated the worker, but no +resume hook entry appeared. A first attempt's incorrect HTTP 409 expectation +is retained and excluded from automatic-finalization acceptance; the existing +late-decision contract is HTTP 404. The final planned attempt was not started. +The new failure is tracked separately as F08; neither it nor the five exact +connection refusals is resolved. + **Guest hook milestone (2026-09-14):** production [worker suspend/resume hooks](./645-p3-lifecycle-hooks.md) now connect the guest barrier to atomic checkpoint writes and retained-credential refresh followed by @@ -284,6 +296,10 @@ handoff, use this order: 200, but no resume hook entry. The prepared service report now includes its API receipts and precise timeline; the new diagnostics have not established the cause. + The [independent PID 1 follow-up](./645-p3-pid1-observer-20260916.md) now + captures process/listener evidence during a separate generic wake-hook + failure. F08 records its service question. An observed listening socket does + not prove the event loop processed the hook or identify the failing connection. 2. The [command-race checks](./645-p3-command-races-20260916.md) now pass cancellation before/after Suspend, during observed `SUSPENDING`, and during restore `PENDING`; approval during an accepted suspension and three consecutive @@ -292,15 +308,19 @@ handoff, use this order: resources were removed. The same follow-up deployed clearer status-read failure guidance in coordinator version 7 and verified the normal task API; image 5.0 and disabled suspension settings remain in place. - Complete the remaining live race/fault matrix: - timeout-winning decision races, service token-retention expiry and recovery + Complete the remaining live fault matrix: + service token-retention expiry and recovery when guest identity evidence is unavailable. Keep each injected failure distinct from an unrelated service failure. The [adjustable-sleep follow-up](./645-p3-user-sleep-20260916.md) completed the late-approval winner and actual credential-refresh denial checks, plus default, off and custom delays. It deployed coordinator version 8 with both gates off. The timeout-winner case was interrupted by the fifth unexplained wake refusal - and remains unaccepted. + and remained unaccepted in that run. A corrected timeout-winning case now + passes on the private PID 1 diagnostic image with unchanged application code, + the original deadline, late HTTP 404 rejection and automatic cleanup. + The diagnostic also reproduced a distinct generic wake failure, so this + individual passing case does not complete final-image enablement. The [durable registration follow-up](./645-p3-registration-20260916.md) passed a lost reply after an actual registration commit, cancellation before identity registration, and explicit operator recovery/termination of a live diff --git a/docs/verification/645-p3-pid1-observer-20260916.md b/docs/verification/645-p3-pid1-observer-20260916.md new file mode 100644 index 000000000..d588cfcb7 --- /dev/null +++ b/docs/verification/645-p3-pid1-observer-20260916.md @@ -0,0 +1,140 @@ +# ADR-021 P3: independent observation with the original PID 1 server + +This bounded diagnostic addresses a gap in the +[resume-refusal investigation](./645-p3-resume-refusal-investigation.md). +The earlier independent observer became the server's parent. Here the launcher +starts an observer child and then executes the original server command in its +own process. The server therefore remains PID 1, as in the five recorded +failures. This is private diagnostic instrumentation; the normal image is +unchanged. + +## Image and measurement + +The diagnostic derives from the exact normal image 5.0 artifact, SHA-256 +`9b3150e9e5991cc9fcc8a4adbb2cfbd8f97399f0c16baa9c2b5fe5b101e7b035`. +Only the Dockerfile command changes and `agent/scripts/pid1_observer.py` is +added. All other 110 artifact entries remain byte-identical. + +Private image: +`arn:aws:lambda:us-west-2::microvm-image:backgroundagent-dev-p3-pid1-observer-20260916`, +version `1.0`. Artifact SHA-256: +`c8f90180a2358678fa850e58a2dab8f5dc6632a00d0814e6b999d889c0cfe077`. +The image retains ARM64, 8,192 MiB, the original application, connection +keepalive settings, six hooks and existing connectors. + +The observer samples `/proc` every 100 ms. It reports the server's state, +thread count, RSS, pending signals, port-8080 listener inodes and whether +the server owns those sockets. It emits state changes, five-second summaries +and gaps in wall or monotonic time. It opens no sockets and records no command +arguments, environment, request bodies or credentials. The guest did not expose +the sampled cgroup `memory.events` file. + +A local ARM64 container control kept the server as PID 1 and the observer as +PID 7. The observer detected an open listener, then its intentional closure +while PID 1 remained alive. The control used no network access and was removed +afterward. The actual AWS worker also showed observer PID 7 independently +reading PID 1's listener. + +## Bounded cases + +The private coordinator uses the production durable handler from `a81c565d`, +with fixed owned task IDs and a repository-free `Read /etc/os-release` approval +gate. Each task allows six turns/$1, has a 150-second approval deadline, and +requests sleep after 30 seconds. The normal policy wakes it 60 seconds before +the original deadline. The observer then waits for the approval row to become +`TIMED_OUT` before submitting a late approval through the normal API. + +Only the private switch is enabled during a run. The four attempts are +sequential, stop on a failing attempt, and have six-minute observer and +1,800-second worker limits. No wake delivery, credentials, guest callback or +service response is deliberately changed. The wrapper records actual +Suspend/Resume request receipts. + +| Attempt | Task | Scope | +|---|---|---| +| 1 | `01M2NRA1BJNSBH4R5WR8VV3SHW` | Valid wake observation; excluded from automatic-finalization acceptance | +| 2 | `01M2NRA1BMK7ZBGDRCB3PBS0WH` | Passed original timeout, late rejection and automatic cleanup | +| 3 | `01M2NRA1BNVMXGD5YDVYD4XH4T` | Unexpected generic resume-hook failure; captured independent listener evidence | +| 4 | `01M2NRA1BN7RKF8E95B75S4VNT` | Not started; the run stopped after attempt 3 | + +Attempt 1 woke successfully. Its fixture expected HTTP 409 after the timeout, +but the deployed approval contract deliberately returns HTTP 404 +`REQUEST_NOT_FOUND` for missing, wrong-owner **and already-decided** approval +rows. The same approval-row condition failed after timeout, so the late +approval was correctly rejected. The fixture's assertion nevertheless triggered +fallback cleanup, including stopping its durable execution. The task finished +`COMPLETED` and the worker was terminated, but this attempt cannot establish +automatic finalization. The raw assertion and cleanup evidence are retained. + +The remaining attempts use the correct HTTP 404 contract. This did not change +the production API. A stale comment claiming an already-decided approval would +return 409 was corrected in `approve-task.ts`. + +The complete timeout case must retain the original deadline, reject the late +approval, return `User timed_out` as an error tool result, finish the task, +terminate the worker, release its reservation and empty its payload prefix +without observer repair. An error tool result here reports the denial; it is +not evidence that the file was read. + +## Failed wake with an observed listener + +Attempt 3 used worker `microvm-da3668f9-04a6-393c-939a-33c059e4b2a0`. +Its single coordinator wake was accepted, then AWS terminated the worker with: + +> Resume lifecycle hook failed. Please check your hook endpoint and application +> logs for more details. + +This is **different wording** from the five connection refusals. It is retained +as a separate failed-wake observation, without assuming a shared cause. + +| UTC, September 16 | Evidence | +|---|---| +| 18:55:13.690 | Coordinator starts Suspend | +| 18:55:13.865 | Suspend accepted; receipt `802e1264-4569-49ef-aada-70be7c26e71b` | +| 18:55:14.013 | Guest checkpoint succeeds; suspend hook finishes HTTP 200 in 91 ms | +| 18:56:13.043 | Coordinator starts Resume | +| 18:56:13.121 | Resume accepted; receipt `11bc8d9b-ada5-440e-aef3-034d8ac04848` | +| 18:56:13.350 | Independent observer runs after a 59.366-second monotonic gap | +| 18:56:13.869 | AWS records termination with the generic resume-hook failure | +| 18:56:15.893 | Observer sees task `FAILED` and durable execution `SUCCEEDED` | + +The post-restoration sample saw PID 1 in `R (running)` state, 29 threads, +156,784 kB RSS, no pending signals and ownership of the port-8080 `LISTEN` +socket, inode `481`. That inode was also present before suspension. +The sample preceded AWS's termination timestamp by approximately 519 ms. +There was no resume hook entry, stage or HTTP access line in the retained +worker log. + +This establishes that the server process and its listening socket existed at +that sample. It does not prove event-loop responsiveness, the state of a reused +HTTP connection, continuous listener health until termination, or the service's +underlying transport error. The observer samples every 100 ms but emits unchanged +state only every five seconds. + +The old classifier persisted this generic wording as +`MICROVM_SUBSTRATE_TERMINATED`, suggesting a retry. The local correction recognizes +the observed message as `MICROVM_RESUME_HOOK_FAILED` and supplies administrator +guidance with `retryable=false`. Classifier and production-finalizer regression +tests passed. This corrects future feedback; it does not repair wake transport +or rewrite the historical task record. + +## Limits and evidence + +Successful wakes do not close the five earlier connection refusals. An +independent child still adds work and can affect scheduling. A listener observed +healthy around a successful wake does not establish its state during another +worker's failure. Service-side connection evidence or a failure captured with +independent guest observations is still needed. + +Evidence directory: +`/tmp/abca-645-p2-clean-20260913/p3-pid1-observer-20260916`. +The audit records one accepted timeout case, one excluded assertion and one +unexpected wake failure, with no audit errors. The accepted trace has zero +dropped events and SHA-256 +`b8d63225e8a09b70a938bad8b36932ed6e08a5f29d928f545baab994395c8896`. +All three workers are terminated, reservations released and payload prefixes +empty. The private coordinator, its role/switch/log group and three zero counters +were removed at **19:02:01.271 UTC**, after saving 1,124 coordinator log events. +The private image, build role, artifact object and image log group were removed +at **19:05:57.158 UTC**, after saving 2,038 image log events. Ownership and +read-only absence checks passed. The normal deployment remained unchanged. diff --git a/docs/verification/645-p3-resume-refusal-investigation.md b/docs/verification/645-p3-resume-refusal-investigation.md index 9153534c2..0dea2b66b 100644 --- a/docs/verification/645-p3-resume-refusal-investigation.md +++ b/docs/verification/645-p3-resume-refusal-investigation.md @@ -1,5 +1,15 @@ # ADR-021 P3: intermittent resume-hook connection refusal +**September 16 follow-up:** a +[diagnostic retaining the original PID 1 server](./645-p3-pid1-observer-20260916.md) +captured a separate wake failure with the generic reason +`Resume lifecycle hook failed.` An independent child observed PID 1 owning its +listening socket after restoration, approximately 519 ms before AWS terminated +the worker. No resume hook entry appeared. This adds process/listener evidence +for that new failure; it does not establish the cause of the five exact +connection refusals below. The service question is tracked separately as +[F08](./645-lambda-microvm-service-feedback.md#f08--generic-wake-hook-failure-while-pid-1-owns-its-listener). + Updated 2026-09-16 UTC. This is an investigation record and a prepared report; it has not been submitted to AWS or published as an issue. The [service-team feedback tracker](./645-lambda-microvm-service-feedback.md) From e560fbeb73b8ea66af8be992087456835a48ed1e Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Wed, 16 Sep 2026 15:36:28 -0400 Subject: [PATCH 059/149] docs: record deployed wake failure guidance (#645) --- .../verification/645-capacity-reservations.md | 4 ++ .../645-p3-implementation-plan.md | 14 +++- .../645-p3-pid1-observer-20260916.md | 19 ++++-- .../645-p3-wake-feedback-20260916.md | 67 +++++++++++++++++++ 4 files changed, 98 insertions(+), 6 deletions(-) create mode 100644 docs/verification/645-p3-wake-feedback-20260916.md diff --git a/docs/verification/645-capacity-reservations.md b/docs/verification/645-capacity-reservations.md index 87ddcf175..e708ebf43 100644 --- a/docs/verification/645-capacity-reservations.md +++ b/docs/verification/645-capacity-reservations.md @@ -61,6 +61,10 @@ execution. This establishes artifact alignment and a point-in-time execution inventory; it does not prove admissions were paused or the upgrade/rollback sequence was rehearsed. Evidence is in `/tmp/abca-645-p2-clean-20260913/p3-capacity-writer-inventory-20260916`. +The [PID 1 evidence record](./645-p3-pid1-observer-20260916.md) links its +permanent archive, which also contains this writer inventory. The subsequent +[feedback-only rollout](./645-p3-wake-feedback-20260916.md) advanced the normal +coordinator to version 9; it did not change the capacity protocol. Local tests prove the application requests and DynamoDB Local's transaction behavior. They do not establish deployed IAM, AWS scaling, successful rollout or MicroVM sleep/wake behavior. Terminal events may repeat or be lost independently of the atomic seat update. diff --git a/docs/verification/645-p3-implementation-plan.md b/docs/verification/645-p3-implementation-plan.md index 7145d8170..dddf7da20 100644 --- a/docs/verification/645-p3-implementation-plan.md +++ b/docs/verification/645-p3-implementation-plan.md @@ -43,6 +43,15 @@ late-decision contract is HTTP 404. The final planned attempt was not started. The new failure is tracked separately as F08; neither it nor the five exact connection refusals is resolved. +**Generic wake feedback deployed (2026-09-16):** the +[reviewed code-only update](./645-p3-wake-feedback-20260916.md) advanced the normal +coordinator to version 9, retaining version 8 and changing no image, policy or +sleep setting. All 11 deployed classifier consumers matched the reviewed ZIPs. +Four normal task-API checks passed, with synthetic rows removed afterward. +The observed generic wake failure now receives specific service/admin guidance; +already-persisted stable codes remain unchanged. This corrects feedback, not +the unresolved wake failure. + **Guest hook milestone (2026-09-14):** production [worker suspend/resume hooks](./645-p3-lifecycle-hooks.md) now connect the guest barrier to atomic checkpoint writes and retained-credential refresh followed by @@ -268,8 +277,9 @@ The detailed batches below preserve the implementation history. For the current handoff, use this order: 1. Resolve the five [resume-hook connection refusals](./645-p3-resume-refusal-investigation.md). - Service-side connection diagnostics and guest/listener health are still - missing. Passing retries, the callback-timeout fix and successful cleanup + Service-side connection diagnostics during those exact failures are still + missing. The later F08 observation supplies partial guest/listener evidence + for a distinct generic failure. Passing retries, the callback-timeout fix and successful cleanup do not close this gate. The [minimal listener experiment](./645-p3-listener-probe-20260916.md) also exposed a separate [pending-wake timer bug](./645-p3-pending-wake.md). diff --git a/docs/verification/645-p3-pid1-observer-20260916.md b/docs/verification/645-p3-pid1-observer-20260916.md index d588cfcb7..2a948deb9 100644 --- a/docs/verification/645-p3-pid1-observer-20260916.md +++ b/docs/verification/645-p3-pid1-observer-20260916.md @@ -112,19 +112,22 @@ underlying transport error. The observer samples every 100 ms but emits unchange state only every five seconds. The old classifier persisted this generic wording as -`MICROVM_SUBSTRATE_TERMINATED`, suggesting a retry. The local correction recognizes +`MICROVM_SUBSTRATE_TERMINATED`, suggesting a retry. The correction recognizes the observed message as `MICROVM_RESUME_HOOK_FAILED` and supplies administrator guidance with `retryable=false`. Classifier and production-finalizer regression tests passed. This corrects future feedback; it does not repair wake transport -or rewrite the historical task record. +or rewrite the historical task record. The subsequent +[normal deployment and API checks](./645-p3-wake-feedback-20260916.md) passed +with coordinator version 9 and automatic suspension still disabled. ## Limits and evidence Successful wakes do not close the five earlier connection refusals. An independent child still adds work and can affect scheduling. A listener observed healthy around a successful wake does not establish its state during another -worker's failure. Service-side connection evidence or a failure captured with -independent guest observations is still needed. +worker's failure. F08 supplies one failed wake with independent process/listener +observations, but its underlying connection error is still unknown. Service-side +connection evidence and event-loop/connection observations remain needed. Evidence directory: `/tmp/abca-645-p2-clean-20260913/p3-pid1-observer-20260916`. @@ -138,3 +141,11 @@ were removed at **19:02:01.271 UTC**, after saving 1,124 coordinator log events. The private image, build role, artifact object and image log group were removed at **19:05:57.158 UTC**, after saving 2,038 image log events. Ownership and read-only absence checks passed. The normal deployment remained unchanged. + +The permanent private archive is +`/Users/sphias/.local/share/abca-verification/645-p3-20260916/pid1-observer-capacity-writers-evidence.tar.gz` +(74 files, 114,478,688 bytes, mode `0600`, SHA-256 +`a2e69354cab8fe7250efbbf8998621a023e26ae705b870e80cff0e1d4268a69f`). +Every archived file was checked against its hash manifest. It includes source +commit `b1837ec5`, diagnostic artifacts, logs, cleanup evidence, and the separate +read-only capacity-writer inventory. diff --git a/docs/verification/645-p3-wake-feedback-20260916.md b/docs/verification/645-p3-wake-feedback-20260916.md new file mode 100644 index 000000000..fa6c13912 --- /dev/null +++ b/docs/verification/645-p3-wake-feedback-20260916.md @@ -0,0 +1,67 @@ +# Generic wake-failure feedback deployment — 2026-09-16 + +The [independent PID 1 diagnostic](./645-p3-pid1-observer-20260916.md) captured +AWS's generic `Resume lifecycle hook failed.` message. The previous classifier +did not recognize that wording and suggested retrying a generic compute failure. +Source commit `35d5515b` now classifies it as `MICROVM_RESUME_HOOK_FAILED`, with +service/admin guidance, the relevant log and request-ID locations, and +`retryable: false`. This fixes the explanation; it does not fix the failed wake. + +## Reviewed deployment + +The `backgroundagent-dev` stack in account ``, `us-west-2`, reached +`UPDATE_COMPLETE`. Change set `p3-wake-feedback-20260916` contained exactly 14 +resource changes: 11 Lambda code-object updates, a new coordinator version, +retention of the previous version, and the live alias update. No continuing +resource was replaced; all 475 continuing/new resource entries were accounted +for. No worker image, environment, policy or AgentCore container property changed. +Execution request ID: `170357ab-2b5c-48b0-9a5a-adecd8a15d11`. + +The live coordinator is version **9**, with code SHA-256 +`zoYSUwO5qeLX0pR78DU8OKmFu6qeU9kes31YRj5up4o=`. +Version **8** remains readable with its original code hash. All 11 deployed ZIPs +were downloaded and their hashes matched Lambda's `CodeSha256`. + +Normal MicroVM image **5.0** remains `ACTIVE`/`SUCCESSFUL`, with **8192 MiB** and +the original server process/connection settings. AgentCore remains version +**5**, `READY`. Both automatic-suspension switches remain **false**. + +The deployment used the exact prior S3 template bytes, SHA-256 +`9741f2d4ea6f43e8dde6deff0cb791780a54efdb933e7046c9851b9e6a2e3f9d`. +The reviewed compact target was 706,153 bytes, SHA-256 +`50331a6fe3bcbe5218bfde4cd7c0185f40987c7bc0f3dd95de24de22ffb82d15`. +An explicit `us-west-2` synthesis supplied the new code assets. The default +build's separately generated `us-east-1` assembly was not deployed. + +## Validation and limits + +The full CDK build passed compilation, lint, synthesis and **5,052 tests in 229 +suites**, plus one snapshot. The 56 optional DynamoDB Local tests were skipped +in this build; the previous real-database run passed and this change does not +alter that protocol. The focused classifier/orchestrator run passed 156 tests. + +Four owned synthetic terminal records verified the normal deployed GetTask +handler: + +| Stored error | Observed response | +|---|---| +| Actual generic AWS wake-failure wording | Service error, specific wake explanation, not retryable | +| Legacy reconciliation message without a stable code | Same specific wake guidance | +| New `MICROVM_RESUME_HOOK_FAILED` code | Same specific wake guidance | +| Previously persisted `MICROVM_SUBSTRATE_TERMINATED` code | Existing generic classification preserved | + +Each temporary row was conditionally deleted and its absence verified. No +worker or durable execution was started. Direct handler invocation supplied a +trusted test identity; it does not test API Gateway authentication. Local +orchestrator regression tests cover the newly persisted failure code; these +four live reads do not constitute another end-to-end wake/finalization run. + +The actual F08 task's historical error is unchanged. Its already-persisted +generic code continues to take precedence over diagnostic wording. The five +connection-refusal failures and the distinct F08 generic failure remain open +in the [service feedback tracker](./645-lambda-microvm-service-feedback.md). +Neither retained versions nor this code-only deployment constitute a completed +capacity-protocol rollback exercise. + +Raw deployment, package and API evidence: +`/tmp/abca-645-p2-clean-20260913/p3-wake-feedback-20260916`. From 94a931f8c6a2b5e90c74dbbb7c81fabe9f905c82 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Wed, 16 Sep 2026 15:47:31 -0400 Subject: [PATCH 060/149] docs: verify capacity protocol upgrade and rollback (#645) --- .../verification/645-capacity-reservations.md | 7 ++ .../645-lambda-microvm-service-feedback.md | 8 +- .../645-p3-capacity-upgrade-20260916.md | 92 +++++++++++++++++++ .../645-p3-implementation-plan.md | 13 +++ .../645-p3-wake-feedback-20260916.md | 7 ++ 5 files changed, 124 insertions(+), 3 deletions(-) create mode 100644 docs/verification/645-p3-capacity-upgrade-20260916.md diff --git a/docs/verification/645-capacity-reservations.md b/docs/verification/645-capacity-reservations.md index e708ebf43..ec583e687 100644 --- a/docs/verification/645-capacity-reservations.md +++ b/docs/verification/645-capacity-reservations.md @@ -52,6 +52,13 @@ terminal reservation while preserving a real waiting worker. The bounded volume and role checks do not complete the old-writer drain/upgrade/rollback procedure or establish arbitrary production scale. +The subsequent [isolated upgrade/rollback rehearsal](./645-p3-capacity-upgrade-20260916.md) +passed admission fences, legacy/current writer drains, lost-finalization replay, +drained rollback and re-upgrade using real AWS functions and restricted table +permissions. All temporary resources were removed. It exercised the table +protocol with fixture-managed task statuses; the normal deployment's complete +admission-route and durable-execution drain remains a separate operational gate. + A subsequent read-only deployment audit checked upload confirmation, coordinator version 8, counter reconciliation, queue pickup and stranded-task reconciliation. All five referenced the same code assets as the `a81c565d` assembly, and each diff --git a/docs/verification/645-lambda-microvm-service-feedback.md b/docs/verification/645-lambda-microvm-service-feedback.md index 92bf8a1b3..31181cb2e 100644 --- a/docs/verification/645-lambda-microvm-service-feedback.md +++ b/docs/verification/645-lambda-microvm-service-feedback.md @@ -121,9 +121,11 @@ coordinator 7 and verified through the normal task API. These controlled failure and passing race cases do not explain F01. F08 exposed another service message, `Resume lifecycle hook failed.`, without -an HTTP status or underlying connection error. The local classifier correction -recognizes that observed wording and prevents misleading retry advice. It does -not establish why the service failed to complete the hook. +an HTTP status or underlying connection error. The +[classifier correction deployed in coordinator 9](./645-p3-wake-feedback-20260916.md) +recognizes that observed wording and prevents misleading retry advice for newly +classified failures. Previously persisted stable error codes remain unchanged. +This does not establish why the service failed to complete the hook. **Ask:** provide a service-side lifecycle attempt timeline or equivalent structured fields: originating API receipt, hook kind/attempt ID, start/end diff --git a/docs/verification/645-p3-capacity-upgrade-20260916.md b/docs/verification/645-p3-capacity-upgrade-20260916.md new file mode 100644 index 000000000..755855f65 --- /dev/null +++ b/docs/verification/645-p3-capacity-upgrade-20260916.md @@ -0,0 +1,92 @@ +# Capacity protocol upgrade and rollback rehearsal — 2026-09-16 + +The isolated AWS rehearsal passed: old counter writers can lose another task's +seat when cleanup repeats; the current reservation protocol preserves it. +Pausing admissions, draining tasks, switching protocols, rolling back after +another drain, and upgrading again all worked in the bounded fixture. + +This is a table-protocol rehearsal. It does not claim that the normal +deployment's admission routes were paused or that its durable executions were +drained. + +## Exact scope + +The fixture ran in account ``, `us-west-2`, from **19:39:32 to +19:39:57 UTC**. It used two private DynamoDB tables, two restricted Lambda roles +and five Lambda functions under +`backgroundagent-dev-p3-capacity-upgrade-20260916`: + +- Separate old admission and release entry points. +- Separate current admission and release entry points. +- The exact deployed concurrency-reconciler ZIP, with only environment table + names redirected to the private tables. + +The old `admissionControl` and `decrementConcurrency` function bodies were +extracted unchanged from +`0f4545c77f01c5d905d9e32b8b375d9910e18324`. The current +`task-concurrency.ts` module was bundled unchanged from source `35d5515b`. +The wrapper admitted only nine fixed task IDs with one fixed owner. Its role +allowed `GetItem`/`UpdateItem` only on the two private tables. The reconciler +used the previously reviewed equivalent table permissions, redirected to those +tables. + +The writer ZIP SHA-256 was +`485093d998cc9c11576f08306b5fbb8e3696c0defabe45650ccd19fad4210261`. +The reconciler's Lambda code hash was +`ZbDOQoSBUD3t4e0efIyopTKifQwKkQBBF8IA2wBLMRw=`, matching the normal deployment. +Source-function hashes, bundle inputs, deployed configurations and role policies +are retained with the evidence. + +## Observed checks + +All **12 checks** passed, using 27 completed Lambda invocations and seven actual +invocation rejections while reserved concurrency was zero. + +| Check | Evidence | +|---|---| +| Enforced admission pause | AWS rejected invocation of the paused old/current admission functions | +| Old replay defect | Two active old tasks gave count 2; cleaning one twice produced 0 while the other remained active | +| Ambiguous legacy task | The current reconciler left the unmarked active task and its counter untouched; the fixture's drain check prevented progression | +| Approval occupancy | A current task awaiting approval retained its reservation; early release returned false | +| Capacity limit | A third admission at limit 2 was rejected without a reservation | +| Lost finalization reply | The wrapper deliberately threw after the real release committed; a repeated release preserved the other task's count of 1 | +| Cancellation | A terminal cancelled task released its own slot | +| Completion | A completed task released its own slot | +| Queue boundary | `QUEUED` could not acquire; changing to `SUBMITTED` permitted acquisition | +| Rollback with active work | Admissions remained paused and the fixture's drain check detected an active held reservation | +| Drained rollback | After all current reservations were released, a fresh old-protocol task admitted and finished with count 0 | +| Re-upgrade | After draining the old protocol again, a fresh current-protocol task admitted and released successfully | + +The fixture's runner explicitly changed task statuses to represent work, +approval, failure and cancellation. It did not start a coding worker, run the +production queue-pickup handler, or reproduce a durable execution's entire +finalization path. The deliberately lost reply occurred after the release helper +returned, before its caller received a successful Lambda result. It establishes +safe repeated release after a committed transaction, not an AWS service outage. + +The drain check is part of this operator rehearsal, **not a newly implemented +production deployment interlock**. Separate admission/release functions let the +fixture stop admissions while allowing cleanup. A normal rollout must identify +and stop every actual admission route without disabling cleanup for old work. + +## Cleanup and remaining gate + +At the final strong reads, all nine owned task records were terminal, no held +reservation remained and the single counter was zero. All five functions were +paused before cleanup. Their 120 log events, final table rows and invocation +receipts were saved. + +All five functions, two roles, two tables and five log groups were deleted. +Ownership checks and **14 absence checks** passed at +**19:42:37.730 UTC**. No normal task record, counter or deployment setting changed. + +The [capacity runbook](./645-capacity-reservations.md) still requires a deployment +specific inventory and drain of all old admission/cleanup writers, including +pending uploads, queue pickups and retained durable executions. The +[600-user scan test](./645-p3-capacity-scan-20260916.md) supplies separate bounded +volume evidence. Together these checks narrow the remaining rollout work; they +do not establish arbitrary retention scale or an already-executed normal-fleet +migration. + +Raw evidence: +`/tmp/abca-645-p2-clean-20260913/p3-capacity-upgrade-20260916`. diff --git a/docs/verification/645-p3-implementation-plan.md b/docs/verification/645-p3-implementation-plan.md index dddf7da20..023a50188 100644 --- a/docs/verification/645-p3-implementation-plan.md +++ b/docs/verification/645-p3-implementation-plan.md @@ -52,6 +52,15 @@ The observed generic wake failure now receives specific service/admin guidance; already-persisted stable codes remain unchanged. This corrects feedback, not the unresolved wake failure. +**Capacity upgrade rehearsal (2026-09-16):** the +[isolated AWS protocol check](./645-p3-capacity-upgrade-20260916.md) passed 12 +checks: admission fences, old/current task drains, safe repeated finalization, +drained rollback and re-upgrade. It used unchanged old function bodies, the +current reservation helper and the exact deployed reconciler artifact. +All five functions, two roles, two tables and five log groups were removed. +This verifies the bounded table protocol, not a completed migration of the +normal deployment's admission routes and retained durable executions. + **Guest hook milestone (2026-09-14):** production [worker suspend/resume hooks](./645-p3-lifecycle-hooks.md) now connect the guest barrier to atomic checkpoint writes and retained-credential refresh followed by @@ -349,6 +358,10 @@ handoff, use this order: deployed writer roles, including realistic scan volume. Retain the verified local transaction and isolated-live results as evidence for their narrower scope. + The [isolated protocol rehearsal](./645-p3-capacity-upgrade-20260916.md) now + passes enforced admission pauses, old/current drains, rollback and re-upgrade. + Its private helper entry points and fixture-managed task states do not replace + the normal deployment's complete admission-route/durable-execution drain. The [AWS scan follow-up](./645-p3-capacity-scan-20260916.md) now passes a 600-user fixture with real multi-page reads, exact normal Lambda artifact, equivalent table permissions, zero writes after interrupted scans, and diff --git a/docs/verification/645-p3-wake-feedback-20260916.md b/docs/verification/645-p3-wake-feedback-20260916.md index fa6c13912..ad937c421 100644 --- a/docs/verification/645-p3-wake-feedback-20260916.md +++ b/docs/verification/645-p3-wake-feedback-20260916.md @@ -65,3 +65,10 @@ capacity-protocol rollback exercise. Raw deployment, package and API evidence: `/tmp/abca-645-p2-clean-20260913/p3-wake-feedback-20260916`. + +The permanent private archive is +`/Users/sphias/.local/share/abca-verification/645-p3-20260916/generic-wake-feedback-evidence.tar.gz` +(60 files, 820,897,318 bytes, mode `0600`, SHA-256 +`ab24324c8b6b12c3fb8efce4fd6f75a9d6f0ecbd1dae25d4fbfbe260939e79e9`). +All file hashes were verified against its manifest. It retains the exact 11 +deployed Lambda ZIPs and source/docs commit `0090a713`. From 224c76bd97d2f7b619b779f5b1c9cb076604288c Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Wed, 16 Sep 2026 15:51:37 -0400 Subject: [PATCH 061/149] docs: correct stale ECS sizing and build comments (#645) --- cdk/src/constructs/ecs-agent-cluster.ts | 73 +++++-------------- .../shared/strategies/ecs-strategy.ts | 2 +- 2 files changed, 18 insertions(+), 57 deletions(-) diff --git a/cdk/src/constructs/ecs-agent-cluster.ts b/cdk/src/constructs/ecs-agent-cluster.ts index 28f01661a..d6567b7ae 100644 --- a/cdk/src/constructs/ecs-agent-cluster.ts +++ b/cdk/src/constructs/ecs-agent-cluster.ts @@ -144,8 +144,8 @@ const HTTPS_PORT = 443; * ~3.1 GB of the 16 GB, because ``MISE_JOBS=1`` serialises the packages so peak * is max-single-package rather than sum-of-all. Nearly 5x headroom. * - * Disk is the tighter constraint and is sized less aggressively for that - * reason: the same build peaked at ~14.7 GiB, so Fargate's 21 GiB floor leaves + * Disk is the tighter constraint: the same build peaked at ~14.7 GiB, so + * Fargate's 20 GiB default leaves * only ~1.4x — a heavier dependency cache or a second build sharing the task * would run it out of space and surface as a spurious build failure. 50 GiB * restores real margin, and ephemeral storage is a small fraction of the @@ -163,16 +163,9 @@ const HTTPS_PORT = 443; * execution role, because splitting them is how a grant silently lands on one * def and not the other. Do not read the name as a privilege boundary. */ -// A MODEST default: 4 vCPU / 16 GB, and Fargate's own 20 GiB disk. -// -// Deliberately not the Fargate ceiling. A default is what an adopter who changes -// nothing gets, and at 16 vCPU / 120 GB that is roughly 5x the per-build cost of -// this size in us-east-1 on-demand. Under-provisioning surfaces as a slow or -// OOM-ing build, which is diagnosable and fixable with one prop; over-provisioning -// surfaces as a bill, which is not. A large TypeScript + Python monorepo genuinely -// needs more — raise it through {@link EcsTaskSizing}, up to Fargate's 16 vCPU / -// 120 GB maximum, and raise ephemeral storage with it if concurrent builds run the -// disk out of space. +// Build defaults: 4 vCPU / 16 GiB RAM / 50 GiB disk. +// Planning defaults: 2 vCPU / 8 GiB RAM / Fargate's 20 GiB default disk. +// EcsTaskSizing overrides these values for the target repository's workload. const DEFAULT_BUILD_TASK_CPU = 4096; const DEFAULT_BUILD_TASK_MEMORY_MIB = 16384; const DEFAULT_BUILD_TASK_EPHEMERAL_STORAGE_GIB = 50; @@ -182,7 +175,7 @@ const DEFAULT_PLANNING_TASK_MEMORY_MIB = 8192; /** * Per-task Fargate sizing overrides. Every field is optional; anything left * unset uses the default above. A consumer with a lighter repo should shrink the - * build task (for example 4 vCPU / 16 GB) to cut cost; a heavy monorepo can keep + * build task to cut cost; a heavy monorepo can keep * or raise it up to the Fargate ceiling of 16 vCPU / 120 GB. Values are passed * straight to the Fargate task definition, so they must be a valid Fargate * cpu/memory combination (see the AWS Fargate docs) — an invalid pair fails at @@ -456,51 +449,19 @@ export class EcsAgentCluster extends Construct { this.taskDefinition = makeTaskDef('TaskDef', buildCpu, buildMemory, { // Heavy CI-parity builds legitimately run longer than the 1800s default. BUILD_VERIFY_TIMEOUT_S: '3600', - // Pin the jest test fleet to an ABSOLUTE worker count on ECS. jest's - // `maxWorkers: 25%` is CORE-relative → 4 workers on this 16-vCPU box. - // Measured, the test suite at 4 workers peaks at only ~2.2 GB (whole process - // tree) — not tens of GB. Container OOMs were not driven by this test - // suite's worker count; they were driven by TOTAL concurrency — a - // full-parallel `mise run build` running every package's test/build legs - // plus the resident coding agent all at once. So the real memory driver is - // cross-package build parallelism, not jest's internal workers. 4 is - // comfortably safe on the 120 GB box even alongside the other packages + - // agent. Kept as an explicit env (not core-relative) so a future bigger box - // can't silently over-spawn. The test script reads JEST_MAX_WORKERS (default - // 25%), so this only pins the shared ECS box — CI (2–4 cores) and dev - // machines keep 25%, unaffected. + // Repositories that honor JEST_MAX_WORKERS use four Jest workers on build + // tasks. An absolute value keeps that limit stable when task CPU changes; + // it does not set worker counts for other test runners. JEST_MAX_WORKERS: '4', - // Serialize the mise task graph so the build steps' peak memory doesn't sum - // and OOM the task. `mise run build` fans out its `depends` (the per-package - // build/quality legs) up to MISE_JOBS in parallel (default 4); each package - // then spawns its OWN worker fleet (jest, pytest, esbuild, cdk synth). The - // measured memory driver of the OOMs was this CROSS-PACKAGE storm summing on - // top of the resident coding agent — not any single package. At 120 GB - // (Fargate's max at 16 vCPU) there is no more RAM to add, so the remedy is to - // cut peak parallelism. MISE_JOBS=1 runs the packages SEQUENTIALLY → peak ≈ - // max(single package) instead of sum(all packages), while still building - // every package and keeping BOTH gates (baseline + post-agent). - // Within-package parallelism (JEST_MAX_WORKERS=4, pytest) is untouched, so a - // single package still uses the box's cores. Cost is wall-clock (~serial - // sum, still minutes) — trivial against BUILD_VERIFY_TIMEOUT_S=3600. Without - // this, the post-agent build OOM'd (exit 137) stacking on the still-resident - // agent; a gate that OOMs verified NOTHING — serializing lets it actually - // COMPLETE and gate. Only affects `mise run ` (the build legs); the - // agent's direct `uv run pytest` calls are unaffected. + // Run one mise task at a time to limit overlap between package builds. + // Individual tools can still run their own workers, and the coding agent + // remains resident. This trades build duration for lower peak memory; + // it does not cap direct pytest/Jest calls or guarantee a workload fits. MISE_JOBS: '1', - // Skip the target repo's pre-push TEST hook inside the agent container. - // `mise run install` installs prek git hooks, incl. a pre-push hook that - // re-runs the FULL cdk+cli+agent test suite on every `git push`. In this - // container that suite already ran TWICE (baseline + post-agent build gate) - // and GitHub CI runs it again — so the pre-push run is pure redundancy, AND - // it runs UNcapped (no JEST_MAX_WORKERS), stacking on the resident agent → - // OOM. The agent's only escape was `git push --no-verify`, which silently - // bypassed ALL hooks (incl. the security scan) and trained a - // skip-verification habit. SKIP is the pre-commit/prek standard env var - // (comma-separated hook ids); scoping it to the tests hook lets the push - // succeed WITHOUT --no-verify while KEEPING the pre-push security scan. - // Propagates to both the platform push (post_hooks.py) and the agent's own - // git-tool pushes via shell.py::_clean_env (blacklist — passes SKIP through). + // For repositories using this pre-commit/prek hook ID, skip that named + // pre-push test hook. Other hook IDs remain enabled. This does not establish + // that the repository's tests already ran; verification depends on its + // configured workflow. shell.py::_clean_env passes SKIP to git subprocesses. SKIP: 'monorepo-tests-pre-push', // Caller overrides win: the values above are tuned for one monorepo's // toolchain, so a deployment with a different build shape replaces them diff --git a/cdk/src/handlers/shared/strategies/ecs-strategy.ts b/cdk/src/handlers/shared/strategies/ecs-strategy.ts index e03ea5dd9..acbd91b39 100644 --- a/cdk/src/handlers/shared/strategies/ecs-strategy.ts +++ b/cdk/src/handlers/shared/strategies/ecs-strategy.ts @@ -147,7 +147,7 @@ export class EcsComputeStrategy implements ComputeStrategy { // The ECS container's default CMD starts the FastAPI server (uvicorn) which // waits for HTTP POST to /invocations — but in standalone ECS nobody sends - // that request. We override the container command to invoke run_task() + // that request. We override the container command to invoke run_task_from_payload() // directly with the full orchestrator payload (including hydrated_context). // This avoids the server entirely and runs the agent in batch mode. if (!ECS_PAYLOAD_BUCKET) { From c9f450ae584982ffbd4527a153fd4ed5d006ecea Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Wed, 16 Sep 2026 15:55:53 -0400 Subject: [PATCH 062/149] docs: distinguish MicroVM credential renewal path (#645) --- agent/src/runner.py | 11 +++++++---- cdk/src/constructs/agent-session-role.ts | 18 ++++++++++-------- 2 files changed, 17 insertions(+), 12 deletions(-) diff --git a/agent/src/runner.py b/agent/src/runner.py index 45743be30..1d765d630 100644 --- a/agent/src/runner.py +++ b/agent/src/runner.py @@ -73,10 +73,13 @@ def _setup_bedrock_cost_attribution(config: TaskConfig) -> None: 1. **Per-user/repo chargeback (CUR 2.0 / Cost Explorer).** Write the SessionRole ARN + ``{user_id, repo, task_id}`` STS tags to a 0600 file - that ``bedrock_creds_helper.py`` reads. Claude Code's managed-settings - ``awsCredentialExport`` runs that helper and signs Bedrock requests with - the tagged assumed-role credentials. Skipped when ``AGENT_SESSION_ROLE_ARN`` - is unset (local/dev) — the helper then fails open to ambient creds. + that ``bedrock_creds_helper.py`` reads on AgentCore/ECS. Claude Code's + managed-settings ``awsCredentialExport`` runs that helper and signs + Bedrock requests with the tagged assumed-role credentials. MicroVM + instead uses the parent's scoped container-credential provider; its + export helper returns no credentials and cannot supply a wake barrier. + File writing is skipped when ``AGENT_SESSION_ROLE_ARN`` is unset + (local/dev); the non-MicroVM helper can fall back to ambient credentials. 2. **Per-call forensics (model-invocation logs).** Set ``X-Amzn-Bedrock-Request-Metadata`` via ``ANTHROPIC_CUSTOM_HEADERS`` on the diff --git a/cdk/src/constructs/agent-session-role.ts b/cdk/src/constructs/agent-session-role.ts index 2d4d56cc6..83992a046 100644 --- a/cdk/src/constructs/agent-session-role.ts +++ b/cdk/src/constructs/agent-session-role.ts @@ -130,9 +130,10 @@ export interface AgentSessionRoleProps { * subprocess is then attributed per `{user_id, repo}` in CUR 2.0 / Cost * Explorer via the session tags this role already carries. * - * The compute role keeps its own Bedrock grant: attribution is a billing - * control that fails open (the credential helper falls back to compute-role - * creds if the assume-role fails), so model invocation never depends on this. + * The compute role keeps its own Bedrock grant. On AgentCore/ECS, the export + * helper can fall back to compute-role credentials if attribution fails. + * MicroVM workers instead use the retained task-scoped credential provider; + * failed renewal blocks wake rather than falling back to ambient credentials. * Omit (e.g. isolated construct tests) to skip the Bedrock grant. */ readonly invokableModels?: bedrock.IBedrockInvokable[]; @@ -162,9 +163,10 @@ export interface AgentSessionRoleProps { * CloudWatch Logs remains on the compute role (shared access). The * compute role *also* keeps `InvokeModel`; this role adds a parallel, session- * tagged Bedrock grant (#215) used by the Claude Code subprocess for cost - * attribution. Long-task safety on the 1-hour-capped chained session is handled - * by Claude Code's `awsCredentialExport` refresh, and the helper falls back to - * the compute role if assume fails — so model invocation never breaks. + * attribution. AgentCore/ECS use `awsCredentialExport` with expiry and an ambient + * fallback for attribution failures. MicroVM uses the parent's retained scoped + * provider with synchronous renewal; the export helper returns no credentials. + * The background export refresh is not a MicroVM wake barrier. */ export class AgentSessionRole extends Construct { /** Actions sufficient for the agent's DynamoDB access. Excludes Scan. */ @@ -278,8 +280,8 @@ export class AgentSessionRole extends Construct { // Reuse grantInvoke so this role's Bedrock permissions exactly mirror the // compute role's (cross-region profiles fan out to the foundation model in // every routed region — replicating that by hand would risk an AccessDenied - // on a cross-region route). Claude Code assumes this role (via its - // awsCredentialExport helper) so InvokeModel rides the session's + // on a cross-region route). Claude Code uses this role through the export + // helper or MicroVM's scoped provider, so InvokeModel rides the session's // {user_id, repo, task_id} tags, surfacing per-user/repo Bedrock spend in // CUR 2.0 / Cost Explorer. No PrincipalTag condition: the tags are for // billing attribution, not access scoping, so a condition would add no From 8b44f76db3899c6542313d3b9404fcf7c6c1a9ae Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Wed, 16 Sep 2026 15:58:11 -0400 Subject: [PATCH 063/149] docs: link capacity evidence and comment review (#645) --- docs/verification/645-p3-capacity-upgrade-20260916.md | 8 ++++++++ docs/verification/645-p3-implementation-plan.md | 7 +++++++ 2 files changed, 15 insertions(+) diff --git a/docs/verification/645-p3-capacity-upgrade-20260916.md b/docs/verification/645-p3-capacity-upgrade-20260916.md index 755855f65..80a71d8f2 100644 --- a/docs/verification/645-p3-capacity-upgrade-20260916.md +++ b/docs/verification/645-p3-capacity-upgrade-20260916.md @@ -90,3 +90,11 @@ migration. Raw evidence: `/tmp/abca-645-p2-clean-20260913/p3-capacity-upgrade-20260916`. + +Permanent private archive: +`/Users/sphias/.local/share/abca-verification/645-p3-20260916/capacity-upgrade-evidence.tar.gz` +(56 files, 9,265,702 bytes, mode `0600`, SHA-256 +`e1499cdde2ad8573996365b09553f4a62fc7ec790a932c0054370b5ded03784e`). +Every file was checked against its hash manifest. It includes source +`2b12ce88`, both exact function ZIPs, the legacy function bodies, all receipts, +logs and cleanup evidence, plus the subsequent ECS sizing-comment check. diff --git a/docs/verification/645-p3-implementation-plan.md b/docs/verification/645-p3-implementation-plan.md index 023a50188..c72fdd41c 100644 --- a/docs/verification/645-p3-implementation-plan.md +++ b/docs/verification/645-p3-implementation-plan.md @@ -61,6 +61,13 @@ All five functions, two roles, two tables and five log groups were removed. This verifies the bounded table protocol, not a completed migration of the normal deployment's admission routes and retained durable executions. +**Comment cleanup follow-up (2026-09-16):** ECS comments now describe the actual +4-vCPU/16-GiB/50-GiB build defaults and the scope of its existing build settings. +Shared role and runner documentation now distinguishes AgentCore/ECS credential +export from MicroVM's retained scoped provider. TypeScript emitted code and +Python executable syntax trees were unchanged. These corrections did not change +sizing, build settings or credential behavior; ECS live acceptance remains open. + **Guest hook milestone (2026-09-14):** production [worker suspend/resume hooks](./645-p3-lifecycle-hooks.md) now connect the guest barrier to atomic checkpoint writes and retained-credential refresh followed by From 268d0336bc88aca0c94fdd5b5c726c96f765878b Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Wed, 16 Sep 2026 18:01:40 -0400 Subject: [PATCH 064/149] fix(microvm): close lifecycle connections before freezing --- agent/src/microvm_http.py | 3 + agent/tests/test_microvm_http.py | 24 ++ cdk/src/handlers/shared/error-classifier.ts | 5 +- .../handlers/shared/error-classifier.test.ts | 1 + .../645-lambda-microvm-service-feedback.md | 67 +++-- .../645-p3-implementation-plan.md | 33 ++- .../645-p3-lifecycle-diagnostics.md | 10 +- .../645-p3-resume-refusal-investigation.md | 46 ++-- .../645-p3-transport-control-20260916.md | 16 +- .../645-p3-wake-transport-20260916.md | 256 ++++++++++++++++++ 10 files changed, 409 insertions(+), 52 deletions(-) create mode 100644 docs/verification/645-p3-wake-transport-20260916.md diff --git a/agent/src/microvm_http.py b/agent/src/microvm_http.py index 3ac2c426f..ba7f526c3 100644 --- a/agent/src/microvm_http.py +++ b/agent/src/microvm_http.py @@ -61,6 +61,9 @@ async def _transition(request: Request, action: Literal["suspend", "resume"]) -> except asyncio.CancelledError as exc: diagnostics.finish(None, "MICROVM_LIFECYCLE_CANCELLED", exc) raise + # A frozen VM retains this connection and its idle timer. Close it with + # the response so the next hook uses a fresh connection after restoration. + response.headers["Connection"] = "close" diagnostics.finish( response.status_code, json.loads(bytes(response.body)).get("code", "acknowledged") ) diff --git a/agent/tests/test_microvm_http.py b/agent/tests/test_microvm_http.py index 753cc63bb..e0c77b9e2 100644 --- a/agent/tests/test_microvm_http.py +++ b/agent/tests/test_microvm_http.py @@ -79,6 +79,30 @@ async def park(context): @pytest.mark.anyio class TestLifecycleHttp: + @pytest.mark.parametrize("action", ["suspend", "resume"]) + @pytest.mark.parametrize( + ("outcome", "expected_status"), + [("acknowledged", 200), ("unavailable", 409), ("invalid", 400), ("failed", 503)], + ) + async def test_lifecycle_responses_close_even_when_client_requests_keep_alive( + self, context, callbacks, client, action, outcome, expected_status + ): + await park(context) + if action == "resume": + assert (await client.post(PREFIX + "/suspend", json={})).status_code == 200 + if outcome == "unavailable": + lifecycle.unregister_task(context) + elif outcome == "failed": + callback = callbacks[0] if action == "suspend" else callbacks[1] + callback.side_effect = RuntimeError("callback failed") + response = await client.post( + PREFIX + "/" + action, + content=b"{" if outcome == "invalid" else b"{}", + headers={"Connection": "keep-alive"}, + ) + assert response.status_code == expected_status + assert response.headers["connection"] == "close" + async def test_hooks_have_distinct_correlated_timelines( self, context, callbacks, client, capsys ): diff --git a/cdk/src/handlers/shared/error-classifier.ts b/cdk/src/handlers/shared/error-classifier.ts index 64dd0d40a..dfe06bd5a 100644 --- a/cdk/src/handlers/shared/error-classifier.ts +++ b/cdk/src/handlers/shared/error-classifier.ts @@ -105,7 +105,8 @@ const MICROVM_TERMINAL_CLASSIFICATIONS: Readonly. ' - + 'A connection refusal can happen before the guest hook logs anything. Check saved task progress and cleanup before starting a replacement task; retrying alone is not a verified fix.', + + 'A transport error can occur before any guest hook log. The service’s "connection was refused" wording alone does not establish that the listener was closed. ' + + 'Check saved task progress and cleanup before starting a replacement task; retrying alone is not a verified fix.', retryable: false, errorClass: ErrorClass.SERVICE, }, @@ -137,7 +138,7 @@ const MICROVM_TERMINAL_CLASSIFICATIONS: Readonly { test.each([ 'Resume lifecycle hook failed. Please check your hook endpoint and application logs for more details.', 'Resume lifecycle hook connection was refused. Please check your hook endpoint and application logs for more details.', + 'Resume lifecycle hook timed out. Please check your hook endpoint and application logs for more details.', 'Resume lifecycle hook returned HTTP status 503.', 'Resume lifecycle hook returned HTTP status 409.', ])('gives actionable wake diagnostics for %s', reason => { diff --git a/docs/verification/645-lambda-microvm-service-feedback.md b/docs/verification/645-lambda-microvm-service-feedback.md index 31181cb2e..9abb50b4b 100644 --- a/docs/verification/645-lambda-microvm-service-feedback.md +++ b/docs/verification/645-lambda-microvm-service-feedback.md @@ -10,11 +10,11 @@ is the receipt that lets the service team find a particular call. | ID | Priority | Topic | Evidence/status | |---|---|---|---| -| F01 | P3 blocker | Accepted wake ends in connection refusal | Five recorded failures, including instrumented image 5.0; responsible component unknown | +| F01 | P3 blocker | Accepted wake ends in connection refusal | Six recorded failures; latest captures idle-timer closure and a live listener; three close-header cases verify closure before freeze | | F02 | High | Supported IAM conditions and misleading permission errors | Reproduced in earlier P2 work; current service behavior needs confirmation | | F03 | Medium | A service-side hook timeline and structured failure details | Diagnostic improvement request based on F01 | | F04 | Medium | `PENDING` also means restoring an existing worker | Observed live; our timer bug is fixed | -| F05 | Medium | HTTP connection handling across suspend/resume | Contract question; no established cause of F01 | +| F05 | Medium | HTTP connection handling across suspend/resume | Exact refusal now correlates with expired idle connection; service dispatch details still needed | | F06 | Medium | Conditional operator-role requirement for VPC connectors | Earlier deployment failure; application setup fixed | | F07 | P2/P3 acceptance gap | Run token retention and recovery without a worker ID | Guest-log recovery verified; maximum retention, post-expiry behavior and recovery without identity logs unknown | | F08 | P3 blocker | Generic wake-hook failure with an observed listener | PID 1 owned its listener after restore, 519 ms before termination; no resume hook entry | @@ -25,8 +25,10 @@ is the receipt that lets the service team find a particular call. The coordinator releases capacity correctly; the requested coding workflow still fails. This prevents enabling automatic suspension for normal tasks. -**Observed:** five failures on images `3.0`, `4.0` and `5.0` in `us-west-2`, September -15–16. AWS accepted `ResumeMicrovm`, then reported: +**Observed:** five failures on normal images `3.0`, `4.0` and `5.0`, plus a sixth +on a private image retaining 5.0's application and connection behavior with +transport probes, in `us-west-2`, September 15–16. AWS accepted `ResumeMicrovm`, +then reported: > Resume lifecycle hook connection was refused. Please check your hook endpoint > and application logs for more details. @@ -37,6 +39,20 @@ and coordinator Resume calls as a necessary cause. **Best starting evidence for the service team:** +- Account ``, region `us-west-2`, September 16, 21:14:27–21:15:30 UTC. +- Worker `microvm-bb1bfd4b-9ce3-3691-a60b-88f263081d43`, private image + `backgroundagent-dev-p3-wake-transport-20260916:1.0`, original server PID 1. +- Resume receipt `74943bd5-cfd1-4781-9f10-9946566089e8`, accepted + 21:15:26.786 UTC. At 21:15:26.948, the restored event loop expired the old + suspend connection's five-second idle timer. At 21:15:26.950, an independent + observer saw PID 1 owning the original port-8080 listener. AWS terminated the + worker at 21:15:27.471 with the exact refusal wording. +- No fresh HTTP connection or resume request appeared in the retained window. + The [transport investigation](./645-p3-wake-transport-20260916.md) + contains the full timeline and successful fresh-connection comparison. + +**Earlier unmodified image 5.0 evidence:** + - Account ``, region `us-west-2`, September 16, 17:09:39–17:10:42 UTC. - Worker `microvm-6ff103ab-a41d-348b-8a56-721c9050b623`, image `backgroundagent-dev-abca-agent:5.0`, original server PID 1. @@ -56,15 +72,17 @@ a process exit, or a networking/restore failure? At what point did the service consider the guest network and listener ready, and what underlying error did it map to this reason? -**Limits:** no root cause is established. The Mac was the test controller; the -failing connection was between AWS and the guest inside AWS. Later successful -controls, including three with the original server as PID 1, do not prove this -defect fixed or establish a failure rate. +**Limits:** the latest guest trace supports an old-connection race but cannot +show the service's actual dispatch error or socket choice. The Mac was the test +controller; the failing connection was inside AWS. Passing controls do not +establish a failure rate or discharge the earlier failures. -**Next:** retain a fresh failure with independent process/listener evidence, -or obtain the service-side evidence above. Service response: pending. -F08 now supplies an independently observed failed wake with different service -wording. Its relationship to these five connection refusals remains unknown. +**Next:** complete normal-image rollout and acceptance of the explicit close +header, retain the exact failure, and obtain service-side dispatch details. +Two unchanged sleeps longer than 90 seconds and three candidate cases passed; +the candidate trace verifies actual connection closure before freeze. +Service response: pending. F08 records an independently observed failed wake +with different wording; a shared cause remains unconfirmed. ## F02 — IAM conditions and errors make correct setup difficult @@ -161,8 +179,10 @@ not reopening the corrected timer bug. **Evidence:** some successful full-agent suspend/resume access logs used the same peer port. An isolated Linux paused-process experiment reproduced a reset of an old HTTP connection after a six-second pause, while every fresh connection -still succeeded. It did **not** reproduce an AWS connection refusal. Some F01 -failures followed suspension by less than five seconds. +still succeeded. It did **not** reproduce an AWS connection refusal. +The interval from an observer's `SUSPENDED` sample is not the server's idle +connection age. The [timing correction](./645-p3-wake-transport-20260916.md#historical-timing-correction) +removes the earlier inference that short observed sleeps ruled out idle expiry. See the [transport control](./645-p3-transport-control-20260916.md) and [process-observer record](./645-p3-process-observer-20260916.md). @@ -177,17 +197,26 @@ cases used fresh ports. No unexpected refusal or reset appeared. Two original mistitled long-hold attempts are retained and excluded from that acceptance; fresh corrected cases supplied the stated durations. +The later [instrumented transport investigation](./645-p3-wake-transport-20260916.md) +captured a 59.451-second event-loop gap, closure of the old suspend connection +by `timeout_keep_alive_handler`, a live PID 1 listener, and the exact F01 refusal +without a new HTTP connection. Its passing control closed the old socket and +accepted a fresh resume connection about two milliseconds later. The candidate +uses an explicit response header, not the earlier timer flag. All three candidate +cases verified response-driven closure before freeze, no armed idle timer on the +suspend connection, and a fresh resume connection. + **Ask:** does the service reuse hook TCP connections across suspend/resume, honor `Connection: close`, and retry a failed reused connection on a fresh socket? How are reset, refused and timeout errors classified? What ordering is guaranteed between guest unfreeze, network restoration and hook delivery? Which clock semantics should guest timeout/keep-alive timers expect across suspension? -**Limits:** matching peer ports suggest reuse but do not establish all transport -behavior. A paused Linux process is not an AWS MicroVM restore. The bounded -AWS comparison establishes working paths with both settings, not an explanation -or fix for F01. Production connection handling remains unchanged. Service response: -pending; retain the service contract questions above. +**Limits:** the guest now records exact connection identity and timer closure, +but it cannot expose the service client's pool or failed dispatch. A paused +Linux process is not an AWS MicroVM restore, and the earlier timer-flag comparison +did not reproduce F01. Normal deployment connection handling remains unchanged. +Service response: pending; retain the service contract questions above. ## F06 — Make the VPC connector role requirement obvious before deployment diff --git a/docs/verification/645-p3-implementation-plan.md b/docs/verification/645-p3-implementation-plan.md index c72fdd41c..b5b2c49c2 100644 --- a/docs/verification/645-p3-implementation-plan.md +++ b/docs/verification/645-p3-implementation-plan.md @@ -6,6 +6,18 @@ Prepared 2026-09-13 from `main` `5e10038c7e28179b302ac4de78b709795aeba3ce`. Read Prerequisite work is tracked here on `fix/645-microvm-readiness`. “Completed” means implemented and checked locally; AWS deployment and live verification have separate completion gates below. +**Wake transport correction (2026-09-16):** the +[instrumented comparison](./645-p3-wake-transport-20260916.md) captured a sixth +exact refusal: the restored event loop expired the old suspend connection while +PID 1 still owned its listener, with no fresh resume connection. Two unchanged +cases sleeping longer than 90 seconds passed. Three private candidate cases +with explicit `Connection: close` passed, each proving socket closure before +freeze, no idle timer on that socket, fresh resume connection, original timeout, +late approval rejection and complete cleanup. This supplies a concrete +application correction; service dispatch traces remain unavailable and normal +deployment plus final-image acceptance remain separate steps. Wake-hook timeouts +also now receive nonretryable service/admin feedback in source. + **Adjustable sleep deployed (2026-09-16):** source `a81c565d` adds `microvm_sleep_after_s` and CLI `--microvm-sleep-after `, with a 600-second default and zero to stay awake. The full build passed 7,939 tests; @@ -198,7 +210,7 @@ is superseded by these records. - [x] Deploy the supervisor and six-hook image with suspension disabled; verify six isolated guest cases, repair the discovered cancellation stop omission, and prove API termination before test cleanup. - [ ] Deploy and verify the complete P3 sleep/wake lifecycle in AWS. - [x] Deploy the explicit SDK callback-timeout fix and repeat long sleep, late wake, approval, denial and cancellation; require final approval/tool evidence as well as cleanup. Image `4.0` passed these checks, including real renewal after credential expiry. -- [ ] Resolve the intermittent resume-hook connection refusal; successful retries do not discharge the five failures across images 3.0, 4.0 and 5.0. See the [investigation and request IDs](./645-p3-resume-refusal-investigation.md). +- [ ] Complete normal-image rollout and acceptance of the explicit connection-close correction; retain all six exact refusals and the generic failure. The [transport comparison](./645-p3-wake-transport-20260916.md) verifies actual closure before freeze on the private candidate. First prerequisite batch completed locally on 2026-09-13: @@ -292,11 +304,14 @@ absence of all temporary infrastructure. The detailed batches below preserve the implementation history. For the current handoff, use this order: -1. Resolve the five [resume-hook connection refusals](./645-p3-resume-refusal-investigation.md). - Service-side connection diagnostics during those exact failures are still - missing. The later F08 observation supplies partial guest/listener evidence - for a distinct generic failure. Passing retries, the callback-timeout fix and successful cleanup - do not close this gate. +1. Complete normal-image rollout and acceptance of the + [connection-close correction](./645-p3-wake-transport-20260916.md), retaining + the six [exact refusals](./645-p3-resume-refusal-investigation.md) and F08. + Service-side dispatch diagnostics remain missing, but the sixth refusal now + captures exact idle-timer closure and a live listener. The private candidate + proves the suspend socket closes before freeze and resume uses a new one. + Passing retries, the callback-timeout fix and cleanup alone do not close + this gate. The [minimal listener experiment](./645-p3-listener-probe-20260916.md) also exposed a separate [pending-wake timer bug](./645-p3-pending-wake.md). Its correction and startup-confirmation follow-up are deployed in coordinator @@ -310,8 +325,10 @@ handoff, use this order: the new hook-stage, AWS request-ID and durable state-change logging. Three isolated AWS workflows verified it, including actual API wake and coordinator recovery. The [normal-stack rollout](./645-p3-diagnostics-rollout-20260916.md) - now runs coordinator version 6 and image 5.0 with both suspension switches off. The original - server remained PID 1. Logging and successful controls do not close the defect. + first deployed coordinator version 6 and image 5.0 with both suspension + switches off; subsequent feedback updates advanced the coordinator to 9. + The original server remained PID 1. Logging and successful controls alone + do not close the defect. The same record covers a bounded AWS comparison with connection reuse disabled: both quick/long wakes and both missing-approval HTTP 409 controls passed, with specific guest-stage diagnostics and deployed task-API feedback. diff --git a/docs/verification/645-p3-lifecycle-diagnostics.md b/docs/verification/645-p3-lifecycle-diagnostics.md index 07f27117d..4b23d4a5b 100644 --- a/docs/verification/645-p3-lifecycle-diagnostics.md +++ b/docs/verification/645-p3-lifecycle-diagnostics.md @@ -108,12 +108,20 @@ continues independently for service diagnosis. refusal before hook entry, independent listener/process or service-side evidence is still required. -Known Resume connection-refused and HTTP 4xx/5xx reasons now produce the stable +Known Resume generic failures, connection-refused, timeout and HTTP 4xx/5xx +reasons produce the stable `MICROVM_RESUME_HOOK_FAILED` task error. It asks an admin to inspect this evidence and saved progress before starting a replacement. It does not promise that retrying repairs the fault. Persisted classification codes take priority over diagnostic words; unrecognized AWS wording keeps the generic terminal code. +The service's connection-refused wording alone does not establish that the +listener was closed. The [instrumented transport failure](./645-p3-wake-transport-20260916.md#exact-refusal-with-connection-and-listener-evidence) +recorded that wording while the original listener was still observed, after an +expired idle timer closed the old HTTP connection. Preserve the raw reason and +receipts; distinguish the guest observations from the service's unavailable +dispatch details. + ## Verification and deployment Local regressions cover hook correlation, sensitive-text exclusion, safe AWS diff --git a/docs/verification/645-p3-resume-refusal-investigation.md b/docs/verification/645-p3-resume-refusal-investigation.md index 0dea2b66b..927cc64c8 100644 --- a/docs/verification/645-p3-resume-refusal-investigation.md +++ b/docs/verification/645-p3-resume-refusal-investigation.md @@ -1,6 +1,15 @@ # ADR-021 P3: intermittent resume-hook connection refusal -**September 16 follow-up:** a +**Latest September 16 follow-up:** the +[HTTP connection investigation](./645-p3-wake-transport-20260916.md#exact-refusal-with-connection-and-listener-evidence) +captured a sixth exact refusal. The old connection's five-second idle timer +expired at restoration; PID 1 still owned its listener, and no fresh HTTP +connection or resume request appeared. Three explicit `Connection: close` +candidate cases passed on a private image, proving closure before freeze and +fresh resume connections. Normal-image rollout and acceptance remain separate +steps; automatic suspension remains disabled. + +An earlier [diagnostic retaining the original PID 1 server](./645-p3-pid1-observer-20260916.md) captured a separate wake failure with the generic reason `Resume lifecycle hook failed.` An independent child observed PID 1 owning its @@ -24,9 +33,10 @@ acknowledges `ResumeMicrovm`. Its final `stateReason` is: > and application logs for more details. The guest previously returned HTTP 200 from `/suspend`. Its retained application -stream has no subsequent `/resume` access log. This does not prove the guest -process stayed healthy: application logs alone cannot distinguish a listener, -process, network-restoration or hook-transport failure. +stream has no subsequent `/resume` access log. Application logs alone cannot +distinguish a listener, process, network-restoration or hook-transport failure. +The latest case adds an independent listener sample and exact idle-timer closure; +the first five cases below did not capture those observations. The coordinator detects termination, marks the task failed and releases its capacity reservation. Successful failure cleanup does not make the requested @@ -37,12 +47,14 @@ the new logging and wake-failure feedback. Three isolated AWS workflows verified the instrumentation with the original server running as PID 1, including actual API wake and coordinator recovery. None reproduced refusal. The [normal-stack rollout](./645-p3-diagnostics-rollout-20260916.md) now runs coordinator -version 7 and image 5.0 after the subsequent command-race follow-up. A fifth -failure on image 5.0 now includes the new diagnostics, as recorded below. +version 9 and image 5.0 after subsequent coordinator fixes. The fifth failure +on image 5.0 includes those application diagnostics, as recorded below. ## Recorded failures -All times are UTC, in `us-west-2`, on image `backgroundagent-dev-abca-agent`. +The first five failures below are on image `backgroundagent-dev-abca-agent`. +All times are UTC, in `us-west-2`. The sixth, on an instrumented private image +derived from 5.0, has its [own detailed timeline](./645-p3-wake-transport-20260916.md#exact-refusal-with-connection-and-listener-evidence). The earlier cases are detailed in the [durable verification record](./645-p3-durable-live-20260915.md). @@ -95,7 +107,7 @@ do not explain or discharge this fifth failure. ### Image 4.0 request timeline -The last case used source `58ed7bbf`'s image, including the explicit approval +The image 4.0 case used source `58ed7bbf`'s image, including the explicit approval callback timeout. Task ID: `01M2KRWNPP4BQCJ80SVG6473Z8`. | Time on Sep 16 | Evidence | @@ -165,9 +177,10 @@ adds the independent parent and listener sampling described below. Its completed fallback and direct API wakes retained a healthy child-owned listener, without reproducing the refusal. Short full-agent cases reused the same client port for `/suspend` and `/resume`, unlike the minimal listener's closed connections. -That is an observed transport difference, not an established cause. None of -the five recorded failures has independent process/listener evidence at restore, -and the guest does not expose the cgroup OOM counters sampled by the observer. +That experiment established a transport difference. None of the first five +recorded failures has independent process/listener evidence at restore; the +sixth now does. The guest does not expose the cgroup OOM counters sampled by +the observer. The subsequent [local transport control](./645-p3-transport-control-20260916.md) reproduced a reset on an old connection after a six-second process pause, while @@ -188,12 +201,11 @@ do not identify F01's cause or justify a production connection-setting change. error and whether the request reached the guest, including any transport retry. 2. Correlate guest process/kernel health and the port-8080 listener at restoration. Application access logs do not provide that missing evidence. - The diagnostic parent now supplies these observations in successful controls; - a refusal must be captured with those observations to make the comparison. - Preserve the original image and use fixed, bounded owned cases if further - testing is justified by a specific transport or process hypothesis. -3. If a transport or application race is identified, make a bounded correction - and test that trigger specifically. Do not hide a failed wake by silently + The latest failure now supplies these observations with the original PID 1. + Compare its expired connection against the successful fresh-connection wake. +3. Verify the explicit close response on a private candidate image, including + its actual socket closure before freeze and fresh connection after restore. + Do not hide a failed wake by silently launching another worker: the approved action and workspace may already have changed. 4. Repeat approval, denial, original/late deadline, multiple-gate persistence and diff --git a/docs/verification/645-p3-transport-control-20260916.md b/docs/verification/645-p3-transport-control-20260916.md index 6b86307af..7906b3e4e 100644 --- a/docs/verification/645-p3-transport-control-20260916.md +++ b/docs/verification/645-p3-transport-control-20260916.md @@ -1,13 +1,15 @@ # ADR-021 P3: paused-server HTTP transport control Date: 2026-09-16. Local diagnostic completed; the original AWS resume refusal -was **not reproduced**. No application code or AWS deployment changed. +was **not reproduced** by this local experiment. It changed no application code +or AWS deployment. Subsequent AWS transport instrumentation is recorded in the +[wake transport investigation](./645-p3-wake-transport-20260916.md). ## Question The full-agent hooks sometimes reuse one HTTP connection across suspension. Could an old connection fail after a pause even though the server still accepts -new connections? This is a narrower question than the four recorded +new connections? This is a narrower question than the recorded [AWS resume refusals](./645-p3-resume-refusal-investigation.md). The earlier full-agent observer measured a 363.501-second wall-clock gap and @@ -53,9 +55,13 @@ container exited successfully and its absence was verified. One reused connection reset after exceeding the server's five-second keep-alive timeout. That is different from a new connection being refused: the listener remained reachable in every case. This does not establish how AWS classifies -its underlying transport errors. Some original failures followed suspension by -less than five seconds, so simple keep-alive expiration does not explain all -four recorded failures. +its underlying transport errors. An earlier interpretation measured from the +observer's `SUSPENDED` sample and used an interval shorter than five seconds to +discount idle expiration. That uses the wrong starting point: the server arms +its idle timer after finishing the preceding HTTP response, before that later +state sample. The [historical timing correction](./645-p3-wake-transport-20260916.md#historical-timing-correction) +retains the original timestamps and explains why they neither establish nor +exclude this cause. The next bounded AWS comparison should: diff --git a/docs/verification/645-p3-wake-transport-20260916.md b/docs/verification/645-p3-wake-transport-20260916.md new file mode 100644 index 000000000..590cddee4 --- /dev/null +++ b/docs/verification/645-p3-wake-transport-20260916.md @@ -0,0 +1,256 @@ +# ADR-021 P3: wake HTTP connection and event-loop investigation + +Comparison completed on September 16, 2026, in `us-west-2`. +A fresh failure now correlates the exact refusal wording with an expired +connection timer and a live listener. Three `Connection: close` cases passed +with actual closure before freeze and fresh resume connections. The normal +deployment was unchanged during this comparison. This record separates +actual AWS wake evidence from local calibration and attempts that failed before +any sleep. Nothing here has been submitted to the service team. + +## Question and instrumentation + +An HTTP keep-alive connection lets a client reuse an existing connection for its +next request. Uvicorn normally closes an unused connection after five seconds. +If its timer advances while a MicroVM is frozen, closing that connection can +race with the next request when the worker wakes. + +The private image was derived from the exact deployed image 5.0 artifact, +SHA-256 `9b3150e9e5991cc9fcc8a4adbb2cfbd8f97399f0c16baa9c2b5fe5b101e7b035`. +It retains the original server command, server as PID 1, default keep-alive +behavior, 8,192 MiB memory and all six hooks. Its additions are: + +- HTTP protocol metadata: connection identity, request arrival, response + completion, idle-timer deadline, and the function that closes a socket. +- A timer on the server's event loop, showing when that loop resumes execution. +- A separate process sampling the server and its listening socket. This observer + opens no network connections. + +The probes do not record request bodies, headers or credentials. The artifact +SHA-256 is `d894f553c1f43983e30591f38a727ce6b9ae7db58dd3f8246feb11415e39f63b`; +109 other ZIP entries are byte-identical to the baseline. The private image is +`backgroundagent-dev-p3-wake-transport-20260916`, version `1.0`. + +## Local calibration + +A disposable ARM64 Linux container used Uvicorn 0.50.0, Python 3.13.13, +FastAPI 0.139.0 and the asyncio/h11 server. It had no external network. +The controller waited for logged response completion and an armed idle timer +before pausing the server process. + +| HTTP response behavior | Process pause | Next request | Separate fresh connection | +|---|---:|---|---| +| Default keep-alive | 0.3 s | 200 | 200 | +| Default keep-alive | 6 s | Old connection failed with `BrokenPipeError` | 200 | +| Explicit `Connection: close` | 0.3 s | 200 | 200 | +| Explicit `Connection: close` | 6 s | 200 | 200 | + +The failing old connection was closed by `timeout_keep_alive_handler`; the server +stayed alive. `Connection: close` tells the client to use a new connection. It +is different from setting the server's idle timer to zero. + +Two preliminary local attempts are retained and excluded. The first started +the heartbeat during module import, before Uvicorn created its event loop. +The second paused immediately after the client read the response body, before +the server finished its connection bookkeeping. It consequently armed the timer +after continuation. Reading a response body is not proof that the server has +already armed its idle timer. + +Pausing a local process does not reproduce AWS snapshot or network restoration. + +## Attempts excluded before wake + +| Case | Task | Reason | Cleanup | +|---|---|---|---| +| 1 | `01M2NZ3AB465YF8ZN6ED3ERAB4` | Diagnostic IAM policy retained old fixture IDs; initial task read denied | Cancelled; no worker or reservation created | +| 2 | `01M2NZ3AB92EGG6GCSG2XT9DVZ` | AWS durable execution stopped with `Stopped.ByService` | Owned worker cancelled and terminated; reservation released | +| 3 | `01M2NZ3AB9PVXP9QHAB79DMSTB` | Second `Stopped.ByService`, during startup | Owned worker cancelled and terminated; reservation released | + +The copied IAM allowlists were corrected and checked against the fresh task +IDs. The two durable service errors contain correlation IDs +`b56d7524-7d7c-49c3-afab-13f343b3ab85` at 20:56:03.507Z and +`75b494cd-f102-4f1d-9a14-02c86533c808` at 20:58:46.239Z. Neither attempted a +Suspend or Resume. They are separate from the MicroVM wake failure. + +Subsequent transport tests use the production coordinator body and supervisor +with a local, single-pass step/poll adapter. Workers and lifecycle calls remain +in AWS. This tests real wake behavior; it does not test durable replay or resolve +the separate durable service stops. Each worker has a 1,800-second maximum +lifetime, with a five-minute local coordinator bound and six-minute watcher. + +## First actual AWS wake + +Task `01M2NZ3ABA61SYKQSRJ9S8ZAZ4` used worker +`microvm-a0df3925-9773-325a-9e64-7ce295f548db`, a 150-second approval window +and a 30-second sleep delay. + +| UTC time | Observation | +|---|---| +| 21:03:48.732 | Suspend response finishes on connection 3; five-second deadline armed | +| 21:03:50.039 | Watcher observes `SUSPENDED` | +| 21:04:46.524 | Sole Resume request starts | +| 21:04:46.779 | AWS accepts, receipt `da6f3495-c60b-44c6-a6ab-542533e43f20` | +| 21:04:46.979 | Server event loop resumes after a 58.306-second gap | +| 21:04:46.979 | Expired idle timer closes connection 3 | +| 21:04:46.981 | Server accepts fresh connection 4 | +| 21:04:46.983 | Resume request arrives on connection 4 | +| 21:04:47.428 | Resume response completes with HTTP 200 | +| 21:05:46.693 | Original approval timeout observed; subsequent approve API returns 404 | +| 21:05:57.960 | Task and cleanup checks pass without watcher repair | + +The independent observer saw the original server still owning its listener. +The original approval became `TIMED_OUT`; the Read did not execute. The task +completed, its worker terminated, its reservation was released, its counter was +zero and its payload absent. The complete trace contained no dropped records. + +This directly shows an idle timer expiring after an AWS freeze. It also shows +a successful fresh connection afterward. Timer expiry alone therefore does not +establish the cause of the intermittent failure. + +## Exact refusal with connection and listener evidence + +A second unchanged-image round completed one more successful wake, then +reproduced the exact refusal. Its third planned case was not started. +The failing task was `01M2P0QTA9M6D2J6B5RM42KDNH`, worker +`microvm-bb1bfd4b-9ce3-3691-a60b-88f263081d43`, on private image `1.0`. +No command or guest failure was injected. + +| UTC time | Observation | +|---|---| +| 21:14:27.502 | Suspend accepted, receipt `26d37f89-141a-4938-ba07-904c3edaf6b9` | +| 21:14:27.565 | Suspend HTTP 200 finishes on connection 3, peer port 55068; idle deadline 157.88734039 on the monotonic clock | +| 21:15:26.540 | Sole Resume request starts | +| 21:15:26.786 | Resume accepted, receipt `74943bd5-cfd1-4781-9f10-9946566089e8` | +| 21:15:26.947 | Event loop resumes after a 59.451-second gap, at monotonic time 212.277492979 | +| 21:15:26.948 | `timeout_keep_alive_handler` closes connection 3 | +| 21:15:26.949 | Connection 3 reports `connection_lost` | +| 21:15:26.950 | Independent observer sees PID 1 running and owning the original port-8080 listener, inode 13544 | +| 21:15:27.471 | AWS terminates the worker with `Resume lifecycle hook connection was refused` | + +The retained window has no fresh HTTP connection, resume request bytes, resume +hook entry or resume access log. The observer is a sample, not continuous proof +of health; its process/socket reads are not atomic. The 271 raw guest events and +immediate service response are retained. + +The coordinator recorded `FAILED` and released the reservation before the +runner's idempotent fallback cleanup. This is a failed wake, not a passing +approval test. Together with the successful fresh-connection control, it +supports an old-connection race. The service's actual dispatch error and +connection choice remain unavailable from customer-visible telemetry. + +## Explicit connection-close candidate + +The candidate adds `Connection: close` to suspend and resume JSON responses, +including error responses. Status codes, response bodies, checkpointing and +approval barriers remain the same. This makes the HTTP response end the +connection before a freeze instead of leaving closure to an idle timer. + +The private image's version `2.0` differs from diagnostic `1.0` only in +`agent/src/microvm_http.py`; 112 other archive entries are byte-identical. +Candidate artifact SHA-256: +`6eeddb08a4314d939e87a4bcc3a517319e107d1bcc3a5eda71d814346f60fbeb`. +The original server remains PID 1 with 8,192 MiB. + +Agent quality checks passed: lint, formatting, type checking, and 1,955 tests +with 11 opt-in DynamoDB Local tests skipped. Eight new route cases verify that +both hooks return the close header for HTTP 200, 400, 409 and 503, even when the +client requests keep-alive. + +The live comparison first ran two unchanged-image cases with a 210-second +approval window and a 30-second sleep delay, checking an actual suspended hold +longer than 90 seconds. These cases used a seven-minute local coordinator bound +and eight-minute watcher. It then ran three candidate-image cases with the +original 150-second approval window and approximately 60-second freeze. +Each phase was configured to stop at its first failure. + +| Case | Image | Observed suspended hold | Result | +|---|---|---:|---| +| `over-90s-1` | Unchanged `1.0` | 117.829 s | Passed | +| `over-90s-2` | Unchanged `1.0` | 117.935 s | Passed | +| `close-1` | Candidate `2.0` | About 58 s | Passed | +| `close-2` | Candidate `2.0` | About 59 s | Passed | +| `close-3` | Candidate `2.0` | About 58 s | Passed | + +The candidate audit additionally checked the actual HTTP transport: + +| Case | Suspend connection closed | First observed `SUSPENDED` | Fresh resume connection | +|---|---|---|---| +| `close-1` | 21:42:17.331 | 21:42:18.562 | 21:43:16.699 | +| `close-2` | 21:46:26.263 | 21:46:27.040 | 21:47:26.058 | +| `close-3` | 21:50:09.743 | 21:50:10.231 | 21:51:08.939 | + +In every candidate, connection 3 closed from Uvicorn's response-send path before +the freeze; no idle deadline was armed and no idle-expiry callback ran on that +connection. Resume arrived on new connection 4. The internal `keep_alive` +boolean remained true even though h11 closed the transport, so that boolean +alone would be an incorrect verification criterion. + +All five comparison tasks preserved their original approval deadline, applied +the timeout, rejected a subsequent approval with HTTP 404 `REQUEST_NOT_FOUND`, +and completed without executing the unapproved Read. Complete traces contained +one Read intent each and no dropped records. Every worker terminated, every +reservation was released, counters reached zero, and payload prefixes were +empty without watcher repair. All four phase audits had zero errors. + +A longer-sleep pass or a finite series of successful resumes alone cannot +establish the service-side cause or eliminate an intermittent failure. The +candidate's direct transport evidence establishes removal of the retained +suspend connection in these tests. It does not reveal the service's failed +dispatch error or prove that every historical wake failure had this cause. + +## Error feedback correction + +Review also found that the timeout wording `Resume lifecycle hook timed out` +fell into the generic retryable substrate category. It now maps to +`MICROVM_RESUME_HOOK_FAILED`, preserving the raw reason and requiring the same +service/admin diagnosis as the other recognized wake failures. The regression +checks current, legacy and raw error forms and the resulting retry guidance. +All 136 focused classifier tests passed, followed by the complete CDK suite: +5,053 passed and 56 opt-in DynamoDB Local tests skipped. TypeScript compilation +and ESLint passed. The remedy explicitly avoids treating the service's refused +wording as proof of a closed listener. + +## Historical timing correction + +The first four refusal records had short intervals measured from an observed +`SUSPENDED` state. Their archived HTTP access timestamps are earlier: + +| Earlier case | Suspend HTTP 200 access log | AWS termination | Access-to-termination interval | Observed `SUSPENDED`-to-termination interval | +|---|---|---|---:|---:| +| Approval, Sep 15 | 21:31:41.279 | 21:31:46.958 | 5.679 s | 2.834 s | +| Start-crash recovery, Sep 15 | 21:56:48.531 | 21:56:56.279 | 7.748 s | 3.006 s | +| Inline Resume failure recovery, Sep 15 | 22:54:39.936 | 22:54:45.753 | 5.817 s | 4.759 s | +| Missing approval, Sep 16 | 00:25:01.657 | 00:25:07.524 | 5.867 s | 2.923 s | + +An access log precedes final response completion and timer arming. AWS's +termination timestamp is not the exact time its failed hook request was sent. +The old records do not contain those precise events. These intervals cannot +prove idle expiry caused a refusal, but a short interval measured from +`SUSPENDED` does not rule it out. + +## Evidence and remaining work + +Raw scripts, exact artifacts, command receipts, paginated guest logs, original +failure timing extracts and audit results are retained privately under: + +- `/tmp/abca-645-p2-clean-20260913/p3-wake-transport-20260916` +- `/tmp/abca-645-p2-clean-20260913/p3-wake-transport-round2-20260916` +- `/tmp/abca-645-p2-clean-20260913/p3-wake-pool-window-20260916` +- `/tmp/abca-645-p2-clean-20260913/p3-wake-connection-close-20260916` + +Cleanup finished at 21:58:39.471Z. All ten created workers were terminated. +The private image and its versions, function and versions, two roles, parameter, +two log groups, two exact artifact objects and ten zero-valued fixture counters +were removed and their absence checked. Ordinary task records remain under +their existing retention policy. The archive retains 6,636 guest log records +and 73 coordinator log records. + +An initial cleanup preflight rejected `/tmp` versus `/private/tmp` spellings of +the same directory before any resource deletion. It was corrected by comparing +resolved paths; that failed preflight and the successful cleanup are both retained. + +The evidence is being preserved in the private permanent archive +`~/.local/share/abca-verification/645-p3-20260916/wake-transport-evidence.tar.gz`. +At comparison cleanup, the normal deployment remained coordinator 9 / image +5.0, 8,192 MiB, with automatic suspension disabled. Normal rollout is a separate +verification step. From ab2a253c82f2edb3ed90505ca14dcafeb469791a Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Wed, 16 Sep 2026 18:54:45 -0400 Subject: [PATCH 065/149] docs(microvm): record connection-close rollout and acceptance --- .../645-lambda-microvm-service-feedback.md | 33 ++- ...45-p3-connection-close-rollout-20260916.md | 196 ++++++++++++++++++ .../645-p3-implementation-plan.md | 75 +++---- .../645-p3-lifecycle-diagnostics.md | 8 +- .../645-p3-resume-refusal-investigation.md | 41 ++-- .../645-p3-wake-transport-20260916.md | 15 +- 6 files changed, 293 insertions(+), 75 deletions(-) create mode 100644 docs/verification/645-p3-connection-close-rollout-20260916.md diff --git a/docs/verification/645-lambda-microvm-service-feedback.md b/docs/verification/645-lambda-microvm-service-feedback.md index 9abb50b4b..0672d66b8 100644 --- a/docs/verification/645-lambda-microvm-service-feedback.md +++ b/docs/verification/645-lambda-microvm-service-feedback.md @@ -10,11 +10,11 @@ is the receipt that lets the service team find a particular call. | ID | Priority | Topic | Evidence/status | |---|---|---|---| -| F01 | P3 blocker | Accepted wake ends in connection refusal | Six recorded failures; latest captures idle-timer closure and a live listener; three close-header cases verify closure before freeze | +| F01 | High; service diagnosis open | Accepted wake ends in connection refusal | Six recorded failures; pre-freeze connection closure verified; image 6.0 correction deployed and four Durable workflows passed | | F02 | High | Supported IAM conditions and misleading permission errors | Reproduced in earlier P2 work; current service behavior needs confirmation | | F03 | Medium | A service-side hook timeline and structured failure details | Diagnostic improvement request based on F01 | | F04 | Medium | `PENDING` also means restoring an existing worker | Observed live; our timer bug is fixed | -| F05 | Medium | HTTP connection handling across suspend/resume | Exact refusal now correlates with expired idle connection; service dispatch details still needed | +| F05 | Medium | HTTP connection handling across suspend/resume | Exact refusal correlates with expired idle connection; explicit close header deployed; service dispatch details still needed | | F06 | Medium | Conditional operator-role requirement for VPC connectors | Earlier deployment failure; application setup fixed | | F07 | P2/P3 acceptance gap | Run token retention and recovery without a worker ID | Guest-log recovery verified; maximum retention, post-expiry behavior and recovery without identity logs unknown | | F08 | P3 blocker | Generic wake-hook failure with an observed listener | PID 1 owned its listener after restore, 519 ms before termination; no resume hook entry | @@ -23,7 +23,8 @@ is the receipt that lets the service team find a particular call. **Impact:** the user approves an action, but the worker stops before continuing. The coordinator releases capacity correctly; the requested coding workflow -still fails. This prevents enabling automatic suspension for normal tasks. +failed in these recorded cases. The application correction below is deployed; +automatic suspension remains disabled pending the remaining P3 acceptance gates. **Observed:** five failures on normal images `3.0`, `4.0` and `5.0`, plus a sixth on a private image retaining 5.0's application and connection behavior with @@ -77,10 +78,16 @@ show the service's actual dispatch error or socket choice. The Mac was the test controller; the failing connection was inside AWS. Passing controls do not establish a failure rate or discharge the earlier failures. -**Next:** complete normal-image rollout and acceptance of the explicit close -header, retain the exact failure, and obtain service-side dispatch details. +**Application correction:** the [normal rollout](./645-p3-connection-close-rollout-20260916.md) +deployed explicit close headers in image 6.0, with coordinator 10 and both sleep +switches off. Approval, denial, timeout and cancellation while asleep passed on +that image through a private real Durable coordinator and normal decision APIs. Two unchanged sleeps longer than 90 seconds and three candidate cases passed; the candidate trace verifies actual connection closure before freeze. + +**Next:** obtain service-side dispatch details for the retained failures and +confirm supported connection handling across suspension. The application +evidence does not establish the exact service error in every historical case. Service response: pending. F08 records an independently observed failed wake with different wording; a shared cause remains unconfirmed. @@ -145,6 +152,12 @@ recognizes that observed wording and prevents misleading retry advice for newly classified failures. Previously persisted stable error codes remain unchanged. This does not establish why the service failed to complete the hook. +The subsequent image 6.0 / coordinator 10 rollout also classifies +`Resume lifecycle hook timed out` as a nonretryable service failure. Six normal +task-handler checks cover raw/legacy/stable timeout forms, refusal, generic +failure and preservation of an older stable code. They use synthetic terminal +records and verify feedback, not a new service failure. + **Ask:** provide a service-side lifecycle attempt timeline or equivalent structured fields: originating API receipt, hook kind/attempt ID, start/end times, connection versus HTTP failure, HTTP status, underlying error code and @@ -206,6 +219,11 @@ uses an explicit response header, not the earlier timer flag. All three candidat cases verified response-driven closure before freeze, no armed idle timer on the suspend connection, and a fresh resume connection. +That correction is now deployed in normal image 6.0. Four real Durable +approval/denial/timeout/cancellation workflows passed on the normal image; +the [rollout record](./645-p3-connection-close-rollout-20260916.md) includes +actual wake receipts, guest acknowledgments and final task/tool evidence. + **Ask:** does the service reuse hook TCP connections across suspend/resume, honor `Connection: close`, and retry a failed reused connection on a fresh socket? How are reset, refused and timeout errors classified? What ordering is guaranteed @@ -215,7 +233,8 @@ semantics should guest timeout/keep-alive timers expect across suspension? **Limits:** the guest now records exact connection identity and timer closure, but it cannot expose the service client's pool or failed dispatch. A paused Linux process is not an AWS MicroVM restore, and the earlier timer-flag comparison -did not reproduce F01. Normal deployment connection handling remains unchanged. +did not reproduce F01. The normal image now sends explicit close headers; these +passing cases do not reveal the service's earlier dispatch error. Service response: pending; retain the service contract questions above. ## F06 — Make the VPC connector role requirement obvious before deployment @@ -306,7 +325,7 @@ added a separate observer child. Application code, connection keepalive, receipt, was the hook attempted on a fresh or reused socket, what bytes/status were received, and was another connection attempted before termination? Does the service retain the hook-attempt timeline separately from the API -receipt? Compare it with the five exact connection refusals in F01. +receipt? Compare it with the six exact connection refusals in F01. **Limits:** the observed process and listener existed 519 ms before termination. This does not prove continuous health or event-loop responsiveness, and it does diff --git a/docs/verification/645-p3-connection-close-rollout-20260916.md b/docs/verification/645-p3-connection-close-rollout-20260916.md new file mode 100644 index 000000000..b00931ea8 --- /dev/null +++ b/docs/verification/645-p3-connection-close-rollout-20260916.md @@ -0,0 +1,196 @@ +# ADR-021 P3: connection-close rollout and normal-image acceptance + +Verified September 16, 2026, in `us-west-2`, account ``. +The normal `backgroundagent-dev` stack now runs MicroVM image **6.0** and +coordinator **10**. Approval, denial, timeout and cancellation while asleep +passed on that image using a private AWS Durable coordinator and normal decision +handlers. Both normal automatic-suspension switches remain **off**. + +## What changed and why + +An HTTP connection is the channel AWS uses to deliver a lifecycle message. +Keeping it open normally saves the work of opening another one. A frozen worker, +however, can wake with both an old connection and an expired timer that wants +to close it. + +The [instrumented investigation](./645-p3-wake-transport-20260916.md) reproduced +the exact refusal while the original server still owned its listening socket. +Its restored event loop closed the old suspend connection; no fresh connection +or resume request appeared. Two unchanged cases sleeping longer than 90 seconds +passed. Three candidate cases directly verified that an explicit +`Connection: close` response closed the suspend connection before freezing and +that resume arrived on a new connection. + +Source `ccab2eecbf8aa28d4b9782e003f790f14078c7a5` adds that header to both +suspend and resume JSON responses, including errors. It leaves the original +server command, five-second idle timeout, checkpointing and approval barriers +intact. It also recognizes `Resume lifecycle hook timed out` as +`MICROVM_RESUME_HOOK_FAILED`, with service/admin guidance rather than automatic +retry advice. + +These guest observations support a specific connection-lifetime correction. +They do not reveal the service client's failed dispatch, establish a failure +rate, or prove that every historical wake failure had the same cause. The +service-side questions remain in [F01, F03, F05 and F08](./645-lambda-microvm-service-feedback.md). +No report was sent to the service team. + +## Reviewed normal deployment + +CloudFormation reached `UPDATE_COMPLETE` at **22:22:56.959 UTC**. + +| Item | Verified result | +|---|---| +| Source | `ccab2eecbf8aa28d4b9782e003f790f14078c7a5` | +| Root resource count | 475 | +| Coordinator live alias | Version 10; previous version 9 retained | +| Coordinator code SHA-256, base64 | `hrLwpI9HJ15OctduM1tv4VelPBXIRQ2qeGOqdq+zdqw=` | +| Normal image | `backgroundagent-dev-abca-agent:6.0`, `ACTIVE` / `SUCCESSFUL` | +| Previous image | Version 5.0 retained | +| Worker memory | 8,192 MiB | +| AgentCore | Version 5 retained; this rollout did not change its container | +| Bootstrap | 1.8.0 | +| Normal sleep gates | Environment `false`; SSM `false`, parameter version 1 | + +The normal image artifact is 486,771 bytes, SHA-256 +`191ae368a2a4d28bb8910caabef6a2b99b5521a474cff60ef6542456fb679490`. +Compared with the exact image 5.0 ZIP, 107 entries are identical and four differ: +the lifecycle HTTP module, a runner docstring with an unchanged executable +syntax tree, and the JSON/Markdown copies of the already-established adjustable +sleep contract. The Dockerfile is identical and the image contains no transport +probes. All image properties other than the source artifact match version 5.0. + +The reviewed change set, `p3-connection-close-20260916`, contained 16 resource +changes: 11 classifier-consumer function code updates, the coordinator version +and alias transition with retention of the old version, the image artifact, and +the build role's exact artifact digest. No modified resource required +replacement. The executed template is 706,153 bytes, SHA-256 +`856bb6aaba07bf2eb1c8d5c0ad874217c6fa7c1f19b84a03b92b0dec14624f51`. +It was prepared from the exact previous S3 template bytes, avoiding the lossy +`GetTemplate` baseline discovered in an earlier rollout. + +After deployment, all 11 function code hashes matched the actual reviewed ZIPs +downloaded from S3. The image version, unchanged image configuration, retained +coordinator version and disabled sleep settings were checked separately. + +## Four real Durable workflows + +The temporary function `backgroundagent-dev-p3-close-acceptance-20260916:1` +ran the production Durable handler with five-second polling. Its wrapper bound +launches and permissions to four fresh fixture identities, the normal image +6.0, and a 1,800-second maximum worker lifetime. Only its private sleep switch +was enabled. Each case had a six-minute watcher, a $1 task budget and a six-turn +limit. No transport failure was injected. + +The normal approval, denial, cancellation and read handlers were invoked through +Lambda with the fixture identities. This tests the deployed handlers, not an +additional API Gateway authentication/ingress path. These were repository-free +tasks requesting one guarded Read of `/etc/os-release`; no notifications or +repository publication were requested. + +| Case | Task | Worker | Result | +|---|---|---|---| +| Approve after sleep | `01M2P4WX07MY6AP3VAJ1X5BQXX` | `microvm-2e038148-adce-37eb-8e83-629ef9c16eda` | Approved; exactly one successful Read; task `COMPLETED` | +| Deny after sleep | `01M2P4WX086HH030697Z4XK8W8` | `microvm-4149fb1d-42ba-3fdc-b509-a548bba70627` | Denied; Read blocked; task `COMPLETED` | +| Timeout wins | `01M2P4WX08SKVPCA55E4AX0W7T` | `microvm-92ff298a-dc28-3e59-9994-1375f79a43a7` | Original approval `TIMED_OUT`; late approve returned HTTP 404; Read blocked | +| Cancel while asleep | `01M2P4WX08WQZH86E72YQGSDNA` | `microvm-b8414dd8-6e85-35c1-b32c-a496a4c553c2` | Task `CANCELLED`; worker terminated without a Resume request | + +All four AWS Durable executions ended `SUCCEEDED`. Each worker terminated, +each reservation was released, each counter reached zero, and each launch +payload prefix was empty before fixture cleanup. No watcher repair was needed. +The three non-cancelled tasks retained complete traces with one Read intent +each and zero dropped records. The denied Read returned `AUTHORITATIVE DENY`; +it did not execute. The cancelled task was not required to produce a completed +agent trace after termination. + +### Actual wake delivery + +The normal sleep switch prevents new automatic suspension. It intentionally +does not prevent a decision handler from waking an already-suspended worker. +Approve and deny therefore used the normal API's immediate wake path; timeout +used the private coordinator's production supervisor. + +| Case | Resume accepted, UTC | AWS Resume request ID | Original PID 1 resume hook | +|---|---|---|---| +| Approve | 22:35:14.410 | `04923c77-f85c-4274-85ee-ebbdc92b2935` | HTTP 200 at 22:35:14.832 | +| Deny | 22:37:32.764 | `5110864c-a2b1-4141-99ef-5d1d7e374e14` | HTTP 200 at 22:37:33.259 | +| Timeout | 22:40:07.397 | `154a247f-2c0e-4e0a-9221-31f8da62c76e` | HTTP 200 at 22:40:07.760 | + +Approve was delivered after more than 60 seconds of observed suspension. The +timeout case woke before its original 150-second approval deadline, then let +that deadline expire; wake did not grant a new window. Its later approval +received HTTP 404 `REQUEST_NOT_FOUND`. + +The cancel case was observed `SUSPENDED` at 22:42:45.314. The normal cancellation +handler returned HTTP 200 at 22:42:56.987, Lambda invocation receipt +`34e57a5e-eba9-459f-aee5-cb7a0c5d7503`. The worker terminated, the approval +remained pending, and no Resume command or successful Read appeared. + +An accepted API call alone was never the pass condition. The audit required +the guest's hook results where applicable, original decision/deadline outcomes, +tool evidence, final Durable/task state and resource cleanup. + +## Error feedback and local checks + +Six synthetic terminal records exercised the normal deployed GetTask handler: +raw, legacy and stable-code timeout forms; raw refusal; raw generic wake +failure; and preservation of an older stable classification code. All passed, +and all six synthetic rows were deleted. Recognized new wake failures receive +nonretryable service/admin guidance explaining that a refused-connection message +alone does not establish a closed listener. Existing stable classifications +keep their precedence. + +The source passed agent lint, formatting and type checks, **1,955 agent tests**, +TypeScript compilation, ESLint, and **5,053 CDK tests** across 229 suites. +The 11 agent and 56 CDK opt-in DynamoDB Local cases were skipped in these runs; +this change did not alter their transaction conditions. Eight new route cases +cover both lifecycle endpoints with HTTP 200/400/409/503 despite an incoming +keep-alive request. The classifier suite covers timeout classification and +resulting retry guidance. + +These four acceptance executions used real AWS Durable execution. Earlier +transport comparisons used a local step adapter after two separate AWS +`Stopped.ByService` errors during startup. Passing this round does not identify +the cause of those earlier service stops. + +## Cleanup and evidence + +The four-case audit passed with zero errors. Scoped cleanup completed at +**22:49:26.544 UTC**. It removed the private coordinator and versions, its role, +private SSM parameter, private function log group and four zero-valued fixture +counters. Ownership tags, terminal executions and task/counter versions were +checked before deletion; absence was verified afterward. All 331 private +function log events were archived. Normal image versions, shared logs and +ordinary task records retain their existing retention policy. + +A fresh normal-state check at **22:49:27.184 UTC** confirmed coordinator 10, +image 6.0 `ACTIVE` / `SUCCESSFUL`, 8,192 MiB, stack `UPDATE_COMPLETE`, and the +normal SSM switch still `false` at version 1. + +Private working evidence is under: + +- `/tmp/abca-645-p2-clean-20260913/p3-connection-close-rollout-20260916` +- `/tmp/abca-645-p2-clean-20260913/p3-close-acceptance-20260916` + +The permanent archive is +`~/.local/share/abca-verification/645-p3-20260916/connection-close-rollout-evidence.tar.gz`. +It contains 122 files, occupies 827,593,551 bytes, and has SHA-256 +`92aaa3c29a6ff5dbc1cf7acf81d7fe354b7aed4fdbcf774359e793ad466ce237`. +Every member's size and hash were verified against the manifest; permissions +are `0600`. It includes the exact rollout templates/artifacts, deployed function +ZIPs, source comparison, fixture code, API receipts, guest logs, Durable history, +task/tool evidence and cleanup proofs. The decisive earlier refusal and +instrumented comparison have their separate archive in the transport report. + +## Remaining completion gates + +The application correction and these four normal-image workflows are complete. +The broader [P3 plan](./645-p3-implementation-plan.md#remaining-work-in-execution-order) +still requires the wider final-image workspace/credential/decision matrix, +other-backend permission and network checks, and the normal deployment's +coordinated capacity upgrade/drain and rollback validation. Service token +retention and recovery without guest identity logs also remain open. + +Automatic suspension stays disabled while those gates are open. The user sleep +setting retains its 600-second default and zero-to-stay-awake option. Service +trace requests remain separate from the verified application change; finite +passing runs cannot certify every future wake. diff --git a/docs/verification/645-p3-implementation-plan.md b/docs/verification/645-p3-implementation-plan.md index b5b2c49c2..b247813c4 100644 --- a/docs/verification/645-p3-implementation-plan.md +++ b/docs/verification/645-p3-implementation-plan.md @@ -14,9 +14,14 @@ cases sleeping longer than 90 seconds passed. Three private candidate cases with explicit `Connection: close` passed, each proving socket closure before freeze, no idle timer on that socket, fresh resume connection, original timeout, late approval rejection and complete cleanup. This supplies a concrete -application correction; service dispatch traces remain unavailable and normal -deployment plus final-image acceptance remain separate steps. Wake-hook timeouts -also now receive nonretryable service/admin feedback in source. +application correction. The [normal rollout and acceptance](./645-p3-connection-close-rollout-20260916.md) +now verify image 6.0 and coordinator 10, 8,192 MiB, both sleep switches off, +and four real Durable workflows: approve, deny, timeout winning over a late +decision, and cancellation while asleep. All four finalized without watcher +repair; temporary infrastructure was removed. Six deployed task-handler checks +also verify wake feedback, including nonretryable service/admin guidance for +hook timeouts. Service dispatch traces and the wider final-image matrix remain +separate gates. **Adjustable sleep deployed (2026-09-16):** source `a81c565d` adds `microvm_sleep_after_s` and CLI `--microvm-sleep-after `, with a @@ -210,7 +215,8 @@ is superseded by these records. - [x] Deploy the supervisor and six-hook image with suspension disabled; verify six isolated guest cases, repair the discovered cancellation stop omission, and prove API termination before test cleanup. - [ ] Deploy and verify the complete P3 sleep/wake lifecycle in AWS. - [x] Deploy the explicit SDK callback-timeout fix and repeat long sleep, late wake, approval, denial and cancellation; require final approval/tool evidence as well as cleanup. Image `4.0` passed these checks, including real renewal after credential expiry. -- [ ] Complete normal-image rollout and acceptance of the explicit connection-close correction; retain all six exact refusals and the generic failure. The [transport comparison](./645-p3-wake-transport-20260916.md) verifies actual closure before freeze on the private candidate. +- [x] Deploy the explicit connection-close correction and verify approve, deny, timeout and cancel-asleep on normal image 6.0 through real Durable execution; retain all six exact refusals and the generic failure. The [transport comparison](./645-p3-wake-transport-20260916.md) proves closure before freeze; the [rollout record](./645-p3-connection-close-rollout-20260916.md) records normal-image acceptance and cleanup. +- [ ] Complete the wider final-image workspace, multiple-gate, late-decision-winner and expired-credential checks before enabling automatic suspension. First prerequisite batch completed locally on 2026-09-13: @@ -304,53 +310,30 @@ absence of all temporary infrastructure. The detailed batches below preserve the implementation history. For the current handoff, use this order: -1. Complete normal-image rollout and acceptance of the - [connection-close correction](./645-p3-wake-transport-20260916.md), retaining - the six [exact refusals](./645-p3-resume-refusal-investigation.md) and F08. - Service-side dispatch diagnostics remain missing, but the sixth refusal now - captures exact idle-timer closure and a live listener. The private candidate - proves the suspend socket closes before freeze and resume uses a new one. - Passing retries, the callback-timeout fix and cleanup alone do not close - this gate. - The [minimal listener experiment](./645-p3-listener-probe-20260916.md) also - exposed a separate [pending-wake timer bug](./645-p3-pending-wake.md). - Its correction and startup-confirmation follow-up are deployed in coordinator - version 5. Local checks, real first-start observations and the exact API-issued - old-worker `PENDING` branch pass in the - [full-agent observer experiment](./645-p3-process-observer-20260916.md). - Its seven workers and temporary infrastructure were removed, and the exact - evidence was privately archived. These timer results do not explain the - separate connection refusals. +1. Complete the wider final-image lifecycle matrix on image 6.0: a cloned + repository with mutable files across multiple approval gates, the late-decision + winner, and wake after actual credential expiry. Earlier image 4.0/5.0 results + remain evidence for their recorded scope. + The [connection-close correction](./645-p3-wake-transport-20260916.md) and + [normal rollout](./645-p3-connection-close-rollout-20260916.md) are complete. + Three instrumented candidates proved socket closure before freeze and a new + resume connection; four normal-image Durable workflows passed approval, + denial, the original timeout winning, and cancellation while asleep. + Coordinator 10 and image 6.0 are deployed with both sleep gates off. + Retain the six [exact refusals](./645-p3-resume-refusal-investigation.md) and + F08, and obtain service-side dispatch details separately. The guest evidence + does not reveal the actual service error in every historical case. The [lifecycle diagnostics guide](./645-p3-lifecycle-diagnostics.md) describes - the new hook-stage, AWS request-ID and durable state-change logging. - Three isolated AWS workflows verified it, including actual API wake and - coordinator recovery. The [normal-stack rollout](./645-p3-diagnostics-rollout-20260916.md) - first deployed coordinator version 6 and image 5.0 with both suspension - switches off; subsequent feedback updates advanced the coordinator to 9. - The original server remained PID 1. Logging and successful controls alone - do not close the defect. - The same record covers a bounded AWS comparison with connection reuse disabled: - both quick/long wakes and both missing-approval HTTP 409 controls passed, - with specific guest-stage diagnostics and deployed task-API feedback. - Two mistitled long-hold attempts were excluded and replaced by fresh measured - cases. No unexpected refusal appeared; production connection handling is unchanged. - A subsequent image 5.0 timeout-race case reproduced the refusal after a normal - pre-deadline wake. The guest logged a successful checkpoint and suspend HTTP - 200, but no resume hook entry. The prepared service report now includes its - API receipts and precise timeline; the new diagnostics have not established - the cause. - The [independent PID 1 follow-up](./645-p3-pid1-observer-20260916.md) now - captures process/listener evidence during a separate generic wake-hook - failure. F08 records its service question. An observed listening socket does - not prove the event loop processed the hook or identify the failing connection. + the deployed hook-stage, AWS request-ID and durable state-change logging, + including the distinction between an accepted Resume and a successful hook. 2. The [command-race checks](./645-p3-command-races-20260916.md) now pass cancellation before/after Suspend, during observed `SUSPENDING`, and during restore `PENDING`; approval during an accepted suspension and three consecutive polling failures also pass. Six required cases used nine workers, with three harness-invalidated attempts explicitly excluded and replaced. All temporary resources were removed. The same follow-up deployed clearer status-read - failure guidance in coordinator version 7 and verified the normal task API; - image 5.0 and disabled suspension settings remain in place. + failure guidance in coordinator version 7 and verified the normal task API. + Those cases used image 5.0; disabled suspension settings remain in place. Complete the remaining live fault matrix: service token-retention expiry and recovery when guest identity evidence is unavailable. Keep each injected failure distinct from an @@ -363,7 +346,9 @@ handoff, use this order: passes on the private PID 1 diagnostic image with unchanged application code, the original deadline, late HTTP 404 rejection and automatic cleanup. The diagnostic also reproduced a distinct generic wake failure, so this - individual passing case does not complete final-image enablement. + individual passing case did not complete final-image enablement. The timeout + winner now also passes on normal image 6.0 through real Durable execution, + with a late HTTP 404 and no unapproved Read; the wider gate in step 1 remains. The [durable registration follow-up](./645-p3-registration-20260916.md) passed a lost reply after an actual registration commit, cancellation before identity registration, and explicit operator recovery/termination of a live diff --git a/docs/verification/645-p3-lifecycle-diagnostics.md b/docs/verification/645-p3-lifecycle-diagnostics.md index 4b23d4a5b..d753f5315 100644 --- a/docs/verification/645-p3-lifecycle-diagnostics.md +++ b/docs/verification/645-p3-lifecycle-diagnostics.md @@ -2,8 +2,12 @@ Updated 2026-09-16. The logging changes passed local checks and three isolated AWS workflows. The [normal-stack rollout](./645-p3-diagnostics-rollout-20260916.md) -now runs coordinator version 6 and image 5.0. They do not resolve the four -[recorded wake refusals](./645-p3-resume-refusal-investigation.md). +first deployed coordinator version 6 and image 5.0. The latest +[connection-close rollout](./645-p3-connection-close-rollout-20260916.md) runs +coordinator 10 and image 6.0, adds specific timeout feedback, and passed four +real Durable workflows. Automatic suspension remains disabled. The six +[recorded wake refusals](./645-p3-resume-refusal-investigation.md) are retained +for service-side diagnosis. ## What the records tell us diff --git a/docs/verification/645-p3-resume-refusal-investigation.md b/docs/verification/645-p3-resume-refusal-investigation.md index 927cc64c8..a3ef7cab9 100644 --- a/docs/verification/645-p3-resume-refusal-investigation.md +++ b/docs/verification/645-p3-resume-refusal-investigation.md @@ -6,8 +6,11 @@ captured a sixth exact refusal. The old connection's five-second idle timer expired at restoration; PID 1 still owned its listener, and no fresh HTTP connection or resume request appeared. Three explicit `Connection: close` candidate cases passed on a private image, proving closure before freeze and -fresh resume connections. Normal-image rollout and acceptance remain separate -steps; automatic suspension remains disabled. +fresh resume connections. The subsequent +[normal rollout](./645-p3-connection-close-rollout-20260916.md) deployed image 6.0 +and coordinator 10; approval, denial, timeout and cancellation while asleep +passed on the normal image using a private real Durable coordinator. +Automatic suspension remains disabled while the remaining P3 gates are open. An earlier [diagnostic retaining the original PID 1 server](./645-p3-pid1-observer-20260916.md) @@ -46,9 +49,9 @@ The [lifecycle diagnostics guide](./645-p3-lifecycle-diagnostics.md) documents the new logging and wake-failure feedback. Three isolated AWS workflows verified the instrumentation with the original server running as PID 1, including actual API wake and coordinator recovery. None reproduced refusal. The -[normal-stack rollout](./645-p3-diagnostics-rollout-20260916.md) now runs coordinator -version 9 and image 5.0 after subsequent coordinator fixes. The fifth failure -on image 5.0 includes those application diagnostics, as recorded below. +[normal-stack rollout](./645-p3-diagnostics-rollout-20260916.md) first installed +these diagnostics in image 5.0. The fifth failure on that image includes them, +as recorded below. The current image and coordinator are identified above. ## Recorded failures @@ -159,7 +162,7 @@ Suspend drains tracked work and checkpoints; it does not intentionally close the HTTP listener. The server's shutdown path is separate. That source review does not exclude a process crash or a lower-level restore problem. -## Next investigation +## Earlier experiments and remaining investigation The [minimal listener experiment](./645-p3-listener-probe-20260916.md) completed four full normal cases and recorded 14 guest resume acknowledgments without an @@ -185,16 +188,17 @@ the observer. The subsequent [local transport control](./645-p3-transport-control-20260916.md) reproduced a reset on an old connection after a six-second process pause, while all eight fresh-connection checks succeeded and the servers remained alive. -This is not a reproduction of the AWS refusal. It supplies a specific comparison -for the next cloud investigation without establishing a production fix. +This was not a reproduction of the AWS refusal. It supplied a specific comparison +for the later instrumented cloud investigation. -The [AWS connection comparison](./645-p3-diagnostics-rollout-20260916.md) has now -tested normal image 5.0 against a private image with HTTP connection reuse -disabled, keeping the original server as PID 1. Both quick and longer wakes +The earlier [AWS connection comparison](./645-p3-diagnostics-rollout-20260916.md) +tested normal image 5.0 against a private image with +`--timeout-keep-alive 0`, keeping the original server as PID 1. Both quick and longer wakes passed, and both missing-approval controls produced the expected guest HTTP 409 with a failed identity-read stage. The longer normal wake used a fresh client port; quick normal wakes reused one. No refusal was reproduced. These results -do not identify F01's cause or justify a production connection-setting change. +alone did not identify F01's cause. The later instrumented refusal and explicit +response-header comparison supply the additional evidence for the deployed fix. 1. Use the recorded worker IDs, region, timestamps and Resume request IDs to inspect service-side lifecycle diagnostics. Determine the actual connection @@ -203,14 +207,15 @@ do not identify F01's cause or justify a production connection-setting change. Application access logs do not provide that missing evidence. The latest failure now supplies these observations with the original PID 1. Compare its expired connection against the successful fresh-connection wake. -3. Verify the explicit close response on a private candidate image, including - its actual socket closure before freeze and fresh connection after restore. - Do not hide a failed wake by silently +3. Retain the verified private-candidate socket closure and normal-image + acceptance evidence. Do not hide a failed wake by silently launching another worker: the approved action and workspace may already have changed. -4. Repeat approval, denial, original/late deadline, multiple-gate persistence and - expired-credential cases on the corrected image. Keep the previous failures - in the evidence record. +4. Approval, denial, the original timeout winning over a late approval, and + cancellation while asleep now pass on image 6.0. Complete the wider final-image + checks, including multiple-gate workspace persistence, the late-decision + winner and actual expired-credential renewal. Earlier-image results remain + evidence for their recorded scope. 5. Leave production automatic suspension off until this gate and the [remaining acceptance plan](./645-p3-implementation-plan.md) are satisfied. diff --git a/docs/verification/645-p3-wake-transport-20260916.md b/docs/verification/645-p3-wake-transport-20260916.md index 590cddee4..aeaaaf0d9 100644 --- a/docs/verification/645-p3-wake-transport-20260916.md +++ b/docs/verification/645-p3-wake-transport-20260916.md @@ -8,6 +8,10 @@ deployment was unchanged during this comparison. This record separates actual AWS wake evidence from local calibration and attempts that failed before any sleep. Nothing here has been submitted to the service team. +The subsequent [normal rollout](./645-p3-connection-close-rollout-20260916.md) +deployed image 6.0 and coordinator 10. Four real Durable workflows passed on that +image; the normal automatic-suspension switches remain off. + ## Question and instrumentation An HTTP keep-alive connection lets a client reuse an existing connection for its @@ -249,8 +253,13 @@ An initial cleanup preflight rejected `/tmp` versus `/private/tmp` spellings of the same directory before any resource deletion. It was corrected by comparing resolved paths; that failed preflight and the successful cleanup are both retained. -The evidence is being preserved in the private permanent archive +The evidence is preserved in the private permanent archive `~/.local/share/abca-verification/645-p3-20260916/wake-transport-evidence.tar.gz`. +It contains 255 files, occupies 17,981,013 bytes, and has SHA-256 +`c949e28f68b0884d84c3b9d0ced6d166ed307d4f2fc8f876b89a1f1474e9b0f8`. +Every member's size and hash were checked against its manifest; archive +permissions are `0600`. + At comparison cleanup, the normal deployment remained coordinator 9 / image -5.0, 8,192 MiB, with automatic suspension disabled. Normal rollout is a separate -verification step. +5.0, 8,192 MiB, with automatic suspension disabled. The subsequent normal rollout +and its separate acceptance archive are recorded in the linked rollout report. From 87802a8c3fc1c212832842dc7d0fe82fa9c84520 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Wed, 16 Sep 2026 20:01:55 -0400 Subject: [PATCH 066/149] fix(ecs): wire approval storage into both task definitions --- cdk/src/constructs/ecs-agent-cluster.ts | 22 ++++++-- cdk/src/stacks/agent.ts | 3 +- cdk/test/constructs/ecs-agent-cluster.test.ts | 56 ++++++++++++++++++- cdk/test/stacks/agent.test.ts | 16 +++++- 4 files changed, 86 insertions(+), 11 deletions(-) diff --git a/cdk/src/constructs/ecs-agent-cluster.ts b/cdk/src/constructs/ecs-agent-cluster.ts index d6567b7ae..4b7cec542 100644 --- a/cdk/src/constructs/ecs-agent-cluster.ts +++ b/cdk/src/constructs/ecs-agent-cluster.ts @@ -41,6 +41,8 @@ export interface EcsAgentClusterProps { readonly agentImageAsset: ecr_assets.DockerImageAsset; readonly taskTable: dynamodb.ITable; readonly taskEventsTable: dynamodb.ITable; + /** Approval storage. Required for human approval gates; optional in isolated tests. */ + readonly taskApprovalsTable?: dynamodb.ITable; readonly userConcurrencyTable: dynamodb.ITable; readonly githubTokenSecret: secretsmanager.ISecret; readonly memoryId?: string; @@ -233,6 +235,7 @@ export interface EcsTaskSizing { const RESERVED_BUILD_ENV_KEYS = new Set([ 'TASK_TABLE_NAME', 'TASK_EVENTS_TABLE_NAME', + 'TASK_APPROVALS_TABLE_NAME', 'USER_CONCURRENCY_TABLE_NAME', 'LOG_GROUP_NAME', 'GITHUB_TOKEN_SECRET_ARN', @@ -297,12 +300,10 @@ export class EcsAgentCluster extends Construct { public readonly taskDefinition: ecs.FargateTaskDefinition; /** * The smaller read-only PLANNING task def (8 GB / 2 vCPU) — for any read-only - * workflow that clones + reads + emits an artifact but never builds. Same - * image/role/env/grants as the build def (shared task+execution role + a shared - * container spec, so a grant present on one def but missing on the other can't - * silently diverge); the ONLY difference is cpu/mem. The orchestrator selects - * this for read-only workflows on an ECS repo, so planning doesn't - * over-allocate the large build task. + * workflow that clones + reads + emits an artifact but never builds. Both + * definitions share their image, roles, grants and platform environment. + * Sizing, disk and build-tool settings differ. The orchestrator selects this + * definition for read-only workflows on an ECS repo. */ public readonly planningTaskDefinition: ecs.FargateTaskDefinition; public readonly securityGroup: ec2.SecurityGroup; @@ -375,6 +376,9 @@ export class EcsAgentCluster extends Construct { CLAUDE_CODE_USE_BEDROCK: '1', TASK_TABLE_NAME: props.taskTable.tableName, TASK_EVENTS_TABLE_NAME: props.taskEventsTable.tableName, + ...(props.taskApprovalsTable && { + TASK_APPROVALS_TABLE_NAME: props.taskApprovalsTable.tableName, + }), USER_CONCURRENCY_TABLE_NAME: props.userConcurrencyTable.tableName, LOG_GROUP_NAME: logGroup.logGroupName, GITHUB_TOKEN_SECRET_ARN: props.githubTokenSecret.secretArn, @@ -486,6 +490,12 @@ export class EcsAgentCluster extends Construct { } else { grantAgentTaskTableAccess(props.taskTable, taskRole, false); props.taskEventsTable.grantReadWriteData(taskRole); + props.taskApprovalsTable?.grant( + taskRole, + 'dynamodb:GetItem', + 'dynamodb:PutItem', + 'dynamodb:UpdateItem', + ); } // Capacity counters are coordinator-owned. The agent never accesses them. diff --git a/cdk/src/stacks/agent.ts b/cdk/src/stacks/agent.ts index d60ed598f..feb5d52c9 100644 --- a/cdk/src/stacks/agent.ts +++ b/cdk/src/stacks/agent.ts @@ -1065,6 +1065,7 @@ export class AgentStack extends Stack { }), taskTable: taskTable.table, taskEventsTable: taskEventsTable.table, + taskApprovalsTable: taskApprovalsTable.table, userConcurrencyTable: userConcurrencyTable.table, githubTokenSecret, memoryId: agentMemory.memory.memoryId, @@ -1074,7 +1075,7 @@ export class AgentStack extends Stack { // without this grant. The AgentCore runtime gets the equivalent grant // where it is created above. agentMemory, - // Read-only grant so the container can fetch its payload from S3. + // The task role reads bootstrap manifests; a signed URL delivers its payload. payloadBucket: ecsPayloadBucket!.bucket, // ECS parity: the same bucket the runtime uses for ARTIFACTS_BUCKET_NAME — // a repo-bound artifact workflow delivers here. Wires the diff --git a/cdk/test/constructs/ecs-agent-cluster.test.ts b/cdk/test/constructs/ecs-agent-cluster.test.ts index 887001439..788d30e65 100644 --- a/cdk/test/constructs/ecs-agent-cluster.test.ts +++ b/cdk/test/constructs/ecs-agent-cluster.test.ts @@ -38,6 +38,7 @@ function createStack(overrides?: { bedrockGeoRegion?: string; withMemory?: boolean; withLinearVault?: boolean; + withApprovals?: boolean; taskSizing?: { buildTaskCpu?: number; buildTaskMemoryMiB?: number; @@ -73,6 +74,12 @@ function createStack(overrides?: { const userConcurrencyTable = new dynamodb.Table(stack, 'UserConcurrencyTable', { partitionKey: { name: 'user_id', type: dynamodb.AttributeType.STRING }, }); + const taskApprovalsTable = overrides?.withApprovals + ? new dynamodb.Table(stack, 'TaskApprovalsTable', { + partitionKey: { name: 'task_id', type: dynamodb.AttributeType.STRING }, + sortKey: { name: 'request_id', type: dynamodb.AttributeType.STRING }, + }) + : undefined; const githubTokenSecret = new secretsmanager.Secret(stack, 'GitHubTokenSecret'); @@ -90,6 +97,7 @@ function createStack(overrides?: { agentImageAsset, taskTable, taskEventsTable, + taskApprovalsTable, userConcurrencyTable, githubTokenSecret, memoryId: overrides?.memoryId, @@ -711,6 +719,7 @@ describe('EcsAgentCluster construct', () => { }); const taskTable = mk('TaskTable'); const taskEventsTable = mk('TaskEventsTable'); + const taskApprovalsTable = mk('TaskApprovalsTable'); const userConcurrencyTable = new dynamodb.Table(stack, 'UserConcurrencyTable', { partitionKey: { name: 'user_id', type: dynamodb.AttributeType.STRING }, }); @@ -722,7 +731,7 @@ describe('EcsAgentCluster construct', () => { }), ], taskTable, - taskScopedTables: [taskEventsTable], + taskScopedTables: [taskEventsTable, taskApprovalsTable], traceArtifactsBucket: new s3.Bucket(stack, 'TraceBucket'), attachmentsBucket: new s3.Bucket(stack, 'AttachmentsBucket'), }); @@ -732,6 +741,7 @@ describe('EcsAgentCluster construct', () => { agentImageAsset, taskTable, taskEventsTable, + taskApprovalsTable, userConcurrencyTable, githubTokenSecret, agentSessionRole: sessionRole, @@ -781,7 +791,7 @@ describe('EcsAgentCluster construct', () => { // All tenant data stays on SessionRole; the agent has no counter access. expect(JSON.stringify(taskRoleStatements)).not.toContain('dynamodb:'); - // Main-task read/update statements plus the events-table grant. + // Main-task read/update statements plus events and approval grants. let conditioned = 0; for (const policy of Object.values(policies)) { for (const s of policy.Properties.PolicyDocument.Statement) { @@ -790,11 +800,51 @@ describe('EcsAgentCluster construct', () => { } } } - expect(conditioned).toBe(3); + expect(conditioned).toBe(4); + }); + + test('both task definitions point at the approval table authorized by the SessionRole', () => { + const tables = sessionTemplate.findResources('AWS::DynamoDB::Table'); + const approvalId = Object.keys(tables).find(id => id.startsWith('TaskApprovalsTable')); + expect(approvalId).toBeDefined(); + const definitions = Object.values(sessionTemplate.findResources('AWS::ECS::TaskDefinition')); + expect(definitions).toHaveLength(2); + for (const definition of definitions) { + expect(definition.Properties.ContainerDefinitions[0].Environment).toContainEqual({ + Name: 'TASK_APPROVALS_TABLE_NAME', + Value: { Ref: approvalId }, + }); + } }); }); }); +describe('EcsAgentCluster approval wiring without a SessionRole', () => { + let template: Template; + beforeAll(() => { template = createStack({ withApprovals: true }).template; }); + + test('grants only the approval item operations used by the agent', () => { + const approvalId = Object.keys(template.findResources('AWS::DynamoDB::Table')) + .find(id => id.startsWith('TaskApprovalsTable')); + const statements = Object.values(template.findResources('AWS::IAM::Policy')) + .flatMap(policy => policy.Properties.PolicyDocument.Statement); + const grants = statements.filter(statement => + JSON.stringify(statement.Resource).includes(approvalId!), + ); + expect(grants).toEqual([expect.objectContaining({ + Effect: 'Allow', + Action: ['dynamodb:GetItem', 'dynamodb:PutItem', 'dynamodb:UpdateItem'], + })]); + }); + + test('rejects a build setting that would erase approval-table wiring', () => { + const node = new Stack(new App({ + context: { ecsExtraBuildEnv: { TASK_APPROVALS_TABLE_NAME: '' } }, + }), 'S').node; + expect(() => resolveEcsTaskSizing(node)).toThrow('TASK_APPROVALS_TABLE_NAME'); + }); +}); + describe('EcsAgentCluster payload bucket (#502)', () => { function createWithPayloadBucket(): Template { const app = new App(); diff --git a/cdk/test/stacks/agent.test.ts b/cdk/test/stacks/agent.test.ts index 4da420417..ee8091788 100644 --- a/cdk/test/stacks/agent.test.ts +++ b/cdk/test/stacks/agent.test.ts @@ -1014,7 +1014,7 @@ describe('AgentStack with the ECS substrate gate (--context compute_type=ecs)', test('provisions an ECS cluster + both Fargate task definitions (build + planning)', () => { template.resourceCountIs('AWS::ECS::Cluster', 1); - // Two task defs — the 64 GB build def and the 8 GB read-only planning def + // Two task defs — the 16 GB build def and the 8 GB read-only planning def // (a read-only workflow runs on the smaller one). See // docs/design/ECS_RIGHTSIZED_PLANNING.md. template.resourceCountIs('AWS::ECS::TaskDefinition', 2); @@ -1024,6 +1024,20 @@ describe('AgentStack with the ECS substrate gate (--context compute_type=ecs)', template.hasOutput('ComputeSubstrate', { Value: 'ecs' }); }); + test('both ECS task definitions receive the same approval table as AgentCore', () => { + const runtime = Object.values(template.findResources('AWS::BedrockAgentCore::Runtime'))[0]; + const approvalTable = runtime.Properties.EnvironmentVariables.TASK_APPROVALS_TABLE_NAME; + expect(approvalTable).toBeDefined(); + const definitions = Object.values(template.findResources('AWS::ECS::TaskDefinition')); + expect(definitions).toHaveLength(2); + for (const definition of definitions) { + expect(definition.Properties.ContainerDefinitions[0].Environment).toContainEqual({ + Name: 'TASK_APPROVALS_TABLE_NAME', + Value: approvalTable, + }); + } + }); + test('the orchestrator gets the PLANNING task-def ARN, not just the build one', () => { // Without this env var the ECS strategy's `readOnly && // ECS_PLANNING_TASK_DEFINITION_ARN` guard is always falsy, so the planning From a6bf4743a64fefd08d7256f159aa1f0e8e2c0cba Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Wed, 16 Sep 2026 20:23:12 -0400 Subject: [PATCH 067/149] docs: record final-image lifecycle and ECS acceptance --- .../645-lambda-microvm-service-feedback.md | 20 +- .../645-p3-final-image-and-ecs-20260917.md | 249 ++++++++++++++++++ .../645-p3-implementation-plan.md | 52 +++- docs/verification/645-p3-readiness-review.md | 13 +- 4 files changed, 318 insertions(+), 16 deletions(-) create mode 100644 docs/verification/645-p3-final-image-and-ecs-20260917.md diff --git a/docs/verification/645-lambda-microvm-service-feedback.md b/docs/verification/645-lambda-microvm-service-feedback.md index 0672d66b8..52a3b3de9 100644 --- a/docs/verification/645-lambda-microvm-service-feedback.md +++ b/docs/verification/645-lambda-microvm-service-feedback.md @@ -1,6 +1,6 @@ # Lambda MicroVM service-team feedback tracker -Updated 2026-09-16. Working notes for the ADR-021 takeover. **Not submitted to the +Updated 2026-09-17. Working notes for the ADR-021 takeover. **Not submitted to the service team.** Keep each item's evidence, question, service response and next action here as verification continues. @@ -10,7 +10,7 @@ is the receipt that lets the service team find a particular call. | ID | Priority | Topic | Evidence/status | |---|---|---|---| -| F01 | High; service diagnosis open | Accepted wake ends in connection refusal | Six recorded failures; pre-freeze connection closure verified; image 6.0 correction deployed and four Durable workflows passed | +| F01 | High; service diagnosis open | Accepted wake ends in connection refusal | Six recorded failures; pre-freeze connection closure verified; image 6.0 correction deployed and seven Durable workflows passed | | F02 | High | Supported IAM conditions and misleading permission errors | Reproduced in earlier P2 work; current service behavior needs confirmation | | F03 | Medium | A service-side hook timeline and structured failure details | Diagnostic improvement request based on F01 | | F04 | Medium | `PENDING` also means restoring an existing worker | Observed live; our timer bug is fixed | @@ -26,6 +26,14 @@ The coordinator releases capacity correctly; the requested coding workflow failed in these recorded cases. The application correction below is deployed; automatic suspension remains disabled pending the remaining P3 acceptance gates. +The [September 17 follow-up](./645-p3-final-image-and-ecs-20260917.md) adds two +repository sleep/wake cycles, late approval winning, and wake after actual STS +expiry on normal image 6.0. The long-sleep worker remained suspended after its +old credentials expired, then renewed with identical identity tags and retained +its original approval deadline. All three workflows finalized without repair. +These are additional application acceptance results, not service-side traces +or a measured failure rate. + **Observed:** five failures on normal images `3.0`, `4.0` and `5.0`, plus a sixth on a private image retaining 5.0's application and connection behavior with transport probes, in `us-west-2`, September 15–16. AWS accepted `ResumeMicrovm`, @@ -268,6 +276,14 @@ state. The installed SDK documents idempotency but gives no token-retention duration. ABCA therefore stops automatic recovery without a saved handle after its own conservative 120-second deadline. +**Public documentation recheck (September 17):** the +[RunMicrovm API reference](https://docs.aws.amazon.com/lambda/latest/microvm-api/API_RunMicrovm.html) +is now reachable. It describes `clientToken` as “A unique, case-sensitive +identifier you provide to ensure the idempotency of the request,” with a +1–128-character limit. It still gives no retention duration or post-expiry +behavior. Its request schema has no per-worker tags field. This resolves the +earlier documentation-access problem, not the missing service contract. + **Impact:** if AWS created a worker but its response was lost, an operator needs to find that exact worker. An undocumented retention boundary prevents proving that a late retry cannot create another one. ABCA must not invent a new token diff --git a/docs/verification/645-p3-final-image-and-ecs-20260917.md b/docs/verification/645-p3-final-image-and-ecs-20260917.md new file mode 100644 index 000000000..3d0707f72 --- /dev/null +++ b/docs/verification/645-p3-final-image-and-ecs-20260917.md @@ -0,0 +1,249 @@ +# ADR-021 P3: final-image persistence, credential expiry and ECS compatibility + +Verification ran September 16–17, 2026, in `us-west-2`, account +``. The normal deployment remains MicroVM image **6.0**, +coordinator **10**, 8,192 MiB, with both automatic-suspension gates **off**. +The [connection-close rollout](./645-p3-connection-close-rollout-20260916.md) +records the preceding four core lifecycle cases. + +## Repository state across two sleeps + +Task `01M2P84R90WESJH1GPTYKZ20PM` used normal image 6.0, its original server as +PID 1, and the production Durable handler in a private coordinator. It cloned +public repository `isadeks/vercel-abca-linear`, main commit +`f5be1e23a964b661f1bf3d98ead55679d65978c4`, into a temporary directory. + +The actual tool sequence was: + +1. Clone, verify the commit, run `npm ci`, lint and tests, then write a marker. +2. Request approval to read the marker, sleep, approve through the normal API, + and read it after waking. +3. Verify the first marker's hash and write a second marker. +4. Request another approval, sleep again, approve and read the second marker. +5. Verify both hashes and the unchanged repository commit, check the tracked + files are unchanged, and run lint and tests again. + +Both lint runs and both Vitest runs passed; the repository has one test. +The marker hashes were: + +| Marker | SHA-256 | +|---|---| +| First | `8437311a6e403b723febea0fbae6509e03b48c198b597aab9f04cd2cf7e30d5e` | +| Second | `cb1e409f22270af5 (SHA-256 prefix)` | + +Worker `microvm-3a114559-4c86-34a0-af36-0a04e6232a5f` recorded two suspend and +two resume HTTP 200 results from PID 1. Independent normal approval-handler +logs supplied both actual `ResumeMicrovm` request IDs. Each gate kept its +original creation time and 600-second timeout. Finalization passed at +**23:25:41.137 UTC**: task `COMPLETED`, Durable execution `SUCCEEDED`, worker +`TERMINATED`, reservation released, counter zero and payload absent. +The watcher performed no repair. + +This verifies repository files and running application state across multiple +sleeps. The task used artifact delivery and explicitly cloned into a temporary +directory. It does **not** exercise the platform's normal repository-bound +clone/PR-delivery workflow. No commit, push, PR or external comment was created. + +## Late approval wins its decision race + +Task `01M2P84R94PY36ZV5D43RVN7MT`, worker +`microvm-f191abfa-de75-3a89-bb5f-d1b4b9865ce1`, used image 6.0 and a 150-second +approval window. The original deadline was **23:28:44 UTC**. + +The private coordinator deliberately withheld its scheduled resume until five +seconds after that deadline. Those omissions were recorded as injected missing +commands, never as successful AWS requests. The normal approval request began +at **23:28:46.349** and returned HTTP 202 at **23:28:47.593**. Its actual resume +request ID was `c3eed6c5-5029-494a-a45b-162de6dfce63`. + +The original PID 1 finished the resume hook with HTTP 200, and the permitted +Read completed. The approval remained `APPROVED` with its original clock. +Task completion, successful Durable execution, worker termination, reservation +release and payload cleanup passed at **23:29:01.093**, without watcher repair. + +This is the existing decision contract: the first committed decision wins. +The agent writes `TIMED_OUT`; the approval API does not independently reject a +request solely because the clock deadline passed. The preceding image 6.0 +acceptance separately verified timeout winning and a later approval returning +HTTP 404. + +The independent two-case audit passed at **23:29:55.019**. Its private +coordinator, role, switch, log group and two zero counters were removed, with +absence verified at **23:31:04.765**. All 237 function log events were retained. + +## Real credential expiry + +The separate task `01M2P7E6SF68775QCBYXS1Z0YP` uses normal image 6.0 and worker +`microvm-ddc38544-f478-3578-a78d-79e1ac7a32fa`. Its initial credentials expire at +**00:07:25 UTC on September 17**, according to the actual STS issuance recorded +in CloudTrail. Its original approval deadline is **00:11:10**. + +An independent `GetMicrovm` observed the worker still `SUSPENDED` at +**00:07:40.565**, after that actual expiry. Request ID: +`1761febc-81d7-43d5-97b2-841c8f553f94`. + +Both the lifecycle audit and the independent credential audit passed. + +| Event | UTC time | +|---|---| +| Initial STS session issued | September 16, 23:07:25 | +| Worker observed suspended | September 16, 23:11:41.565 | +| Initial credentials expired | September 17, 00:07:25 | +| Independent observation still `SUSPENDED` | 00:07:40.565 | +| Actual Resume accepted | 00:10:11.100 | +| New STS session issued | 00:10:12 | +| Original PID 1 finished resume with HTTP 200 | 00:10:12.242 | +| Original approval deadline reached | 00:11:10 | +| Late approval returned HTTP 404 | 00:11:12.097 | +| Finalization verified without watcher repair | 00:11:22.157 | + +The new STS session expires at **01:10:12**. CloudTrail request +`e1320fee-c291-44da-8f32-dff210efcd38` records the same session role, session name, +task tag and user tag as the initial request +`85c3d6af-9bf7-4bfc-8555-ab523f28eb64`. CloudTrail timestamps have second +precision. The original PID 1 completed its refresh/reconciliation hook in +132 ms; the actual Resume request ID was +`1a817784-6702-43ed-828e-532a18ec06c4`. + +The gate became `TIMED_OUT`, and the unapproved Read did not succeed. The task +then completed its refusal response, the Durable execution succeeded, and the +worker terminated. Its reservation was released, counter was zero and payload +was absent. The retained trace had no dropped events. This distinguishes +completing the task's refusal response from permitting the timed-out tool. + +The renewal record became visible in CloudTrail about three minutes after +issuance. The independent audit required every pre-observation session to +have expired, a successful subsequent issuance with unchanged identity tags, +and the successful resume/lifecycle evidence. No credential values were saved. + +## ECS approval configuration fix + +ECS runs the same agent in a container. Its task definition is the recipe that +tells the container its image, memory, permissions and configuration. + +The production ECS recipe omitted `TASK_APPROVALS_TABLE_NAME`. The agent's +approval code therefore raised `ApprovalTablesUnavailable`, even though the +shared session role already had approval-table permissions. + +Local commit `c14d2ba0` fixes this in the source: + +- Pass the real approval table from `AgentStack` into `EcsAgentCluster`. +- Give both build and planning containers the same approval-table name. +- Prevent deployment-time build settings from erasing that name. +- Preserve session-role ownership of production data permissions. The + optional path without a session role gets only approval Get/Put/Update. +- Correct nearby comments about task sizing, shared environment and signed + payload delivery. + +Compilation, ESLint and 186 targeted construct/stack tests passed. The full +CDK suite passed **5,057 tests in 229 suites**, with one snapshot; the existing +56 optional DynamoDB tests were skipped. No Python runtime change was needed. + +## ECS live approval and cancellation + +A separate 23-resource stack used the production `EcsAgentCluster`, +`AgentSessionRole` and `EcsPayloadBucket` constructs in the existing private +subnets. It created its own roles, buckets, cluster, security group, task +definitions, switch and logs. The normal deployment and its roles were not +modified. + +The unchanged Dockerfile image was built for ARM64 and published to an +immutable tag in a private verification repository. Its OCI index digest was +`sha256:e892253aaa2565d2064da2fbb13c09f85061cbb5095355b8993fa249fa10c2c0`. +Both workers reported that digest. Production build sizing remained +4 vCPU / 16 GiB / 50 GiB; planning remained 2 vCPU / 8 GiB. + +The production Durable handler ran in private coordinator version 2. Approval +used the normal API. Cancellation used the production cancellation handler in +a private function with its own restricted role, because the normal deployment +has no ECS substrate or ECS cancellation grant. A `timeout` wrapper bounded +each test container to 900 seconds; it ran the normal batch bootstrap command +inside that limit. Coordinator polling was shortened to five seconds. + +| Case | Task | Result | +|---|---|---| +| Approve while waiting | `01M2PA7M84P99ZDV97920NRSEX` | `COMPLETED`; one successful Read | +| Cancel while waiting | `01M2PA7M891EW8VW3FMW9APFXV` | `CANCELLED`; no successful Read | + +Each task requested a five-second MicroVM sleep delay but remained ECS +`RUNNING` throughout an approval wait longer than 20 seconds. Neither acquired +MicroVM lifecycle/start metadata. Both retained their original approval clocks, +finished their actual Durable executions successfully, reached ECS `STOPPED`, +released their reservations and deleted their payloads without watcher repair. +The independent audit passed at **00:01:21.042 UTC**. + +The first deployment attempt used an incorrect managed-policy name in the +scratch cancellation role and rolled back before any task ran. A subsequent +watcher incorrectly treated the JSON-text approval preview as an object; that +task was stopped and repaired, retained as excluded evidence, and replaced with +fresh IDs. The final audit also corrected its event-name lookup to the actual +`agent_tool_result` schema. These verification errors are not counted as product +failures or successful unattended acceptance. + +## Actual ECS permissions and network ports + +A separate planning container used the real ECS task role and the production +scoped-session provider. It passed **19 checks**, exited zero, and reached +`STOPPED`: + +| Check | Result | +|---|---| +| Ambient task/counter reads | Denied | +| Ambient own bootstrap marker read | Allowed | +| Ambient payload read and payload-bucket listing | Denied | +| Scoped own task read | Allowed | +| Scoped other-task and counter reads | Denied | +| Scoped reporting update shape | Authorized; deliberately false condition prevented mutation | +| Scoped owner, compute handle, start receipt, reservation and lifecycle updates | Denied | +| Scoped bootstrap read | Denied | +| Scoped own artifact write / other-task artifact write | Allowed / denied | +| TCP 443 / TCP 80 to the same address | Connected in 8 ms / timed out after 5 seconds | + +Both task records were unchanged afterward. The operator could reach both ports +on that same external address before and after the guest probe. The deployed +security group has no ingress and permits only outbound TCP 443; public IP +assignment was disabled. The network probe verifies this egress restriction, +not remote-MCP behavior or a live public-ingress negative test. + +The session role limits an existing tagged session to its task. The compute role +chooses the tags when assuming it; these checks do not establish protection +against a compromised worker minting a different session identity. + +## Cleanup, evidence and remaining scope + +The long-expiry coordinator, role, private switch, log group and zero counter +were removed with absence verified at **00:15:38.640 UTC**. All 1,524 private +function log events were retained. Together with the earlier two-case cleanup, +all three verification deployments in this record have been removed. + +ECS teardown was verified at **00:07:31.420 UTC**. The private stack, both +functions and all versions, five roles, three buckets, registry and private +logs were removed. All four ECS workers were stopped, and the three owned +zero counters were deleted. The automatically created Container Insights +performance log was archived and removed separately from CloudFormation. + +The image is an OCI index referencing an ARM64 manifest and a build-attestation +manifest. Cleanup first stopped on an incorrect one-record assumption, then +verified and removed the exact three-digest set. The exact image is retained +locally as a Docker archive. Both inactive task definitions were submitted for +deletion and reported `DELETE_IN_PROGRESS`; these are service records, not +running containers. + +Private evidence directories: + +- `/tmp/abca-645-p2-clean-20260913/p3-image6-matrix-20260916` +- `/tmp/abca-645-p2-clean-20260913/p3-image6-expiry-20260916` +- `/tmp/abca-645-p2-clean-20260913/p3-ecs-compatibility-20260916` + +Permanent private archive: +`/Users/sphias/.local/share/abca-verification/645-p3-20260916/final-image-and-ecs-evidence.tar.gz`. +It contains 188 files, including the exact Docker image, and is 701,261,332 bytes +with mode `0600`. SHA-256: +`576a9d5171fc6b97562310d5f12809e601a88d73e30ac6a3c500d8d9d9bb093f`. +Every archived file hash was checked against the retained manifest. + +The normal automatic-suspension switches stay off. Service-side wake traces, +Run-token retention and recovery without guest identity logs, the remaining +backend/network/MCP checks, normal repository-bound delivery, and the applicable +deployment/drain/rollback procedure remain separately tracked in the +[implementation plan](./645-p3-implementation-plan.md). diff --git a/docs/verification/645-p3-implementation-plan.md b/docs/verification/645-p3-implementation-plan.md index b247813c4..882ed2ed7 100644 --- a/docs/verification/645-p3-implementation-plan.md +++ b/docs/verification/645-p3-implementation-plan.md @@ -6,6 +6,20 @@ Prepared 2026-09-13 from `main` `5e10038c7e28179b302ac4de78b709795aeba3ce`. Read Prerequisite work is tracked here on `fix/645-microvm-readiness`. “Completed” means implemented and checked locally; AWS deployment and live verification have separate completion gates below. +**Final-image and ECS follow-up (2026-09-17):** the +[new acceptance record](./645-p3-final-image-and-ecs-20260917.md) verifies +repository marker persistence through two sleeps on image 6.0, unchanged +repository commit and passing lint/tests before and after, and late approval +winning with the original deadline. Both private Durable cases finalized +without repair and their infrastructure was removed. The same follow-up found +and fixed missing ECS approval-table configuration in source `c14d2ba0`. +Two fresh ECS approval/cancellation cases and 19 real permission/port checks +passed; the isolated ECS deployment was removed. The full CDK suite passed +5,057 tests. The real image 6.0 credential-expiry case also passed: its worker +was observed asleep after expiry, CloudTrail confirmed a new session preserving +identity, and the original deadline and automatic finalization held. Normal +sleep gates remain off. + **Wake transport correction (2026-09-16):** the [instrumented comparison](./645-p3-wake-transport-20260916.md) captured a sixth exact refusal: the restored event loop expired the old suspend connection while @@ -182,7 +196,8 @@ is superseded by these records. - [x] Bind configuration to IAM-authenticated deployment manifests and use single-object payload links for ECS/MicroVM (#817 / #700). - [x] Verify MicroVM manifest/download transport, malformed or mismatched inputs, URL expiry/revocation, a foreign private-bucket denial and >1 MiB transport in AWS; verify concurrent/repeated/conflicting S3 preparation with operator credentials. - [x] Verify deployed MicroVM-role metadata/S3 permissions, public-object denial and actual signer-credential expiry in AWS; see the [effective IAM evidence](./645-effective-iam-20260915.md) for scope. -- [ ] Complete other-backend roles, runtime network paths and the ECS/coordinated-rollout matrix in AWS. +- [x] Fix ECS approval-table configuration and verify current-container approval/cancellation, v2 payload cleanup, ambient/scoped permissions and port 443/80 controls in a bounded deployment; remove its infrastructure. +- [ ] Complete the remaining AgentCore permission checks, runtime ingress/remote-MCP paths and coordinated-rollout matrix in AWS. - [x] Implement saved MicroVM start receipts, stable tokens, input fingerprints and handle recovery. - [x] Verify immediate identical `RunMicrovm` replay returns the same worker ID in the live payload probes. - [x] Verify simultaneous identical Run calls, changed-parameter rejection, and replay after termination through roughly five minutes against AWS; distinguish cached Run responses from fresh VM state. @@ -216,7 +231,7 @@ is superseded by these records. - [ ] Deploy and verify the complete P3 sleep/wake lifecycle in AWS. - [x] Deploy the explicit SDK callback-timeout fix and repeat long sleep, late wake, approval, denial and cancellation; require final approval/tool evidence as well as cleanup. Image `4.0` passed these checks, including real renewal after credential expiry. - [x] Deploy the explicit connection-close correction and verify approve, deny, timeout and cancel-asleep on normal image 6.0 through real Durable execution; retain all six exact refusals and the generic failure. The [transport comparison](./645-p3-wake-transport-20260916.md) proves closure before freeze; the [rollout record](./645-p3-connection-close-rollout-20260916.md) records normal-image acceptance and cleanup. -- [ ] Complete the wider final-image workspace, multiple-gate, late-decision-winner and expired-credential checks before enabling automatic suspension. +- [x] Complete the wider image 6.0 workspace, multiple-gate, late-decision-winner and expired-credential checks, with actual CloudTrail renewal proof and private cleanup. The repository case uses a temporary clone/artifact workflow; normal repository-bound delivery remains separate. First prerequisite batch completed locally on 2026-09-13: @@ -310,10 +325,12 @@ absence of all temporary infrastructure. The detailed batches below preserve the implementation history. For the current handoff, use this order: -1. Complete the wider final-image lifecycle matrix on image 6.0: a cloned - repository with mutable files across multiple approval gates, the late-decision - winner, and wake after actual credential expiry. Earlier image 4.0/5.0 results - remain evidence for their recorded scope. +1. The wider final-image lifecycle matrix on image 6.0 is now complete for its + recorded scope: a temporary repository clone with mutable files across two + approval gates, the late-decision winner, and wake after actual credential + expiry with CloudTrail renewal proof. See the + [final-image follow-up](./645-p3-final-image-and-ecs-20260917.md). + Earlier image 4.0/5.0 results remain evidence for their recorded scope. The [connection-close correction](./645-p3-wake-transport-20260916.md) and [normal rollout](./645-p3-connection-close-rollout-20260916.md) are complete. Three instrumented candidates proved socket closure before freeze and a new @@ -348,21 +365,30 @@ handoff, use this order: The diagnostic also reproduced a distinct generic wake failure, so this individual passing case did not complete final-image enablement. The timeout winner now also passes on normal image 6.0 through real Durable execution, - with a late HTTP 404 and no unapproved Read; the wider gate in step 1 remains. + with a late HTTP 404 and no unapproved Read. The wider image 6.0 matrix in + step 1 has subsequently passed too. The [durable registration follow-up](./645-p3-registration-20260916.md) passed a lost reply after an actual registration commit, cancellation before identity registration, and explicit operator recovery/termination of a live worker whose ID never reached coordinator state. The latter required an exact task/worker pair in the guest log; it does not establish post-retention behavior or a recovery path when those logs are unavailable. -3. Complete effective permissions and network checks for the other backends, - plus runtime/remote-MCP connectivity. Verify a full cloned-repository P3 - workflow on the final image, including mutable workspace state and normal - P2 behavior. Respect the target repository's publication checks. +3. Complete the remaining AgentCore permission and runtime/network negatives, + plus remote-MCP connectivity. The + [ECS follow-up](./645-p3-final-image-and-ecs-20260917.md) now verifies real + ambient/scoped permission requests, port 443 success versus port 80 denial, + approval/cancellation, payload deletion and stopped workers on the current + shared container. Its source fix supplies the missing approval-table name + to both ECS task definitions. Trace and nudge environment parity are separate + existing ECS gaps; they were not silently supplied by the fixture. + The final-image repository test proves mutable files across two sleeps, + but uses an explicit temporary clone inside an artifact task. Verify the + normal repository-bound clone/delivery path separately, respecting the + target repository's publication checks. The [AgentCore follow-up](./645-p3-agentcore-20260916.md) now verifies its current shared container, approval/cancellation, exclusion from MicroVM sleep, - reservation release and owned-session cleanup. The stack has no ECS resources; - those checks require a separate bounded deployment. + reservation release and owned-session cleanup. The normal stack has no ECS + resources; the separate bounded ECS deployment has now been tested and removed. 4. Exercise the coordinated capacity upgrade/drain and rollback procedure under deployed writer roles, including realistic scan volume. Retain the verified local transaction and isolated-live results as evidence for their narrower diff --git a/docs/verification/645-p3-readiness-review.md b/docs/verification/645-p3-readiness-review.md index 9192f5682..c534fabb7 100644 --- a/docs/verification/645-p3-readiness-review.md +++ b/docs/verification/645-p3-readiness-review.md @@ -4,6 +4,17 @@ Reviewed 2026-09-13 against `main` commit `5e10038c7e28179b302ac4de78b709795aeba This review covers the existing MicroVM implementation, related P2 follow-ups, comment accuracy and nested-stack feasibility. It includes local experiments, not an AWS deployment or a new live smoke run. The associated [implementation plan](./645-p3-implementation-plan.md) turns the findings into ordered work. +**Current status (2026-09-17):** subsequent +[normal-image acceptance](./645-p3-connection-close-rollout-20260916.md) and the +[final-image/ECS follow-up](./645-p3-final-image-and-ecs-20260917.md) verify seven +real Durable workflows on image 6.0: approval, denial, timeout, cancellation, +repository files across two sleeps, late approval winning, and wake after real +credential expiry. The ECS follow-up fixes missing approval-table configuration +and passes approval/cancellation plus 19 permission/port checks. Private +verification infrastructure was removed. Normal automatic sleep remains off +until the separately listed delivery, permission/network and deployment gates +are complete. Earlier paragraphs below preserve their dated findings. + **Live update (2026-09-14):** subsequent work completed a [clean deployment](./645-p2-clean-deployment-20260913.md) and [real coding, PR iteration and cancellation tests](./645-p2-live-task-20260914.md), including Memory writes and runtime logging. A normal [image rebuild](./645-microvm-image-rebuild-20260914.md) activated version `2.0`; [11 live payload cases](./645-p2-payload-live-20260914.md) then verified transport/rejection, URL expiry/revocation and immediate Run replay. Those records supersede the corresponding gaps in the historical notes below. Full P2 acceptance and integrated P3 sleep/wake remain open; the original review findings are retained as a dated baseline. **Implementation update (2026-09-13):** subsequent local batches fix thread isolation, deletion/error/byte/contract bugs, approval heartbeat, stable MicroVM start recovery, atomic capacity reservations and coordinator metadata permissions. The latest batch implements v2 authenticated deployment manifests and single-object payload links for both ECS and MicroVM (#817/#700), with no old unsigned fallback. See the [bootstrap runbook](./645-payload-bootstrap.md) and [implementation progress](./645-p3-implementation-plan.md#implementation-progress). A further local batch removes unused logging counters in favor of structured stdout failures and verifies large registry assets through v2 delivery and the local loader. At that batch's completion, effective AWS policies, expiry/networking, stdout ingestion, remote-tool connectivity, clean deployment and P3 sleep/wake were pending; the live update above records later evidence. Findings below preserve the original reviewed baseline, rather than describing all of them as current defects. @@ -58,7 +69,7 @@ An **IAM role** is a permission badge. A **trust policy** says who may wear that |---|---|---| | P1 | Build the computer, start it, deliver a task, check it and stop it | Merged in [#689](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/689). Strategy, infrastructure, bootstrap permissions, packaging, types, `/ready` and `/run` exist. | | P2 | Make a real coding task work with configuration, permissions, logs and progress | Merged in [#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733). `/validate`, `/terminate`, warm-up, runtime grants and heartbeat support exist. The September 14 clean rerun passed coding, iteration and cancellation without manual IAM workarounds, followed by image 2.0 payload/start checks. The broader deployed recovery, effective IAM and network matrix remains open. | -| P3 | Sleep during a human approval wait, wake correctly, and keep deadlines/credentials safe | Local foundations include intent/policy, guest pause control, scoped credential renewal, production HTTP checkpoint/wake hooks and per-worker image capability. Durable supervision and post-commit approval wake are now connected locally with bounded recovery and scoped IAM. Live acceptance remains open; automatic sleep defaults off. | +| P3 | Sleep during a human approval wait, wake correctly, and keep deadlines/credentials safe | Implemented and deployed. Seven image 6.0 Durable workflows verify the core lifecycle, repository-file persistence and real expired-credential renewal. Broader delivery, permission/network and deployment gates remain open; automatic sleep stays off. | | P4 | — | ADR-021 defines no P4. Verification runbooks have their own numbered phases; those are not extra ADR milestones. | The old unchecked checklist and the word “proposed” do not erase the merged work. Conversely, merged code is not proof that the final deployment path works unattended. From cd99f223cc64c10a51de3c38baf61d514234ab64 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Wed, 16 Sep 2026 20:45:33 -0400 Subject: [PATCH 068/149] docs: verify repository wake path and correct workflow comments --- agent/src/models.py | 29 ++-- agent/workflows/coding/pr-review-v1.yaml | 7 +- agent/workflows/default/agent-v1.yaml | 6 +- .../645-lambda-microvm-service-feedback.md | 15 +- .../645-p3-implementation-plan.md | 19 ++- docs/verification/645-p3-readiness-review.md | 9 +- .../645-p3-repository-path-20260917.md | 136 ++++++++++++++++++ 7 files changed, 193 insertions(+), 28 deletions(-) create mode 100644 docs/verification/645-p3-repository-path-20260917.md diff --git a/agent/src/models.py b/agent/src/models.py index 14a6c1f2a..ab6b89ab0 100644 --- a/agent/src/models.py +++ b/agent/src/models.py @@ -180,24 +180,21 @@ class TaskConfig(BaseModel): # scheme is unchanged; read-only enforcement no longer keys off this # principal — it keys off ``read_only`` below. policy_principal: str = "new_task" - # Whether the resolved workflow is read-only (may not mutate the working - # tree). Threaded into the Cedar request ``context.read_only`` so the - # hard-deny Write/Edit rules fire for *any* read-only workflow, and drives the - # runner's allowed_tools tightening. + # The workflow's read-only policy flag. Threaded into Cedar's + # ``context.read_only`` to hard-deny Write/Edit, and used by the runner to + # remove those tools from SDK auto-approval. Other tools still depend on + # their own policy rules; this flag does not remove Bash from the surface. read_only: bool = False - # The SDK tool surface for this task, from the resolved workflow's - # ``agent_config.allowed_tools``. This is the second enforcement layer - # alongside ``read_only``: ``run_agent`` passes it to - # ``ClaudeAgentOptions.allowed_tools`` verbatim, and drops ``Write``/``Edit`` - # when ``read_only`` is true. Empty list means "fall back to the built-in - # full surface" so legacy/batch callers that never resolved a workflow keep - # working unchanged; a workflow that wants to restrict tools MUST declare a - # non-empty list (every shipped workflow does). + # SDK auto-approval list from ``agent_config.allowed_tools``. The runner + # drops Write/Edit when read_only is true; an empty list falls back to the + # built-in list for legacy/batch callers. This does not restrict available + # tools: unlisted tools fall through to the SDK permission mode. Actual + # restrictions require disallowed_tools or applicable Cedar forbid rules. allowed_tools: list[str] = Field(default_factory=list) - # Whether the resolved workflow requires a repo. False for repo-less - # knowledge workflows: the pipeline skips clone/build/PR and drives the agent - # + deliver_artifact steps through the workflow runner. Defaults True so - # coding tasks (and any caller that omits it) keep the repo-bound path. + # Whether the workflow requires a repo. False makes the repo optional: + # the pipeline skips repository setup only when no repo was supplied. + # A supplied repo still takes the repository-bound path. Defaults True for + # coding tasks and callers that omit the field. requires_repo: bool = True # True when the resolved workflow operates on an existing PR (pr_* coding # workflows) — gates the "resume existing branch / resolve PR" behavior that diff --git a/agent/workflows/coding/pr-review-v1.yaml b/agent/workflows/coding/pr-review-v1.yaml index 486ecfc0b..6b757852a 100644 --- a/agent/workflows/coding/pr-review-v1.yaml +++ b/agent/workflows/coding/pr-review-v1.yaml @@ -2,9 +2,10 @@ # mutating it. Selected via workflow_ref (pr_review → coding/pr-review-v1). # # read_only:true with Write/Edit dropped from allowed_tools (validator rule 6 — -# read-only tier may not grant mutating tools). Read-only is enforced by Cedar -# `context.read_only == true` (hard-deny read_only_forbid_write/edit) AND by the -# trimmed allowed_tools — defense in depth. It resolves the existing PR URL +# read-only tier may not list Write/Edit). Cedar's `context.read_only == true` +# hard-denies Write/Edit. The SDK allowed_tools list controls auto-approval; +# omitting a tool from it does not remove that tool from the available surface. +# Bash commands still depend on their own policy rules. It resolves the PR URL # without pushing via ensure_pr(strategy: resolve); the post_review step is not # used (its registered handler is intentionally unimplemented). The Cedar # principal is the id-derived audit identity "pr_review" — enforcement keys off diff --git a/agent/workflows/default/agent-v1.yaml b/agent/workflows/default/agent-v1.yaml index 5cbfb5033..1061f5b3b 100644 --- a/agent/workflows/default/agent-v1.yaml +++ b/agent/workflows/default/agent-v1.yaml @@ -6,7 +6,7 @@ # hatch — so its behavior is auditable and overridable per-repo via the Blueprint. # # Deliberately conservative because it runs when *nothing* was specified: -# requires_repo:false (repository optional), a read-leaning tool set (no Bash/Write/Edit), +# requires_repo:false (repository optional), a read-leaning auto-approval list, # and delivery to S3 + a comment milestone. A caller who wants coding selects # (or maps to) a coding workflow. soft_deny is still mandatory (read_only:false) # so any future tool addition stays HITL-gated. @@ -35,8 +35,8 @@ hydration: sources: [task_description, attachments, memory] agent_config: tier: standard - # Conservative: no Bash/Write/Edit by default. The default must not silently - # mutate a filesystem or push code on a submission that never asked for it. + # No Bash/Write/Edit in SDK auto-approval. This list does not hide those tools; + # applicable Cedar rules govern approval/denial when they are requested. allowed_tools: [Read, Glob, Grep, WebFetch] cedar_policy_modules: [builtin/hard_deny, builtin/soft_deny] repo_config: diff --git a/docs/verification/645-lambda-microvm-service-feedback.md b/docs/verification/645-lambda-microvm-service-feedback.md index 52a3b3de9..cbb7ad081 100644 --- a/docs/verification/645-lambda-microvm-service-feedback.md +++ b/docs/verification/645-lambda-microvm-service-feedback.md @@ -10,7 +10,7 @@ is the receipt that lets the service team find a particular call. | ID | Priority | Topic | Evidence/status | |---|---|---|---| -| F01 | High; service diagnosis open | Accepted wake ends in connection refusal | Six recorded failures; pre-freeze connection closure verified; image 6.0 correction deployed and seven Durable workflows passed | +| F01 | High; service diagnosis open | Accepted wake ends in connection refusal | Six recorded failures; pre-freeze connection closure verified; image 6.0 correction deployed and eight Durable workflows passed | | F02 | High | Supported IAM conditions and misleading permission errors | Reproduced in earlier P2 work; current service behavior needs confirmation | | F03 | Medium | A service-side hook timeline and structured failure details | Diagnostic improvement request based on F01 | | F04 | Medium | `PENDING` also means restoring an existing worker | Observed live; our timer bug is fixed | @@ -34,6 +34,10 @@ its original approval deadline. All three workflows finalized without repair. These are additional application acceptance results, not service-side traces or a measured failure rate. +The [normal repository follow-up](./645-p3-repository-path-20260917.md) also +passes clone/setup, approval sleep/wake, build/lint and existing-PR resolution +on image 6.0, with unchanged GitHub content and verified private cleanup. + **Observed:** five failures on normal images `3.0`, `4.0` and `5.0`, plus a sixth on a private image retaining 5.0's application and connection behavior with transport probes, in `us-west-2`, September 15–16. AWS accepted `ResumeMicrovm`, @@ -284,6 +288,15 @@ identifier you provide to ensure the idempotency of the request,” with a behavior. Its request schema has no per-worker tags field. This resolves the earlier documentation-access problem, not the missing service contract. +The same recheck of [ListMicrovms](https://docs.aws.amazon.com/lambda/latest/microvm-api/API_ListMicrovms.html) +and [GetMicrovm](https://docs.aws.amazon.com/lambda/latest/microvm-api/API_GetMicrovm.html) +found no returned client token, task identity, environment or per-worker tags. +List supports image/version filtering and returns the worker ID, image, +start time and state. Get adds the endpoint, execution role, connectors, +duration/idle settings and termination details. These fields can narrow an +operator's search, but shared image/role/time matches do not uniquely identify +the worker for a lost launch response. + **Impact:** if AWS created a worker but its response was lost, an operator needs to find that exact worker. An undocumented retention boundary prevents proving that a late retry cannot create another one. ABCA must not invent a new token diff --git a/docs/verification/645-p3-implementation-plan.md b/docs/verification/645-p3-implementation-plan.md index 882ed2ed7..c7c1d0955 100644 --- a/docs/verification/645-p3-implementation-plan.md +++ b/docs/verification/645-p3-implementation-plan.md @@ -6,6 +6,15 @@ Prepared 2026-09-13 from `main` `5e10038c7e28179b302ac4de78b709795aeba3ce`. Read Prerequisite work is tracked here on `fix/645-microvm-readiness`. “Completed” means implemented and checked locally; AWS deployment and live verification have separate completion gates below. +**Normal repository path (2026-09-17):** the +[repository acceptance record](./645-p3-repository-path-20260917.md) verifies +normal clone/setup, one approval sleep/wake, passing post-build/lint and +resolve-only delivery for existing PR #584 on image 6.0. Private event storage +and supported prompt configuration excluded publication; GitHub snapshots +were unchanged. The worker finalized without repair and all private +infrastructure was removed. The initial guardrail rejection is retained as +an excluded attempt. This adds the eighth successful image 6.0 Durable workflow. + **Final-image and ECS follow-up (2026-09-17):** the [new acceptance record](./645-p3-final-image-and-ecs-20260917.md) verifies repository marker persistence through two sleeps on image 6.0, unchanged @@ -381,10 +390,12 @@ handoff, use this order: shared container. Its source fix supplies the missing approval-table name to both ECS task definitions. Trace and nudge environment parity are separate existing ECS gaps; they were not silently supplied by the fixture. - The final-image repository test proves mutable files across two sleeps, - but uses an explicit temporary clone inside an artifact task. Verify the - normal repository-bound clone/delivery path separately, respecting the - target repository's publication checks. + The final-image repository test proves mutable files across two sleeps + using an explicit temporary clone inside an artifact task. The subsequent + [normal repository check](./645-p3-repository-path-20260917.md) also passes + clone/setup, approval sleep/wake, build/lint and existing-PR resolution. + New-PR publication remains covered by the earlier P2 record for its stated + image/scope; this final-image fixture deliberately publishes nothing. The [AgentCore follow-up](./645-p3-agentcore-20260916.md) now verifies its current shared container, approval/cancellation, exclusion from MicroVM sleep, reservation release and owned-session cleanup. The normal stack has no ECS diff --git a/docs/verification/645-p3-readiness-review.md b/docs/verification/645-p3-readiness-review.md index c534fabb7..98d533def 100644 --- a/docs/verification/645-p3-readiness-review.md +++ b/docs/verification/645-p3-readiness-review.md @@ -15,6 +15,13 @@ verification infrastructure was removed. Normal automatic sleep remains off until the separately listed delivery, permission/network and deployment gates are complete. Earlier paragraphs below preserve their dated findings. +The later [normal repository check](./645-p3-repository-path-20260917.md) adds +an eighth successful image 6.0 Durable workflow: normal clone/setup, +approval sleep/wake, build/lint and existing-PR resolution. It uses isolated +notification storage and changes no GitHub content. Its private infrastructure +was removed. The review also corrects stale source comments that treated SDK +auto-approval as a tool restriction and repository-optional as repository-free. + **Live update (2026-09-14):** subsequent work completed a [clean deployment](./645-p2-clean-deployment-20260913.md) and [real coding, PR iteration and cancellation tests](./645-p2-live-task-20260914.md), including Memory writes and runtime logging. A normal [image rebuild](./645-microvm-image-rebuild-20260914.md) activated version `2.0`; [11 live payload cases](./645-p2-payload-live-20260914.md) then verified transport/rejection, URL expiry/revocation and immediate Run replay. Those records supersede the corresponding gaps in the historical notes below. Full P2 acceptance and integrated P3 sleep/wake remain open; the original review findings are retained as a dated baseline. **Implementation update (2026-09-13):** subsequent local batches fix thread isolation, deletion/error/byte/contract bugs, approval heartbeat, stable MicroVM start recovery, atomic capacity reservations and coordinator metadata permissions. The latest batch implements v2 authenticated deployment manifests and single-object payload links for both ECS and MicroVM (#817/#700), with no old unsigned fallback. See the [bootstrap runbook](./645-payload-bootstrap.md) and [implementation progress](./645-p3-implementation-plan.md#implementation-progress). A further local batch removes unused logging counters in favor of structured stdout failures and verifies large registry assets through v2 delivery and the local loader. At that batch's completion, effective AWS policies, expiry/networking, stdout ingestion, remote-tool connectivity, clean deployment and P3 sleep/wake were pending; the live update above records later evidence. Findings below preserve the original reviewed baseline, rather than describing all of them as current defects. @@ -69,7 +76,7 @@ An **IAM role** is a permission badge. A **trust policy** says who may wear that |---|---|---| | P1 | Build the computer, start it, deliver a task, check it and stop it | Merged in [#689](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/689). Strategy, infrastructure, bootstrap permissions, packaging, types, `/ready` and `/run` exist. | | P2 | Make a real coding task work with configuration, permissions, logs and progress | Merged in [#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733). `/validate`, `/terminate`, warm-up, runtime grants and heartbeat support exist. The September 14 clean rerun passed coding, iteration and cancellation without manual IAM workarounds, followed by image 2.0 payload/start checks. The broader deployed recovery, effective IAM and network matrix remains open. | -| P3 | Sleep during a human approval wait, wake correctly, and keep deadlines/credentials safe | Implemented and deployed. Seven image 6.0 Durable workflows verify the core lifecycle, repository-file persistence and real expired-credential renewal. Broader delivery, permission/network and deployment gates remain open; automatic sleep stays off. | +| P3 | Sleep during a human approval wait, wake correctly, and keep deadlines/credentials safe | Implemented and deployed. Eight image 6.0 Durable workflows verify the core lifecycle, repository-file persistence, normal clone/PR resolution and real expired-credential renewal. Broader permission/network and deployment gates remain open; automatic sleep stays off. | | P4 | — | ADR-021 defines no P4. Verification runbooks have their own numbered phases; those are not extra ADR milestones. | The old unchecked checklist and the word “proposed” do not erase the merged work. Conversely, merged code is not proof that the final deployment path works unattended. diff --git a/docs/verification/645-p3-repository-path-20260917.md b/docs/verification/645-p3-repository-path-20260917.md new file mode 100644 index 000000000..d71b84897 --- /dev/null +++ b/docs/verification/645-p3-repository-path-20260917.md @@ -0,0 +1,136 @@ +# ADR-021 P3: normal repository workflow across sleep and wake + +Verified September 17, 2026, in account ``, `us-west-2`. This +extends the [final-image matrix](./645-p3-final-image-and-ecs-20260917.md) +from an explicit temporary clone to the platform's normal repository setup +and existing-PR resolution path. + +## Test boundary + +The test used normal image **6.0**, its original server as PID 1, and **8,192 +MiB**. The packaged `coding/pr-review-v1` workflow cloned +`isadeks/vercel-abca-linear` and checked out existing +[PR #584](https://github.com/isadeks/vercel-abca-linear/pull/584). + +A private coordinator ran the production Durable handler. Its temporary +configuration limited the worker to 900 seconds, the task to five turns/$1, +and the approval to 300 seconds. Automatic sleep was enabled only for this +fixture, with a 30-second delay. The shared deployment's switches stayed off. + +The test deliberately excluded publication: + +- A private event table had no DynamoDB stream or notification consumer. +- Private copies of the deployed worker/session roles substituted the session + role and event-table ARNs. The session role trusted only the private worker + role. Normal deployed roles were unchanged. +- The supported `system_prompt_overrides` setting restricted the agent to one + README read and a final response. Every tool call required approval; the + watcher approved only that exact Read. +- Normal post-hooks used the workflow's `resolve` strategy, which looks up the + existing PR without pushing. Private build/lint settings selected + `npm ci --no-audit --no-fund && npm test` and `npm run lint`; repository + configuration was unchanged. + +The normal approval handler supplied the decision using a synthetic trusted +user context. This verifies that handler and its AWS wake call, not API Gateway +authentication. Its approval/wake audit events used the normal event table; +the agent/coordinator progress and terminal events used the private table. + +This isolation matters: the GitHub notification handler posts for a task with +a repository and PR number even when its source is `api`. Its entry point +currently ignores per-task notification overrides. Separately, the PR-review +system prompt tells the agent to post reviews and comments through Bash. +Neither the `read_only` flag nor `ensure_pr(strategy: resolve)` alone disables +those publication paths. + +## Successful run + +Task: `01M2PCJYY9Z0DK7Z0F4R4QFX0Y`. +Worker: `microvm-4e50cefb-66c7-34d5-8e28-32f1766c0c68`. +Private coordinator version: **3**, code SHA-256: +`DipaZN3h7+L0uJgyWSD9KIhhjX4XvVIshvxVItsvwnM=`. + +| Event | UTC time | +|---|---| +| Task created | 00:36:00.258 | +| Worker observed running | 00:36:10.401 | +| Original approval created | 00:37:07 | +| Worker observed suspended | 00:37:38.708 | +| Normal approval API returned HTTP 202 | 00:37:51.476 | +| Worker observed running after wake | 00:37:52.616 | +| Task observed completed | 00:38:13.514 | +| Finalization verified without repair | 00:38:17.812 | + +Approval `01M2PCQE38M9NB9ZJ3NXYP6TNP` retained its original 300-second +deadline, **00:42:07**. The actual normal-handler `ResumeMicrovm` request ID +was `96e85ab7-0936-499d-91f5-399bca675488`. PID 1 completed suspend and resume +with HTTP 200, in 91 ms and 141 ms respectively. + +The independent audit passed at **00:39:09.507**. It verified: + +- The normal clone/setup path and exactly one successful Read of + `/workspace/01M2PCJYY9Z0DK7Z0F4R4QFX0Y/README.md`. +- The expected `# vercel-abca-linear` heading, passing build/test and lint + results, and resolve-only post-hooks returning the existing PR URL. +- Task `COMPLETED`, Durable execution `SUCCEEDED`, worker `TERMINATED`, + reservation released, counter zero and task payload absent. +- No watcher repair and no dropped trace events. Trace SHA-256: + `537bb8894d8da3c34308456fa8c1753c1a1a0e9a5ae8bbfaa0eae27bceb8d66f`. +- Identical GitHub snapshots before and after: head + `fd509fa63fa089df356574bdd654123c8621b12e`, branch, base, state, update + timestamp, all six comments and zero reviews. + +The reference README was 1,744 bytes with SHA-256 +`dfa6bc9de7dc213fac31c35f6b4f22717f2df1371525eeb2fa195b6dbcc7649c`. +The reference hash identifies the expected remote file; the runtime audit +checks the recorded Read and heading, not a separate guest-side file hash. + +## Excluded attempts and diagnostics + +The initial infrastructure setup encountered IAM propagation: the newly +created worker role was not yet accepted as a trust-policy principal. +The exact error was retained, and a bounded retry for that specific error +completed setup. No worker existed during this failure. + +Private coordinator version 1 was never invoked. Version 2 ran task +`01M2PC1VQ0N3X3KTY0HDC37HX1`, whose verbose test request was rejected during +PR context screening as `CONTENT/PROMPT_ATTACK (MEDIUM)`. It created no +worker and released its reservation before the watcher performed redundant +failure cleanup. Its failed Durable execution remains excluded. + +A read-only hydration comparison with the same PR and the concise request +“Read the README and report its first heading” passed the unchanged filter. +The fresh version 3 task used that wording and passed its own normal screening. +This suggests the extra test instructions contributed to the rejection; it +does not identify an exact offending span or establish deterministic classifier +behavior. No guardrail setting was disabled or weakened. + +## Evidence and remaining scope + +Raw scripts, exact bundles for all three versions, role snapshots, events, +logs, traces and GitHub comparisons are retained under +`/tmp/abca-645-p2-clean-20260913/p3-repository-path-20260917`. + +Cleanup completed at **00:40:40.868 UTC**. The private coordinator and all three +versions, three roles, switch, log group and event table were removed, along +with both owned zero counters. The archive retains 91 function log events and +24 private task events. Subsequent comparisons confirmed the normal worker +and session-role policies and repository configuration were unchanged. + +The permanent private archive is +`/Users/sphias/.local/share/abca-verification/645-p3-20260916/repository-path-evidence.tar.gz`. +It contains 60 files, is 43,934,346 bytes, and has mode `0600`. Every member +was verified against its manifest. SHA-256: +`75850ffb172f3ed5c15198dc0da1ee8c194a87a0edeb89b317ddb0b025c20a11`. + +The accompanying source review corrected comment-only claims in +`agent/src/models.py` and the default/PR-review workflow YAML files: +SDK `allowed_tools` controls auto-approval, and `requires_repo: false` makes +the repository optional. Neither setting is the stronger restriction that +the old comments described. No executable Python or workflow values changed. + +This verifies normal cloning and resolve-only delivery after wake. It does not +verify creating/pushing a new PR or publishing review comments on image 6.0. +Earlier P2 publication results retain their recorded scope. Remaining +permission/network, service-contract and deployment gates are tracked in the +[implementation plan](./645-p3-implementation-plan.md). From 9274d262baf0e6b4e72199ccdceecc1ae4a784ec Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Wed, 16 Sep 2026 21:03:52 -0400 Subject: [PATCH 069/149] docs: verify AgentCore task permission boundaries --- .../645-p3-agentcore-permissions-20260917.md | 108 ++++++++++++++++++ .../645-p3-implementation-plan.md | 16 ++- 2 files changed, 121 insertions(+), 3 deletions(-) create mode 100644 docs/verification/645-p3-agentcore-permissions-20260917.md diff --git a/docs/verification/645-p3-agentcore-permissions-20260917.md b/docs/verification/645-p3-agentcore-permissions-20260917.md new file mode 100644 index 000000000..5aed0957e --- /dev/null +++ b/docs/verification/645-p3-agentcore-permissions-20260917.md @@ -0,0 +1,108 @@ +# ADR-021 P3: actual AgentCore permission checks + +Verified September 17, 2026, in account ``, `us-west-2`. +The normal AgentCore runtime's `DEFAULT` endpoint remained **READY**, version +**5**. This extends the earlier +[AgentCore approval/cancellation checks](./645-p3-agentcore-20260916.md) +with real permission requests from that runtime. + +## Test and results + +Task `01M2PDAMMMMH8GMSS0YJP2J9R6` ran one exact Python diagnostic through +the agent's Bash tool. It had no repository or notification destination, +a four-turn/$1 limit, and an eight-minute watcher limit. A private Lambda +ran the production Durable handler. Its code SHA-256 was +`T4bq5saVKeGxQd0yUHN3TRCPKTrqUNH+TJHVT2YyVtc=`. + +The diagnostic removed inherited static credential overrides in its child +process before loading the production `aws_session` providers. This made +the ambient checks use AgentCore's container credential provider. Actual STS +identity requests verified the normal runtime role and the task session role. +No credential values were printed or saved. + +The independent audit passed at **00:55:43.375 UTC**: + +| Requests | Count | Result | +|---|---:|---| +| Ambient and scoped STS identities | 2 | Expected roles | +| Ambient task/counter reads | 2 | Denied | +| Ambient artifact read/list | 2 | Denied | +| Scoped own task / other owned task / counter reads | 3 | Allowed / denied / denied | +| Scoped reporting update with deliberately false condition | 1 | Authorized shape; condition rejected | +| Scoped writes to owner, compute handle, start receipt, reservation and lifecycle fields | 5 | Denied | +| Scoped own artifact write/read | 2 | Allowed / denied | +| Scoped write under another owned task's artifact prefix | 1 | Denied | +| Scoped MicroVM bootstrap-object read | 1 | Denied | + +All **19** checks retained actual AWS request IDs. Artifact access is +deliberately write-only in the deployed session role; the read denial is +expected. Every DynamoDB update used a false condition, so even unexpected +authorization could not change an existing task row. The other owned task's +complete record was unchanged. + +The audit verified exactly one Bash call matching the prepared command, one +successful tool result, zero approval records and zero dropped trace events. +The task completed, and its Durable execution succeeded. No MicroVM SDK call, +start receipt or lifecycle state appeared. + +Permission-result SHA-256: +`07237297d088b12743b5237a8655bfd5fca0cc072f84bddbdec3d15fa1913fdc`. +Trace SHA-256: +`9dc90a5df6077b2d27ef777d88400c505cb6c5e75dd2d634a12cb9599a50b377`. + +These checks establish the observed grants for the issued credentials. +They do not prove isolation against a compromised worker choosing different +tags when assuming the session role. + +## Watcher timing error + +The original watcher failed after the diagnostic had completed. It read +the task while its slot was still held, then read the newer Durable status +`SUCCEEDED`, and incorrectly asserted against the older task snapshot. + +| Evidence | UTC time | +|---|---| +| Actual reservation release | 00:53:28.261 | +| Durable `finalize` step succeeded | 00:53:28.301 | +| Durable execution succeeded | 00:53:28.318 | +| Watcher's stale-snapshot assertion failed | 00:53:28.391 | + +The fallback invoked the idempotent release helper after release had already +completed. It also explicitly stopped the owned AgentCore session. The +independent audit records this watcher failure and verifies the original +permission results separately; it does not relabel the watcher as passing. + +The original script is retained as `run-executed.cjs`. The corrected watcher +rereads the task after observing Durable completion. That corrected script +was syntax-checked but was not used to rerun the already completed AWS checks. + +## Cleanup and retained evidence + +Session `adbb6743-2a03-4161-9f87-f206aa738088` was stopped at +**00:53:40.358**, request ID `9077a23c-f13b-47ca-883f-26dac1c066c9`, +and subsequently returned `ResourceNotFoundException`. + +Private infrastructure cleanup was verified at **00:58:36.593**. The +coordinator and all versions, role, switch, log group and zero counter were +removed. Both harmless probe objects were archived and removed; the negative +write created no object. All 32 private function log events were retained. +Normal task/trace records follow their existing retention policy. + +The initial immediate absence check briefly still saw the deleted Lambda. +A subsequent bounded, read-only check verified absence. The original cleanup +script and failure, the follow-up verification, and the corrected future +cleanup script are retained. + +The normal runtime endpoint's full before/after snapshots and ambient-role +policies were identical. Normal MicroVM automatic sleep stayed off. +Evidence directory: +`/tmp/abca-645-p2-clean-20260913/p3-agentcore-permissions-20260917`. + +The permanent private archive is +`/Users/sphias/.local/share/abca-verification/645-p3-20260916/agentcore-permissions-evidence.tar.gz`. +It contains **45 files**, **14,912,232 bytes**, with mode `0600`; every member's +hash was checked against the manifest. Archive SHA-256: +`34123cd0d4599598a0489d9716d34ab86ccb97c0c5944b53072dbf33f21aa2fc`. + +Runtime ingress, remote MCP connectivity and the remaining deployment/service +gates remain in the [implementation plan](./645-p3-implementation-plan.md). diff --git a/docs/verification/645-p3-implementation-plan.md b/docs/verification/645-p3-implementation-plan.md index c7c1d0955..e0884f13a 100644 --- a/docs/verification/645-p3-implementation-plan.md +++ b/docs/verification/645-p3-implementation-plan.md @@ -6,6 +6,13 @@ Prepared 2026-09-13 from `main` `5e10038c7e28179b302ac4de78b709795aeba3ce`. Read Prerequisite work is tracked here on `fix/645-microvm-readiness`. “Completed” means implemented and checked locally; AWS deployment and live verification have separate completion gates below. +**AgentCore effective permissions (2026-09-17):** the +[actual runtime probe](./645-p3-agentcore-permissions-20260917.md) passes 19 +ambient/scoped AWS permission checks on normal runtime version 5. The +independent audit verifies the exact tool command and AWS receipts, and +separately records a watcher snapshot race after normal finalization. +Its session, private infrastructure and harmless probe objects were removed. + **Normal repository path (2026-09-17):** the [repository acceptance record](./645-p3-repository-path-20260917.md) verifies normal clone/setup, one approval sleep/wake, passing post-build/lint and @@ -206,7 +213,8 @@ is superseded by these records. - [x] Verify MicroVM manifest/download transport, malformed or mismatched inputs, URL expiry/revocation, a foreign private-bucket denial and >1 MiB transport in AWS; verify concurrent/repeated/conflicting S3 preparation with operator credentials. - [x] Verify deployed MicroVM-role metadata/S3 permissions, public-object denial and actual signer-credential expiry in AWS; see the [effective IAM evidence](./645-effective-iam-20260915.md) for scope. - [x] Fix ECS approval-table configuration and verify current-container approval/cancellation, v2 payload cleanup, ambient/scoped permissions and port 443/80 controls in a bounded deployment; remove its infrastructure. -- [ ] Complete the remaining AgentCore permission checks, runtime ingress/remote-MCP paths and coordinated-rollout matrix in AWS. +- [x] Verify 19 actual ambient/scoped permission requests from normal AgentCore runtime 5 and remove the private fixture. +- [ ] Complete runtime ingress/remote-MCP paths and the coordinated-rollout matrix in AWS. - [x] Implement saved MicroVM start receipts, stable tokens, input fingerprints and handle recovery. - [x] Verify immediate identical `RunMicrovm` replay returns the same worker ID in the live payload probes. - [x] Verify simultaneous identical Run calls, changed-parameter rejection, and replay after termination through roughly five minutes against AWS; distinguish cached Run responses from fresh VM state. @@ -382,8 +390,10 @@ handoff, use this order: worker whose ID never reached coordinator state. The latter required an exact task/worker pair in the guest log; it does not establish post-retention behavior or a recovery path when those logs are unavailable. -3. Complete the remaining AgentCore permission and runtime/network negatives, - plus remote-MCP connectivity. The +3. Complete the remaining runtime/network negatives and remote-MCP connectivity. + The [AgentCore permission probe](./645-p3-agentcore-permissions-20260917.md) + now passes 19 actual requests on normal runtime version 5, including ambient + denial, scoped task access and protected-field denials. The [ECS follow-up](./645-p3-final-image-and-ecs-20260917.md) now verifies real ambient/scoped permission requests, port 443 success versus port 80 denial, approval/cancellation, payload deletion and stopped workers on the current From 6ba7fd92c9e0288f052d0f99019fe09549a466ec Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Wed, 16 Sep 2026 21:31:49 -0400 Subject: [PATCH 070/149] fix: clarify dispatch feedback and verify remote MCP wake path --- cdk/src/constructs/lambda-microvm-compute.ts | 5 + cdk/src/handlers/confirm-uploads.ts | 6 +- .../strategies/lambda-microvm-strategy.ts | 6 +- .../verification/645-capacity-reservations.md | 9 + .../645-p3-implementation-plan.md | 20 ++- .../645-p3-mcp-network-20260917.md | 160 ++++++++++++++++++ .../645-p3-normal-rollout-review-20260917.md | 125 ++++++++++++++ docs/verification/645-p3-readiness-review.md | 12 +- 8 files changed, 335 insertions(+), 8 deletions(-) create mode 100644 docs/verification/645-p3-mcp-network-20260917.md create mode 100644 docs/verification/645-p3-normal-rollout-review-20260917.md diff --git a/cdk/src/constructs/lambda-microvm-compute.ts b/cdk/src/constructs/lambda-microvm-compute.ts index 89cc049f0..eb9ceb39b 100644 --- a/cdk/src/constructs/lambda-microvm-compute.ts +++ b/cdk/src/constructs/lambda-microvm-compute.ts @@ -359,6 +359,11 @@ const CPU_ARCHITECTURE = 'ARM_64'; * control**, not an omission: the strategy passes the `NO_INGRESS` connector on * every `RunMicrovm`. The ARN shape is taken verbatim from that observed * `HTTP_INGRESS` ARN, with the connector name swapped. + * + * A service endpoint URL is still returned with `NO_INGRESS` (verified + * 2026-09-17). Its existence does not prove that requests reach the guest: + * endpoint requests require a MicroVM auth token. An unauthenticated 403 tests + * that authentication boundary, not the connector's handling of valid tokens. */ export const MICROVM_NO_INGRESS_CONNECTOR_RESOURCE = 'aws-network-connector:NO_INGRESS'; diff --git a/cdk/src/handlers/confirm-uploads.ts b/cdk/src/handlers/confirm-uploads.ts index be3758055..35bcdb127 100644 --- a/cdk/src/handlers/confirm-uploads.ts +++ b/cdk/src/handlers/confirm-uploads.ts @@ -515,7 +515,7 @@ async function transitionToSubmitted( }); } catch (orchErr) { orchestratorInvokeFailed = true; - logger.error('Failed to invoke orchestrator after confirm-uploads — task will be picked up by StrandedTaskReconciler', { + logger.error('Orchestrator dispatch could not be confirmed after confirm-uploads; stranded-task cleanup fails tasks that remain stuck', { error: orchErr instanceof Error ? orchErr.message : String(orchErr), task_id: taskId, request_id: requestId, @@ -533,8 +533,8 @@ async function transitionToSubmitted( }; const responseBody = toTaskDetail(updatedTask); if (orchestratorInvokeFailed) { - (responseBody as any).warning = 'Task was submitted successfully but orchestration dispatch failed. ' + - 'The task will be picked up automatically within minutes by the background reconciler.'; + (responseBody as any).warning = 'Task was submitted, but starting the worker could not be confirmed. ' + + 'Check task status before retrying. Background cleanup marks tasks that remain stuck as failed.'; } return successResponse(200, responseBody, requestId); } diff --git a/cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts b/cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts index 1cf80c93e..9d4ab2e4b 100644 --- a/cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts +++ b/cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts @@ -504,8 +504,10 @@ export class LambdaMicrovmComputeStrategy implements ComputeStrategy { // Explicit ingress control (F7, live 2026-07-31): `RunMicrovm` does NOT // default to "no ingress" — omitting the field attaches the AWS-managed - // PUBLIC `HTTP_INGRESS` connector and mints a public - // `*.lambda-microvm..on.aws` endpoint. So the field is ALWAYS sent. + // PUBLIC `HTTP_INGRESS` connector. So the field is ALWAYS sent. + // A service endpoint URL is also returned with `NO_INGRESS`; the URL alone + // does not establish guest reachability. Requests require a MicroVM auth + // token, and an unauthenticated 403 proves only that authentication check. // // The env var is unconditional in every CDK-deployed stack (its prop is // required), so in practice this always takes the `configuredIngress` branch diff --git a/docs/verification/645-capacity-reservations.md b/docs/verification/645-capacity-reservations.md index ec583e687..7824f0e85 100644 --- a/docs/verification/645-capacity-reservations.md +++ b/docs/verification/645-capacity-reservations.md @@ -73,6 +73,15 @@ permanent archive, which also contains this writer inventory. The subsequent [feedback-only rollout](./645-p3-wake-feedback-20260916.md) advanced the normal coordinator to version 9; it did not change the capacity protocol. +The [September 17 normal-rollout review](./645-p3-normal-rollout-review-20260917.md) +extends the inventory through coordinator version 10 and identifies all nine +actual admission producers. The installation was idle at observation and already +used task-owned reservations in its clean-deployment baseline. A live admission +fence was not applied. Its asynchronous processors have no failure-retention +destination; setting their reserved concurrency to zero can discard input +instead of holding it for later. Follow the retained-input and image-rollback +steps in that review before a normal rollout rehearsal. + Local tests prove the application requests and DynamoDB Local's transaction behavior. They do not establish deployed IAM, AWS scaling, successful rollout or MicroVM sleep/wake behavior. Terminal events may repeat or be lost independently of the atomic seat update. The reservation/start markers share the task row. Subsequent prerequisite work restricts agent updates to reporting/approval attributes and removes whole-row replacement/deletion plus direct worker access to the counter. Public-API omission alone was not protection. See [coordinator metadata verification](./645-coordinator-metadata.md) for the writer inventory, actual policy boundary, remaining status/tag trust limits and required AWS authorization checks. These local transaction tests do not prove that security boundary. diff --git a/docs/verification/645-p3-implementation-plan.md b/docs/verification/645-p3-implementation-plan.md index e0884f13a..a9675a39a 100644 --- a/docs/verification/645-p3-implementation-plan.md +++ b/docs/verification/645-p3-implementation-plan.md @@ -214,7 +214,8 @@ is superseded by these records. - [x] Verify deployed MicroVM-role metadata/S3 permissions, public-object denial and actual signer-credential expiry in AWS; see the [effective IAM evidence](./645-effective-iam-20260915.md) for scope. - [x] Fix ECS approval-table configuration and verify current-container approval/cancellation, v2 payload cleanup, ambient/scoped permissions and port 443/80 controls in a bounded deployment; remove its infrastructure. - [x] Verify 19 actual ambient/scoped permission requests from normal AgentCore runtime 5 and remove the private fixture. -- [ ] Complete runtime ingress/remote-MCP paths and the coordinated-rollout matrix in AWS. +- [x] Verify actual remote MCP calls, runtime HTTPS/port-80 behavior and unauthenticated endpoint denial before/after sleep on image 6.0; retain observer limitations. +- [ ] Complete the installation-specific coordinated-rollout matrix in AWS. - [x] Implement saved MicroVM start receipts, stable tokens, input fingerprints and handle recovery. - [x] Verify immediate identical `RunMicrovm` replay returns the same worker ID in the live payload probes. - [x] Verify simultaneous identical Run calls, changed-parameter rejection, and replay after termination through roughly five minutes against AWS; distinguish cached Run responses from fresh VM state. @@ -390,7 +391,14 @@ handoff, use this order: worker whose ID never reached coordinator state. The latter required an exact task/worker pair in the guest log; it does not establish post-retention behavior or a recovery path when those logs are unavailable. -3. Complete the remaining runtime/network negatives and remote-MCP connectivity. +3. Runtime networking and remote-MCP connectivity now pass the + [September 17 independent audit](./645-p3-mcp-network-20260917.md): + real documentation-tool calls before/after a 60-second sleep, HTTPS success + versus port-80 timeout on the same IP, and unauthenticated endpoint 403s while + the worker was running. This adds the ninth successful image 6.0 Durable + workflow. The record retains the excluded discovery-policy attempt and + watcher bookkeeping failure; product finalization preceded its fallback. + Valid-token ingress and exact TCP socket reuse were not tested. The [AgentCore permission probe](./645-p3-agentcore-permissions-20260917.md) now passes 19 actual requests on normal runtime version 5, including ambient denial, scoped task access and protected-field denials. The @@ -426,6 +434,14 @@ handoff, use this order: production migration or arbitrary retention scale. A subsequent measurement found 86 task rows and 36 counter rows in the normal development deployment, below the fixture's 600 rows per table. + The [September 17 normal-rollout review](./645-p3-normal-rollout-review-20260917.md) + now inventories all nine actual producers and retained coordinator versions + 2–10. It found 112 terminal tasks, no held reservations and 36 zero counters. + The clean installation already used the current reservation protocol. + Admission was not fenced: its async processors lack failure retention, so + setting their concurrency to zero could lose incoming work. A safe retained + input/replay procedure remains necessary. Coordinator rollback also requires + an explicit image plan because the managed runtime selects latest active. 5. Perform the final compatible rollout, including shared runtime changes for ECS/AgentCore, pinned-version retention and rollback checks. Enable automatic suspension only after the remaining gates pass, then finish the ADR/runbook diff --git a/docs/verification/645-p3-mcp-network-20260917.md b/docs/verification/645-p3-mcp-network-20260917.md new file mode 100644 index 000000000..d293b32de --- /dev/null +++ b/docs/verification/645-p3-mcp-network-20260917.md @@ -0,0 +1,160 @@ +# P3 remote MCP and runtime networking — September 17, 2026 + +The independent audit passed on normal MicroVM image **6.0**, with the original +PID 1 server and **8192 MiB**. A real remote documentation tool worked before +and after suspension. Runtime HTTPS connections worked in both phases, port 80 +timed out, and unauthenticated requests to the worker endpoint returned 403. + +The original watcher did not pass: it missed its after-wake ingress observation. +That observation was collected independently while the worker was running. +The original tool results and Durable history establish the results below. + +## Scope + +Account ``, Region `us-west-2`, profile `sphia-dev`. +Task `01M2PENMJXNHESFSP95ECR8MMD`, worker +`microvm-1685f819-8853-377e-a764-4e409211e19b`. + +A private coordinator ran the production Durable handler and normal repository +pipeline against existing PR 584 in `isadeks/vercel-abca-linear`. Its version 2 +code hash was `GZhMPmybPf3//lClZWOkomI9BzA2I63mFEgftlFk8Lc=`. +The fixture capped the worker at 900 seconds and the agent at ten turns/$2. + +The wrapper supplied one explicit resolved MCP asset to the normal payload +transport and project loader: + +```json +{ + "kind": "mcp_server", + "namespace": "verification", + "name": "aws_knowledge", + "version": "1.0.0", + "runtime": { + "transport": "http", + "url": "https://knowledge-mcp.global.api.aws" + } +} +``` + +This tests the actual Claude CLI/SDK remote-tool path and project configuration. +It does not claim that a registry lookup occurred in this fixture. The normal +registry was unchanged. The official public AWS Knowledge MCP endpoint required +no credentials. + +Private copies of the deployed worker/session policies redirected TaskEvents to +a private table without a notification consumer. A supported prompt override +and locally verified Cedar rules allowed only the exact network command, the +read-only documentation tool and tool discovery. The watcher approved only the +exact README Read. The normal repository, role policies and PR were unchanged. + +## Actual requests before and after sleep + +The agent made five sequential substantive calls: network diagnostic, remote +documentation read, approved README Read, a new remote documentation read, and +the same network diagnostic. It also used `ToolSearch` once to load the deferred +MCP tool definition. + +Both MCP calls invoked +`mcp__verification__aws_knowledge__aws___read_documentation` with the same +GetMicrovm documentation URL and `max_length: 2500`. + +| Check | Before sleep | After wake | +|---|---|---| +| Remote MCP tool | `SUCCESS`, 01:14:02.596 UTC | `SUCCESS`, 01:15:49.755 UTC | +| TCP 443 to `docs.aws.amazon.com`, IP `52.85.31.75` | Connected, TLS 1.3 | Connected, TLS 1.3 | +| TCP 80 to that same IP | Timeout after 4.02 seconds | Timeout after 4.02 seconds | +| Unauthenticated `GET /ping` on the worker endpoint | HTTP 403 | HTTP 403 | + +The laptop control connected to both ports on that same IP. Actual EC2 rules +for runtime security group `sg-07217b11d74ddc22f` allowed only IPv4 TCP 443 +outbound, with no inbound rules. Cloud Control confirmed the active runtime +connector used that group. Its rules were identical after the test. + +Both endpoint checks were bracketed by `GetMicrovm` returning `RUNNING`. +The exact response was **“Request missing authentication.”** Current +[networking documentation](https://docs.aws.amazon.com/lambda/latest/dg/microvms-networking.html) +requires an `X-aws-proxy-auth` token for endpoint requests. Thus these checks +prove unauthenticated denial. They do not establish how `NO_INGRESS` handles a +valid token; no token was minted. The returned endpoint URL alone does not prove +guest reachability. + +The output scanner redacted part of the public documentation response. Both +explicit `SUCCESS` results and exact tool inputs remained available. This +redaction does not indicate a transport failure. + +## Sleep, wake and completion + +Approval `01M2PEV9TFC74GEAQHVR5VGR71` retained its original creation time +**01:14:11** and 300-second timeout. The watcher observed `SUSPENDED` at +**01:14:42.966** and called the normal approval handler after more than 60 seconds. +The actual `ResumeMicrovm` request ID was +`60c6af18-966f-4442-a308-f6db63465b86`. + +PID 1 acknowledged `/suspend` with HTTP 200 in **84 ms**, and `/resume` with +HTTP 200 in **149 ms**. The same worker resumed and performed the remaining +calls. This establishes successful remote-client use across the freeze; it does +not prove that the underlying TCP socket was reused. + +| Completion evidence | UTC time | +|---|---| +| Original task reservation released | 01:16:40.183 | +| Durable `finalize` step succeeded | 01:16:40.383 | +| Durable execution succeeded | 01:16:40.399 | +| Worker observed `TERMINATED` | 01:16:41.999 | +| Watcher assertion failed | 01:16:44.459 | + +The task completed with passing build/lint, existing-PR resolution, zero counter +and absent launch payload. PR head, branch, state, update time, six comments and +zero reviews remained unchanged. + +The independent audit passed at **01:18:52.171**. It verified exact tool order +and inputs, real tool results, original approval timing, request IDs and the +completion timeline. Trace SHA-256: +`439f4a120ee5bb409e70837ed14e28f9728bb9597b6fc7e2c253f750d4a201f1`; +zero trace events were dropped. + +## Fixture and observer failures retained + +The first task, `01M2PEC4NZHE3G1PNZ27GCRAVT`, used private coordinator version 1. +Its first network diagnostic passed, but the fixture had not permitted +`ToolSearch`, which CLI 2.1.191 requires for deferred MCP discovery. The watcher +rejected that unexpected approval, stopped the execution, terminated the worker +and released its slot. No MCP call or suspension occurred. This attempt is +excluded from wake acceptance. + +Version 2 permitted read-only discovery. Its watcher later searched for the +approval using `awaiting_approval_request_id`, which production correctly clears +after approval. It therefore skipped the after-wake ingress observation and +eventually asserted `1 !== 2`. An independent observer had already captured the +second 403 at **01:16:35.621–01:16:36.197**, while the worker remained running. + +The fallback called idempotent payload-deletion and reservation-release helpers +after product finalization. No execution stop or worker termination request was +needed. The independent audit verifies the earlier completion instead of +relabeling this watcher as passing. The original scripts and failures are +retained; the corrected future watcher was syntax-checked, not rerun. + +## Cleanup and archive + +All private infrastructure was removed by **01:21:57.979**: coordinator and +both versions, three roles, switch, log group, event table and two zero counters. +Both workers were terminated and their launch payload prefixes empty. +Normal task/trace records retain their existing retention policy. + +The first cleanup exhausted its 45-second polling window after AWS accepted +`DeleteFunction`. A later read returned `ResourceNotFoundException`; the +follow-up verified completed deletions and removed the remaining table/roles. +The original cleanup failure and subsequent verification are retained. +All 142 private function log events and 58 private TaskEvents were archived. + +Normal runtime role trust/policies and firewall rules were unchanged. Normal +image 6.0 remained active, and normal automatic suspension remained disabled. + +Permanent private archive: +`/Users/sphias/.local/share/abca-verification/645-p3-20260916/mcp-network-evidence.tar.gz`. +It contains **85 files**, **29,507,839 bytes**, mode `0600`, with every member +verified against its hash manifest. Archive SHA-256: +`68b806d250e6f7c829a021d8a1b7b9fb6d53c3819f18fed535b87eb7c23e2a58`. + +The [implementation plan](./645-p3-implementation-plan.md) retains the remaining +deployment and service-contract gates. diff --git a/docs/verification/645-p3-normal-rollout-review-20260917.md b/docs/verification/645-p3-normal-rollout-review-20260917.md new file mode 100644 index 000000000..6f7e826ef --- /dev/null +++ b/docs/verification/645-p3-normal-rollout-review-20260917.md @@ -0,0 +1,125 @@ +# P3 normal deployment rollout review — September 17, 2026 + +This was a read-only review of `backgroundagent-dev` in account ``, +Region `us-west-2`. No admission settings, aliases, roles or task records were +changed. It identifies the remaining operational work; it is not a completed +normal deployment drain or rollback rehearsal. + +## Observed installation + +At **01:27:41.617 UTC**: + +- Normal coordinator versions **2–10** had no running Durable executions. +- All **112 task rows** were terminal; none held a capacity reservation. +- All **36 counter rows** had `active_count: 0`. +- Nine deployed functions referenced the normal coordinator's `live` alias. +- None of those nine had a configured dead-letter queue or asynchronous + on-failure destination. +- The normal coordinator alias remained on version **10**. Automatic MicroVM + suspension remained disabled. + +The initial clean-deployment source, `29dcaa74`, already contained +`acquireTaskSlot`, `reservation_version` and task-owned release markers. +This installation does not require conversion from the older counter protocol. +The [isolated legacy/current rehearsal](./645-p3-capacity-upgrade-20260916.md) +remains evidence for installations that do need that migration. +The [600-user scan check](./645-p3-capacity-scan-20260916.md) exceeds this +installation's observed table volume, without establishing arbitrary scale. + +An idle snapshot is not a fence: new work could arrive immediately afterward. + +## Actual admission producers + +Source callers and deployed `ORCHESTRATOR_FUNCTION_ARN` configuration identify +these nine entry points: + +| Producer | Work it can start | +|---|---| +| TaskApi CreateTask | Direct submissions | +| TaskApi WebhookCreateTask | Webhook submissions | +| TaskApi ConfirmUploads | Tasks whose uploads become ready | +| AdmissionQueuePickup | Previously queued tasks | +| Slack CommandProcessor | Slack submissions | +| Linear WebhookProcessor | Linear submissions and follow-ups | +| Jira WebhookProcessor | Jira submissions and follow-ups | +| OrchestrationReconciler | Released dependent/child tasks | +| StrandedOrchestrationReconciler | Recovery that can release child tasks | + +Stopping only the public create-task route does not stop the other eight. +Stopping the coordinator itself would also prevent existing Durable executions +from continuing through cleanup, so that is not a suitable drain mechanism. + +## Two unsafe shortcuts + +Slack, Linear and Jira use asynchronous Lambda processor invocation. +[AWS documents](https://docs.aws.amazon.com/lambda/latest/dg/invocation-async-retain-records.html) +that setting reserved concurrency to zero sends new asynchronous events +directly to a configured dead-letter queue or failure destination **without +retries**. The observed functions have neither configured. The zero-concurrency +method used on private fixture functions must not be copied blindly to these +normal processors. + +Denying the producers' coordinator-invocation permission is also insufficient. +`createTaskCore` and upload confirmation persist `SUBMITTED` before invoking +the coordinator. A denied or uncertain invocation can leave a submitted task +without a running execution. The stranded-task reconciler eventually marks +stuck tasks failed; it does not start them. + +The upload-confirmation log and user warning previously promised automatic +pickup. The local source now explains that startup could not be confirmed, +advises checking status before retrying, and describes failed cleanup of stuck +tasks. The existing 13 upload-confirmation tests, compilation and lint pass. +This text correction has not been deployed. + +## Image rollback is separate from coordinator rollback + +Normal coordinator versions 9 and 10 both name the image +`backgroundagent-dev-abca-agent` without `MICROVM_IMAGE_VERSION`. +The managed-image construct deliberately selects the latest active version: +the latest-active attribute can be empty during the initial image build. + +Consequently, moving the coordinator alias from 10 to 9 does **not** restore an +older MicroVM image. Once a worker starts, its saved handle records the actual +returned image version, but that does not select the version for future starts. + +Before rehearsing rollback, make image selection explicit in the reviewed +rollout procedure. The existing external-image path supports a version pin, +but switching an existing managed image resource to that path is an +infrastructure migration and must be reviewed for deletion/replacement. +Do not remove a managed image merely to obtain a pin. + +## Remaining execution sequence + +1. Prepare an admission pause that retains asynchronous input and has a tested + replay procedure. Inventory the receivers and scheduled/stream triggers + feeding all nine producers. Verify retention and restoration before relying + on a pause. Keep coordinator continuations, approvals, cancellation and + task finalization available. +2. Choose and record compatible coordinator **and image** rollback targets. + Preserve the current exact template, code hashes, image identity, role + policies, live switch and original producer/trigger settings. +3. Exercise the pause, then wait for already-running producer invocations and + previously accepted dispatches to settle. Repeatedly inventory all retained + coordinator versions, task states, compute handles and reservations. + Account explicitly for queued tasks, pending uploads and dependent tasks. +4. With admission fenced and the drain verified, run reconciliation and compare + held reservations with counters. This installation uses the current protocol; + do not introduce old writers merely to simulate a legacy migration. +5. Exercise the reviewed compatible rollout/rollback/restore sequence. Verify + actual returned worker image versions, completion and task-owned release. + Restore input delivery, replay retained input through its normal deduplication + path, and verify that no accepted work was lost. +6. Include the upload-feedback correction in the controlled deployment. Keep + automatic suspension disabled until the remaining applicable rollout and + service-contract gates in the [P3 plan](./645-p3-implementation-plan.md) close. + +Raw read-only evidence is in +`/tmp/abca-645-p2-clean-20260913/p3-normal-drain-20260917`. +It contains exact function names/hashes, asynchronous settings, version +inventories, table scans and the AWS documentation used for this review. + +Permanent private archive: +`/Users/sphias/.local/share/abca-verification/645-p3-20260916/normal-rollout-review-evidence.tar.gz`. +It contains nine files, 34,608 bytes, mode `0600`; every member was checked +against its hash manifest. Archive SHA-256: +`4a1e4dfd93b78c85463fdad590fe8cf93547cb01733d190080e5b8153ec65b54`. diff --git a/docs/verification/645-p3-readiness-review.md b/docs/verification/645-p3-readiness-review.md index 98d533def..014b87b19 100644 --- a/docs/verification/645-p3-readiness-review.md +++ b/docs/verification/645-p3-readiness-review.md @@ -22,6 +22,16 @@ notification storage and changes no GitHub content. Its private infrastructure was removed. The review also corrects stale source comments that treated SDK auto-approval as a tool restriction and repository-optional as repository-free. +The [remote MCP/network check](./645-p3-mcp-network-20260917.md) adds a ninth +successful image 6.0 Durable workflow. Actual remote tool calls and HTTPS +connections worked before/after sleep; port 80 timed out in both phases. +Unauthenticated endpoint requests returned 403 with the worker running. +The independent audit retains the fixture/observer failures and verifies +product finalization before fallback cleanup. All private resources were removed. +The [AgentCore permission check](./645-p3-agentcore-permissions-20260917.md) +also passes 19 actual ambient/scoped requests. Normal rollout and service-contract +gates remain open. + **Live update (2026-09-14):** subsequent work completed a [clean deployment](./645-p2-clean-deployment-20260913.md) and [real coding, PR iteration and cancellation tests](./645-p2-live-task-20260914.md), including Memory writes and runtime logging. A normal [image rebuild](./645-microvm-image-rebuild-20260914.md) activated version `2.0`; [11 live payload cases](./645-p2-payload-live-20260914.md) then verified transport/rejection, URL expiry/revocation and immediate Run replay. Those records supersede the corresponding gaps in the historical notes below. Full P2 acceptance and integrated P3 sleep/wake remain open; the original review findings are retained as a dated baseline. **Implementation update (2026-09-13):** subsequent local batches fix thread isolation, deletion/error/byte/contract bugs, approval heartbeat, stable MicroVM start recovery, atomic capacity reservations and coordinator metadata permissions. The latest batch implements v2 authenticated deployment manifests and single-object payload links for both ECS and MicroVM (#817/#700), with no old unsigned fallback. See the [bootstrap runbook](./645-payload-bootstrap.md) and [implementation progress](./645-p3-implementation-plan.md#implementation-progress). A further local batch removes unused logging counters in favor of structured stdout failures and verifies large registry assets through v2 delivery and the local loader. At that batch's completion, effective AWS policies, expiry/networking, stdout ingestion, remote-tool connectivity, clean deployment and P3 sleep/wake were pending; the live update above records later evidence. Findings below preserve the original reviewed baseline, rather than describing all of them as current defects. @@ -76,7 +86,7 @@ An **IAM role** is a permission badge. A **trust policy** says who may wear that |---|---|---| | P1 | Build the computer, start it, deliver a task, check it and stop it | Merged in [#689](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/689). Strategy, infrastructure, bootstrap permissions, packaging, types, `/ready` and `/run` exist. | | P2 | Make a real coding task work with configuration, permissions, logs and progress | Merged in [#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733). `/validate`, `/terminate`, warm-up, runtime grants and heartbeat support exist. The September 14 clean rerun passed coding, iteration and cancellation without manual IAM workarounds, followed by image 2.0 payload/start checks. The broader deployed recovery, effective IAM and network matrix remains open. | -| P3 | Sleep during a human approval wait, wake correctly, and keep deadlines/credentials safe | Implemented and deployed. Eight image 6.0 Durable workflows verify the core lifecycle, repository-file persistence, normal clone/PR resolution and real expired-credential renewal. Broader permission/network and deployment gates remain open; automatic sleep stays off. | +| P3 | Sleep during a human approval wait, wake correctly, and keep deadlines/credentials safe | Implemented and deployed. Nine image 6.0 Durable workflows verify the core lifecycle, repository-file persistence, normal clone/PR resolution, real expired-credential renewal and remote MCP/network behavior. Normal rollout and service-contract gates remain open; automatic sleep stays off. | | P4 | — | ADR-021 defines no P4. Verification runbooks have their own numbered phases; those are not extra ADR milestones. | The old unchecked checklist and the word “proposed” do not erase the merged work. Conversely, merged code is not proof that the final deployment path works unattended. From a2dc0c908cc18926900a466b2d30ee9b8b428dc9 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Thu, 17 Sep 2026 09:22:54 -0400 Subject: [PATCH 071/149] feat: nest MicroVM compute and complete approval cancellation feedback --- agent/src/hooks.py | 13 + agent/tests/test_hooks.py | 38 +++ cdk/bootstrap/BOOTSTRAP_HASH | 2 +- cdk/bootstrap/BOOTSTRAP_VERSION | 2 +- cdk/bootstrap/bootstrap-template.yaml | 10 +- .../policies/compute-lambda-microvm.json | 4 +- cdk/scripts/README.md | 6 +- cdk/scripts/package-microvm-artifact.sh | 30 +- .../policies/compute-lambda-microvm.ts | 7 +- cdk/src/bootstrap/version.ts | 5 +- cdk/src/constructs/fanout-consumer.ts | 7 + cdk/src/constructs/lambda-microvm-compute.ts | 64 +++-- cdk/src/constructs/lambda-microvm-stack.ts | 67 +++++ cdk/src/constructs/slack-integration.ts | 8 + cdk/src/constructs/task-api.ts | 11 +- cdk/src/handlers/cancel-task.ts | 68 ++--- cdk/src/handlers/fanout-task-events.ts | 111 ++++---- cdk/src/handlers/get-pending.ts | 98 +++++-- .../handlers/shared/approval-notifications.ts | 171 +++++++++++ .../handlers/shared/deny-reason-scanner.ts | 4 +- cdk/src/handlers/shared/microvm-lifecycle.ts | 2 +- cdk/src/handlers/shared/response.ts | 1 + cdk/src/handlers/shared/slack-blocks.ts | 2 +- cdk/src/handlers/shared/task-cancellation.ts | 175 ++++++++++++ cdk/src/handlers/shared/types.ts | 25 +- cdk/src/handlers/slack-interactions.ts | 44 ++- cdk/src/handlers/slack-notify.ts | 20 +- cdk/src/stacks/agent.ts | 79 ++++-- .../__snapshots__/version.test.ts.snap | 2 +- cdk/test/bootstrap/bootstrap-template.test.ts | 2 + cdk/test/bootstrap/policies.test.ts | 18 ++ cdk/test/constructs/fanout-consumer.test.ts | 20 ++ .../constructs/lambda-microvm-compute.test.ts | 8 +- .../constructs/lambda-microvm-stack.test.ts | 170 +++++++++++ cdk/test/constructs/slack-integration.test.ts | 25 ++ .../constructs/task-api-microvm-wake.test.ts | 21 ++ cdk/test/handlers/cancel-task.test.ts | 23 ++ cdk/test/handlers/fanout-task-events.test.ts | 151 +++++++--- cdk/test/handlers/get-pending.test.ts | 137 ++++++++- .../shared/approval-notifications.test.ts | 143 ++++++++++ .../handlers/shared/task-cancellation.test.ts | 144 ++++++++++ cdk/test/handlers/slack-interactions.test.ts | 32 ++- cdk/test/handlers/slack-notify.test.ts | 39 +++ cdk/test/stacks/agent.test.ts | 151 ++++++++-- cli/src/commands/watch.ts | 27 +- cli/src/types.ts | 1 + cli/test/commands/watch.test.ts | 15 + ...ADR-021-lambda-microvms-compute-backend.md | 8 +- docs/design/CEDAR_HITL_GATES.md | 102 +++++-- docs/design/DEPLOYMENT_ROLES.md | 13 +- .../docs/architecture/Cedar-hitl-gates.md | 102 +++++-- .../docs/architecture/Deployment-roles.md | 13 +- ...Adr-021-lambda-microvms-compute-backend.md | 8 +- .../645-p3-approval-ux-20260917.md | 267 ++++++++++++++++++ .../645-p3-implementation-plan.md | 165 ++++++++++- docs/verification/645-p3-nested-stack.md | 192 +++++++++++++ scripts/check-types-sync.ts | 1 + 57 files changed, 2678 insertions(+), 396 deletions(-) create mode 100644 cdk/src/constructs/lambda-microvm-stack.ts create mode 100644 cdk/src/handlers/shared/approval-notifications.ts create mode 100644 cdk/src/handlers/shared/task-cancellation.ts create mode 100644 cdk/test/constructs/lambda-microvm-stack.test.ts create mode 100644 cdk/test/handlers/shared/approval-notifications.test.ts create mode 100644 cdk/test/handlers/shared/task-cancellation.test.ts create mode 100644 docs/verification/645-p3-approval-ux-20260917.md create mode 100644 docs/verification/645-p3-nested-stack.md diff --git a/agent/src/hooks.py b/agent/src/hooks.py index 27f6e509e..b5b5d409c 100644 --- a/agent/src/hooks.py +++ b/agent/src/hooks.py @@ -802,6 +802,11 @@ async def _handle_require_approval( row_reread = None outcome = _reconcile_late_decision(outcome, row_reread, progress, request_id) + if outcome.get("status") == "CANCELLED": + # The platform closed the approval in the same transaction as task + # cancellation. Do not try to restore RUNNING or cache a human denial. + return _deny_response("Task was cancelled; this approval is closed.") + # Step 11 — resume transition (RUNNING). The ``awaiting_approval_request_id`` # condition prevents resuming a cancelled task or racing with another # approval. @@ -953,6 +958,7 @@ def _reconcile_late_decision( - ``row["status"] == "APPROVED"`` → rebuild as APPROVED (allow flow). - ``row["status"] == "DENIED"`` → rebuild as DENIED (deny flow). + - ``row["status"] == "CANCELLED"`` → close the wait without resuming the task. - Anything else (row gone, still PENDING) → fall through with the original TIMED_OUT (the fail-closed branch). @@ -977,6 +983,8 @@ def _reconcile_late_decision( "decided_at": row.get("decided_at"), "decided_by": row.get("user_id"), } + if status == "CANCELLED": + return {"status": "CANCELLED", "reason": row.get("cancellation_reason")} if status == "DENIED": if progress is not None: _try_progress( @@ -1070,6 +1078,11 @@ async def _poll_for_decision( "reason": row.get("deny_reason") or "denied", "decided_at": row.get("decided_at"), } + if status == "CANCELLED": + return { + "status": "CANCELLED", + "reason": row.get("cancellation_reason"), + } # Compute sleep interval based on elapsed since poll started. elapsed = time.monotonic() - start diff --git a/agent/tests/test_hooks.py b/agent/tests/test_hooks.py index 297079fa4..2ce4faed0 100644 --- a/agent/tests/test_hooks.py +++ b/agent/tests/test_hooks.py @@ -1336,6 +1336,44 @@ def test_denied_returns_deny_queues_injection_and_caches( assert "write_approval_denied" in progress.milestones() +class TestCancelledPath: + @pytest.mark.parametrize("during_timeout", [False, True]) + def test_cancellation_closes_wait_without_resuming_or_caching_a_human_denial( + self, fake_task_state, progress, engine_with_soft_gate, monkeypatch, during_timeout + ): + _fast_poll(monkeypatch) + cancelled = {"status": "CANCELLED", "cancellation_reason": "Task cancelled by its owner"} + if during_timeout: + fake_task_state.best_effort_return = False + fake_task_state.reread_row = cancelled + else: + _prime_approval(fake_task_state, cancelled) + + result = _run( + pre_tool_use_hook( + _hook_input(), + "tu-1", + {}, + engine=engine_with_soft_gate, + task_id="01KTASK", + user_id="u-1", + progress=progress, + task_state_module=fake_task_state, + ) + ) + + assert result["hookSpecificOutput"]["permissionDecision"] == "deny" + assert "cancelled" in result["hookSpecificOutput"]["permissionDecisionReason"] + assert not fake_task_state.resume_calls + if not during_timeout: + assert not fake_task_state.update_calls + assert not engine_with_soft_gate.drain_denial_injections() + follow_up = engine_with_soft_gate.evaluate_tool_use("Bash", {"command": "echo foo"}) + assert "Recent DENIED" not in follow_up.reason + assert "write_approval_denied" not in progress.milestones() + assert "write_approval_timed_out" not in progress.milestones() + + class TestPersistentGateCount: """Chunk 7 (§13.6): REQUIRE_APPROVAL path must bump BOTH the session counter and the TaskTable-persisted counter so a container restart diff --git a/cdk/bootstrap/BOOTSTRAP_HASH b/cdk/bootstrap/BOOTSTRAP_HASH index 49d71c9a7..61b09e65c 100644 --- a/cdk/bootstrap/BOOTSTRAP_HASH +++ b/cdk/bootstrap/BOOTSTRAP_HASH @@ -1 +1 @@ -0f557449b9b55426b3ae54e28ca76807e2eb74ead7472ace5f035f590b651bd0 +dc6301b65558c8fb51200e7b712a754e1b85adda7a512b14ec4f647dc954f541 diff --git a/cdk/bootstrap/BOOTSTRAP_VERSION b/cdk/bootstrap/BOOTSTRAP_VERSION index 27f9cd322..f8e233b27 100644 --- a/cdk/bootstrap/BOOTSTRAP_VERSION +++ b/cdk/bootstrap/BOOTSTRAP_VERSION @@ -1 +1 @@ -1.8.0 +1.9.0 diff --git a/cdk/bootstrap/bootstrap-template.yaml b/cdk/bootstrap/bootstrap-template.yaml index 265a486bf..934df554c 100644 --- a/cdk/bootstrap/bootstrap-template.yaml +++ b/cdk/bootstrap/bootstrap-template.yaml @@ -1,7 +1,7 @@ # GENERATED FILE - DO NOT EDIT DIRECTLY # This template is generated by: npx tsx scripts/generate-bootstrap-template.ts -# ABCA Bootstrap Policy Version: 1.8.0 -# ABCA Bootstrap Policy Hash: 0f557449b9b55426b3ae54e28ca76807e2eb74ead7472ace5f035f590b651bd0 +# ABCA Bootstrap Policy Version: 1.9.0 +# ABCA Bootstrap Policy Hash: dc6301b65558c8fb51200e7b712a754e1b85adda7a512b14ec4f647dc954f541 # # Based on the default CDK bootstrap template with the following modifications: # - BootstrapVariant set to "ABCA: Least-Privilege Bootstrap" @@ -868,7 +868,7 @@ Resources: ManagedPolicyName: Fn::Sub: cdk-${Qualifier}-IaCRole-ABCA-Compute-LambdaMicrovms-${AWS::AccountId}-${AWS::Region} PolicyDocument: >- - {"Statement":[{"Action":["lambda:CreateMicrovmImage","lambda:GetMicrovmImage","lambda:UpdateMicrovmImage","lambda:DeleteMicrovmImage","lambda:ListMicrovmImages","lambda:GetMicrovmImageVersion","lambda:UpdateMicrovmImageVersion","lambda:DeleteMicrovmImageVersion","lambda:ListMicrovmImageVersions","lambda:GetMicrovmImageBuild","lambda:ListMicrovmImageBuilds","lambda:ListManagedMicrovmImages","lambda:ListManagedMicrovmImageVersions","lambda:CreateNetworkConnector","lambda:GetNetworkConnector","lambda:UpdateNetworkConnector","lambda:DeleteNetworkConnector","lambda:ListNetworkConnectors","lambda:PassNetworkConnector"],"Effect":"Allow","Resource":"*","Sid":"LambdaMicrovms"},{"Action":"iam:PassRole","Effect":"Allow","Resource":["arn:aws:iam::*:role/backgroundagent-dev-LambdaMicrovmComputeBuild*","arn:aws:iam::*:role/backgroundagent-dev-LambdaMicrovmComputeConnector*"],"Sid":"MicrovmPassRoles"},{"Action":["ssm:GetParameters","ssm:PutParameter","ssm:DeleteParameter","ssm:AddTagsToResource","ssm:RemoveTagsFromResource","ssm:ListTagsForResource"],"Effect":"Allow","Resource":"arn:aws:ssm:*:*:parameter/backgroundagent-*/microvm-approval-suspend-enabled","Sid":"MicrovmSuspendConfiguration"}],"Version":"2012-10-17"} + {"Statement":[{"Action":["lambda:CreateMicrovmImage","lambda:GetMicrovmImage","lambda:UpdateMicrovmImage","lambda:DeleteMicrovmImage","lambda:ListMicrovmImages","lambda:GetMicrovmImageVersion","lambda:UpdateMicrovmImageVersion","lambda:DeleteMicrovmImageVersion","lambda:ListMicrovmImageVersions","lambda:GetMicrovmImageBuild","lambda:ListMicrovmImageBuilds","lambda:ListManagedMicrovmImages","lambda:ListManagedMicrovmImageVersions","lambda:CreateNetworkConnector","lambda:GetNetworkConnector","lambda:UpdateNetworkConnector","lambda:DeleteNetworkConnector","lambda:ListNetworkConnectors","lambda:PassNetworkConnector"],"Effect":"Allow","Resource":"*","Sid":"LambdaMicrovms"},{"Action":"iam:PassRole","Effect":"Allow","Resource":["arn:aws:iam::*:role/backgroundagent-dev-LambdaMicrovmComputeBuild*","arn:aws:iam::*:role/backgroundagent-dev-LambdaMicrovmComputeConnector*","arn:aws:iam::*:role/backgroundagent-dev-MicrovmBuildRole","arn:aws:iam::*:role/backgroundagent-dev-MicrovmConnectorRole"],"Sid":"MicrovmPassRoles"},{"Action":["ssm:GetParameters","ssm:PutParameter","ssm:DeleteParameter","ssm:AddTagsToResource","ssm:RemoveTagsFromResource","ssm:ListTagsForResource"],"Effect":"Allow","Resource":"arn:aws:ssm:*:*:parameter/backgroundagent-*/microvm-approval-suspend-enabled","Sid":"MicrovmSuspendConfiguration"}],"Version":"2012-10-17"} Description: 'ABCA Bootstrap: IaCRole-ABCA-Compute-LambdaMicrovms permissions for CloudFormation execution role' Condition: IncludeComputeLambdaMicrovms Outputs: @@ -899,10 +899,10 @@ Outputs: Value: '32' BootstrapPolicyVersion: Description: The version of the ABCA bootstrap policy bundle - Value: 1.8.0 + Value: 1.9.0 BootstrapPolicyHash: Description: SHA-256 hash of the ABCA bootstrap policy bundle for drift detection - Value: 0f557449b9b55426b3ae54e28ca76807e2eb74ead7472ace5f035f590b651bd0 + Value: dc6301b65558c8fb51200e7b712a754e1b85adda7a512b14ec4f647dc954f541 BootstrapPolicySet: Description: Comma-separated list of active ABCA bootstrap policy names Value: diff --git a/cdk/bootstrap/policies/compute-lambda-microvm.json b/cdk/bootstrap/policies/compute-lambda-microvm.json index 3f09b0b0f..227dd2cda 100644 --- a/cdk/bootstrap/policies/compute-lambda-microvm.json +++ b/cdk/bootstrap/policies/compute-lambda-microvm.json @@ -31,7 +31,9 @@ "Effect": "Allow", "Resource": [ "arn:aws:iam::*:role/backgroundagent-dev-LambdaMicrovmComputeBuild*", - "arn:aws:iam::*:role/backgroundagent-dev-LambdaMicrovmComputeConnector*" + "arn:aws:iam::*:role/backgroundagent-dev-LambdaMicrovmComputeConnector*", + "arn:aws:iam::*:role/backgroundagent-dev-MicrovmBuildRole", + "arn:aws:iam::*:role/backgroundagent-dev-MicrovmConnectorRole" ], "Sid": "MicrovmPassRoles" }, diff --git a/cdk/scripts/README.md b/cdk/scripts/README.md index 548533c34..c42b4b20c 100644 --- a/cdk/scripts/README.md +++ b/cdk/scripts/README.md @@ -8,4 +8,8 @@ Bundling for Lambda assets is handled at synth time; the **`bundle`** task in ** | `generate-bootstrap-template.ts` | Regenerates `cdk/bootstrap/bootstrap-template.yaml` (least-privilege CDK bootstrap, `ComputeTypes`-gated compute policies) | `mise //cdk:bootstrap:generate` | | `package-microvm-artifact.sh` | Packages `agent/` + `contracts/` + `Dockerfile` into the zip artifact an `AWS::Lambda::MicrovmImage` builds from, and uploads it to the CDK-created artifact bucket (ADR-021) | run directly — see the script header for the full bootstrap sequence | -`package-microvm-artifact.sh` exists because CloudFormation cannot produce its own MicroVM `codeArtifact`: the image resource consumes a zip that must already be in S3, and there is no CDK asset type for "zip + Dockerfile a MicroVM image builds from". Everything else on that backend (buckets, roles, network connector, log group, the image resource itself) is CDK-managed in `src/constructs/lambda-microvm-compute.ts`. +`package-microvm-artifact.sh` exists because CloudFormation cannot produce its own MicroVM `codeArtifact`: the image resource consumes a zip that must already be in S3, and there is no CDK asset type for "zip + Dockerfile a MicroVM image builds from". Everything else on that backend (buckets, roles, network connectors, log group, the image resource itself) is CDK-managed by `src/constructs/lambda-microvm-compute.ts`, normally inside `lambda-microvm-stack.ts`. + +The nested layout requires bootstrap bundle 1.9.0. Existing flat deployments must +keep `microvm_nested_stack=false` until completing the +[resource migration](../../docs/verification/645-p3-nested-stack.md). diff --git a/cdk/scripts/package-microvm-artifact.sh b/cdk/scripts/package-microvm-artifact.sh index 8d7d1b4d0..70b989498 100755 --- a/cdk/scripts/package-microvm-artifact.sh +++ b/cdk/scripts/package-microvm-artifact.sh @@ -18,6 +18,10 @@ # --------------------------------------------------------------------------- # BOOTSTRAP SEQUENCE (first time) # --------------------------------------------------------------------------- +# 0. Use bootstrap policy bundle >= 1.9.0 for the default nested layout. +# Existing flat deployments must retain microvm_nested_stack=false until +# their resources are migrated; see docs/verification/645-p3-nested-stack.md. +# # 1. Deploy the MicroVM substrate WITHOUT an image. Synth warns that no image # is configured; that is expected — the artifact bucket must exist before # you can upload to it. @@ -342,26 +346,22 @@ if [[ "${CREATE_IMAGE}" -eq 0 ]]; then CDK-managed (recommended) — redeploy with the base image and artifact pinned. Keep this digest with the deployment's context; it identifies these exact ZIP bytes. - !! BOOTSTRAP POLICY BUNDLE >= 1.7.0 REQUIRED !! - This path took two live-verified fixes to work. The first (ADR-021 P2-F2: the L1 - sent hook paths and \`arm64\` where CloudFormation wants ENABLED / ARM_64) is - DISCHARGED — change-set early validation now passes. The second (ADR-021 - P2r2-F9) is a BOOTSTRAP change: CloudFormation could not pass the MicroVM build - role, because the deploy role's \`iam:PassRole\` carried an - \`iam:PassedToService\` condition the Lambda MicroVMs service does not satisfy. - The fix is the \`MicrovmPassRoles\` statement in the conditional - IaCRole-ABCA-Compute-LambdaMicrovms policy, which only reaches your account when - you re-bootstrap: + The default nested layout requires bootstrap policy bundle >= 1.9.0. + Flat P3 deployments require >= 1.8.0. Check the installed bundle: aws cloudformation describe-stacks --stack-name CDKToolkit \\ --query 'Stacks[0].Outputs[?OutputKey==\`BootstrapPolicyVersion\`].OutputValue' --output text - # if that is below 1.7.0: MISE_EXPERIMENTAL=1 mise //cdk:bootstrap # ComputeTypes must include lambda-microvm - Without it the image resource fails with - "is not authorized to perform: iam:PassRole on resource: - ...LambdaMicrovmComputeBuildRole... (Service: LambdaMicrovms, Status Code: 403)". - Then: + An older bundle may deny iam:PassRole for the image's build role. Updating the + source bundle alone does not update the account's installed policies. + + Existing flat stacks: keep --context microvm_nested_stack=false on deployment + commands until the resource migration is complete. Changing the layout directly + can replace resources or delete bucket contents. See: + docs/verification/645-p3-nested-stack.md + + For a new nested deployment (or a completed migration): aws lambda-microvms list-managed-microvm-images MISE_EXPERIMENTAL=1 mise //cdk:deploy -- \\ diff --git a/cdk/src/bootstrap/policies/compute-lambda-microvm.ts b/cdk/src/bootstrap/policies/compute-lambda-microvm.ts index e3f0f9f9c..670f59e70 100644 --- a/cdk/src/bootstrap/policies/compute-lambda-microvm.ts +++ b/cdk/src/bootstrap/policies/compute-lambda-microvm.ts @@ -134,7 +134,8 @@ export function computeLambdaMicrovmPolicy(): iam.PolicyDocument { // shared statement would have dropped that constraint for ~30 roles to // fix two. // - // SCOPE. Two name-prefix patterns, not `role/backgroundagent-dev-*`, and + // SCOPE. Build/operator prefixes for flat deployments and exact names for + // nested deployments, not `role/backgroundagent-dev-*`, and // deliberately NOT the execution role — CloudFormation never passes that one // (the orchestrator does, at `RunMicrovm`), so including it here would widen // the unconditioned pass to the role that runs untrusted repo code for no @@ -155,6 +156,10 @@ export function computeLambdaMicrovmPolicy(): iam.PolicyDocument { // Passed as `operatorRole` on AWS::Lambda::NetworkConnector (required // for VPC_EGRESS connectors). 'arn:aws:iam::*:role/backgroundagent-dev-LambdaMicrovmComputeConnector*', + // Nested deployments pin these two physical names to the parent name; + // no generated child-stack prefix or role-name wildcard is needed. + 'arn:aws:iam::*:role/backgroundagent-dev-MicrovmBuildRole', + 'arn:aws:iam::*:role/backgroundagent-dev-MicrovmConnectorRole', ], }), // A shared value is required because durable executions retain their diff --git a/cdk/src/bootstrap/version.ts b/cdk/src/bootstrap/version.ts index 387a458f5..a2634e16e 100644 --- a/cdk/src/bootstrap/version.ts +++ b/cdk/src/bootstrap/version.ts @@ -58,8 +58,11 @@ import { allPolicies } from './policies'; * * 1.7.0 → 1.8.0 adds scoped SSM parameter lifecycle/tag permissions for the P3 * MicroVM suspension switch. Re-bootstrap before deploying the live parameter. + * + * 1.8.0 → 1.9.0 admits the exact nested MicroVM build/operator role names to + * the backend-specific PassRole statement. Re-bootstrap before the nested split. */ -export const BOOTSTRAP_VERSION = '1.8.0'; +export const BOOTSTRAP_VERSION = '1.9.0'; function canonicalize(value: unknown): unknown { if (Array.isArray(value)) return value.map(canonicalize); diff --git a/cdk/src/constructs/fanout-consumer.ts b/cdk/src/constructs/fanout-consumer.ts index f232897e3..5ab025043 100644 --- a/cdk/src/constructs/fanout-consumer.ts +++ b/cdk/src/constructs/fanout-consumer.ts @@ -63,6 +63,9 @@ export interface FanOutConsumerProps { */ readonly taskTable?: dynamodb.ITable; + /** Approval state and successful per-request notification receipts. */ + readonly taskApprovalsTable?: dynamodb.ITable; + /** * RepoTable — GitHub dispatcher reads per-repo * `github_token_secret_arn` overrides. Optional: if omitted, falls @@ -214,6 +217,10 @@ export class FanOutConsumer extends Construct { props.taskTable.grantReadWriteData(this.fn); this.fn.addEnvironment('TASK_TABLE_NAME', props.taskTable.tableName); } + if (props.taskApprovalsTable) { + props.taskApprovalsTable.grant(this.fn, 'dynamodb:GetItem', 'dynamodb:UpdateItem'); + this.fn.addEnvironment('TASK_APPROVALS_TABLE_NAME', props.taskApprovalsTable.tableName); + } if (props.repoTable) { props.repoTable.grantReadData(this.fn); this.fn.addEnvironment('REPO_TABLE_NAME', props.repoTable.tableName); diff --git a/cdk/src/constructs/lambda-microvm-compute.ts b/cdk/src/constructs/lambda-microvm-compute.ts index eb9ceb39b..294e0d6dc 100644 --- a/cdk/src/constructs/lambda-microvm-compute.ts +++ b/cdk/src/constructs/lambda-microvm-compute.ts @@ -486,6 +486,20 @@ export function isLambdaMicrovmImageConfigured(inputs: LambdaMicrovmImageInputs) * Properties for {@link LambdaMicrovmCompute}. */ export interface LambdaMicrovmComputeProps extends LambdaMicrovmImageInputs { + /** Stable parent deployment name for image/connector names when nested. */ + readonly deploymentName?: string; + + /** + * Parent-owned execution role. Keeping this beside AgentSessionRole avoids + * a parent/child cycle through the session role's trust and AssumeRole grant. + * Omitted by standalone constructs, which create their own execution role. + */ + readonly executionRole?: iam.Role; + + /** Explicit names used by nested deployments to preserve bootstrap PassRole scope. */ + readonly buildRoleName?: string; + readonly connectorOperatorRoleName?: string; + /** * Platform VPC. Egress leaves the MicroVM through a `AWS::Lambda::NetworkConnector` * bound to this VPC's private-with-egress subnets, so the DNS Firewall / @@ -856,6 +870,10 @@ export class LambdaMicrovmCompute extends Construct { assertLambdaMicrovmRegionSupported(this); const stack = Stack.of(this); + const deploymentName = props.deploymentName ?? stack.stackName; + if (Token.isUnresolved(deploymentName)) { + throw new Error('Nested MicroVM resources require a concrete deploymentName from the parent stack'); + } const managedImage = Boolean(props.baseImageArn && props.baseImageVersion); if ((managedImage || props.artifactSha256 !== undefined) && !/^[a-f0-9]{64}$/.test(props.artifactSha256 ?? '')) { @@ -868,7 +886,7 @@ export class LambdaMicrovmCompute extends Construct { this.artifactObjectKey = props.artifactSha256 ? `${this.artifactBaseObjectKey.replace(/\.zip$/, '')}-${props.artifactSha256}.zip` : this.artifactBaseObjectKey; - this.imageName = props.imageName ?? sanitizeImageName(`${stack.stackName}-abca-agent`); + this.imageName = props.imageName ?? sanitizeImageName(`${deploymentName}-abca-agent`); // Fail at SYNTH on an unsupported memory size. The service enumerates the // sizes a base image accepts and rejects anything else at create time, which @@ -970,6 +988,7 @@ export class LambdaMicrovmCompute extends Construct { // ("NO source-key condition on any MicroVM-facing role trust"). const microvmAssumedBy = new iam.ServicePrincipal('lambda.amazonaws.com'); this.connectorOperatorRole = new iam.Role(this, 'ConnectorOperatorRole', { + roleName: props.connectorOperatorRoleName, assumedBy: microvmAssumedBy, description: 'ABCA Lambda MicroVMs network-connector operator role: lets Lambda manage the connector ' @@ -1011,7 +1030,7 @@ export class LambdaMicrovmCompute extends Construct { }).subnetIds; this.egressConnector = new lambda.CfnNetworkConnector(this, 'EgressConnector', { - name: sanitizeImageName(`${stack.stackName}-microvm-egress`), + name: sanitizeImageName(`${deploymentName}-microvm-egress`), operatorRole: this.connectorOperatorRole.roleArn, configuration: { vpcEgressConfiguration: { @@ -1028,7 +1047,7 @@ export class LambdaMicrovmCompute extends Construct { // (443 + 80). Referenced by the image resource / packaging script, never by // `RunMicrovm`, so a running agent still gets the 443-only posture. this.buildEgressConnector = new lambda.CfnNetworkConnector(this, 'BuildEgressConnector', { - name: sanitizeImageName(`${stack.stackName}-microvm-build-egress`), + name: sanitizeImageName(`${deploymentName}-microvm-build-egress`), operatorRole: this.connectorOperatorRole.roleArn, configuration: { vpcEgressConfiguration: { @@ -1111,6 +1130,7 @@ export class LambdaMicrovmCompute extends Construct { // {@link grantTagSession} adds. this.buildRole = new iam.Role(this, 'BuildRole', { + roleName: props.buildRoleName, assumedBy: microvmAssumedBy, description: 'ABCA Lambda MicroVMs image-build role: reads the zip+Dockerfile artifact from S3 ' @@ -1128,13 +1148,7 @@ export class LambdaMicrovmCompute extends Construct { // Build role: keeps `logs:CreateLogGroup` — see `grantMicrovmLogWrites`. this.grantMicrovmLogWrites(this.buildRole, { allowCreateLogGroup: true }); - this.executionRole = new iam.Role(this, 'ExecutionRole', { - assumedBy: microvmAssumedBy, - description: - 'ABCA Lambda MicroVMs execution role: assumed by the running MicroVM and its runtime ' - + 'lifecycle hooks; writes logs and reads deployment bootstrap manifests.', - }); - grantTagSession(this.executionRole, microvmAssumedBy); + this.executionRole = props.executionRole ?? createMicrovmExecutionRole(this, 'ExecutionRole'); // Execution role: NO `logs:CreateLogGroup`. It runs untrusted repo code and // live evidence shows it only ever writes into the pre-created group — see // `grantMicrovmLogWrites` for the runbook citations and the re-verify note. @@ -1291,7 +1305,7 @@ export class LambdaMicrovmCompute extends Construct { } this.image = new lambda.CfnMicrovmImage(this, 'Image', { name: this.imageName, - description: `ABCA agent snapshot for ${stack.stackName} (ADR-021 lambda-microvm backend)`, + description: `ABCA agent snapshot for ${deploymentName} (ADR-021 lambda-microvm backend)`, baseImageArn: props.baseImageArn, baseImageVersion: props.baseImageVersion, buildRoleArn: this.buildRole.roleArn, @@ -1420,8 +1434,8 @@ export class LambdaMicrovmCompute extends Construct { if (this.imageIdentifier) { // Emitted on EVERY deploy that configures an image, in both image states. - // Clean coding runs exist; they do not cover the broader failure/security - // matrix. Keep the warning scoped to those remaining acceptance gates. + // Prior image acceptance does not verify a newly configured image or turn + // on automatic suspension. Keep the warning scoped to deployment gates. // // The id is deliberately UNCHANGED across P1→P2 (operators grep for it, and a // rename would read as "the old warning is gone, so it must be fine"). @@ -1430,11 +1444,13 @@ export class LambdaMicrovmCompute extends Construct { 'A MicroVM image is configured. Clean P2 deployment with bootstrap bundle 1.7.0 and ' + 'coding, iteration and cancellation runs passed on 2026-09-14 without manual IAM changes. ' + 'The agent serves /ready, /validate, /run, /terminate, /suspend and /resume; managed images declare all six. ' - + 'Heartbeat, logs, Memory writes and cleanup have live evidence. Full P2 acceptance ' - + 'still needs the failure/recovery, effective IAM and networking matrix in ' - + 'docs/verification/645-p3-implementation-plan.md. P3 checks the actual launched image version; ' + + 'Image 6.0 has nine successful isolated P3 approval workflows, including expired-credential renewal ' + + 'and repository/network checks after the Connection: close lifecycle fix. This does not verify ' + + 'a different image. P3 checks the actual launched image version; ' + 'P3 requires bootstrap bundle 1.8.0 and defaults new suspension off; supervisor integration is implemented. ' - + 'Live sleep/wake acceptance remains open. The warning ID is retained ' + + 'Nested deployments require bundle 1.9.0 and a reviewed migration from existing flat stacks. ' + + 'Normal automatic-suspension activation remains open; follow docs/verification/645-p3-implementation-plan.md ' + + 'and docs/verification/645-p3-nested-stack.md. The warning ID is retained ' + 'across phases for existing operator filters.', ); } @@ -1563,6 +1579,20 @@ export class LambdaMicrovmCompute extends Construct { } } +/** Create the runtime role in its owning stack, independently of image resources. */ +export function createMicrovmExecutionRole(scope: Construct, id: string): iam.Role { + const principal = new iam.ServicePrincipal('lambda.amazonaws.com'); + const role = new iam.Role(scope, id, { + assumedBy: principal, + description: + 'ABCA Lambda MicroVMs execution role: assumed by the running MicroVM and its runtime ' + + 'lifecycle hooks; writes logs and reads deployment bootstrap manifests.', + }); + grantTagSession(role, principal); + Tags.of(role).add(MICROVM_BACKEND_TAG_KEY, MICROVM_BACKEND_TAG_VALUE); + return role; +} + /** * Add `sts:TagSession` alongside the `sts:AssumeRole` CDK's `assumedBy` emits. * diff --git a/cdk/src/constructs/lambda-microvm-stack.ts b/cdk/src/constructs/lambda-microvm-stack.ts new file mode 100644 index 000000000..76359069e --- /dev/null +++ b/cdk/src/constructs/lambda-microvm-stack.ts @@ -0,0 +1,67 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import { NestedStack, Stack, Tags, Token } from 'aws-cdk-lib'; +import type * as iam from 'aws-cdk-lib/aws-iam'; +import { Construct } from 'constructs'; +import { + LambdaMicrovmCompute, + type LambdaMicrovmComputeProps, + MICROVM_BACKEND_TAG_KEY, + MICROVM_BACKEND_TAG_VALUE, +} from './lambda-microvm-compute'; + +export interface LambdaMicrovmStackProps extends Omit< + LambdaMicrovmComputeProps, 'buildRoleName' | 'connectorOperatorRoleName' +> { + readonly deploymentName: string; + readonly executionRole: iam.Role; +} + +/** Child deployment for MicroVM resources; shared runtime trust stays in the parent. */ +export class LambdaMicrovmStack extends NestedStack { + public readonly compute: LambdaMicrovmCompute; + + constructor(scope: Construct, id: string, props: LambdaMicrovmStackProps) { + super(scope, id, { + description: 'ABCA Lambda MicroVM image, storage, build roles and network connectors', + }); + if (Stack.of(props.executionRole) !== this.nestedStackParent) { + throw new Error('LambdaMicrovmStack executionRole must be owned by its parent stack'); + } + // CDK creates bucket-cleanup providers at stack scope, outside Compute. + // Those providers are also MicroVM-specific in this child. + Tags.of(this).add(MICROVM_BACKEND_TAG_KEY, MICROVM_BACKEND_TAG_VALUE); + // Automatic IAM names include the generated nested-stack name and can lose + // their discriminating role suffix to truncation. Explicit parent names keep + // the bootstrap's unconditioned PassRole grant limited to these two roles. + const roleName = (suffix: string): string => { + const name = `${props.deploymentName}-${suffix}`; + if (Token.isUnresolved(name) || !/^[A-Za-z0-9+=,.@_-]{1,64}$/.test(name)) { + throw new Error(`Nested MicroVM role name must be concrete and at most 64 characters: ${suffix}`); + } + return name; + }; + this.compute = new LambdaMicrovmCompute(this, 'Compute', { + ...props, + buildRoleName: roleName('MicrovmBuildRole'), + connectorOperatorRoleName: roleName('MicrovmConnectorRole'), + }); + } +} diff --git a/cdk/src/constructs/slack-integration.ts b/cdk/src/constructs/slack-integration.ts index fef7a117a..4140a91fe 100644 --- a/cdk/src/constructs/slack-integration.ts +++ b/cdk/src/constructs/slack-integration.ts @@ -64,6 +64,9 @@ export interface SlackIntegrationProps { /** The DynamoDB task events table (must have DynamoDB Streams enabled). */ readonly taskEventsTable: dynamodb.ITable; + /** Approval records closed atomically when their owner cancels a task. */ + readonly taskApprovalsTable?: dynamodb.ITable; + /** Monthly user/team budget configuration and spend table. */ readonly budgetTable?: dynamodb.ITable; @@ -351,6 +354,9 @@ export class SlackIntegration extends Construct { environment: { SLACK_SIGNING_SECRET_ARN: this.signingSecret.secretArn, TASK_TABLE_NAME: props.taskTable.tableName, + TASK_EVENTS_TABLE_NAME: props.taskEventsTable.tableName, + TASK_RETENTION_DAYS: String(props.taskRetentionDays ?? DEFAULT_TASK_RETENTION_DAYS), + ...(props.taskApprovalsTable && { TASK_APPROVALS_TABLE_NAME: props.taskApprovalsTable.tableName }), SLACK_USER_MAPPING_TABLE_NAME: this.userMappingTable.tableName, }, bundling: commonBundling, @@ -358,6 +364,8 @@ export class SlackIntegration extends Construct { this.signingSecret.grantRead(slackInteractionsFn); slackInteractionsFn.addToRolePolicy(readSlackSecretsPolicy); props.taskTable.grantReadWriteData(slackInteractionsFn); + props.taskEventsTable.grant(slackInteractionsFn, 'dynamodb:PutItem'); + props.taskApprovalsTable?.grant(slackInteractionsFn, 'dynamodb:GetItem', 'dynamodb:UpdateItem'); this.userMappingTable.grantReadData(slackInteractionsFn); // --- Slash Command Acknowledger --- diff --git a/cdk/src/constructs/task-api.ts b/cdk/src/constructs/task-api.ts index b0959f57d..8c538bdaf 100644 --- a/cdk/src/constructs/task-api.ts +++ b/cdk/src/constructs/task-api.ts @@ -652,6 +652,9 @@ export class TaskApi extends Construct { }); const cancelTaskEnv: Record = { ...commonEnv }; + if (props.taskApprovalsTable) { + cancelTaskEnv.TASK_APPROVALS_TABLE_NAME = props.taskApprovalsTable.tableName; + } const stopSessionArn = props.agentCoreStopSessionRuntimeArn; if (stopSessionArn) { cancelTaskEnv.RUNTIME_ARN = stopSessionArn; @@ -667,10 +670,8 @@ export class TaskApi extends Construct { architecture: Architecture.ARM_64, environment: cancelTaskEnv, bundling: commonBundling, - // Cancel performs: DDB GetItem + DDB UpdateItem + ECS StopTask or - // AgentCore StopRuntimeSession + DDB PutItem. The default 3s timeout - // is not enough once cold-start TLS handshakes for bedrock-agentcore - // are added. 15s gives comfortable headroom. + // Cancel reads state, atomically closes an unanswered approval, stops + // compute and writes an event. The state operation has its own 5s bound. timeout: Duration.seconds(API_HANDLER_TIMEOUT_SECONDS), memorySize: API_HANDLER_MEMORY_MB, }); @@ -709,6 +710,7 @@ export class TaskApi extends Construct { props.taskEventsTable.grantReadWriteData(createTaskFn); props.taskTable.grantReadWriteData(cancelTaskFn); props.taskEventsTable.grantReadWriteData(cancelTaskFn); + props.taskApprovalsTable?.grant(cancelTaskFn, 'dynamodb:GetItem', 'dynamodb:UpdateItem'); if (stopSessionArn) { cancelTaskFn.addToRolePolicy(new iam.PolicyStatement({ @@ -1046,6 +1048,7 @@ export class TaskApi extends Construct { timeout: Duration.seconds(10), memorySize: API_HANDLER_MEMORY_MB, }); + props.taskTable.grant(getPendingFn, 'dynamodb:BatchGetItem'); // Least-privilege: GetPendingFn only reads (Query on // user_id-status-index for the user's pending rows) and writes // a synthetic ``RATE##PENDING`` rate-limit row diff --git a/cdk/src/handlers/cancel-task.ts b/cdk/src/handlers/cancel-task.ts index d0db27f0a..08b339d10 100644 --- a/cdk/src/handlers/cancel-task.ts +++ b/cdk/src/handlers/cancel-task.ts @@ -20,13 +20,14 @@ import { BedrockAgentCoreClient, StopRuntimeSessionCommand } from '@aws-sdk/client-bedrock-agentcore'; import { ECSClient, StopTaskCommand } from '@aws-sdk/client-ecs'; import { LambdaMicrovmsClient, TerminateMicrovmCommand } from '@aws-sdk/client-lambda-microvms'; -import { GetCommand, PutCommand, UpdateCommand } from '@aws-sdk/lib-dynamodb'; +import { GetCommand, PutCommand } from '@aws-sdk/lib-dynamodb'; import type { APIGatewayProxyEvent, APIGatewayProxyResult } from 'aws-lambda'; import { ulid } from 'ulid'; import { TaskStatus, TERMINAL_STATUSES } from '../constructs/task-status'; import { extractUserId } from './shared/gateway'; import { logger } from './shared/logger'; import { ErrorCode, errorResponse, successResponse } from './shared/response'; +import { cancelTaskState, TaskCancellationError } from './shared/task-cancellation'; import type { TaskRecord } from './shared/types'; import { makeClient, makeDocClient } from './shared/ua'; import { computeTtlEpoch } from './shared/validation'; @@ -37,6 +38,7 @@ const ecsClient = makeClient(ECSClient); const microvmClient = makeClient(LambdaMicrovmsClient); const TABLE_NAME = process.env.TASK_TABLE_NAME!; const EVENTS_TABLE_NAME = process.env.TASK_EVENTS_TABLE_NAME!; +const APPROVALS_TABLE_NAME = process.env.TASK_APPROVALS_TABLE_NAME; const TASK_RETENTION_DAYS = Number(process.env.TASK_RETENTION_DAYS ?? '90'); const RUNTIME_ARN = process.env.RUNTIME_ARN; const ECS_CLUSTER_ARN = process.env.ECS_CLUSTER_ARN; @@ -64,6 +66,7 @@ export async function handler(event: APIGatewayProxyEvent): Promise> ...TERMINAL_EVENT_TYPES, 'pr_created', ]), - // Linear posts deterministic status comments on the platform tier - // (ADR-016: Linear is fully deterministic — the agent has no Linear MCP - // and posts nothing itself). Two events: + // Linear posts approval requests/outcomes and deterministic status comments + // on the platform tier (ADR-016: the agent has no Linear MCP). // * ``pr_created`` — the first-run "🔗 PR opened" courtesy comment (or, // for a comment-iteration, matures the threaded reply to "🔄 Working"). // This replaces the agent's old step-2 MCP save_comment. @@ -185,12 +161,12 @@ export const CHANNEL_DEFAULTS: Record> // OOM) before any PR, which is the motivating case: without a // platform-side comment the requester gets no completion signal at all. // - // Linear's `save_comment` doesn't support edit, so each is post-once (no - // live updates a la GitHub edit-in-place), idempotent across partial-batch - // retries via per-event markers. The start "🤖 Starting" comment is posted + // Approval messages and first-run status comments use delivery receipts; + // iteration replies are edited in place. The start "🤖 Starting" comment is posted // even earlier, at task-admission in the webhook processor (ADR-016 P4.5). linear: new Set([ ...TERMINAL_EVENT_TYPES, + ...APPROVAL_NOTIFICATION_EVENTS, 'pr_created', // Include task_timed_out so a Linear standalone iteration // that TIMES OUT still settles (its 👀→✅/❌ + terminal reply come through the @@ -354,7 +330,7 @@ export function parseStreamRecord(record: DynamoDBRecord): FanOutEvent | null { * ``trajectory_uploaded``, ``trace_truncated``. Only ``pr_created`` * is currently in any channel's default filter (§6.2 Slack + GitHub). */ -const ROUTABLE_MILESTONES: ReadonlySet = new Set(['pr_created']); +const ROUTABLE_MILESTONES: ReadonlySet = new Set(['pr_created', ...APPROVAL_NOTIFICATION_EVENTS]); /** * Unwrap ``agent_milestone`` events to their milestone name for @@ -481,6 +457,7 @@ async function loadTaskForComment(taskId: string): Promise { const result = await ddb.send(new GetCommand({ TableName: tableName, Key: { task_id: taskId }, + ConsistentRead: true, })); return (result.Item as TaskRecord | undefined) ?? null; } @@ -1177,6 +1154,32 @@ async function dispatchToLinear(event: FanOutEvent): Promise { return; } + const effectiveType = effectiveEventType(event); + if (isApprovalNotification(effectiveType)) { + const notification = await loadApprovalNotification(ddb, task, effectiveType, event.metadata ?? {}, 'linear'); + if (!notification) return; + const result = await postIssueComment( + { linearWorkspaceId: workspaceId, registryTableName }, + issueId, + approvalNotificationMarkdown(notification), + ); + if (!result.ok) { + logger.warn('Linear approval notification failed', { + event: 'fanout.linear.approval_post_failed', + task_id: task.task_id, + request_id: notification.requestId, + retryable: result.retryable, + }); + if (result.retryable) throw new Error('Retryable Linear approval notification failure'); + return; + } + await markApprovalNotificationDelivered(ddb, notification); + logger.info('Linear approval notification delivered', { + event: 'fanout.linear.approval_dispatched', task_id: task.task_id, request_id: notification.requestId, + }); + return; // An approval message must never settle the task's final reply. + } + // Iteration-UX: this task is a comment-iteration when it carries a maturing // reply id (set at trigger time). For those, the progress + terminal status // lives in that ONE edited reply, NOT in fresh top-level comments. @@ -1264,8 +1267,8 @@ async function dispatchToLinear(event: FanOutEvent): Promise { return; // milestones never post the terminal status comment } - // Idempotency across partial-batch retries: Linear has no comment - // edit API, so a re-run of this dispatcher (e.g. a sibling channel's + // Idempotency across partial-batch retries: a re-run of this dispatcher + // (e.g. a sibling channel's // infra rejection pushed the whole stream record into // ``batchItemFailures``) would post a duplicate final-status comment. // The marker is persisted after the first successful post below. @@ -1940,8 +1943,7 @@ export async function routeEvent( * ``constructs/fanout-consumer.ts``) can honor partial-batch semantics. * Without a structured return, a single poisonous record would cause * Lambda to retry the **entire batch** from the stream checkpoint, - * replaying every sibling event and defeating the per-task ordering - * guarantee promised by ``ParallelizationFactor: 1`` upstream. + * replaying successful earlier events unnecessarily. * * Partial-failure surface (per-record try/catch below): * - ``routeEvent`` wraps each dispatcher in ``Promise.allSettled``, so @@ -1957,10 +1959,10 @@ export async function routeEvent( * future refactor (e.g. a stricter ``parseStreamRecord``) from * crashing the whole batch. * - * On any caught throw we push ``{ itemIdentifier: record.eventID }`` so - * Lambda retries ONLY that record, isolating the poison pill per - * design §6 + §8.9 expectations. Successful records are NOT in - * ``batchItemFailures`` and advance the stream checkpoint normally. + * Failures return the DynamoDB ``SequenceNumber``, not the opaque ``eventID``. + * Lambda checkpoints at the lowest failed sequence and retries that record and + * the following records. Successful later records can therefore be delivered + * again; channel delivery receipts still matter. * * Two review findings shaped this shape: the fanout handler used to return * ``void`` despite ``reportBatchItemFailures: true``, and a ``routeEvent`` @@ -1983,6 +1985,15 @@ export const handler = async ( let processed = 0; let dispatched = 0; let dropped = 0; + const retryRecord = (record: DynamoDBRecord): void => { + const sequence = record.dynamodb?.SequenceNumber; + if (!sequence) { + // A malformed failed record must not acknowledge the batch or return an + // invalid retry cursor. Reject the invocation so Lambda retries the batch. + throw new Error('Failed DynamoDB record is missing its sequence number'); + } + batchItemFailures.push({ itemIdentifier: sequence }); + }; // v1: no per-task override; every event uses the channel defaults. // Chunk K wires a DDB read here to load ``TaskRecord.notifications``. @@ -2025,21 +2036,13 @@ export const handler = async ( // attempt has a chance to succeed. Without this push, a transient // failure would be silently dropped — the regression that // motivated this fix. - if (outcome.infraRejections.length > 0 && record.eventID !== undefined) { - batchItemFailures.push({ itemIdentifier: record.eventID }); - } + if (outcome.infraRejections.length > 0) retryRecord(record); } catch (err) { // Poison-pill isolation: one record's unhandled throw must not // crash the batch. See the handler doc block for the full list of // paths that can reach here (notably AccessDeniedException from // ``resolveTokenSecretArn``). // - // ``eventID`` is the stream-record identifier Lambda uses for the - // retry cursor; on Kinesis-style event-source-mappings with - // ``reportBatchItemFailures: true`` the service retries all - // records at-or-after the lowest-sequence failure. Returning even - // one failed itemIdentifier is enough to preserve ordering across - // the whole batch for that task. const eventID = record.eventID; logger.warn('[fanout] record threw — flagging for partial-batch retry', { event: 'fanout.record.failed', @@ -2047,9 +2050,7 @@ export const handler = async ( error: err instanceof Error ? err.message : String(err), error_name: err instanceof Error ? err.name : undefined, }); - if (eventID !== undefined) { - batchItemFailures.push({ itemIdentifier: eventID }); - } + retryRecord(record); } } diff --git a/cdk/src/handlers/get-pending.ts b/cdk/src/handlers/get-pending.ts index 713a231dd..08455ae58 100644 --- a/cdk/src/handlers/get-pending.ts +++ b/cdk/src/handlers/get-pending.ts @@ -17,7 +17,7 @@ * SOFTWARE. */ -import { QueryCommand, UpdateCommand } from '@aws-sdk/lib-dynamodb'; +import { BatchGetCommand, QueryCommand, type QueryCommandInput, UpdateCommand } from '@aws-sdk/lib-dynamodb'; import type { APIGatewayProxyEvent, APIGatewayProxyResult } from 'aws-lambda'; import { ulid } from 'ulid'; import { extractUserId } from './shared/gateway'; @@ -29,12 +29,15 @@ import { makeDocClient } from './shared/ua'; const ddb = makeDocClient(); const TASK_APPROVALS_TABLE_NAME = process.env.TASK_APPROVALS_TABLE_NAME; -if (!TASK_APPROVALS_TABLE_NAME) { - throw new Error('get-pending handler requires TASK_APPROVALS_TABLE_NAME env var'); +const TASK_TABLE_NAME = process.env.TASK_TABLE_NAME ?? ''; +if (!TASK_APPROVALS_TABLE_NAME || !TASK_TABLE_NAME) { + throw new Error('get-pending handler requires TASK_APPROVALS_TABLE_NAME and TASK_TABLE_NAME env vars'); } const USER_STATUS_INDEX_NAME = process.env.USER_STATUS_INDEX_NAME ?? 'user_id-status-index'; const PENDING_RATE_LIMIT_PER_MINUTE = Number(process.env.PENDING_RATE_LIMIT_PER_MINUTE ?? '10'); const PENDING_LIST_LIMIT = 100; +const PENDING_READ_TIMEOUT_MS = 5_000; +const TASK_READ_ATTEMPTS = 3; /** * GET /v1/pending — List pending approvals owned by the caller (§7.7). @@ -100,20 +103,16 @@ export async function handler(event: APIGatewayProxyEvent): Promise>; - const pending: PendingApprovalSummary[] = items.map((row) => { + const { liveItems, omitted } = await readLivePendingRows(userId); + if (omitted > 0) { + logger.info('Omitted approvals whose tasks are no longer waiting for them', { + event: 'pending_approvals_inactive', + user_id: userId, + omitted, + request_id: requestId, + }); + } + const pending: PendingApprovalSummary[] = liveItems.map((row) => { const created_at = String(row.created_at ?? ''); const timeout_s = Number(row.timeout_s ?? 0); const expires_at = computeExpiresAt(created_at, timeout_s); @@ -148,6 +147,71 @@ export async function handler(event: APIGatewayProxyEvent): Promise>; + omitted: number; +}> { + const liveItems: Array> = []; + let omitted = 0; + let lastKey: QueryCommandInput['ExclusiveStartKey']; + const abortSignal = AbortSignal.timeout(PENDING_READ_TIMEOUT_MS); + do { + // The limit applies to displayed requests, not stale rows examined. Older + // cancellations can otherwise fill page one and hide every current request. + abortSignal.throwIfAborted(); + const result = await ddb.send(new QueryCommand({ + TableName: TASK_APPROVALS_TABLE_NAME, + IndexName: USER_STATUS_INDEX_NAME, + KeyConditionExpression: 'user_id = :user AND #status = :pending', + ExpressionAttributeNames: { '#status': 'status' }, + ExpressionAttributeValues: { ':user': userId, ':pending': 'PENDING' }, + Limit: PENDING_LIST_LIMIT - liveItems.length, + ...(lastKey && { ExclusiveStartKey: lastKey }), + }), { abortSignal }); + const items = (result.Items ?? []) as ReadonlyArray>; + const tasks = await readWaitingTasks(items, abortSignal); + for (const row of items) { + const task = tasks.get(String(row.task_id)); + // Bind the eventually consistent GSI row to explicit task/request state. + // This does not assess whether the requested action remains relevant. + if (task?.user_id === userId && task.status === 'AWAITING_APPROVAL' + && task.awaiting_approval_request_id === row.request_id) { + liveItems.push(row); + } else { + omitted++; + } + } + lastKey = result.LastEvaluatedKey; + } while (lastKey && liveItems.length < PENDING_LIST_LIMIT); + return { liveItems, omitted }; +} + +async function readWaitingTasks( + items: ReadonlyArray>, + abortSignal: AbortSignal, +): Promise>> { + const tasks = new Map>(); + let keys = [...new Set(items.map(row => String(row.task_id)))].map(task_id => ({ task_id })); + for (let attempt = 0; keys.length > 0 && attempt < TASK_READ_ATTEMPTS; attempt++) { + const response = await ddb.send(new BatchGetCommand({ + RequestItems: { + [TASK_TABLE_NAME]: { + Keys: keys, + ConsistentRead: true, + ProjectionExpression: 'task_id, user_id, #status, awaiting_approval_request_id', + ExpressionAttributeNames: { '#status': 'status' }, + }, + }, + }), { abortSignal }); + for (const task of response.Responses?.[TASK_TABLE_NAME] ?? []) { + tasks.set(String(task.task_id), task); + } + keys = (response.UnprocessedKeys?.[TASK_TABLE_NAME]?.Keys ?? []) as typeof keys; + } + if (keys.length > 0) throw new Error('Pending approval task reads remained unprocessed'); + return tasks; +} + function coerceSeverity(value: unknown): Severity { if (value === 'low' || value === 'medium' || value === 'high') { return value; diff --git a/cdk/src/handlers/shared/approval-notifications.ts b/cdk/src/handlers/shared/approval-notifications.ts new file mode 100644 index 000000000..8158287b7 --- /dev/null +++ b/cdk/src/handlers/shared/approval-notifications.ts @@ -0,0 +1,171 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import { GetCommand, UpdateCommand, type DynamoDBDocumentClient } from '@aws-sdk/lib-dynamodb'; +import { scanDenyReason } from './deny-reason-scanner'; +import { logger } from './logger'; +import type { TaskRecord } from './types'; +import { TaskStatus, TERMINAL_STATUSES } from '../../constructs/task-status'; + +const REQUEST_ID_MAX_LENGTH = 128; +const SEVERITY_MAX_LENGTH = 20; +const SCOPE_MAX_LENGTH = 150; + +/** Decisions are acknowledged at commit time, even when the worker cannot wake. */ +export const APPROVAL_NOTIFICATION_EVENTS = [ + 'approval_requested', 'approval_decision_recorded', 'approval_timed_out', + 'approval_cancelled', 'approval_stranded', +] as const; + +export function isApprovalNotification(eventType: string): boolean { + return (APPROVAL_NOTIFICATION_EVENTS as readonly string[]).includes(eventType); +} + +export interface ApprovalNotification { + readonly title: string; + readonly text: string; + readonly taskId: string; + readonly requestId: string; + readonly userId: string; + readonly marker: string; +} + +function text(value: unknown, max = 500): string { + if (typeof value !== 'string') return ''; + // Redact before truncation so cutting a token cannot hide its recognizable shape. + return scanDenyReason(value) + .replace(/\u001b\[[0-?]*[ -/]*[@-~]/g, '') + .replace(/[\u0000-\u0008\u000b-\u001f\u007f\u200e\u200f\u202a-\u202e\u2066-\u2069]/g, '') + .slice(0, max); +} + +/** Read saved state instead of showing an old, delayed "please approve" event. */ +export async function loadApprovalNotification( + ddb: DynamoDBDocumentClient, + task: TaskRecord, + eventType: string, + metadata: Record, + channel: 'slack' | 'linear', +): Promise { + if (!isApprovalNotification(eventType)) return null; + const tableName = process.env.TASK_APPROVALS_TABLE_NAME; + if (!tableName) throw new Error('Approval notifications require TASK_APPROVALS_TABLE_NAME'); + // The deployed stranded-task reconciler closes the task, leaves its approval + // row PENDING, and emits a milestone without request_id. Recover that identity + // only from the consistently read, failed owning task and its saved cause. + const legacyStranded = eventType === 'approval_stranded' + && task.status === TaskStatus.FAILED + && metadata.reason === 'STRANDED_NO_HEARTBEAT' + && task.error_message?.startsWith('Approval stranded:'); + const requestId = metadata.request_id ?? (legacyStranded ? task.awaiting_approval_request_id : undefined); + if (!requestId && eventType === 'approval_stranded') { + logger.warn('Stranded approval notification has no saved request identity', { + event: 'approval_notification_request_missing', task_id: task.task_id, + }); + return null; + } + if (typeof requestId !== 'string' || !requestId || requestId.length > REQUEST_ID_MAX_LENGTH) { + throw new Error('Approval notification is missing a valid request_id'); + } + const response = await ddb.send(new GetCommand({ + TableName: tableName, Key: { task_id: task.task_id, request_id: requestId }, ConsistentRead: true, + })); + const row = response.Item; + const marker = `notified_${channel}_${eventType}`; + if (!row || row.user_id !== task.user_id || row[marker]) return null; + const status = legacyStranded && row.status === 'PENDING' + && task.awaiting_approval_request_id === requestId ? 'STRANDED' : row.status; + const expected = eventType === 'approval_requested' ? ['PENDING'] + : eventType === 'approval_decision_recorded' ? ['APPROVED', 'DENIED'] + : eventType === 'approval_timed_out' ? ['TIMED_OUT'] + : eventType === 'approval_cancelled' ? ['CANCELLED'] : ['STRANDED']; + if (!expected.includes(status)) return null; + if (status === 'PENDING' + && (task.status !== 'AWAITING_APPROVAL' || task.awaiting_approval_request_id !== requestId)) return null; + + const title = status === 'PENDING' ? 'Approval needed' + : status === 'APPROVED' ? 'Approval recorded' + : status === 'DENIED' ? 'Denial recorded' + : status === 'TIMED_OUT' ? 'Approval request timed out' + : status === 'CANCELLED' ? 'Approval request cancelled' : 'Approval wait could not continue'; + const lines = [title, `Task: ${task.task_id}`, `Request: ${requestId}`]; + if (status === 'PENDING') { + lines.push(`Tool: ${text(row.tool_name, 100)}`, `Severity: ${text(row.severity, SEVERITY_MAX_LENGTH)}`, + `Reason: ${text(row.reason)}`, `Action preview: ${text(row.tool_input_preview)}`); + const deadline = Date.parse(row.created_at) + Number(row.timeout_s) * 1000; + if (Number.isFinite(deadline) && Number(row.timeout_s) > 0) { + lines.push(`Decision deadline: ${new Date(deadline).toISOString()}`); + } + // Never interpolate untrusted text into suggested shell commands. + if ([task.task_id, requestId].every(id => /^[A-Za-z0-9_-]{1,128}$/.test(id))) { + lines.push('Respond using the CLI while signed in as the task owner:', + `bgagent approve ${task.task_id} ${requestId} --scope this_call`, + `bgagent deny ${task.task_id} ${requestId}`); + } + lines.push('Run bgagent pending to see currently open requests.'); + } else if (status === 'APPROVED') { + lines.push(`Scope: ${text(row.scope, SCOPE_MAX_LENGTH)}`); + lines.push(TERMINAL_STATUSES.includes(task.status) + ? `The decision is saved. Task status: ${task.status}. The task has already ended.` + : 'The decision is saved. The agent will continue when its worker is ready.'); + } else if (status === 'DENIED') { + lines.push(`Reason: ${text(row.deny_reason) || 'No reason supplied.'}`); + lines.push(TERMINAL_STATUSES.includes(task.status) + ? `The decision is saved. Task status: ${task.status}. The task has already ended.` + : 'The decision is saved. The agent will receive the denial when its worker is ready.'); + } else { + // reason is the original policy explanation, not the cause of closure. + // The guest stores a polling failure in deny_reason on its timeout path. + const reason = status === 'TIMED_OUT' + ? text(row.deny_reason) || 'The configured decision deadline was reached.' + : status === 'CANCELLED' + ? text(row.cancellation_reason) || 'Task cancelled by its owner.' + : text(task.error_message) || 'The agent could not continue this approval wait.'; + lines.push(`Reason: ${reason}`); + } + return { title, text: lines.join('\n'), taskId: task.task_id, requestId, userId: task.user_id, marker }; +} + +/** Persist only after the external service accepted the message, so failures retry. */ +export async function markApprovalNotificationDelivered( + ddb: DynamoDBDocumentClient, + notification: ApprovalNotification, +): Promise { + try { + await ddb.send(new UpdateCommand({ + TableName: process.env.TASK_APPROVALS_TABLE_NAME!, + Key: { task_id: notification.taskId, request_id: notification.requestId }, + UpdateExpression: 'SET #marker = :now', + ConditionExpression: 'attribute_exists(task_id) AND user_id = :user', + ExpressionAttributeNames: { '#marker': notification.marker }, + ExpressionAttributeValues: { ':now': new Date().toISOString(), ':user': notification.userId }, + })); + } catch (error) { + if ((error as { name?: string }).name !== 'ConditionalCheckFailedException') throw error; + // TTL removal after a successful post must not recreate the approval row. + logger.info('Approval disappeared before its notification receipt was saved', { + event: 'approval_notification_receipt_missing', task_id: notification.taskId, request_id: notification.requestId, + }); + } +} + +/** Keep previews literal: repository text must not create mentions or Markdown links. */ +export function approvalNotificationMarkdown(notification: ApprovalNotification): string { + return `\`\`\`text\n${notification.text.replace(/`/g, 'ˋ')}\n\`\`\``; +} diff --git a/cdk/src/handlers/shared/deny-reason-scanner.ts b/cdk/src/handlers/shared/deny-reason-scanner.ts index e38af68ed..b5f455038 100644 --- a/cdk/src/handlers/shared/deny-reason-scanner.ts +++ b/cdk/src/handlers/shared/deny-reason-scanner.ts @@ -33,8 +33,8 @@ * * Keep the pattern set in sync with `agent/src/output_scanner.py`; the * agent-side scanner is the canonical source for PostToolUse output - * redaction, and this module is the REST-side port for the deny-reason - * path only. If the agent-side patterns change, update here too. + * redaction, and this module also protects approval notification previews. + * If the agent-side patterns change, update here too. */ interface SecretPattern { diff --git a/cdk/src/handlers/shared/microvm-lifecycle.ts b/cdk/src/handlers/shared/microvm-lifecycle.ts index 9c5d8593a..b8a88e5ce 100644 --- a/cdk/src/handlers/shared/microvm-lifecycle.ts +++ b/cdk/src/handlers/shared/microvm-lifecycle.ts @@ -77,7 +77,7 @@ export const MICROVM_LIFECYCLE_STORE_TIMEOUT_MS = 5_000; const ddb = makeDocClient(); const TASK_TABLE = process.env.TASK_TABLE_NAME!; const APPROVALS_TABLE = process.env.TASK_APPROVALS_TABLE_NAME!; -const APPROVAL_STATUSES: readonly ApprovalStatus[] = ['PENDING', 'APPROVED', 'DENIED', 'TIMED_OUT', 'STRANDED']; +const APPROVAL_STATUSES: readonly ApprovalStatus[] = ['PENDING', 'APPROVED', 'DENIED', 'CANCELLED', 'TIMED_OUT', 'STRANDED']; const LIVE_TASK_STATUSES: readonly TaskStatusType[] = [TaskStatus.HYDRATING, TaskStatus.RUNNING, TaskStatus.AWAITING_APPROVAL]; function storeSignal(options?: SessionControlOptions): AbortSignal { diff --git a/cdk/src/handlers/shared/response.ts b/cdk/src/handlers/shared/response.ts index 4c92d1f03..70ce8ee5a 100644 --- a/cdk/src/handlers/shared/response.ts +++ b/cdk/src/handlers/shared/response.ts @@ -30,6 +30,7 @@ export const ErrorCode = { TRACE_NOT_AVAILABLE: 'TRACE_NOT_AVAILABLE', DUPLICATE_TASK: 'DUPLICATE_TASK', TASK_ALREADY_TERMINAL: 'TASK_ALREADY_TERMINAL', + TASK_STATE_CONFLICT: 'TASK_STATE_CONFLICT', RATE_LIMIT_EXCEEDED: 'RATE_LIMIT_EXCEEDED', WEBHOOK_NOT_FOUND: 'WEBHOOK_NOT_FOUND', WEBHOOK_ALREADY_REVOKED: 'WEBHOOK_ALREADY_REVOKED', diff --git a/cdk/src/handlers/shared/slack-blocks.ts b/cdk/src/handlers/shared/slack-blocks.ts index 86f116cc1..eaf5f47a4 100644 --- a/cdk/src/handlers/shared/slack-blocks.ts +++ b/cdk/src/handlers/shared/slack-blocks.ts @@ -39,7 +39,7 @@ interface PlainText { /** Section block: a single line/paragraph of mrkdwn content. */ export interface SectionBlock { readonly type: 'section'; - readonly text: MrkdwnText; + readonly text: MrkdwnText | PlainText; readonly block_id?: string; } diff --git a/cdk/src/handlers/shared/task-cancellation.ts b/cdk/src/handlers/shared/task-cancellation.ts new file mode 100644 index 000000000..ad3017f27 --- /dev/null +++ b/cdk/src/handlers/shared/task-cancellation.ts @@ -0,0 +1,175 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import { randomUUID } from 'node:crypto'; +import { GetCommand, TransactWriteCommand, UpdateCommand } from '@aws-sdk/lib-dynamodb'; +import { ulid } from 'ulid'; +import { logger } from './logger'; +import type { TaskRecord } from './types'; +import { makeDocClient } from './ua'; +import { computeTtlEpoch } from './validation'; +import { TaskStatus, TERMINAL_STATUSES } from '../../constructs/task-status'; + +const ddb = makeDocClient(); +const MAX_ATTEMPTS = 3; +const STATE_TIMEOUT_MS = 5_000; + +export class TaskCancellationError extends Error { + constructor(public readonly reason: 'missing' | 'forbidden' | 'terminal' | 'conflict') { + super(`Task cancellation ${reason}`); + this.name = 'TaskCancellationError'; + } +} + +interface CancellationOptions { + readonly userId: string; + readonly taskTable: string; + readonly approvalsTable?: string; + readonly eventsTable?: string; + readonly retentionDays: number; +} + +/** The latest pre-cancel record supplies the compute handle for active cleanup. */ +export interface TaskCancellationResult { + readonly task: TaskRecord; + readonly cancelledAt: string; + readonly cancelledRequestId?: string; +} + +function conditionalConflict(error: unknown): boolean { + const value = error as { name?: string; CancellationReasons?: Array<{ Code?: string }> }; + return value?.name === 'ConditionalCheckFailedException' + || (value?.name === 'TransactionCanceledException' + && Boolean(value.CancellationReasons?.some(reason => reason.Code === 'ConditionalCheckFailed'))); +} + +/** + * Cancel exactly the observed task/gate. Approval, timeout and gate changes race + * through conditional writes; a conflict reloads state before another attempt. + * A decision that already committed remains a decision, never a cancellation. + */ +export async function cancelTaskState( + initialTask: TaskRecord, + options: CancellationOptions, +): Promise { + const abortSignal = AbortSignal.timeout(STATE_TIMEOUT_MS); + let task = initialTask; + for (let attempt = 0; attempt < MAX_ATTEMPTS; attempt++) { + if (task.user_id !== options.userId) throw new TaskCancellationError('forbidden'); + if (TERMINAL_STATUSES.includes(task.status)) throw new TaskCancellationError('terminal'); + const requestId = task.awaiting_approval_request_id; + let closeApproval = false; + if (requestId) { + if (!options.approvalsTable || !options.eventsTable) { + throw new Error('Cancellation of an approval wait requires approvals and events tables'); + } + const response = await ddb.send(new GetCommand({ + TableName: options.approvalsTable, + Key: { task_id: task.task_id, request_id: requestId }, + ConsistentRead: true, + }), { abortSignal }); + const approval = response.Item; + closeApproval = approval?.status === 'PENDING' && approval.user_id === options.userId; + if (approval && approval.user_id !== options.userId) { + // A malformed approval must not prevent the owner stopping their task. + // Do not change another user's approval record. + logger.error('Cancellation found an approval owner mismatch', { + event: 'approval_cancel_owner_mismatch', task_id: task.task_id, request_id: requestId, + }); + } + } + const now = new Date().toISOString(); + const update = { + TableName: options.taskTable, + Key: { task_id: task.task_id }, + UpdateExpression: 'SET #status = :cancelled, updated_at = :now, completed_at = :now, status_created_at = :sca, #ttl = :ttl', + ConditionExpression: 'attribute_exists(task_id) AND user_id = :user AND #status = :observed AND ' + + (requestId + ? 'awaiting_approval_request_id = :request' + : '(attribute_not_exists(awaiting_approval_request_id) OR awaiting_approval_request_id = :request)'), + ExpressionAttributeNames: { '#status': 'status', '#ttl': 'ttl' }, + ExpressionAttributeValues: { + ':cancelled': TaskStatus.CANCELLED, + ':now': now, + ':sca': `${TaskStatus.CANCELLED}#${now}`, + ':ttl': computeTtlEpoch(options.retentionDays), + ':user': options.userId, + ':observed': task.status, + ':request': requestId ?? null, + }, + }; + try { + if (closeApproval) { + await ddb.send(new TransactWriteCommand({ + ClientRequestToken: randomUUID(), + TransactItems: [ + { Update: update }, + { + Update: { + TableName: options.approvalsTable!, + Key: { task_id: task.task_id, request_id: requestId! }, + UpdateExpression: 'SET #status = :cancelled, decided_at = :now, cancellation_reason = :reason', + ConditionExpression: '#status = :pending AND user_id = :user', + ExpressionAttributeNames: { '#status': 'status' }, + ExpressionAttributeValues: { + ':cancelled': 'CANCELLED', + ':pending': 'PENDING', + ':user': options.userId, + ':now': now, + ':reason': 'Task cancelled by its owner', + }, + }, + }, + { + Put: { + TableName: options.eventsTable!, + Item: { + task_id: task.task_id, + event_id: ulid(), + event_type: 'approval_cancelled', + timestamp: now, + ttl: computeTtlEpoch(options.retentionDays), + metadata: { + request_id: requestId, + status: 'CANCELLED', + reason: 'Task cancelled by its owner', + }, + }, + }, + }, + ], + }), { abortSignal }); + } else { + await ddb.send(new UpdateCommand(update), { abortSignal }); + } + return { task, cancelledAt: now, ...(closeApproval && { cancelledRequestId: requestId }) }; + } catch (error) { + if (!conditionalConflict(error)) throw error; + logger.info('Task cancellation raced with a state change; refreshing', { + event: 'task_cancel_retry', task_id: task.task_id, attempt: attempt + 1, + }); + const fresh = await ddb.send(new GetCommand({ + TableName: options.taskTable, Key: { task_id: task.task_id }, ConsistentRead: true, + }), { abortSignal }); + if (!fresh.Item) throw new TaskCancellationError('missing'); + task = fresh.Item as TaskRecord; + } + } + throw new TaskCancellationError('conflict'); +} diff --git a/cdk/src/handlers/shared/types.ts b/cdk/src/handlers/shared/types.ts index d291fb512..7933c9610 100644 --- a/cdk/src/handlers/shared/types.ts +++ b/cdk/src/handlers/shared/types.ts @@ -1307,14 +1307,14 @@ export type ApprovalStatus = | 'PENDING' | 'APPROVED' | 'DENIED' + | 'CANCELLED' | 'TIMED_OUT' | 'STRANDED'; /** - * Cedar HITL severity, surfaced in the CLI approval prompt and used - * for severity-gated channel routing (§11.2: high-severity rules - * skip Slack-button auto-approval). Shared alias so the same literal - * union is not redefined inline in `ApprovalRecord`, + * Cedar HITL severity, surfaced in approval prompts and notifications. + * Native Slack approval buttons remain unimplemented. Shared alias so the + * same literal union is not redefined inline in `ApprovalRecord`, * `PendingApprovalSummary`, `PolicyRuleSummary`, etc. */ export type Severity = 'low' | 'medium' | 'high'; @@ -1366,14 +1366,22 @@ export interface DeniedApprovalRecord extends ApprovalRecordBase { readonly deny_reason?: string; } -/** TIMED_OUT approval row — decided_at required (server-set). */ +/** CANCELLED approval row — its owning task was explicitly cancelled. */ +export interface CancelledApprovalRecord extends ApprovalRecordBase { + readonly status: 'CANCELLED'; + readonly decided_at: string; + readonly cancellation_reason: string; +} + +/** TIMED_OUT approval row — the guest's older timeout writer omits decided_at. */ export interface TimedOutApprovalRecord extends ApprovalRecordBase { readonly status: 'TIMED_OUT'; - readonly decided_at: string; + readonly decided_at?: string; + readonly deny_reason?: string; } -/** STRANDED approval row — decided_at required (set by the - * stranded-task reconciler). No user decision was ever recorded. */ +/** STRANDED approval row, when explicitly recorded. The current stranded-task + * reconciler closes the owning task and leaves its approval row PENDING. */ export interface StrandedApprovalRecord extends ApprovalRecordBase { readonly status: 'STRANDED'; readonly decided_at: string; @@ -1392,6 +1400,7 @@ export type ApprovalRecord = | PendingApprovalRecord | ApprovedApprovalRecord | DeniedApprovalRecord + | CancelledApprovalRecord | TimedOutApprovalRecord | StrandedApprovalRecord; diff --git a/cdk/src/handlers/slack-interactions.ts b/cdk/src/handlers/slack-interactions.ts index 85a1e552e..9040fa312 100644 --- a/cdk/src/handlers/slack-interactions.ts +++ b/cdk/src/handlers/slack-interactions.ts @@ -17,16 +17,20 @@ * SOFTWARE. */ -import { GetCommand, UpdateCommand } from '@aws-sdk/lib-dynamodb'; +import { GetCommand } from '@aws-sdk/lib-dynamodb'; import type { APIGatewayProxyEvent, APIGatewayProxyResult } from 'aws-lambda'; import { logger } from './shared/logger'; import { getSlackSecret, SLACK_SECRET_PREFIX, verifySlackRequest } from './shared/slack-verify'; +import { cancelTaskState, TaskCancellationError } from './shared/task-cancellation'; +import type { TaskRecord } from './shared/types'; import { makeDocClient } from './shared/ua'; const ddb = makeDocClient(); const SIGNING_SECRET_ARN = process.env.SLACK_SIGNING_SECRET_ARN!; const TASK_TABLE = process.env.TASK_TABLE_NAME!; +const APPROVALS_TABLE = process.env.TASK_APPROVALS_TABLE_NAME; +const EVENTS_TABLE = process.env.TASK_EVENTS_TABLE_NAME; const USER_MAPPING_TABLE = process.env.SLACK_USER_MAPPING_TABLE_NAME!; interface SlackInteractionPayload { @@ -119,6 +123,7 @@ async function handleCancelAction(payload: SlackInteractionPayload, actionId: st const taskResult = await ddb.send(new GetCommand({ TableName: TASK_TABLE, Key: { task_id: taskId }, + ConsistentRead: true, })); if (!taskResult.Item) { @@ -131,32 +136,19 @@ async function handleCancelAction(payload: SlackInteractionPayload, actionId: st return; } - // Attempt to cancel. QUEUED (#441) is cancellable — it removes the - // task from the admission queue (no compute or concurrency to release). - const CANCELLABLE_STATUSES = ['PENDING_UPLOADS', 'QUEUED', 'SUBMITTED', 'HYDRATING', 'RUNNING', 'AWAITING_APPROVAL', 'FINALIZING']; + // Share the REST path's atomic task/approval cancellation and race handling. try { - await ddb.send(new UpdateCommand({ - TableName: TASK_TABLE, - Key: { task_id: taskId }, - UpdateExpression: 'SET #s = :cancelled, updated_at = :now', - ConditionExpression: '#s IN (:s1, :s2, :s3, :s4, :s5, :s6, :s7)', - ExpressionAttributeNames: { '#s': 'status' }, - ExpressionAttributeValues: { - ':cancelled': 'CANCELLED', - ':now': new Date().toISOString(), - ':s1': CANCELLABLE_STATUSES[0], - ':s2': CANCELLABLE_STATUSES[1], - ':s3': CANCELLABLE_STATUSES[2], - ':s4': CANCELLABLE_STATUSES[3], - ':s5': CANCELLABLE_STATUSES[4], - ':s6': CANCELLABLE_STATUSES[5], - ':s7': CANCELLABLE_STATUSES[6], - }, - })); + const cancelled = await cancelTaskState(taskResult.Item as TaskRecord, { + userId: platformUserId, + taskTable: TASK_TABLE, + approvalsTable: APPROVALS_TABLE, + eventsTable: EVENTS_TABLE, + retentionDays: Number(process.env.TASK_RETENTION_DAYS ?? '90'), + }); // Instant feedback: replace the Cancel button message with "Cancelling..." // then clean up all intermediate messages. - const channelMeta = taskResult.Item.channel_metadata as Record | undefined; + const channelMeta = cancelled.task.channel_metadata as Record | undefined; const channelId = payload.channel?.id ?? channelMeta?.slack_channel_id; if (channelMeta && channelId) { const botToken = await getSlackSecret(`${SLACK_SECRET_PREFIX}${teamId}`); @@ -172,8 +164,10 @@ async function handleCancelAction(payload: SlackInteractionPayload, actionId: st } } } catch (err) { - if ((err as Error)?.name === 'ConditionalCheckFailedException') { - await postToResponseUrl(payload.response_url, ':warning: Task is already in a terminal state.'); + if (err instanceof TaskCancellationError) { + await postToResponseUrl(payload.response_url, err.reason === 'conflict' + ? ':warning: Task state changed. Please retry cancellation.' + : ':warning: Task is no longer available for cancellation.'); } else { throw err; } diff --git a/cdk/src/handlers/slack-notify.ts b/cdk/src/handlers/slack-notify.ts index 4b391ac70..61242f83c 100644 --- a/cdk/src/handlers/slack-notify.ts +++ b/cdk/src/handlers/slack-notify.ts @@ -51,8 +51,9 @@ import { type DynamoDBDocumentClient, GetCommand, UpdateCommand } from '@aws-sdk // so importing the type back creates no runtime cycle. ``import type`` // is erased after compile, so the bundler sees a one-way dep. import type { FanOutEvent } from './fanout-task-events'; +import { APPROVAL_NOTIFICATION_EVENTS, isApprovalNotification, loadApprovalNotification, markApprovalNotificationDelivered } from './shared/approval-notifications'; import { logger } from './shared/logger'; -import { renderSlackBlocks } from './shared/slack-blocks'; +import { renderSlackBlocks, type SlackMessage } from './shared/slack-blocks'; import { getSlackSecret, SLACK_SECRET_PREFIX } from './shared/slack-verify'; import type { TaskRecord } from './shared/types'; @@ -97,8 +98,8 @@ const SLACK_DEDUP_ATTRIBUTE: Record = { * Slack entries in ``CHANNEL_DEFAULTS`` (see fanout-task-events.ts) — * drift means the router subscribes Slack to events that the * dispatcher silently ignores, which lies in batch telemetry - * (issue #64 review Cat 7). Forward-compat ``approval_required`` and - * ``status_response`` are deliberately absent until their emitters + * (issue #64 review Cat 7). Forward-compat ``status_response`` is + * deliberately absent until its emitter * ship; until then they fall through and are dropped at this gate. * ``pr_created`` is intentionally omitted from Slack — the * ``task_completed`` block already carries the View PR button, so a @@ -114,6 +115,7 @@ export const NOTIFIABLE_EVENTS = new Set([ 'task_timed_out', 'task_stranded', 'agent_error', + ...APPROVAL_NOTIFICATION_EVENTS, ]); /** @@ -229,10 +231,15 @@ export async function dispatchSlackEvent( const taskResult = await ddb.send(new GetCommand({ TableName: tableName, Key: { task_id: taskId }, + ConsistentRead: true, })); const task = taskResult.Item as TaskRecord | undefined; if (!task || task.channel_source !== 'slack') return; + const approval = isApprovalNotification(eventType) + ? await loadApprovalNotification(ddb, task, eventType, event.metadata ?? {}, 'slack') : null; + if (isApprovalNotification(eventType) && !approval) return; + // Dedup any event that should only ever post once per task even // under partial-batch retry (terminals, agent_error). The orchestrator // can also write multiple events of the same kind (retries, @@ -290,7 +297,10 @@ export async function dispatchSlackEvent( // ``metadata: { S: ... }`` shape itself from the raw stream record. const eventMetadata = event.metadata; - const message = renderSlackBlocks(eventType, task, eventMetadata); + const message: SlackMessage = approval ? { + text: `${approval.title} for task ${taskId}`, + blocks: [{ type: 'section', text: { type: 'plain_text', text: approval.text } }], + } : renderSlackBlocks(eventType, task, eventMetadata); const threadTs = channelMeta.slack_thread_ts; @@ -348,6 +358,8 @@ export async function dispatchSlackEvent( throw new SlackApiError(failureMessage); } + if (approval) await markApprovalNotificationDelivered(ddb, approval); + // Reactions always use the real channel id even for DMs. const reactionChannel = channelMeta.slack_channel_id; const reactionTarget = threadTs ?? result.ts; diff --git a/cdk/src/stacks/agent.ts b/cdk/src/stacks/agent.ts index feb5d52c9..f0364fa12 100644 --- a/cdk/src/stacks/agent.ts +++ b/cdk/src/stacks/agent.ts @@ -55,9 +55,11 @@ import { IterationHeartbeat } from '../constructs/iteration-heartbeat'; import { JiraIntegration } from '../constructs/jira-integration'; import { LambdaMicrovmCompute, + createMicrovmExecutionRole, isLambdaMicrovmImageConfigured, type LambdaMicrovmImageInputs, } from '../constructs/lambda-microvm-compute'; +import { LambdaMicrovmStack } from '../constructs/lambda-microvm-stack'; import { LinearIdentityVault } from '../constructs/linear-identity-vault'; import { LinearIntegration } from '../constructs/linear-integration'; import { LinearVaultConsentPageStack } from '../constructs/linear-vault-consent-page'; @@ -346,6 +348,13 @@ export class AgentStack extends Stack { // MicroVM termination grant (ADR-021 sub-decision 4). const computeType = this.node.tryGetContext('compute_type') ?? 'agentcore'; const lambdaMicrovmEnabled = computeType === 'lambda-microvm'; + const microvmNestedContext = this.node.tryGetContext('microvm_nested_stack'); + if (microvmNestedContext !== undefined && ![true, false, 'true', 'false'].includes(microvmNestedContext)) { + throw new Error('microvm_nested_stack must be true or false'); + } + // Existing flat deployments retain their resource paths with false until + // their reviewed resource-migration procedure is complete. + const microvmNested = microvmNestedContext !== false && microvmNestedContext !== 'false'; const suspendContext = this.node.tryGetContext('microvm_approval_suspend_enabled'); if (suspendContext !== undefined && ![true, false, 'true', 'false'].includes(suspendContext)) { throw new Error('microvm_approval_suspend_enabled must be true or false'); @@ -1106,33 +1115,47 @@ export class AgentStack extends Stack { // agentcore-only. The construct itself enforces the ADR's Region gate, so a // deploy into a Region without Lambda MicroVMs fails at synth rather than on // the first task. - const lambdaMicrovm = lambdaMicrovmEnabled - ? new LambdaMicrovmCompute(this, 'LambdaMicrovmCompute', { - vpc: agentVpc.vpc, - // Per-session IAM scoping (#209): the MicroVM execution role is admitted - // to the same per-task SessionRole the AgentCore runtime and the Fargate - // task role use, so tenant-data access is tag-scoped on every substrate. - agentSessionRole, - // ADR-021 P2 runtime parity on the MicroVM execution role. Same two props - // EcsAgentCluster takes, for the same reasons: the PAT is read at startup - // before the SessionRole is assumed, and MEMORY_ID (already delivered in - // agent_payload) makes the agent ATTEMPT a memory write that fails closed - // without the grant. The remaining parity grants (channel OAuth, Bedrock, - // AZ describe) need no stack input and are wired inside the construct. - githubTokenSecret, - agentMemory, - // ADR-021 P2-F4: the SAME log group whose name travels to the guest in - // `agentPlatformConfig.logGroupName` below (→ `LOG_GROUP_NAME`). P2 - // delivered the name without the grant, so the agent's structured per-task - // lines and its METRICS_REPORT were AccessDenied on - // logs:CreateLogStream and the platform's canonical observability streams - // were empty on this backend. Passing the construct (not the name) keeps the - // grant and the delivered value derived from one object. - applicationLogGroup, - // Resolved above TaskApi — see `microvmImageInputs`. - ...microvmImageInputs, - }) - : undefined; + const microvmProps = { + vpc: agentVpc.vpc, + // Per-session IAM scoping (#209): the MicroVM execution role is admitted + // to the same per-task SessionRole the AgentCore runtime and the Fargate + // task role use, so tenant-data access is tag-scoped on every substrate. + agentSessionRole, + // ADR-021 P2 runtime parity on the MicroVM execution role. Same two props + // EcsAgentCluster takes, for the same reasons: the PAT is read at startup + // before the SessionRole is assumed, and MEMORY_ID (already delivered in + // agent_payload) makes the agent ATTEMPT a memory write that fails closed + // without the grant. The remaining parity grants (channel OAuth, Bedrock, + // AZ describe) need no stack input and are wired inside the construct. + githubTokenSecret, + agentMemory, + // ADR-021 P2-F4: the SAME log group whose name travels to the guest in + // `agentPlatformConfig.logGroupName` below (→ `LOG_GROUP_NAME`). P2 + // delivered the name without the grant, so the agent's structured per-task + // lines and its METRICS_REPORT were AccessDenied on + // logs:CreateLogStream and the platform's canonical observability streams + // were empty on this backend. Passing the construct (not the name) keeps the + // grant and the delivered value derived from one object. + applicationLogGroup, + // Resolved above TaskApi — see `microvmImageInputs`. + ...microvmImageInputs, + }; + let lambdaMicrovm: LambdaMicrovmCompute | undefined; + if (lambdaMicrovmEnabled) { + if (microvmNested) { + // Preserve the execution role's original parent path and therefore its + // logical ID. Moving it with the image creates a SessionRole trust cycle. + const roleScope = new Construct(this, 'LambdaMicrovmCompute'); + const executionRole = createMicrovmExecutionRole(roleScope, 'ExecutionRole'); + lambdaMicrovm = new LambdaMicrovmStack(this, 'Microvm', { + ...microvmProps, + deploymentName: this.stackName, + executionRole, + }).compute; + } else { + lambdaMicrovm = new LambdaMicrovmCompute(this, 'LambdaMicrovmCompute', microvmProps); + } + } // Resolve the image ARN used by TaskApi's cancel and wake grants. The invariant the // Lazy's `produce` guards: `microvmImageConfigured` (computed from the same @@ -1384,6 +1407,7 @@ export class AgentStack extends Stack { userPool: taskApi.userPool, taskTable: taskTable.table, taskEventsTable: taskEventsTable.table, + taskApprovalsTable: taskApprovalsTable.table, budgetTable: budgetTable.table, repoTable: repoTable.table, orchestratorFunctionArn: orchestrator.alias.functionArn, @@ -1882,6 +1906,7 @@ export class AgentStack extends Stack { const fanOutConsumer = new FanOutConsumer(this, 'FanOutConsumer', { taskEventsTable: taskEventsTable.table, taskTable: taskTable.table, + taskApprovalsTable: taskApprovalsTable.table, repoTable: repoTable.table, githubTokenSecret, // Slack bot-token grant is guarded on this prop — pass the diff --git a/cdk/test/bootstrap/__snapshots__/version.test.ts.snap b/cdk/test/bootstrap/__snapshots__/version.test.ts.snap index bb6744d54..cf7c721d9 100644 --- a/cdk/test/bootstrap/__snapshots__/version.test.ts.snap +++ b/cdk/test/bootstrap/__snapshots__/version.test.ts.snap @@ -1,3 +1,3 @@ // Jest Snapshot v1, https://jestjs.io/docs/snapshot-testing -exports[`bootstrap version module hash is stable 1`] = `"0f557449b9b55426b3ae54e28ca76807e2eb74ead7472ace5f035f590b651bd0"`; +exports[`bootstrap version module hash is stable 1`] = `"dc6301b65558c8fb51200e7b712a754e1b85adda7a512b14ec4f647dc954f541"`; diff --git a/cdk/test/bootstrap/bootstrap-template.test.ts b/cdk/test/bootstrap/bootstrap-template.test.ts index a3006fa00..5be0d5790 100644 --- a/cdk/test/bootstrap/bootstrap-template.test.ts +++ b/cdk/test/bootstrap/bootstrap-template.test.ts @@ -132,6 +132,8 @@ describe('Bootstrap template', () => { expect(passRole!.Resource).toEqual([ 'arn:aws:iam::*:role/backgroundagent-dev-LambdaMicrovmComputeBuild*', 'arn:aws:iam::*:role/backgroundagent-dev-LambdaMicrovmComputeConnector*', + 'arn:aws:iam::*:role/backgroundagent-dev-MicrovmBuildRole', + 'arn:aws:iam::*:role/backgroundagent-dev-MicrovmConnectorRole', ]); }); diff --git a/cdk/test/bootstrap/policies.test.ts b/cdk/test/bootstrap/policies.test.ts index 2abf0558e..ceafbcd2b 100644 --- a/cdk/test/bootstrap/policies.test.ts +++ b/cdk/test/bootstrap/policies.test.ts @@ -556,6 +556,8 @@ describe('computeLambdaMicrovmPolicy', () => { expect(resources).toEqual([ 'arn:aws:iam::*:role/backgroundagent-dev-LambdaMicrovmComputeBuild*', 'arn:aws:iam::*:role/backgroundagent-dev-LambdaMicrovmComputeConnector*', + 'arn:aws:iam::*:role/backgroundagent-dev-MicrovmBuildRole', + 'arn:aws:iam::*:role/backgroundagent-dev-MicrovmConnectorRole', ]); // NOT the stack-wide role prefix the conditioned statement uses: an // unconditioned pass on `backgroundagent-dev-*` would drop the @@ -601,6 +603,22 @@ describe('computeLambdaMicrovmPolicy', () => { expect(matched).toBe(true); } }); + + it('limits nested PassRole to the two explicit build and operator names', () => { + const resources = passRoleStatement().Resource as string[]; + const matches = (name: string) => resources.some(pattern => + new RegExp(`^${pattern.replace(/[.+?^${}()|[\]\\]/g, '\\$&').replace(/\*/g, '.*')}$`) + .test(`arn:aws:iam::123456789012:role/${name}`)); + expect(matches('backgroundagent-dev-MicrovmBuildRole')).toBe(true); + expect(matches('backgroundagent-dev-MicrovmConnectorRole')).toBe(true); + for (const name of [ + 'backgroundagent-dev-MicrovmExecutionRole', + 'backgroundagent-dev-MicrovmBuildRoleOther', + 'backgroundagent-dev-OtherBuildRole', + ]) { + expect(matches(name)).toBe(false); + } + }); }); it('covers both CFN resource types the construct synthesizes', () => { diff --git a/cdk/test/constructs/fanout-consumer.test.ts b/cdk/test/constructs/fanout-consumer.test.ts index 9a7b15149..52a6ae14a 100644 --- a/cdk/test/constructs/fanout-consumer.test.ts +++ b/cdk/test/constructs/fanout-consumer.test.ts @@ -39,6 +39,26 @@ function makeTaskEventsTable(stack: Stack): dynamodb.Table { } describe('FanOutConsumer', () => { + test('binds approval notifications to only GetItem and UpdateItem on the approvals table', () => { + const stack = new Stack(new App(), 'ApprovalNotifications'); + const table = new dynamodb.Table(stack, 'Approvals', { + partitionKey: { name: 'task_id', type: dynamodb.AttributeType.STRING }, + }); + new FanOutConsumer(stack, 'FanOut', { + taskEventsTable: makeTaskEventsTable(stack), taskApprovalsTable: table, + }); + const template = Template.fromStack(stack); + template.hasResourceProperties('AWS::Lambda::Function', { + Environment: { Variables: Match.objectLike({ TASK_APPROVALS_TABLE_NAME: stack.resolve(table.tableName) }) }, + }); + template.hasResourceProperties('AWS::IAM::Policy', { + PolicyDocument: { + Statement: Match.arrayWith([Match.objectLike({ + Action: ['dynamodb:GetItem', 'dynamodb:UpdateItem'], Resource: [stack.resolve(table.tableArn)], + })]), + }, + }); + }); test('attaches a single DynamoEventSource on the TaskEventsTable stream', () => { const app = new App(); const stack = new Stack(app, 'TestStack'); diff --git a/cdk/test/constructs/lambda-microvm-compute.test.ts b/cdk/test/constructs/lambda-microvm-compute.test.ts index ae3c85331..4d92364eb 100644 --- a/cdk/test/constructs/lambda-microvm-compute.test.ts +++ b/cdk/test/constructs/lambda-microvm-compute.test.ts @@ -990,8 +990,7 @@ describe('LambdaMicrovmCompute — image provisioned from a managed base image', }); test('distinguishes clean P2 smoke evidence from remaining acceptance and P3 hooks', () => { - // Coding, iteration and cancellation have live evidence. The warning must - // identify the remaining matrix and retain the served/undeclared hook list. + // Prior image evidence does not establish new-image or normal rollout acceptance. const warnings = built.construct.node.metadata.filter(m => m.type === 'aws:cdk:warning'); const message = warnings.map(w => String(w.data)).join('\n'); expect(JSON.stringify(built.construct.node.metadata)) @@ -1001,7 +1000,7 @@ describe('LambdaMicrovmCompute — image provisioned from a managed base image', .not.toContain('abca:microvm-image-p1-not-runnable'); expect(message).toContain('Clean P2 deployment'); expect(message).toContain('2026-09-14 without manual IAM changes'); - expect(message).toContain('failure/recovery, effective IAM and networking matrix'); + expect(message).toContain('This does not verify a different image'); expect(message).toContain('P2'); // It must state what IS true now, or it reads as the old (wrong) claim — and // the hook list here is what an operator compares against a failed build or a @@ -1011,7 +1010,8 @@ describe('LambdaMicrovmCompute — image provisioned from a managed base image', } expect(message).toContain('supervisor integration is implemented'); expect(message).toContain('P3 requires bootstrap bundle 1.8.0 and defaults new suspension off'); - expect(message).toContain('Live sleep/wake acceptance remains open'); + expect(message).toContain('Nested deployments require bundle 1.9.0'); + expect(message).toContain('Normal automatic-suspension activation remains open'); }); test('enables every hook the agent serves, and only those (rendered form)', () => { diff --git a/cdk/test/constructs/lambda-microvm-stack.test.ts b/cdk/test/constructs/lambda-microvm-stack.test.ts new file mode 100644 index 000000000..52ed743af --- /dev/null +++ b/cdk/test/constructs/lambda-microvm-stack.test.ts @@ -0,0 +1,170 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import { App, CfnOutput, NestedStack, Stack } from 'aws-cdk-lib'; +import { Match, Template } from 'aws-cdk-lib/assertions'; +import * as dynamodb from 'aws-cdk-lib/aws-dynamodb'; +import * as ec2 from 'aws-cdk-lib/aws-ec2'; +import * as iam from 'aws-cdk-lib/aws-iam'; +import * as s3 from 'aws-cdk-lib/aws-s3'; +import { Construct } from 'constructs'; +import { AgentSessionRole } from '../../src/constructs/agent-session-role'; +import { + createMicrovmExecutionRole, + LambdaMicrovmCompute, + type LambdaMicrovmImageInputs, +} from '../../src/constructs/lambda-microvm-compute'; +import { LambdaMicrovmStack } from '../../src/constructs/lambda-microvm-stack'; + +const ENV = { account: '123456789012', region: 'us-east-1' }; +const IMAGE_INPUTS: Array<[string, LambdaMicrovmImageInputs]> = [ + ['bootstrap', {}], + ['imported', { externalImageIdentifier: 'existing-image', externalImageVersion: '6.0' }], + ['managed', { + baseImageArn: 'arn:aws:lambda:us-east-1:aws:microvm-image:al2023-1', + baseImageVersion: '1', + artifactSha256: 'a'.repeat(64), + }], +]; + +describe.each(IMAGE_INPUTS)('LambdaMicrovmStack %s', (mode, imageInputs) => { + let parent: Template; + let child: Template; + let compute: LambdaMicrovmCompute; + + beforeAll(() => { + const stack = new Stack(new App(), 'backgroundagent-dev', { env: ENV }); + const vpc = new ec2.Vpc(stack, 'Vpc', { maxAzs: 2 }); + const executionRole = createMicrovmExecutionRole( + new Construct(stack, 'LambdaMicrovmCompute'), 'ExecutionRole', + ); + const sessionRole = new AgentSessionRole(stack, 'AgentSessionRole', { + assumingRoles: [new iam.Role(stack, 'RuntimeRole', { + assumedBy: new iam.ServicePrincipal('bedrock-agentcore.amazonaws.com'), + })], + taskTable: new dynamodb.Table(stack, 'Tasks', { + partitionKey: { name: 'task_id', type: dynamodb.AttributeType.STRING }, + }), + taskScopedTables: [], + traceArtifactsBucket: new s3.Bucket(stack, 'TraceBucket'), + attachmentsBucket: new s3.Bucket(stack, 'AttachmentsBucket'), + }); + const nested = new LambdaMicrovmStack(stack, 'Microvm', { + ...imageInputs, + vpc, + executionRole, + agentSessionRole: sessionRole, + deploymentName: stack.stackName, + }); + compute = nested.compute; + // Exercise actual parent consumers and child outputs, not just a detached child. + new CfnOutput(stack, 'ArtifactBucket', { value: compute.artifactBucket.bucketName }); + new CfnOutput(stack, 'PayloadBucket', { value: compute.payloadBucket.bucketArn }); + if (compute.imageArn) new CfnOutput(stack, 'ImageArn', { value: compute.imageArn }); + parent = Template.fromStack(stack); // Also rejects parent/child dependency cycles. + child = Template.fromStack(nested); + }); + + test('moves backend resources into the child, preserving parent-owned runtime trust', () => { + parent.resourceCountIs('AWS::Lambda::NetworkConnector', 0); + parent.resourceCountIs('AWS::Lambda::MicrovmImage', 0); + child.resourceCountIs('AWS::Lambda::NetworkConnector', 2); + child.resourceCountIs('AWS::S3::Bucket', 2); + child.resourceCountIs('AWS::Lambda::MicrovmImage', mode === 'managed' ? 1 : 0); + const parentRoles = parent.findResources('AWS::IAM::Role'); + const executionId = Object.keys(parentRoles).find(id => id.startsWith('LambdaMicrovmComputeExecutionRole'))!; + expect(executionId).toBeDefined(); + parent.hasResourceProperties('AWS::IAM::Role', { + AssumeRolePolicyDocument: { + Statement: Match.arrayWith([Match.objectLike({ + Principal: { AWS: { 'Fn::GetAtt': [executionId, 'Arn'] } }, + })]), + }, + }); + expect(Object.keys(child.findResources('AWS::IAM::Role')).some(id => id.includes('ExecutionRole'))).toBe(false); + }); + + test('uses concrete parent-derived service and narrowly scoped bootstrap role names', () => { + expect(compute.imageName).toBe('backgroundagent-dev-abca-agent'); + child.hasResourceProperties('AWS::Lambda::NetworkConnector', { Name: 'backgroundagent-dev-microvm-egress' }); + child.hasResourceProperties('AWS::Lambda::NetworkConnector', { Name: 'backgroundagent-dev-microvm-build-egress' }); + child.hasResourceProperties('AWS::IAM::Role', { RoleName: 'backgroundagent-dev-MicrovmBuildRole' }); + child.hasResourceProperties('AWS::IAM::Role', { RoleName: 'backgroundagent-dev-MicrovmConnectorRole' }); + expect(JSON.stringify(child.toJSON())).not.toContain('Token-TOKEN'); + }); + + test('retains build/runtime network separation and backend tags', () => { + const groups = Object.values(child.findResources('AWS::EC2::SecurityGroup')); + const egressPorts = groups.map(group => + group.Properties.SecurityGroupEgress.map((rule: { FromPort: number }) => rule.FromPort).sort()); + expect(egressPorts).toEqual(expect.arrayContaining([[443], [443, 80]])); + child.hasResourceProperties('AWS::Lambda::NetworkConnector', { + Tags: Match.arrayWith([{ Key: 'abca:compute-backend', Value: 'lambda-microvm' }]), + }); + // CDK's raw cleanup-provider resources inherit tags from CloudFormation; + // they do not expose a CDK TagManager that emits resource-level Tags. + parent.hasResourceProperties('AWS::CloudFormation::Stack', { + Tags: Match.arrayWith([{ Key: 'abca:compute-backend', Value: 'lambda-microvm' }]), + }); + const executionPolicies = Object.entries(parent.findResources('AWS::IAM::Policy')) + .filter(([id]) => id.startsWith('LambdaMicrovmComputeExecutionRole')); + expect(executionPolicies).toHaveLength(1); + expect(JSON.stringify(executionPolicies)).not.toContain('dynamodb:'); + }); + + test('preserves image selection semantics across the boundary', () => { + if (mode === 'bootstrap') { + expect(compute.imageIdentifier).toBeUndefined(); + } else if (mode === 'imported') { + expect(compute.imageArn).toContain(':microvm-image:existing-image'); + expect(compute.imageVersion).toBe('6.0'); + } else { + child.hasResourceProperties('AWS::Lambda::MicrovmImage', { + Name: 'backgroundagent-dev-abca-agent', + Resources: [{ MinimumMemoryInMiB: 8192 }], + Hooks: { MicrovmHooks: Match.objectLike({ Suspend: 'ENABLED', Resume: 'ENABLED' }) }, + }); + expect(compute.imageVersion).toBeUndefined(); + } + }); +}); + +test('rejects nested token sanitization and an execution role in the wrong stack', () => { + const app = new App(); + const parent = new Stack(app, 'Parent', { env: ENV }); + const vpc = new ec2.Vpc(parent, 'Vpc', { maxAzs: 2 }); + const child = new NestedStack(parent, 'UnconfiguredChild'); + expect(() => new LambdaMicrovmCompute(child, 'Compute', { vpc })) + .toThrow(/concrete deploymentName/); + const sibling = new Stack(app, 'Sibling', { env: ENV }); + expect(() => new LambdaMicrovmStack(parent, 'WrongRole', { + vpc, + deploymentName: 'Parent', + executionRole: createMicrovmExecutionRole(sibling, 'ExecutionRole'), + })).toThrow(/owned by its parent/); +}); + +test('rejects a deployment name that would exceed IAM role-name limits', () => { + const parent = new Stack(new App(), 'LongParent', { env: ENV }); + expect(() => new LambdaMicrovmStack(parent, 'Microvm', { + vpc: new ec2.Vpc(parent, 'Vpc', { maxAzs: 2 }), + deploymentName: 'a'.repeat(64), + executionRole: createMicrovmExecutionRole(parent, 'ExecutionRole'), + })).toThrow(/at most 64 characters/); +}); diff --git a/cdk/test/constructs/slack-integration.test.ts b/cdk/test/constructs/slack-integration.test.ts index 702e9bf30..12a02f575 100644 --- a/cdk/test/constructs/slack-integration.test.ts +++ b/cdk/test/constructs/slack-integration.test.ts @@ -47,11 +47,36 @@ describe('SlackIntegration construct', () => { userPool, taskTable, taskEventsTable, + taskApprovalsTable: dynamodb.Table.fromTableArn( + stack, 'Approvals', 'arn:aws:dynamodb:us-east-1:123456789012:table/Approvals', + ), }); template = Template.fromStack(stack); }); + test('interaction cancellation can settle approvals and publish their closure event', () => { + const functions = Object.entries(template.findResources('AWS::Lambda::Function')); + const interaction = functions.find(([id]) => id.includes('SlackInteractionsFn'))![1]; + expect(interaction.Properties.Environment.Variables).toMatchObject({ + TASK_APPROVALS_TABLE_NAME: 'Approvals', + TASK_EVENTS_TABLE_NAME: expect.anything(), + TASK_RETENTION_DAYS: expect.any(String), + }); + const policy = Object.entries(template.findResources('AWS::IAM::Policy')) + .find(([id]) => id.includes('SlackInteractionsFn'))![1]; + expect(policy.Properties.PolicyDocument.Statement).toEqual(expect.arrayContaining([ + expect.objectContaining({ + Action: ['dynamodb:GetItem', 'dynamodb:UpdateItem'], + Resource: ['arn:aws:dynamodb:us-east-1:123456789012:table/Approvals'], + }), + expect.objectContaining({ + Action: 'dynamodb:PutItem', + Resource: expect.arrayContaining([expect.objectContaining({ 'Fn::GetAtt': expect.arrayContaining([expect.stringContaining('TaskEventsTable')]) })]), + }), + ])); + }); + test('creates three Slack DynamoDB tables (installation + user mapping + channel mapping)', () => { // TaskTable + TaskEventsTable + SlackInstallation + SlackUserMapping + SlackChannelMapping = 5 template.resourceCountIs('AWS::DynamoDB::Table', 5); diff --git a/cdk/test/constructs/task-api-microvm-wake.test.ts b/cdk/test/constructs/task-api-microvm-wake.test.ts index ce745fb3f..4b45e7c37 100644 --- a/cdk/test/constructs/task-api-microvm-wake.test.ts +++ b/cdk/test/constructs/task-api-microvm-wake.test.ts @@ -79,3 +79,24 @@ test('the decision APIs keep their15s Lambda budget and read worker IDs from sav expect(resource.Properties.Environment.Variables.MICROVM_IMAGE_IDENTIFIER).toBeUndefined(); } }); + +test('cancellation and pending listing receive their approval workflow permissions', () => { + const functions = configured.findResources('AWS::Lambda::Function'); + const cancel = Object.entries(functions).find(([id]) => id.includes('CancelTaskFn'))![1]; + expect(cancel.Properties.Environment.Variables.TASK_APPROVALS_TABLE_NAME).toBeDefined(); + const policies = Object.entries(configured.findResources('AWS::IAM::Policy')); + const cancelPolicy = policies.find(([id]) => id.includes('CancelTaskFn'))![1]; + expect(cancelPolicy.Properties.PolicyDocument.Statement).toEqual(expect.arrayContaining([ + expect.objectContaining({ + Action: ['dynamodb:GetItem', 'dynamodb:UpdateItem'], + Resource: expect.arrayContaining([expect.objectContaining({ 'Fn::GetAtt': expect.arrayContaining([expect.stringContaining('Approvals')]) })]), + }), + ])); + const pendingPolicy = policies.find(([id]) => id.includes('GetPendingFn'))![1]; + expect(pendingPolicy.Properties.PolicyDocument.Statement).toEqual(expect.arrayContaining([ + expect.objectContaining({ + Action: 'dynamodb:BatchGetItem', + Resource: expect.arrayContaining([expect.objectContaining({ 'Fn::GetAtt': expect.arrayContaining([expect.stringContaining('Tasks')]) })]), + }), + ])); +}); diff --git a/cdk/test/handlers/cancel-task.test.ts b/cdk/test/handlers/cancel-task.test.ts index 5b4db883e..515fce01b 100644 --- a/cdk/test/handlers/cancel-task.test.ts +++ b/cdk/test/handlers/cancel-task.test.ts @@ -42,12 +42,14 @@ jest.mock('@aws-sdk/lib-dynamodb', () => ({ GetCommand: jest.fn((input: unknown) => ({ _type: 'Get', input })), UpdateCommand: jest.fn((input: unknown) => ({ _type: 'Update', input })), PutCommand: jest.fn((input: unknown) => ({ _type: 'Put', input })), + TransactWriteCommand: jest.fn((input: unknown) => ({ _type: 'TransactWrite', input })), })); jest.mock('ulid', () => ({ ulid: jest.fn(() => 'REQ-ULID') })); process.env.TASK_TABLE_NAME = 'Tasks'; process.env.TASK_EVENTS_TABLE_NAME = 'TaskEvents'; +process.env.TASK_APPROVALS_TABLE_NAME = 'Approvals'; process.env.TASK_RETENTION_DAYS = '90'; process.env.RUNTIME_ARN = 'arn:aws:bedrock-agentcore:us-east-1:123456789012:runtime/default'; process.env.ECS_CLUSTER_ARN = 'arn:aws:ecs:us-east-1:123456789012:cluster/agent-cluster'; @@ -127,6 +129,26 @@ beforeEach(() => { }); describe('cancel-task handler', () => { + test('closes an unanswered approval atomically before stopping its worker', async () => { + mockSend.mockReset(); + mockSend.mockResolvedValueOnce({ + Item: { + ...RUNNING_TASK, status: 'AWAITING_APPROVAL', awaiting_approval_request_id: 'request-1', + }, + }) + .mockResolvedValueOnce({ Item: { status: 'PENDING', user_id: RUNNING_TASK.user_id } }) + .mockResolvedValueOnce({}) + .mockResolvedValueOnce({}); + const response = await handler(makeEvent()); + expect(response.statusCode).toBe(200); + const transaction = mockSend.mock.calls.find(([command]) => command._type === 'TransactWrite')![0].input; + expect(transaction.TransactItems[0].Update.ExpressionAttributeValues[':cancelled']).toBe('CANCELLED'); + expect(transaction.TransactItems[1].Update.Key.request_id).toBe('request-1'); + expect(transaction.TransactItems[1].Update.ExpressionAttributeValues[':cancelled']).toBe('CANCELLED'); + expect(transaction.TransactItems[2].Put.Item.event_type).toBe('approval_cancelled'); + expect(mockAgentCoreSend).toHaveBeenCalledTimes(1); + }); + test.each(['HYDRATING', 'RUNNING', 'AWAITING_APPROVAL', 'FINALIZING'])('stops an AgentCore session when cancelling %s', async (status) => { mockSend.mockReset(); mockSend @@ -203,6 +225,7 @@ describe('cancel-task handler', () => { const condError = new Error('Condition not met'); condError.name = 'ConditionalCheckFailedException'; mockSend.mockRejectedValueOnce(condError); + mockSend.mockResolvedValueOnce({ Item: { ...RUNNING_TASK, status: 'COMPLETED' } }); const result = await handler(makeEvent()); diff --git a/cdk/test/handlers/fanout-task-events.test.ts b/cdk/test/handlers/fanout-task-events.test.ts index 2ef81be26..72df12d2e 100644 --- a/cdk/test/handlers/fanout-task-events.test.ts +++ b/cdk/test/handlers/fanout-task-events.test.ts @@ -165,6 +165,7 @@ jest.mock('../../src/handlers/shared/jira-feedback', () => ({ process.env.TASK_TABLE_NAME = 'Tasks'; process.env.GITHUB_TOKEN_SECRET_ARN = 'arn:aws:secretsmanager:us-east-1:0:secret:platform'; process.env.LINEAR_WORKSPACE_REGISTRY_TABLE_NAME = 'LinearWorkspaceRegistry'; +process.env.TASK_APPROVALS_TABLE_NAME = 'Approvals'; process.env.JIRA_WORKSPACE_REGISTRY_TABLE_NAME = 'JiraWorkspaceRegistry'; /** Flatten the stubbed ADF (`{ _adf: paragraphs }`) back to a newline-joined @@ -192,6 +193,8 @@ import { renderJiraFinishedPointer, } from '../../src/handlers/shared/jira-status-comment'; +let streamSequence = 1; + function mkRecord( eventName: 'INSERT' | 'MODIFY' | 'REMOVE', newImage: Record }> | undefined, @@ -200,7 +203,7 @@ function mkRecord( eventID: `evt-${Math.random().toString(36).slice(2)}`, eventName, eventSource: 'aws:dynamodb', - dynamodb: newImage ? { NewImage: newImage as never } : {}, + dynamodb: { SequenceNumber: String(streamSequence++), ...(newImage ? { NewImage: newImage as never } : {}) }, } as unknown as DynamoDBRecord; } @@ -308,8 +311,11 @@ describe('fanout-task-events: per-channel filter contract (design §6.2)', () => const f = CHANNEL_DEFAULTS.slack; expect([...f].sort()).toEqual([ 'agent_error', + 'approval_cancelled', + 'approval_decision_recorded', 'approval_requested', 'approval_stranded', + 'approval_timed_out', 'session_started', 'status_response', 'task_cancelled', @@ -328,23 +334,12 @@ describe('fanout-task-events: per-channel filter contract (design §6.2)', () => }); test('every Slack-default event the dispatcher actually renders today is in NOTIFIABLE_EVENTS (drift guard)', () => { - // The router subscribes Slack to events the dispatcher must - // render. ``approval_requested``, ``approval_stranded``, and - // ``status_response`` are forward-compat (no Slack-side renderer - // today — the CLI surfaces approval UX; Slack is only in the - // channel-defaults set so a future Slack-button renderer can - // light up without changing the router filter). They're allowed - // to be in CHANNEL_DEFAULTS.slack but absent from - // NOTIFIABLE_EVENTS — when their emitters land, this test will - // start failing and force the dispatcher update at the same time. - // Every OTHER Slack default must be renderable, otherwise - // telemetry lies. Use ``requireActual`` to bypass the - // slack-notify mock and read the real exported NOTIFIABLE_EVENTS - // set. + // Approval emitters are live and must have a renderer. Only status_response + // remains a placeholder. Read the actual module despite the dispatcher mock. const real = jest.requireActual( '../../src/handlers/slack-notify', ); - const forwardCompat = new Set(['approval_requested', 'approval_stranded', 'status_response']); + const forwardCompat = new Set(['status_response']); const expectedRenderable = [...CHANNEL_DEFAULTS.slack].filter( e => !forwardCompat.has(e), ); @@ -379,11 +374,16 @@ describe('fanout-task-events: per-channel filter contract (design §6.2)', () => ]); }); - test('Linear subscribes to pr_created + terminal events + task_timed_out (ADR-016 P4.5 courtesy comment + post-once final-status)', () => { + test('Linear subscribes to approvals, pr_created and terminal events', () => { // review should-fix: task_timed_out added so a Linear standalone iteration // that times out still settles (matches Jira/Slack, which already had it). const f = CHANNEL_DEFAULTS.linear; expect([...f].sort()).toEqual([ + 'approval_cancelled', + 'approval_decision_recorded', + 'approval_requested', + 'approval_stranded', + 'approval_timed_out', 'pr_created', 'task_cancelled', 'task_completed', @@ -890,7 +890,7 @@ describe('fanout-task-events: GitHub dispatcher (Chunk J)', () => { const event = { Records: [mkEvent('task_completed', 't-gh')] } as DynamoDBStreamEvent; const result = await handler(event); expect(result.batchItemFailures).toHaveLength(1); - expect(result.batchItemFailures[0].itemIdentifier).toBe(event.Records[0].eventID); + expect(result.batchItemFailures[0].itemIdentifier).toBe(event.Records[0].dynamodb!.SequenceNumber); // No UpdateCommand fires (no id to persist from a failed upsert). const updateCalls = mockDdbSend.mock.calls.filter( c => (c[0] as { _type?: string })._type === 'Update', @@ -1055,7 +1055,7 @@ describe('fanout-task-events: GitHub dispatcher (Chunk J)', () => { // Record IS in batchItemFailures — Lambda will replay until // the rate-limit window opens. Critical: the swallow-as-terminal // path would have produced an empty array (silent drop). - expect(result.batchItemFailures).toEqual([{ itemIdentifier: record.eventID }]); + expect(result.batchItemFailures).toEqual([{ itemIdentifier: record.dynamodb!.SequenceNumber }]); }, ); @@ -1383,7 +1383,7 @@ describe('fanout-task-events: Slack dispatcher', () => { const record = mkEvent('task_completed', 't-slack-fail'); const result = await handler({ Records: [record] }); - expect(result.batchItemFailures).toEqual([{ itemIdentifier: record.eventID }]); + expect(result.batchItemFailures).toEqual([{ itemIdentifier: record.dynamodb!.SequenceNumber }]); }); test('Slack dispatcher SlackApiError swallow does NOT escalate to retry', async () => { @@ -1483,6 +1483,54 @@ describe('fanout-task-events: Linear dispatcher', () => { }); }; + test.each([false, true])('routes wrapped approval to Linear without settling the task (post failure=%s)', async fails => { + mockDdbSend.mockImplementation(async command => { + if (command._type !== 'Get') return {}; + if (command.input.TableName === 'Approvals') { + return { + Item: { + user_id: 'u-1', + status: 'PENDING', + tool_name: 'Bash', + severity: 'medium', + reason: 'Review this', + tool_input_preview: 'git push', + timeout_s: 1800, + created_at: '2026-09-16T12:00:00Z', + }, + }; + } + return { + Item: { + ...TASK_RECORD_LINEAR, + status: 'AWAITING_APPROVAL', + awaiting_approval_request_id: 'g1', + channel_metadata: { ...TASK_RECORD_LINEAR.channel_metadata, trigger_comment_id: 'trigger', iteration_reply_comment_id: 'reply' }, + }, + }; + }); + if (fails) mockPostIssueComment.mockResolvedValueOnce({ ok: false, retryable: true }); + const outcome = await routeEvent({ + task_id: 't-lin', + event_id: 'approval-event', + event_type: 'agent_milestone', + timestamp: '2026-09-16T12:00:00Z', + metadata: { milestone: 'approval_requested', request_id: 'g1' }, + }); + expect(mockPostIssueComment).toHaveBeenCalledTimes(1); + expect(mockPostIssueComment.mock.calls[0][2]).toContain('bgagent approve t-lin g1 --scope this_call'); + expect(mockUpsertThreadedReply).not.toHaveBeenCalled(); + const updates = mockDdbSend.mock.calls.filter(([c]) => c._type === 'Update'); + if (fails) { + expect(outcome.infraRejections).toHaveLength(1); + expect(updates).toHaveLength(0); + } else { + expect(outcome.infraRejections).toHaveLength(0); + expect(updates).toHaveLength(1); + expect(updates[0][0].input.ExpressionAttributeNames).toEqual({ '#marker': 'notified_linear_approval_requested' }); + } + }); + test('task_completed posts ✅ comment with cost / turns / duration on linked Linear issue', async () => { mockGet(TASK_RECORD_LINEAR); @@ -1612,7 +1660,7 @@ describe('fanout-task-events: Linear dispatcher', () => { // Transient failure → record enters batchItemFailures so Lambda retries. expect(result.batchItemFailures).toHaveLength(1); - expect(result.batchItemFailures[0].itemIdentifier).toBe(event.Records[0].eventID); + expect(result.batchItemFailures[0].itemIdentifier).toBe(event.Records[0].dynamodb!.SequenceNumber); // Marker NOT persisted — the retry will re-post. const updateCalls = mockDdbSend.mock.calls.filter((c) => (c[0] as { _type?: string })._type === 'Update'); expect(updateCalls).toHaveLength(0); @@ -1717,7 +1765,7 @@ describe('fanout-task-events: Linear dispatcher', () => { expect(mockPostIssueComment).toHaveBeenCalledTimes(1); expect(result.batchItemFailures).toHaveLength(1); - expect(result.batchItemFailures[0]).toEqual({ itemIdentifier: records[0].eventID }); + expect(result.batchItemFailures[0]).toEqual({ itemIdentifier: records[0].dynamodb!.SequenceNumber }); // And no marker write: the retry must be allowed to post. const updates = mockDdbSend.mock.calls @@ -1902,8 +1950,8 @@ describe('fanout-task-events: Linear dispatcher', () => { // task — so the thumbnail renders without depending on the racy comment edit. const PNG = 'https://cdn.example/screenshots/iter.png'; const DEPLOY = 'https://app.vercel.app'; - mockDdbSend.mockReset().mockImplementation((cmd: { _type?: string; input?: { ConsistentRead?: boolean } }) => { - if (cmd?._type === 'Get' && cmd.input?.ConsistentRead) { + mockDdbSend.mockReset().mockImplementation((cmd: { _type?: string; input?: { ConsistentRead?: boolean; ProjectionExpression?: string } }) => { + if (cmd?._type === 'Get' && cmd.input?.ConsistentRead && cmd.input.ProjectionExpression === 'screenshot_url, screenshot_preview_url') { // the late re-read: screenshot has landed durably by now return Promise.resolve({ Item: { screenshot_url: PNG, screenshot_preview_url: DEPLOY } }); } @@ -2460,7 +2508,7 @@ describe('fanout-task-events: Jira dispatcher', () => { const result = await handler({ Records: [record] }); expect(result.batchItemFailures).toEqual([ - { itemIdentifier: record.eventID }, + { itemIdentifier: record.dynamodb!.SequenceNumber }, ]); const releases = mockDdbSend.mock.calls .map(([command]) => command as { @@ -2525,7 +2573,7 @@ describe('fanout-task-events: Jira dispatcher', () => { const result = await handler({ Records: [record] }); - expect(result.batchItemFailures).toEqual([{ itemIdentifier: record.eventID }]); + expect(result.batchItemFailures).toEqual([{ itemIdentifier: record.dynamodb!.SequenceNumber }]); expect(mockUpdateIssueCommentAdf).toHaveBeenCalledTimes(1); const releases = mockDdbSend.mock.calls .map(([command]) => command as { @@ -2619,7 +2667,7 @@ describe('fanout-task-events: Jira dispatcher', () => { expect(mockPostIssueCommentAdf).toHaveBeenCalledTimes(1); expect(result.batchItemFailures).toHaveLength(1); - expect(result.batchItemFailures[0]).toEqual({ itemIdentifier: records[0].eventID }); + expect(result.batchItemFailures[0]).toEqual({ itemIdentifier: records[0].dynamodb!.SequenceNumber }); // No marker write — the retry must be allowed to post. const updates = mockDdbSend.mock.calls @@ -3019,10 +3067,8 @@ describe('fanout-task-events: agent_milestone routing (effective event type)', ( // --------------------------------------------------------------------------- /** - * Stream record with a caller-supplied ``eventID`` so the test can - * assert which record surfaces in ``batchItemFailures``. ``mkEvent`` - * uses ``Math.random()`` for the id which is fine for parse tests but - * useless when we need to cross-reference the failure identifier. + * Stream record with distinct event and sequence identifiers. The event ID + * identifies log entries; only the numeric sequence is a valid retry cursor. */ function mkEventWithId(type: string, eventID: string, taskId = 't-fail'): DynamoDBRecord { return { @@ -3030,6 +3076,7 @@ function mkEventWithId(type: string, eventID: string, taskId = 't-fail'): Dynamo eventName: 'INSERT', eventSource: 'aws:dynamodb', dynamodb: { + SequenceNumber: String(streamSequence++), NewImage: { task_id: { S: taskId }, event_id: { S: `01ABC${type}` }, @@ -3045,9 +3092,8 @@ describe('fanout-task-events: partial-batch response', () => { // The construct sets ``reportBatchItemFailures: true`` on // the event-source-mapping, so the handler must return a batch response // rather than ``void``. Returning ``void`` makes Lambda retry the WHOLE - // batch on any unhandled throw — replaying every sibling event and - // defeating the per-task ordering guarantee promised upstream by - // ``ParallelizationFactor: 1``. + // batch on an unhandled throw. A sequence cursor avoids replaying + // successful earlier records; later records still need deduplication. // // The architecturally reachable poison-pill path is a // throw that bypasses ``routeEvent``'s ``Promise.allSettled``. The @@ -3063,6 +3109,7 @@ describe('fanout-task-events: partial-batch response', () => { beforeEach(() => { mockDdbSend.mockReset().mockResolvedValue({ Item: undefined }); + mockDispatchSlackEvent.mockReset().mockResolvedValue(undefined); mockUpsertTaskComment.mockReset(); mockRenderCommentBody.mockReset().mockReturnValue('rendered body'); mockLoadRepoConfig.mockReset().mockResolvedValue(null); @@ -3070,6 +3117,25 @@ describe('fanout-task-events: partial-batch response', () => { mockClearTokenCache.mockReset(); }); + test.each([true, false])('returns the AWS sequence cursor when eventID is present=%s', async hasEventId => { + const record = mkEventWithId('task_created', 'opaque-event-id'); + record.dynamodb!.SequenceNumber = '400000000000000000000001'; + if (!hasEventId) delete record.eventID; + mockDispatchSlackEvent.mockRejectedValueOnce(new Error('temporary delivery failure')); + await expect(handler({ Records: [record] })).resolves.toEqual({ + batchItemFailures: [{ itemIdentifier: '400000000000000000000001' }], + }); + }); + + test.each([undefined, ''])('rejects the batch when a failed record has an invalid sequence (%s)', async sequence => { + const record = mkEventWithId('task_created', 'opaque-event-id'); + record.dynamodb!.SequenceNumber = sequence; + mockDispatchSlackEvent.mockRejectedValueOnce(new Error('temporary delivery failure')); + await expect(handler({ Records: [record] })).rejects.toThrow( + 'Failed DynamoDB record is missing its sequence number', + ); + }); + test('AccessDeniedException from resolveTokenSecretArn lands in infraRejections and flags the record for retry', async () => { // An earlier version swallowed this rejection inside // ``Promise.allSettled``, which silently dropped transient infra @@ -3109,8 +3175,8 @@ describe('fanout-task-events: partial-batch response', () => { const result = await handler(event); // Record is flagged for partial-batch retry — Lambda will replay - // this single eventID, leaving siblings alone. - expect(result.batchItemFailures).toEqual([{ itemIdentifier: poisonId }]); + // from its sequence onward, including any successful later records. + expect(result.batchItemFailures).toEqual([{ itemIdentifier: event.Records[0].dynamodb!.SequenceNumber }]); // The rejection is observable through the dispatcher-rejected // warn so operators can alarm distinctly from the generic @@ -3131,11 +3197,8 @@ describe('fanout-task-events: partial-batch response', () => { // loop throws past ``routeEvent``'s containment (simulated here by // making ``logger.warn`` throw on the rate-limit path — the // closest real non-``routeEvent`` code path), the handler's - // per-record try/catch must push the record's ``eventID`` into - // ``batchItemFailures`` so Lambda retries ONLY that record. A handler - // that returned void would make Lambda retry the ENTIRE - // batch, replaying every sibling event and defeating per-task - // ordering. + // per-record try/catch must return its ``SequenceNumber`` so Lambda + // retries from the failed record onward. const loggerModule = await import('../../src/handlers/shared/logger'); // Rate-limit warn on the 21st event throws; earlier events succeed. let warnCalls = 0; @@ -3175,7 +3238,7 @@ describe('fanout-task-events: partial-batch response', () => { // from the handler's perspective (``routeEvent`` short-circuits // on "task not found" since the shared DDB mock returns no Item). expect(result.batchItemFailures).toEqual([ - { itemIdentifier: 'evt-20' }, + { itemIdentifier: records[20].dynamodb!.SequenceNumber }, ]); } finally { warnSpy.mockRestore(); @@ -3186,7 +3249,7 @@ describe('fanout-task-events: partial-batch response', () => { // Mixed batch: one record throws past routeEvent (via the same // rate-limit-warn trick as above but in a simpler shape — we make // the second record specifically trigger the throw), the other - // routes cleanly. The response must list ONLY the failing eventID. + // routes cleanly. The response must list only the failed sequence. const loggerModule = await import('../../src/handlers/shared/logger'); const warnSpy = jest.spyOn(loggerModule.logger, 'warn').mockImplementation( (_msg: string, meta?: Record) => { @@ -3206,9 +3269,9 @@ describe('fanout-task-events: partial-batch response', () => { const result = await handler({ Records: records }); expect(result.batchItemFailures).toHaveLength(1); - expect(result.batchItemFailures[0]).toEqual({ itemIdentifier: 'evt-chatty-20' }); + expect(result.batchItemFailures[0]).toEqual({ itemIdentifier: records[21].dynamodb!.SequenceNumber }); // Specifically NOT the successful record. - expect(result.batchItemFailures.map(f => f.itemIdentifier)).not.toContain('evt-ok'); + expect(result.batchItemFailures.map(f => f.itemIdentifier)).not.toContain(records[0].dynamodb!.SequenceNumber); } finally { warnSpy.mockRestore(); } diff --git a/cdk/test/handlers/get-pending.test.ts b/cdk/test/handlers/get-pending.test.ts index b19b438c5..66ee88f36 100644 --- a/cdk/test/handlers/get-pending.test.ts +++ b/cdk/test/handlers/get-pending.test.ts @@ -27,6 +27,7 @@ jest.mock('@aws-sdk/client-dynamodb', () => ({ jest.mock('@aws-sdk/lib-dynamodb', () => ({ DynamoDBDocumentClient: { from: jest.fn(() => ({ send: mockSend })) }, QueryCommand: jest.fn((input: unknown) => ({ _type: 'Query', input })), + BatchGetCommand: jest.fn((input: unknown) => ({ _type: 'BatchGet', input })), UpdateCommand: jest.fn((input: unknown) => ({ _type: 'Update', input })), })); @@ -34,6 +35,7 @@ let ulidCounter = 0; jest.mock('ulid', () => ({ ulid: jest.fn(() => `ULID${ulidCounter++}`) })); process.env.TASK_APPROVALS_TABLE_NAME = 'Approvals'; +process.env.TASK_TABLE_NAME = 'Tasks'; process.env.PENDING_RATE_LIMIT_PER_MINUTE = '10'; import { handler } from '../../src/handlers/get-pending'; @@ -76,21 +78,138 @@ beforeEach(() => { }); /** - * Two-mock chain used by every test that returns a non-empty pending - * list: (1) rate-limit Update passes, (2) GSI Query returns ``items``. - * - * ``matching_rule_ids`` is projected onto the ``user_id-status-index`` - * GSI directly (see the construct), so the handler maps each field - * from the Query result without a second read — keeping the mock - * chain short. + * GSI metadata plus strongly consistent reads of the owning task states. */ -function setupPendingMocks(items: ReadonlyArray>): void { +function setupPendingMocks( + items: ReadonlyArray>, + tasks = items.map(row => ({ + task_id: row.task_id, + user_id: 'user-alice', + status: 'AWAITING_APPROVAL', + awaiting_approval_request_id: row.request_id, + })), +): void { mockSend .mockResolvedValueOnce({}) // rate-limit - .mockResolvedValueOnce({ Items: items }); + .mockResolvedValueOnce({ Items: items }) + .mockResolvedValueOnce({ Responses: { Tasks: tasks } }); } describe('get-pending', () => { + test('finds a current request after a full page of cancelled legacy requests', async () => { + const oldRows = Array.from({ length: 100 }, (_, i) => ({ task_id: `old-${i}`, request_id: 'r' })); + const lastKey = { task_id: 'old-99', request_id: 'r', user_id: 'user-alice', status: 'PENDING' }; + mockSend.mockResolvedValueOnce({}) + .mockResolvedValueOnce({ Items: oldRows, LastEvaluatedKey: lastKey }) + .mockResolvedValueOnce({ + Responses: { + Tasks: oldRows.map(row => ({ + ...row, user_id: 'user-alice', status: 'CANCELLED', + })), + }, + }) + .mockResolvedValueOnce({ Items: [{ task_id: 'current', request_id: 'new' }] }) + .mockResolvedValueOnce({ + Responses: { + Tasks: [{ + task_id: 'current', + user_id: 'user-alice', + status: 'AWAITING_APPROVAL', + awaiting_approval_request_id: 'new', + }], + }, + }); + + const response = await handler(makeEvent()); + expect(response.statusCode).toBe(200); + expect(JSON.parse(response.body).data.pending).toEqual([ + expect.objectContaining({ task_id: 'current', request_id: 'new' }), + ]); + const queries = mockSend.mock.calls.filter(([command]) => command._type === 'Query'); + expect(queries).toHaveLength(2); + expect(queries[1][0].input.ExclusiveStartKey).toEqual(lastKey); + expect(queries[0][1].abortSignal).toBe(queries[1][1].abortSignal); + }); + + test('stops paging at the display limit even when the index has more rows', async () => { + const rows = Array.from({ length: 100 }, (_, i) => ({ task_id: `live-${i}`, request_id: 'r' })); + mockSend.mockResolvedValueOnce({}) + .mockResolvedValueOnce({ Items: rows, LastEvaluatedKey: { task_id: 'more' } }) + .mockResolvedValueOnce({ + Responses: { + Tasks: rows.map(row => ({ + task_id: row.task_id, + user_id: 'user-alice', + status: 'AWAITING_APPROVAL', + awaiting_approval_request_id: 'r', + })), + }, + }); + const response = await handler(makeEvent()); + expect(response.statusCode).toBe(200); + expect(JSON.parse(response.body).data.pending).toHaveLength(100); + expect(mockSend.mock.calls.filter(([command]) => command._type === 'Query')).toHaveLength(1); + }); + + test('reports a later-page read failure instead of returning a misleading empty list', async () => { + mockSend.mockResolvedValueOnce({}) + .mockResolvedValueOnce({ Items: [], LastEvaluatedKey: { task_id: 'more' } }) + .mockRejectedValueOnce(new Error('Later page unavailable')); + const response = await handler(makeEvent()); + expect(response.statusCode).toBe(500); + }); + + test.each(['CANCELLED', 'COMPLETED', 'FAILED', 'TIMED_OUT', 'RUNNING'])( + 'omits a legacy pending approval when its task is %s', + async (status) => { + setupPendingMocks([{ task_id: 't', request_id: 'r' }], [{ + task_id: 't', user_id: 'user-alice', status, awaiting_approval_request_id: 'r', + }]); + const response = await handler(makeEvent()); + expect(response.statusCode).toBe(200); + expect(JSON.parse(response.body).data.pending).toEqual([]); + const batch = mockSend.mock.calls.find(([command]) => command._type === 'BatchGet')![0].input; + expect(batch.RequestItems.Tasks.ConsistentRead).toBe(true); + }, + ); + + test('omits a replaced gate, missing task and mismatched owner', async () => { + setupPendingMocks([ + { task_id: 'old', request_id: 'r' }, + { task_id: 'missing', request_id: 'r' }, + { task_id: 'foreign', request_id: 'r' }, + ], [ + { task_id: 'old', user_id: 'user-alice', status: 'AWAITING_APPROVAL', awaiting_approval_request_id: 'new' }, + { task_id: 'foreign', user_id: 'someone-else', status: 'AWAITING_APPROVAL', awaiting_approval_request_id: 'r' }, + ]); + const response = await handler(makeEvent()); + expect(JSON.parse(response.body).data.pending).toEqual([]); + }); + + test('retries unprocessed task reads instead of dropping the request', async () => { + mockSend.mockResolvedValueOnce({}) + .mockResolvedValueOnce({ Items: [{ task_id: 't', request_id: 'r' }] }) + .mockResolvedValueOnce({ UnprocessedKeys: { Tasks: { Keys: [{ task_id: 't' }] } } }) + .mockResolvedValueOnce({ + Responses: { + Tasks: [{ + task_id: 't', user_id: 'user-alice', status: 'AWAITING_APPROVAL', awaiting_approval_request_id: 'r', + }], + }, + }); + const response = await handler(makeEvent()); + expect(response.statusCode).toBe(200); + expect(JSON.parse(response.body).data.pending).toHaveLength(1); + }); + + test('reports incomplete task reads as an error, not an empty pending list', async () => { + mockSend.mockResolvedValueOnce({}) + .mockResolvedValueOnce({ Items: [{ task_id: 't', request_id: 'r' }] }) + .mockResolvedValue({ UnprocessedKeys: { Tasks: { Keys: [{ task_id: 't' }] } } }); + const response = await handler(makeEvent()); + expect(response.statusCode).toBe(500); + }); + test('401 when no Cognito claims', async () => { const event = makeEvent(); (event.requestContext.authorizer as { claims: Record }).claims = {}; diff --git a/cdk/test/handlers/shared/approval-notifications.test.ts b/cdk/test/handlers/shared/approval-notifications.test.ts new file mode 100644 index 000000000..532fc87d3 --- /dev/null +++ b/cdk/test/handlers/shared/approval-notifications.test.ts @@ -0,0 +1,143 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import type { DynamoDBDocumentClient } from '@aws-sdk/lib-dynamodb'; +import { TaskStatus } from '../../../src/constructs/task-status'; +import { approvalNotificationMarkdown, loadApprovalNotification, markApprovalNotificationDelivered } from '../../../src/handlers/shared/approval-notifications'; +import type { TaskRecord } from '../../../src/handlers/shared/types'; + +const send = jest.fn(); +const ddb = { send } as unknown as DynamoDBDocumentClient; +const task = { + task_id: 'task', + user_id: 'owner', + status: TaskStatus.AWAITING_APPROVAL, + awaiting_approval_request_id: 'gate', +} as TaskRecord; +const row = { + status: 'PENDING', + user_id: 'owner', + tool_name: 'Bash', + severity: 'high', + reason: 'Destructive command', + tool_input_preview: 'git push --force', + created_at: '2026-09-16T12:00:00Z', + timeout_s: 1800, +}; +beforeEach(() => { + process.env.TASK_APPROVALS_TABLE_NAME = 'Approvals'; + send.mockReset().mockResolvedValue({ Item: row }); +}); + +test('shows the saved action and exact deadline, with one-call CLI approval', async () => { + const message = await loadApprovalNotification(ddb, task, 'approval_requested', { request_id: 'gate', reason: 'wrong' }, 'slack'); + expect(message?.text).toContain('Destructive command'); + expect(message?.text).toContain('git push --force'); + expect(message?.text).toContain('2026-09-16T12:30:00.000Z'); + expect(message?.text).toContain('bgagent approve task gate --scope this_call'); + expect(message?.text).toContain('bgagent deny task gate'); + expect(send.mock.calls[0][0].input.ConsistentRead).toBe(true); +}); + +test.each([ + ['cancelled task', { ...task, status: TaskStatus.CANCELLED }, row], + ['new gate', { ...task, awaiting_approval_request_id: 'new' }, row], + ['already decided', task, { ...row, status: 'APPROVED' }], + ['foreign owner', task, { ...row, user_id: 'other' }], + ['already delivered', task, { ...row, notified_slack_approval_requested: '2026-09-16' }], + ['missing approval', task, undefined], +])('suppresses a delayed request: %s', async (_label, owningTask, approval) => { + send.mockResolvedValue({ Item: approval }); + expect(await loadApprovalNotification(ddb, owningTask, 'approval_requested', { request_id: 'gate' }, 'slack')).toBeNull(); +}); + +test.each([ + ['APPROVED', 'approval_decision_recorded', 'Approval recorded'], + ['DENIED', 'approval_decision_recorded', 'Denial recorded'], + ['TIMED_OUT', 'approval_timed_out', 'Approval request timed out'], + ['CANCELLED', 'approval_cancelled', 'Approval request cancelled'], + ['STRANDED', 'approval_stranded', 'Approval wait could not continue'], +])('reports saved %s even when the task is now terminal', async (status, event, title) => { + send.mockResolvedValue({ Item: { ...row, status, cancellation_reason: 'Task cancelled by its owner' } }); + const message = await loadApprovalNotification(ddb, { ...task, status: TaskStatus.CANCELLED }, event, { request_id: 'gate' }, 'linear'); + expect(message?.title).toBe(title); + expect(message?.text).not.toContain('bgagent approve'); + if (status === 'CANCELLED') expect(message?.text).toContain('Task cancelled by its owner'); + if (status === 'APPROVED' || status === 'DENIED') { + expect(message?.text).toContain('Task status: CANCELLED. The task has already ended.'); + expect(message?.text).not.toContain('when its worker is ready'); + } +}); + +test.each([ + [undefined, 'The configured decision deadline was reached.'], + ['poll failed 3 consecutive times', 'poll failed 3 consecutive times'], +])('reports timeout closure instead of the original policy reason (%s)', async (denyReason, expected) => { + send.mockResolvedValue({ Item: { ...row, status: 'TIMED_OUT', deny_reason: denyReason } }); + const message = await loadApprovalNotification(ddb, task, 'approval_timed_out', { request_id: 'gate' }, 'slack'); + expect(message?.text).toContain(expected); + expect(message?.text).not.toContain(row.reason); +}); + +test('handles the actual stranded-reconciler event without changing the pending approval', async () => { + const failed = { ...task, status: TaskStatus.FAILED, error_message: 'Approval stranded: task paused for approval for 7200s with no resume transition.' }; + const message = await loadApprovalNotification(ddb, failed, 'approval_stranded', { reason: 'STRANDED_NO_HEARTBEAT' }, 'linear'); + expect(message).toMatchObject({ title: 'Approval wait could not continue', requestId: 'gate' }); + expect(message?.text).toContain(failed.error_message); + expect(message?.text).not.toContain('bgagent approve'); + expect(send).toHaveBeenCalledTimes(1); + expect(send.mock.calls[0][0].input.Key).toEqual({ task_id: 'task', request_id: 'gate' }); +}); + +test.each([ + [task, row], + [{ ...task, status: TaskStatus.FAILED, error_message: 'Unrelated failure' }, row], + [{ ...task, status: TaskStatus.FAILED, error_message: 'Approval stranded: old wait', awaiting_approval_request_id: undefined }, row], + [{ ...task, status: TaskStatus.FAILED, error_message: 'Approval stranded: old wait' }, { ...row, user_id: 'foreign' }], + [{ ...task, status: TaskStatus.FAILED, error_message: 'Approval stranded: old wait' }, { ...row, status: 'APPROVED' }], +])('does not infer a stranded request from an unrelated or decided task', async (owningTask, approval) => { + send.mockResolvedValue({ Item: approval }); + expect(await loadApprovalNotification(ddb, owningTask, 'approval_stranded', { reason: 'STRANDED_NO_HEARTBEAT' }, 'linear')).toBeNull(); +}); + +test('delivery receipts are separate for each request and channel and cannot recreate a deleted row', async () => { + const message = (await loadApprovalNotification(ddb, task, 'approval_requested', { request_id: 'gate' }, 'linear'))!; + await markApprovalNotificationDelivered(ddb, message); + expect(send.mock.calls[1][0].input).toMatchObject({ + Key: { task_id: 'task', request_id: 'gate' }, + ConditionExpression: 'attribute_exists(task_id) AND user_id = :user', + ExpressionAttributeNames: { '#marker': 'notified_linear_approval_requested' }, + }); +}); + +test('does not turn repository content into markdown or shell command substitutions', async () => { + send.mockResolvedValue({ Item: { ...row, tool_input_preview: '``` @everyone $(touch file)' } }); + const message = (await loadApprovalNotification(ddb, { ...task, awaiting_approval_request_id: '$(evil)' }, 'approval_requested', { request_id: '$(evil)' }, 'linear'))!; + expect(message.text).not.toContain('bgagent approve'); + expect(approvalNotificationMarkdown(message).match(/```/g)).toHaveLength(2); +}); + +test('redacts known credential shapes before posting a preview to a shared channel', async () => { + const token = `ghp_${'a'.repeat(36)}`; + send.mockResolvedValue({ Item: { ...row, tool_input_preview: `\u001b[31mTOKEN=${token}\u202eecho` } }); + const message = (await loadApprovalNotification(ddb, task, 'approval_requested', { request_id: 'gate' }, 'slack'))!; + expect(message.text).toContain('[REDACTED-GITHUB_TOKEN]'); + expect(message.text).not.toContain(token); + expect(message.text).not.toMatch(/[\u001b\u202e]/); +}); diff --git a/cdk/test/handlers/shared/task-cancellation.test.ts b/cdk/test/handlers/shared/task-cancellation.test.ts new file mode 100644 index 000000000..ff64d364c --- /dev/null +++ b/cdk/test/handlers/shared/task-cancellation.test.ts @@ -0,0 +1,144 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +// Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + +import { GetCommand, TransactWriteCommand, UpdateCommand } from '@aws-sdk/lib-dynamodb'; +import { TaskStatus } from '../../../src/constructs/task-status'; +import { cancelTaskState } from '../../../src/handlers/shared/task-cancellation'; +import type { TaskRecord } from '../../../src/handlers/shared/types'; + +const mockSend = jest.fn(); +jest.mock('../../../src/handlers/shared/ua', () => ({ + makeDocClient: () => ({ send: (...args: unknown[]) => mockSend(...args) }), +})); + +const task = { + task_id: 'task', + user_id: 'owner', + status: TaskStatus.AWAITING_APPROVAL, + awaiting_approval_request_id: 'gate', +} as TaskRecord; +const options = { + userId: 'owner', taskTable: 'Tasks', approvalsTable: 'Approvals', eventsTable: 'Events', retentionDays: 90, +}; +const pending = { status: 'PENDING', user_id: 'owner' }; +const conflict = { name: 'ConditionalCheckFailedException' }; +const transactionConflict = { + name: 'TransactionCanceledException', + CancellationReasons: [{ Code: 'None' }, { Code: 'ConditionalCheckFailed' }, { Code: 'None' }], +}; +beforeEach(() => mockSend.mockReset()); + +test('atomically closes the exact task, pending approval and audit event', async () => { + mockSend.mockResolvedValueOnce({ Item: pending }).mockResolvedValueOnce({}); + const result = await cancelTaskState(task, options); + expect(result.cancelledRequestId).toBe('gate'); + const command = mockSend.mock.calls[1][0] as TransactWriteCommand; + expect(command).toBeInstanceOf(TransactWriteCommand); + const [taskWrite, approvalWrite, eventWrite] = command.input.TransactItems!; + expect(taskWrite.Update).toMatchObject({ + Key: { task_id: 'task' }, + ConditionExpression: expect.stringContaining('awaiting_approval_request_id = :request'), + ExpressionAttributeValues: { ':observed': 'AWAITING_APPROVAL', ':request': 'gate', ':user': 'owner' }, + }); + expect(approvalWrite.Update).toMatchObject({ + Key: { task_id: 'task', request_id: 'gate' }, + ExpressionAttributeValues: { ':pending': 'PENDING', ':cancelled': 'CANCELLED', ':user': 'owner' }, + }); + expect(eventWrite.Put?.Item).toMatchObject({ + event_type: 'approval_cancelled', metadata: { request_id: 'gate', status: 'CANCELLED' }, + }); +}); + +test.each(['APPROVED', 'DENIED', 'TIMED_OUT', 'CANCELLED', 'STRANDED'])( + 'preserves an already %s approval', + async status => { + mockSend.mockResolvedValueOnce({ Item: { ...pending, status } }).mockResolvedValueOnce({}); + expect((await cancelTaskState(task, options)).cancelledRequestId).toBeUndefined(); + expect(mockSend.mock.calls[1][0]).toBeInstanceOf(UpdateCommand); + }, +); + +test('an approval winning the transaction race stays approved while its task is cancelled', async () => { + mockSend.mockResolvedValueOnce({ Item: pending }) + .mockRejectedValueOnce(transactionConflict) + .mockResolvedValueOnce({ Item: task }) + .mockResolvedValueOnce({ Item: { ...pending, status: 'APPROVED' } }) + .mockResolvedValueOnce({}); + await cancelTaskState(task, options); + expect(mockSend.mock.calls.map(([command]) => command.constructor)).toEqual([ + GetCommand, TransactWriteCommand, GetCommand, GetCommand, UpdateCommand, + ]); + expect(mockSend.mock.calls[2][0].input.ConsistentRead).toBe(true); +}); + +test('refreshes a newly created gate rather than cancelling only the old task snapshot', async () => { + const running = { ...task, status: TaskStatus.RUNNING, awaiting_approval_request_id: null }; + mockSend.mockRejectedValueOnce(conflict) + .mockResolvedValueOnce({ Item: task }) + .mockResolvedValueOnce({ Item: pending }) + .mockResolvedValueOnce({}); + const result = await cancelTaskState(running, options); + expect(result.cancelledRequestId).toBe('gate'); + expect(result.task).toEqual(task); + expect(mockSend.mock.calls[0][0].input.ConditionExpression).toContain('attribute_not_exists(awaiting_approval_request_id)'); + expect(mockSend.mock.calls[3][0]).toBeInstanceOf(TransactWriteCommand); +}); + +test.each([ + [undefined, 'missing'], + [{ ...task, status: TaskStatus.COMPLETED }, 'terminal'], + [{ ...task, user_id: 'other' }, 'forbidden'], +])('does not overwrite state after a conflicting cancellation (%s)', async (fresh, reason) => { + mockSend.mockResolvedValueOnce({ Item: pending }) + .mockRejectedValueOnce(transactionConflict) + .mockResolvedValueOnce({ Item: fresh }); + await expect(cancelTaskState(task, options)).rejects.toMatchObject({ reason }); + expect(mockSend).toHaveBeenCalledTimes(3); +}); + +test('does not fall back to a task-only cancellation after an unexpected transaction failure', async () => { + const error = new Error('DynamoDB unavailable'); + mockSend.mockResolvedValueOnce({ Item: pending }).mockRejectedValueOnce(error); + await expect(cancelTaskState(task, options)).rejects.toBe(error); + expect(mockSend).toHaveBeenCalledTimes(2); +}); + +test('bounded conflicts leave a retryable error, never an unconditional update', async () => { + for (let i = 0; i < 3; i++) { + mockSend.mockResolvedValueOnce({ Item: pending }) + .mockRejectedValueOnce(transactionConflict).mockResolvedValueOnce({ Item: task }); + } + await expect(cancelTaskState(task, options)).rejects.toMatchObject({ reason: 'conflict' }); + expect(mockSend).toHaveBeenCalledTimes(9); +}); + +test('a malformed foreign approval does not prevent cancellation or change that approval', async () => { + mockSend.mockResolvedValueOnce({ Item: { ...pending, user_id: 'other' } }).mockResolvedValueOnce({}); + await cancelTaskState(task, options); + expect(mockSend.mock.calls[1][0]).toBeInstanceOf(UpdateCommand); + expect(mockSend.mock.calls[1][0].input.TableName).toBe('Tasks'); +}); + +test('fails before mutating an approval wait when required tables are not wired', async () => { + await expect(cancelTaskState(task, { ...options, approvalsTable: undefined })) + .rejects.toThrow('requires approvals and events tables'); + expect(mockSend).not.toHaveBeenCalled(); +}); diff --git a/cdk/test/handlers/slack-interactions.test.ts b/cdk/test/handlers/slack-interactions.test.ts index 9d1356022..eac51a432 100644 --- a/cdk/test/handlers/slack-interactions.test.ts +++ b/cdk/test/handlers/slack-interactions.test.ts @@ -26,6 +26,7 @@ jest.mock('@aws-sdk/lib-dynamodb', () => ({ DynamoDBDocumentClient: { from: jest.fn(() => ({ send: ddbSend })) }, GetCommand: jest.fn((input: unknown) => ({ _type: 'Get', input })), UpdateCommand: jest.fn((input: unknown) => ({ _type: 'Update', input })), + TransactWriteCommand: jest.fn((input: unknown) => ({ _type: 'TransactWrite', input })), })); const smSend = jest.fn(); @@ -39,6 +40,8 @@ const fetchMock = jest.fn(); process.env.SLACK_SIGNING_SECRET_ARN = 'arn:aws:secretsmanager:us-east-1:123:secret:bgagent/slack/signing-I'; process.env.TASK_TABLE_NAME = 'Tasks'; +process.env.TASK_APPROVALS_TABLE_NAME = 'Approvals'; +process.env.TASK_EVENTS_TABLE_NAME = 'Events'; process.env.SLACK_USER_MAPPING_TABLE_NAME = 'SlackMap'; import { invalidateSlackSecretCache } from '../../src/handlers/shared/slack-verify'; @@ -127,7 +130,7 @@ describe('slack-interactions handler', () => { // 1. user mapping lookup → platform user id ddbSend.mockResolvedValueOnce({ Item: { platform_user_id: 'user-42' } }); // 2. task lookup → same owner - ddbSend.mockResolvedValueOnce({ Item: { task_id: 'task-42', user_id: 'user-42', channel_metadata: {} } }); + ddbSend.mockResolvedValueOnce({ Item: { task_id: 'task-42', user_id: 'user-42', status: 'RUNNING', channel_metadata: {} } }); // 3. update → success ddbSend.mockResolvedValueOnce({}); @@ -159,22 +162,37 @@ describe('slack-interactions handler', () => { test('cancel_task on already-terminal task warns the user', async () => { ddbSend.mockResolvedValueOnce({ Item: { platform_user_id: 'user-42' } }); - ddbSend.mockResolvedValueOnce({ Item: { task_id: 'task-42', user_id: 'user-42' } }); - // ConditionalCheckFailedException => already in terminal state - const err = new Error('conditional failed'); - err.name = 'ConditionalCheckFailedException'; - ddbSend.mockRejectedValueOnce(err); + ddbSend.mockResolvedValueOnce({ Item: { task_id: 'task-42', user_id: 'user-42', status: 'COMPLETED' } }); const event = makeInteractionEvent(interactionPayload('cancel_task:task-42')); const result = await handler(event); expect(result.statusCode).toBe(200); const posted = fetchMock.mock.calls.find( ([url, opts]) => - isSlackHooksRequestUrl(url) && String((opts as { body: string }).body).includes('terminal state'), + isSlackHooksRequestUrl(url) && String((opts as { body: string }).body).includes('no longer available'), ); expect(posted).toBeTruthy(); }); + test('cancel_task closes the linked approval in the same transaction', async () => { + ddbSend.mockResolvedValueOnce({ Item: { platform_user_id: 'user-42' } }) + .mockResolvedValueOnce({ + Item: { + task_id: 'task-42', + user_id: 'user-42', + status: 'AWAITING_APPROVAL', + awaiting_approval_request_id: 'request-42', + }, + }) + .mockResolvedValueOnce({ Item: { user_id: 'user-42', status: 'PENDING' } }) + .mockResolvedValueOnce({}); + const response = await handler(makeInteractionEvent(interactionPayload('cancel_task:task-42'))); + expect(response.statusCode).toBe(200); + const transaction = ddbSend.mock.calls.find(([command]) => command._type === 'TransactWrite')![0].input; + expect(transaction.TransactItems[1].Update.Key.request_id).toBe('request-42'); + expect(transaction.TransactItems[2].Put.Item.event_type).toBe('approval_cancelled'); + }); + test('unknown action_id is ignored silently', async () => { const event = makeInteractionEvent(interactionPayload('other_action:xyz')); const result = await handler(event); diff --git a/cdk/test/handlers/slack-notify.test.ts b/cdk/test/handlers/slack-notify.test.ts index b7b10ccf3..e421dfe30 100644 --- a/cdk/test/handlers/slack-notify.test.ts +++ b/cdk/test/handlers/slack-notify.test.ts @@ -40,6 +40,7 @@ const fetchMock = jest.fn(); (global as unknown as { fetch: unknown }).fetch = fetchMock; process.env.TASK_TABLE_NAME = 'Tasks'; +process.env.TASK_APPROVALS_TABLE_NAME = 'Approvals'; import type { DynamoDBDocumentClient } from '@aws-sdk/lib-dynamodb'; import { dispatchSlackEvent, SlackApiError, type SlackDispatchEvent } from '../../src/handlers/slack-notify'; @@ -72,6 +73,44 @@ describe('dispatchSlackEvent', () => { }); }); + test.each([false, true])('approval delivery records success only after posting (retryable failure=%s)', async fails => { + ddbSend.mockResolvedValueOnce({ + Item: { + task_id: 't1', + user_id: 'u1', + status: 'AWAITING_APPROVAL', + awaiting_approval_request_id: 'g1', + channel_source: 'slack', + channel_metadata: { slack_team_id: 'T1', slack_channel_id: 'C1', slack_thread_ts: 'thread' }, + }, + }).mockResolvedValueOnce({ + Item: { + user_id: 'u1', + status: 'PENDING', + tool_name: 'Bash', + severity: 'high', + reason: 'Review this', + tool_input_preview: ' echo hello', + created_at: '2026-09-16T12:00:00Z', + timeout_s: 1800, + }, + }).mockResolvedValue({}); + if (fails) fetchMock.mockRejectedValueOnce(new Error('network unavailable')); + const dispatch = dispatchSlackEvent(mkEvent('t1', 'approval_requested', { request_id: 'g1' }), ddb); + if (fails) { + await expect(dispatch).rejects.toThrow('network unavailable'); + expect(ddbSend).toHaveBeenCalledTimes(2); + } else { + await dispatch; + const payload = JSON.parse(fetchMock.mock.calls[0][1].body); + expect(payload.thread_ts).toBe('thread'); + expect(payload.blocks[0].text.type).toBe('plain_text'); + expect(payload.blocks[0].text.text).toContain('bgagent approve t1 g1 --scope this_call'); + expect(ddbSend.mock.calls[2][0].input.ExpressionAttributeNames).toEqual({ '#marker': 'notified_slack_approval_requested' }); + expect(fetchMock.mock.invocationCallOrder[0]).toBeLessThan(ddbSend.mock.invocationCallOrder[2]); + } + }); + test('skips non-slack tasks without touching Slack', async () => { // A Slack-subscribed event on a non-Slack task must still short- // circuit cheaply — one DDB Get, no dedup write, no API call. diff --git a/cdk/test/stacks/agent.test.ts b/cdk/test/stacks/agent.test.ts index ee8091788..b51d5a366 100644 --- a/cdk/test/stacks/agent.test.ts +++ b/cdk/test/stacks/agent.test.ts @@ -19,7 +19,7 @@ import * as fs from 'fs'; import * as path from 'path'; -import { App, AspectPriority, Aspects } from 'aws-cdk-lib'; +import { App, AspectPriority, Aspects, NestedStack } from 'aws-cdk-lib'; import { Match, Template } from 'aws-cdk-lib/assertions'; import { BEDROCK_GEO_REGION_CONTEXT_KEY, @@ -27,6 +27,7 @@ import { DEFAULT_BEDROCK_MODEL_IDS, } from '../../src/constructs/bedrock-models'; import * as lambdaMicrovmCompute from '../../src/constructs/lambda-microvm-compute'; +import { LambdaMicrovmStack } from '../../src/constructs/lambda-microvm-stack'; import { buildAppId, SolutionUaAspect } from '../../src/constructs/solution-ua-aspect'; import { AgentStack } from '../../src/stacks/agent'; @@ -1097,6 +1098,28 @@ describe('AgentStack with the Lambda MicroVMs substrate gate (--context compute_ const BASE_IMAGE_ARN = 'arn:aws:lambda:us-east-1:aws:microvm-image:al2023-1'; let template: Template; + let childTemplate: Template; + + function assertChildReference(value: unknown, resourceType: string, attribute: string, prefix = ''): void { + const [stackId, outputPath] = (value as { 'Fn::GetAtt': [string, string] })['Fn::GetAtt']; + expect(stackId).toMatch(/^MicrovmNestedStack/); + expect(outputPath).toMatch(/^Outputs\./); + const childOutput = childTemplate.toJSON().Outputs[outputPath.slice('Outputs.'.length)]; + const ids = Object.keys(childTemplate.findResources(resourceType)).filter(id => id.startsWith(prefix)); + expect(ids.length).toBeGreaterThan(0); + expect(childOutput.Value).toEqual({ 'Fn::GetAtt': [expect.stringMatching(new RegExp(`^(${ids.join('|')})$`)), attribute] }); + } + + function orchestratorEnvironment(): Record { + return Object.entries(template.findResources('AWS::Lambda::Function')) + .find(([id]) => id.includes('TaskOrchestratorOrchestratorFn'))![1] + .Properties.Environment.Variables; + } + + function imageResources(): unknown[] { + const arn = orchestratorEnvironment().MICROVM_IMAGE_IDENTIFIER; + return [arn, { 'Fn::Join': ['', [arn, ':*']] }]; + } beforeAll(() => { // Gate ON *and* an image configured — the steady state. The intermediate @@ -1115,13 +1138,16 @@ describe('AgentStack with the Lambda MicroVMs substrate gate (--context compute_ env: { account: '123456789012', region: 'us-east-1' }, }); template = Template.fromStack(stack); + childTemplate = Template.fromStack(stack.node.findChild('Microvm') as LambdaMicrovmStack); }); test('provisions the MicroVM image + BOTH egress network connectors', () => { - template.resourceCountIs('AWS::Lambda::MicrovmImage', 1); + template.resourceCountIs('AWS::Lambda::MicrovmImage', 0); + template.resourceCountIs('AWS::Lambda::NetworkConnector', 0); + childTemplate.resourceCountIs('AWS::Lambda::MicrovmImage', 1); // Runtime (443) + build-time (443 + 80, for apt-get) — see ADR-021's // build-time-egress security-table row. - template.resourceCountIs('AWS::Lambda::NetworkConnector', 2); + childTemplate.resourceCountIs('AWS::Lambda::NetworkConnector', 2); }); test('does NOT provision the ECS substrate (the gates are mutually exclusive)', () => { @@ -1154,7 +1180,7 @@ describe('AgentStack with the Lambda MicroVMs substrate gate (--context compute_ const key = `microvm-images/agent-artifact-${'a'.repeat(64)}.zip`; template.hasOutput('MicrovmArtifactObjectKey', { Value: key }); template.hasOutput('MicrovmArtifactBaseObjectKey', { Value: 'microvm-images/agent-artifact.zip' }); - const image = Object.values(template.findResources('AWS::Lambda::MicrovmImage'))[0]!; + const image = Object.values(childTemplate.findResources('AWS::Lambda::MicrovmImage'))[0]!; expect(JSON.stringify(image.Properties.CodeArtifact.Uri)).toContain(key); }); @@ -1197,8 +1223,7 @@ describe('AgentStack with the Lambda MicroVMs substrate gate (--context compute_ const [, orchestrator] = Object.entries(fns) .find(([id]) => id.includes('TaskOrchestratorOrchestratorFn'))!; const env = orchestrator.Properties.Environment.Variables as Record; - expect(JSON.stringify(env.MICROVM_IMAGE_IDENTIFIER)) - .toMatch(/"Fn::GetAtt":\["LambdaMicrovmComputeImage[^"]*","ImageArn"\]/); + assertChildReference(env.MICROVM_IMAGE_IDENTIFIER, 'AWS::Lambda::MicrovmImage', 'ImageArn'); }); test('grants the orchestrator its launch, state, cleanup and image-capability actions, image-scoped', () => { @@ -1221,9 +1246,7 @@ describe('AgentStack with the Lambda MicroVMs substrate gate (--context compute_ ]); // Every MicroVM lifecycle action authorizes against the *image* resource, // which is why "scoped to platform-created images" is achievable at all. - expect(JSON.stringify(lifecycle.Resource)).toMatch( - /"Fn::GetAtt":\["LambdaMicrovmComputeImage[^"]*","ImageArn"\]/, - ); + expect(lifecycle.Resource).toEqual(imageResources()); // PassNetworkConnector supports no resource-level permissions. const pass = statements.find(s => s.Sid === 'MicrovmPassNetworkConnector')!; @@ -1252,7 +1275,7 @@ describe('AgentStack with the Lambda MicroVMs substrate gate (--context compute_ Resource: unknown; }>); const payloadStatements = statements.filter(s => - JSON.stringify(s.Resource).includes('LambdaMicrovmComputePayloadBucket')); + JSON.stringify(s.Resource).includes('ComputePayloadBucket')); const actions = payloadStatements.flatMap(s => Array.isArray(s.Action) ? s.Action : [s.Action]); expect(actions).toContain('s3:PutObject'); @@ -1260,9 +1283,12 @@ describe('AgentStack with the Lambda MicroVMs substrate gate (--context compute_ expect(actions).toContain('s3:GetObject'); expect(actions).toContain('s3:ListBucket'); const deletion = payloadStatements.find(s => Array.isArray(s.Action) && s.Action.includes('s3:DeleteObject')); + const resources = deletion!.Resource as Array<{ 'Fn::Join': [string, unknown[]] }>; + const payloadArn = resources[0]['Fn::Join'][1][0]; + assertChildReference(payloadArn, 'AWS::S3::Bucket', 'Arn', 'ComputePayloadBucket'); expect(deletion!.Resource).toEqual(['payload.json', 'launch.json'].map(filename => ({ 'Fn::Join': ['', [ - { 'Fn::GetAtt': [expect.stringMatching(/^LambdaMicrovmComputePayloadBucket/), 'Arn'] }, + payloadArn, `/*/${filename}`, ]], }))); @@ -1284,7 +1310,7 @@ describe('AgentStack with the Lambda MicroVMs substrate gate (--context compute_ // (TaskApi is built before the MicroVM construct), so the grant names ONE // image instead of an account/Region-wide `microvm-image:*`. const rendered = JSON.stringify(microvmStatements[0]!.Resource); - expect(rendered).toMatch(/LambdaMicrovmComputeImage[^"]*","ImageArn"/); + expect(microvmStatements[0]!.Resource).toEqual(imageResources()); expect(rendered).not.toContain('microvm-image:*'); }); @@ -1334,10 +1360,10 @@ describe('AgentStack with the Lambda MicroVMs substrate gate (--context compute_ }); test('MicroVM resources carry the backend cost-allocation tag', () => { - template.hasResourceProperties('AWS::Lambda::MicrovmImage', { + childTemplate.hasResourceProperties('AWS::Lambda::MicrovmImage', { Tags: Match.arrayWith([{ Key: 'abca:compute-backend', Value: 'lambda-microvm' }]), }); - template.hasResourceProperties('AWS::Lambda::NetworkConnector', { + childTemplate.hasResourceProperties('AWS::Lambda::NetworkConnector', { Tags: Match.arrayWith([{ Key: 'abca:compute-backend', Value: 'lambda-microvm' }]), }); }); @@ -1360,9 +1386,11 @@ describe('AgentStack with the Lambda MicroVMs substrate gate (--context compute_ microvm_artifact_sha256: 'a'.repeat(64), }, }); - overriddenTemplate = Template.fromStack(new AgentStack(app, 'TestAgentStackMicrovmOverride', { + const stack = new AgentStack(app, 'TestAgentStackMicrovmOverride', { env: { account: '123456789012', region: 'eu-central-1' }, - })); + }); + Template.fromStack(stack); + overriddenTemplate = Template.fromStack(stack.node.findChild('Microvm') as LambdaMicrovmStack); }); test('fails synth when the stack Region has no Lambda MicroVMs', () => { @@ -1414,6 +1442,7 @@ describe('AgentStack default (agentcore) deploy — MicroVM substrate absent', ( describe('AgentStack with the MicroVM gate on but no image configured (first deploy)', () => { let template: Template; + let childTemplate: Template; beforeAll(() => { // The bootstrap state: substrate provisioned so the artifact bucket exists, @@ -1425,10 +1454,13 @@ describe('AgentStack with the MicroVM gate on but no image configured (first dep env: { account: '123456789012', region: 'us-east-1' }, }); template = Template.fromStack(stack); + childTemplate = Template.fromStack(stack.node.findChild('Microvm') as LambdaMicrovmStack); }); test('provisions the substrate (buckets, roles, both connectors) but no image', () => { - template.resourceCountIs('AWS::Lambda::NetworkConnector', 2); + template.resourceCountIs('AWS::Lambda::NetworkConnector', 0); + childTemplate.resourceCountIs('AWS::Lambda::NetworkConnector', 2); + childTemplate.resourceCountIs('AWS::Lambda::MicrovmImage', 0); template.resourceCountIs('AWS::Lambda::MicrovmImage', 0); template.hasOutput('MicrovmArtifactBucketName', {}); // The build-time connector output is what the packaging script reads next, so @@ -1454,6 +1486,38 @@ describe('AgentStack with the MicroVM gate on but no image configured (first dep }); }); +describe('AgentStack MicroVM flat-layout migration compatibility', () => { + let template: Template; + + beforeAll(() => { + const app = new App({ + context: { + compute_type: 'lambda-microvm', + microvm_nested_stack: false, + microvm_base_image_arn: 'arn:aws:lambda:us-east-1:aws:microvm-image:al2023-1', + microvm_base_image_version: '1', + microvm_artifact_sha256: 'a'.repeat(64), + }, + }); + template = Template.fromStack(new AgentStack(app, 'FlatMicrovmStack', { + env: { account: '123456789012', region: 'us-east-1' }, + })); + }); + + test('retains original resource paths and unnamed IAM roles with nesting disabled', () => { + template.resourceCountIs('AWS::Lambda::MicrovmImage', 1); + template.resourceCountIs('AWS::Lambda::NetworkConnector', 2); + expect(Object.keys(template.findResources('AWS::Lambda::MicrovmImage'))[0]) + .toMatch(/^LambdaMicrovmComputeImage/); + expect(Object.keys(template.findResources('AWS::CloudFormation::Stack'))) + .not.toEqual(expect.arrayContaining([expect.stringMatching(/^MicrovmNestedStack/)])); + const roles = Object.entries(template.findResources('AWS::IAM::Role')) + .filter(([id]) => id.startsWith('LambdaMicrovmCompute')); + expect(roles).toHaveLength(3); + for (const [, role] of roles) expect(role.Properties.RoleName).toBeUndefined(); + }); +}); + describe('AgentStack MicroVM image ARN invariant', () => { let synthError: unknown; @@ -1886,29 +1950,66 @@ describe('AgentStack CloudFormation resource budget 500 with cushion', () => { // `AWS::CDK::Metadata`. Budget the synthesized number, so add that resource back. const SYNTH_ONLY_RESOURCES = 1; - const COMPUTE_TYPES = ['agentcore', 'ecs', 'lambda-microvm']; - const CELLS = COMPUTE_TYPES.flatMap(computeType => - [false, true].map(enableToolGateway => ({ computeType, enableToolGateway })), + const CONFIGURATIONS = [ + { name: 'agentcore', context: { compute_type: 'agentcore' } }, + { name: 'ecs', context: { compute_type: 'ecs' } }, + { name: 'microvm-bootstrap', context: { compute_type: 'lambda-microvm' } }, + { + name: 'microvm-imported', + context: { + compute_type: 'lambda-microvm', + microvm_image_identifier: 'arn:aws:lambda:us-east-1:123456789012:microvm-image:existing-agent', + microvm_image_version: '6.0', + }, + }, + { + name: 'microvm-managed', + context: { + compute_type: 'lambda-microvm', + microvm_base_image_arn: 'arn:aws:lambda:us-east-1:aws:microvm-image:al2023-1', + microvm_base_image_version: '1', + microvm_artifact_sha256: 'a'.repeat(64), + }, + }, + ]; + const CELLS = CONFIGURATIONS.flatMap(configuration => + [false, true].map(enableToolGateway => ({ ...configuration, enableToolGateway })), ); describe.each(CELLS)( - 'compute_type=$computeType enableToolGateway=$enableToolGateway', - ({ computeType, enableToolGateway }) => { + '$name enableToolGateway=$enableToolGateway', + ({ context, enableToolGateway }) => { let template: Template; + let templates: Template[]; beforeAll(() => { - const app = new App({ context: { compute_type: computeType, enableToolGateway } }); + const app = new App({ context: { ...context, enableToolGateway } }); const stack = new AgentStack(app, 'BudgetStack', { env: { account: '123456789012', region: 'us-east-1' }, }); // Throws `TooManyResourcesInStack` if this cell is over the hard quota, so // reaching the assertions below is itself part of the guard. template = Template.fromStack(stack); + templates = [ + template, + ...stack.node.findAll().filter((child): child is NestedStack => NestedStack.isNestedStack(child)) + .map(child => Template.fromStack(child)), + ]; }); test('stays inside the resource budget', () => { - const resourceCount = Object.keys(template.toJSON().Resources ?? {}).length; - expect(resourceCount + SYNTH_ONLY_RESOURCES).toBeLessThanOrEqual(RESOURCE_BUDGET); + for (const item of templates) { + const rendered = item.toJSON(); + const resourceCount = Object.keys(rendered.Resources ?? {}).length; + expect(resourceCount + SYNTH_ONLY_RESOURCES).toBeLessThanOrEqual(RESOURCE_BUDGET); + expect(Buffer.byteLength(JSON.stringify(rendered))).toBeLessThanOrEqual(1024 * 1024); + expect(Object.keys(rendered.Parameters ?? {}).length).toBeLessThanOrEqual(200); + expect(Object.keys(rendered.Outputs ?? {}).length).toBeLessThanOrEqual(200); + } + // Even changing every resource must fit the nested-operation quota. + const total = templates.reduce((count, item) => + count + Object.keys(item.toJSON().Resources ?? {}).length + SYNTH_ONLY_RESOURCES, 0); + expect(total).toBeLessThanOrEqual(2500); }); test('emits no Lambda permission for the API Gateway console test-invoke stage', () => { diff --git a/cli/src/commands/watch.ts b/cli/src/commands/watch.ts index 6cefbfab8..6f5612f03 100644 --- a/cli/src/commands/watch.ts +++ b/cli/src/commands/watch.ts @@ -113,6 +113,8 @@ const PROGRESS_EVENT_TYPES = new Set([ 'agent_cost_update', 'agent_error', 'agent_blocked', + 'approval_decision_recorded', + 'approval_cancelled', ]); /** Format an event timestamp to a short local time string. */ @@ -154,7 +156,9 @@ function renderMilestoneSuffix(meta: Record): string { if (meta.severity != null) parts.push(`[sev=${String(meta.severity)}]`); if (meta.request_id != null) parts.push(`request_id=${String(meta.request_id)}`); if (meta.scope != null) parts.push(`scope=${String(meta.scope)}`); + if (meta.status != null) parts.push(`status=${String(meta.status)}`); if (meta.timeout_s != null) parts.push(`timeout=${String(meta.timeout_s)}s`); + if (meta.reason != null) parts.push(`reason=${String(meta.reason)}`); const ruleIds = meta.matching_rule_ids; if (Array.isArray(ruleIds) && ruleIds.length > 0) { parts.push(`rules=${ruleIds.map(String).join(',')}`); @@ -177,7 +181,7 @@ function renderMilestoneSuffix(meta: Record): string { } /** Render a single progress event as a human-readable line. */ -export function renderEvent(event: TaskEvent): string { +export function renderEvent(event: TaskEvent, taskId?: string): string { const time = formatTime(event.timestamp); const meta = event.metadata; @@ -208,8 +212,21 @@ export function renderEvent(event: TaskEvent): string { } case 'agent_milestone': { const milestone = String(meta.milestone ?? ''); - return `[${time}] ★ ${milestone}${renderMilestoneSuffix(meta)}`; + let line = `[${time}] ★ ${milestone}${renderMilestoneSuffix(meta)}`; + if (milestone === 'approval_requested') { + if (meta.input_preview) line += `\n Action: ${String(meta.input_preview)}`; + const requestId = String(meta.request_id ?? ''); + if (taskId && [taskId, requestId].every(id => /^[A-Za-z0-9_-]{1,128}$/.test(id))) { + line += `\n bgagent approve ${taskId} ${requestId} --scope this_call`; + line += `\n bgagent deny ${taskId} ${requestId}`; + } + } + return line; } + case 'approval_decision_recorded': + return `[${time}] Decision saved${renderMilestoneSuffix(meta)}`; + case 'approval_cancelled': + return `[${time}] Approval request closed${renderMilestoneSuffix(meta)}`; case 'agent_cost_update': { const cost = meta.cost_usd != null ? `$${Number(meta.cost_usd).toFixed(COST_USD_DECIMALS)}` : '$?'; const input = meta.input_tokens ?? 0; @@ -302,7 +319,7 @@ interface Formatter { emit(ev: TaskEvent): void; } -export function makeFormatter(isJson: boolean): Formatter { +export function makeFormatter(isJson: boolean, taskId?: string): Formatter { return { emit(ev: TaskEvent): void { if (isJson) { @@ -310,7 +327,7 @@ export function makeFormatter(isJson: boolean): Formatter { return; } if (PROGRESS_EVENT_TYPES.has(ev.event_type)) { - console.log(renderEvent(ev)); + console.log(renderEvent(ev, taskId)); } }, }; @@ -561,7 +578,7 @@ export function makeWatchCommand(): Command { throw e; } - const formatter = makeFormatter(isJson); + const formatter = makeFormatter(isJson, taskId); // Task already terminated — print the snapshot tail and exit. if ((TERMINAL_STATUSES as readonly string[]).includes(snapshot.taskStatus)) { diff --git a/cli/src/types.ts b/cli/src/types.ts index 8a809722d..8b033768a 100644 --- a/cli/src/types.ts +++ b/cli/src/types.ts @@ -707,6 +707,7 @@ export type ApprovalStatus = | 'PENDING' | 'APPROVED' | 'DENIED' + | 'CANCELLED' | 'TIMED_OUT' | 'STRANDED'; diff --git a/cli/test/commands/watch.test.ts b/cli/test/commands/watch.test.ts index 2cc65a32f..972e4cd1c 100644 --- a/cli/test/commands/watch.test.ts +++ b/cli/test/commands/watch.test.ts @@ -133,6 +133,8 @@ describe('renderEvent', () => { timeout_s: 300, matching_rule_ids: ['rule-1', 'rule-2'], scope: 'this_call', + reason: 'Needs your permission', + input_preview: 'git push --force', }, }); const output = renderEvent(event); @@ -142,6 +144,19 @@ describe('renderEvent', () => { expect(output).toContain('timeout=300s'); expect(output).toContain('rules=rule-1,rule-2'); expect(output).toContain('scope=this_call'); + expect(output).toContain('Needs your permission'); + expect(output).toContain('Action: git push --force'); + expect(renderEvent(event, 'task-1')).toContain('bgagent approve task-1 req-abc --scope this_call'); + expect(renderEvent(event, 'task-1')).toContain('bgagent deny task-1 req-abc'); + }); + + test('renders a closed approval with its cancellation reason', () => { + const output = renderEvent(makeEvent({ + event_type: 'approval_cancelled', + metadata: { request_id: 'gate', status: 'CANCELLED', reason: 'Task cancelled by its owner' }, + })); + expect(output).toContain('Approval request closed'); + expect(output).toContain('Task cancelled by its owner'); }); test('renders policy_decision milestone metadata', () => { diff --git a/docs/decisions/ADR-021-lambda-microvms-compute-backend.md b/docs/decisions/ADR-021-lambda-microvms-compute-backend.md index 6c0ce5527..ea42ce55e 100644 --- a/docs/decisions/ADR-021-lambda-microvms-compute-backend.md +++ b/docs/decisions/ADR-021-lambda-microvms-compute-backend.md @@ -1,8 +1,8 @@ # ADR-021: AWS Lambda MicroVMs as a third ComputeStrategy backend -> **Implementation status (2026-09-16):** P1 ([#689](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/689)) and P2 ([#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733)) are merged. The takeover has a [clean deployment](../verification/645-p2-clean-deployment-20260913.md), [live repository workflow evidence](../verification/645-p2-live-task-20260914.md), and integrated P3 guest hooks, credentials, durable intent and approval/supervisor handling. The current normal deployment uses agent image **5.0**, coordinator **6** and bootstrap **1.8.0**. [Long-sleep credential/deadline acceptance](../verification/645-p3-callback-live-20260915.md) and the [pending-wake timer corrections](../verification/645-p3-pending-wake.md) passed live checks. +> **Implementation status (2026-09-17):** P1 ([#689](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/689)) and P2 ([#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733)) are merged. The takeover has a [clean deployment](../verification/645-p2-clean-deployment-20260913.md), [live repository workflow evidence](../verification/645-p2-live-task-20260914.md), and integrated P3 guest hooks, credentials, durable intent and approval/supervisor handling. The normal deployment uses agent image **7.0**, coordinator live alias **10** and bootstrap **1.9.0**. The [approval UX update](../verification/645-p3-approval-ux-20260917.md) deploys atomic cancellation, channel notification support and guest cancellation handling. The [final image tests](../verification/645-p3-final-image-and-ecs-20260917.md), [repository workflow](../verification/645-p3-repository-path-20260917.md), and [MCP/network checks](../verification/645-p3-mcp-network-20260917.md) provide nine successful image 6.0 approval workflows, including real credential renewal after expiry. > -> **P3 remains incomplete; automatic suspension is disabled.** Four [resume connection refusals](../verification/645-p3-resume-refusal-investigation.md) remain unexplained, alongside remaining race, permission/network, recovery and rollout checks. New [lifecycle diagnostics and wake feedback](../verification/645-p3-lifecycle-diagnostics.md) passed isolated AWS workflows and are [deployed to the normal stack](../verification/645-p3-diagnostics-rollout-20260916.md). Six further [live command-race cases](../verification/645-p3-command-races-20260916.md) now pass, including cancellation during suspension/restore and bounded polling failure. Follow the [current implementation plan](../verification/645-p3-implementation-plan.md) and [service-team feedback tracker](../verification/645-lambda-microvm-service-feedback.md). This ADR defines P1–P3, not a P4. +> **P3 remains incomplete; automatic suspension is disabled.** Lifecycle responses now explicitly close their HTTP connections before freezing, with [transport evidence and a deployed fix](../verification/645-p3-connection-close-rollout-20260916.md). Historical service-side dispatch traces remain unavailable; passing workflows do not supply those traces. Remaining work includes normal activation/off-switch acceptance, approval UX work tracked in the [current implementation plan](../verification/645-p3-implementation-plan.md), and deployment of the requested [nested MicroVM stack](../verification/645-p3-nested-stack.md). Bootstrap **1.9.0** is installed; moving the existing flat deployment still requires a reviewed resource migration. Follow the [service-team feedback tracker](../verification/645-lambda-microvm-service-feedback.md) for open service questions. This ADR defines P1–P3, not a P4. **Status:** proposed **Date:** 2026-07-29 @@ -138,7 +138,7 @@ The handshake must respect the existing approval mechanics: the agent **discover It does **not** relieve the orchestrator of anything, because the two cases are disjoint. The service reaps a hook *result* it did not like; it has no view of the guest once the hook returned 200. So a task that starts normally — the overwhelming majority — has no service-side reaper at all, and a VM whose pipeline finished, crashed after `/run`, or hung is reaped by nobody but `TerminateMicrovm`. A leaked handle therefore remains a cost incident that bills until the 8 h cap; only the "the guest rejected its own payload" corner now cleans itself up. - **Concurrency slot stays held** during suspend. Cedar decision #7's rationale ("container alive, consuming memory") weakens under suspend, and the harder replacement rationale — "AWS counts `SUSPENDED` MicroVMs toward the account memory quota, so releasing ABCA's slot would not free real capacity" — is **undischarged**: the suspended VM stayed in `list-microvms` at every checkpoint, but that only proves *listed*. `L-CD1C0CC4` (1024 GB, account-scoped) exposes no `UsageMetric`, `AWS/Usage` carries only `CallCount` per API, and no MicroVM memory metric exists in any namespace, so consumption is **not observable safely** — proving it would need a large concurrent fleet. The conclusion (hold the slot) stands as the conservative choice, not as a verified fact. Size the arithmetic against the 32 GiB **peak** rather than the 8 GiB baseline: a busy fleet scales up, so peak is what actually competes for the account quota. -In P3, the agent's `/suspend` hook must acknowledge durable progress before returning 200 within its hook budget; `/resume` must refresh credentials and reseed application PRNG state from fresh OS entropy. These hooks are implemented locally with atomic checkpoints, retained credential refresh and original-gate reconciliation; matching deployment and live acceptance remain open. A non-cryptographic PRNG such as Python `random` remains unsuitable for secrets even after reseeding; security-sensitive randomness must use an OS-backed cryptographic source. +In P3, the agent's `/suspend` hook must acknowledge durable progress before returning 200 within its hook budget; `/resume` must refresh credentials and reseed application PRNG state from fresh OS entropy. These hooks are deployed in image 6.0 with atomic checkpoints, retained credential refresh and original-gate reconciliation, and have passed the isolated live workflows linked above. Automatic suspension on the normal task path remains disabled pending rollout acceptance. A non-cryptographic PRNG such as Python `random` remains unsuitable for secrets even after reseeding; security-sensitive randomness must use an OS-backed cryptographic source. **Amended (P2 review): Application PRNG reseeding is P3 scope, and the P2 exposure is negligible.** The original EARS requirement below implied a `/run`-time reseed had to land with P2; it did not, and shipping P2 with the requirement unmet-but-asserted was itself the defect. Measured exposure, which is what changed the scoping: @@ -176,7 +176,7 @@ The phasing is therefore: | `/run` | **P1** (construct sets `hooks.microvmHooks.run`) | **P1** | The payload-delivery channel. Must be served in P1 because `/ready` forces hooks to exist at all, and a hook-less image cannot accept `runHookPayload`. Since P2 it is also the **platform-configuration** channel (see "Platform configuration delivery" below). | | `/validate` | **P2** (construct sets `hooks.microvmImageHooks.validate`) | **P2** | An **image** (build-time) hook, and a **shallow self-check only**: server alive, hook routes registered, interpreter + contract sanity. It runs under the BUILD role, which deliberately holds no Bedrock / Secrets Manager / DynamoDB grants, so it must make **zero AWS API calls** and must not touch credential resolution — the "deeper warm-up assertions (Bedrock reachability, Memory access, tool availability)" this ADR originally assigned here are **not implementable**: every one of them would `AccessDenied` and fail every image build. They belong to the first task's own error handling. 200 when the checks pass, 503 while still initialising. | | `/terminate` | **P2** (construct sets `hooks.microvmHooks.terminate`) | **P2** | Best-effort final flush: a final structured log line, then 200 — always, inside the hook budget, even with nothing running. It must **not** write terminal task status (the orchestrator normally finalizes before `TerminateMicrovm`, and also attempts cleanup if database finalization fails, so a terminate hook that wrote a status would race that finalization and could clobber the real outcome) and must not join the pipeline thread. `ProgressWriter` writes synchronously but drops failed events, so this hook cannot guarantee that every progress write succeeded. P3 supplies a separate acknowledged checkpoint for suspension. "Always 200" also covers the BODY: the handler reads the raw request rather than a typed model, because a typed body is validated before the handler runs and would answer 422 to malformed JSON — a reported hook failure on a successful teardown. Safe to declare because `TerminateMicrovm` removes the VM with or without in-guest cooperation. **Correction (P2-F8):** the service sends `microvmId: ""` on this hook, unlike `/run` where it is populated, so an empty id is expected-normal and this hook cannot join the guest record to the control-plane one — `/run`'s accepted line carries that correlation instead. | -| `/suspend`, `/resume` | **P3**, implemented locally | **P3**, implemented locally | Managed images declare the served checkpoint/credential/gate hooks with the shared 30-second service timeout and protocol marker. The coordinator verifies the actual launched version before allowing new suspension. Supervisor integration is implemented locally with a default-off rollout flag; live acceptance remains open. | +| `/suspend`, `/resume` | **P3**, deployed in image 6.0 | **P3**, deployed in image 6.0 | Managed images declare the served checkpoint/credential/gate hooks with the shared 30-second service timeout and protocol marker. The coordinator verifies the actual launched version before allowing new suspension. Lifecycle responses close the connection before freezing. Isolated live workflows pass; normal automatic-suspension activation remains gated off. | Consequence to state plainly, replacing the original "a P1-built MicroVM image is not runnable end to end": **a P1 image is creatable, launchable and payload-deliverable, but carries no smoke-parity guarantee.** P1 delivers the strategy, the construct, the roles/buckets/connectors, the image resource, the packaging script, and the `/ready` + `/run` endpoints — so a `lambda-microvm` task can start a MicroVM and hand it a payload. What P1 has **not** established is anything P2 owns: AgentCore Memory grants and `MEMORY_ID` delivery, the agent's non-secret env parity inside the snapshot, egress specifics from a running MicroVM, and heartbeat/progress behaviour end to end. At the P1 handoff, no clone → change → PR run had happened. P2 initially demonstrated that path with an IAM workaround; the 2026-09-14 clean deployment and coding/iteration/cancellation runs later passed without manual IAM changes. The broader failure/recovery, effective IAM and networking matrix remains open. The construct and the packaging script surface that remaining scope at synth/run time (`abca:microvm-image-p1-smoke-unverified`) so an operator cannot mistake a launchable substrate for a verified one. diff --git a/docs/design/CEDAR_HITL_GATES.md b/docs/design/CEDAR_HITL_GATES.md index 90c8008f6..88a01e69b 100644 --- a/docs/design/CEDAR_HITL_GATES.md +++ b/docs/design/CEDAR_HITL_GATES.md @@ -5,6 +5,13 @@ > **Design locked:** 2026-04-23 (Sam ↔ assistant discussion). > **Rev:** 5 (2026-05-06 — fold in parallel adversarial + advocate review of the timeout design: late-approval re-read on TIMED_OUT ConditionCheckFailed; user-visible timeout-cap milestones; ceiling-shrink milestone; Runtime JWT bound verified as auto-refreshed IAM; three new tuning metrics; explicit off-hours trade-off section; notification-delivery-failure boundary. IMPL-24 through IMPL-28 added.). > **Implementation:** Core shipped. The 3-outcome engine (`agent/src/policy.py`), default policy sets (`agent/policies/hard_deny.cedar`, `agent/policies/soft_deny.cedar`), approval Lambdas (`cdk/src/handlers/{approve-task,deny-task,get-pending,get-policies}.ts`) wired into `cdk/src/constructs/task-api.ts` (routes `/tasks/{id}/approve`, `/deny`, `/pending`, `/repos/{repo_id}/policies`), the cross-engine parity fixtures (`contracts/cedar-parity/`), and the exact engine pins are all on `main`. §15's task list is preserved as a historical implementation record; see the note at the top of §15 for what (if anything) remains unbuilt. +> +> **Current behavior clarified (2026-09-17):** closed approval-row conditions return +> `404 REQUEST_NOT_FOUND`; task-only conflicts return 409. Cancellation atomically +> closes a pending request. The stranded-task reconciler currently fails the task +> but leaves its approval row pending; the pending endpoint filters such rows. +> CLI response instructions are implemented for Slack/Linear notifications; +> native channel decisions and approvals without an expiry remain future work. --- @@ -273,7 +280,7 @@ Narrative walk-through of the happy path. Sequence diagrams in the round-trip Me - ConditionalUpdate on `TaskApprovalsTable`: `#status = :pending AND user_id = :caller AND task_id = :task_id` → flip to APPROVED - ConditionalUpdate on `TaskTable`: `#status = :awaiting AND awaiting_approval_request_id = :rid` → (no-op update, pure state guard; keeps status AWAITING_APPROVAL until the agent's resume transaction flips it RUNNING) Both conditions must hold or the entire transaction is cancelled. No TOCTOU window, no "approved a cancelled task" 202 surprise. - - On `TransactionCanceledException` with per-item `CancellationReasons`: distinguishes between (a) approvals row missing (404 `REQUEST_NOT_FOUND`), (b) approvals row wrong user (404 `REQUEST_NOT_FOUND` — don't leak existence), (c) approvals row wrong status (409 `REQUEST_ALREADY_DECIDED`), (d) task no longer AWAITING_APPROVAL (409 `TASK_NOT_AWAITING_APPROVAL`). + - On `TransactionCanceledException` with per-item `CancellationReasons`: returns 404 `REQUEST_NOT_FOUND` for any approval-row condition failure (missing, foreign-owned or already closed), or 409 `TASK_NOT_AWAITING_APPROVAL` for a task-only condition failure. - Records audit event to TaskEventsTable directly (`approval_decision_recorded`) so the 90-day audit trail is owned by the Lambda, not dependent on agent milestones. - Returns 202 `{task_id, request_id, status: "APPROVED", scope, decided_at}` or error. 24. Agent's poll reads the `APPROVED` row on next tick (within 2-5s). @@ -950,8 +957,7 @@ Content-Type: application/json | 202 | — | Success | `{task_id, request_id, status: "APPROVED", scope, decided_at}` | | 400 | `VALIDATION_ERROR` | Bad scope format, missing fields | `{error, message, field}` | | 401 | `UNAUTHORIZED` | Missing/invalid JWT | — | -| 404 | `REQUEST_NOT_FOUND` | Row missing OR wrong user (both surfaces 404 to prevent enumeration) | — | -| 409 | `REQUEST_ALREADY_DECIDED` | Approvals row status != PENDING | `{error, message, current_status}` | +| 404 | `REQUEST_NOT_FOUND` | Approval-row condition failed: missing, foreign-owned or already closed | — | | 409 | `TASK_NOT_AWAITING_APPROVAL` | Task's current status is not AWAITING_APPROVAL | `{error, message, current_status}` | | 429 | `RATE_LIMIT_EXCEEDED` | Per-user > 30 approve/min | — | | 503 | `SERVICE_UNAVAILABLE` | DDB throttled or upstream failure | — | @@ -1000,11 +1006,12 @@ await ddb.transactWriteItems({ }); ``` -On `TransactionCanceledException`, `ApproveTaskFn` inspects the per-item `CancellationReasons` to distinguish cases: -- ApprovalsTable condition failed with `OldImage` absent → 404 `REQUEST_NOT_FOUND` -- ApprovalsTable condition failed with `OldImage.user_id != caller` → 404 (same code, prevent existence oracle) -- ApprovalsTable condition failed with `OldImage.status != "PENDING"` → 409 `REQUEST_ALREADY_DECIDED` -- TaskTable condition failed (status changed) → 409 `TASK_NOT_AWAITING_APPROVAL` +On `TransactionCanceledException`, `ApproveTaskFn` inspects per-item +`CancellationReasons`. It does not request or classify old item images: + +- Any approval-row condition failure → 404 `REQUEST_NOT_FOUND`, including + cancellation, timeout and an already-recorded decision. +- A task-only condition failure → 409 `TASK_NOT_AWAITING_APPROVAL`. This is symmetric with the agent-side `TransactWriteItems` pattern (§4 step 25a) used for the resume transition — Lambdas and agent speak the same atomic-update contract. @@ -1014,7 +1021,17 @@ After successful transaction, `ApproveTaskFn` writes an audit event to `TaskEven **Scenario (finding #6):** Three months from now, a platform engineer adds a multi-tenant mode where Cognito `sub` becomes `tenant-abc:01JXZ...`. They update the agent's row-write path to prefix-strip: `user_id = sub.split(":", 1)[1]`, storing `01JXZ...` on TaskApprovalsTable. They forget to update `ApproveTaskFn`. Now the Lambda reads `sub = "tenant-abc:01JXZ..."` from the JWT and compares it against the stored `01JXZ...` — condition fails, 404 on every approve, all tasks stranded. The fix as written: "the Cognito sub is compared verbatim; any transformation must happen at write time, not at compare time" — if the agent writes the full `sub`, the Lambda compares the full `sub`; if either side transforms, both sides must. The CI assertion is a unit test that extracts `user_id` from a sample row and asserts it matches the `sub` claim of a sample JWT byte-for-byte. This test would fail on the prefix-strip refactor above and force the engineer to update both sides. Without this hard rule, ownership-in-condition silently breaks under any future identity refactor. -**Scenario (finding #7):** A user submits a risky task at 10:00 AM. At 10:05 AM the agent hits a soft-deny gate. At 10:05:30 AM the user on Terminal B runs `bgagent cancel 01KPW...`, which lands as CancelTaskFn writes `status=CANCELLING`. At 10:05:31 AM the user — forgetting they just cancelled, or running from a different terminal where they didn't see the cancel — runs `bgagent approve 01KPW... 01KPR...`. Without the cross-table transaction, the Lambda's GetItem on TaskTable (separate call) might read the stale RUNNING state, then UpdateItem on TaskApprovalsTable succeeds because the approvals row is still PENDING → 202 returned. The user sees "approved!" but the task is dying. With the TransactWriteItems pattern, both conditions must hold: the TaskTable guard `status = AWAITING_APPROVAL` fails (because it's now CANCELLING), the entire transaction rolls back, the Lambda returns 409 `TASK_NOT_AWAITING_APPROVAL` with `current_status: CANCELLING`. The user sees "cannot approve: task is already cancelling" and correctly understands state. The cost is one extra table in the transaction (two instead of one) — still within DDB's 100-item limit and nowhere near the 4 MB request size. Symmetric with the agent's resume transaction, which already does the cross-table guard. +**Scenario (finding #7):** A user cancels a task and then approves its old request +from another terminal. Cancellation writes `CANCELLED` directly; there is no +`CANCELLING` task state. The approval transaction cannot commit because the task +is no longer `AWAITING_APPROVAL`. The P3 cancellation update also atomically closes +the linked `PENDING` approval as `CANCELLED` and writes an `approval_cancelled` +event. A later decision is rejected: the existing API returns +`404 REQUEST_NOT_FOUND` for missing, foreign or already-decided approval rows, +including a cancelled row. If approval committed first, cancellation preserves +the recorded decision while cancelling the task. See the +[P3 approval verification record](../verification/645-p3-approval-ux-20260917.md) +for source versus deployment status. ### 7.2 `POST /v1/tasks/{task_id}/deny` @@ -1133,7 +1150,11 @@ Rate-limited 30/min/user; cached 5min per repo in-Lambda. ### 7.7 `GET /v1/pending` — list pending approvals across user's active tasks -Returns all approvals with `status=PENDING` owned by the caller. Backing index: `user_id-status-index` GSI on `TaskApprovalsTable` (see §10.1). +Returns up to 100 caller-owned pending approvals whose consistently read tasks +are still awaiting that exact request. The `user_id-status-index` GSI supplies +candidates; pagination continues past cancelled/orphaned rows within a shared +five-second read budget. A read failure returns an error rather than a misleading +partial list. **Request**: `GET /v1/pending` with Cognito auth. @@ -1285,17 +1306,21 @@ stateDiagram-v2 PENDING --> APPROVED: ApproveTaskFn
(cross-table transaction) PENDING --> DENIED: DenyTaskFn
(cross-table transaction) PENDING --> TIMED_OUT: Agent poll timeout
(best-effort update) - PENDING --> STRANDED: Reconciler detects
orphan (age > 2×timeout_s) + PENDING --> CANCELLED: Task owner cancels
(cross-table transaction) APPROVED --> [*]: terminal DENIED --> [*]: terminal TIMED_OUT --> [*]: terminal - STRANDED --> [*]: terminal + CANCELLED --> [*]: terminal note right of APPROVED TTL = created_at + timeout_s + 120s DDB reaps row after TTL end note ``` +`STRANDED` remains a recognized row status in the type contract. The current +reconciler does not write it: it fails the owning task and leaves the row +`PENDING`, as described below. + ### 9.3 Orchestrator impact - `waitStrategy` adds `AWAITING_APPROVAL` as non-terminal. @@ -1328,14 +1353,21 @@ This has a concrete implication: the hook computes an `effective_timeout` bounde ### 9.6 Stranded-approval reconciliation -`reconcile-stranded-tasks.ts` gains an AWAITING_APPROVAL-aware branch: +`reconcile-stranded-tasks.ts` has an AWAITING_APPROVAL-aware branch: -- Detects tasks in AWAITING_APPROVAL with `age > 2 * timeout_s` -- Best-effort conditional-updates TaskApprovalsTable row → `STRANDED` status -- Transitions TaskTable → `FAILED` with reason `"approval stranded (container eviction)"` -- Emits `approval_stranded` event to TaskEventsTable +- Uses `APPROVAL_STRANDED_TIMEOUT_SECONDS`, default 7,200 seconds, measured from + entry into the current status. It does not calculate twice each row's timeout. +- Conditionally changes the task to `FAILED` if it is still awaiting approval, + recording the elapsed wait and a recovery suggestion. +- Emits `task_stranded`, `task_failed` and a wrapped `approval_stranded` milestone. + The legacy milestone has no request ID. +- Leaves the approval row unchanged. The pending endpoint hides it because its + owning task is terminal; late approval is rejected by the task-state guard. -This closes the container-eviction gap. Without this, a container restart mid-approval would leave the task hanging until the user manually cancelled. +The notification helper can recover that legacy milestone's request identity +from the consistently read failed task and its saved stranded cause. It verifies +approval ownership before rendering feedback and does not change the row. +The timer is a backstop, not proof of a particular container failure. `reconcile-concurrency.ts` (scheduled every 5 min) already scans for orphaned concurrency counters; with `AWAITING_APPROVAL` added to `ACTIVE_STATUSES` it correctly counts awaiting tasks as active. @@ -1401,9 +1433,9 @@ Attributes: | `reason` | S | Yes | Cedar matching rule description | | `severity` | S | Yes | "low" \| "medium" \| "high" | | `matching_rule_ids` | L | Yes | List (not Set — can be empty) of soft-deny rule IDs | -| `status` | S | Yes | PENDING \| APPROVED \| DENIED \| TIMED_OUT \| STRANDED | +| `status` | S | Yes | PENDING \| APPROVED \| DENIED \| TIMED_OUT \| STRANDED \| CANCELLED | | `created_at` | S | Yes | ISO8601 | -| `decided_at` | S | No | Set when status != PENDING | +| `decided_at` | S | No | Set by approve/deny/cancel; the current guest timeout writer can omit it | | `scope` | S | No | Set on APPROVED | | `deny_reason` | S | No | Set on DENIED; sanitized user text | | `timeout_s` | N | Yes | Resolved timeout for audit | @@ -1481,16 +1513,28 @@ Emitted to both `ProgressWriter` (DDB, 90d) and `sse_adapter` (live stream). Plu ### 11.2 Fan-out plane interaction — Slack button → Cognito mapping -Approval events flow to the fan-out Lambda via TaskEventsTable Streams (the existing Phase 1b path). They are dispatched to Slack / GitHub / Email stubs. +Approval events flow to the fan-out Lambda via TaskEventsTable Streams. The P3 +implementation adds Slack and Linear notifications with the saved action, +reason, decision deadline and exact CLI approve/deny commands. Recorded decisions, +cancellations, timeouts and stranded waits also produce messages. The response +path uses the CLI owner's authentication. Native Slack approval buttons and Linear +approval replies are not implemented by this notification change; the OAuth/button +design below remains proposed. Email remains a log-only stub and GitHub does not +receive approval messages. Deployment status is recorded in the +[P3 verification record](../verification/645-p3-approval-ux-20260917.md). **TaskApprovalsTable Streams are not consumed by the fan-out Lambda**. The approval row is working state; the audit trail is in TaskEventsTable. Enabling Streams on TaskApprovalsTable would be redundant and add noise. Final design: TaskApprovalsTable DOES NOT have Streams enabled. (Retains the `stream` attribute commented out for future use if needed.) -Fan-out dispatch rules (extending Phase 1b stubs): -- Slack: on `approval_requested` OR `approval_stranded` — "Agent @task_id requests approval for Bash: `git push --force`" -- Email: on `approval_requested` with `severity: high` -- GitHub: none +Slack and Linear route `approval_requested`, `approval_decision_recorded`, +`approval_timed_out`, `approval_cancelled` and `approval_stranded`. The dispatcher +reads the current approval row and owning task before displaying a pending request, +and records successful delivery per request/channel. Delivery failure does not mark +the message delivered. A post that succeeds just before receipt persistence fails +can still produce a duplicate on retry. -**Rate-limited per-user**: 10 approval-related fan-out messages per user per minute. Prevents notification-spam from malicious users driving up approval-gate count. +**Proposed notification rate limit:** 10 approval-related messages per user per +minute. This dispatcher limit is not implemented. Existing gate-creation caps and +API rate limits remain, but they are not a substitute for notification throttling. **Notification plane is observability, not state (see §13.14).** Notification delivery failures do NOT pause the approval timer — coupling the two creates a bypass where an adversary who takes down the webhook gets an unbounded approval window. The timer runs on the agent's local clock keyed to `created_at`; `bgagent pending` is the recovery path for users who suspect notifications are broken (backed by `user_id-status-index` GSI, §7.7). For the off-hours / unattended trade-off that this posture implies, see §14.8. @@ -1640,11 +1684,11 @@ Authorization + approvals-state + task-state transition all atomic. A compromise - User's CLI writes `APPROVED WHERE status = :pending` (via TransactWriteItems) - One wins atomically - The loser: - - If TIMED_OUT wins: user gets 409 `REQUEST_ALREADY_DECIDED`. User sees "approval expired". + - If TIMED_OUT wins: a later decision gets 404 `REQUEST_NOT_FOUND`; the saved row is already closed. - If APPROVED wins: agent's poll reads APPROVED on next tick. Agent proceeds. **Race 2 — double-approve**: -- Two concurrent CLI invocations. Second gets 409 `REQUEST_ALREADY_DECIDED`. Idempotent. +- Two concurrent CLI invocations. Only one decision commits; the second gets 404 `REQUEST_NOT_FOUND`. **Race 3 — cancel during AWAITING_APPROVAL**: - Agent writes `RUNNING WHERE status = :awaiting AND awaiting_approval_request_id = :rid` @@ -2187,7 +2231,7 @@ See §17.18 for the off-hours escalation future-work primitive, and §13.14 for - Cancel during AWAITING_APPROVAL (agent-side resume race) - Cancel during approve Lambda (cross-table transaction catches it — finding #7) - Cancel during deny with queued denial injection (between-turns hook pre-empted; `permissionDecisionReason` still delivered — finding #2) - - Late approval after TIMED_OUT (expect 409) + - Late approval after TIMED_OUT (expect 404 `REQUEST_NOT_FOUND`) - **VM-throttle + late-approve race (IMPL-24, §13.12)**: user's APPROVE lands in DDB before `_best_effort_update_status("TIMED_OUT")` can claim the row; agent must re-read with ConsistentRead and honor APPROVED (not return stale TIMED_OUT). Includes the DENIED variant and the "still PENDING" fall-through where neither side wins. - **Chaos tests**: - Container restart mid-approval (simulated via kill + reconciler) diff --git a/docs/design/DEPLOYMENT_ROLES.md b/docs/design/DEPLOYMENT_ROLES.md index c73a33117..fe905ca85 100644 --- a/docs/design/DEPLOYMENT_ROLES.md +++ b/docs/design/DEPLOYMENT_ROLES.md @@ -845,6 +845,15 @@ The second statement, `MicrovmPassRoles`, is the one exception to the rule that > **Operators must re-bootstrap for this.** The statement ships in bootstrap policy bundle **1.6.0**; a CDKToolkit stack bootstrapped at 1.5.0 or earlier will fail the CDK-managed MicroVM image deploy with a caller-side `iam:PassRole` AccessDenied on the build role. Check `CDKToolkit`'s `BootstrapPolicyVersion` output, and re-run `mise //cdk:bootstrap` (with `ComputeTypes` including `lambda-microvm`) if it is behind. P3 additionally requires **bundle 1.8.0** for `MicrovmSuspendConfiguration`. + +The nested MicroVM layout requires **bundle 1.9.0**. Its child stack uses the +explicit parent-derived names `backgroundagent-dev-MicrovmBuildRole` and +`backgroundagent-dev-MicrovmConnectorRole`; `MicrovmPassRoles` admits those two +exact names in addition to the legacy flat-layout prefixes. The execution role +stays in the parent and is still excluded. Re-bootstrap before deploying the +child stack. Existing flat deployments must keep `microvm_nested_stack=false` +until their resource migration is reviewed; changing ownership is not an ordinary +in-place update. See the [nested-stack runbook](../verification/645-p3-nested-stack.md). This statement lets CloudFormation manage and tag the live suspension setting at `//microvm-approval-suspend-enabled`. The coordinator gets only `GetParameter` on its exact parameter. Existing durable @@ -885,7 +894,9 @@ suspension, so disable can reach executions already running. "Effect": "Allow", "Resource": [ "arn:aws:iam::*:role/backgroundagent-dev-LambdaMicrovmComputeBuild*", - "arn:aws:iam::*:role/backgroundagent-dev-LambdaMicrovmComputeConnector*" + "arn:aws:iam::*:role/backgroundagent-dev-LambdaMicrovmComputeConnector*", + "arn:aws:iam::*:role/backgroundagent-dev-MicrovmBuildRole", + "arn:aws:iam::*:role/backgroundagent-dev-MicrovmConnectorRole" ], "Sid": "MicrovmPassRoles" }, diff --git a/docs/src/content/docs/architecture/Cedar-hitl-gates.md b/docs/src/content/docs/architecture/Cedar-hitl-gates.md index 1539a5a2f..dcadae836 100644 --- a/docs/src/content/docs/architecture/Cedar-hitl-gates.md +++ b/docs/src/content/docs/architecture/Cedar-hitl-gates.md @@ -9,6 +9,13 @@ title: Cedar hitl gates > **Design locked:** 2026-04-23 (Sam ↔ assistant discussion). > **Rev:** 5 (2026-05-06 — fold in parallel adversarial + advocate review of the timeout design: late-approval re-read on TIMED_OUT ConditionCheckFailed; user-visible timeout-cap milestones; ceiling-shrink milestone; Runtime JWT bound verified as auto-refreshed IAM; three new tuning metrics; explicit off-hours trade-off section; notification-delivery-failure boundary. IMPL-24 through IMPL-28 added.). > **Implementation:** Core shipped. The 3-outcome engine (`agent/src/policy.py`), default policy sets (`agent/policies/hard_deny.cedar`, `agent/policies/soft_deny.cedar`), approval Lambdas (`cdk/src/handlers/{approve-task,deny-task,get-pending,get-policies}.ts`) wired into `cdk/src/constructs/task-api.ts` (routes `/tasks/{id}/approve`, `/deny`, `/pending`, `/repos/{repo_id}/policies`), the cross-engine parity fixtures (`contracts/cedar-parity/`), and the exact engine pins are all on `main`. §15's task list is preserved as a historical implementation record; see the note at the top of §15 for what (if anything) remains unbuilt. +> +> **Current behavior clarified (2026-09-17):** closed approval-row conditions return +> `404 REQUEST_NOT_FOUND`; task-only conflicts return 409. Cancellation atomically +> closes a pending request. The stranded-task reconciler currently fails the task +> but leaves its approval row pending; the pending endpoint filters such rows. +> CLI response instructions are implemented for Slack/Linear notifications; +> native channel decisions and approvals without an expiry remain future work. --- @@ -277,7 +284,7 @@ Narrative walk-through of the happy path. Sequence diagrams in the round-trip Me - ConditionalUpdate on `TaskApprovalsTable`: `#status = :pending AND user_id = :caller AND task_id = :task_id` → flip to APPROVED - ConditionalUpdate on `TaskTable`: `#status = :awaiting AND awaiting_approval_request_id = :rid` → (no-op update, pure state guard; keeps status AWAITING_APPROVAL until the agent's resume transaction flips it RUNNING) Both conditions must hold or the entire transaction is cancelled. No TOCTOU window, no "approved a cancelled task" 202 surprise. - - On `TransactionCanceledException` with per-item `CancellationReasons`: distinguishes between (a) approvals row missing (404 `REQUEST_NOT_FOUND`), (b) approvals row wrong user (404 `REQUEST_NOT_FOUND` — don't leak existence), (c) approvals row wrong status (409 `REQUEST_ALREADY_DECIDED`), (d) task no longer AWAITING_APPROVAL (409 `TASK_NOT_AWAITING_APPROVAL`). + - On `TransactionCanceledException` with per-item `CancellationReasons`: returns 404 `REQUEST_NOT_FOUND` for any approval-row condition failure (missing, foreign-owned or already closed), or 409 `TASK_NOT_AWAITING_APPROVAL` for a task-only condition failure. - Records audit event to TaskEventsTable directly (`approval_decision_recorded`) so the 90-day audit trail is owned by the Lambda, not dependent on agent milestones. - Returns 202 `{task_id, request_id, status: "APPROVED", scope, decided_at}` or error. 24. Agent's poll reads the `APPROVED` row on next tick (within 2-5s). @@ -954,8 +961,7 @@ Content-Type: application/json | 202 | — | Success | `{task_id, request_id, status: "APPROVED", scope, decided_at}` | | 400 | `VALIDATION_ERROR` | Bad scope format, missing fields | `{error, message, field}` | | 401 | `UNAUTHORIZED` | Missing/invalid JWT | — | -| 404 | `REQUEST_NOT_FOUND` | Row missing OR wrong user (both surfaces 404 to prevent enumeration) | — | -| 409 | `REQUEST_ALREADY_DECIDED` | Approvals row status != PENDING | `{error, message, current_status}` | +| 404 | `REQUEST_NOT_FOUND` | Approval-row condition failed: missing, foreign-owned or already closed | — | | 409 | `TASK_NOT_AWAITING_APPROVAL` | Task's current status is not AWAITING_APPROVAL | `{error, message, current_status}` | | 429 | `RATE_LIMIT_EXCEEDED` | Per-user > 30 approve/min | — | | 503 | `SERVICE_UNAVAILABLE` | DDB throttled or upstream failure | — | @@ -1004,11 +1010,12 @@ await ddb.transactWriteItems({ }); ``` -On `TransactionCanceledException`, `ApproveTaskFn` inspects the per-item `CancellationReasons` to distinguish cases: -- ApprovalsTable condition failed with `OldImage` absent → 404 `REQUEST_NOT_FOUND` -- ApprovalsTable condition failed with `OldImage.user_id != caller` → 404 (same code, prevent existence oracle) -- ApprovalsTable condition failed with `OldImage.status != "PENDING"` → 409 `REQUEST_ALREADY_DECIDED` -- TaskTable condition failed (status changed) → 409 `TASK_NOT_AWAITING_APPROVAL` +On `TransactionCanceledException`, `ApproveTaskFn` inspects per-item +`CancellationReasons`. It does not request or classify old item images: + +- Any approval-row condition failure → 404 `REQUEST_NOT_FOUND`, including + cancellation, timeout and an already-recorded decision. +- A task-only condition failure → 409 `TASK_NOT_AWAITING_APPROVAL`. This is symmetric with the agent-side `TransactWriteItems` pattern (§4 step 25a) used for the resume transition — Lambdas and agent speak the same atomic-update contract. @@ -1018,7 +1025,17 @@ After successful transaction, `ApproveTaskFn` writes an audit event to `TaskEven **Scenario (finding #6):** Three months from now, a platform engineer adds a multi-tenant mode where Cognito `sub` becomes `tenant-abc:01JXZ...`. They update the agent's row-write path to prefix-strip: `user_id = sub.split(":", 1)[1]`, storing `01JXZ...` on TaskApprovalsTable. They forget to update `ApproveTaskFn`. Now the Lambda reads `sub = "tenant-abc:01JXZ..."` from the JWT and compares it against the stored `01JXZ...` — condition fails, 404 on every approve, all tasks stranded. The fix as written: "the Cognito sub is compared verbatim; any transformation must happen at write time, not at compare time" — if the agent writes the full `sub`, the Lambda compares the full `sub`; if either side transforms, both sides must. The CI assertion is a unit test that extracts `user_id` from a sample row and asserts it matches the `sub` claim of a sample JWT byte-for-byte. This test would fail on the prefix-strip refactor above and force the engineer to update both sides. Without this hard rule, ownership-in-condition silently breaks under any future identity refactor. -**Scenario (finding #7):** A user submits a risky task at 10:00 AM. At 10:05 AM the agent hits a soft-deny gate. At 10:05:30 AM the user on Terminal B runs `bgagent cancel 01KPW...`, which lands as CancelTaskFn writes `status=CANCELLING`. At 10:05:31 AM the user — forgetting they just cancelled, or running from a different terminal where they didn't see the cancel — runs `bgagent approve 01KPW... 01KPR...`. Without the cross-table transaction, the Lambda's GetItem on TaskTable (separate call) might read the stale RUNNING state, then UpdateItem on TaskApprovalsTable succeeds because the approvals row is still PENDING → 202 returned. The user sees "approved!" but the task is dying. With the TransactWriteItems pattern, both conditions must hold: the TaskTable guard `status = AWAITING_APPROVAL` fails (because it's now CANCELLING), the entire transaction rolls back, the Lambda returns 409 `TASK_NOT_AWAITING_APPROVAL` with `current_status: CANCELLING`. The user sees "cannot approve: task is already cancelling" and correctly understands state. The cost is one extra table in the transaction (two instead of one) — still within DDB's 100-item limit and nowhere near the 4 MB request size. Symmetric with the agent's resume transaction, which already does the cross-table guard. +**Scenario (finding #7):** A user cancels a task and then approves its old request +from another terminal. Cancellation writes `CANCELLED` directly; there is no +`CANCELLING` task state. The approval transaction cannot commit because the task +is no longer `AWAITING_APPROVAL`. The P3 cancellation update also atomically closes +the linked `PENDING` approval as `CANCELLED` and writes an `approval_cancelled` +event. A later decision is rejected: the existing API returns +`404 REQUEST_NOT_FOUND` for missing, foreign or already-decided approval rows, +including a cancelled row. If approval committed first, cancellation preserves +the recorded decision while cancelling the task. See the +[P3 approval verification record](/sample-autonomous-cloud-coding-agents/architecture/645-p3-approval-ux-20260917) +for source versus deployment status. ### 7.2 `POST /v1/tasks/{task_id}/deny` @@ -1137,7 +1154,11 @@ Rate-limited 30/min/user; cached 5min per repo in-Lambda. ### 7.7 `GET /v1/pending` — list pending approvals across user's active tasks -Returns all approvals with `status=PENDING` owned by the caller. Backing index: `user_id-status-index` GSI on `TaskApprovalsTable` (see §10.1). +Returns up to 100 caller-owned pending approvals whose consistently read tasks +are still awaiting that exact request. The `user_id-status-index` GSI supplies +candidates; pagination continues past cancelled/orphaned rows within a shared +five-second read budget. A read failure returns an error rather than a misleading +partial list. **Request**: `GET /v1/pending` with Cognito auth. @@ -1289,17 +1310,21 @@ stateDiagram-v2 PENDING --> APPROVED: ApproveTaskFn
(cross-table transaction) PENDING --> DENIED: DenyTaskFn
(cross-table transaction) PENDING --> TIMED_OUT: Agent poll timeout
(best-effort update) - PENDING --> STRANDED: Reconciler detects
orphan (age > 2×timeout_s) + PENDING --> CANCELLED: Task owner cancels
(cross-table transaction) APPROVED --> [*]: terminal DENIED --> [*]: terminal TIMED_OUT --> [*]: terminal - STRANDED --> [*]: terminal + CANCELLED --> [*]: terminal note right of APPROVED TTL = created_at + timeout_s + 120s DDB reaps row after TTL end note ``` +`STRANDED` remains a recognized row status in the type contract. The current +reconciler does not write it: it fails the owning task and leaves the row +`PENDING`, as described below. + ### 9.3 Orchestrator impact - `waitStrategy` adds `AWAITING_APPROVAL` as non-terminal. @@ -1332,14 +1357,21 @@ This has a concrete implication: the hook computes an `effective_timeout` bounde ### 9.6 Stranded-approval reconciliation -`reconcile-stranded-tasks.ts` gains an AWAITING_APPROVAL-aware branch: +`reconcile-stranded-tasks.ts` has an AWAITING_APPROVAL-aware branch: -- Detects tasks in AWAITING_APPROVAL with `age > 2 * timeout_s` -- Best-effort conditional-updates TaskApprovalsTable row → `STRANDED` status -- Transitions TaskTable → `FAILED` with reason `"approval stranded (container eviction)"` -- Emits `approval_stranded` event to TaskEventsTable +- Uses `APPROVAL_STRANDED_TIMEOUT_SECONDS`, default 7,200 seconds, measured from + entry into the current status. It does not calculate twice each row's timeout. +- Conditionally changes the task to `FAILED` if it is still awaiting approval, + recording the elapsed wait and a recovery suggestion. +- Emits `task_stranded`, `task_failed` and a wrapped `approval_stranded` milestone. + The legacy milestone has no request ID. +- Leaves the approval row unchanged. The pending endpoint hides it because its + owning task is terminal; late approval is rejected by the task-state guard. -This closes the container-eviction gap. Without this, a container restart mid-approval would leave the task hanging until the user manually cancelled. +The notification helper can recover that legacy milestone's request identity +from the consistently read failed task and its saved stranded cause. It verifies +approval ownership before rendering feedback and does not change the row. +The timer is a backstop, not proof of a particular container failure. `reconcile-concurrency.ts` (scheduled every 5 min) already scans for orphaned concurrency counters; with `AWAITING_APPROVAL` added to `ACTIVE_STATUSES` it correctly counts awaiting tasks as active. @@ -1405,9 +1437,9 @@ Attributes: | `reason` | S | Yes | Cedar matching rule description | | `severity` | S | Yes | "low" \| "medium" \| "high" | | `matching_rule_ids` | L | Yes | List (not Set — can be empty) of soft-deny rule IDs | -| `status` | S | Yes | PENDING \| APPROVED \| DENIED \| TIMED_OUT \| STRANDED | +| `status` | S | Yes | PENDING \| APPROVED \| DENIED \| TIMED_OUT \| STRANDED \| CANCELLED | | `created_at` | S | Yes | ISO8601 | -| `decided_at` | S | No | Set when status != PENDING | +| `decided_at` | S | No | Set by approve/deny/cancel; the current guest timeout writer can omit it | | `scope` | S | No | Set on APPROVED | | `deny_reason` | S | No | Set on DENIED; sanitized user text | | `timeout_s` | N | Yes | Resolved timeout for audit | @@ -1485,16 +1517,28 @@ Emitted to both `ProgressWriter` (DDB, 90d) and `sse_adapter` (live stream). Plu ### 11.2 Fan-out plane interaction — Slack button → Cognito mapping -Approval events flow to the fan-out Lambda via TaskEventsTable Streams (the existing Phase 1b path). They are dispatched to Slack / GitHub / Email stubs. +Approval events flow to the fan-out Lambda via TaskEventsTable Streams. The P3 +implementation adds Slack and Linear notifications with the saved action, +reason, decision deadline and exact CLI approve/deny commands. Recorded decisions, +cancellations, timeouts and stranded waits also produce messages. The response +path uses the CLI owner's authentication. Native Slack approval buttons and Linear +approval replies are not implemented by this notification change; the OAuth/button +design below remains proposed. Email remains a log-only stub and GitHub does not +receive approval messages. Deployment status is recorded in the +[P3 verification record](/sample-autonomous-cloud-coding-agents/architecture/645-p3-approval-ux-20260917). **TaskApprovalsTable Streams are not consumed by the fan-out Lambda**. The approval row is working state; the audit trail is in TaskEventsTable. Enabling Streams on TaskApprovalsTable would be redundant and add noise. Final design: TaskApprovalsTable DOES NOT have Streams enabled. (Retains the `stream` attribute commented out for future use if needed.) -Fan-out dispatch rules (extending Phase 1b stubs): -- Slack: on `approval_requested` OR `approval_stranded` — "Agent @task_id requests approval for Bash: `git push --force`" -- Email: on `approval_requested` with `severity: high` -- GitHub: none +Slack and Linear route `approval_requested`, `approval_decision_recorded`, +`approval_timed_out`, `approval_cancelled` and `approval_stranded`. The dispatcher +reads the current approval row and owning task before displaying a pending request, +and records successful delivery per request/channel. Delivery failure does not mark +the message delivered. A post that succeeds just before receipt persistence fails +can still produce a duplicate on retry. -**Rate-limited per-user**: 10 approval-related fan-out messages per user per minute. Prevents notification-spam from malicious users driving up approval-gate count. +**Proposed notification rate limit:** 10 approval-related messages per user per +minute. This dispatcher limit is not implemented. Existing gate-creation caps and +API rate limits remain, but they are not a substitute for notification throttling. **Notification plane is observability, not state (see §13.14).** Notification delivery failures do NOT pause the approval timer — coupling the two creates a bypass where an adversary who takes down the webhook gets an unbounded approval window. The timer runs on the agent's local clock keyed to `created_at`; `bgagent pending` is the recovery path for users who suspect notifications are broken (backed by `user_id-status-index` GSI, §7.7). For the off-hours / unattended trade-off that this posture implies, see §14.8. @@ -1644,11 +1688,11 @@ Authorization + approvals-state + task-state transition all atomic. A compromise - User's CLI writes `APPROVED WHERE status = :pending` (via TransactWriteItems) - One wins atomically - The loser: - - If TIMED_OUT wins: user gets 409 `REQUEST_ALREADY_DECIDED`. User sees "approval expired". + - If TIMED_OUT wins: a later decision gets 404 `REQUEST_NOT_FOUND`; the saved row is already closed. - If APPROVED wins: agent's poll reads APPROVED on next tick. Agent proceeds. **Race 2 — double-approve**: -- Two concurrent CLI invocations. Second gets 409 `REQUEST_ALREADY_DECIDED`. Idempotent. +- Two concurrent CLI invocations. Only one decision commits; the second gets 404 `REQUEST_NOT_FOUND`. **Race 3 — cancel during AWAITING_APPROVAL**: - Agent writes `RUNNING WHERE status = :awaiting AND awaiting_approval_request_id = :rid` @@ -2191,7 +2235,7 @@ See §17.18 for the off-hours escalation future-work primitive, and §13.14 for - Cancel during AWAITING_APPROVAL (agent-side resume race) - Cancel during approve Lambda (cross-table transaction catches it — finding #7) - Cancel during deny with queued denial injection (between-turns hook pre-empted; `permissionDecisionReason` still delivered — finding #2) - - Late approval after TIMED_OUT (expect 409) + - Late approval after TIMED_OUT (expect 404 `REQUEST_NOT_FOUND`) - **VM-throttle + late-approve race (IMPL-24, §13.12)**: user's APPROVE lands in DDB before `_best_effort_update_status("TIMED_OUT")` can claim the row; agent must re-read with ConsistentRead and honor APPROVED (not return stale TIMED_OUT). Includes the DENIED variant and the "still PENDING" fall-through where neither side wins. - **Chaos tests**: - Container restart mid-approval (simulated via kill + reconciler) diff --git a/docs/src/content/docs/architecture/Deployment-roles.md b/docs/src/content/docs/architecture/Deployment-roles.md index f095038ab..c115a54ba 100644 --- a/docs/src/content/docs/architecture/Deployment-roles.md +++ b/docs/src/content/docs/architecture/Deployment-roles.md @@ -849,6 +849,15 @@ The second statement, `MicrovmPassRoles`, is the one exception to the rule that > **Operators must re-bootstrap for this.** The statement ships in bootstrap policy bundle **1.6.0**; a CDKToolkit stack bootstrapped at 1.5.0 or earlier will fail the CDK-managed MicroVM image deploy with a caller-side `iam:PassRole` AccessDenied on the build role. Check `CDKToolkit`'s `BootstrapPolicyVersion` output, and re-run `mise //cdk:bootstrap` (with `ComputeTypes` including `lambda-microvm`) if it is behind. P3 additionally requires **bundle 1.8.0** for `MicrovmSuspendConfiguration`. + +The nested MicroVM layout requires **bundle 1.9.0**. Its child stack uses the +explicit parent-derived names `backgroundagent-dev-MicrovmBuildRole` and +`backgroundagent-dev-MicrovmConnectorRole`; `MicrovmPassRoles` admits those two +exact names in addition to the legacy flat-layout prefixes. The execution role +stays in the parent and is still excluded. Re-bootstrap before deploying the +child stack. Existing flat deployments must keep `microvm_nested_stack=false` +until their resource migration is reviewed; changing ownership is not an ordinary +in-place update. See the [nested-stack runbook](/sample-autonomous-cloud-coding-agents/architecture/645-p3-nested-stack). This statement lets CloudFormation manage and tag the live suspension setting at `//microvm-approval-suspend-enabled`. The coordinator gets only `GetParameter` on its exact parameter. Existing durable @@ -889,7 +898,9 @@ suspension, so disable can reach executions already running. "Effect": "Allow", "Resource": [ "arn:aws:iam::*:role/backgroundagent-dev-LambdaMicrovmComputeBuild*", - "arn:aws:iam::*:role/backgroundagent-dev-LambdaMicrovmComputeConnector*" + "arn:aws:iam::*:role/backgroundagent-dev-LambdaMicrovmComputeConnector*", + "arn:aws:iam::*:role/backgroundagent-dev-MicrovmBuildRole", + "arn:aws:iam::*:role/backgroundagent-dev-MicrovmConnectorRole" ], "Sid": "MicrovmPassRoles" }, diff --git a/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md b/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md index fe841a7d4..7499a11c5 100644 --- a/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md +++ b/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md @@ -4,9 +4,9 @@ title: Adr 021 lambda microvms compute backend # ADR-021: AWS Lambda MicroVMs as a third ComputeStrategy backend -> **Implementation status (2026-09-16):** P1 ([#689](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/689)) and P2 ([#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733)) are merged. The takeover has a [clean deployment](/sample-autonomous-cloud-coding-agents/architecture/645-p2-clean-deployment-20260913), [live repository workflow evidence](/sample-autonomous-cloud-coding-agents/architecture/645-p2-live-task-20260914), and integrated P3 guest hooks, credentials, durable intent and approval/supervisor handling. The current normal deployment uses agent image **5.0**, coordinator **6** and bootstrap **1.8.0**. [Long-sleep credential/deadline acceptance](/sample-autonomous-cloud-coding-agents/architecture/645-p3-callback-live-20260915) and the [pending-wake timer corrections](/sample-autonomous-cloud-coding-agents/architecture/645-p3-pending-wake) passed live checks. +> **Implementation status (2026-09-17):** P1 ([#689](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/689)) and P2 ([#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733)) are merged. The takeover has a [clean deployment](/sample-autonomous-cloud-coding-agents/architecture/645-p2-clean-deployment-20260913), [live repository workflow evidence](/sample-autonomous-cloud-coding-agents/architecture/645-p2-live-task-20260914), and integrated P3 guest hooks, credentials, durable intent and approval/supervisor handling. The normal deployment uses agent image **7.0**, coordinator live alias **10** and bootstrap **1.9.0**. The [approval UX update](/sample-autonomous-cloud-coding-agents/architecture/645-p3-approval-ux-20260917) deploys atomic cancellation, channel notification support and guest cancellation handling. The [final image tests](/sample-autonomous-cloud-coding-agents/architecture/645-p3-final-image-and-ecs-20260917), [repository workflow](/sample-autonomous-cloud-coding-agents/architecture/645-p3-repository-path-20260917), and [MCP/network checks](/sample-autonomous-cloud-coding-agents/architecture/645-p3-mcp-network-20260917) provide nine successful image 6.0 approval workflows, including real credential renewal after expiry. > -> **P3 remains incomplete; automatic suspension is disabled.** Four [resume connection refusals](/sample-autonomous-cloud-coding-agents/architecture/645-p3-resume-refusal-investigation) remain unexplained, alongside remaining race, permission/network, recovery and rollout checks. New [lifecycle diagnostics and wake feedback](/sample-autonomous-cloud-coding-agents/architecture/645-p3-lifecycle-diagnostics) passed isolated AWS workflows and are [deployed to the normal stack](/sample-autonomous-cloud-coding-agents/architecture/645-p3-diagnostics-rollout-20260916). Six further [live command-race cases](/sample-autonomous-cloud-coding-agents/architecture/645-p3-command-races-20260916) now pass, including cancellation during suspension/restore and bounded polling failure. Follow the [current implementation plan](/sample-autonomous-cloud-coding-agents/architecture/645-p3-implementation-plan) and [service-team feedback tracker](/sample-autonomous-cloud-coding-agents/architecture/645-lambda-microvm-service-feedback). This ADR defines P1–P3, not a P4. +> **P3 remains incomplete; automatic suspension is disabled.** Lifecycle responses now explicitly close their HTTP connections before freezing, with [transport evidence and a deployed fix](/sample-autonomous-cloud-coding-agents/architecture/645-p3-connection-close-rollout-20260916). Historical service-side dispatch traces remain unavailable; passing workflows do not supply those traces. Remaining work includes normal activation/off-switch acceptance, approval UX work tracked in the [current implementation plan](/sample-autonomous-cloud-coding-agents/architecture/645-p3-implementation-plan), and deployment of the requested [nested MicroVM stack](/sample-autonomous-cloud-coding-agents/architecture/645-p3-nested-stack). Bootstrap **1.9.0** is installed; moving the existing flat deployment still requires a reviewed resource migration. Follow the [service-team feedback tracker](/sample-autonomous-cloud-coding-agents/architecture/645-lambda-microvm-service-feedback) for open service questions. This ADR defines P1–P3, not a P4. **Status:** proposed **Date:** 2026-07-29 @@ -142,7 +142,7 @@ The handshake must respect the existing approval mechanics: the agent **discover It does **not** relieve the orchestrator of anything, because the two cases are disjoint. The service reaps a hook *result* it did not like; it has no view of the guest once the hook returned 200. So a task that starts normally — the overwhelming majority — has no service-side reaper at all, and a VM whose pipeline finished, crashed after `/run`, or hung is reaped by nobody but `TerminateMicrovm`. A leaked handle therefore remains a cost incident that bills until the 8 h cap; only the "the guest rejected its own payload" corner now cleans itself up. - **Concurrency slot stays held** during suspend. Cedar decision #7's rationale ("container alive, consuming memory") weakens under suspend, and the harder replacement rationale — "AWS counts `SUSPENDED` MicroVMs toward the account memory quota, so releasing ABCA's slot would not free real capacity" — is **undischarged**: the suspended VM stayed in `list-microvms` at every checkpoint, but that only proves *listed*. `L-CD1C0CC4` (1024 GB, account-scoped) exposes no `UsageMetric`, `AWS/Usage` carries only `CallCount` per API, and no MicroVM memory metric exists in any namespace, so consumption is **not observable safely** — proving it would need a large concurrent fleet. The conclusion (hold the slot) stands as the conservative choice, not as a verified fact. Size the arithmetic against the 32 GiB **peak** rather than the 8 GiB baseline: a busy fleet scales up, so peak is what actually competes for the account quota. -In P3, the agent's `/suspend` hook must acknowledge durable progress before returning 200 within its hook budget; `/resume` must refresh credentials and reseed application PRNG state from fresh OS entropy. These hooks are implemented locally with atomic checkpoints, retained credential refresh and original-gate reconciliation; matching deployment and live acceptance remain open. A non-cryptographic PRNG such as Python `random` remains unsuitable for secrets even after reseeding; security-sensitive randomness must use an OS-backed cryptographic source. +In P3, the agent's `/suspend` hook must acknowledge durable progress before returning 200 within its hook budget; `/resume` must refresh credentials and reseed application PRNG state from fresh OS entropy. These hooks are deployed in image 6.0 with atomic checkpoints, retained credential refresh and original-gate reconciliation, and have passed the isolated live workflows linked above. Automatic suspension on the normal task path remains disabled pending rollout acceptance. A non-cryptographic PRNG such as Python `random` remains unsuitable for secrets even after reseeding; security-sensitive randomness must use an OS-backed cryptographic source. **Amended (P2 review): Application PRNG reseeding is P3 scope, and the P2 exposure is negligible.** The original EARS requirement below implied a `/run`-time reseed had to land with P2; it did not, and shipping P2 with the requirement unmet-but-asserted was itself the defect. Measured exposure, which is what changed the scoping: @@ -180,7 +180,7 @@ The phasing is therefore: | `/run` | **P1** (construct sets `hooks.microvmHooks.run`) | **P1** | The payload-delivery channel. Must be served in P1 because `/ready` forces hooks to exist at all, and a hook-less image cannot accept `runHookPayload`. Since P2 it is also the **platform-configuration** channel (see "Platform configuration delivery" below). | | `/validate` | **P2** (construct sets `hooks.microvmImageHooks.validate`) | **P2** | An **image** (build-time) hook, and a **shallow self-check only**: server alive, hook routes registered, interpreter + contract sanity. It runs under the BUILD role, which deliberately holds no Bedrock / Secrets Manager / DynamoDB grants, so it must make **zero AWS API calls** and must not touch credential resolution — the "deeper warm-up assertions (Bedrock reachability, Memory access, tool availability)" this ADR originally assigned here are **not implementable**: every one of them would `AccessDenied` and fail every image build. They belong to the first task's own error handling. 200 when the checks pass, 503 while still initialising. | | `/terminate` | **P2** (construct sets `hooks.microvmHooks.terminate`) | **P2** | Best-effort final flush: a final structured log line, then 200 — always, inside the hook budget, even with nothing running. It must **not** write terminal task status (the orchestrator normally finalizes before `TerminateMicrovm`, and also attempts cleanup if database finalization fails, so a terminate hook that wrote a status would race that finalization and could clobber the real outcome) and must not join the pipeline thread. `ProgressWriter` writes synchronously but drops failed events, so this hook cannot guarantee that every progress write succeeded. P3 supplies a separate acknowledged checkpoint for suspension. "Always 200" also covers the BODY: the handler reads the raw request rather than a typed model, because a typed body is validated before the handler runs and would answer 422 to malformed JSON — a reported hook failure on a successful teardown. Safe to declare because `TerminateMicrovm` removes the VM with or without in-guest cooperation. **Correction (P2-F8):** the service sends `microvmId: ""` on this hook, unlike `/run` where it is populated, so an empty id is expected-normal and this hook cannot join the guest record to the control-plane one — `/run`'s accepted line carries that correlation instead. | -| `/suspend`, `/resume` | **P3**, implemented locally | **P3**, implemented locally | Managed images declare the served checkpoint/credential/gate hooks with the shared 30-second service timeout and protocol marker. The coordinator verifies the actual launched version before allowing new suspension. Supervisor integration is implemented locally with a default-off rollout flag; live acceptance remains open. | +| `/suspend`, `/resume` | **P3**, deployed in image 6.0 | **P3**, deployed in image 6.0 | Managed images declare the served checkpoint/credential/gate hooks with the shared 30-second service timeout and protocol marker. The coordinator verifies the actual launched version before allowing new suspension. Lifecycle responses close the connection before freezing. Isolated live workflows pass; normal automatic-suspension activation remains gated off. | Consequence to state plainly, replacing the original "a P1-built MicroVM image is not runnable end to end": **a P1 image is creatable, launchable and payload-deliverable, but carries no smoke-parity guarantee.** P1 delivers the strategy, the construct, the roles/buckets/connectors, the image resource, the packaging script, and the `/ready` + `/run` endpoints — so a `lambda-microvm` task can start a MicroVM and hand it a payload. What P1 has **not** established is anything P2 owns: AgentCore Memory grants and `MEMORY_ID` delivery, the agent's non-secret env parity inside the snapshot, egress specifics from a running MicroVM, and heartbeat/progress behaviour end to end. At the P1 handoff, no clone → change → PR run had happened. P2 initially demonstrated that path with an IAM workaround; the 2026-09-14 clean deployment and coding/iteration/cancellation runs later passed without manual IAM changes. The broader failure/recovery, effective IAM and networking matrix remains open. The construct and the packaging script surface that remaining scope at synth/run time (`abca:microvm-image-p1-smoke-unverified`) so an operator cannot mistake a launchable substrate for a verified one. diff --git a/docs/verification/645-p3-approval-ux-20260917.md b/docs/verification/645-p3-approval-ux-20260917.md new file mode 100644 index 000000000..f5e68fdcb --- /dev/null +++ b/docs/verification/645-p3-approval-ux-20260917.md @@ -0,0 +1,267 @@ +# P3 cancellation and approval notifications + +This records the September 16–17 implementation while the user was away. +The cancellation/API/notification update is deployed on `backgroundagent-dev` in +`us-west-2`. The subsequent managed-image update deploys the guest hook change in +image 7.0. Normal automatic sleep remains disabled. + +## Cancellation behavior + +REST/CLI and Slack cancellation use one shared state operation. Cancelling a task +with a pending approval writes all three changes in one DynamoDB transaction: + +1. The task becomes `CANCELLED`. +2. Its unanswered request becomes `CANCELLED`, with the owner's cancellation reason. +3. An `approval_cancelled` audit event is saved. + +The write checks ownership, the observed task status and the exact request ID. +If a decision or new request wins the race, cancellation reloads state and retries +up to three times within its five-second budget. A previously recorded approval or +denial remains recorded. Unexpected database failures do not fall back to cancelling +only half the state. + +The existing approve/deny API deliberately reports `404 REQUEST_NOT_FOUND` for +any failed approval-row condition, including already-decided/cancelled requests. +It cannot distinguish missing rows from foreign ownership using the current +transaction response. A task-only condition failure returns 409. The cancellation +reason is recorded in the approval row and closure event. + +The pending endpoint checks the owning tasks with consistent batch reads. Requests +whose tasks are terminal, missing or waiting on another request are omitted. This +also hides orphaned requests created by older cancellation code. It does not judge +whether the requested action is still relevant. + +A subsequent review found that a full first page of legacy cancelled requests +could hide a current request on the next page. The source now follows the index +cursor until it finds 100 live requests or exhausts the results, sharing a +five-second read budget. Read failures return an error instead of an incomplete +empty list. The 21 focused tests pass. The follow-up is deployed and its real +101-request check returned the single current request from page two after +examining 100 cancelled requests on page one. All 202 temporary task/approval +records were removed and consistent reads confirmed their absence. +The deployed pagination proof and cleanup are archived at +`/Users/sphias/.local/share/abca-verification/645-p3-20260916/pending-pagination`. + +The guest hook recognizes cancellation immediately, including when cancellation +wins a timeout race. It denies that tool call without restoring the task to +`RUNNING` or inventing a human denial for the agent's decision cache. + +## Notification and response path + +Slack and Linear receive pending requests, saved decisions and closure reasons. +The message identifies the task/request, tool, reason, action preview and current +decision deadline. It supplies these commands with the actual IDs: + +```text +bgagent approve TASK_ID REQUEST_ID --scope this_call +bgagent deny TASK_ID REQUEST_ID +``` + +The user runs them while signed in as the task owner. Slack messages stay in the +original thread; Linear posts on the originating issue. Native Slack buttons and +Linear comment-to-decision parsing remain unimplemented. + +The router now unwraps approval milestones from `agent_milestone`, as it already +did for PR creation. The Slack renderer no longer drops those events, and Linear +has explicit approval subscriptions. Approval handling returns before Linear's +terminal-reply path, so asking a question cannot accidentally mark a task finished. + +The dispatcher reads saved approval state rather than trusting delayed event +previews. It suppresses old pending notifications after closure and records +successful delivery separately for each request, event type and channel. External +post success followed by a lost receipt write can still duplicate a notification. +Known credential shapes are redacted before truncation. Slack uses plain text; +Linear keeps previews inside a literal code block. Repository text cannot supply +the generated approval command. + +CLI watch shows the action, reason, exact response commands and cancellation +feedback. A saved decision is described as recorded; it does not claim the sleeping +worker has successfully resumed. + +The final review also corrects three closure-feedback cases. Timeout messages +use the saved polling failure or explain that the deadline elapsed, instead of +repeating the original policy reason. A delayed approval/denial message reports +an already-ended task's status instead of promising that it will continue. +The current stranded-task reconciler emits `approval_stranded` without a +request ID and leaves the approval row `PENDING`; the notifier handles that +legacy shape only when the consistently read failed task, saved stranded cause, +exact pending request and ownership agree. This supplies feedback without changing +the approval row or adding a relevance check. Missing identity is logged. + +The review also found a pre-existing stream retry defect. Failed deliveries +returned the opaque event ID; AWS requires the DynamoDB record's sequence number. +The handler now returns the sequence, retaining it as a string, and rejects the +whole batch if a failed record has no sequence. AWS retries from the lowest +failed sequence onward, so successful later records can repeat and still need +delivery receipts. See the +[AWS partial-batch contract](https://docs.aws.amazon.com/lambda/latest/dg/services-ddb-batchfailurereporting.html). + +## Verification + +- Focused cancellation/store/handler checks: 149 CDK tests passed. +- Full Python quality: 1,957 passed, 11 skipped; lint/format/type checks passed. +- Full CLI build: 943 tests passed; compilation and lint passed. +- Full CDK suite: 5,126 tests passed, 56 skipped; the snapshot passed. +- Documentation sync and the 77-page site build passed. +- Additional construct IAM checks: 17 passed. Type synchronization, final + compilation and lint passed. +- After the pagination follow-up: 21 pending-handler tests, compilation and lint + passed. The packaging instruction cleanup passed seven existing regression tests + and Bash syntax validation. +- After the final closure/retry corrections: 241 notification and routing tests, + compilation, lint and type synchronization passed. + +The production TypeScript helpers also ran against three isolated real DynamoDB +tables in ``, `us-west-2`. Seven scenario groups passed: atomic pending +cancellation, late approval refusal, preservation of an existing approval, a new +gate racing an old task snapshot, cancellation without a gate, 12 simultaneous +approve/cancel transactions, and per-channel delivery receipts/closure feedback. +Approval won first in all 12 simultaneous races; cancellation preserved that +decision and cancelled the task. The separate cancel-first case rejected approval. + +All temporary tables were deleted and `DescribeTable` returned +`ResourceNotFoundException` for each. No Slack or Linear messages were sent during +verification. Raw evidence: `/tmp/abca-645-p3-approval-live-20260916`; permanent +archive: `/Users/sphias/.local/share/abca-verification/645-p3-20260916/approval-ux`. + +## Bounded deployment + +CloudFormation completed `p3-approval-ux-20260917-r2`, started at +`2026-09-17T03:47:12.105Z`. The reviewed template SHA-256 is +`2d75f6c84ea625ce2ff12cf189795476b22ee8e342ca3e7daddde59313b7a6e3`. +It changes five function packages (cancel, pending, Slack interactions, fan-out +and upload confirmation) and four additive table-access policies. CloudFormation +also lists eight existing API references because they depend on those function +ARNs. Their template definitions are unchanged. + +All 475 resource identities, compute resources, image 6.0, coordinator alias 10 and +parent outputs are preserved. The upload-confirmation package includes the earlier +correction explaining that startup was not confirmed, instead of promising an +automatic retry that does not exist. + +The initial review caught a synthesis-region mismatch before deployment. The final +assembly explicitly uses `AWS_REGION=us-west-2`, `AWS_DEFAULT_REGION=us-west-2` and +checks the manifest environment. The first change set was deleted without execution +so the final packages include the lint cleanup. Deployment artifacts and postchecks +are under `/tmp/abca-645-p3-approval-rollout`. + +Post-deployment verification at `2026-09-17T12:06:26Z` confirmed all five function +code hashes against the actual S3 ZIP bytes and exercised the deployed handlers +with two owned, API-origin task fixtures. Pending requests on cancelled tasks were +hidden; cancelling the live fixture atomically closed its task and request and +saved the audit event; a later approval returned `404 REQUEST_NOT_FOUND`; the +pending list was then empty. These calls used the handler's authenticated-event +shape through Lambda invocation, not an additional API Gateway login test. +Neither fixture launched a worker or sent a channel message. + +All owned task, approval, event and rate-limit records were removed. Eight +independent consistent reads confirmed their absence. Alias 10 and the live sleep +switch set to `false` were verified again. The two earlier verification attempts +used an incomplete request body and an incorrect expected HTTP status; their +fixtures were also cleaned up, and their logs are retained separately. + +The reviewed template, deployment receipts, code hashes, postchecks and cleanup +proof are archived with SHA-256 hashes at +`/Users/sphias/.local/share/abca-verification/645-p3-20260916/approval-rollout`. + +## Guest image deployment + +The subsequent `p3-cancel-image-20260917-r2` update deployed image **7.0**. +The artifact has 111 files and is 486,955 bytes, SHA-256 +`b4bc0c628f4c7976e18850ff47f92e78ecdccec683bc38974e0da82fc8600049`. +Compared with image 6.0, only `hooks.py` changes executable behavior; three other +files contain comment corrections and 107 files, including the Dockerfile, are +identical. + +The executed template SHA-256 is +`aa27f9e38107ad28b3bfeac7773797daadfae954b550cd78d21fd71f8fcc4c11`. +It changes the image artifact and its exact build read permission. The five +additional CloudFormation notices refer to unchanged consumers of the image ARN. +An expanded preview removed automatic notices for the unchanged existing nested +stacks; the initial preview was deleted without execution. + +Postchecks confirm `ACTIVE` / `SUCCESSFUL`, 8,192 MiB, unchanged image configuration, +retention of image 6.0, all 475 physical resource identities, coordinator alias 10 +and both sleep switches off. + +Two real Durable workflows then passed on image 7.0 using a private coordinator, +the production handler and normal decision APIs: + +| Case | Task | Result | +| --- | --- | --- | +| Approve after 60 seconds asleep | `01M2QMKXPQMHF2HZWS49R2G2S0` | Original PID 1 resumed with HTTP 200, one Read succeeded, task completed | +| Cancel while asleep | `01M2QMKXPZVE13FED8FG4EAYT8` | Task and approval became `CANCELLED`, one closure event, worker terminated without Resume or a successful Read | + +The approved wake has AWS request ID `233838bc-a1a9-4527-9a16-39e91bfe8b7e`. +Both executions finalized without watcher repair, released their reservations and +deleted their launch payloads. The cancellation case verifies platform closure and +termination; the guest's immediate cancelled-poll branch is covered by the Python +tests. Normal automatic sleep stayed disabled. + +The private function, role, switch, log group and two empty capacity counters were +deleted. All 147 function log events were archived. The function remained visible +beyond the original 45-second deletion-verification window; independent reads at +`2026-09-17T12:33:47Z` confirmed all private resources absent. No deletion was repeated. +Normal task/audit records retain their existing retention policy. + +The 67-file image and acceptance archive, with a SHA-256 manifest, is +`/Users/sphias/.local/share/abca-verification/645-p3-20260916/image7-cancellation`. + +## Pagination follow-up deployment + +`p3-pending-pages-20260917` changed only the pending function package; the two other +change-set notices were unchanged API references. Template SHA-256: +`128bfd2135643fd43b0763dbc52c85e0f5c41664bdf03b4e46e2a9693b535680`. +The deployed ZIP hash is +`W++CETdMQwJ2hrSQIcyCQ+6yTAKkjzByAzVo6NBlwQg=`. +The live 101-request regression passed at `2026-09-17T12:36:47Z`, invocation request +ID `4a488ff1-37ab-49ac-80d9-a142c6dda9c4`; cleanup was verified five seconds later. +Evidence is under `/tmp/abca-645-p3-pending-pages-20260917`. + +## Final closure-feedback deployment + +The reviewed `p3-approval-closure-feedback-20260917-r2` change set +`08d5af7d-b938-4d75-9343-5240f63642c6` changed only the fan-out function's code. +CloudFormation reached `UPDATE_COMPLETE`; its update started at +`2026-09-17T13:16:58.640Z`. The current template SHA-256 is +`1be45cbc55d63fe8370a638984f226366a57e6b29317b9b064c8a67b0baf1dc7`. +The live code hash matches the reviewed ZIP: +`fYNQWLpyeKEz9Bz0qlDdxKYitLX/pYoLrnrcqPzkQAE=`. +All 475 physical identities, parent outputs, image 7.0, coordinator alias 10 and +the disabled sleep setting are preserved. + +Three deployed checks passed at `2026-09-17T13:18:04Z`: + +1. The actual legacy stranded-event shape rendered from saved task/approval + state, then exited at the intentionally absent chat destination. +2. A malformed approval event returned its numeric stream sequence as the + retry cursor, preserving the value as a string. +3. The same failed event without a sequence rejected the invocation so a batch + cannot be silently acknowledged. + +The approval row remained unchanged. Both owned fixture rows were deleted; +consistent reads confirmed their absence. No chat destinations, worker sessions +or external messages were involved. These checks verify the deployed handler +contract, not a live external delivery or the event-source mapping's retry loop. + +The initial pretty-printed template exceeded CloudFormation's size limit and +was rejected before preview creation. Compact JSON preserved the same content +within the limit. The first valid preview was deleted without execution so the +final package also contains the stream retry correction. + +The 38-file evidence archive and SHA-256 manifest are at +`/Users/sphias/.local/share/abca-verification/645-p3-20260916/closure-feedback`. + +## Remaining boundaries + +This change does not make requests indefinite. The current default is still +300 seconds, with a 3,600-second submission maximum and worker-lifetime clipping. +Keeping unanswered requests available beyond a worker's lifetime needs durable +continuation of the agent's workspace/session and the waiting tool call. Simply +removing the deadline would leave an apparently actionable request pointing at +a worker that can no longer resume. + +The separate per-user notification throttle proposed in the Cedar design is also +not implemented. Native chat responses, indefinite waiting, nested resource +migration and normal automatic-sleep activation remain on the +[P3 checklist](./645-p3-implementation-plan.md). diff --git a/docs/verification/645-p3-implementation-plan.md b/docs/verification/645-p3-implementation-plan.md index a9675a39a..3eb353e75 100644 --- a/docs/verification/645-p3-implementation-plan.md +++ b/docs/verification/645-p3-implementation-plan.md @@ -2,6 +2,13 @@ Prepared 2026-09-13 from `main` `5e10038c7e28179b302ac4de78b709795aeba3ce`. Read the [review](./645-p3-readiness-review.md) for evidence and the beginner introduction. This document tracks implementation and validation; local completion does not mark P3 live acceptance complete. +**Approval UX scope decision:** leave assessment of whether an action is still +relevant to the agent's ordinary reasoning. The user has excluded a separate +approval rechecking system, including automatic context/staleness detection, +from this plan. The platform continues to enforce authenticated ownership, +the task/request identity, approval or denial, and task cancellation. This scope +decision does not change the currently implemented approval deadlines. + ## Implementation progress Prerequisite work is tracked here on `fix/645-microvm-readiness`. “Completed” means implemented and checked locally; AWS deployment and live verification have separate completion gates below. @@ -329,7 +336,132 @@ Second P3 foundation batch implemented locally (2026-09-13): - CDK lint/compilation passed. The broad handler/session-role run passed **158 suites / 3,738 tests**, including **23 lifecycle** and **15 existing capacity** DynamoDB Local tests. Five relevant suites passed **218 overlapping tests** and exited normally. The broad run exited successfully after a delay (about 72 seconds total versus 17.5 seconds reported test execution), with no open-handle trace; its cause is not established. Documentation sync, the **77-page** build and link checks pass. No Python source changed. The temporary local database was removed. - At that foundation milestone, no production caller used this policy/store. No IAM grants, image hooks or automatic suspension were enabled. Durable poll failure/recovery tracking, guest barriers, supervisor/decision-handler wiring and live AWS gates remain unfinished. -## Remaining work in execution order +## Current completion checklist + +The implemented P3 sleep/wake backend has nine successful image 6.0 Durable +workflows and two focused image 7.0 approval/cancellation workflows. +Normal automatic sleep remains disabled. +The recent approval UX discussion adds product work that those lifecycle tests +do not cover: + +- [x] Implement the requested [nested MicroVM stack](./645-p3-nested-stack.md), + preserve parent execution-role identity and outputs, generate bootstrap 1.9.0, + and validate the bundled managed-image templates. The parent has 460 resources; + the MicroVM child has 19. AWS structural template validation passed. +- [x] Deploy bootstrap 1.9.0 and verify its exact installed policy, bundle hash, + and allowed/denied role names under the CloudFormation execution role. +- [x] Deploy an isolated fresh nested stack through that execution role, build + the managed image, verify hooks/roles/network rules and delete its resources. + The parent has three resources and the test child has 18; image 1.0 reached + `ACTIVE` / `SUCCESSFUL`. All 14 cleanup checks passed and normal resource + identities were preserved. This does not migrate the existing flat stack. +- [ ] Rehearse and review the resource migration, then + migrate the existing flat deployment and verify its image build/task path. + The existing deployment remains flat; nesting is part of the requested scope. +- [ ] Implement the agreed human waiting policy: keep unanswered + requests available, separate the human decision window from the configurable + 600-second worker sleep delay, and retain an explicit timeout option. Today, + requests still default to 300 seconds with a maximum submission setting of + 3,600 seconds. Waiting beyond a worker's lifetime needs a defined recovery + path; current same-worker suspend/resume does not provide that continuation. +- [x] Implement [actionable Slack/Linear notifications and CLI response + instructions](./645-p3-approval-ux-20260917.md), including saved decisions and + closure reasons. Unwrap the actual agent approval milestones and keep approval + messages out of Linear's terminal-reply path. Verify retry and receipt behavior. +- [x] Deploy the notification router, channel renderers and required table access. + Verify the deployed function packages against the reviewed S3 assets. +- [ ] Verify live channel delivery/response. Native Slack buttons, + Linear approval replies and the proposed per-user notification throttle remain + separate unfinished pieces; the implemented response path uses the signed-in CLI. +- [x] Atomically close pending approvals when REST/CLI or Slack cancels the task, + filter the pending list against owning task state, and stop the guest wait without + restoring `RUNNING`. Real DynamoDB transaction/race and notification-receipt checks + passed; all temporary tables were removed. +- [x] Deploy the cancellation/API changes. Deployed handler checks confirm atomic + cancellation, closure events, pending-list filtering and late-approval rejection; + eight consistent reads verify removal of all owned verification records. +- [x] Deploy the guest cancellation handling in image 7.0; preserve image 6.0, + all 475 physical resources, coordinator 10 and both disabled sleep switches. +- [x] Complete the image 7.0 approval/cancellation acceptance and cleanup. + Both Durable executions finalized without repair. Approve resumed PID 1 and + allowed one Read; cancel closed the request and terminated the sleeping worker. +- [x] Fix and deploy pending-list pagination so 100 cancelled legacy requests + cannot hide a live request on the next page. The real 101-request check passed; + all 202 temporary task/approval records were removed and verified absent. +- [x] Correct and deploy stranded/timeout/terminal-task notification feedback + and the stream retry cursor. The actual legacy stranded event, sequence-number + response and missing-sequence failure passed against the deployed function. + No channel messages were sent; both owned fixture rows were removed. +- [ ] Perform controlled activation on the normal deployment with the compatible + image/coordinator and the configurable 600-second sleep default. Verify an + ordinary submission through notification, decision, wake, continuation and + cleanup. For the current timeout-based implementation, use an explicitly + longer approval window: a default five-minute approval cannot reach a + ten-minute sleep delay. +- [ ] Verify the normal deployment's live off switch prevents new suspensions + while existing sleeping workers can still wake and finish. Retain compatible + coordinator versions, image and permissions. +- [ ] Finish the relevant comment/error-feedback cleanup, rollout documentation + and ADR/issue handoff, distinguishing shipped behavior from proposed approval + changes. The upload-dispatch feedback correction is now deployed and its + function package matches the reviewed S3 asset. + +The human waiting policy remains unimplemented; notification and cancellation +API changes are deployed, with guest handling and channel acceptance tracked above. +A separate approval relevance/staleness rechecking system is excluded by the +user's scope decision at the top of this plan. + +### Unanswered approvals: implementation order + +The product direction is agreed: unanswered requests stay available; a short +expiry is an explicit option; the worker's sleep delay defaults to 600 seconds. +The following work is still required before changing the current deadline default: + +1. Prove recovery of a paused agent on a **replacement worker**. The current + MicroVM checkpoint records task/gate identity and lifecycle acknowledgements; + it does not export the workspace or the SDK conversation. `runner.py` records + the SDK session ID in its result but does not currently start a resumed SDK + session. First test the SDK's supported recovery behavior at a pending tool + hook and define how the saved decision reaches the agent's normal reasoning. +2. Add a durable continuation checkpoint: workspace changes, conversation/session + state and exact pending request identity. Store it under task-scoped + permissions, exclude credentials, and confirm the write before releasing the + worker. A failed checkpoint must produce actionable feedback, never a claim + that the work was saved. Test restoration of uncommitted and untracked files. +3. Separate the task from each worker attempt. Add an attempt generation to + launch tokens, saved handles, writes and reservations so a replaced worker + cannot overwrite its successor. Release capacity while the task waits and + reacquire it on continuation. Duplicate decisions, lost launch replies, + cancellation and coordinator replay must not launch two workers or release + another attempt's reservation. +4. Let an open request have no decision deadline. Keep the explicit timeout + setting, retain existing finite deadlines for already-running tasks and remove + retention TTLs from unanswered requests. Start retention when a request + closes. Update validation, types, pending responses, CLI displays, notifications + and metrics together. The 600-second sleep setting remains independent. +5. Replace worker-lifetime failure with checkpoint/release for this supported + waiting state. The eight-hour service limit still applies to each worker; + the task and unanswered request survive it. Update the stranded-task + reconciler's current two-hour approval backstop so it does not fail a + deliberately parked task. A reply starts continuation if the old worker is + gone; the existing worker resumes when it is still usable. +6. Verify short sleep, a reply after the old five-minute deadline, reply after + worker replacement, explicit timeout, cancel while parked, simultaneous + replies, lost checkpoint/launch replies and permission failures. Confirm + single decision consumption, preserved files, bounded worker cost, accurate + feedback and cleanup. Then change the default and enable it through a + reviewed compatible rollout. + +These are recovery and resource-accounting requirements. They do not introduce a +separate system to judge whether the proposed action is still relevant; the agent +continues to make that judgment. + +Service token-retention/unknown-worker recovery and installation-specific legacy +capacity migration remain broader backend follow-ups. They are not automatically +prerequisites for enabling the existing sleep implementation on this clean, +compatible installation. The user has separately requested nested infrastructure. + +## Verification evidence and broader follow-ups Keep service questions and evidence in the [Lambda MicroVM service-team feedback tracker](./645-lambda-microvm-service-feedback.md). @@ -340,8 +472,8 @@ after expiry, and coordinator cleanup passed without watcher repair. The [live record](./645-p3-callback-live-20260915.md) records the evidence and verified absence of all temporary infrastructure. -The detailed batches below preserve the implementation history. For the current -handoff, use this order: +The detailed batches below preserve the implementation history and the limits of +the recorded evidence. Use the current checklist above for the next work. 1. The wider final-image lifecycle matrix on image 6.0 is now complete for its recorded scope: a temporary repository clone with mutable files across two @@ -442,13 +574,13 @@ handoff, use this order: setting their concurrency to zero could lose incoming work. A safe retained input/replay procedure remains necessary. Coordinator rollback also requires an explicit image plan because the managed runtime selects latest active. -5. Perform the final compatible rollout, including shared runtime changes for - ECS/AgentCore, pinned-version retention and rollback checks. Enable automatic - suspension only after the remaining gates pass, then finish the ADR/runbook - and issue handoff with the actual results. +5. Complete the compatible rollout and normal activation checks listed above, + retaining the required runtime versions and rollback options. Record shared + ECS/AgentCore changes for installations that use them. Keep the broader + service and legacy-migration follow-ups separately visible in the handoff. -Nested CloudFormation stacks remain optional. The current root has 475 -resources; moving existing resources is a separate migration decision. +Nested CloudFormation stacks are now requested. The last deployed flat root had +475 resources; moving its existing resources requires a reviewed migration. ## The result we want @@ -595,7 +727,7 @@ The writer inventory found no production callers of Python `write_submitted` or **Remaining trust limits:** agent status/results remain reports from the agent. Compute roles choose their session tags; existing trust does not independently bind those choices to a task. This patch protects coordinator attributes from the resulting session permissions; it does not establish complete hostile-worker tenant isolation. The replay tests in 1D/1F also do not prove AWS authorization. -## 2. Nest infrastructure if adopting the split +## 2. Nest infrastructure (requested) 1. Introduce `LambdaMicrovmStack` as a `NestedStack` wrapper and an explicit way to supply the MicroVM execution role from the parent. Keep shared session-role trust and runtime-role ownership in the parent. Preserve exact grant behavior. 2. Pass a stable deployment name for image and both connector names. Never sanitize unresolved CDK tokens. Preserve name/ARN resolution for imported images as well as managed images. @@ -606,11 +738,20 @@ The writer inventory found no production callers of Python `write_submitted` or 7. Remove or replace the #857 guard only when the supported combination actually passes the current matrix. Keep a size regression test; do not retain “505 resources” as a permanent error message. 8. Review the CloudFormation change set for replacement/deletion. For the first experimental rollout, prefer a dedicated test deployment. For migration of an existing one, stop new admissions, drain/terminate existing tasks, preserve required artifacts/state and use a service-supported migration procedure. Define rollback before applying the change set. Do not let `autoDeleteObjects` silently erase an active payload/artifact bucket. -**Done:** cycle-free bundled templates, least-privilege bootstrap coverage, preserved outputs, safe migration/change-set evidence and a clean deployed image build. The local prototype alone meets none of the live gates. +**Implementation status:** the wrapper, parent execution role, stable names, +preserved outputs, bootstrap 1.9.0 artifacts and child-template assertions are +implemented. Production managed-image synthesis and AWS structural template +validation passed. The unit budget matrix covers AgentCore, ECS and all three +MicroVM image modes with gateway on/off. The #857 guard remains. + +**Completion requires:** the remaining bundled deployment combinations, safe +migration/change-set evidence, installation of bootstrap 1.9.0, and a clean +deployed image build. The normal deployment has not been migrated. See the +[nested-stack verification record](./645-p3-nested-stack.md). ## 3. Re-prove P2 on the final infrastructure -Use a supported Region and an isolated development repository/account deployment. Record the actual deployed bootstrap bundle (at least 1.8.0 for this source, including exact-self CloudFormation PassRole and the scoped live-suspension parameter permissions, or a newer required bundle). Compare effective policies as well as the displayed version. Update bootstrap deliberately when required; a command that skips an already bootstrapped stack is not evidence of refresh. +Use a supported Region and an isolated development repository/account deployment. Record the actual deployed bootstrap bundle (at least 1.9.0 for the nested layout; 1.8.0 for the explicit flat compatibility layout). Compare effective policies as well as the displayed version, including exact-self CloudFormation PassRole, scoped live-suspension parameter permissions and the nested build/operator role names. Update bootstrap deliberately when required; a command that skips an already bootstrapped stack is not evidence of refresh. The [2026-09-13–14 clean deployment](./645-p2-clean-deployment-20260913.md) completed the infrastructure, managed-image creation and build-hook checks in diff --git a/docs/verification/645-p3-nested-stack.md b/docs/verification/645-p3-nested-stack.md new file mode 100644 index 000000000..e92c7cf9c --- /dev/null +++ b/docs/verification/645-p3-nested-stack.md @@ -0,0 +1,192 @@ +# ADR-021 nested MicroVM stack + +The user requested this split after the P3 approval UX review. The implementation, +local checks and an isolated fresh AWS deployment, image build and deletion are +complete. The normal `backgroundagent-dev` deployment still uses the flat layout; +migration of those existing resources remains open. This infrastructure split +does not nest virtual machines. + +## Resource ownership + +`AgentStack` creates a `LambdaMicrovmStack` child named `Microvm`. The child owns +the managed image when configured, two network connectors and security groups, +artifact and payload buckets, the log group, and build/operator roles. + +The execution role stays at the original parent path +`LambdaMicrovmCompute/ExecutionRole`, beside `AgentSessionRole`. This preserves +its logical ID and avoids a parent/child cycle through session-role trust. +Existing parent `Microvm*` outputs and orchestrator/API consumers refer to child +outputs; packaging and CLI discovery keep their existing output names. + +Image, connector and log names derive from the concrete parent deployment name. +Never sanitize the generated child stack name: it is an unresolved token during +synthesis. Child IAM roles have explicit names `-MicrovmBuildRole` +and `-MicrovmConnectorRole`; names exceeding IAM's 64-character limit +are rejected at synthesis. + +## Configuration and bootstrap + +For `compute_type=lambda-microvm`, nesting defaults to enabled. +`microvm_nested_stack=false` preserves the existing flat resource paths for +deployments that have not migrated. The setting accepts booleans or the strings +`true` and `false`. + +Nested deployments require bootstrap bundle **1.9.0**. It adds only the two exact +child build/operator names to the backend-specific, unconditioned `iam:PassRole` +statement. Legacy role prefixes remain for flat deployments; the runtime +execution role is not added to this statement. Generic nested-stack execution +permissions already existed in bundle 1.7.0. + +Bundle 1.9.0 is now installed in account ``, `us-west-2`. +CloudFormation completed the reviewed one-policy update on September 17. +The installed policy exactly matches source and the bundle hash is +`dc6301b65558c8fb51200e7b712a754e1b85adda7a512b14ec4f647dc954f541`. +Effective-role simulation without a service condition allows both exact names +and denies the runtime role and a similarly named extra role. Existing parameters +and all other bootstrap resources are unchanged. Evidence is archived under +`/Users/sphias/.local/share/abca-verification/645-p3-20260916/bootstrap19`. +Installing this prerequisite does not move the normal stack's resources. + +Nesting does not change the 8,192 MiB memory baseline, hook configuration, +runtime HTTPS-only egress, separate build HTTP/HTTPS egress, task-scoped +permissions or sleep gates. The MicroVM/Linear-vault combination remains gated +by #857 until its own deployment verification is complete. + +## Existing deployment migration + +Do not apply the default nested template directly to an existing flat stack. +CloudFormation sees the old resources removed and child resources added. +The existing image, connector and log names may collide, and the old bucket +auto-delete resources can erase artifacts or pending task payloads. + +Keep `microvm_nested_stack=false` in the existing deployment configuration while +preparing a concrete migration: + +1. Record the actual templates, physical IDs, image versions, grants and + retained coordinator versions; preserve the build artifacts. +2. Rehearse the chosen resource-transfer or replacement procedure in an isolated + deployment, including its rollback. Inspect each change set for deletions, + replacements, custom-resource effects and named-resource conflicts. +3. Stop new admissions through a procedure that preserves accepted inputs and + drain active work before switching resource ownership. An idle inventory alone + does not prevent new tasks from arriving. +4. Deploy bootstrap 1.9.0 and apply only the reviewed migration. Confirm outputs, + permission boundaries, managed image build and normal task lifecycle. +5. Keep both sleep gates off until the normal activation checks are complete; + remove temporary migration resources only after verification. + +Read-only `GetTemplateSummary` on the normal stack returned identifiers +`ImageArn`/`Name` for `AWS::Lambda::MicrovmImage`, `Arn`/`Name` for +`AWS::Lambda::NetworkConnector`, and `BucketName` for S3. That is useful migration +metadata, not proof that a nested import/refactor will succeed. + +[CloudFormation stack refactoring](https://docs.aws.amazon.com/AWSCloudFormation/latest/UserGuide/stack-refactoring.html) +supports moves between nested stacks. Live `DescribeType` reports both MicroVM +resource types as `FULLY_MUTABLE`, with `Name` their only create-only property. +However, refactoring cannot simultaneously add/delete resources, change their +configuration, or add/change parameters, conditions or mappings. The new explicit +IAM role names and child parameters therefore need staged migration templates; +the final application template is not itself a resource-transfer plan. A successful +refactor preview and rehearsal are still required. + +## Verification + +The dedicated construct tests exercise bootstrap, imported-image and +managed-image modes, actual parent consumers, session-role trust, stable names, +network separation, backend tags and invalid nesting inputs. Stack tests inspect +both templates and preserve a flat-layout compatibility case. Bootstrap tests +cover the exact new role names, exclusions, generated artifacts and policy hash. + +Production `mise //cdk:synth` with managed-image inputs in `us-west-2` succeeded, +including Lambda asset bundling and the production aspects. The corrected output is +under `/tmp/abca-645-nested-20260916/managed-us-west-2`; its manifest explicitly +reports `aws:///us-west-2`. This was synthesis, not deployment. + +The earlier `managed` assembly actually targeted the profile's `us-east-1` default, +despite the previous version of this record saying `us-west-2`. Setting only +`CDK_DEFAULT_REGION` did not override the CDK CLI's environment selection. The +corrected run explicitly sets `AWS_REGION` and `AWS_DEFAULT_REGION` to `us-west-2` +and checks the generated manifest. Neither synthesis run deployed a nested stack; +the separate live deployment below subsequently exercised `us-west-2`. +AWS `ValidateTemplate` also accepted the MicroVM child, reporting three parameters +and `CAPABILITY_NAMED_IAM`. That API checks template structure; it does not build +the image or validate a migration. + +| Template | Resources | File bytes | Parameters | Outputs | +| --- | ---: | ---: | ---: | ---: | +| Parent | 460 | 680,945 | 1 | 49 | +| MicroVM child | 19 | 43,895 | 3 | 8 | +| Existing registry child | 20 | 52,259 | 0 | 2 | +| Existing registry API child | 36 | 64,560 | 3 | 3 | + +Every template fits the 500-resource, 1 MiB and 200-parameter/output limits. +The hierarchy totals 535 resources, below the 2,500-resource nested-operation +limit even if every resource changed. Unit synthesis additionally covers +AgentCore, ECS and all three MicroVM image modes, each with the tool gateway +disabled and enabled. It preserves the separate #857 vault guard. + +The flat and nested templates retain the execution role logical ID +`LambdaMicrovmComputeExecutionRoleAA0C4A0D`, the same trust document and the same +parent execution-role output reference. The child stack carries the backend tag +so its stack-scoped S3 cleanup helpers inherit it as well. + +The first broad test attempt exhausted local disk while staging repeated Docker +source contexts. Generated test directories were cleared. The rerun uses +`CDK_CONTEXT_JSON='{"aws:cdk:disable-asset-staging":true}'` for structural unit +tests; the production synthesis above used actual staging and bundling. + +Final local checks: + +- CDK suite after the approval work: **232 passed, 2 skipped**; + **5,126 tests passed, 56 skipped**; + the snapshot passed. +- The final child-stack tag assertions also passed in a focused **14-test** run. +- TypeScript compilation, ESLint and whitespace checks passed. +- Documentation sync and the **77-page** site build passed. +- Bootstrap golden-baseline, artifact-sync, hash and policy coverage tests passed + as part of the CDK suite. + +Templates, test/build logs, measurements, role-identity comparison and a SHA-256 +manifest are archived at +`/Users/sphias/.local/share/abca-verification/645-p3-20260916/nested-us-west-2`. +The older `nested-stack` archive remains historical evidence of the original +structural checks, with its region correction recorded here. + +## Fresh nested deployment and cleanup + +The isolated stack `backgroundagent-dev-p3-nested-20260917` uses the production +`LambdaMicrovmStack` construct and normal CloudFormation execution role. It +imports the existing VPC and uses separate image/connector names. Its owner tag +is `abca:verification=645-p3-nested-20260917`. No normal resources were moved. + +The bootstrap phase created three parent resources and 17 child resources. +The second reviewed change set added only the managed image to the child. +Image `abca-645-nested-probe-20260917:1.0` reached `ACTIVE` / `SUCCESSFUL`; +the stack reached `UPDATE_COMPLETE`. + +Verification at `2026-09-17T12:58:18.525Z` confirmed: + +- All six lifecycle hooks, the base image, CPU, 8,192 MiB memory and environment + match normal image 7.0. +- The build uses artifact SHA-256 + `b4bc0c628f4c7976e18850ff47f92e78ecdccec683bc38974e0da82fc8600049` + from its own tagged artifact bucket. +- The build and connector roles use the exact bootstrap 1.9.0 names. + The parent execution-role identity is unchanged between phases. +- Both connectors are active in the expected subnets. Runtime egress allows + port 443; build egress allows 80 and 443; neither security group has ingress. + +No worker was launched. After archiving 1,471 image-log events, the owned stack +was deleted. Cleanup verification at `2026-09-17T13:07:47.380Z` confirms absence +of both buckets, connectors, security groups, image/version, roles, provider +function and log groups. The implicitly created provider log group was archived +and removed separately. + +The normal deployment retains all 475 physical resource identities, image 7.0, +coordinator alias 10, its disabled live sleep switch and the shared VPC. +Both phases' exact templates, 36 evidence files and a SHA-256 manifest are archived at +`/Users/sphias/.local/share/abca-verification/645-p3-20260916/nested-live`. + +This proves fresh nested deployment, image build and deletion. Existing P3 worker +lifecycle evidence remains in its dated records. Moving the existing normal stack +and testing its task path after migration are still required. diff --git a/scripts/check-types-sync.ts b/scripts/check-types-sync.ts index bf2f2555b..13318086d 100644 --- a/scripts/check-types-sync.ts +++ b/scripts/check-types-sync.ts @@ -91,6 +91,7 @@ const CDK_ONLY_ALLOWLIST = new Set([ 'PendingApprovalRecord', 'ApprovedApprovalRecord', 'DeniedApprovalRecord', + 'CancelledApprovalRecord', // Persisted cancellation row; CLI consumes summaries, not DDB records. 'TimedOutApprovalRecord', 'StrandedApprovalRecord', 'NudgeRecord', From 0467de94afb9fdf5eaf2a5d7afef9541cabeee20 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Thu, 17 Sep 2026 09:59:10 -0400 Subject: [PATCH 072/149] docs: record MicroVM refactor limitation and session recovery --- .../645-lambda-microvm-service-feedback.md | 48 +++++++- .../645-p3-implementation-plan.md | 9 ++ docs/verification/645-p3-nested-stack.md | 103 +++++++++++++++++- .../645-p3-session-recovery-20260917.md | 64 +++++++++++ 4 files changed, 220 insertions(+), 4 deletions(-) create mode 100644 docs/verification/645-p3-session-recovery-20260917.md diff --git a/docs/verification/645-lambda-microvm-service-feedback.md b/docs/verification/645-lambda-microvm-service-feedback.md index cbb7ad081..ecf04662d 100644 --- a/docs/verification/645-lambda-microvm-service-feedback.md +++ b/docs/verification/645-lambda-microvm-service-feedback.md @@ -10,14 +10,15 @@ is the receipt that lets the service team find a particular call. | ID | Priority | Topic | Evidence/status | |---|---|---|---| -| F01 | High; service diagnosis open | Accepted wake ends in connection refusal | Six recorded failures; pre-freeze connection closure verified; image 6.0 correction deployed and eight Durable workflows passed | +| F01 | High; service diagnosis open | Accepted wake ends in connection refusal | Six recorded failures; pre-freeze connection closure verified; image 6.0 correction deployed and nine Durable workflows passed | | F02 | High | Supported IAM conditions and misleading permission errors | Reproduced in earlier P2 work; current service behavior needs confirmation | | F03 | Medium | A service-side hook timeline and structured failure details | Diagnostic improvement request based on F01 | | F04 | Medium | `PENDING` also means restoring an existing worker | Observed live; our timer bug is fixed | | F05 | Medium | HTTP connection handling across suspend/resume | Exact refusal correlates with expired idle connection; explicit close header deployed; service dispatch details still needed | | F06 | Medium | Conditional operator-role requirement for VPC connectors | Earlier deployment failure; application setup fixed | | F07 | P2/P3 acceptance gap | Run token retention and recovery without a worker ID | Guest-log recovery verified; maximum retention, post-expiry behavior and recovery without identity logs unknown | -| F08 | P3 blocker | Generic wake-hook failure with an observed listener | PID 1 owned its listener after restore, 519 ms before termination; no resume hook entry | +| F08 | High; service diagnosis open | Generic wake-hook failure with an observed listener | Historical image 5.0 failure; PID 1 owned its listener after restore, with no resume hook entry; connection-close correction now deployed | +| F09 | Blocks native image refactoring | Image move passes preview but execution rejects tag schema | Exact one-image move rolled back; original image, version, tags and resource identities preserved | ## F01 — Wake request accepted, then the hook connection is refused @@ -363,6 +364,49 @@ can affect scheduling. The [full record](./645-p3-pid1-observer-20260916.md) retains the passing comparison, excluded fixture assertion, sampled state, timestamps and failure. Service response: pending. +## F09 — Image refactor preview passes but execution rejects the tag schema + +**Observed:** September 17, 13:52–13:54 UTC, account ``, +region `us-west-2`. A freshly built isolated nested deployment used the production +MicroVM construct and image artifact. The test attempted to move only the image's +CloudFormation ownership from child to parent; no worker was running. + +- Image: `abca-645-refactor-probe-20260917:1.0`. +- Parent stack: `backgroundagent-dev-p3-refactor-20260917`, ID suffix + `434bf610-b29c-11f1-8640-0624a84210c3`. +- Child stack ID suffix: `44ff4b60-b29c-11f1-8386-06d8ca482b91`. +- Refactor: `b439bca8-a95f-4716-970d-dfad7c5b30d7`. +- Create request: `12b4130e-07ec-4540-aa80-81be74899243`. +- Execute request: `6f195c46-c1a5-4f40-912c-9fdc362eeac6`, + accepted at `2026-09-17T13:52:48.582Z`. +- Preview: `CREATE_COMPLETE` / `AVAILABLE`, one `MOVE`, with + `No configuration changes detected.` Both user tags would be preserved. +- Execution: `ROLLBACK_COMPLETE`, with: + `Stack Refactor does not support AWS::Lambda::MicrovmImage because the resource type defines an unsupported tag schema.` + +**Impact:** existing images cannot use this native stack-refactor path to move +into the requested nested layout. The successful preview does not expose the +failure before execution. The normal deployment was not modified. Independent +rollback checks confirmed the image's original ARN, only version 1.0, settings, +tags and all 21 resource identities. + +**Schema evidence:** `DescribeType` reports `FULLY_MUTABLE` and updatable tags. +`Tags` is an array of objects whose `required` list contains only `Key`; `Value` +is optional. `AWS::Lambda::NetworkConnector` has the same shape. By comparison, +S3 requires both `Key` and `Value`. The exact internal rejection condition and +connector behavior have not been established. + +**Ask:** can the MicroVM image provider support CloudFormation stack refactoring, +or document a supported image-preserving retain/import procedure? Is the +optional tag `Value` causing the schema rejection, and does the connector need +the same correction? Can the preview perform this validation before making the +operation executable? Please also confirm how refactoring preserves an explicit +resource tag that overlaps a source-stack tag. + +The [nested-stack record](./645-p3-nested-stack.md#image-ownership-refactor-execution-rejected-rollback-verified) +contains the preparation controls, exact receipts and rollback evidence. +Service response: pending. This feedback has not been submitted. + ## Updating this tracker For each new finding, add the actual trigger, UTC window, region, worker/image diff --git a/docs/verification/645-p3-implementation-plan.md b/docs/verification/645-p3-implementation-plan.md index 3eb353e75..6a7b344af 100644 --- a/docs/verification/645-p3-implementation-plan.md +++ b/docs/verification/645-p3-implementation-plan.md @@ -358,6 +358,10 @@ do not cover: - [ ] Rehearse and review the resource migration, then migrate the existing flat deployment and verify its image build/task path. The existing deployment remains flat; nesting is part of the requested scope. + The isolated native image refactor passed preview but execution rejected the + resource's unsupported tag schema. Automatic rollback preserved the image and + all identities. A retain/import or explicit replacement procedure now needs + its own supported, reversible rehearsal; see the nested-stack record and F09. - [ ] Implement the agreed human waiting policy: keep unanswered requests available, separate the human decision window from the configurable 600-second worker sleep delay, and retain an explicit timeout option. Today, @@ -423,6 +427,11 @@ The following work is still required before changing the current deadline defaul the SDK session ID in its result but does not currently start a resumed SDK session. First test the SDK's supported recovery behavior at a pending tool hook and define how the saved decision reaches the agent's normal reasoning. + The [local pinned-SDK recovery probe](./645-p3-session-recovery-20260917.md) + now verifies conversation recovery from a copied session after abrupt process + loss. The pending tool call is omitted from the restored model context; a + newly proposed call receives a new ID and hook. Cloud-worker recovery and the + explicit transfer of the saved action/decision remain required. 2. Add a durable continuation checkpoint: workspace changes, conversation/session state and exact pending request identity. Store it under task-scoped permissions, exclude credentials, and confirm the write before releasing the diff --git a/docs/verification/645-p3-nested-stack.md b/docs/verification/645-p3-nested-stack.md index e92c7cf9c..0b879b834 100644 --- a/docs/verification/645-p3-nested-stack.md +++ b/docs/verification/645-p3-nested-stack.md @@ -6,6 +6,12 @@ complete. The normal `backgroundagent-dev` deployment still uses the flat layout migration of those existing resources remains open. This infrastructure split does not nest virtual machines. +An isolated image-ownership refactor was also exercised on September 17. +CloudFormation accepted the preview but rejected execution because +`AWS::Lambda::MicrovmImage` has an unsupported tag schema. Its automatic rollback +preserved the original image and every resource identity. Native image refactoring +is therefore not an available migration path with the provider tested here. + ## Resource ownership `AgentStack` creates a `LambdaMicrovmStack` child named `Microvm`. The child owns @@ -86,8 +92,24 @@ resource types as `FULLY_MUTABLE`, with `Name` their only create-only property. However, refactoring cannot simultaneously add/delete resources, change their configuration, or add/change parameters, conditions or mappings. The new explicit IAM role names and child parameters therefore need staged migration templates; -the final application template is not itself a resource-transfer plan. A successful -refactor preview and rehearsal are still required. +the final application template is not itself a resource-transfer plan. + +Live resource-provider inspection also reports `AWS::IAM::Policy` as +`NON_PROVISIONABLE`. CDK emits these separate inline-policy resources alongside +the roles. They must not be assumed eligible for a refactor just because their +roles are `FULLY_MUTABLE`. A full migration needs a separately reviewed procedure +for inline policies and S3 auto-delete custom resources, preserving permissions +and bucket contents throughout. Existing generated role names also differ from +the new explicit names; moving ownership must not silently rename those roles. + +The execution failure below means the normal migration must now choose and +rehearse another supported procedure. A retain/remove/import sequence is a +candidate only after an actual import of this resource type succeeds in isolation; +identifier discovery does not prove import support. An explicit replacement +procedure must preserve artifacts, pending payloads and compatible images while +using non-conflicting names. Neither alternative has been executed or accepted. +Do not treat the failed native refactor as a reason to apply the final nested +template directly to the existing stack. ## Verification @@ -190,3 +212,80 @@ Both phases' exact templates, 36 evidence files and a SHA-256 manifest are archi This proves fresh nested deployment, image build and deletion. Existing P3 worker lifecycle evidence remains in its dated records. Moving the existing normal stack and testing its task path after migration are still required. + +## Image ownership refactor: execution rejected, rollback verified + +The separate stack `backgroundagent-dev-p3-refactor-20260917` used the same +production construct, bootstrap 1.9.0 and artifact as the successful fresh build. +Its image `abca-645-refactor-probe-20260917:1.0` reached `ACTIVE` / `SUCCESSFUL`. +Baseline verification at `2026-09-17T13:47:15.644Z` checked all six hooks, roles, +network rules and the three parent / 18 child resources; 1,450 image-log events +were retained. No worker was launched. + +The planned transfer moved only child resource `ComputeImageC9058F98` to parent +resource `MovedMicrovmImage`. Image settings resolved to the same values through +existing child outputs. Every other resource remained in its original stack. +Moving the image back was conditional on successful execution and verification. + +Two preview findings required staging: + +- Changing the parent's nested-stack `TemplateURL` produced + `Found an action type that is not permitted during refactor operations: Modify`. + Keeping that property unchanged and providing the child's revised template as + its own `StackDefinition` produced an accepted preview. +- The preview removed source-stack tags and applied destination-stack tags. It + would have dropped `abca:compute-backend`, despite that tag also appearing on + the image resource. A separately reviewed tag-only update aligned this + MicroVM-only test parent's tags. Its two root changes had `Tags` scope, + identical before/after resource properties and no replacements. Image 1.0 + remained the only version. This staging choice must not be copied blindly to + the mixed-backend normal parent stack. + +Final refactor `b439bca8-a95f-4716-970d-dfad7c5b30d7` reached `CREATE_COMPLETE` / +`AVAILABLE`. Its only action was the expected image `MOVE`, described as +`No configuration changes detected.` Both user tags were preserved by the +proposed remove/reapply operations. Execution was accepted at +`2026-09-17T13:52:48.582Z`, request ID +`6f195c46-c1a5-4f40-912c-9fdc362eeac6`, then automatically rolled back: + +> Stack Refactor does not support AWS::Lambda::MicrovmImage because the resource type defines an unsupported tag schema. + +The refactor reached `ROLLBACK_COMPLETE`; both stacks reached +`UPDATE_ROLLBACK_COMPLETE`. Verification at `2026-09-17T13:54:29.943Z` proved: + +- The image retained its ARN, settings, creation time, `ACTIVE` / `SUCCESSFUL` + state and only version **1.0**. +- The original three parent and 18 child physical identities and all parent + outputs were preserved. +- The image retained both user tags and its original child-stack ownership tags. + +No successful ownership transfer occurred, so the planned reverse move was not +attempted. This is a reproduced provider limitation, not a passing migration +rehearsal. It is tracked as +[service feedback F09](./645-lambda-microvm-service-feedback.md#f09--image-refactor-preview-passes-but-execution-rejects-the-tag-schema). + +The live provider schema declares `FULLY_MUTABLE`, updatable tags and +`ImageArn` as its identifier. Its tag object requires `Key` but makes `Value` +optional. The connector has the same requirement; S3's tag object requires both. +This comparison supplies a service-team diagnostic question, not proof of the +internal validator's cause or of connector-refactor failure. + +Evidence includes all rejected previews, the accepted action list, execution +receipts, schema snapshots and rollback checks. Cleanup verification at +`2026-09-17T13:57:26.518Z` passed 14 independent absence checks for the owned image, +roles, buckets, connectors, security groups, provider function and log groups. +The three uploaded verification template versions were removed from their exact +toolkit-bucket prefix, with no remaining versions or delete markers. Normal CDK +assets were retained. + +The normal deployment still has all 475 original resource identities, image 7.0, +coordinator alias 10, its disabled live sleep switch and the shared VPC. The +rehearsal resources are fully deleted. Evidence and a SHA-256 manifest are +archived at +`/Users/sphias/.local/share/abca-verification/645-p3-20260916/nested-refactor`. + +One preparation guard also caught a CDK asset-key assumption: a published +template's key matched the hash of compact JSON, while its stored bytes used +formatted JSON. Their parsed contents were identical. The rehearsal therefore +used its own prefix and hashes of the actual uploaded bytes, without replacing +the existing CDK object. diff --git a/docs/verification/645-p3-session-recovery-20260917.md b/docs/verification/645-p3-session-recovery-20260917.md new file mode 100644 index 000000000..407865f99 --- /dev/null +++ b/docs/verification/645-p3-session-recovery-20260917.md @@ -0,0 +1,64 @@ +# ADR-021 pending approval session recovery + +A local diagnostic on September 17 verified that the pinned agent SDK can resume +a saved conversation in a new process after its original process dies while +waiting in a tool hook. It does **not** resume that pending hook. This is one +prerequisite for retaining unanswered approvals beyond a worker's lifetime; +production continuation remains unimplemented. + +## What was exercised + +The probe used Python `claude-agent-sdk` **0.2.110** and its bundled Claude Code +**2.1.191**, matching the reviewed application versions. The model endpoint was +a loopback HTTP server returning deterministic responses. AWS credentials were +synthetic, configuration directories were isolated, and the only available tool +was `Read` against an owned temporary marker file. No model service, repository +or notification channel was contacted. + +1. The first process received a model response proposing `Read`, with tool ID + `toolu_original`. Its `PreToolUse` callback remained blocked. +2. The probe waited until the assistant tool call appeared in the saved session + file, then copied the session directory and workspace. +3. It killed the original process group, including the bundled CLI. No + `PostToolUse` callback occurred in that process. +4. It restored the workspace at the same path and started a separate process + with the copied configuration and `ClaudeAgentOptions.resume` set to the + original session ID. +5. A new user turn told the agent that a decision had arrived. The simulated model + proposed a new `Read`, ID `toolu_restored`. That call passed through a new + permission hook, read the restored marker once, and completed in the same + conversation session. + +## What recovery preserved + +The restored model request contained the original user prompt and the new +continuation prompt. The original pending tool call was absent; its assistant +turn was represented as `No response requested.` The new tool call had a new ID. +The session ID remained `bb94901c-a91f-4854-8e63-edbc7a476f91`. + +Therefore, passing `resume=` is not sufficient to consume the saved +approval or recover its exact proposed action. The application must retain that +request identity, action and decision separately and make them available to the +agent after restoration. Newly proposed tool calls still pass through the normal +authorization hooks. The recorded approval must not become blanket permission +for a different action. + +This requirement does not add a separate relevance or staleness checker. The +agent assesses relevance through its ordinary reasoning, as requested. + +## Limits and next steps + +This test proves local SDK conversation recovery with an abrupt process loss and +a copied workspace/configuration directory. Its deterministic model demonstrates +the transport behavior; it does not prove how a real model will interpret the +continuation prompt. + +The probe does not yet verify recovery on a replacement cloud worker, portable +workspace paths, durable object storage, exclusion of credentials from a +production checkpoint, task-attempt fencing, exactly-once decision consumption, +denial, cancellation or loss of a checkpoint/launch response. Those remain part +of the [unanswered-approval implementation order](./645-p3-implementation-plan.md#unanswered-approvals-implementation-order). + +The executable probe, exact model requests, hook audits, copied synthetic session +files and verification result are archived with a SHA-256 manifest under +`/Users/sphias/.local/share/abca-verification/645-p3-20260916/session-recovery`. From 2d6edb41f945459bb104c44afbd0f325f646f105 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Thu, 17 Sep 2026 11:24:14 -0400 Subject: [PATCH 073/149] feat: add verified conversation checkpoints for MicroVM continuation (#645) --- agent/README.md | 48 +- agent/src/continuation_session.py | 437 ++++++++++++++++++ agent/tests/test_continuation_sdk_probe.py | 336 ++++++++++++++ agent/tests/test_continuation_session.py | 379 +++++++++++++++ .../645-p3-implementation-plan.md | 20 +- .../645-p3-session-recovery-20260917.md | 146 +++++- 6 files changed, 1343 insertions(+), 23 deletions(-) create mode 100644 agent/src/continuation_session.py create mode 100644 agent/tests/test_continuation_sdk_probe.py create mode 100644 agent/tests/test_continuation_session.py diff --git a/agent/README.md b/agent/README.md index bc214705e..f87750104 100644 --- a/agent/README.md +++ b/agent/README.md @@ -228,7 +228,7 @@ Immediate response (acceptance): Final metrics (PR URL, cost, turns, build status, etc.) appear in **container logs**, in **DynamoDB** when configured, and in the **REST API** for deployed tasks (`GET /v1/tasks/{task_id}` via the `bgagent` CLI or HTTP client). -### AWS Lambda MicroVMs lifecycle hooks (ADR-021 P1 + P2) +### AWS Lambda MicroVMs lifecycle hooks (ADR-021 P1–P3) The same uvicorn process also serves the **Lambda MicroVMs** lifecycle hooks, on the same port (8080 — the port declared in the image's `hooks.port`). On that backend there is no `InvokeAgentRuntime` and no orchestrator→agent HTTP path at all: the task payload arrives as the `/run` hook body and nothing else dials in. @@ -253,7 +253,7 @@ Baked secrets are **reported, not enforced**: `warnings` lists the names (never `microvmId` is parsed defensively and **arrives empty in practice**: the service sends `""` here, unlike `/run` where it is populated (live-verified, ADR-021 P2-F8). So an empty id is expected-normal, not a degraded read — and this hook therefore **cannot** join the guest's record to the control-plane one. `/run`'s `hook accepted task_id=… microvm_id=…` line carries that correlation; `/terminate`'s value is the pipeline-state snapshot it reports. -**`POST /aws/lambda-microvms/runtime/v1/run`** — Authenticate and download a task, install `platform_config` (below), start the pipeline in a background thread, and return 200 inside the hook budget. Protocol v2 is implemented locally; real AWS permission/network/expiry verification remains pending (repository runbook: `docs/verification/645-payload-bootstrap.md`). +**`POST /aws/lambda-microvms/runtime/v1/run`** — Authenticate and download a task, install `platform_config` (below), start the pipeline in a background thread, and return 200 inside the hook budget. Protocol v2 is deployed for MicroVM; [11 live transport/failure cases](../docs/verification/645-p2-payload-live-20260914.md) verified worker downloads/rejections, URL expiry/revocation and immediate launch replay. The [runbook](../docs/verification/645-payload-bootstrap.md) tracks the broader authorization and recovery matrix separately. `runHookPayload` is a JSON **string** passed through by `RunMicrovm`, containing: @@ -305,6 +305,50 @@ Rejections are structured so they are readable in the MicroVM log group: `400 MI `/suspend` and `/resume` are served by `microvm_http.py` and declared in managed images with 30-second service timeouts and a shared non-secret protocol marker. Pause requires an active original approval gate, matching coordinator intent, drained activity and an acknowledged checkpoint; wake renews credentials and atomically rechecks the original task/gate before allowing coding. Each handler has a 20-second total budget. `/validate` rejects a supplied incompatible image marker without contacting AWS. The coordinator checks the actual launched image version and persists support on that worker; missing support disables new suspension. See [image capability verification](../docs/verification/645-p3-image-capability.md). Supervisor integration and live P3 acceptance remain open. +#### Conversation continuation checkpoints (P3 prerequisite) + +`src/continuation_session.py` provides an SDK conversation store and versioned S3 +checkpoint adapter. It is **not yet connected to the production runner or worker +release** and does not change approval deadlines. The existing MicroVM lifecycle +checkpoint resumes the same frozen process; this new component supports future +continuation after that process is gone. + +`CheckpointSessionStore` implements the pinned SDK's public `SessionStore` +contract. A caller supplies it with `session_store_flush="eager"` and waits for +`checkpoint_pending()` to acknowledge the exact assistant tool call. Enabling +mirroring alone is insufficient because the SDK copies the transcript +asynchronously. A missing or rejected transcript batch prevents acknowledgement. +The envelope retains the session, full proposed action and task/attempt/request +identity, with a 16 MiB / 50,000-entry limit. + +`S3ContinuationCheckpoints` saves under +`continuations////.json`, verifies a +read-back, and returns a receipt pinned to the S3 object version and checksum. +Its default client requires task-scoped credentials. A versioned bucket and +task-prefix `s3:PutObject`, `s3:GetObject` and `s3:GetObjectVersion` permissions +are required; the current production artifact grants do not supply these reads. +The adapter neither lists nor deletes objects. Retention must be arranged by the +future integration after the request closes. + +Before releasing a worker, that integration must also preserve the workspace, +hold the lifecycle barrier and conditionally publish the verified checkpoint for +the same task attempt. A failed save must leave the worker available and report +the failure. Checkpoints are private task data: transcripts and proposed actions +can contain sensitive content. The module does not read CLI authentication files +or the process environment, but does not redact conversation contents. + +The opt-in test uses the actual pinned SDK/CLI with a deterministic loopback +model. It kills the original process, deletes its configuration, restores from +the new store, and verifies approve and deny through a fresh tool hook: + +```bash +cd agent +ABCA_TEST_SDK_CONTINUATION=1 uv run pytest tests/test_continuation_sdk_probe.py --no-cov +``` + +See the [session recovery and storage evidence](../docs/verification/645-p3-session-recovery-20260917.md) +for the live S3 permission checks and remaining replacement-worker requirements. + ### Testing Server Mode Locally Use `run.sh --server` to build and start the server locally. It handles credentials, port mapping, and resource constraints automatically: diff --git a/agent/src/continuation_session.py b/agent/src/continuation_session.py new file mode 100644 index 000000000..9fc878434 --- /dev/null +++ b/agent/src/continuation_session.py @@ -0,0 +1,437 @@ +# Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. +# SPDX-License-Identifier: MIT-0 + +"""Acknowledged SDK conversation checkpoints for future worker continuation. + +This module does not release workers or change approval deadlines. A caller must +hold the lifecycle barrier, checkpoint the workspace, and conditionally publish +the returned receipt for the same task/attempt/request before releasing compute. +Only SDK transcript entries are captured; CLI authentication files and process +environment are never read. Transcript/action contents remain private task data. +""" + +from __future__ import annotations + +import asyncio +import base64 +import hashlib +import json +import math +import re +from dataclasses import asdict, dataclass +from typing import TYPE_CHECKING, Any +from uuid import UUID + +from claude_agent_sdk import SessionStore + +if TYPE_CHECKING: + from claude_agent_sdk import SessionKey, SessionStoreEntry + +CHECKPOINT_VERSION = 1 +MAX_CHECKPOINT_BYTES = 16 * 1024 * 1024 +MAX_CHECKPOINT_ENTRIES = 50_000 +MAX_ID_LENGTH = 128 +_ID = re.compile(r"[A-Za-z0-9][A-Za-z0-9_-]{0,127}\Z") +_HASH = re.compile(r"[0-9a-f]{64}\Z") + + +class ContinuationCheckpointError(RuntimeError): + """A checkpoint cannot be acknowledged or safely restored.""" + + +def _encode(value: Any) -> bytes: + try: + return json.dumps(value, sort_keys=True, separators=(",", ":"), allow_nan=False).encode() + except (TypeError, ValueError, RecursionError) as exc: + raise ContinuationCheckpointError("Checkpoint contains invalid JSON data") from exc + + +def _copy(value: Any) -> Any: + return json.loads(_encode(value)) + + +def _identifier(value: Any) -> bool: + return isinstance(value, str) and _ID.fullmatch(value) is not None + + +def _session_id(value: Any) -> bool: + if not isinstance(value, str): + return False + try: + return str(UUID(value)) == value + except ValueError: + return False + + +def _action_hash(tool_input: dict) -> str: + # Match the approval row / RecentDecisionCache encoding, including spaces + # and ASCII escaping. Invalid input is rejected rather than stringified. + _encode(tool_input) + return hashlib.sha256(json.dumps(tool_input, sort_keys=True).encode()).hexdigest() + + +@dataclass(frozen=True) +class CheckpointIdentity: + task_id: str + attempt_id: str + request_id: str + user_id: str + repo: str + + def __post_init__(self) -> None: + if ( + not all( + _identifier(value) for value in (self.task_id, self.attempt_id, self.request_id) + ) + or not isinstance(self.user_id, str) + or not self.user_id + or len(self.user_id) > MAX_ID_LENGTH + or not isinstance(self.repo, str) + ): + raise ContinuationCheckpointError("Checkpoint identity is invalid") + + @property + def prefix(self) -> str: + return f"continuations/{self.task_id}/{self.attempt_id}/{self.request_id}/" + + +@dataclass(frozen=True) +class CheckpointReceipt: + """Pin a verified object version, never an overwriteable current key.""" + + key: str + version_id: str + sha256: str + size_bytes: int + + def validate(self, identity: CheckpointIdentity) -> None: + if ( + not isinstance(self.sha256, str) + or not _HASH.fullmatch(self.sha256) + or self.key != identity.prefix + self.sha256 + ".json" + or not isinstance(self.version_id, str) + or not self.version_id + or self.version_id == "null" + or type(self.size_bytes) is not int + or not 0 < self.size_bytes <= MAX_CHECKPOINT_BYTES + ): + raise ContinuationCheckpointError("Checkpoint receipt is invalid or outside this task") + + +def _validate_entries(entries: Any) -> list[dict]: + if not isinstance(entries, list) or len(entries) > MAX_CHECKPOINT_ENTRIES: + raise ContinuationCheckpointError("Checkpoint transcript entry limit exceeded") + for entry in entries: + if not isinstance(entry, dict) or not isinstance(entry.get("type"), str): + raise ContinuationCheckpointError("Checkpoint transcript entry is invalid") + if "uuid" in entry and (not isinstance(entry["uuid"], str) or not entry["uuid"]): + raise ContinuationCheckpointError("Checkpoint transcript entry UUID is invalid") + return entries + + +def _pending_action_present(entries: list[dict], action: dict) -> bool: + found = False + for entry in entries: + message = entry.get("message") + content = message.get("content") if isinstance(message, dict) else None + if not isinstance(content, list): + continue + for block in content: + if not isinstance(block, dict): + continue + if ( + block.get("type") == "tool_result" + and block.get("tool_use_id") == action["tool_use_id"] + ): + raise ContinuationCheckpointError("Pending action already has a tool result") + if block.get("type") == "tool_use" and block.get("id") == action["tool_use_id"]: + if found: + raise ContinuationCheckpointError("Pending action has duplicate transcript IDs") + if ( + entry.get("type") != "assistant" + or block.get("name") != action["tool_name"] + or _encode(block.get("input")) != _encode(action["tool_input"]) + ): + raise ContinuationCheckpointError( + "Pending action disagrees with the transcript" + ) + found = True + return found + + +def decode_checkpoint(body: bytes, identity: CheckpointIdentity) -> dict: + """Validate the complete private envelope before using any restored data.""" + if not body or len(body) > MAX_CHECKPOINT_BYTES: + raise ContinuationCheckpointError("Checkpoint byte limit exceeded") + try: + envelope = json.loads(body) + except (ValueError, UnicodeError, RecursionError) as exc: + raise ContinuationCheckpointError("Checkpoint is not valid JSON") from exc + if ( + not isinstance(envelope, dict) + or set(envelope) + != {"version", "identity", "project_key", "session_id", "action", "entries"} + or type(envelope.get("version")) is not int + or envelope["version"] != CHECKPOINT_VERSION + or envelope.get("identity") != asdict(identity) + or not _session_id(envelope.get("session_id")) + or not isinstance(envelope.get("project_key"), str) + or not envelope["project_key"] + or not isinstance(envelope.get("action"), dict) + ): + raise ContinuationCheckpointError("Checkpoint envelope or identity is invalid") + action = envelope["action"] + if ( + set(action) != {"tool_use_id", "tool_name", "tool_input", "tool_input_sha256"} + or not _identifier(action.get("tool_use_id")) + or not _identifier(action.get("tool_name")) + or not isinstance(action.get("tool_input"), dict) + or action.get("tool_input_sha256") != _action_hash(action["tool_input"]) + ): + raise ContinuationCheckpointError("Checkpoint pending action is invalid") + entries = _validate_entries(envelope.get("entries")) + if any( + "sessionId" in entry and entry["sessionId"] != envelope["session_id"] for entry in entries + ): + raise ContinuationCheckpointError("Transcript contains another SDK session") + if not _pending_action_present(entries, action): + raise ContinuationCheckpointError("Pending action is missing from the transcript") + # Reject non-finite numbers even though Python's JSON decoder accepts them. + _encode(envelope) + return envelope + + +class CheckpointSessionStore(SessionStore): + """Single SDK session buffer implementing its public append/load protocol. + + Eager mirroring is still asynchronous. ``checkpoint_pending`` waits for the + exact assistant action to reach this buffer; enabling mirroring alone is not + an acknowledgement. Buffers belong to the SDK runner's event loop. + """ + + def __init__(self, project_key: str) -> None: + if not isinstance(project_key, str) or not project_key: + raise ContinuationCheckpointError("SDK project key is unavailable") + self.project_key = project_key + self._session: str | None = None + self._entries: list[dict] = [] + self._changed = asyncio.Condition() + self._failure: str | None = None + + def _check_key(self, key: SessionKey) -> str: + if ( + not isinstance(key, dict) + or key.get("project_key") != self.project_key + or not _session_id(key.get("session_id")) + or "subpath" in key + or (self._session is not None and key["session_id"] != self._session) + ): + raise ContinuationCheckpointError("Unexpected SDK session or subagent transcript") + return key["session_id"] + + async def append(self, key: SessionKey, entries: list[SessionStoreEntry]) -> None: + async with self._changed: + try: + session = self._check_key(key) + incoming = _validate_entries(_copy(entries)) + if any( + "sessionId" in entry and entry["sessionId"] != session for entry in incoming + ): + raise ContinuationCheckpointError("Transcript contains another SDK session") + updated = _copy(self._entries) + positions = {entry["uuid"]: i for i, entry in enumerate(updated) if "uuid" in entry} + for entry in incoming: + entry_id = entry.get("uuid") + if entry_id in positions: + updated[positions[entry_id]] = entry + else: + if entry_id is not None: + positions[entry_id] = len(updated) + updated.append(entry) + if ( + len(updated) > MAX_CHECKPOINT_ENTRIES + or len(_encode(updated)) > MAX_CHECKPOINT_BYTES + ): + raise ContinuationCheckpointError("SDK transcript exceeds checkpoint limits") + self._session, self._entries = session, updated + except ContinuationCheckpointError as exc: + # The SDK can continue after a dropped mirror batch. Once a gap + # is possible, no later matching action may certify completeness. + self._failure = str(exc) + raise + finally: + self._changed.notify_all() + + async def load(self, key: SessionKey) -> list[SessionStoreEntry] | None: + async with self._changed: + self._check_key(key) + if self._failure: + raise ContinuationCheckpointError(self._failure) + return _copy(self._entries) if self._session else None + + async def checkpoint_pending( + self, + identity: CheckpointIdentity, + *, + session_id: str, + tool_use_id: str, + tool_name: str, + tool_input: dict, + timeout_s: float = 5, + ) -> bytes: + """Require mirrored action coverage; workspace/barrier checks are separate.""" + if ( + not _session_id(session_id) + or not _identifier(tool_use_id) + or not _identifier(tool_name) + or not isinstance(tool_input, dict) + or isinstance(timeout_s, bool) + or not isinstance(timeout_s, (float, int)) + or not math.isfinite(timeout_s) + or timeout_s <= 0 + ): + raise ContinuationCheckpointError("Pending SDK action identity is invalid") + action = { + "tool_use_id": tool_use_id, + "tool_name": tool_name, + "tool_input": _copy(tool_input), + "tool_input_sha256": _action_hash(tool_input), + } + try: + async with asyncio.timeout(timeout_s), self._changed: + while True: + if self._failure: + raise ContinuationCheckpointError(self._failure) + if self._session and self._session != session_id: + raise ContinuationCheckpointError("Pending SDK session identity changed") + if _pending_action_present(self._entries, action): + body = _encode( + { + "version": CHECKPOINT_VERSION, + "identity": asdict(identity), + "project_key": self.project_key, + "session_id": session_id, + "action": action, + "entries": self._entries, + } + ) + decode_checkpoint(body, identity) + return body + await self._changed.wait() + except TimeoutError as exc: + raise ContinuationCheckpointError( + "SDK mirror did not acknowledge the pending action" + ) from exc + + @classmethod + def restore(cls, body: bytes, identity: CheckpointIdentity) -> CheckpointSessionStore: + envelope = decode_checkpoint(body, identity) + store = cls(envelope["project_key"]) + store._session = envelope["session_id"] + store._entries = envelope["entries"] + return store + + +class S3ContinuationCheckpoints: + """Store immutable, encrypted checkpoints under a task/attempt/request key. + + The bucket must have versioning enabled. The task role needs PutObject, + GetObject and GetObjectVersion for its own continuations// prefix. + This adapter never lists buckets, deletes data or falls back to ambient AWS + credentials. The caller must arrange retention after the owning request + closes; this adapter does not configure object expiration. + """ + + def __init__(self, bucket: str, *, client: Any = None) -> None: + if not isinstance(bucket, str) or not bucket or "/" in bucket: + raise ContinuationCheckpointError("Checkpoint bucket is unavailable") + if client is None: + from botocore.config import Config + + from aws_session import is_scoped, tenant_client + + if not is_scoped(): + raise ContinuationCheckpointError( + "Checkpoint storage requires task-scoped credentials" + ) + client = tenant_client( + "s3", + config=Config(connect_timeout=2, read_timeout=5, retries={"total_max_attempts": 1}), + ) + self.bucket, self.client = bucket, client + + def _read(self, key: str, *, version_id: str | None = None) -> tuple[bytes, str]: + kwargs = {"Bucket": self.bucket, "Key": key, "ChecksumMode": "ENABLED"} + if version_id is not None: + kwargs["VersionId"] = version_id + response = self.client.get_object(**kwargs) + stream = response["Body"] + try: + size = response.get("ContentLength") + if type(size) is not int or not 0 < size <= MAX_CHECKPOINT_BYTES: + raise ContinuationCheckpointError("Stored checkpoint byte limit exceeded") + body = stream.read(MAX_CHECKPOINT_BYTES + 1) + if len(body) != size: + raise ContinuationCheckpointError("Stored checkpoint is incomplete") + finally: + stream.close() + actual_version = response.get("VersionId") + if ( + not isinstance(actual_version, str) + or not actual_version + or actual_version == "null" + or (version_id is not None and actual_version != version_id) + ): + raise ContinuationCheckpointError("Checkpoint storage requires an exact object version") + checksum = base64.b64encode(hashlib.sha256(body).digest()).decode() + if response.get("ChecksumSHA256") != checksum: + raise ContinuationCheckpointError( + "Stored checkpoint checksum is unavailable or incorrect" + ) + return body, actual_version + + def save(self, body: bytes, identity: CheckpointIdentity) -> CheckpointReceipt: + """Read back and verify even after an ambiguous write response.""" + decode_checkpoint(body, identity) + digest = hashlib.sha256(body).hexdigest() + key = identity.prefix + digest + ".json" + write_error: Exception | None = None + try: + self.client.put_object( + Bucket=self.bucket, + Key=key, + Body=body, + ContentType="application/json", + ServerSideEncryption="AES256", + ChecksumSHA256=base64.b64encode(hashlib.sha256(body).digest()).decode(), + IfNoneMatch="*", + ) + except Exception as exc: + # A timeout can follow a committed write; a repeated save can also + # return 412. Only exact read-back permits an acknowledgement. + write_error = exc + try: + stored, version = self._read(key) + if stored != body: + raise ContinuationCheckpointError( + "Stored checkpoint differs from the prepared data" + ) + except Exception as exc: + cause = write_error if write_error is not None else exc + raise ContinuationCheckpointError( + "Checkpoint could not be verified; keep the current worker available" + ) from cause + receipt = CheckpointReceipt(key, version, digest, len(body)) + receipt.validate(identity) + return receipt + + def load(self, receipt: CheckpointReceipt, identity: CheckpointIdentity) -> bytes: + receipt.validate(identity) + try: + body, _ = self._read(receipt.key, version_id=receipt.version_id) + except Exception as exc: + raise ContinuationCheckpointError("Saved checkpoint could not be read") from exc + if len(body) != receipt.size_bytes or hashlib.sha256(body).hexdigest() != receipt.sha256: + raise ContinuationCheckpointError("Saved checkpoint does not match its receipt") + decode_checkpoint(body, identity) + return body diff --git a/agent/tests/test_continuation_sdk_probe.py b/agent/tests/test_continuation_sdk_probe.py new file mode 100644 index 000000000..4723e3adb --- /dev/null +++ b/agent/tests/test_continuation_sdk_probe.py @@ -0,0 +1,336 @@ +# Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. +# SPDX-License-Identifier: MIT-0 + +"""Opt-in real SDK/CLI continuation test with a synthetic loopback model. + +Run ABCA_TEST_SDK_CONTINUATION=1 uv run pytest tests/test_continuation_sdk_probe.py +--no-cov. No model service or real AWS credentials are used. +""" + +from __future__ import annotations + +import argparse +import asyncio +import importlib.metadata +import json +import os +import shutil +import signal +import subprocess +import sys +import threading +import time +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer +from pathlib import Path + +import pytest + +AGENT_ROOT = Path(__file__).resolve().parents[1] +sys.path.insert(0, str(AGENT_ROOT)) +sys.path.insert(0, str(AGENT_ROOT / "src")) + +from continuation_session import CheckpointIdentity, CheckpointSessionStore, decode_checkpoint +from scripts.verify_microvm_credentials import MODEL, event_frame, model_response + +IDENTITY = CheckpointIdentity("sdk-probe", "attempt-1", "request-1", "owner", "probe/owned") + + +def _tool_response(target: Path, tool_id: str) -> bytes: + events = [ + { + "type": "message_start", + "message": { + "id": "msg_" + tool_id, + "type": "message", + "role": "assistant", + "model": MODEL, + "content": [], + "stop_reason": None, + "stop_sequence": None, + "usage": {"input_tokens": 1, "output_tokens": 0}, + }, + }, + { + "type": "content_block_start", + "index": 0, + "content_block": {"type": "tool_use", "id": tool_id, "name": "Read", "input": {}}, + }, + { + "type": "content_block_delta", + "index": 0, + "delta": { + "type": "input_json_delta", + "partial_json": json.dumps({"file_path": str(target)}), + }, + }, + {"type": "content_block_stop", "index": 0}, + { + "type": "message_delta", + "delta": {"stop_reason": "tool_use", "stop_sequence": None}, + "usage": {"output_tokens": 1}, + }, + {"type": "message_stop"}, + ] + return b"".join(event_frame(event) for event in events) + + +async def _worker(directory: Path, endpoint: str, phase: str, decision: str) -> None: + from claude_agent_sdk import ( + ClaudeAgentOptions, + ClaudeSDKClient, + ResultMessage, + project_key_for_directory, + ) + from claude_agent_sdk.types import HookMatcher + + workspace = directory / "workspace" + target = workspace / "owned.txt" + original = phase == "original" + store = ( + CheckpointSessionStore(project_key_for_directory(str(workspace))) + if original + else CheckpointSessionStore.restore((directory / "checkpoint.json").read_bytes(), IDENTITY) + ) + saved = ( + None + if original + else decode_checkpoint((directory / "checkpoint.json").read_bytes(), IDENTITY) + ) + audit: list[dict] = [] + + def record(kind: str, **data) -> None: + audit.append({"kind": kind, **data}) + (directory / f"{phase}-audit.json").write_text(json.dumps(audit)) + + async def pre(data, tool_id, context): + assert data["tool_name"] == "Read" + assert data["tool_input"] == {"file_path": str(target)} + record("pre", session_id=data["session_id"], tool_id=tool_id) + if original: + body = await store.checkpoint_pending( + IDENTITY, + session_id=data["session_id"], + tool_use_id=tool_id, + tool_name=data["tool_name"], + tool_input=data["tool_input"], + ) + pending = directory / "checkpoint.tmp" + pending.write_bytes(body) + pending.chmod(0o600) + pending.replace(directory / "checkpoint.json") + await asyncio.sleep(100) + raise RuntimeError("Original process was not stopped") + return { + "hookSpecificOutput": { + "hookEventName": "PreToolUse", + "permissionDecision": "allow" if decision == "approve" else "deny", + "permissionDecisionReason": "Owned continuation diagnostic decision", + } + } + + async def post(data, tool_id, context): + record("post", tool_id=tool_id) + return {} + + env = { + "CLAUDE_CONFIG_DIR": str(directory / f"{phase}-config"), + "CLAUDE_CODE_USE_BEDROCK": "1", + "CLAUDE_CODE_DISABLE_NONESSENTIAL_TRAFFIC": "1", + "CLAUDE_CODE_MAX_RETRIES": "0", + "DISABLE_TELEMETRY": "1", + "DISABLE_ERROR_REPORTING": "1", + "DISABLE_AUTOUPDATER": "1", + "ANTHROPIC_BEDROCK_BASE_URL": endpoint, + "AWS_ENDPOINT_URL": endpoint, + "AWS_REGION": "us-west-2", + "AWS_DEFAULT_REGION": "us-west-2", + "AWS_CONFIG_FILE": str(directory / "aws-config"), + "AWS_SHARED_CREDENTIALS_FILE": str(directory / "aws-credentials"), + "AWS_ACCESS_KEY_ID": "SYNTHETIC_NOT_VALID", + "AWS_SECRET_ACCESS_KEY": "synthetic-not-valid-in-aws", + "AWS_SESSION_TOKEN": "synthetic-not-valid-in-aws", + "AWS_EC2_METADATA_DISABLED": "true", + } + options = ClaudeAgentOptions( + model=MODEL, + max_turns=3, + cwd=str(workspace), + tools=["Read"], + permission_mode="bypassPermissions", + setting_sources=[], + settings=str(directory / "settings.json"), + env=env, + resume=saved["session_id"] if saved else None, + session_store=store, + session_store_flush="eager", + stderr=lambda line: record("stderr", text=line), + hooks={ + "PreToolUse": [HookMatcher(hooks=[pre], timeout=100)], + "PostToolUse": [HookMatcher(hooks=[post])], + }, + ) + async with ClaudeSDKClient(options=options) as client: + prompt = ( + "Continue the saved task. The human decision was " + + decision + + ". The saved pending action was " + + json.dumps(saved["action"]) + if saved + else "Read the owned marker once, then stop." + ) + await client.query(prompt) + async for message in client.receive_response(): + if isinstance(message, ResultMessage): + record("result", session_id=message.session_id, is_error=message.is_error) + + +@pytest.mark.skipif( + os.environ.get("ABCA_TEST_SDK_CONTINUATION") != "1", + reason="Opt-in pinned SDK/CLI subprocess diagnostic", +) +@pytest.mark.parametrize("decision", ["approve", "deny"]) +def test_sdk_resumes_from_checkpoint_without_original_config(tmp_path, decision): + import claude_agent_sdk + + assert importlib.metadata.version("claude-agent-sdk") == "0.2.110" + cli = Path(claude_agent_sdk.__file__).parent / "_bundled/claude" + version = subprocess.check_output([str(cli), "--version"], text=True, timeout=10).strip() + assert version == "2.1.191 (Claude Code)" + workspace = tmp_path / "workspace" + workspace.mkdir() + target = workspace / "owned.txt" + target.write_text("OWNED_CONTINUATION_MARKER\n") + for phase in ("original", "restored"): + (tmp_path / f"{phase}-config").mkdir() + auth_sentinel = "SYNTHETIC_AUTH_FILE_MUST_NOT_BE_COPIED" + (tmp_path / "original-config/.credentials.json").write_text( + json.dumps({"sentinel": auth_sentinel}) + ) + (tmp_path / "settings.json").write_text("{}") + (tmp_path / "aws-config").write_text("[default]\nregion = us-west-2\n") + (tmp_path / "aws-credentials").write_text("") + phase = ["original"] + requests = [] + + class Handler(BaseHTTPRequestHandler): + def log_message(self, format, *args): + del format, args + + def do_POST(self): + body = json.loads(self.rfile.read(int(self.headers["Content-Length"]))) + requests.append({"phase": phase[0], "body": body}) + results = [ + item + for message in body.get("messages", []) + for item in ( + message.get("content", []) if isinstance(message.get("content"), list) else [] + ) + if isinstance(item, dict) and item.get("type") == "tool_result" + ] + response = model_response() if results else _tool_response(target, "toolu_" + phase[0]) + self.send_response(200) + self.send_header("Content-Type", "application/vnd.amazon.eventstream") + self.send_header("Content-Length", str(len(response))) + self.end_headers() + self.wfile.write(response) + + server = ThreadingHTTPServer(("127.0.0.1", 0), Handler) + thread = threading.Thread(target=server.serve_forever, daemon=True) + thread.start() + endpoint = f"http://127.0.0.1:{server.server_port}" + children = [] + logs = [] + + def start(name): + log = (tmp_path / f"{name}-process.log").open("w") + logs.append(log) + child = subprocess.Popen( + [ + sys.executable, + str(__file__), + "--worker", + name, + "--directory", + str(tmp_path), + "--endpoint", + endpoint, + "--decision", + decision, + ], + stdout=log, + stderr=log, + start_new_session=True, + ) + children.append(child) + return child + + try: + first = start("original") + limit = time.monotonic() + 30 + while not (tmp_path / "checkpoint.json").exists(): + assert first.poll() is None, (tmp_path / "original-process.log").read_text() + assert time.monotonic() < limit, (tmp_path / "original-audit.json").read_text() + time.sleep(0.05) + body = (tmp_path / "checkpoint.json").read_bytes() + saved = decode_checkpoint(body, IDENTITY) + assert auth_sentinel.encode() not in body + os.killpg(first.pid, signal.SIGKILL) + assert first.wait(timeout=10) == -signal.SIGKILL + original_audit = json.loads((tmp_path / "original-audit.json").read_text()) + assert not any(event["kind"] == "post" for event in original_audit) + shutil.rmtree(tmp_path / "original-config") + shutil.copytree(workspace, tmp_path / "workspace-copy") + shutil.rmtree(workspace) + shutil.copytree(tmp_path / "workspace-copy", workspace) + phase[0] = "restored" + second = start("restored") + assert second.wait(timeout=35) == 0, (tmp_path / "restored-process.log").read_text() + audit = json.loads((tmp_path / "restored-audit.json").read_text()) + assert [event["tool_id"] for event in audit if event["kind"] == "pre"] == ["toolu_restored"] + posts = [event["tool_id"] for event in audit if event["kind"] == "post"] + assert posts == (["toolu_restored"] if decision == "approve" else []) + result = next(event for event in audit if event["kind"] == "result") + assert not result["is_error"] and result["session_id"] == saved["session_id"] + assert target.read_text() == "OWNED_CONTINUATION_MARKER\n" + restored_requests = [request for request in requests if request["phase"] == "restored"] + assert any( + saved["action"]["tool_use_id"] in json.dumps(r["body"]) for r in restored_requests + ) + proof = { + "sdk": "0.2.110", + "cli": version, + "decision": decision, + "session_id": saved["session_id"], + "original_config_deleted": True, + "pending_action_acknowledged": True, + "auth_file_excluded": True, + "restored_posts": posts, + "only_synthetic_loopback": True, + } + (tmp_path / "verification.json").write_text(json.dumps(proof, indent=2)) + finally: + for child in children: + if child.poll() is None: + os.killpg(child.pid, signal.SIGKILL) + child.wait(timeout=10) + server.shutdown() + server.server_close() + thread.join(timeout=2) + for log in logs: + log.close() + (tmp_path / "model-requests.json").write_text(json.dumps(requests, indent=2)) + + +if __name__ == "__main__": + parser = argparse.ArgumentParser() + parser.add_argument("--worker", required=True) + parser.add_argument("--directory", type=Path, required=True) + parser.add_argument("--endpoint", required=True) + parser.add_argument("--decision", choices=["approve", "deny"], required=True) + args = parser.parse_args() + for key in tuple(os.environ): + if key.startswith(("AWS_", "ANTHROPIC_", "CLAUDE_", "OTEL_", "BEDROCK_")): + del os.environ[key] + asyncio.run( + asyncio.wait_for(_worker(args.directory, args.endpoint, args.worker, args.decision), 110) + ) diff --git a/agent/tests/test_continuation_session.py b/agent/tests/test_continuation_session.py new file mode 100644 index 000000000..09f758dc0 --- /dev/null +++ b/agent/tests/test_continuation_session.py @@ -0,0 +1,379 @@ +# Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. +# SPDX-License-Identifier: MIT-0 + +"""Conversation recovery must acknowledge exact data, not merely a write attempt.""" + +from __future__ import annotations + +import asyncio +import base64 +import hashlib +import io +import json +from dataclasses import replace +from typing import TYPE_CHECKING, Any +from unittest.mock import Mock + +import pytest +from botocore.exceptions import ClientError, EndpointConnectionError + +import continuation_session as checkpoint +from hooks import _sha256_tool_input_for_row + +if TYPE_CHECKING: + from claude_agent_sdk import SessionKey + +SESSION_ID = "11111111-1111-4111-8111-aaaaaaaaaaaa" +PROJECT = "-workspace-task" +KEY: SessionKey = {"project_key": PROJECT, "session_id": SESSION_ID} +IDENTITY = checkpoint.CheckpointIdentity("task", "attempt", "request", "user", "owner/repo") +TOOL_INPUT = {"file_path": "/workspace/task/雪.txt", "offset": 1} + + +def assistant(**overrides) -> Any: + entry = { + "type": "assistant", + "uuid": "entry-1", + "sessionId": SESSION_ID, + "message": { + "role": "assistant", + "content": [ + {"type": "tool_use", "id": "toolu_owned", "name": "Read", "input": TOOL_INPUT} + ], + }, + } + entry.update(overrides) + return entry + + +async def capture(store, **kwargs): + return await store.checkpoint_pending( + IDENTITY, + session_id=SESSION_ID, + tool_use_id="toolu_owned", + tool_name="Read", + tool_input=TOOL_INPUT, + timeout_s=0.1, + **kwargs, + ) + + +@pytest.fixture +def body(): + async def prepare(): + store = checkpoint.CheckpointSessionStore(PROJECT) + await store.append(KEY, [assistant()]) + return await capture(store) + + return asyncio.run(prepare()) + + +class TestSessionMirror: + def test_checkpoint_waits_for_exact_action_and_preserves_full_input(self): + async def scenario(): + store = checkpoint.CheckpointSessionStore(PROJECT) + pending = asyncio.create_task(capture(store)) + await asyncio.sleep(0) + assert not pending.done() + await store.append(KEY, [assistant()]) + result = checkpoint.decode_checkpoint(await pending, IDENTITY) + assert result["action"]["tool_input"] == TOOL_INPUT + assert result["action"]["tool_input_sha256"] == _sha256_tool_input_for_row(TOOL_INPUT) + assert result["session_id"] == SESSION_ID + + asyncio.run(scenario()) + + def test_uuid_upserts_keep_order_and_opaque_fields_without_deduplicating_markers(self): + async def scenario(): + store = checkpoint.CheckpointSessionStore(PROJECT) + user: Any = {"type": "user", "uuid": "user-1", "opaque": {"future": [1, 2]}} + marker: Any = {"type": "mode", "data": "plan"} + await store.append(KEY, [user, assistant(), marker]) + replacement = assistant(opaque={"changed": True}) + await store.append(KEY, [replacement, marker]) + expected = [user, replacement, marker, marker] + assert await store.load(KEY) == expected + replacement["opaque"]["changed"] = False + loaded: Any = await store.load(KEY) + loaded[0]["opaque"]["future"].append(3) + unchanged: Any = await store.load(KEY) + assert unchanged[1]["opaque"] == {"changed": True} + assert unchanged[0]["opaque"]["future"] == [1, 2] + + asyncio.run(scenario()) + + def test_restore_round_trip_through_public_store_contract(self, body): + async def scenario(): + restored = checkpoint.CheckpointSessionStore.restore(body, IDENTITY) + assert await restored.load(KEY) == [assistant()] + reply: Any = {"type": "user", "uuid": "reply", "decision": "deny"} + await restored.append(KEY, [reply]) + loaded = await restored.load(KEY) + assert loaded is not None and len(loaded) == 2 + + asyncio.run(scenario()) + + def test_missing_mirror_never_certifies_the_checkpoint(self): + async def scenario(): + with pytest.raises(checkpoint.ContinuationCheckpointError, match="did not acknowledge"): + await capture(checkpoint.CheckpointSessionStore(PROJECT)) + + asyncio.run(scenario()) + + @pytest.mark.parametrize( + "bad_key", + [ + {**KEY, "project_key": "other"}, + {**KEY, "session_id": "../../escape"}, + {**KEY, "subpath": "subagents/agent-other"}, + ], + ) + def test_a_mirror_gap_prevents_later_checkpoint_acknowledgement(self, bad_key): + async def scenario(): + store = checkpoint.CheckpointSessionStore(PROJECT) + with pytest.raises(checkpoint.ContinuationCheckpointError): + await store.append(bad_key, [assistant()]) + await store.append(KEY, [assistant()]) + with pytest.raises(checkpoint.ContinuationCheckpointError, match="Unexpected SDK"): + await capture(store) + + asyncio.run(scenario()) + + def test_mismatched_action_fails_instead_of_waiting_for_a_later_match(self): + async def scenario(): + store = checkpoint.CheckpointSessionStore(PROJECT) + entry = assistant() + entry["message"]["content"][0]["input"] = {"file_path": "/different"} + await store.append(KEY, [entry]) + with pytest.raises(checkpoint.ContinuationCheckpointError, match="disagrees"): + await capture(store) + + asyncio.run(scenario()) + + def test_completed_tool_cannot_be_recorded_as_pending(self): + async def scenario(): + store = checkpoint.CheckpointSessionStore(PROJECT) + result: Any = { + "type": "user", + "message": {"content": [{"type": "tool_result", "tool_use_id": "toolu_owned"}]}, + } + await store.append( + KEY, + [assistant(), result], + ) + with pytest.raises(checkpoint.ContinuationCheckpointError, match="already has"): + await capture(store) + + asyncio.run(scenario()) + + def test_buffer_limit_failure_does_not_silently_drop_entries(self, monkeypatch): + async def scenario(): + store = checkpoint.CheckpointSessionStore(PROJECT) + monkeypatch.setattr(checkpoint, "MAX_CHECKPOINT_ENTRIES", 1) + with pytest.raises(checkpoint.ContinuationCheckpointError, match="limit"): + await store.append(KEY, [assistant(), {"type": "mode"}]) + with pytest.raises(checkpoint.ContinuationCheckpointError, match="limit"): + await capture(store) + + asyncio.run(scenario()) + + def test_cancelled_checkpoint_wait_propagates_cancellation(self): + async def scenario(): + task = asyncio.create_task(capture(checkpoint.CheckpointSessionStore(PROJECT))) + await asyncio.sleep(0) + task.cancel() + with pytest.raises(asyncio.CancelledError): + await task + + asyncio.run(scenario()) + + def test_unwritten_session_load_returns_none(self): + assert asyncio.run(checkpoint.CheckpointSessionStore(PROJECT).load(KEY)) is None + + def test_transcript_cannot_mix_another_session_into_the_saved_conversation(self): + async def scenario(): + store = checkpoint.CheckpointSessionStore(PROJECT) + with pytest.raises(checkpoint.ContinuationCheckpointError, match="another SDK session"): + await store.append( + KEY, [assistant(sessionId="22222222-2222-4222-8222-bbbbbbbbbbbb")] + ) + + asyncio.run(scenario()) + + def test_duplicate_tool_identity_is_ambiguous(self): + async def scenario(): + store = checkpoint.CheckpointSessionStore(PROJECT) + await store.append(KEY, [assistant(), assistant(uuid="different-entry")]) + with pytest.raises(checkpoint.ContinuationCheckpointError, match="duplicate"): + await capture(store) + + asyncio.run(scenario()) + + +class VersionedS3: + """A versioned object service with commit-then-disconnect failure injection.""" + + def __init__(self): + self.objects = {} + self.calls = [] + self.write_error = None + self.lose_write_reply = False + self.versioned = True + self.bad_checksum = False + self.bad_length = False + self.streams = [] + + def put_object(self, **kwargs): + self.calls.append(("put", kwargs)) + if self.write_error: + raise self.write_error + object_key = (kwargs["Bucket"], kwargs["Key"]) + if kwargs.get("IfNoneMatch") == "*" and object_key in self.objects: + raise ClientError({"Error": {"Code": "PreconditionFailed"}}, "PutObject") + versions = self.objects.setdefault(object_key, []) + version = f"version-{len(versions) + 1}" if self.versioned else "null" + versions.append((version, kwargs["Body"])) + if self.lose_write_reply: + raise EndpointConnectionError(endpoint_url="https://s3.invalid") + return {"VersionId": version} + + def get_object(self, **kwargs): + self.calls.append(("get", kwargs)) + versions = self.objects.get((kwargs["Bucket"], kwargs["Key"])) + if not versions: + raise ClientError({"Error": {"Code": "NoSuchKey"}}, "GetObject") + requested = kwargs.get("VersionId") + version, body = ( + next(v for v in versions if v[0] == requested) if requested else versions[-1] + ) + stream = io.BytesIO(body) + self.streams.append(stream) + return { + "Body": stream, + "VersionId": version, + "ContentLength": len(body) + int(self.bad_length), + "ChecksumSHA256": "wrong" + if self.bad_checksum + else base64.b64encode(hashlib.sha256(body).digest()).decode(), + } + + +class TestImmutableStorage: + @pytest.fixture + def storage(self): + client = VersionedS3() + return checkpoint.S3ContinuationCheckpoints("private-checkpoints", client=client), client + + def test_default_client_refuses_ambient_credentials(self, monkeypatch): + import aws_session + + client_factory = Mock() + monkeypatch.setattr(aws_session, "is_scoped", lambda: False) + monkeypatch.setattr(aws_session, "tenant_client", client_factory) + with pytest.raises(checkpoint.ContinuationCheckpointError, match="task-scoped credentials"): + checkpoint.S3ContinuationCheckpoints("private-checkpoints") + client_factory.assert_not_called() + + def test_default_client_uses_attributed_scoped_factory(self, monkeypatch): + import aws_session + + client_factory = Mock(return_value=VersionedS3()) + monkeypatch.setattr(aws_session, "is_scoped", lambda: True) + monkeypatch.setattr(aws_session, "tenant_client", client_factory) + store = checkpoint.S3ContinuationCheckpoints("private-checkpoints") + assert store.client is client_factory.return_value + assert client_factory.call_args.args == ("s3",) + assert client_factory.call_args.kwargs["config"].retries == {"total_max_attempts": 1} + + def test_save_requires_read_back_and_returns_a_version_pinned_receipt(self, storage, body): + store, client = storage + receipt = store.save(body, IDENTITY) + assert [call[0] for call in client.calls] == ["put", "get"] + assert receipt.key == IDENTITY.prefix + hashlib.sha256(body).hexdigest() + ".json" + assert client.calls[0][1]["IfNoneMatch"] == "*" + assert client.calls[0][1]["ServerSideEncryption"] == "AES256" + assert store.load(receipt, IDENTITY) == body + assert client.calls[-1][1]["VersionId"] == receipt.version_id + assert all(stream.closed for stream in client.streams) + + def test_lost_write_reply_recovers_from_exact_read_back(self, storage, body): + store, client = storage + client.lose_write_reply = True + receipt = store.save(body, IDENTITY) + assert receipt.version_id == "version-1" + assert store.save(body, IDENTITY) == receipt + assert len(client.objects[(store.bucket, receipt.key)]) == 1 + + def test_uncommitted_write_failure_never_returns_a_receipt(self, storage, body): + store, client = storage + client.write_error = ClientError({"Error": {"Code": "AccessDenied"}}, "PutObject") + with pytest.raises(checkpoint.ContinuationCheckpointError, match="keep the current worker"): + store.save(body, IDENTITY) + assert not client.objects + + @pytest.mark.parametrize("failure", ["versioned", "bad_checksum", "bad_length"]) + def test_unverifiable_storage_never_acknowledges(self, storage, body, failure): + store, client = storage + setattr(client, failure, failure != "versioned") + with pytest.raises(checkpoint.ContinuationCheckpointError, match="could not be verified"): + store.save(body, IDENTITY) + assert all(stream.closed for stream in client.streams) + + def test_load_keeps_the_original_version_even_if_current_key_changes(self, storage, body): + store, client = storage + receipt = store.save(body, IDENTITY) + client.put_object(Bucket=store.bucket, Key=receipt.key, Body=b"replacement") + assert store.load(receipt, IDENTITY) == body + + @pytest.mark.parametrize("field", ["task_id", "attempt_id", "request_id", "user_id", "repo"]) + def test_cross_identity_data_is_rejected_before_any_io(self, storage, body, field): + store, client = storage + wrong = replace(IDENTITY, **{field: "other"}) + with pytest.raises(checkpoint.ContinuationCheckpointError, match="identity"): + store.save(body, wrong) + assert not client.calls + + def test_receipt_cannot_redirect_a_read_outside_its_task(self, storage, body): + store, client = storage + receipt = store.save(body, IDENTITY) + count = len(client.calls) + with pytest.raises(checkpoint.ContinuationCheckpointError, match="outside this task"): + store.load(replace(receipt, key="continuations/other/data"), IDENTITY) + assert len(client.calls) == count + + def test_valid_transport_checksum_does_not_replace_receipt_integrity(self, storage, body): + store, client = storage + receipt = store.save(body, IDENTITY) + client.objects[(store.bucket, receipt.key)][0] = (receipt.version_id, b"x" * len(body)) + with pytest.raises(checkpoint.ContinuationCheckpointError, match="does not match"): + store.load(receipt, IDENTITY) + + +class TestEnvelope: + @pytest.mark.parametrize( + "change", + [ + lambda data: data.update(version=2), + lambda data: data.update(version=True), + lambda data: data.update(session_id="../../other"), + lambda data: data["action"].update(tool_input_sha256="0" * 64), + lambda data: data["action"].update(tool_input={"file_path": "changed"}), + lambda data: data.update(entries=[]), + lambda data: data.update(environment={"AWS_SECRET_ACCESS_KEY": "synthetic"}), + ], + ) + def test_corrupt_or_incompatible_envelope_fails(self, body, change): + data = json.loads(body) + change(data) + with pytest.raises(checkpoint.ContinuationCheckpointError): + checkpoint.decode_checkpoint(json.dumps(data).encode(), IDENTITY) + + @pytest.mark.parametrize("body", [b"{", b"\xff", b"null", b""]) + def test_invalid_json_fails(self, body): + with pytest.raises(checkpoint.ContinuationCheckpointError): + checkpoint.decode_checkpoint(body, IDENTITY) + + @pytest.mark.parametrize("value", ["../task", "", "task/other", "x" * 129]) + def test_path_components_cannot_escape_task_prefix(self, value): + with pytest.raises(checkpoint.ContinuationCheckpointError): + replace(IDENTITY, task_id=value) diff --git a/docs/verification/645-p3-implementation-plan.md b/docs/verification/645-p3-implementation-plan.md index 6a7b344af..06592e166 100644 --- a/docs/verification/645-p3-implementation-plan.md +++ b/docs/verification/645-p3-implementation-plan.md @@ -368,6 +368,12 @@ do not cover: requests still default to 300 seconds with a maximum submission setting of 3,600 seconds. Waiting beyond a worker's lifetime needs a defined recovery path; current same-worker suspend/resume does not provide that continuation. +- [x] Implement and verify the [conversation/action checkpoint prerequisite](./645-p3-session-recovery-20260917.md). + The public SDK store restores approve/deny in a new process after removing the + old configuration. Immutable S3 version reads and task-scoped access passed nine + live checks; all temporary resources were removed. Full agent quality passed + 2,004 tests. Workspace preservation and production worker-replacement wiring + remain required before changing approval deadlines or releasing a waiting worker. - [x] Implement [actionable Slack/Linear notifications and CLI response instructions](./645-p3-approval-ux-20260917.md), including saved decisions and closure reasons. Unwrap the actual agent approval milestones and keep approval @@ -428,15 +434,21 @@ The following work is still required before changing the current deadline defaul session. First test the SDK's supported recovery behavior at a pending tool hook and define how the saved decision reaches the agent's normal reasoning. The [local pinned-SDK recovery probe](./645-p3-session-recovery-20260917.md) - now verifies conversation recovery from a copied session after abrupt process - loss. The pending tool call is omitted from the restored model context; a - newly proposed call receives a new ID and hook. Cloud-worker recovery and the - explicit transfer of the saved action/decision remain required. + now verifies conversation recovery through the public `SessionStore` after + abrupt process loss and deletion of the old configuration. The pending call + is omitted from the restored model context; the checkpoint carries its full + action separately. A continuation prompt transfers that action and decision, + and the newly proposed call receives a new ID and hook. Local approve/deny + pass; replacement cloud-worker recovery remains required. 2. Add a durable continuation checkpoint: workspace changes, conversation/session state and exact pending request identity. Store it under task-scoped permissions, exclude credentials, and confirm the write before releasing the worker. A failed checkpoint must produce actionable feedback, never a claim that the work was saved. Test restoration of uncommitted and untracked files. + `agent/src/continuation_session.py` now implements acknowledged conversation/ + action storage with immutable S3 version receipts; nine live AWS permission + and persistence checks passed. Workspace preservation, production IAM/retention, + lifecycle-barrier integration and conditional receipt publication remain open. 3. Separate the task from each worker attempt. Add an attempt generation to launch tokens, saved handles, writes and reservations so a replaced worker cannot overwrite its successor. Release capacity while the task waits and diff --git a/docs/verification/645-p3-session-recovery-20260917.md b/docs/verification/645-p3-session-recovery-20260917.md index 407865f99..af7935d41 100644 --- a/docs/verification/645-p3-session-recovery-20260917.md +++ b/docs/verification/645-p3-session-recovery-20260917.md @@ -1,12 +1,15 @@ # ADR-021 pending approval session recovery -A local diagnostic on September 17 verified that the pinned agent SDK can resume -a saved conversation in a new process after its original process dies while -waiting in a tool hook. It does **not** resume that pending hook. This is one -prerequisite for retaining unanswered approvals beyond a worker's lifetime; -production continuation remains unimplemented. +September 17 diagnostics verified that the pinned agent SDK can resume a saved +conversation in a new process after its original process dies while waiting in +a tool hook. It does **not** resume that pending hook. The implemented SDK +conversation store now retains the exact proposed action alongside the session; +approve and deny passed after deleting the original configuration. Nine live +S3 checks also passed, including task isolation and reads pinned to an immutable +object version. These are prerequisites for retaining unanswered approvals +beyond a worker's lifetime; production continuation remains unimplemented. -## What was exercised +## Initial recovery diagnostic The probe used Python `claude-agent-sdk` **0.2.110** and its bundled Claude Code **2.1.191**, matching the reviewed application versions. The model endpoint was @@ -46,19 +49,128 @@ for a different action. This requirement does not add a separate relevance or staleness checker. The agent assesses relevance through its ordinary reasoning, as requested. +## Supported conversation checkpoint implementation + +The initial configuration-copy diagnostic established the SDK behavior. The new +`agent/src/continuation_session.py` uses the SDK's supported `SessionStore` +contract instead. It never reads or copies CLI configuration or authentication +files. + +- `CheckpointSessionStore.append()` preserves opaque SDK journal entries, updates + existing UUID entries in place and retains entries without UUIDs. It rejects + mixed sessions and subagent transcripts; detached/subagent work is outside the + current runner's supported waiting state. +- With `session_store_flush="eager"`, the SDK still mirrors asynchronously. + `checkpoint_pending()` waits for the exact assistant tool ID, name and full + input to appear. A missing batch, mismatch or existing tool result prevents + acknowledgement. A rejected batch poisons the buffer, so later data cannot + conceal the possible gap. +- The versioned envelope contains task, attempt, request, user and repository + identity; SDK project/session identity; full pending action and its approval-row + compatible hash; and transcript entries. It is bounded to 16 MiB and 50,000 + entries. Extra top-level configuration or environment fields are rejected. +- `S3ContinuationCheckpoints.save()` conditionally creates a SHA-256-addressed + object with encryption and checksum. It returns a receipt only after checking + the exact bytes and a real S3 version. A lost write reply or repeated save can + recover through read-back; failure to prove persistence never reports success. +- `load()` reads the receipt's exact version, validates its checksum, size, + envelope and identity, then supplies the journal to a new store for SDK + materialization. Later changes to the current object cannot change that + version's saved data. + +Storage uses `continuations////.json`. +The default AWS client requires an active task-scoped session before using the +attributed factory. This explicit check matters because `tenant_client()` alone +can fall back to ambient credentials. The future deployment needs a private, +versioned bucket and `PutObject`, `GetObject`, `GetObjectVersion` permission only +under the task's prefix. Existing artifact-write permissions are insufficient. +This component does not configure retention or delete checkpoints. + +Transcript and action contents remain private task data and may themselves +contain sensitive text. Excluding authentication files does not mean the +conversation is safe to log or publish. + +## Real SDK approve and deny checks + +`agent/tests/test_continuation_sdk_probe.py` uses the same SDK **0.2.110** and CLI +**2.1.191** with synthetic AWS credentials and a deterministic loopback model: + +1. The original `Read` hook waits for the new store to acknowledge the exact + proposed action and writes that checkpoint. +2. The test kills the entire original process group, verifies that its tool did + not execute, and deletes the original configuration directory. +3. It restores the owned workspace marker at the same path and starts a fresh + process with `resume`, the restored store and eager mirroring. This workspace + copy is a test fixture, not the production workspace checkpoint. +4. The continuation prompt carries the full saved action and human decision. + The model proposes `toolu_restored`; the fresh permission hook allows one + `Read` for approval and denies it for denial. Both sessions complete with + their original session ID. +5. A synthetic authentication-file sentinel is absent from the saved checkpoint. + No original CLI configuration is used during restoration. + +The final full agent quality run enabled both tests: + +```bash +ABCA_TEST_SDK_CONTINUATION=1 MISE_EXPERIMENTAL=1 mise run //agent:quality +``` + +Lint, formatting and type checks passed; **2,004 tests passed, 11 skipped**, with +86.57% total coverage. This includes **45 checkpoint unit tests** and both real SDK +cases. The skipped cases are the existing opt-in DynamoDB Local checks. One +dependency warning concerns Starlette's deprecated `httpx` test-client support. +Unit failure injection covers dropped mirror data, corrupt or cross-task +checkpoints, cancelled waits, ambiguous write replies and unverifiable storage. +The agent Bandit high-severity check and documentation build passed. The full +repository silent-success scan reported 92 findings in 50 files unchanged from +the pre-change commit `95fd6746`; none were in this component. The corresponding +change-only scan against that commit passed. + +## Live S3 verification and cleanup + +The isolated AWS probe used the development account, region `us-west-2`, private +versioned bucket `abca-645-checkpoint--20260917` and role +`abca-645-checkpoint-probe-20260917`. The role allowed only +`PutObject`, `GetObject` and `GetObjectVersion` under +`continuations/${aws:PrincipalTag/task_id}/*`, with task/user/repository session +tags. It used a real checkpoint produced by the SDK test above. + +| Check | Result | +| --- | --- | +| Save, read and repeat the same save | Same verified version receipt | +| Object encryption and SHA-256 checksum | AES256 and matching checksum | +| Overwrite current object, then load saved receipt | Original version and bytes restored | +| Another task reads current object | Access denied | +| Another task reads pinned version | Access denied | +| Another task writes this task's prefix | Access denied | +| Owning task lists the bucket | Access denied | +| Owning task deletes current object | Access denied | +| Owning task deletes pinned version | Access denied | + +All nine checks passed. Cleanup verified ownership tags, removed both exact +object versions, the bucket, inline policy and role, and confirmed bucket `404` +and role `NoSuchEntity`. No normal deployment resource or setting changed. + ## Limits and next steps -This test proves local SDK conversation recovery with an abrupt process loss and -a copied workspace/configuration directory. Its deterministic model demonstrates -the transport behavior; it does not prove how a real model will interpret the -continuation prompt. +These tests prove local SDK conversation recovery after abrupt process loss, +explicit transfer of the action/decision, and the S3 storage contract under real +task-scoped permissions. The deterministic model demonstrates transport and hook +behavior; it does not prove how a real model will interpret the continuation +prompt. -The probe does not yet verify recovery on a replacement cloud worker, portable -workspace paths, durable object storage, exclusion of credentials from a -production checkpoint, task-attempt fencing, exactly-once decision consumption, -denial, cancellation or loss of a checkpoint/launch response. Those remain part -of the [unanswered-approval implementation order](./645-p3-implementation-plan.md#unanswered-approvals-implementation-order). +Production `runner.py` does not yet install this store or resume from it. Still +required: workspace preservation including uncommitted/untracked files; recovery +on a replacement cloud worker; a stable workspace path; lifecycle barrier and +conditional publication of the acknowledged receipt; task-attempt fencing and +reservation transfer; exactly-once decision consumption; cancellation while +parked; lost launch replies; and request retention/deadline changes. A conversation +receipt alone never permits worker release. Those requirements remain in the +[unanswered-approval implementation order](./645-p3-implementation-plan.md#unanswered-approvals-implementation-order). -The executable probe, exact model requests, hook audits, copied synthetic session -files and verification result are archived with a SHA-256 manifest under +The initial executable probe, model requests, hook audits and synthetic session +files are archived with a SHA-256 manifest under `/Users/sphias/.local/share/abca-verification/645-p3-20260916/session-recovery`. +The new implementation snapshot, SDK fixtures, test logs and live S3 policy, +verification and cleanup receipts are archived separately under +`/Users/sphias/.local/share/abca-verification/645-p3-20260916/continuation-checkpoint`. From 51e36ce1f65927c66066cad5e52bfc1b59191e99 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Thu, 17 Sep 2026 11:51:33 -0400 Subject: [PATCH 074/149] feat: preserve workspace state for MicroVM continuation (#645) --- agent/README.md | 18 +- agent/src/continuation_workspace.py | 771 ++++++++++++++++++ agent/tests/test_continuation_sdk_probe.py | 38 +- agent/tests/test_continuation_workspace.py | 480 +++++++++++ .../645-p3-implementation-plan.md | 16 +- .../645-p3-session-recovery-20260917.md | 17 +- .../645-p3-workspace-recovery-20260917.md | 134 +++ 7 files changed, 1457 insertions(+), 17 deletions(-) create mode 100644 agent/src/continuation_workspace.py create mode 100644 agent/tests/test_continuation_workspace.py create mode 100644 docs/verification/645-p3-workspace-recovery-20260917.md diff --git a/agent/README.md b/agent/README.md index f87750104..f1ee7b456 100644 --- a/agent/README.md +++ b/agent/README.md @@ -330,16 +330,24 @@ are required; the current production artifact grants do not supply these reads. The adapter neither lists nor deletes objects. Retention must be arranged by the future integration after the request closes. -Before releasing a worker, that integration must also preserve the workspace, -hold the lifecycle barrier and conditionally publish the verified checkpoint for -the same task attempt. A failed save must leave the worker available and report +`src/continuation_workspace.py` now captures and restores the local workspace: +Git history, staged/unstaged changes, and untracked/ignored files, with a 1 GiB / +100,000-entry default limit. It rebuilds Git configuration and never replaces an +existing destination. Unsupported Git/filesystem states or detected concurrent +writes prevent capture. This archive still needs durable upload and integration; +see the [workspace recovery boundaries](../docs/verification/645-p3-workspace-recovery-20260917.md). + +Before releasing a worker, the integration must hold the lifecycle barrier and +conditionally publish verified conversation and workspace receipts for the same +task attempt. A failed save must leave the worker available and report the failure. Checkpoints are private task data: transcripts and proposed actions can contain sensitive content. The module does not read CLI authentication files or the process environment, but does not redact conversation contents. The opt-in test uses the actual pinned SDK/CLI with a deterministic loopback -model. It kills the original process, deletes its configuration, restores from -the new store, and verifies approve and deny through a fresh tool hook: +model. It kills the original process, deletes its configuration and workspace, +restores from the conversation store and workspace archive, and verifies approve +and deny through a fresh tool hook: ```bash cd agent diff --git a/agent/src/continuation_workspace.py b/agent/src/continuation_workspace.py new file mode 100644 index 000000000..c0bbe51bc --- /dev/null +++ b/agent/src/continuation_workspace.py @@ -0,0 +1,771 @@ +# Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. +# SPDX-License-Identifier: MIT-0 + +"""Offline workspace archives for future worker continuation. + +The caller must stop workspace writers with the lifecycle barrier. Capture is +local only: durable upload/read-back and conditional publication alongside the +conversation receipt are separate requirements before releasing a worker. + +Preserve all working files, including ignored files, rather than guessing which +can be regenerated. Rebuild Git administration from a bundle and staged patch; +never copy the original Git config, hooks, home directory or process environment. +Repository contents themselves remain private and may contain sensitive data. +""" + +from __future__ import annotations + +import base64 +import hashlib +import io +import json +import os +import re +import selectors +import shutil +import signal +import stat +import subprocess +import tarfile +import tempfile +import time +from dataclasses import asdict, dataclass +from pathlib import Path, PurePosixPath +from typing import BinaryIO, NoReturn + +from continuation_session import CheckpointIdentity, ContinuationCheckpointError + +_OID = re.compile(r"[0-9a-f]{40}\Z") +_SHA256 = re.compile(r"[0-9a-f]{64}\Z") +_REPO = re.compile(r"[A-Za-z0-9_.-]+/[A-Za-z0-9_.-]+\Z") +_CHUNK = 1024 * 1024 +_ADMIN = {"git.bundle", "index.patch", "manifest.json"} +_MAX_PATH_BYTES = 4096 +_PERMISSION_BITS = 0o777 +_MAX_EXCLUDE_BYTES = 1024 * 1024 + + +class WorkspaceCheckpointError(ContinuationCheckpointError): + """A content-free stage code suitable for future lifecycle feedback.""" + + def __init__(self, code: str, message: str) -> None: + self.code = code + super().__init__(message) + + +@dataclass(frozen=True) +class WorkspaceLimits: + max_bytes: int = 1024 * 1024 * 1024 + max_entries: int = 100_000 + git_timeout_s: int = 60 + + def __post_init__(self) -> None: + if any( + type(value) is not int or value <= 0 + for value in (self.max_bytes, self.max_entries, self.git_timeout_s) + ): + raise ValueError("Workspace limits must be positive integers") + + +@dataclass(frozen=True) +class WorkspaceArchive: + sha256: str + size_bytes: int + head: str + branch: str | None + entries: int + + +_DEFAULT_LIMITS = WorkspaceLimits() + + +def _fail(code: str, message: str) -> NoReturn: + raise WorkspaceCheckpointError(code, message) + + +def _git( + root: Path, + args: list[str], + limits: WorkspaceLimits, + *, + output: BinaryIO | None = None, + max_bytes: int | None = None, + source: BinaryIO | None = None, +) -> bytes: + """Bound Git output/time, disable hooks/helpers, and never log repository data.""" + env = {key: value for key, value in os.environ.items() if not key.startswith("GIT_")} + env.update( + GIT_CONFIG_NOSYSTEM="1", + GIT_CONFIG_GLOBAL=os.devnull, + GIT_ATTR_NOSYSTEM="1", + GIT_NO_REPLACE_OBJECTS="1", + GIT_TERMINAL_PROMPT="0", + GIT_OPTIONAL_LOCKS="0", + LC_ALL="C", + ) + command = [ + "git", + "-c", + f"core.hooksPath={os.devnull}", + "-c", + "core.fsmonitor=false", + "-c", + "gc.auto=0", + "-c", + "protocol.allow=never", + "-c", + f"safe.directory={root}", + *args, + ] + target = output if output is not None else io.BytesIO() + bound = max_bytes if max_bytes is not None else min(limits.max_bytes, 16 * 1024 * 1024) + count = 0 + # stderr may contain repository contents/configuration; only the stage and + # exit code cross this boundary. The caller retains its original workspace. + with ( + subprocess.Popen( + command, + cwd=root, + env=env, + stdin=source if source is not None else subprocess.DEVNULL, + stdout=subprocess.PIPE, + stderr=subprocess.DEVNULL, + start_new_session=True, + ) as process, + selectors.DefaultSelector() as selector, + ): + if process.stdout is None: + _fail("git_failed", "Workspace Git output pipe is unavailable") + selector.register(process.stdout, selectors.EVENT_READ) + deadline = time.monotonic() + limits.git_timeout_s + try: + while selector.get_map(): + remaining = deadline - time.monotonic() + if remaining <= 0: + _fail("git_timeout", f"Workspace Git {args[0]} timed out") + for key, _ in selector.select(min(remaining, 0.1)): + block = os.read(key.fd, _CHUNK) + if not block: + selector.unregister(key.fd) + continue + count += len(block) + if count > bound: + _fail("size_limit", f"Workspace Git {args[0]} exceeds the byte limit") + target.write(block) + try: + code = process.wait(timeout=max(0.01, deadline - time.monotonic())) + except subprocess.TimeoutExpired as exc: + raise WorkspaceCheckpointError("git_timeout", "Workspace Git timed out") from exc + if code: + _fail("git_failed", f"Workspace Git {args[0]} failed (exit {code})") + finally: + if process.poll() is None: + process.stdout.close() + try: + process.wait(timeout=0.1) + except subprocess.TimeoutExpired: + try: + os.killpg(process.pid, signal.SIGKILL) + except (ProcessLookupError, PermissionError): + # Git may exit between wait and killpg. An error is + # ignorable only after wait confirms that it exited. + process.wait(timeout=0.1) + process.wait() + return target.getvalue() if isinstance(target, io.BytesIO) else b"" + + +def _path(value: str) -> PurePosixPath: + if not isinstance(value, str) or not value or len(value.encode()) > _MAX_PATH_BYTES: + _fail("invalid_path", "Workspace archive path is invalid") + path = PurePosixPath(value) + if ( + path.is_absolute() + or str(path) != value + or "\\" in value + or "\0" in value + or any(part in {".", ".."} or part.lower() == ".git" for part in path.parts) + ): + _fail("invalid_path", "Workspace archive path escapes the working tree") + return path + + +def _stamp(info: os.stat_result) -> tuple[int, ...]: + return ( + info.st_mode, + info.st_dev, + info.st_ino, + info.st_size, + info.st_mtime_ns, + info.st_ctime_ns, + info.st_nlink, + ) + + +def _scan(root: Path, limits: WorkspaceLimits) -> dict[str, tuple[int, ...]]: + entries = {} + device = root.stat().st_dev + for directory, dirs, files, fd in os.fwalk(root, follow_symlinks=False): + relative = Path(directory).relative_to(root) + if relative == Path("."): + dirs[:] = [name for name in dirs if name != ".git"] + for name in sorted(dirs + files): + path = str(relative / name) + _path(path) + info = os.stat(name, dir_fd=fd, follow_symlinks=False) + if ( + not ( + stat.S_ISREG(info.st_mode) + or stat.S_ISDIR(info.st_mode) + or stat.S_ISLNK(info.st_mode) + ) + or info.st_dev != device + or (stat.S_ISREG(info.st_mode) and info.st_nlink != 1) + or info.st_mode & (stat.S_ISUID | stat.S_ISGID | stat.S_ISVTX) + ): + _fail("unsupported_file", "Workspace has a special, linked or mounted file") + entries[path] = _stamp(info) + if len(entries) > limits.max_entries: + _fail("entry_limit", "Workspace exceeds the entry limit") + return dict(sorted(entries.items())) + + +def _open_file(root: Path, path: str) -> BinaryIO: + """Open through directory descriptors; never traverse a substituted symlink.""" + parts = _path(path).parts + fd = os.open(root, os.O_RDONLY | os.O_DIRECTORY | os.O_NOFOLLOW) + try: + for part in parts[:-1]: + child = os.open(part, os.O_RDONLY | os.O_DIRECTORY | os.O_NOFOLLOW, dir_fd=fd) + os.close(fd) + fd = child + result = os.open(parts[-1], os.O_RDONLY | os.O_NOFOLLOW | os.O_NONBLOCK, dir_fd=fd) + return os.fdopen(result, "rb") + finally: + os.close(fd) + + +def _git_state(root: Path, limits: WorkspaceLimits) -> dict: + gitdir = root / ".git" + if gitdir.is_symlink() or not gitdir.is_dir(): + _fail("unsupported_git", "Workspace requires a standalone Git checkout") + for name in ( + "MERGE_HEAD", + "CHERRY_PICK_HEAD", + "REVERT_HEAD", + "rebase-apply", + "rebase-merge", + "sequencer", + "index.lock", + "shallow", + "objects/info/alternates", + "info/grafts", + "info/sparse-checkout", + ): + if os.path.lexists(gitdir / name): + _fail("unsupported_git", "Workspace has an active or unsupported Git operation") + if ( + _git(root, ["rev-parse", "--show-toplevel"], limits).decode().strip() != str(root) + or Path(_git(root, ["rev-parse", "--absolute-git-dir"], limits).decode().strip()) != gitdir + or _git(root, ["rev-parse", "--show-object-format"], limits).strip() != b"sha1" + ): + _fail("unsupported_git", "Workspace Git layout or object format is unsupported") + head = _git(root, ["rev-parse", "HEAD"], limits).decode().strip() + branch_value = _git(root, ["rev-parse", "--abbrev-ref", "HEAD"], limits).decode().strip() + branch = None if branch_value == "HEAD" else branch_value + refs = {} + for line in _git( + root, ["for-each-ref", "--format=%(objectname) %(refname)"], limits + ).splitlines(): + oid, ref = line.decode().split(" ", 1) + if ref.startswith("refs/replace/"): + _fail("unsupported_git", "Workspace replacement refs cannot be checkpointed") + refs[ref] = oid + index = _git(root, ["ls-files", "--stage", "-z"], limits) + for line in index.split(b"\0"): + if not line: + continue + details, path = line.split(b"\t", 1) + mode, _, stage = details.split() + _path(path.decode()) + if mode == b"160000" or stage != b"0": + _fail( + "unsupported_git", + "Workspace submodules or unresolved index entries are unsupported", + ) + flags = _git(root, ["ls-files", "-v", "-z"], limits).split(b"\0") + if any(entry and entry[:1] != b"H" for entry in flags): + _fail("unsupported_git", "Workspace has sparse or assume-unchanged index flags") + visible = _git( + root, + ["diff", "--cached", "--raw", "--no-ext-diff", "--no-textconv", "--ita-visible-in-index"], + limits, + ) + invisible = _git( + root, + ["diff", "--cached", "--raw", "--no-ext-diff", "--no-textconv", "--ita-invisible-in-index"], + limits, + ) + if visible != invisible: + _fail("unsupported_git", "Workspace intent-to-add entries cannot be checkpointed") + exclude_b64 = None + if os.path.lexists(gitdir / "info/exclude"): + with _open_file(gitdir, "info/exclude") as source: + if not stat.S_ISREG(os.fstat(source.fileno()).st_mode): + _fail("unsupported_git", "Workspace Git ignore file must be regular") + exclude = source.read(_MAX_EXCLUDE_BYTES + 1) + if len(exclude) > _MAX_EXCLUDE_BYTES: + _fail("size_limit", "Workspace Git ignore file exceeds the byte limit") + exclude_b64 = base64.b64encode(exclude).decode() + return { + "head": head, + "branch": branch, + "refs": refs, + "index_sha256": hashlib.sha256(index).hexdigest(), + "exclude_b64": exclude_b64, + } + + +class _DigestReader: + def __init__(self, stream: BinaryIO) -> None: + self.stream = stream + self.digest = hashlib.sha256() + + def read(self, size: int = -1) -> bytes: + block = self.stream.read(size) + self.digest.update(block) + return block + + +def _file_digest(path: Path) -> str: + with path.open("rb") as stream: + return hashlib.file_digest(stream, "sha256").hexdigest() + + +def _repository_identity(identity: CheckpointIdentity) -> None: + if not _REPO.fullmatch(identity.repo) or any( + part in {".", ".."} for part in identity.repo.split("/") + ): + _fail("invalid_identity", "Workspace repository identity is invalid") + + +def capture_workspace( + workspace: Path, + destination: Path, + identity: CheckpointIdentity, + *, + limits: WorkspaceLimits = _DEFAULT_LIMITS, +) -> WorkspaceArchive: + """Create an owned local archive; never overwrite a previously saved file.""" + try: + return _capture_workspace(workspace, destination, identity, limits) + except WorkspaceCheckpointError: + raise + except ( + OSError, + ValueError, + TypeError, + RecursionError, + subprocess.SubprocessError, + tarfile.TarError, + ) as exc: + raise WorkspaceCheckpointError( + "capture_failed", "Workspace capture failed; keep the worker available" + ) from exc + + +def _capture_workspace( + workspace: Path, destination: Path, identity: CheckpointIdentity, limits: WorkspaceLimits +) -> WorkspaceArchive: + _repository_identity(identity) + root = workspace.absolute() + destination = destination.absolute() + if root.resolve() != root or not root.is_dir(): + _fail("invalid_path", "Workspace must be an existing canonical directory") + if destination.is_relative_to(root) or destination.parent.resolve() != destination.parent: + _fail("invalid_path", "Workspace archive must be outside the working tree") + if os.path.lexists(destination): + _fail("destination_exists", "Workspace archive destination already exists") + original = _git_state(root, limits) + before = _scan(root, limits) + with tempfile.TemporaryDirectory( + prefix=".workspace-capture-", dir=destination.parent + ) as scratch: + temp = Path(scratch) + bundle, patch = temp / "git.bundle", temp / "index.patch" + with bundle.open("wb") as output: + _git( + root, + ["bundle", "create", "-", "--all", "HEAD"], + limits, + output=output, + max_bytes=limits.max_bytes, + ) + with patch.open("wb") as output: + _git( + root, + [ + "diff", + "--cached", + "--binary", + "--full-index", + "--no-ext-diff", + "--no-textconv", + "HEAD", + "--", + ], + limits, + output=output, + max_bytes=limits.max_bytes - bundle.stat().st_size, + ) + records = [] + total = bundle.stat().st_size + patch.stat().st_size + archive = temp / "workspace.tar" + with tarfile.open(archive, "w", format=tarfile.PAX_FORMAT) as tar: + for path, stamp in before.items(): + mode, _, _, size, *_ = stamp + member = tarfile.TarInfo("files/" + path) + member.mode = stat.S_IMODE(mode) + record = {"path": path, "mode": member.mode} + if stat.S_ISDIR(mode): + member.type = tarfile.DIRTYPE + record["kind"] = "directory" + tar.addfile(member) + elif stat.S_ISLNK(mode): + member.type = tarfile.SYMTYPE + member.linkname = os.readlink(root / path) + record.update(kind="symlink", target=member.linkname) + tar.addfile(member) + else: + total += size + if total > limits.max_bytes: + _fail("size_limit", "Workspace exceeds the byte limit") + member.size = size + with _open_file(root, path) as stream: + if _stamp(os.fstat(stream.fileno())) != stamp: + _fail("workspace_changed", "Workspace changed during capture") + reader = _DigestReader(stream) + tar.addfile(member, reader) + if _stamp(os.fstat(stream.fileno())) != stamp: + _fail("workspace_changed", "Workspace changed during capture") + record.update(kind="file", size=size, sha256=reader.digest.hexdigest()) + records.append(record) + metadata = { + "version": 1, + "identity": asdict(identity), + "workspace": str(root), + "git": original, + "files": records, + "bundle_sha256": _file_digest(bundle), + "patch_sha256": _file_digest(patch), + } + for path in (bundle, patch): + member = tarfile.TarInfo(path.name) + member.mode, member.size = 0o600, path.stat().st_size + with path.open("rb") as stream: + tar.addfile(member, stream) + body = json.dumps(metadata, sort_keys=True, separators=(",", ":")).encode() + member = tarfile.TarInfo("manifest.json") + member.mode, member.size = 0o600, len(body) + tar.addfile(member, io.BytesIO(body)) + if archive.stat().st_size > limits.max_bytes: + _fail("size_limit", "Workspace archive exceeds the byte limit") + if _scan(root, limits) != before or _git_state(root, limits) != original: + _fail("workspace_changed", "Workspace or Git state changed during capture") + with tarfile.open(archive, "r:") as tar: + _validated_manifest(tar, identity, root, limits) + digest = _file_digest(archive) + size = archive.stat().st_size + # A hard link publishes the completed file without replacing a racing + # writer's destination. The temporary name is removed by its owned scope. + archive.chmod(0o600) + os.link(archive, destination) + return WorkspaceArchive(digest, size, original["head"], original["branch"], len(records)) + + +def _validated_manifest( + tar: tarfile.TarFile, identity: CheckpointIdentity, root: Path, limits: WorkspaceLimits +) -> tuple[dict, dict[str, tarfile.TarInfo]]: + members = {} + for member in tar: + name = member.name.rstrip("/") if member.isdir() else member.name + if ( + name in members + or len(members) >= limits.max_entries + len(_ADMIN) + or member.size < 0 + or member.size > limits.max_bytes + or member.mode < 0 + or member.mode > _PERMISSION_BITS + or member.sparse is not None + or set(member.pax_headers) - {"path", "linkpath"} + ): + _fail("invalid_archive", "Workspace archive has duplicate or unsupported entries") + if member.name in _ADMIN: + if not member.isfile(): + _fail("invalid_archive", "Workspace archive metadata is not a regular file") + elif member.name.startswith("files/"): + _path( + member.name.removeprefix("files/").removesuffix("/") + if member.isdir() + else member.name.removeprefix("files/") + ) + if member.type not in {tarfile.REGTYPE, tarfile.DIRTYPE, tarfile.SYMTYPE}: + _fail("invalid_archive", "Workspace archive file type is unsupported") + else: + _fail("invalid_archive", "Workspace archive has an unknown entry") + members[name] = member + if not _ADMIN.issubset(members) or members["manifest.json"].size > 32 * 1024 * 1024: + _fail("invalid_archive", "Workspace archive manifest is unavailable") + stream = tar.extractfile(members["manifest.json"]) + if stream is None: + _fail("invalid_archive", "Workspace archive manifest has no data") + with stream: + manifest = json.load(stream) + if ( + not isinstance(manifest, dict) + or set(manifest) + != {"version", "identity", "workspace", "git", "files", "bundle_sha256", "patch_sha256"} + or type(manifest["version"]) is not int + or manifest["version"] != 1 + or manifest["identity"] != asdict(identity) + or manifest["workspace"] != str(root) + or not isinstance(manifest["files"], list) + or len(manifest["files"]) > limits.max_entries + ): + _fail("invalid_archive", "Workspace checkpoint identity, path or version differs") + state = manifest["git"] + if ( + not isinstance(state, dict) + or set(state) != {"head", "branch", "refs", "index_sha256", "exclude_b64"} + or not isinstance(state["head"], str) + or not _OID.fullmatch(state["head"]) + or (state["branch"] is not None and not isinstance(state["branch"], str)) + or not isinstance(state["refs"], dict) + or not isinstance(state["index_sha256"], str) + or not _SHA256.fullmatch(state["index_sha256"]) + ): + _fail("invalid_archive", "Workspace Git metadata is invalid") + exclude = state["exclude_b64"] + if exclude is not None: + if not isinstance(exclude, str) or len(exclude) > 4 * (_MAX_EXCLUDE_BYTES // 3 + 1): + _fail("invalid_archive", "Workspace Git ignore data is invalid") + if len(base64.b64decode(exclude, validate=True)) > _MAX_EXCLUDE_BYTES: + _fail("invalid_archive", "Workspace Git ignore data exceeds its limit") + paths = {} + folded = set() + for record in manifest["files"]: + if not isinstance(record, dict): + _fail("invalid_archive", "Workspace file manifest is invalid") + path = str(_path(record.get("path"))) + if path.casefold() in folded: + _fail("invalid_archive", "Workspace file paths collide") + folded.add(path.casefold()) + kind = record.get("kind") + fields = {"path", "mode", "kind"} | ( + {"sha256", "size"} if kind == "file" else {"target"} if kind == "symlink" else set() + ) + member = members.get("files/" + path) + if ( + set(record) != fields + or kind not in {"file", "directory", "symlink"} + or member is None + or type(record["mode"]) is not int + or member.mode != record["mode"] + or member.type + != {"file": tarfile.REGTYPE, "directory": tarfile.DIRTYPE, "symlink": tarfile.SYMTYPE}[ + kind + ] + ): + _fail("invalid_archive", "Workspace file manifest disagrees with its archive") + if kind == "file": + if type(record["size"]) is not int or member.size != record["size"]: + _fail("invalid_archive", "Workspace file size disagrees with its archive") + _verify_member(tar, member, record["sha256"]) + elif kind == "symlink": + if ( + not isinstance(record["target"], str) + or not record["target"] + or "\0" in record["target"] + or len(record["target"].encode()) > _MAX_PATH_BYTES + or member.linkname != record["target"] + or member.size != 0 + ): + _fail("invalid_archive", "Workspace symlink metadata is invalid") + elif member.size != 0: + _fail("invalid_archive", "Workspace directory contains data") + paths[path] = kind + if set(members) != _ADMIN | {"files/" + path for path in paths}: + _fail("invalid_archive", "Workspace archive has unlisted files") + for path in paths: + for parent in PurePosixPath(path).parents: + if str(parent) != "." and paths.get(str(parent)) != "directory": + _fail( + "invalid_archive", "Workspace archive traverses a symlink or missing directory" + ) + for name, key in (("git.bundle", "bundle_sha256"), ("index.patch", "patch_sha256")): + _verify_member(tar, members[name], manifest[key]) + return manifest, members + + +def _verify_member(tar: tarfile.TarFile, member: tarfile.TarInfo, digest: str) -> None: + if not isinstance(digest, str) or not _SHA256.fullmatch(digest): + _fail("invalid_archive", "Workspace file checksum is invalid") + stream = tar.extractfile(member) + if stream is None: + _fail("invalid_archive", "Workspace archive file has no data") + with stream: + actual = hashlib.sha256() + while block := stream.read(_CHUNK): + actual.update(block) + if actual.hexdigest() != digest: + _fail("checksum_mismatch", "Workspace file checksum did not match") + + +def restore_workspace( + archive: Path, + workspace: Path, + identity: CheckpointIdentity, + *, + expected_sha256: str, + limits: WorkspaceLimits = _DEFAULT_LIMITS, +) -> WorkspaceArchive: + """Restore only into a nonexistent stable path; validate before publication.""" + try: + return _restore_workspace(archive, workspace, identity, expected_sha256, limits) + except WorkspaceCheckpointError: + raise + except ( + OSError, + ValueError, + TypeError, + RecursionError, + subprocess.SubprocessError, + tarfile.TarError, + ) as exc: + raise WorkspaceCheckpointError( + "restore_failed", "Workspace restoration failed; continuation cannot start" + ) from exc + + +def _restore_workspace( + archive: Path, + workspace: Path, + identity: CheckpointIdentity, + expected_sha256: str, + limits: WorkspaceLimits, +) -> WorkspaceArchive: + root = workspace.absolute() + if root.parent.resolve() != root.parent or os.path.lexists(root): + _fail("destination_exists", "Workspace restore requires a new canonical destination") + _repository_identity(identity) + size = archive.stat().st_size + if not 0 < size <= limits.max_bytes: + _fail("size_limit", "Workspace archive exceeds the byte limit") + if not isinstance(expected_sha256, str) or not _SHA256.fullmatch(expected_sha256): + _fail("checksum_mismatch", "Workspace archive receipt checksum is invalid") + # The only tree removed on error is this newly-created owned staging tree. + # Never use extractall: paths/types/parents are validated, and links are leaves. + fd = os.open(archive, os.O_RDONLY | os.O_NOFOLLOW | os.O_NONBLOCK) + with ( + os.fdopen(fd, "rb") as archive_stream, + tempfile.TemporaryDirectory(prefix=".workspace-restore-", dir=root.parent) as scratch, + ): + archive_stamp = _stamp(os.fstat(archive_stream.fileno())) + if not stat.S_ISREG(archive_stamp[0]) or archive_stamp[3] != size: + _fail("invalid_archive", "Workspace archive is not a stable regular file") + if hashlib.file_digest(archive_stream, "sha256").hexdigest() != expected_sha256: + _fail("checksum_mismatch", "Workspace archive differs from its receipt") + archive_stream.seek(0) + staging = Path(scratch) / "tree" + staging.mkdir(mode=0o700) + with tarfile.open(fileobj=archive_stream, mode="r:") as tar: + manifest, members = _validated_manifest(tar, identity, root, limits) + for name in ("git.bundle", "index.patch"): + source = tar.extractfile(members[name]) + if source is None: + _fail("invalid_archive", "Workspace Git archive has no data") + with source, (Path(scratch) / name).open("xb") as output: + shutil.copyfileobj(source, output, _CHUNK) + records = manifest["files"] + for record in sorted( + records, key=lambda record: len(PurePosixPath(record["path"]).parts) + ): + target = staging / record["path"] + if record["kind"] == "directory": + target.mkdir(mode=0o700) + elif record["kind"] == "symlink": + os.symlink(record["target"], target) + else: + source = tar.extractfile(members["files/" + record["path"]]) + if source is None: + _fail("invalid_archive", "Workspace archive file has no data") + with source, target.open("xb") as output: + shutil.copyfileobj(source, output, _CHUNK) + target.chmod(record["mode"]) + state = manifest["git"] + _git(staging, ["init", "--template=", "--quiet"], limits) + for ref, oid in state["refs"].items(): + if ( + not isinstance(ref, str) + or not ref.startswith("refs/") + or ref.startswith("refs/replace/") + or not isinstance(oid, str) + or not _OID.fullmatch(oid) + ): + _fail("invalid_archive", "Workspace saved Git ref is invalid") + _git(staging, ["check-ref-format", ref], limits) + bundle = str(Path(scratch) / "git.bundle") + _git(staging, ["bundle", "verify", bundle], limits) + advertised = _git(staging, ["bundle", "unbundle", bundle], limits) + bundle_refs = dict(line.decode().split(" ", 1)[::-1] for line in advertised.splitlines()) + if bundle_refs != {**state["refs"], "HEAD": state["head"]}: + _fail("invalid_archive", "Workspace Git bundle refs disagree with the manifest") + ref_updates = Path(scratch) / "refs.txt" + ref_updates.write_text( + "".join(f"update {ref} {oid}\n" for ref, oid in state["refs"].items()) + ) + with ref_updates.open("rb") as source: + _git(staging, ["update-ref", "--stdin"], limits, source=source) + branch = state["branch"] + if branch is None: + _git(staging, ["update-ref", "--no-deref", "HEAD", state["head"]], limits) + else: + _git(staging, ["check-ref-format", "refs/heads/" + branch], limits) + if state["refs"].get("refs/heads/" + branch) != state["head"]: + _fail("invalid_archive", "Workspace branch disagrees with saved HEAD") + _git(staging, ["symbolic-ref", "HEAD", "refs/heads/" + branch], limits) + _git(staging, ["read-tree", "HEAD"], limits) + patch = Path(scratch) / "index.patch" + if patch.stat().st_size: + _git( + staging, + ["apply", "--cached", "--binary", "--whitespace=nowarn", str(patch)], + limits, + ) + index = _git(staging, ["ls-files", "--stage", "-z"], limits) + if hashlib.sha256(index).hexdigest() != state["index_sha256"]: + _fail("invalid_archive", "Workspace staged state was not restored exactly") + _git( + staging, ["remote", "add", "origin", f"https://github.com/{identity.repo}.git"], limits + ) + _git(staging, ["config", "--local", "credential.helper", "!gh auth git-credential"], limits) + if state["exclude_b64"] is not None: + ignore_file = staging / ".git/info/exclude" + ignore_file.parent.mkdir(exist_ok=True) + ignore_file.write_bytes(base64.b64decode(state["exclude_b64"], validate=True)) + for record in sorted( + records, key=lambda record: len(PurePosixPath(record["path"]).parts), reverse=True + ): + if record["kind"] == "directory": + (staging / record["path"]).chmod(record["mode"]) + if _stamp(os.fstat(archive_stream.fileno())) != archive_stamp: + _fail("workspace_changed", "Workspace archive changed during restoration") + # Reserve the final directory exclusively, then move the owned contents. + # A partial move is rolled back; an existing destination is never cleared. + root.mkdir(mode=0o700) + try: + for entry in staging.iterdir(): + os.rename(entry, root / entry.name) + except BaseException: + shutil.rmtree(root) + raise + return WorkspaceArchive(expected_sha256, size, state["head"], branch, len(records)) diff --git a/agent/tests/test_continuation_sdk_probe.py b/agent/tests/test_continuation_sdk_probe.py index 4723e3adb..dfae7b637 100644 --- a/agent/tests/test_continuation_sdk_probe.py +++ b/agent/tests/test_continuation_sdk_probe.py @@ -30,6 +30,7 @@ sys.path.insert(0, str(AGENT_ROOT / "src")) from continuation_session import CheckpointIdentity, CheckpointSessionStore, decode_checkpoint +from continuation_workspace import capture_workspace, restore_workspace from scripts.verify_microvm_credentials import MODEL, event_frame, model_response IDENTITY = CheckpointIdentity("sdk-probe", "attempt-1", "request-1", "owner", "probe/owned") @@ -200,6 +201,31 @@ def test_sdk_resumes_from_checkpoint_without_original_config(tmp_path, decision) workspace.mkdir() target = workspace / "owned.txt" target.write_text("OWNED_CONTINUATION_MARKER\n") + git_env = {key: value for key, value in os.environ.items() if not key.startswith("GIT_")} + git_env.update(GIT_CONFIG_GLOBAL=os.devnull, GIT_CONFIG_NOSYSTEM="1") + for command in ( + ["init", "--quiet", "--template=", "--initial-branch=work/probe"], + ["add", "owned.txt"], + [ + "-c", + "user.name=Fixture", + "-c", + "user.email=fixture@example.invalid", + "commit", + "--quiet", + "-m", + "owned", + ], + ): + subprocess.run( + ["git", "-c", f"core.hooksPath={os.devnull}", *command], + cwd=workspace, + env=git_env, + check=True, + capture_output=True, + timeout=10, + ) + (workspace / "untracked.txt").write_text("UNTRACKED_CONTINUATION_MARKER\n") for phase in ("original", "restored"): (tmp_path / f"{phase}-config").mkdir() auth_sentinel = "SYNTHETIC_AUTH_FILE_MUST_NOT_BE_COPIED" @@ -274,14 +300,17 @@ def start(name): body = (tmp_path / "checkpoint.json").read_bytes() saved = decode_checkpoint(body, IDENTITY) assert auth_sentinel.encode() not in body + workspace_archive = tmp_path / "workspace.tar" + workspace_receipt = capture_workspace(workspace, workspace_archive, IDENTITY) os.killpg(first.pid, signal.SIGKILL) assert first.wait(timeout=10) == -signal.SIGKILL original_audit = json.loads((tmp_path / "original-audit.json").read_text()) assert not any(event["kind"] == "post" for event in original_audit) shutil.rmtree(tmp_path / "original-config") - shutil.copytree(workspace, tmp_path / "workspace-copy") shutil.rmtree(workspace) - shutil.copytree(tmp_path / "workspace-copy", workspace) + restore_workspace( + workspace_archive, workspace, IDENTITY, expected_sha256=workspace_receipt.sha256 + ) phase[0] = "restored" second = start("restored") assert second.wait(timeout=35) == 0, (tmp_path / "restored-process.log").read_text() @@ -292,6 +321,7 @@ def start(name): result = next(event for event in audit if event["kind"] == "result") assert not result["is_error"] and result["session_id"] == saved["session_id"] assert target.read_text() == "OWNED_CONTINUATION_MARKER\n" + assert (workspace / "untracked.txt").read_text() == "UNTRACKED_CONTINUATION_MARKER\n" restored_requests = [request for request in requests if request["phase"] == "restored"] assert any( saved["action"]["tool_use_id"] in json.dumps(r["body"]) for r in restored_requests @@ -304,6 +334,10 @@ def start(name): "original_config_deleted": True, "pending_action_acknowledged": True, "auth_file_excluded": True, + "workspace_archive_sha256": workspace_receipt.sha256, + "workspace_head": workspace_receipt.head, + "workspace_restored_from_archive": True, + "untracked_file_preserved": True, "restored_posts": posts, "only_synthetic_loopback": True, } diff --git a/agent/tests/test_continuation_workspace.py b/agent/tests/test_continuation_workspace.py new file mode 100644 index 000000000..4ec9a8879 --- /dev/null +++ b/agent/tests/test_continuation_workspace.py @@ -0,0 +1,480 @@ +# Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. +# SPDX-License-Identifier: MIT-0 + +"""Real offline Git/working-tree recovery and hostile archive boundary checks.""" + +from __future__ import annotations + +import hashlib +import io +import json +import os +import stat +import subprocess +import tarfile +import time +from dataclasses import replace +from pathlib import Path + +import pytest + +import continuation_workspace as workspace +from continuation_session import CheckpointIdentity + +IDENTITY = CheckpointIdentity("task", "attempt", "approval", "owner", "example/repository") + + +def git(root, *args): + env = {key: value for key, value in os.environ.items() if not key.startswith("GIT_")} + env.update( + GIT_CONFIG_NOSYSTEM="1", + GIT_CONFIG_GLOBAL=os.devnull, + GIT_AUTHOR_NAME="Fixture", + GIT_AUTHOR_EMAIL="fixture@example.invalid", + GIT_COMMITTER_NAME="Fixture", + GIT_COMMITTER_EMAIL="fixture@example.invalid", + LC_ALL="C", + ) + return subprocess.check_output( + ["git", "-c", f"core.hooksPath={os.devnull}", *args], + cwd=root, + env=env, + stderr=subprocess.PIPE, + timeout=10, + ) + + +@pytest.fixture +def repo(tmp_path): + root = tmp_path / "repo" + root.mkdir() + git(root, "init", "--quiet", "--template=", "--initial-branch=work/task") + (root / "tracked.txt").write_text("committed\n") + (root / "binary.dat").write_bytes(b"\0original\xff") + (root / "deleted.txt").write_text("delete from working tree") + (root / "renamed.txt").write_text("rename in index") + (root / "run.sh").write_text("#!/bin/sh\nexit 0\n") + (root / ".gitignore").write_text("ignored/\n") + git(root, "add", ".") + git(root, "commit", "--quiet", "-m", "base") + (root / "local-commit.txt").write_text("unpushed work") + git(root, "add", ".") + git(root, "commit", "--quiet", "-m", "unpushed commit") + git(root, "tag", "local-tag") + git(root, "branch", "another-branch") + git(root, "remote", "add", "origin", "https://github.com/example/repository.git") + (root / "tracked.txt").write_text("staged\n") + (root / "binary.dat").write_bytes(b"\0staged\xfe") + git(root, "add", "tracked.txt", "binary.dat") + git(root, "mv", "renamed.txt", "renamed-new.txt") + (root / "tracked.txt").write_text("unstaged\n") + (root / "binary.dat").write_bytes(b"\0unstaged\xfd") + (root / "deleted.txt").unlink() + (root / "run.sh").chmod(0o755) + (root / "untracked 雪\nfile.txt").write_text("untracked") + (root / "ignored").mkdir() + (root / "ignored/important.bin").write_bytes(b"\0ignored but required") + (root / ".git/info").mkdir(exist_ok=True) + (root / ".git/info/exclude").write_text("local-ignored.txt\n") + (root / "local-ignored.txt").write_text("local ignore rules must survive") + (root / "empty").mkdir() + os.symlink("tracked.txt", root / "relative-link") + git(root, "config", "--local", "http.extraHeader", "SYNTHETIC_AUTH_MUST_NOT_BE_COPIED") + hooks = root / ".git/hooks" + hooks.mkdir(exist_ok=True) + marker = tmp_path / "hook-ran" + hook = hooks / "post-checkout" + hook.write_text(f"#!/bin/sh\ntouch '{marker}'\n") + hook.chmod(0o755) + return root + + +def saved(repo): + archive = repo.parent / "workspace.tar" + receipt = workspace.capture_workspace(repo, archive, IDENTITY) + return archive, receipt + + +def remove_original(repo): + # Preserve the original owned test tree for comparison without leaving it at + # the stable path that the replacement worker must use. + original = repo.with_name("original") + repo.rename(original) + return original + + +def inventory(root): + result = {} + for directory, dirs, files in os.walk(root, followlinks=False): + if Path(directory) == root: + dirs.remove(".git") + for name in dirs + files: + path = Path(directory) / name + info = path.lstat() + value = ( + os.readlink(path) + if path.is_symlink() + else path.read_bytes() + if path.is_file() + else None + ) + result[str(path.relative_to(root))] = ( + stat.S_IFMT(info.st_mode), + info.st_mode & 0o777, + value, + ) + return result + + +def rewrite(archive, *, mutate_manifest=None, mutate_member=None, extras=()): + records = [] + with tarfile.open(archive, "r:") as tar: + for member in tar: + stream = tar.extractfile(member) if member.isfile() else None + data = stream.read() if stream else None + if stream: + stream.close() + if member.name == "manifest.json" and mutate_manifest: + assert data is not None + manifest = json.loads(data) + mutate_manifest(manifest) + data = json.dumps(manifest).encode() + member.size = len(data) + if mutate_member: + member, data = mutate_member(member, data) + records.append((member, data)) + with tarfile.open(archive, "w", format=tarfile.PAX_FORMAT) as tar: + for member, data in [*records, *extras]: + tar.addfile(member, io.BytesIO(data) if data is not None else None) + return hashlib.sha256(archive.read_bytes()).hexdigest() + + +class TestWorkspaceRoundTrip: + def test_preserves_commits_staged_unstaged_binary_deleted_untracked_ignored_and_modes( + self, repo + ): + expected_files = inventory(repo) + expected_index = git(repo, "ls-files", "--stage", "-z") + expected_diff = git(repo, "diff", "--binary", "--no-ext-diff", "--no-textconv") + expected_refs = git(repo, "for-each-ref", "--format=%(objectname) %(refname)") + expected_ignored = git( + repo, "ls-files", "--others", "--ignored", "--exclude-standard", "-z" + ) + archive, receipt = saved(repo) + assert receipt.head == git(repo, "rev-parse", "HEAD").decode().strip() + assert receipt.branch == "work/task" + assert archive.stat().st_mode & 0o777 == 0o600 + assert b"SYNTHETIC_AUTH_MUST_NOT_BE_COPIED" not in archive.read_bytes() + original = remove_original(repo) + restored = workspace.restore_workspace( + archive, repo, IDENTITY, expected_sha256=receipt.sha256 + ) + assert restored == receipt + assert inventory(repo) == expected_files + assert git(repo, "ls-files", "--stage", "-z") == expected_index + assert git(repo, "diff", "--binary", "--no-ext-diff", "--no-textconv") == expected_diff + assert git(repo, "for-each-ref", "--format=%(objectname) %(refname)") == expected_refs + assert ( + git(repo, "ls-files", "--others", "--ignored", "--exclude-standard", "-z") + == expected_ignored + ) + assert git(repo, "log", "--format=%s").splitlines() == [b"unpushed commit", b"base"] + assert ( + git(repo, "remote", "get-url", "origin").strip() + == b"https://github.com/example/repository.git" + ) + assert ( + git(repo, "config", "--local", "credential.helper").strip() + == b"!gh auth git-credential" + ) + assert "SYNTHETIC_AUTH" not in (repo / ".git/config").read_text() + assert not (repo / ".git/hooks/post-checkout").exists() + assert not (repo.parent / "hook-ran").exists() + assert inventory(original) == expected_files + assert not list(repo.parent.glob(".workspace-*")) + + def test_detached_head_is_preserved(self, repo): + git(repo, "checkout", "--detach") + archive, receipt = saved(repo) + remove_original(repo) + workspace.restore_workspace(archive, repo, IDENTITY, expected_sha256=receipt.sha256) + assert receipt.branch is None + assert git(repo, "rev-parse", "--abbrev-ref", "HEAD").strip() == b"HEAD" + + def test_external_symlink_is_saved_as_a_leaf_without_reading_its_target(self, repo): + secret = repo.parent / "outside" + secret.write_text("SYNTHETIC_OUTSIDE_DATA_NOT_IN_ARCHIVE") + os.symlink(secret, repo / "outside-link") + archive, receipt = saved(repo) + assert secret.read_bytes() not in archive.read_bytes() + remove_original(repo) + workspace.restore_workspace(archive, repo, IDENTITY, expected_sha256=receipt.sha256) + assert os.readlink(repo / "outside-link") == str(secret) + assert secret.read_text() == "SYNTHETIC_OUTSIDE_DATA_NOT_IN_ARCHIVE" + + def test_external_diff_configuration_does_not_execute_during_capture(self, repo): + script = repo.parent / "external-diff" + marker = repo.parent / "diff-ran" + script.write_text(f"#!/bin/sh\ntouch '{marker}'\nexit 1\n") + script.chmod(0o755) + git(repo, "config", "diff.external", str(script)) + saved(repo) + assert not marker.exists() + + +class TestCaptureFailures: + def test_git_deadline_stops_the_process_group(self, repo): + git(repo, "config", "alias.checkpoint-wait", "!sleep 10") + started = time.monotonic() + with pytest.raises(workspace.WorkspaceCheckpointError) as error: + workspace._git(repo, ["checkpoint-wait"], workspace.WorkspaceLimits(git_timeout_s=1)) + assert error.value.code == "git_timeout" + assert time.monotonic() - started < 5 + + @pytest.mark.parametrize( + "repo_name", ["", "../repo", "owner/..", "https://github.com/owner/repo"] + ) + def test_repository_identity_must_also_be_restorable(self, repo, repo_name): + destination = repo.parent / "workspace.tar" + with pytest.raises(workspace.WorkspaceCheckpointError) as error: + workspace.capture_workspace(repo, destination, replace(IDENTITY, repo=repo_name)) + assert error.value.code == "invalid_identity" + assert not destination.exists() + + def test_submodule_index_entry_fails_without_fetching_it(self, repo): + head = git(repo, "rev-parse", "HEAD").decode().strip() + git(repo, "update-index", "--add", "--cacheinfo", f"160000,{head},submodule") + with pytest.raises(workspace.WorkspaceCheckpointError) as error: + saved(repo) + assert error.value.code == "unsupported_git" + + @pytest.mark.parametrize( + "limits,code", + [ + (workspace.WorkspaceLimits(max_bytes=32), "size_limit"), + (workspace.WorkspaceLimits(max_entries=1), "entry_limit"), + ], + ) + def test_limits_do_not_publish_partial_archive(self, repo, limits, code): + expected = inventory(repo) + destination = repo.parent / "limited.tar" + with pytest.raises(workspace.WorkspaceCheckpointError) as error: + workspace.capture_workspace(repo, destination, IDENTITY, limits=limits) + assert error.value.code == code + assert not destination.exists() + assert inventory(repo) == expected + assert not list(repo.parent.glob(".workspace-capture-*")) + + @pytest.mark.parametrize( + "path", ["MERGE_HEAD", "index.lock", "shallow", "objects/info/alternates"] + ) + def test_active_or_nonportable_git_layout_fails(self, repo, path): + target = repo / ".git" / path + target.parent.mkdir(exist_ok=True, parents=True) + target.touch() + with pytest.raises(workspace.WorkspaceCheckpointError) as error: + saved(repo) + assert error.value.code == "unsupported_git" + + @pytest.mark.parametrize("flag", ["--assume-unchanged", "--skip-worktree", "intent-to-add"]) + def test_nonportable_index_flags_fail_instead_of_losing_state(self, repo, flag): + if flag == "intent-to-add": + (repo / "intent.txt").write_text("not staged yet") + git(repo, "add", "-N", "intent.txt") + else: + git(repo, "update-index", flag, "tracked.txt") + with pytest.raises(workspace.WorkspaceCheckpointError) as error: + saved(repo) + assert error.value.code == "unsupported_git" + + @pytest.mark.parametrize("kind", ["fifo", "hardlink", "nested-git"]) + def test_special_files_are_rejected(self, repo, kind): + target = repo / "unsupported" + if kind == "fifo": + os.mkfifo(target) + elif kind == "hardlink": + os.link(repo / "tracked.txt", target) + else: + target.mkdir() + (target / ".git").mkdir() + with pytest.raises(workspace.WorkspaceCheckpointError): + saved(repo) + assert not (repo.parent / "workspace.tar").exists() + + def test_existing_archive_is_not_overwritten(self, repo): + destination = repo.parent / "workspace.tar" + destination.write_bytes(b"keep") + with pytest.raises(workspace.WorkspaceCheckpointError) as error: + saved(repo) + assert error.value.code == "destination_exists" + assert destination.read_bytes() == b"keep" + + def test_archive_cannot_be_written_inside_the_workspace(self, repo): + with pytest.raises(workspace.WorkspaceCheckpointError) as error: + workspace.capture_workspace(repo, repo / "backup.tar", IDENTITY) + assert error.value.code == "invalid_path" + + def test_worktree_mutation_prevents_acknowledgement(self, repo, monkeypatch): + real_read = workspace._DigestReader.read + changed = False + + def mutate(reader, size=-1): + nonlocal changed + block = real_read(reader, size) + if not changed: + changed = True + (repo / "created-during-capture").write_text("concurrent writer") + return block + + monkeypatch.setattr(workspace._DigestReader, "read", mutate) + with pytest.raises(workspace.WorkspaceCheckpointError) as error: + saved(repo) + assert error.value.code == "workspace_changed" + assert not (repo.parent / "workspace.tar").exists() + + def test_racing_destination_is_not_replaced(self, repo, monkeypatch): + real_link = os.link + destination = repo.parent / "workspace.tar" + + def race(source, target): + destination.write_bytes(b"other writer") + real_link(source, target) + + monkeypatch.setattr(workspace.os, "link", race) + with pytest.raises(workspace.WorkspaceCheckpointError): + saved(repo) + assert destination.read_bytes() == b"other writer" + + +class TestRestoreBoundaries: + def test_existing_workspace_is_never_cleared(self, repo): + archive, receipt = saved(repo) + before = inventory(repo) + with pytest.raises(workspace.WorkspaceCheckpointError) as error: + workspace.restore_workspace(archive, repo, IDENTITY, expected_sha256=receipt.sha256) + assert error.value.code == "destination_exists" + assert inventory(repo) == before + + def test_receipt_checksum_is_required(self, repo): + archive, _ = saved(repo) + remove_original(repo) + with pytest.raises(workspace.WorkspaceCheckpointError) as error: + workspace.restore_workspace(archive, repo, IDENTITY, expected_sha256="0" * 64) + assert error.value.code == "checksum_mismatch" + assert not repo.exists() + + @pytest.mark.parametrize("field", ["task_id", "attempt_id", "request_id", "user_id", "repo"]) + def test_cross_identity_restore_is_rejected(self, repo, field): + archive, receipt = saved(repo) + remove_original(repo) + wrong = replace(IDENTITY, **{field: "other/repo" if field == "repo" else "other"}) + with pytest.raises(workspace.WorkspaceCheckpointError): + workspace.restore_workspace(archive, repo, wrong, expected_sha256=receipt.sha256) + assert not repo.exists() + + def test_workspace_path_must_remain_stable(self, repo): + archive, receipt = saved(repo) + target = repo.parent / "different-workspace" + with pytest.raises(workspace.WorkspaceCheckpointError): + workspace.restore_workspace(archive, target, IDENTITY, expected_sha256=receipt.sha256) + assert not target.exists() + + @pytest.mark.parametrize( + "change", + [ + lambda m: m.update(version=True), + lambda m: m.update(environment={"SYNTHETIC": "not allowed"}), + lambda m: m["git"].update(index_sha256="0" * 64), + lambda m: m["git"]["refs"].update({"refs/heads/../escape": "a" * 40}), + lambda m: m["git"].update(branch="different"), + lambda m: m["git"].update(exclude_b64="not valid base64"), + lambda m: m["files"][0].update(path="../escape"), + ], + ) + def test_invalid_manifest_does_not_publish_a_workspace(self, repo, change): + archive, _ = saved(repo) + remove_original(repo) + digest = rewrite(archive, mutate_manifest=change) + with pytest.raises(workspace.WorkspaceCheckpointError): + workspace.restore_workspace(archive, repo, IDENTITY, expected_sha256=digest) + assert not repo.exists() + assert not list(repo.parent.glob(".workspace-restore-*")) + + def test_changed_file_bytes_fail_inner_checksum_even_with_valid_outer_receipt(self, repo): + archive, _ = saved(repo) + remove_original(repo) + + def corrupt(member, data): + if member.name == "files/tracked.txt": + data = b"x" * len(data) + return member, data + + digest = rewrite(archive, mutate_member=corrupt) + with pytest.raises(workspace.WorkspaceCheckpointError) as error: + workspace.restore_workspace(archive, repo, IDENTITY, expected_sha256=digest) + assert error.value.code == "checksum_mismatch" + assert not repo.exists() + + @pytest.mark.parametrize( + "name,kind", + [ + ("files/../escape", tarfile.REGTYPE), + ("/absolute", tarfile.REGTYPE), + ("files/.git/config", tarfile.REGTYPE), + ("files/unlisted", tarfile.REGTYPE), + ("files/empty/", tarfile.DIRTYPE), + ("files/link", tarfile.LNKTYPE), + ], + ) + def test_extra_traversal_duplicate_and_hardlink_members_are_rejected(self, repo, name, kind): + archive, _ = saved(repo) + remove_original(repo) + extra = tarfile.TarInfo(name) + extra.type = kind + extra.linkname = "files/tracked.txt" if kind == tarfile.LNKTYPE else "" + digest = rewrite(archive, extras=[(extra, b"" if kind == tarfile.REGTYPE else None)]) + with pytest.raises(workspace.WorkspaceCheckpointError): + workspace.restore_workspace(archive, repo, IDENTITY, expected_sha256=digest) + assert not repo.exists() + assert not (repo.parent / "escape").exists() + + def test_archive_cannot_write_through_a_symlink(self, repo): + archive, _ = saved(repo) + remove_original(repo) + extra = tarfile.TarInfo("files/relative-link/escape") + extra.mode = 0o644 + extra.size = 1 + + def add_child(manifest): + manifest["files"].append( + { + "path": "relative-link/escape", + "kind": "file", + "mode": 0o644, + "size": 1, + "sha256": hashlib.sha256(b"x").hexdigest(), + } + ) + + digest = rewrite(archive, mutate_manifest=add_child, extras=[(extra, b"x")]) + with pytest.raises(workspace.WorkspaceCheckpointError) as error: + workspace.restore_workspace(archive, repo, IDENTITY, expected_sha256=digest) + assert error.value.code == "invalid_archive" + assert not repo.exists() + + def test_archive_change_during_restore_does_not_publish(self, repo, monkeypatch): + archive, receipt = saved(repo) + remove_original(repo) + original_git = workspace._git + + def change(root, args, *rest, **kwargs): + if args[0] == "init": + with archive.open("ab") as stream: + stream.write(b"changed") + return original_git(root, args, *rest, **kwargs) + + monkeypatch.setattr(workspace, "_git", change) + with pytest.raises(workspace.WorkspaceCheckpointError) as error: + workspace.restore_workspace(archive, repo, IDENTITY, expected_sha256=receipt.sha256) + assert error.value.code == "workspace_changed" + assert not repo.exists() diff --git a/docs/verification/645-p3-implementation-plan.md b/docs/verification/645-p3-implementation-plan.md index 06592e166..13e39cf21 100644 --- a/docs/verification/645-p3-implementation-plan.md +++ b/docs/verification/645-p3-implementation-plan.md @@ -372,8 +372,16 @@ do not cover: The public SDK store restores approve/deny in a new process after removing the old configuration. Immutable S3 version reads and task-scoped access passed nine live checks; all temporary resources were removed. Full agent quality passed - 2,004 tests. Workspace preservation and production worker-replacement wiring - remain required before changing approval deadlines or releasing a waiting worker. + 2,004 tests at that milestone. Workspace storage and production + worker-replacement wiring remain required before changing approval deadlines + or releasing a waiting worker. +- [x] Add the [local workspace archive](./645-p3-workspace-recovery-20260917.md): + preserve local commits, staged/unstaged edits and untracked/ignored files, + validate restoration without original Git config or hooks, and connect it to + the actual SDK approve/deny recovery test. Local Git ignore rules also survive. + Final agent quality passes 2,054 tests, including 50 workspace cases and both + SDK decisions. Durable workspace upload, saved + workflow context and replacement cloud-worker integration remain open. - [x] Implement [actionable Slack/Linear notifications and CLI response instructions](./645-p3-approval-ux-20260917.md), including saved decisions and closure reasons. Unwrap the actual agent approval milestones and keep approval @@ -447,7 +455,9 @@ The following work is still required before changing the current deadline defaul that the work was saved. Test restoration of uncommitted and untracked files. `agent/src/continuation_session.py` now implements acknowledged conversation/ action storage with immutable S3 version receipts; nine live AWS permission - and persistence checks passed. Workspace preservation, production IAM/retention, + and persistence checks passed. `continuation_workspace.py` now preserves the + complete supported working tree and Git state in a bounded local archive. + Its durable upload/read-back, saved workflow context, production IAM/retention, lifecycle-barrier integration and conditional receipt publication remain open. 3. Separate the task from each worker attempt. Add an attempt generation to launch tokens, saved handles, writes and reservations so a replaced worker diff --git a/docs/verification/645-p3-session-recovery-20260917.md b/docs/verification/645-p3-session-recovery-20260917.md index af7935d41..ab71972a7 100644 --- a/docs/verification/645-p3-session-recovery-20260917.md +++ b/docs/verification/645-p3-session-recovery-20260917.md @@ -99,9 +99,11 @@ conversation is safe to log or publish. proposed action and writes that checkpoint. 2. The test kills the entire original process group, verifies that its tool did not execute, and deletes the original configuration directory. -3. It restores the owned workspace marker at the same path and starts a fresh - process with `resume`, the restored store and eager mirroring. This workspace - copy is a test fixture, not the production workspace checkpoint. +3. The initial store test copied an owned marker as its workspace fixture. The + [workspace follow-up](./645-p3-workspace-recovery-20260917.md) now deletes the + original workspace and restores its Git history, tracked marker and untracked + file using the new archive component. A fresh process starts at the same + path with `resume`, the restored store and eager mirroring. 4. The continuation prompt carries the full saved action and human decision. The model proposes `toolu_restored`; the fresh permission hook allows one `Read` for approval and denies it for denial. Both sessions complete with @@ -109,7 +111,7 @@ conversation is safe to log or publish. 5. A synthetic authentication-file sentinel is absent from the saved checkpoint. No original CLI configuration is used during restoration. -The final full agent quality run enabled both tests: +The initial conversation-component full agent quality run enabled both tests: ```bash ABCA_TEST_SDK_CONTINUATION=1 MISE_EXPERIMENTAL=1 mise run //agent:quality @@ -159,9 +161,10 @@ task-scoped permissions. The deterministic model demonstrates transport and hook behavior; it does not prove how a real model will interpret the continuation prompt. -Production `runner.py` does not yet install this store or resume from it. Still -required: workspace preservation including uncommitted/untracked files; recovery -on a replacement cloud worker; a stable workspace path; lifecycle barrier and +Production `runner.py` does not yet install this store or resume from it. The +workspace follow-up implements local file/Git preservation; its durable storage +and integration remain required. Other remaining work includes recovery on a +replacement cloud worker; enforcing the stable workspace path; lifecycle barrier and conditional publication of the acknowledged receipt; task-attempt fencing and reservation transfer; exactly-once decision consumption; cancellation while parked; lost launch replies; and request retention/deadline changes. A conversation diff --git a/docs/verification/645-p3-workspace-recovery-20260917.md b/docs/verification/645-p3-workspace-recovery-20260917.md new file mode 100644 index 000000000..5e9f12670 --- /dev/null +++ b/docs/verification/645-p3-workspace-recovery-20260917.md @@ -0,0 +1,134 @@ +# ADR-021 workspace recovery prerequisite + +`agent/src/continuation_workspace.py` adds offline capture and restoration of a +coding workspace. Together with the +[conversation checkpoint](./645-p3-session-recovery-20260917.md), it lets a new +local agent process recover its conversation and files. This is a prerequisite; +the production pipeline does not yet use either component for worker replacement. + +In plain language, a Git commit saves a named version of the code. The **index** +holds changes selected for the next commit. The **working tree** contains the +files currently on disk, including changes that have not been selected or +committed. Recovery must preserve all three, rather than merely clone the last +version uploaded to GitHub. + +## What the archive preserves + +- A Git bundle containing reachable history, local commits, refs and tags, + together with the current branch or detached HEAD. +- A binary staged patch and checksum of the original index entries, preserving + the difference between staged and unstaged changes. +- The repository-local `.git/info/exclude` rules, so its ignored files remain + ignored after restoration. +- Every regular working file, including untracked and ignored files, binary + contents, executable permissions, directories and leaf symlinks. +- Task, attempt, approval request, owner and repository identity; the original + absolute workspace path; and checksums for the archive and each data member. + +Ignored files are included because a Git ignore rule does not prove a file is +disposable. The default limit is **1 GiB including the archive**, with **100,000 +working-tree entries** and a **60-second limit per Git command**. An oversized +workspace fails capture and must keep its worker available; it is not silently +trimmed to fit. + +The original `.git` administration directory is not copied. Restoration rebuilds +it from the bundle and staged patch, sets a credential-free GitHub origin URL and +the `gh` credential helper, and verifies the restored index. Original Git hooks, +config, credential headers and global configuration are not restored. Platform +Git identity and commit-attribution hooks must be installed by the future +pipeline integration. This archive does not replace the separate saved +`RepoSetup`/workflow context needed by that integration. +The local ignore file is an explicit data-only exception; symlinks to an external +ignore file are rejected. Global ignore/filter configuration still belongs to +the fresh worker's trusted setup. + +The module does not read the home directory, copy authentication files from it, +or serialize environment variables. Repository contents can still contain +sensitive data, so the complete archive is private task data. A symlink is +preserved as a leaf, including an external target address, without reading the +target's contents. Archive extraction cannot write through a symlink. +External tools and home-directory caches are not captured. The future worker +setup must recreate required tool installations before using links to them. + +## Capture and restore boundaries + +The caller must hold the lifecycle barrier to stop agent/tool writes. Capture +also compares file metadata and Git state before and after writing. It opens +regular files through directory descriptors without following substituted +symlinks. A detected change prevents publication. + +Capture writes to an owned temporary directory, validates the archive, then +publishes its completed file without replacing an existing destination. Failures +remove the temporary capture data. The original workspace is not modified. + +Restore requires the expected archive checksum and identity and the original +stable workspace path, normally `/workspace/`. It validates member +checksums, paths, types, parent relationships, manifest and Git refs before +publishing a new workspace. It rebuilds the repository offline, with hooks and +external diff helpers disabled. A failed restoration removes only its newly +created staging data; an existing workspace is never cleared. + +The following currently fail with explicit feedback instead of losing state: + +- Linked Git worktrees, shallow/sparse repositories, alternate object stores, + replacement refs, submodules, merge conflicts and active Git operations. +- Intent-to-add, skip-worktree and assume-unchanged index flags. +- Hard-linked regular files, special files, mounted directories and special + permission bits; nested `.git` entries and case-colliding paths. +- Changed or corrupt archives, another task/request's identity, a different + workspace path, unlisted/duplicate members, path traversal and symlink parents. + +`WorkspaceCheckpointError.code` identifies stages such as `size_limit`, +`entry_limit`, `unsupported_git`, `workspace_changed`, `checksum_mismatch`, +`git_timeout` and `destination_exists`. Error messages omit file contents and +raw Git stderr. The future lifecycle integration must expose these failures +without claiming that the worker can be released. + +## Verification + +The unit tests use real offline Git repositories. They preserve two commits, +local branches/tags, staged and unstaged binary/text edits, a staged rename, +an unstaged deletion, ignored and untracked files, executable permissions and +symlinks. Failure injection covers limits, concurrent changes, existing +destinations, unsupported states and hostile archive members. + +The opt-in SDK test now uses this archive instead of copying the working +directory as a fixture: + +1. Start the pinned SDK/CLI and block its original `Read` permission hook. +2. Acknowledge the exact conversation action and capture the workspace. +3. Kill the original process group and delete its configuration and workspace. +4. Restore the files from the archive and the conversation from `SessionStore`. +5. Verify approval executes one new `Read`, denial executes none, the original + session ID is retained and the untracked marker survives. + +The model is deterministic and runs on loopback. No AWS resources, paid model +service or notification channel are used by these tests. + +The final source passed `ABCA_TEST_SDK_CONTINUATION=1 mise run //agent:quality`: +lint, formatting, types and **2,054 tests**, with **11 skipped** and **86.66%** +total coverage. This includes **50 workspace tests** and both composed SDK cases. +The skipped tests are the existing opt-in DynamoDB Local cases; the one warning +is Starlette's existing `httpx` test-client deprecation. + +The agent Bandit high-severity scan, staged secrets scan and silent-success scan +for changes since `f9ee0b49` passed. The full repository silent-success scan's +previously recorded findings are not claimed fixed by this patch. + +The source snapshot, final test logs, actual SDK model requests, hook audits, +workspace/conversation archives and checksums are retained under +`/Users/sphias/.local/share/abca-verification/645-p3-20260916/workspace-checkpoint`. + +## Remaining integration + +The archive is local only. It must still be uploaded under scoped permissions, +read back and pinned to an immutable object version. The existing conversation +S3 adapter accepts its JSON envelope and **does not upload this tar archive**. +Both receipts must be conditionally published together for the owning attempt +before any worker release. + +Production work still includes saved workflow/baseline context, fresh credential +setup, restoring without the destructive fresh-clone path, worker-attempt +fencing, capacity transfer and replacement cloud-worker acceptance. Approval +deadlines and normal automatic sleep are unchanged. See the +[P3 implementation order](./645-p3-implementation-plan.md#unanswered-approvals-implementation-order). From 63e6bcac3f1405b912bea093d17de647c5f96585 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Fri, 18 Sep 2026 08:55:58 -0400 Subject: [PATCH 075/149] feat(compute): complete P3 continuation and rollout integration Preserve the later continuation, approval, identity-vault, documentation and verification changes included in the original P3 publication snapshot after the restored development commits. Sanitize deployment identifiers for public review without changing the published final implementation. --- .dockerignore | 4 + agent/README.md | 7 +- agent/policies/soft_deny.cedar | 18 +- agent/src/config.py | 74 +- agent/src/continuation_runtime.py | 427 ++++++ agent/src/continuation_session.py | 9 +- agent/src/continuation_storage.py | 477 +++++++ agent/src/continuation_usage.py | 86 ++ agent/src/continuation_workspace.py | 24 +- agent/src/hooks.py | 106 +- agent/src/microvm_checkpoint.py | 13 +- agent/src/microvm_lifecycle.py | 130 +- agent/src/payload_bootstrap.py | 19 +- agent/src/pipeline.py | 108 +- agent/src/policy.py | 51 +- agent/src/progress_writer.py | 9 +- agent/src/repo.py | 14 + agent/src/runner.py | 55 +- agent/src/server.py | 81 +- agent/src/task_state.py | 329 ++++- agent/src/workflow/runner.py | 13 +- agent/tests/test_approval_retention.py | 37 + agent/tests/test_config.py | 40 +- agent/tests/test_continuation_runtime.py | 476 +++++++ agent/tests/test_continuation_sdk_probe.py | 28 +- agent/tests/test_continuation_storage.py | 283 ++++ agent/tests/test_continuation_usage.py | 80 ++ agent/tests/test_continuation_workspace.py | 4 +- agent/tests/test_hooks.py | 229 +++- agent/tests/test_microvm_checkpoint.py | 53 +- agent/tests/test_microvm_lifecycle.py | 96 ++ agent/tests/test_payload_bootstrap.py | 20 + agent/tests/test_policy_three_outcome.py | 13 +- agent/tests/test_poll_for_decision.py | 4 +- agent/tests/test_runner.py | 67 + agent/tests/test_server.py | 78 +- agent/tests/test_task_state.py | 117 +- agent/tests/test_workflow_runner.py | 67 + cdk/src/constructs/agent-session-role.ts | 11 +- .../agent-task-write-attributes.json | 1 + cdk/src/constructs/continuation-bucket.ts | 73 ++ cdk/src/constructs/lambda-microvm-compute.ts | 902 +++---------- cdk/src/constructs/lambda-microvm-stack.ts | 11 + cdk/src/constructs/linear-identity-vault.ts | 2 +- cdk/src/constructs/linear-integration.ts | 7 +- .../microvm-continuation-manager.ts | 99 ++ .../constructs/stranded-task-reconciler.ts | 17 +- cdk/src/constructs/task-api.ts | 25 +- cdk/src/constructs/task-approvals-table.ts | 5 +- cdk/src/constructs/task-orchestrator.ts | 25 +- cdk/src/constructs/tool-gateway.ts | 9 +- cdk/src/handlers/approve-task.ts | 4 +- cdk/src/handlers/deny-task.ts | 4 +- cdk/src/handlers/get-pending.ts | 3 +- cdk/src/handlers/linear-webhook-processor.ts | 19 +- cdk/src/handlers/linear-webhook.ts | 4 +- cdk/src/handlers/orchestrate-task.ts | 53 +- cdk/src/handlers/reconcile-concurrency.ts | 18 +- .../reconcile-microvm-continuations.ts | 167 +++ cdk/src/handlers/reconcile-stranded-tasks.ts | 43 +- cdk/src/handlers/shared/builtin-policies.ts | 18 +- .../handlers/shared/close-task-approvals.ts | 96 ++ cdk/src/handlers/shared/compute-strategy.ts | 2 + cdk/src/handlers/shared/create-task-core.ts | 9 +- .../handlers/shared/microvm-approval-wake.ts | 5 +- .../shared/microvm-continuation-dispatch.ts | 110 ++ .../shared/microvm-continuation-retirement.ts | 303 +++++ .../shared/microvm-continuation-runner.ts | 283 ++++ .../shared/microvm-continuation-start.ts | 207 +++ .../shared/microvm-continuation-storage.ts | 287 ++++ .../shared/microvm-continuation-timing.ts | 27 + .../shared/microvm-continuation-types.ts | 98 ++ .../shared/microvm-lifecycle-policy.ts | 2 +- cdk/src/handlers/shared/microvm-lifecycle.ts | 14 +- cdk/src/handlers/shared/microvm-start.ts | 64 +- cdk/src/handlers/shared/microvm-supervisor.ts | 9 +- cdk/src/handlers/shared/microvm-task-poll.ts | 106 ++ .../handlers/shared/microvm-worker-lease.ts | 93 ++ cdk/src/handlers/shared/orchestrator.ts | 6 + cdk/src/handlers/shared/payload-bootstrap.ts | 24 +- .../strategies/lambda-microvm-strategy.ts | 28 +- cdk/src/handlers/shared/task-concurrency.ts | 22 + cdk/src/handlers/shared/types.ts | 29 +- cdk/src/stacks/agent.ts | 97 +- .../constructs/agent-session-role.test.ts | 8 +- .../constructs/continuation-bucket.test.ts | 56 + .../constructs/lambda-microvm-compute.test.ts | 66 +- .../constructs/lambda-microvm-stack.test.ts | 33 + .../microvm-continuation-manager.test.ts | 64 + cdk/test/handlers/get-pending.test.ts | 4 +- cdk/test/handlers/get-policies.test.ts | 4 +- .../reconcile-microvm-continuations.test.ts | 142 ++ .../handlers/reconcile-stranded-tasks.test.ts | 6 +- .../shared/close-task-approvals.test.ts | 72 ++ .../microvm-continuation-dispatch.test.ts | 110 ++ .../microvm-continuation-retirement.test.ts | 195 +++ .../microvm-continuation-runner.test.ts | 186 +++ .../shared/microvm-continuation-start.test.ts | 145 +++ .../microvm-continuation-storage.test.ts | 176 +++ .../shared/microvm-worker-lease.test.ts | 83 ++ .../handlers/shared/payload-bootstrap.test.ts | 25 + .../lambda-microvm-strategy.test.ts | 23 +- .../handlers/shared/task-concurrency.test.ts | 25 + cdk/test/stacks/agent.test.ts | 128 +- .../stacks/microvm-managed-image-nag.test.ts | 29 +- cli/src/commands/pending.ts | 2 +- cli/src/commands/submit.ts | 6 +- cli/src/commands/watch.ts | 2 +- cli/src/types.ts | 7 +- contracts/constants.json | 16 +- contracts/constants.md | 26 +- .../ADR-016-pluggable-identity-and-auth.md | 2 +- ...ADR-021-lambda-microvms-compute-backend.md | 493 ++----- docs/design/CEDAR_HITL_GATES.md | 65 +- docs/design/COMPUTE.md | 23 +- docs/design/DEPLOYMENT_ROLES.md | 9 + docs/design/ORCHESTRATOR.md | 10 +- docs/guides/CEDAR_POLICY_GUIDE.md | 7 +- docs/guides/LINEAR_SETUP_GUIDE.md | 17 +- docs/guides/USER_GUIDE.md | 27 +- .../docs/architecture/Cedar-hitl-gates.md | 65 +- docs/src/content/docs/architecture/Compute.md | 23 +- .../docs/architecture/Deployment-roles.md | 9 + .../content/docs/architecture/Orchestrator.md | 10 +- .../docs/customizing/Cedar-policies.md | 7 +- .../Adr-016-pluggable-identity-and-auth.md | 2 +- ...Adr-021-lambda-microvms-compute-backend.md | 493 ++----- .../docs/using/Approval-gates-cedar-hitl.md | 25 +- .../content/docs/using/Linear-setup-guide.md | 17 +- docs/src/content/docs/using/Using-the-cli.md | 2 +- .../645-lambda-microvm-service-feedback.md | 11 +- .../645-p3-cloud-continuation-20260917.md | 129 ++ ...45-p3-connection-close-rollout-20260916.md | 2 +- .../645-p3-continuation-protocol-20260917.md | 77 ++ .../645-p3-durable-live-20260915.md | 2 +- .../645-p3-final-image-and-ecs-20260917.md | 6 +- .../645-p3-implementation-plan.md | 1149 +++-------------- .../645-p3-live-deployment-20260915.md | 2 +- docs/verification/645-p3-nested-stack.md | 102 +- .../645-p3-normal-closure-20260918.md | 187 +++ .../645-p3-normal-rollout-review-20260917.md | 14 +- .../645-p3-progress-race-20260918.md | 66 + scripts/check-constants-sync.ts | 25 +- 143 files changed, 8812 insertions(+), 3092 deletions(-) create mode 100644 agent/src/continuation_runtime.py create mode 100644 agent/src/continuation_storage.py create mode 100644 agent/src/continuation_usage.py create mode 100644 agent/tests/test_approval_retention.py create mode 100644 agent/tests/test_continuation_runtime.py create mode 100644 agent/tests/test_continuation_storage.py create mode 100644 agent/tests/test_continuation_usage.py create mode 100644 cdk/src/constructs/continuation-bucket.ts create mode 100644 cdk/src/constructs/microvm-continuation-manager.ts create mode 100644 cdk/src/handlers/reconcile-microvm-continuations.ts create mode 100644 cdk/src/handlers/shared/close-task-approvals.ts create mode 100644 cdk/src/handlers/shared/microvm-continuation-dispatch.ts create mode 100644 cdk/src/handlers/shared/microvm-continuation-retirement.ts create mode 100644 cdk/src/handlers/shared/microvm-continuation-runner.ts create mode 100644 cdk/src/handlers/shared/microvm-continuation-start.ts create mode 100644 cdk/src/handlers/shared/microvm-continuation-storage.ts create mode 100644 cdk/src/handlers/shared/microvm-continuation-timing.ts create mode 100644 cdk/src/handlers/shared/microvm-continuation-types.ts create mode 100644 cdk/src/handlers/shared/microvm-task-poll.ts create mode 100644 cdk/src/handlers/shared/microvm-worker-lease.ts create mode 100644 cdk/test/constructs/continuation-bucket.test.ts create mode 100644 cdk/test/constructs/microvm-continuation-manager.test.ts create mode 100644 cdk/test/handlers/reconcile-microvm-continuations.test.ts create mode 100644 cdk/test/handlers/shared/close-task-approvals.test.ts create mode 100644 cdk/test/handlers/shared/microvm-continuation-dispatch.test.ts create mode 100644 cdk/test/handlers/shared/microvm-continuation-retirement.test.ts create mode 100644 cdk/test/handlers/shared/microvm-continuation-runner.test.ts create mode 100644 cdk/test/handlers/shared/microvm-continuation-start.test.ts create mode 100644 cdk/test/handlers/shared/microvm-continuation-storage.test.ts create mode 100644 cdk/test/handlers/shared/microvm-worker-lease.test.ts create mode 100644 docs/verification/645-p3-cloud-continuation-20260917.md create mode 100644 docs/verification/645-p3-continuation-protocol-20260917.md create mode 100644 docs/verification/645-p3-normal-closure-20260918.md create mode 100644 docs/verification/645-p3-progress-race-20260918.md diff --git a/.dockerignore b/.dockerignore index 32526d373..eb755aa97 100644 --- a/.dockerignore +++ b/.dockerignore @@ -47,6 +47,10 @@ abca-worktrees/ # Test/coverage output coverage/ **/coverage/ +.jest-cache/ +**/.jest-cache/ +test-reports/ +**/test-reports/ .pytest_cache/ **/.pytest_cache/ # Coverage data FILES, not just the directories above. pytest-cov writes diff --git a/agent/README.md b/agent/README.md index f1ee7b456..cde506666 100644 --- a/agent/README.md +++ b/agent/README.md @@ -294,16 +294,19 @@ Each snake_case key installs into its UPPER_SNAKE env var, and a payload value * | `log_group_name` | `LOG_GROUP_NAME` | | | `artifacts_bucket_name` | `ARTIFACTS_BUCKET_NAME` | | | `trace_artifacts_bucket_name` | `TRACE_ARTIFACTS_BUCKET_NAME` | | +| `continuation_bucket_name` | `CONTINUATION_BUCKET_NAME` | | | `linear_oauth_secret_arn` | `LINEAR_OAUTH_SECRET_ARN` | | +| `linear_vault_enabled` | `LINEAR_VAULT_ENABLED` | | +| `linear_workload_identity_name` | `LINEAR_WORKLOAD_IDENTITY_NAME` | | | `jira_oauth_secret_arn` | `JIRA_OAUTH_SECRET_ARN` | | | `aws_sdk_ua_app_id` | `AWS_SDK_UA_APP_ID` | | | `anthropic_default_haiku_model` | `ANTHROPIC_DEFAULT_HAIKU_MODEL` | | -Values are **non-secret identifiers only** — secrets are still fetched at `/run` time from Secrets Manager using the ARNs delivered here, so task secrets need not be baked into the snapshot. The build-hook warning is not proof that a hand-built image contains no secrets. The allowlist **fails closed**: these values land in `os.environ` of the process that spawns the agent's tool subprocesses, so an unrecognised key is an env-injection attempt (`LD_PRELOAD`, `AWS_ENDPOINT_URL`, …) and the whole run is rejected with nothing installed. Blank/`null` values for optional keys are skipped rather than clobbering an image value; blank required keys are rejected. A verified MicroVM payload must contain `platform_config`; there is no baked-environment fallback. Control characters and inconsistent ARN account/partition fields are also rejected. ARN consistency alone is not deployment authentication: v2 supplies that through the IAM-read manifest and exact configuration comparison, including same-account workspace identifiers ([#817](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/817)). +Values are **non-secret configuration only**. Credentials are fetched at task startup from Secrets Manager or AgentCore Identity. Linear vault use requires `linear_vault_enabled="true"` and the workload identity name; the task's channel metadata identifies the workspace grant. The image carries no task credentials. The allowlist **fails closed**: an unrecognised key rejects the whole block. Blank optional values are skipped; blank required values, control characters and inconsistent ARN account/partition fields are rejected. Deployment authentication comes from the IAM-read manifest and exact configuration comparison, including same-account workspace identifiers ([#817](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/817)). Rejections are structured so they are readable in the MicroVM log group: `400 MICROVM_RUN_PAYLOAD_INVALID` (unusable envelope — retrying the same body cannot help), `500 MICROVM_RUN_PAYLOAD_UNREADABLE` (manifest/payload read or stored bytes failed), `400 MICROVM_RUN_PLATFORM_CONFIG_INVALID` (key off the allowlist, non-object block, or non-string value — fix the producer), `400 MICROVM_RUN_PLATFORM_CONFIG_INCOMPLETE` (a required key missing or blank — fix the deployment wiring), `400 TASK_RECORD_INCOMPLETE` (same validator and vocabulary as `/invocations`). -`/suspend` and `/resume` are served by `microvm_http.py` and declared in managed images with 30-second service timeouts and a shared non-secret protocol marker. Pause requires an active original approval gate, matching coordinator intent, drained activity and an acknowledged checkpoint; wake renews credentials and atomically rechecks the original task/gate before allowing coding. Each handler has a 20-second total budget. `/validate` rejects a supplied incompatible image marker without contacting AWS. The coordinator checks the actual launched image version and persists support on that worker; missing support disables new suspension. See [image capability verification](../docs/verification/645-p3-image-capability.md). Supervisor integration and live P3 acceptance remain open. +`/suspend` and `/resume` are served by `microvm_http.py` and declared in managed images with 30-second service timeouts and a shared non-secret protocol marker. Pause requires an active original approval gate, matching coordinator intent, drained activity and an acknowledged checkpoint; wake renews credentials and atomically rechecks the original task/gate before allowing coding. Each handler has a 20-second total budget. `/validate` rejects a supplied incompatible image marker without contacting AWS. The coordinator checks the actual launched image version and persists support on that worker; missing support disables new suspension. See [image capability verification](../docs/verification/645-p3-image-capability.md) and [completed P3 verification](../docs/verification/645-p3-nested-stack.md). #### Conversation continuation checkpoints (P3 prerequisite) diff --git a/agent/policies/soft_deny.cedar b/agent/policies/soft_deny.cedar index fd04ae7f2..8104baded 100644 --- a/agent/policies/soft_deny.cedar +++ b/agent/policies/soft_deny.cedar @@ -16,20 +16,20 @@ permit (principal, action, resource); // Every rule in this file MUST carry: // @tier("soft") // @rule_id("...") — stable ID for --pre-approve rule:X -// @approval_timeout_s — integer seconds >= 30 (<120 emits WARN per IMPL-25) // @severity — "low" | "medium" | "high" // @category — optional free-form UX grouping +// An optional @approval_timeout_s sets a positive explicit deadline (>= 30s). +// Omitting it uses the task setting, whose default is no automatic expiry. // // Blueprints may OPT OUT of specific rules here via // `security.cedarPolicies.disable: [rule_id]`. They may NOT disable any // rule in hard_deny.cedar (blueprint loader rejects those at task start). -// Gate any git --force / -f push. 300s default approval window, medium severity. +// Gate any git --force / -f push, with medium severity. // Covers both long-form (--force) and short-form (-f) variants, including // the bare `git push -f` invocation with no branch argument. @tier("soft") @rule_id("force_push_any") -@approval_timeout_s("300") @severity("medium") @category("destructive") forbid (principal, action == Agent::Action::"execute_bash", resource) @@ -37,12 +37,11 @@ when { context.command like "*git push --force*" || context.command like "*git push -f *" || context.command like "*git push -f" }; -// Force-push to main/prod specifically — longer window, higher severity. +// Force-push to main/prod specifically — higher severity. // Multi-match with force_push_any is expected: the engine's annotation -// merging picks min(300, 600)=300s and max(medium, high)=high. +// merging picks max(medium, high)=high. There is no implicit approval expiry. @tier("soft") @rule_id("force_push_main") -@approval_timeout_s("600") @severity("high") @category("destructive") forbid (principal, action == Agent::Action::"execute_bash", resource) @@ -55,7 +54,6 @@ when { context.command like "*git push --force origin main*" // agent bypasses PR workflow by pushing directly. @tier("soft") @rule_id("push_to_protected_branch") -@approval_timeout_s("300") @severity("medium") @category("destructive") forbid (principal, action == Agent::Action::"execute_bash", resource) @@ -64,20 +62,18 @@ when { context.command like "*git push origin main*" || context.command like "*git push origin prod*" || context.command like "*git push origin release/*" }; -// Writes to `.env` files typically contain secrets. 600s window, high severity. +// Writes to `.env` files typically contain secrets. High severity. @tier("soft") @rule_id("write_env_files") -@approval_timeout_s("600") @severity("high") @category("filesystem") forbid (principal, action == Agent::Action::"write_file", resource) when { context.file_path like "*.env" }; // Writes to any path containing "credentials" — SSH keys, AWS creds, -// service-account JSON, etc. 300s window, high severity. +// service-account JSON, etc. High severity. @tier("soft") @rule_id("write_credentials") -@approval_timeout_s("300") @severity("high") @category("auth") forbid (principal, action == Agent::Action::"write_file", resource) diff --git a/agent/src/config.py b/agent/src/config.py index 3a8991d23..beebfaf51 100644 --- a/agent/src/config.py +++ b/agent/src/config.py @@ -93,10 +93,8 @@ def _resolve_linear_token_via_vault( applicable / unavailable for any reason — the caller then falls back to the Secrets-Manager token, so a vault hiccup must never raise. - Self-mints through boto3 (``get_workload_access_token_for_user_id`` then - ``get_resource_oauth2_token``) rather than reading the AgentCore-injected - ``WorkloadAccessToken`` header, so it behaves identically on the AgentCore - and ECS substrates (both authorize against the agent session role's IAM). + Uses the compute execution role through ``platform_client`` on AgentCore, + ECS and MicroVM. The task-scoped SessionRole does not mint vault tokens. The grant must already be consented (done at ``bgagent linear setup`` time); if the vault returns an ``authorizationUrl`` instead of a token, that means consent is required and we fall back rather than block on a browser. @@ -186,43 +184,21 @@ def _resolve_linear_token_via_vault( def resolve_linear_api_token(channel_metadata: dict[str, str] | None = None) -> str: - """Resolve the Linear OAuth access token from Secrets Manager. - - Phase 2.0b-O2: the orchestrator stamps ``linear_oauth_secret_arn`` - into the task record's ``channel_metadata`` at task-creation time. - Pass that dict in via ``channel_metadata`` (the pipeline does this - automatically). We fetch the per-workspace secret, parse the token - JSON, refresh if expiring, and cache the access_token in - ``LINEAR_API_TOKEN`` so the one remaining consumer — - ``linear_reactions.py``'s direct-GraphQL Authorization header (reactions + - state transitions) — keeps working. (ADR-016: there is no Linear MCP; this - token no longer feeds an ``.mcp.json`` placeholder.) - - For local development, a pre-set ``LINEAR_API_TOKEN`` env var - short-circuits the lookup so the agent can run outside the runtime. - - Returns an empty string when the credential is absent — ``linear_reactions`` - then skips its reactions/state calls (best-effort, logged). This function is - only called when ``channel_source == 'linear'``. - - RFC #249 Phase 1 re-introduces AgentCore Identity as the PREFERRED path - (``_resolve_linear_token_via_vault``, tried first above) for workspaces - onboarded through the vault; Secrets Manager remains the fallback and the - only path for SM-only installs. The Phase-0 spike confirmed the - USER_FEDERATION service-side issue that parked Phase 2.0a no longer - reproduces and that the flow forwards the ``actor=app`` agent install. + """Resolve the Linear token for direct GraphQL reactions and state updates. + + Reuses ``LINEAR_API_TOKEN`` when set, otherwise tries the configured vault + grant before the workspace's Secrets Manager fallback. Task channel metadata + supplies the provider, vault user and fallback secret identifiers. + + Successful resolution caches the token in ``LINEAR_API_TOKEN``. Missing or + unavailable credentials return an empty string; reactions then skip their + calls without stopping the coding task. """ cached = os.environ.get("LINEAR_API_TOKEN", "") if cached: return cached - # RFC #249 Phase 1: when the vault is enabled AND this task carries a - # provider name (vault-onboarded workspace), mint the token through the - # AgentCore Identity Token Vault first. Any failure returns "" here, so we - # fall through to the Secrets-Manager path below — the vault never blocks a - # task. The Phase-0 spike proved USER_FEDERATION forwards the actor=app - # install and re-enables the path that was parked in 2.0a (the service-side - # USER_FEDERATION bug it cited no longer reproduces). + # Vault-managed workspaces may no longer have a usable fallback secret. vault_token = _resolve_linear_token_via_vault(channel_metadata) if vault_token: os.environ["LINEAR_API_TOKEN"] = vault_token @@ -250,7 +226,7 @@ def resolve_linear_api_token(channel_metadata: dict[str, str] | None = None) -> # boto3 is imported here (not just via platform_client, which imports it # lazily at call time) so a missing SDK still degrades gracefully — skip - # Linear MCP — instead of raising an uncaught ImportError. (#319) + # Linear reactions — instead of raising an uncaught ImportError. (#319) import boto3 # noqa: F401 -- availability probe for the graceful skip below from botocore.exceptions import BotoCoreError, ClientError @@ -273,7 +249,10 @@ def _fetch_token() -> dict | None: """ resp = sm.get_secret_value(SecretId=secret_arn) try: - return json.loads(resp["SecretString"]) + payload = json.loads(resp["SecretString"]) + if not isinstance(payload, dict): + raise TypeError("expected a JSON object") + return payload except (json.JSONDecodeError, KeyError, TypeError) as e: log( "ERROR", @@ -314,6 +293,19 @@ def _try_refresh_once(current: dict) -> tuple[str, dict | None]: except ImportError: return ("failure", None) + missing = [ + key + for key in ("refresh_token", "client_id", "client_secret") + if not isinstance(current.get(key), str) or not current[key].strip() + ] + if missing: + log( + "WARN", + "linear_oauth_refresh_unavailable: fallback secret lacks " + f"{', '.join(missing)}; skipping refresh", + ) + return ("failure", None) + body = urllib.parse.urlencode( { "grant_type": "refresh_token", @@ -349,10 +341,6 @@ def _try_refresh_once(current: dict) -> tuple[str, dict | None]: return ("invalid_grant", None) return ("failure", None) except (urllib.error.URLError, OSError) as e: - # Genuine network failures (DNS, timeout, TCP reset). Other - # exceptions (KeyError on missing field, TypeError on bad - # JSON shape) are programmer errors and should propagate - # with a clear stack trace rather than being swallowed. log("WARN", f"resolve_linear_api_token refresh failed: {type(e).__name__}: {e}") return ("failure", None) @@ -373,7 +361,7 @@ def _try_refresh_once(current: dict) -> tuple[str, dict | None]: "access_token": payload["access_token"], "refresh_token": payload.get("refresh_token", current["refresh_token"]), "expires_at": expires_at_iso, - "scope": payload.get("scope", current["scope"]), + "scope": payload.get("scope", current.get("scope", "")), "updated_at": now.isoformat().replace("+00:00", "Z"), } diff --git a/agent/src/continuation_runtime.py b/agent/src/continuation_runtime.py new file mode 100644 index 000000000..fab65ee58 --- /dev/null +++ b/agent/src/continuation_runtime.py @@ -0,0 +1,427 @@ +# Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. +# SPDX-License-Identifier: MIT-0 + +"""Compose conversation, workflow and workspace recovery at an approval barrier.""" + +from __future__ import annotations + +import asyncio +import os +import subprocess +import tempfile +import time +from contextlib import contextmanager +from contextvars import ContextVar +from dataclasses import replace +from pathlib import Path +from typing import TYPE_CHECKING, Any + +from claude_agent_sdk import project_key_for_directory + +from continuation_session import ( + CheckpointIdentity, + CheckpointSessionStore, + ContinuationCheckpointError, + decode_checkpoint, +) +from continuation_storage import ( + ContinuationContext, + ContinuationManifest, + FileReceipt, + S3ContinuationStorage, +) +from continuation_usage import TOKEN_FIELDS, read_usage +from continuation_workspace import WorkspaceLimits, capture_workspace, restore_workspace + +if TYPE_CHECKING: + from microvm_lifecycle import ApprovalPark, MicrovmLifecycle + from models import TaskConfig + from policy import PolicyEngine + + +_current: ContextVar[ContinuationRuntime | None] = ContextVar("continuation_runtime", default=None) + + +def current_runtime() -> ContinuationRuntime | None: + return _current.get() + + +@contextmanager +def bind_runtime(runtime: ContinuationRuntime | None): + token = _current.set(runtime) + try: + yield + finally: + _current.reset(token) + + +def create_runtime(context: ContinuationContext, task_id: str) -> ContinuationRuntime | None: + from microvm_lifecycle import get_context + + bucket = os.environ.get("CONTINUATION_BUCKET_NAME", "") + if not bucket or get_context(task_id) is None: + return None + return ContinuationRuntime(context, S3ContinuationStorage(bucket)) + + +def restore_for_task(config: TaskConfig) -> ContinuationRuntime | None: + """Read a coordinator-owned replacement assignment; payloads do not choose S3 keys.""" + import task_state + from config import AGENT_WORKSPACE + from microvm_lifecycle import get_context + + bucket = os.environ.get("CONTINUATION_BUCKET_NAME", "") + lifecycle = get_context(config.task_id) + if not bucket or lifecycle is None: + return None + registration_deadline = time.monotonic() + 30 + while True: + task = task_state.get_task(config.task_id, consistent_read=True) + record = task.get("continuation") if task else None + if not isinstance(record, dict) or record.get("state") != "STARTING": + break + # /run can start the pipeline before RunMicrovm returns its handle to + # the coordinator. Never turn that registration race into a fresh clone. + if time.monotonic() >= registration_deadline: + raise ContinuationCheckpointError( + "Replacement worker registration was not acknowledged" + ) + time.sleep(0.2) + if ( + task is None + or task.get("user_id") != config.user_id + or (task.get("repo") or "") != config.repo_url + ): + raise ContinuationCheckpointError("Continuation task identity is unavailable") + record = task.get("continuation") + if record is None: + return None + if not isinstance(record, dict) or record.get("state") != "RESTORING": + raise ContinuationCheckpointError("Existing continuation cannot start as a fresh task") + identity = CheckpointIdentity(**record["identity"]) + if ( + record.get("version") != 1 + or record.get("worker_id") != lifecycle.microvm_id + or task.get("session_id") != lifecycle.microvm_id + or task.get("status") != "AWAITING_APPROVAL" + or task.get("awaiting_approval_request_id") != identity.request_id + or (identity.task_id, identity.user_id, identity.repo) + != (config.task_id, config.user_id, config.repo_url) + ): + raise ContinuationCheckpointError("Replacement worker does not own this continuation") + workflow = config.resolved_workflow or {} + runtime = ContinuationRuntime.restore( + S3ContinuationStorage(bucket), + FileReceipt.from_record(record["manifest"], identity), + identity, + expected_workspace=Path(AGENT_WORKSPACE).resolve() / config.task_id, + workflow_id=workflow.get("id", ""), + workflow_version=workflow.get("version", ""), + ) + approval = task_state.get_approval_row( + config.task_id, identity.request_id, consistent_read=True + ) + if ( + approval is None + or approval.get("user_id") != config.user_id + or (approval.get("repo") or "") != config.repo_url + or runtime.restored is None + or approval.get("tool_input_sha256") != runtime.restored["action"]["tool_input_sha256"] + or approval.get("status") not in {"APPROVED", "DENIED", "TIMED_OUT"} + ): + raise ContinuationCheckpointError("Continuation human decision is unavailable") + # No SDK process exists yet. Only after restoration and this conditional + # claim may its freshly gated tools start. + task_state.consume_restored_continuation(config.task_id, lifecycle.microvm_id, record) + runtime.resume_prompt = runtime.decision_prompt( + decision=approval["status"], reason=approval.get("deny_reason") or "" + ) + runtime.human_decision = approval + config.initial_approval_gate_count = max( + int(task.get("approval_gate_count", 0)), runtime.context.approval_gate_count + ) + return runtime + + +def prepare_repoless_runtime( + config: TaskConfig, + *, + user_prompt: str, + system_prompt: str, + workflow_id: str, + workflow_version: str, +) -> ContinuationRuntime | None: + """Give a MicroVM task a private scratch workspace with a local-only Git baseline.""" + from config import AGENT_WORKSPACE + from microvm_lifecycle import get_context + from models import RepoSetup + + if not os.environ.get("CONTINUATION_BUCKET_NAME") or get_context(config.task_id) is None: + return None + if config.repo_url: + raise ContinuationCheckpointError( + "Scratch recovery is only for a task without a repository" + ) + restored = restore_for_task(config) + if restored is not None: + return restored + # Validate the task component before joining it to a filesystem path. + CheckpointIdentity(config.task_id, "initial", "initial", config.user_id, "") + workspace = Path(AGENT_WORKSPACE).resolve() / config.task_id + workspace.parent.mkdir(parents=True, exist_ok=True) + workspace.mkdir(mode=0o700, exist_ok=False) + env = {key: value for key, value in os.environ.items() if not key.startswith("GIT_")} + env.update(GIT_CONFIG_GLOBAL=os.devnull, GIT_CONFIG_NOSYSTEM="1", GIT_TERMINAL_PROMPT="0") + + def git(*args: str) -> str: + return subprocess.run( + [ + "git", + "-C", + str(workspace), + "-c", + "core.hooksPath=/dev/null", + "-c", + "user.name=ABCA", + "-c", + "user.email=workspace@invalid", + *args, + ], + env=env, + capture_output=True, + text=True, + check=True, + timeout=30, + ).stdout.strip() + + git("init", "-b", "scratch") + git("commit", "--allow-empty", "--no-gpg-sign", "-m", "Private task workspace") + return create_runtime( + ContinuationContext( + RepoSetup( + repo_dir=str(workspace), branch="scratch", head_sha_before=git("rev-parse", "HEAD") + ), + user_prompt, + system_prompt, + workflow_id, + workflow_version, + ), + config.task_id, + ) + + +class ContinuationRuntime: + """A single runner's state; SDK append/checkpoint calls share its event loop.""" + + def __init__( + self, + context: ContinuationContext, + storage: S3ContinuationStorage, + *, + store: CheckpointSessionStore | None = None, + restored: dict | None = None, + ) -> None: + self.context = context + self.storage = storage + self.store = store or CheckpointSessionStore( + project_key_for_directory(context.setup.repo_dir) + ) + self.restored = restored + self.resume_prompt: str | None = None + self.human_decision: dict | None = None + self._approval_consumed = False + self.client: Any = None + # These remain fixed for this SDK process. Repeated captures must not + # add its cumulative counters to an earlier capture of the same process. + self.prior_cost_usd = context.cost_usd if restored is not None else 0.0 + self.prior_token_usage = dict(context.token_usage) if restored is not None else {} + + def seed_policy(self, engine: PolicyEngine) -> None: + """Carry session grants and the recorded denial across worker replacement.""" + for scope in self.context.approval_scopes: + engine.allowlist.add(scope) + decision = self.human_decision + if self.restored is None or decision is None: + return + action = self.restored["action"] + status = decision["status"] + if status == "APPROVED": + scope = decision.get("scope") or "this_call" + if scope != "this_call": + engine.allowlist.add(scope) + elif status in {"DENIED", "TIMED_OUT"}: + reason = decision.get("deny_reason") or "" + engine.recent_decisions.record( + action["tool_name"], + action["tool_input_sha256"], + decision=status, + reason=reason, + original_decision_ts=decision.get("decided_at"), + ) + if status == "DENIED": + for rule_id in decision.get("matching_rule_ids", []): + engine.recent_decisions.record_rule_decision( + action["tool_name"], + rule_id, + decision="DENIED", + reason=reason, + original_decision_ts=decision.get("decided_at"), + ) + + def consume_approved_action(self, tool_name: str, tool_input_sha256: str) -> dict | None: + """Consume one exact saved proposal, only after the normal policy check.""" + decision = self.human_decision + if ( + self.restored is None + or decision is None + or decision.get("status") != "APPROVED" + or self._approval_consumed + or self.restored["action"]["tool_name"] != tool_name + or self.restored["action"]["tool_input_sha256"] != tool_input_sha256 + ): + return None + self._approval_consumed = True + return decision + + async def capture( + self, + lifecycle: MicrovmLifecycle, + park: ApprovalPark, + *, + session_id: str, + tool_name: str, + tool_input: dict, + approval_scopes: tuple[str, ...] = (), + approval_gate_count: int = 0, + ) -> tuple[CheckpointIdentity, FileReceipt]: + if park.record is None: + raise ContinuationCheckpointError("Approval identity is unavailable for continuation") + identity = CheckpointIdentity( + park.task_id, + park.microvm_id, + park.request_id, + park.record.user_id, + park.record.repo, + ) + async with lifecycle.continuation_checkpoint(park): + body = await self.store.checkpoint_pending( + identity, + session_id=session_id, + tool_use_id=park.tool_use_id, + tool_name=tool_name, + tool_input=tool_input, + timeout_s=10, + ) + entries = decode_checkpoint(body, identity)["entries"] + usage = await read_usage(self.client) + turns = set() + for index, entry in enumerate(entries): + message = entry.get("message") + if entry.get("type") != "assistant" or not isinstance(message, dict): + continue + message_id = message.get("id") + turns.add( + message_id + if isinstance(message_id, str) and message_id + else entry.get("uuid") or f"entry-{index}" + ) + self.context = replace( + self.context, + approval_scopes=approval_scopes, + approval_gate_count=max(self.context.approval_gate_count, approval_gate_count), + turns_used=max(self.context.turns_used, len(turns)), + cost_usd=self.prior_cost_usd + usage.cost_usd, + token_usage={ + key: self.prior_token_usage.get(key, 0) + usage.tokens.get(key, 0) + for key in TOKEN_FIELDS + }, + ) + receipt = await asyncio.to_thread(self._save, body, identity) + return identity, receipt + + def _save(self, body: bytes, identity: CheckpointIdentity) -> FileReceipt: + # Staging is outside the workspace so it cannot recursively archive + # itself. TemporaryDirectory is private and is removed on every exit. + with tempfile.TemporaryDirectory(prefix="abca-continuation-") as directory: + archive = Path(directory).resolve() / "workspace.tar" + capture_workspace( + Path(self.context.setup.repo_dir), + archive, + identity, + limits=WorkspaceLimits(max_bytes=self.storage.limits.max_workspace_bytes), + ) + workspace = self.storage.save_workspace(archive, identity) + conversation = self.storage.conversations.save(body, identity) + return self.storage.save_manifest( + ContinuationManifest(identity, conversation, workspace, self.context) + ) + + @classmethod + def restore( + cls, + storage: S3ContinuationStorage, + receipt: FileReceipt, + identity: CheckpointIdentity, + *, + expected_workspace: Path, + workflow_id: str, + workflow_version: str, + ) -> ContinuationRuntime: + manifest = storage.load_manifest(receipt, identity) + context = manifest.context + if ( + context.setup.repo_dir != str(expected_workspace) + or context.workflow_id != workflow_id + or context.workflow_version != workflow_version + ): + raise ContinuationCheckpointError("Continuation workspace or workflow changed") + body = storage.conversations.load(manifest.conversation, identity) + restored = decode_checkpoint(body, identity) + if restored["project_key"] != project_key_for_directory(str(expected_workspace)): + raise ContinuationCheckpointError( + "Continuation SDK project does not match the workspace" + ) + with tempfile.TemporaryDirectory(prefix="abca-restore-") as directory: + archive = Path(directory).resolve() / "workspace.tar" + storage.download_workspace(manifest.workspace, identity, archive) + restore_workspace( + archive, + expected_workspace, + identity, + expected_sha256=manifest.workspace.sha256, + limits=WorkspaceLimits(max_bytes=storage.limits.max_workspace_bytes), + ) + return cls( + context, + storage, + store=CheckpointSessionStore.restore(body, identity), + restored=restored, + ) + + def decision_prompt(self, *, decision: str, reason: str = "") -> str: + """Supply the saved action as context; every newly proposed tool is gated.""" + import json + + if self.restored is None or decision not in {"APPROVED", "DENIED", "TIMED_OUT"}: + raise ContinuationCheckpointError("A resolved continuation decision is required") + action = self.restored["action"] + return ( + "Continue the saved task from the restored workspace and conversation. " + "The previous worker stopped while waiting for a human decision; its pending " + "tool call was not executed. The recorded decision for that request is " + + decision + + ". Human feedback: " + + json.dumps(reason) + + ". Saved tool proposal: " + + json.dumps({"tool_name": action["tool_name"], "tool_input": action["tool_input"]}) + + ( + ". If you perform this approved proposal, use exactly the saved tool name " + "and every saved input field and value, including descriptions and timeouts. " + "Do not paraphrase the description or add optional arguments: any input change " + "is a new proposal and requires its own permission check" + if decision == "APPROVED" + else "" + ) + + ". Decide how to continue using this context. A denial must not be worked around. " + "Any new tool proposal still goes through the normal permission checks." + ) diff --git a/agent/src/continuation_session.py b/agent/src/continuation_session.py index 9fc878434..c7d3ebf45 100644 --- a/agent/src/continuation_session.py +++ b/agent/src/continuation_session.py @@ -1,7 +1,7 @@ # Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. # SPDX-License-Identifier: MIT-0 -"""Acknowledged SDK conversation checkpoints for future worker continuation. +"""Acknowledged SDK conversation checkpoints for worker continuation. This module does not release workers or change approval deadlines. A caller must hold the lifecycle barrier, checkpoint the workspace, and conditionally publish @@ -24,11 +24,13 @@ from claude_agent_sdk import SessionStore +from shared_constants import SHARED_CONSTANTS + if TYPE_CHECKING: from claude_agent_sdk import SessionKey, SessionStoreEntry CHECKPOINT_VERSION = 1 -MAX_CHECKPOINT_BYTES = 16 * 1024 * 1024 +MAX_CHECKPOINT_BYTES = SHARED_CONSTANTS["microvm_continuation"]["max_conversation_bytes"] MAX_CHECKPOINT_ENTRIES = 50_000 MAX_ID_LENGTH = 128 _ID = re.compile(r"[A-Za-z0-9][A-Za-z0-9_-]{0,127}\Z") @@ -92,7 +94,8 @@ def __post_init__(self) -> None: @property def prefix(self) -> str: - return f"continuations/{self.task_id}/{self.attempt_id}/{self.request_id}/" + prefix = SHARED_CONSTANTS["microvm_continuation"]["object_key_prefix"] + return f"{prefix}{self.task_id}/{self.attempt_id}/{self.request_id}/" @dataclass(frozen=True) diff --git a/agent/src/continuation_storage.py b/agent/src/continuation_storage.py new file mode 100644 index 000000000..0cc40ad43 --- /dev/null +++ b/agent/src/continuation_storage.py @@ -0,0 +1,477 @@ +# Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. +# SPDX-License-Identifier: MIT-0 + +"""Bounded, version-pinned storage for complete continuation checkpoints. + +The small manifest is the commit record: it is written only after the workspace +and SDK conversation have both been read back successfully. Publishing that +manifest to the task's conditional state record is a separate operation. An S3 +upload alone never grants permission to stop a worker. +""" + +from __future__ import annotations + +import base64 +import hashlib +import json +import os +import re +import shutil +import stat +import tempfile +import time +from dataclasses import asdict, dataclass, field +from decimal import Decimal +from pathlib import Path +from typing import Any, BinaryIO + +from continuation_session import ( + CheckpointIdentity, + CheckpointReceipt, + ContinuationCheckpointError, + S3ContinuationCheckpoints, +) +from models import RepoSetup +from shared_constants import SHARED_CONSTANTS + +_CHUNK = 1024 * 1024 +_MAX_MANIFEST_BYTES = SHARED_CONSTANTS["microvm_continuation"]["max_manifest_bytes"] +_SHA256 = re.compile(r"[0-9a-f]{64}\Z") +_KINDS = {"workspace": "tar", "manifest": "json"} +_VERSION = SHARED_CONSTANTS["microvm_continuation"]["version"] +_MAX_VERSION_ID_LENGTH = 1024 + + +class ContinuationStorageError(ContinuationCheckpointError): + """Content-free failure classification for lifecycle diagnostics.""" + + def __init__(self, code: str, message: str) -> None: + self.code = code + super().__init__(message) + + +@dataclass(frozen=True) +class StorageLimits: + max_workspace_bytes: int = SHARED_CONSTANTS["microvm_continuation"]["max_workspace_bytes"] + disk_reserve_bytes: int = 128 * _CHUNK + transfer_timeout_s: int = 300 + + def __post_init__(self) -> None: + if any( + type(value) is not int or value <= 0 + for value in ( + self.max_workspace_bytes, + self.disk_reserve_bytes, + self.transfer_timeout_s, + ) + ): + raise ValueError("Continuation storage limits must be positive integers") + + +_DEFAULT_LIMITS = StorageLimits() + + +@dataclass(frozen=True) +class FileReceipt: + kind: str + key: str + version_id: str + sha256: str + size_bytes: int + + @classmethod + def from_record( + cls, value: Any, identity: CheckpointIdentity, limits: StorageLimits = _DEFAULT_LIMITS + ) -> FileReceipt: + """Accept DynamoDB's integral Decimal without coercing strings/floats/bools.""" + if not isinstance(value, dict) or set(value) != { + "kind", + "key", + "version_id", + "sha256", + "size_bytes", + }: + raise ContinuationStorageError( + "invalid_receipt", "Continuation file receipt is invalid" + ) + data = dict(value) + size = data["size_bytes"] + if isinstance(size, Decimal) and size.is_finite() and size == size.to_integral_value(): + data["size_bytes"] = int(size) + receipt = cls(**data) + receipt.validate(identity, limits) + return receipt + + def validate(self, identity: CheckpointIdentity, limits: StorageLimits) -> None: + suffix = _KINDS.get(self.kind) + bound = limits.max_workspace_bytes if self.kind == "workspace" else _MAX_MANIFEST_BYTES + if ( + suffix is None + or not isinstance(self.sha256, str) + or not _SHA256.fullmatch(self.sha256) + or self.key != f"{identity.prefix}{self.kind}/{self.sha256}.{suffix}" + or not isinstance(self.version_id, str) + or not self.version_id + or self.version_id == "null" + or len(self.version_id) > _MAX_VERSION_ID_LENGTH + or type(self.size_bytes) is not int + or not 0 < self.size_bytes <= bound + ): + raise ContinuationStorageError( + "invalid_receipt", "Continuation file receipt is invalid" + ) + + +@dataclass(frozen=True) +class ContinuationContext: + """Workflow products needed after recovery; never serialize TaskConfig secrets.""" + + setup: RepoSetup + user_prompt: str + system_prompt: str + workflow_id: str + workflow_version: str + approval_scopes: tuple[str, ...] = () + approval_gate_count: int = 0 + turns_used: int = 0 + cost_usd: float = 0.0 + token_usage: dict[str, int] = field(default_factory=dict) + started_reaction_id: str | None = None + + def to_dict(self) -> dict: + value = asdict(self) + value["setup"] = self.setup.model_dump(mode="json") + value["approval_scopes"] = list(self.approval_scopes) + return value + + @classmethod + def from_dict(cls, value: Any) -> ContinuationContext: + from continuation_usage import valid_cost, valid_tokens + + text_fields = {"user_prompt", "system_prompt", "workflow_id", "workflow_version"} + fields = text_fields | { + "setup", + "approval_scopes", + "approval_gate_count", + "turns_used", + "cost_usd", + "token_usage", + "started_reaction_id", + } + if ( + not isinstance(value, dict) + or set(value) != fields + or any(not isinstance(value[name], str) for name in text_fields) + or not value["workflow_id"] + or not value["workflow_version"] + or not isinstance(value["setup"], dict) + or set(value["setup"]) != set(RepoSetup.model_fields) + or not isinstance(value["approval_scopes"], list) + or any(not isinstance(scope, str) for scope in value["approval_scopes"]) + or any( + type(value[name]) is not int or value[name] < 0 + for name in ("approval_gate_count", "turns_used") + ) + or not valid_cost(value["cost_usd"]) + or not valid_tokens(value["token_usage"]) + or ( + value["started_reaction_id"] is not None + and not isinstance(value["started_reaction_id"], str) + ) + ): + raise ContinuationStorageError( + "invalid_context", "Continuation workflow context is invalid" + ) + try: + setup = RepoSetup.model_validate(value["setup"], strict=True) + except ValueError as exc: + raise ContinuationStorageError( + "invalid_context", "Continuation repository context is invalid" + ) from exc + if not Path(setup.repo_dir).is_absolute(): + raise ContinuationStorageError( + "invalid_context", "Continuation workspace must be absolute" + ) + return cls( + setup, + value["user_prompt"], + value["system_prompt"], + value["workflow_id"], + value["workflow_version"], + tuple(value["approval_scopes"]), + value["approval_gate_count"], + value["turns_used"], + float(value["cost_usd"]), + value["token_usage"], + value["started_reaction_id"], + ) + + +@dataclass(frozen=True) +class ContinuationManifest: + identity: CheckpointIdentity + conversation: CheckpointReceipt + workspace: FileReceipt + context: ContinuationContext + + def encode(self, limits: StorageLimits) -> bytes: + self.conversation.validate(self.identity) + self.workspace.validate(self.identity, limits) + if self.workspace.kind != "workspace": + raise ContinuationStorageError("invalid_manifest", "Manifest workspace kind is invalid") + context = self.context.to_dict() + ContinuationContext.from_dict(context) + try: + body = json.dumps( + { + "version": _VERSION, + "identity": asdict(self.identity), + "conversation": asdict(self.conversation), + "workspace": asdict(self.workspace), + "context": context, + }, + sort_keys=True, + separators=(",", ":"), + allow_nan=False, + ).encode() + except (TypeError, ValueError, RecursionError) as exc: + raise ContinuationStorageError("invalid_manifest", "Manifest data is invalid") from exc + if len(body) > _MAX_MANIFEST_BYTES: + raise ContinuationStorageError( + "size_limit", "Continuation manifest exceeds its byte limit" + ) + return body + + @classmethod + def decode( + cls, body: bytes, identity: CheckpointIdentity, limits: StorageLimits + ) -> ContinuationManifest: + if not body or len(body) > _MAX_MANIFEST_BYTES: + raise ContinuationStorageError( + "size_limit", "Continuation manifest exceeds its byte limit" + ) + try: + value = json.loads(body) + if ( + not isinstance(value, dict) + or set(value) != {"version", "identity", "conversation", "workspace", "context"} + or type(value["version"]) is not int + or value["version"] != _VERSION + or value["identity"] != asdict(identity) + ): + raise ValueError("Invalid envelope") + result = cls( + identity, + CheckpointReceipt(**value["conversation"]), + FileReceipt(**value["workspace"]), + ContinuationContext.from_dict(value["context"]), + ) + result.encode(limits) + return result + except (TypeError, ValueError, KeyError, RecursionError) as exc: + raise ContinuationStorageError( + "invalid_manifest", "Continuation manifest is invalid" + ) from exc + + +def _stamp(info: os.stat_result) -> tuple[int, ...]: + return (info.st_dev, info.st_ino, info.st_size, info.st_mtime_ns, info.st_ctime_ns) + + +def _time_left(deadline: float) -> None: + if time.monotonic() >= deadline: + raise ContinuationStorageError( + "transfer_timeout", "Continuation transfer exceeded its deadline" + ) + + +def _space(path: Path, required: int, limits: StorageLimits) -> None: + if shutil.disk_usage(path).free < required + limits.disk_reserve_bytes: + raise ContinuationStorageError( + "disk_pressure", "Continuation transfer has insufficient disk space" + ) + + +class S3ContinuationStorage: + def __init__( + self, bucket: str, *, client: Any = None, limits: StorageLimits = _DEFAULT_LIMITS + ) -> None: + # Reuse the mandatory task-scoped credential check and conservative SDK + # timeouts. An explicitly injected client is only for tests/operators. + self.conversations = S3ContinuationCheckpoints(bucket, client=client) + self.bucket = self.conversations.bucket + self.client = self.conversations.client + self.limits = limits + + def _read( + self, + receipt: FileReceipt, + identity: CheckpointIdentity, + destination: BinaryIO | None, + *, + deadline: float, + current: bool = False, + disk_path: Path | None = None, + ) -> FileReceipt: + receipt.validate(identity, self.limits) + _time_left(deadline) + response = self.client.get_object( + Bucket=self.bucket, + Key=receipt.key, + ChecksumMode="ENABLED", + **({} if current else {"VersionId": receipt.version_id}), + ) + stream = response["Body"] + try: + version = response.get("VersionId") + if ( + not isinstance(version, str) + or not version + or version == "null" + or (not current and version != receipt.version_id) + or type(response.get("ContentLength")) is not int + or response["ContentLength"] != receipt.size_bytes + or response.get("ChecksumSHA256") + != base64.b64encode(bytes.fromhex(receipt.sha256)).decode() + ): + raise ContinuationStorageError( + "integrity_failed", "Stored continuation metadata differs" + ) + digest = hashlib.sha256() + count = 0 + while True: + _time_left(deadline) + chunk = stream.read(min(_CHUNK, receipt.size_bytes - count + 1)) + if not chunk: + break + count += len(chunk) + if count > receipt.size_bytes: + raise ContinuationStorageError( + "size_limit", "Stored continuation exceeds its receipt" + ) + digest.update(chunk) + if destination is not None: + if disk_path is not None: + _space(disk_path, len(chunk), self.limits) + destination.write(chunk) + if count != receipt.size_bytes or digest.hexdigest() != receipt.sha256: + raise ContinuationStorageError( + "integrity_failed", "Stored continuation content differs" + ) + verified = FileReceipt(receipt.kind, receipt.key, version, receipt.sha256, count) + verified.validate(identity, self.limits) + return verified + finally: + stream.close() + + def _save( + self, source: BinaryIO, identity: CheckpointIdentity, kind: str, size: int + ) -> FileReceipt: + deadline = time.monotonic() + self.limits.transfer_timeout_s + digest = hashlib.sha256() + count = 0 + while chunk := source.read(_CHUNK): + _time_left(deadline) + count += len(chunk) + if count > size: + raise ContinuationStorageError( + "source_changed", "Prepared continuation changed size" + ) + digest.update(chunk) + if count != size: + raise ContinuationStorageError("source_changed", "Prepared continuation is incomplete") + sha256 = digest.hexdigest() + key = f"{identity.prefix}{kind}/{sha256}.{_KINDS[kind]}" + # The provisional version is never returned: _read must supply and + # validate a real S3 version, including after a lost PutObject reply. + provisional = FileReceipt(kind, key, "unverified", sha256, size) + provisional.validate(identity, self.limits) + source.seek(0) + write_error = None + try: + _time_left(deadline) + self.client.put_object( + Bucket=self.bucket, + Key=key, + Body=source, + ContentLength=size, + ContentType="application/x-tar" if kind == "workspace" else "application/json", + ServerSideEncryption="AES256", + ChecksumSHA256=base64.b64encode(digest.digest()).decode(), + IfNoneMatch="*", + ) + except Exception as exc: + write_error = exc + try: + return self._read(provisional, identity, None, deadline=deadline, current=True) + except Exception as exc: + raise ContinuationStorageError( + "unverified_upload", "Continuation upload could not be verified; retain the worker" + ) from (write_error if write_error is not None else exc) + + def save_workspace(self, archive: Path, identity: CheckpointIdentity) -> FileReceipt: + with os.fdopen( + os.open(archive, os.O_RDONLY | os.O_NOFOLLOW | os.O_NONBLOCK), "rb" + ) as source: + before = os.fstat(source.fileno()) + if ( + not stat.S_ISREG(before.st_mode) + or before.st_nlink != 1 + or not 0 < before.st_size <= self.limits.max_workspace_bytes + ): + raise ContinuationStorageError( + "invalid_source", "Prepared workspace archive is invalid" + ) + receipt = self._save(source, identity, "workspace", before.st_size) + if _stamp(os.fstat(source.fileno())) != _stamp(before): + raise ContinuationStorageError( + "source_changed", "Workspace archive changed during upload" + ) + return receipt + + def save_manifest(self, manifest: ContinuationManifest) -> FileReceipt: + import io + + body = manifest.encode(self.limits) + return self._save(io.BytesIO(body), manifest.identity, "manifest", len(body)) + + def load_manifest( + self, receipt: FileReceipt, identity: CheckpointIdentity + ) -> ContinuationManifest: + import io + + if receipt.kind != "manifest": + raise ContinuationStorageError("invalid_receipt", "Expected a continuation manifest") + target = io.BytesIO() + self._read( + receipt, + identity, + target, + deadline=time.monotonic() + self.limits.transfer_timeout_s, + ) + return ContinuationManifest.decode(target.getvalue(), identity, self.limits) + + def download_workspace( + self, receipt: FileReceipt, identity: CheckpointIdentity, destination: Path + ) -> None: + if receipt.kind != "workspace": + raise ContinuationStorageError("invalid_receipt", "Expected a workspace archive") + receipt.validate(identity, self.limits) + _space(destination.parent, receipt.size_bytes, self.limits) + # The caller supplies a private staging directory. Never truncate an + # existing destination; publish the verified file with a no-replace link. + fd, name = tempfile.mkstemp(prefix=".continuation-", dir=destination.parent) + try: + with os.fdopen(fd, "wb") as target: + self._read( + receipt, + identity, + target, + deadline=time.monotonic() + self.limits.transfer_timeout_s, + disk_path=destination.parent, + ) + target.flush() + os.fsync(target.fileno()) + os.link(name, destination) + finally: + os.unlink(name) diff --git a/agent/src/continuation_usage.py b/agent/src/continuation_usage.py new file mode 100644 index 000000000..a43a5eced --- /dev/null +++ b/agent/src/continuation_usage.py @@ -0,0 +1,86 @@ +# Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. +# SPDX-License-Identifier: MIT-0 + +"""Version-checked access to exact CLI spending at a paused approval hook.""" + +from __future__ import annotations + +import asyncio +import importlib.metadata +import math +from dataclasses import dataclass +from typing import Any + +from continuation_session import ContinuationCheckpointError +from shared_constants import SHARED_CONSTANTS + +TOKEN_FIELDS = { + "input_tokens": "inputTokens", + "output_tokens": "outputTokens", + "cache_read_input_tokens": "cacheReadInputTokens", + "cache_creation_input_tokens": "cacheCreationInputTokens", +} + + +@dataclass(frozen=True) +class UsageSnapshot: + cost_usd: float + tokens: dict[str, int] + + +def valid_cost(value: Any) -> bool: + return type(value) in (int, float) and math.isfinite(value) and value >= 0 + + +def valid_tokens(value: Any) -> bool: + return ( + isinstance(value, dict) + and set(value) <= set(TOKEN_FIELDS) + and all(type(count) is int and count >= 0 for count in value.values()) + ) + + +async def read_usage(client: Any) -> UsageSnapshot: + """Read cumulative spending without interrupting or issuing another query. + + SDK 0.2.110 bundles CLI 2.1.191, whose experimental ``get_usage`` control + request exposes exact session dollars and per-model token counters. Python + has no public wrapper yet. Keep this dependency isolated and covered by the + opt-in real SDK probe; an upgrade must verify it before publishing checkpoints. + """ + if ( + importlib.metadata.version("claude-agent-sdk") + != SHARED_CONSTANTS["microvm_continuation"]["verified_sdk_version"] + ): + raise ContinuationCheckpointError( + "Continuation accounting requires the verified SDK version" + ) + query = getattr(client, "_query", None) + send = getattr(query, "_send_control_request", None) + if not callable(send): + raise ContinuationCheckpointError("Continuation accounting client is unavailable") + response = await asyncio.wait_for(send({"subtype": "get_usage"}), timeout=5) + session = response.get("session") if isinstance(response, dict) else None + if not isinstance(session, dict) or not valid_cost(session.get("total_cost_usd")): + raise ContinuationCheckpointError("Continuation accounting response has no valid cost") + models = session.get("model_usage") + if not isinstance(models, dict): + raise ContinuationCheckpointError("Continuation accounting response has no model usage") + tokens = dict.fromkeys(TOKEN_FIELDS, 0) + model_cost = 0.0 + for usage in models.values(): + if ( + not isinstance(usage, dict) + or not valid_cost(usage.get("costUSD")) + or any( + type(usage.get(key)) is not int or usage[key] < 0 for key in TOKEN_FIELDS.values() + ) + ): + raise ContinuationCheckpointError("Continuation model usage is invalid") + model_cost += usage["costUSD"] + for target, source in TOKEN_FIELDS.items(): + tokens[target] += usage[source] + cost = float(session["total_cost_usd"]) + if not math.isclose(model_cost, cost, rel_tol=1e-9, abs_tol=1e-12): + raise ContinuationCheckpointError("Continuation model costs do not match session spending") + return UsageSnapshot(cost, tokens) diff --git a/agent/src/continuation_workspace.py b/agent/src/continuation_workspace.py index c0bbe51bc..a15767a31 100644 --- a/agent/src/continuation_workspace.py +++ b/agent/src/continuation_workspace.py @@ -1,7 +1,7 @@ # Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. # SPDX-License-Identifier: MIT-0 -"""Offline workspace archives for future worker continuation. +"""Offline workspace archives for worker continuation. The caller must stop workspace writers with the lifecycle barrier. Capture is local only: durable upload/read-back and conditional publication alongside the @@ -34,6 +34,7 @@ from typing import BinaryIO, NoReturn from continuation_session import CheckpointIdentity, ContinuationCheckpointError +from shared_constants import SHARED_CONSTANTS _OID = re.compile(r"[0-9a-f]{40}\Z") _SHA256 = re.compile(r"[0-9a-f]{64}\Z") @@ -46,7 +47,7 @@ class WorkspaceCheckpointError(ContinuationCheckpointError): - """A content-free stage code suitable for future lifecycle feedback.""" + """A content-free stage code for lifecycle feedback.""" def __init__(self, code: str, message: str) -> None: self.code = code @@ -55,7 +56,7 @@ def __init__(self, code: str, message: str) -> None: @dataclass(frozen=True) class WorkspaceLimits: - max_bytes: int = 1024 * 1024 * 1024 + max_bytes: int = SHARED_CONSTANTS["microvm_continuation"]["max_workspace_bytes"] max_entries: int = 100_000 git_timeout_s: int = 60 @@ -342,6 +343,8 @@ def _file_digest(path: Path) -> str: def _repository_identity(identity: CheckpointIdentity) -> None: + if identity.repo == "": + return # Repository-free tasks use a private Git baseline with no remote. if not _REPO.fullmatch(identity.repo) or any( part in {".", ".."} for part in identity.repo.split("/") ): @@ -744,10 +747,17 @@ def _restore_workspace( index = _git(staging, ["ls-files", "--stage", "-z"], limits) if hashlib.sha256(index).hexdigest() != state["index_sha256"]: _fail("invalid_archive", "Workspace staged state was not restored exactly") - _git( - staging, ["remote", "add", "origin", f"https://github.com/{identity.repo}.git"], limits - ) - _git(staging, ["config", "--local", "credential.helper", "!gh auth git-credential"], limits) + if identity.repo: + _git( + staging, + ["remote", "add", "origin", f"https://github.com/{identity.repo}.git"], + limits, + ) + _git( + staging, + ["config", "--local", "credential.helper", "!gh auth git-credential"], + limits, + ) if state["exclude_b64"] is not None: ignore_file = staging / ".git/info/exclude" ignore_file.parent.mkdir(exist_ok=True) diff --git a/agent/src/hooks.py b/agent/src/hooks.py index b5b5d409c..44a92a9dd 100644 --- a/agent/src/hooks.py +++ b/agent/src/hooks.py @@ -66,7 +66,7 @@ class _ApprovalDeadline: """One gate's deadline, retained across polling and a possible VM sleep. - UTC counts time while the guest's monotonic clock is frozen. The monotonic + UTC counts elapsed time even if the guest's monotonic clock freezes. The monotonic cap prevents a backward UTC correction from extending the original window. Create this with the approval row, before database writes or notifications; never recreate it on resume. @@ -77,6 +77,8 @@ class _ApprovalDeadline: @classmethod def from_recorded(cls, created_at: str, timeout_s: int) -> _ApprovalDeadline: + if timeout_s == 0: + return cls(float("inf"), float("inf")) created_epoch = ( datetime.strptime(created_at, "%Y-%m-%dT%H:%M:%SZ").replace(tzinfo=UTC).timestamp() ) @@ -532,6 +534,25 @@ async def pre_tool_use_hook( return _deny_response(decision.reason) # -- REQUIRE_APPROVAL path ---------------------------------------------- + from continuation_runtime import current_runtime + + continuation = current_runtime() + saved_decision = ( + continuation.consume_approved_action(tool_name, _sha256_tool_input_for_row(tool_input)) + if continuation is not None + else None + ) + if saved_decision is not None: + if progress is not None: + _try_progress( + progress, + "write_approval_granted", + request_id=saved_decision["request_id"], + scope=saved_decision.get("scope") or "this_call", + decided_at=saved_decision.get("decided_at"), + created_at=saved_decision.get("created_at"), + ) + return _allow_response("User approved this saved action before worker continuation") return await _handle_require_approval( decision=decision, tool_name=tool_name, @@ -542,6 +563,7 @@ async def pre_tool_use_hook( progress=progress, ts=ts_module, tool_use_id=tool_use_id, + session_id=hook_input.get("session_id", ""), ) @@ -556,6 +578,7 @@ async def _handle_require_approval( progress: Any, ts: Any, tool_use_id: str | None = None, + session_id: str = "", ) -> dict: """REQUIRE_APPROVAL branch of ``pre_tool_use_hook``. @@ -601,7 +624,11 @@ async def _handle_require_approval( # Step 3 — effective timeout with floor/ceiling math. Emit # ``approval_timeout_capped`` when the caller's ask is clipped so the # user can see why. - remaining_lifetime = _remaining_maxlifetime_s() + # A durable checkpoint can move to a replacement worker. Its request's + # deadline is independent of this process's remaining service lifetime. + from continuation_runtime import current_runtime + + remaining_lifetime = None if current_runtime() is not None else _remaining_maxlifetime_s() effective_timeout, clip_reason, requested_timeout = _compute_effective_timeout( decision_timeout_s=decision.timeout_s, task_default_timeout_s=engine.task_default_timeout_s, @@ -659,10 +686,18 @@ async def _handle_require_approval( "status": "PENDING", "created_at": _iso_now(), "timeout_s": effective_timeout, - "ttl": int(time.time()) + effective_timeout + CLEANUP_MARGIN_120S, "user_id": user_id or "", "repo": engine.repo, } + if effective_timeout > 0: + row["deadline_epoch"] = ( + int( + datetime.strptime(row["created_at"], "%Y-%m-%dT%H:%M:%SZ") + .replace(tzinfo=UTC) + .timestamp() + ) + + effective_timeout + ) deadline = _ApprovalDeadline.from_recorded(row["created_at"], effective_timeout) # Step 6 — bump counters BEFORE the write so cap/rate checks on @@ -752,6 +787,44 @@ async def _handle_require_approval( if lifecycle else None ) + from continuation_runtime import current_runtime + + continuation = current_runtime() + checkpoint_held = False + if continuation is not None and lifecycle is not None and park is not None: + try: + identity, receipt = await continuation.capture( + lifecycle, + park, + session_id=session_id, + tool_name=tool_name, + tool_input=tool_input, + approval_scopes=engine.allowlist.snapshot_scopes(), + approval_gate_count=engine.approval_gate_count, + ) + checkpoint_held = True + await asyncio.to_thread( + ts.publish_continuation_checkpoint, + identity, + receipt, + tool_input_sha256=tool_input_sha256, + cost_usd=continuation.context.cost_usd, + turns_used=continuation.context.turns_used, + ) + log( + "TASK", + f"Continuation checkpoint ready task_id={task_id} request_id={request_id} " + f"attempt_id={identity.attempt_id}", + ) + except Exception as exc: + # A failed or uncertain publish cannot authorize worker release. + # If a receipt was saved, keep tools held until the conditional + # RUNNING claim resolves any race with the coordinator. + log( + "WARN", + f"Continuation checkpoint unavailable task_id={task_id} request_id={request_id} " + f"code={getattr(exc, 'code', 'checkpoint_failed')} error_type={type(exc).__name__}", + ) try: outcome = await _poll_for_decision( task_id=task_id, @@ -760,8 +833,12 @@ async def _handle_require_approval( progress=progress, ts=ts, ) + except BaseException: + if checkpoint_held and lifecycle: + lifecycle.close() + raise finally: - if lifecycle and park: + if lifecycle and park and not checkpoint_held: # Clear the safe point before ANY approval/task-state mutation or # hook return. A concurrent suspend owns the barrier until resume. await lifecycle.leave_approval(park) @@ -805,6 +882,8 @@ async def _handle_require_approval( if outcome.get("status") == "CANCELLED": # The platform closed the approval in the same transaction as task # cancellation. Do not try to restore RUNNING or cache a human denial. + if checkpoint_held and lifecycle: + lifecycle.close() return _deny_response("Task was cancelled; this approval is closed.") # Step 11 — resume transition (RUNNING). The ``awaiting_approval_request_id`` @@ -820,6 +899,8 @@ async def _handle_require_approval( request_id=request_id, error=f"cancelled: {exc.cancellation_reasons}", ) + if checkpoint_held and lifecycle: + lifecycle.close() return _deny_response("task no longer awaiting approval") except Exception as exc: log("WARN", f"approval resume raised: {type(exc).__name__}: {exc}") @@ -830,8 +911,13 @@ async def _handle_require_approval( request_id=request_id, error=f"{type(exc).__name__}: {exc}", ) + if checkpoint_held and lifecycle: + lifecycle.close() return _deny_response("approval resume failed") + if checkpoint_held and lifecycle and park: + await lifecycle.release_continuation(park) + # Step 12 — terminal branches. status = outcome.get("status") if status == "APPROVED": @@ -1123,6 +1209,8 @@ def _compute_effective_timeout( decision_value = ( decision_timeout_s if decision_timeout_s is not None else task_default_timeout_s ) + if decision_value == 0: + return 0, None, requested # Start with the decision value (already clipped by rule vs task default # in the engine) and apply the remaining-lifetime ceiling here. @@ -1953,13 +2041,9 @@ async def _stop( # The callback transport must outlive the approval loop. A frozen MicroVM # may wake after the gate deadline during supervisor recovery, so keep that - # callback alive for the bounded VM lifetime. The original gate deadline - # still decides permission; this does not give the user more approval time. - callback_window_s = ( - SHARED_CONSTANTS["microvm_lifecycle"]["maximum_duration_seconds"] - if lifecycle - else SHARED_CONSTANTS["approval_timeout_s"]["max"] - ) + # callback alive for the bounded VM lifetime. Explicit gate deadlines remain + # unchanged; a retained request can outlive this callback through continuation. + callback_window_s = SHARED_CONSTANTS["microvm_lifecycle"]["maximum_duration_seconds"] matchers = { "PreToolUse": [ HookMatcher( diff --git a/agent/src/microvm_checkpoint.py b/agent/src/microvm_checkpoint.py index 95e2e06f6..c51c81a08 100644 --- a/agent/src/microvm_checkpoint.py +++ b/agent/src/microvm_checkpoint.py @@ -27,14 +27,13 @@ def _integer(value: Any) -> bool: ) -def _record(park: ApprovalPark) -> tuple[ApprovalRecord, int]: +def _record(park: ApprovalPark) -> tuple[ApprovalRecord, int | None]: record = park.record if ( record is None or not record.user_id or not isinstance(record.repo, str) or not _integer(record.timeout_s) - or record.timeout_s <= 0 ): raise LifecycleUnavailable("Original approval identity is unavailable") try: @@ -43,12 +42,15 @@ def _record(park: ApprovalPark) -> tuple[ApprovalRecord, int]: raise LifecycleUnavailable("Original approval timestamp is invalid") from exc if created.strftime("%Y-%m-%dT%H:%M:%SZ") != record.created_at: raise LifecycleUnavailable("Original approval timestamp is not canonical") - return record, int(created.timestamp() * 1000) + record.timeout_s * 1000 + deadline_ms = ( + int(created.timestamp() * 1000) + record.timeout_s * 1000 if record.timeout_s > 0 else None + ) + return record, deadline_ms def _read( park: ApprovalPark, action: Literal["suspend", "resume"] -) -> tuple[Any, str, str, dict, int]: +) -> tuple[Any, str, str, dict, int | None]: """Read current task/gate identity strongly; absence and mismatch fail closed.""" from boto3.dynamodb.types import TypeDeserializer from botocore.config import Config @@ -98,7 +100,8 @@ def read_item(table: str, key: dict) -> dict: or intent.get("request_id") != park.request_id or intent.get("action") != action or not _integer(intent.get("requested_at_ms")) - or not _integer(intent.get("deadline_ms")) + or "deadline_ms" not in intent + or (intent["deadline_ms"] is not None and not _integer(intent["deadline_ms"])) or intent["deadline_ms"] != deadline_ms ): raise LifecycleUnavailable("Coordinator lifecycle intent does not match this approval") diff --git a/agent/src/microvm_lifecycle.py b/agent/src/microvm_lifecycle.py index c40447134..8f1f393ff 100644 --- a/agent/src/microvm_lifecycle.py +++ b/agent/src/microvm_lifecycle.py @@ -73,11 +73,12 @@ class MicrovmLifecycle: can commit a transition, and its generation check rejects late completion. """ - def __init__(self, task_id: str, microvm_id: str) -> None: + def __init__(self, task_id: str, microvm_id: str, *, attempt_id: str | None = None) -> None: if not task_id or not microvm_id: raise ValueError("Lifecycle requires task and MicroVM identity") self.task_id = task_id self.microvm_id = microvm_id + self.attempt_id = attempt_id or task_id self._lock = threading.Lock() self._tools: set[str] = set() self._park: ApprovalPark | None = None @@ -88,6 +89,7 @@ def __init__(self, task_id: str, microvm_id: str) -> None: self._suspend_ineligible = False self._slept_request_id: str | None = None self._last_resume_park: ApprovalPark | None = None + self._continuation_park: ApprovalPark | None = None def _check_open(self) -> bool: if self._phase in {"closed", "failed"}: @@ -172,6 +174,11 @@ def park_approval( async def leave_approval(self, park: ApprovalPark) -> None: """Remove the safe point *before* the hook changes task state/returns.""" + with self._lock: + if self._continuation_park is park: + raise LifecycleUnavailable( + "Continuation must claim task ownership before releasing tools" + ) while True: await self.wait_until_open() with self._lock: @@ -184,15 +191,23 @@ async def leave_approval(self, park: ApprovalPark) -> None: return @contextmanager - def activity(self): + def activity(self, *, checkpoint_safe: bool = False): """Drain synchronous progress and heartbeat work before suspension. Best-effort writers must explicitly report missing/failed acknowledgments with progress_write_failed(). A later successful event cannot recover a dropped earlier event, so that failure remains latched for the task. + Progress events may finish during a continuation upload: they do not + modify the workspace, and capture drains their acknowledgments again + before publishing. Other activity remains paused during that upload. """ with self._lock: - if not self._check_open(): + capture_progress = checkpoint_safe and self._phase == "checkpointing" + if ( + not self._check_open() + and self._phase != "checkpoint-ready" + and not capture_progress + ): # Do not silently write with pre-wake credentials. The progress # writer catches this like its existing best-effort failures. raise LifecycleUnavailable("Guest activity is paused") @@ -207,12 +222,11 @@ def activity(self): async def approval_poll(self): """Pause new approval reads and drain an already-running read safely.""" while True: - await self.wait_until_open() with self._lock: - if not self._check_open(): - continue - self._activities += 1 - break + if self._check_open() or self._phase == "checkpoint-ready": + self._activities += 1 + break + await asyncio.sleep(0.02) try: yield finally: @@ -229,6 +243,86 @@ def _budget(seconds: float) -> float: raise ValueError("Lifecycle budget must be finite and positive") return time.monotonic() + seconds + @asynccontextmanager + async def continuation_checkpoint(self, park: ApprovalPark, *, drain_budget_s: float = 5): + """Hold the sole pending tool while saving a replacement-worker checkpoint. + + This runs on the SDK loop before decision polling starts. It is separate + from the short HTTP suspend hook: a full workspace transfer may take + minutes. Cancellation keeps the barrier closed because a cancelled + ``to_thread`` capture can still be reading the workspace. + """ + end = self._budget(drain_budget_s) + with self._lock: + if ( + self._park is not park + or self._phase != "parked" + or self._tools != {park.tool_use_id} + or self._suspend_ineligible + or self._progress_failed + ): + raise LifecycleUnavailable( + "Task is not safely parked for a continuation checkpoint" + ) + self._phase = "checkpointing" + self._generation += 1 + generation = self._generation + cancelled = False + complete = False + try: + while True: + with self._lock: + self._assert_transition(generation, "checkpointing") + if self._progress_failed: + raise LifecycleUnavailable("Progress was not acknowledged") + drained = self._activities == 0 + if drained: + break + if time.monotonic() >= end: + raise TimeoutError("Progress did not drain before continuation capture") + await asyncio.sleep(0.02) + yield + # SDK progress can arrive while the full workspace upload awaits + # a thread. Do not drop it or publish before its write is known. + # The upload can take minutes, so this drain gets its own budget. + end = self._budget(drain_budget_s) + while True: + with self._lock: + self._assert_transition(generation, "checkpointing") + if self._progress_failed: + raise LifecycleUnavailable("Progress was not acknowledged") + if self._activities == 0: + self._continuation_park = park + complete = True + break + if time.monotonic() >= end: + raise TimeoutError("Progress did not drain after continuation capture") + await asyncio.sleep(0.02) + except BaseException as exc: + cancelled = not isinstance(exc, Exception) + raise + finally: + with self._lock: + if self._generation == generation and self._phase == "checkpointing": + self._phase = ( + "failed" if cancelled else ("checkpoint-ready" if complete else "parked") + ) + self._generation += 1 + + async def release_continuation(self, park: ApprovalPark) -> None: + """Open tools only after a conditional RUNNING claim by this same worker.""" + while True: + with self._lock: + self._check_open() + if self._continuation_park is not park or self._park is not park: + raise LifecycleUnavailable("Continuation approval changed") + if self._phase == "checkpoint-ready": + self._continuation_park = None + self._park = None + self._phase = "active" + return + await asyncio.sleep(0.02) + async def suspend( self, checkpoint: Callable[[ApprovalPark], None], *, budget_s: float ) -> ApprovalPark: @@ -252,7 +346,7 @@ async def suspend( return park if ( park is None - or self._phase != "parked" + or self._phase not in {"parked", "checkpoint-ready"} or self._suspend_ineligible or park.request_id == self._slept_request_id or self._progress_failed @@ -289,7 +383,9 @@ async def suspend( # progress may finish later; new suspension stays off after failure. with self._lock: if self._generation == generation and self._phase == "suspending": - self._phase = "parked" + self._phase = ( + "checkpoint-ready" if self._continuation_park is park else "parked" + ) self._suspend_ineligible = True self._generation += 1 raise @@ -307,7 +403,10 @@ async def resume( end = self._budget(budget_s) with self._lock: park = self._park - if self._phase in {"active", "parked"} and self._last_resume_park is not None: + if ( + self._phase in {"active", "parked", "checkpoint-ready"} + and self._last_resume_park is not None + ): # A duplicate wake acknowledgment cannot renew the approval # timeout or re-run credential refresh on an executing task. return self._last_resume_park @@ -322,7 +421,7 @@ async def resume( reseed_random() with self._lock: self._assert_transition(generation, "resuming") - self._phase = "parked" + self._phase = "checkpoint-ready" if self._continuation_park is park else "parked" # One sleep per approval gate. A duplicate suspend must not # race the newly released decision loop. self._slept_request_id = park.request_id @@ -354,6 +453,7 @@ def close(self) -> None: self._phase = "closed" self._generation += 1 self._park = None + self._continuation_park = None self._tools.clear() @@ -361,11 +461,13 @@ def close(self) -> None: _contexts: dict[str, MicrovmLifecycle] = {} -def register_task(task_id: str, microvm_id: str) -> MicrovmLifecycle: +def register_task( + task_id: str, microvm_id: str, *, attempt_id: str | None = None +) -> MicrovmLifecycle: with _registry_lock: if _contexts: raise LifecycleUnavailable("A MicroVM pipeline is already registered") - context = MicrovmLifecycle(task_id, microvm_id) + context = MicrovmLifecycle(task_id, microvm_id, attempt_id=attempt_id) _contexts[task_id] = context return context diff --git a/agent/src/payload_bootstrap.py b/agent/src/payload_bootstrap.py index b5133731b..84b39e999 100644 --- a/agent/src/payload_bootstrap.py +++ b/agent/src/payload_bootstrap.py @@ -91,7 +91,7 @@ def _manifest(uri: str, backend: str) -> tuple[str, dict]: return parsed.netloc, manifest["platform_config"] -def _payload_url(url: str, bucket: str, task_id: str) -> None: +def _payload_url(url: str, bucket: str, task_id: str, attempt_id: str | None = None) -> None: """Permit only the exact object's regional S3 HTTPS endpoint, without redirects.""" try: parsed = urlsplit(url) @@ -116,9 +116,10 @@ def _payload_url(url: str, bucket: str, task_id: str) -> None: raise ValueError suffix = "amazonaws.com.cn" if region.startswith("cn-") else "amazonaws.com" host = f"s3.{region}.{suffix}" + prefix = f"{task_id}/{attempt_id}" if attempt_id is not None else task_id valid = ( - parsed.netloc == f"{bucket}.{host}" and parsed.path == f"/{task_id}/payload.json" - ) or (parsed.netloc == host and parsed.path == f"/{bucket}/{task_id}/payload.json") + parsed.netloc == f"{bucket}.{host}" and parsed.path == f"/{prefix}/payload.json" + ) or (parsed.netloc == host and parsed.path == f"/{bucket}/{prefix}/payload.json") if not valid: raise ValueError except (KeyError, IndexError, TypeError, ValueError): @@ -152,6 +153,7 @@ def resolve_payload_reference(reference: Any, backend: str) -> tuple[dict, dict] "payload bootstrap v2 is required; deploy a matching coordinator and image" ) task_id = reference.get("task_id") + attempt_id = reference.get("attempt_id") uri = reference.get("bootstrap_s3_uri") url = reference.get("payload_url") if ( @@ -159,10 +161,18 @@ def resolve_payload_reference(reference: Any, backend: str) -> tuple[dict, dict] or not re.fullmatch(r"[A-Za-z0-9_-]{1,128}", task_id) or not isinstance(uri, str) or not isinstance(url, str) + or ( + attempt_id is not None + and ( + backend != "lambda-microvm" + or not isinstance(attempt_id, str) + or not re.fullmatch(r"[A-Za-z0-9_-]{1,128}", attempt_id) + ) + ) ): raise ValueError("payload reference is missing task identity or download coordinates") bucket, config = _manifest(uri, backend) - _payload_url(url, bucket, task_id) + _payload_url(url, bucket, task_id, attempt_id) document = _download(url) payload = document.get("agent_payload") if ( @@ -170,6 +180,7 @@ def resolve_payload_reference(reference: Any, backend: str) -> tuple[dict, dict] or document.get("task_id") != task_id or not isinstance(payload, dict) or payload.get("task_id") != task_id + or (attempt_id is not None and payload.get("attempt_id") != attempt_id) ): raise ValueError("downloaded payload does not belong to the referenced task") if document.get("platform_config") != config: diff --git a/agent/src/pipeline.py b/agent/src/pipeline.py index 709301d7c..482259155 100644 --- a/agent/src/pipeline.py +++ b/agent/src/pipeline.py @@ -214,6 +214,9 @@ def _execute_agent_step( hydrated, trajectory, progress, + *, + continuation=None, + started_reaction_id=None, ): """Run the agentic step through the workflow step runner. @@ -249,7 +252,22 @@ def _execute_agent_step( # and post-hooks stay on the inline path. only_kinds keeps the runner from # re-running the deterministic steps the pipeline already owns (double clone # / double PR). - result = run_workflow(wf, ctx, only_kinds={"run_agent"}) + from continuation_runtime import bind_runtime, create_runtime + from continuation_storage import ContinuationContext + + runtime = continuation or create_runtime( + ContinuationContext( + setup, + prompt, + system_prompt, + wf.id, + wf.version, + started_reaction_id=started_reaction_id, + ), + config.task_id, + ) + with bind_runtime(runtime): + result = run_workflow(wf, ctx, only_kinds={"run_agent"}) if ctx.agent_result is None: # The run_agent step did not produce a result — i.e. its handler raised @@ -299,6 +317,21 @@ def _run_repoless_task( workflow_id = (config.resolved_workflow or {}).get("id", "default/agent-v1") wf = load_workflow(workflow_id) system_prompt = build_repoless_system_prompt(config, hc, system_prompt_overrides) + from continuation_runtime import bind_runtime, prepare_repoless_runtime + from microvm_lifecycle import get_context as get_microvm_context + + continuation = prepare_repoless_runtime( + config, + user_prompt=prompt, + system_prompt=system_prompt, + workflow_id=wf.id, + workflow_version=wf.version, + ) + if continuation is not None and continuation.restored is not None: + if continuation.resume_prompt is None: + raise RuntimeError("Restored continuation has no recorded decision prompt") + prompt = continuation.resume_prompt + system_prompt = continuation.context.system_prompt ctx = StepContext( workflow=wf, @@ -306,7 +339,7 @@ def _run_repoless_task( hydrated=hc, progress=progress, trajectory=trajectory, - setup=None, # repo-less: no RepoSetup + setup=continuation.context.setup if continuation is not None else None, system_prompt=system_prompt, user_prompt=prompt, ) @@ -314,8 +347,12 @@ def _run_repoless_task( # deliver_artifact. The deliverer uploads the agent's result text to # artifacts/{task_id}/ (and/or surfaces it as a comment), so the declared # terminal outcome is actually produced (#248 Phase 3). - with task_span("task.agent_execution"): + with bind_runtime(continuation), task_span("task.agent_execution"): wf_result = run_workflow(wf, ctx) + lifecycle = get_microvm_context(config.task_id) + if lifecycle and lifecycle.diagnostic_snapshot()["phase"] in {"closed", "failed"}: + log("TASK", "Worker execution is closed; coordinator owns continuation or cleanup.") + return {"task_id": config.task_id, "status": "parked"} agent_result = ctx.agent_result if agent_result is None: @@ -903,6 +940,7 @@ def run_task( repo=config.repo_url, task_id=config.task_id, ) + task_state.verify_worker_lease(config.task_id) # Surface the credential-scoping posture once per task so every task's logs # state plainly whether tenant-data isolation was active. is_scoped() # resolves the session; if scoping was requested but unbuildable it raises @@ -1058,6 +1096,9 @@ def _on_trace_truncated(max_bytes: int, first_dropped: int) -> None: os.environ["GIT_COMMITTER_EMAIL"] = "bgagent@noreply.github.com" os.environ["GITHUB_TOKEN"] = config.github_token os.environ["GH_TOKEN"] = config.github_token + from continuation_runtime import restore_for_task + + continuation = restore_for_task(config) # Set env vars for the prepare-commit-msg hook BEFORE setup_repo() # so the hook has access to TASK_ID/PROMPT_VERSION from the start. @@ -1094,17 +1135,23 @@ def _on_trace_truncated(max_bytes: int, first_dropped: int) -> None: # pr-review) never transition — the orchestration panel owns the # parent's state, and a planning run shouldn't advance the issue. linear_transition_state = not config.read_only - linear_eyes_reaction_id = react_task_started( - config.channel_source, - config.channel_metadata, - transition_state=linear_transition_state, + linear_eyes_reaction_id = ( + continuation.context.started_reaction_id + if continuation is not None + else react_task_started( + config.channel_source, + config.channel_metadata, + transition_state=linear_transition_state, + ) ) # "Starting" comment on the Jira issue through the Forge app actor # (or legacy OAuth fallback). No-op for non-Jira tasks. # Best-effort; failures are logged, never block. workflow_id = (config.resolved_workflow or {}).get("id", "coding/new-task-v1") - if _should_post_start_comment(config.channel_source, workflow_id): + if continuation is None and _should_post_start_comment( + config.channel_source, workflow_id + ): comment_task_started( config.channel_source, config.channel_metadata, @@ -1116,10 +1163,11 @@ def _on_trace_truncated(max_bytes: int, first_dropped: int) -> None: # Part of the Early-ACK block (moved before setup_repo with the 👀 # and start comment) so board state updates immediately, not after # the multi-minute baseline build. - transition_task_started( - config.channel_source, - config.channel_metadata, - ) + if continuation is None: + transition_task_started( + config.channel_source, + config.channel_metadata, + ) # Setup repo (deterministic pre-hooks). A failure/timeout/OOM in the # pre-agent baseline build raises here; it needs no local handler — @@ -1130,7 +1178,13 @@ def _on_trace_truncated(max_bytes: int, first_dropped: int) -> None: # with no visible signal; posting the 👀 earlier is what makes the # outer handler's ❌-swap actually visible for setup failures. with task_span("task.repo_setup") as setup_span: - setup = setup_repo(config, progress=progress) + if continuation is None: + setup = setup_repo(config, progress=progress) + else: + from repo import prepare_restored_repo + + setup = continuation.context.setup + prepare_restored_repo(setup.repo_dir) setup_span.set_attribute("build.before", setup.build_before) progress.write_agent_milestone( "repo_setup_complete", @@ -1178,9 +1232,11 @@ def _on_trace_truncated(max_bytes: int, first_dropped: int) -> None: "asset (ADR-016 — the agent must have no Linear tools)", ) - # Download attachments from S3 (version-pinned, integrity-verified) + # Recovery already restored .attachments and the prompt's exact + # references. Re-downloading would depend on the original attachment + # bucket's shorter retention and could overwrite saved local work. prepared_attachments: list = [] - if config.attachments: + if config.attachments and continuation is None: from attachments import download_attachments try: @@ -1221,6 +1277,17 @@ def _on_trace_truncated(max_bytes: int, first_dropped: int) -> None: if prepared_attachments: prompt = _inject_attachment_context(prompt, prepared_attachments) + if continuation is not None: + if continuation.resume_prompt is None: + raise RuntimeError("Restored continuation has no recorded decision prompt") + prompt = continuation.resume_prompt + system_prompt = continuation.context.system_prompt + progress.write_agent_milestone( + "continuation_restored", + "Saved conversation and workspace restored; " + "continuing the recorded human decision.", + ) + # Run agent disk_before = get_disk_usage(AGENT_WORKSPACE) start_time = time.time() @@ -1247,6 +1314,8 @@ def _on_trace_truncated(max_bytes: int, first_dropped: int) -> None: hc, trajectory, progress, + continuation=continuation, + started_reaction_id=linear_eyes_reaction_id, ) except Exception as e: # Fatal agent error: mirror to APPLICATION_LOGS so @@ -1259,6 +1328,15 @@ def _on_trace_truncated(max_bytes: int, first_dropped: int) -> None: agent_span.set_status(StatusCode.ERROR, str(e)) agent_span.record_exception(e) agent_result = AgentResult(status="error", error=str(e)) + from microvm_lifecycle import get_context as get_microvm_context + + lifecycle = get_microvm_context(config.task_id) + if lifecycle and lifecycle.diagnostic_snapshot()["phase"] in {"closed", "failed"}: + # A transferred/cancelled worker may finish unwinding its SDK. + # It must not run deterministic commit/PR/channel post-hooks. + log("TASK", "Worker execution is closed; coordinator owns continuation or cleanup.") + return {"task_id": config.task_id, "status": "parked"} + progress.write_agent_milestone( "agent_execution_complete", f"status={agent_result.status} turns={agent_result.turns}", diff --git a/agent/src/policy.py b/agent/src/policy.py index a18aa3842..6cdf62420 100644 --- a/agent/src/policy.py +++ b/agent/src/policy.py @@ -12,9 +12,9 @@ 2. Approval allowlist fast-path (tool_type, tool_group, bash_pattern, write_path, all_session scopes). Skips human prompt for pre-approved patterns. - 2.5 Recent-decision cache: same (tool_name, input_sha256) within 60s of a - DENIED/TIMED_OUT outcome auto-denies. Session-scoped, cleared on - container restart (§12.8). + 2.5 Recent-decision cache: the same (tool_name, input_sha256) auto-denies + for 60s after inserting a DENIED/TIMED_OUT outcome. Ordinary restarts + clear it; continuation restores the saved decision before SDK startup. 3. Soft-deny Cedar eval (agent/policies/soft_deny.cedar + blueprint soft). Match → REQUIRE_APPROVAL with merged annotations; rule-scope allowlist match → ALLOW; no match → fall through to step 4. @@ -26,10 +26,11 @@ ``context.file_path``, because Cedar entity UIDs cannot contain arbitrary characters. -**Annotations** expected on every rule in hard_deny/soft_deny files -(§5.2): ``@rule_id`` (globally unique, kebab/snake_case), ``@tier`` -("hard"|"soft"), ``@approval_timeout_s`` (int seconds ≥ 30; soft-deny -only), ``@severity`` ("low"|"medium"|"high"; soft-deny only), ``@category`` +**Annotations** used by rules in hard_deny/soft_deny files (§5.2): +``@rule_id`` (globally unique, kebab/snake_case), ``@tier`` +("hard"|"soft"), optional ``@approval_timeout_s`` (int seconds ≥ 30; +soft-deny only; omission uses the task setting), ``@severity`` +("low"|"medium"|"high"; soft-deny only), ``@category`` (free-form; UX grouping). Annotation recovery goes through cedarpy's ``policies_to_json_str()``; the round-trip contract is locked by ``tests/test_cedarpy_annotations_contract.py``. @@ -95,7 +96,7 @@ def _validate_constants() -> None: raise ValueError( f"contracts/constants.json: approval_timeout_s.min must be > 0, got {FLOOR_TIMEOUT_S}" ) - if DEFAULT_TASK_TIMEOUT_S < FLOOR_TIMEOUT_S: + if DEFAULT_TASK_TIMEOUT_S != 0 and DEFAULT_TASK_TIMEOUT_S < FLOOR_TIMEOUT_S: raise ValueError( f"contracts/constants.json: approval_timeout_s.default ({DEFAULT_TASK_TIMEOUT_S}) " f"must be >= min ({FLOOR_TIMEOUT_S})" @@ -399,6 +400,19 @@ def rule_ids(self) -> frozenset[str]: """Snapshot of rule-ID scopes, checked post-soft-deny in the engine.""" return frozenset(self._rule_ids) + def snapshot_scopes(self) -> tuple[str, ...]: + """Export normalized session grants for an acknowledged continuation.""" + scopes = ["all_session"] if self._all_session else [] + for prefix, values in ( + ("tool_type", self._tool_types), + ("tool_group", self._tool_groups), + ("rule", self._rule_ids), + ("bash_pattern", self._bash_patterns), + ("write_path", self._write_path_patterns), + ): + scopes.extend(f"{prefix}:{value}" for value in sorted(set(values))) + return tuple(scopes) + def matches(self, tool_name: str, tool_input: dict) -> bool: """Return True if a non-rule scope pre-approves this tool call.""" if self._all_session: @@ -438,8 +452,9 @@ class RecentDecisionCache: DENIED/TIMED_OUT — NEVER on APPROVED (so a just-approved call does not auto-deny on the next identical invocation). - **Session-scoped**: cleared on container restart. Documented caveat in - §12.8 — not a bug. Persistent cache is §17.5 future work. + **Session-scoped**: ordinary process restarts clear this cache. A saved + MicroVM continuation seeds its recorded denial or timeout during restoration; + it does not serialize the entire cache. """ def __init__( @@ -698,15 +713,16 @@ def _merge_annotations( if rule is None: continue rule_ids.append(rule.rule_id or pid) - if rule.approval_timeout_s is not None: + if rule.approval_timeout_s is not None and rule.approval_timeout_s > 0: timeouts.append(rule.approval_timeout_s) severities.append(rule.severity or _DEFAULT_SEVERITY) - timeouts.append(task_default_timeout_s) + if task_default_timeout_s > 0: + timeouts.append(task_default_timeout_s) # Defensive only: load-time validation already rejects below-floor values, # but the clamp costs nothing and protects against a future caller that # bypasses validation (e.g. programmatic rule injection). - effective_timeout = max(FLOOR_TIMEOUT_S, min(timeouts)) + effective_timeout = max(FLOOR_TIMEOUT_S, min(timeouts)) if timeouts else 0 if severities: effective_severity = max(severities, key=lambda s: _SEVERITY_ORDER.get(s, 0)) @@ -1220,7 +1236,10 @@ def evaluate_tool_use(self, tool_name: str, tool_input: dict) -> PolicyDecision: # engine pure — policy.py never calls the progress writer. return PolicyDecision( outcome=Outcome.DENY, - reason=f"Recent {cached.decision} within {int(CACHE_TTL_S)}s: {cached.reason}", + reason=( + f"Recorded {cached.decision} at {cached.original_decision_ts}: " + f"{cached.reason}" + ), duration_ms=(time.monotonic() - start) * 1000, cache_hit_metadata={ "tool_name": tool_name, @@ -1273,8 +1292,8 @@ def evaluate_tool_use(self, tool_name: str, tool_input: dict) -> PolicyDecision: return PolicyDecision( outcome=Outcome.DENY, reason=( - f"Recent {cached.decision} on rule {matched_rule_id!r} " - f"within {int(CACHE_TTL_S)}s: {cached.reason}" + f"Recorded {cached.decision} on rule {matched_rule_id!r} " + f"at {cached.original_decision_ts}: {cached.reason}" ), duration_ms=(time.monotonic() - start) * 1000, cache_hit_metadata={ diff --git a/agent/src/progress_writer.py b/agent/src/progress_writer.py index ff2fdc1e8..13a651724 100644 --- a/agent/src/progress_writer.py +++ b/agent/src/progress_writer.py @@ -502,7 +502,7 @@ def _put_event(self, event_type: str, metadata: dict) -> None: self._put_event_best_effort(event_type, metadata) return try: - with lifecycle.activity(): + with lifecycle.activity(checkpoint_safe=True): acknowledged = False try: acknowledged = self._put_event_best_effort(event_type, metadata) @@ -511,7 +511,12 @@ def _put_event(self, event_type: str, metadata: dict) -> None: lifecycle.progress_write_failed() except LifecycleUnavailable: lifecycle.progress_write_failed() - print("[progress] lifecycle barrier closed — event not acknowledged", flush=True) + print( + f"[progress] lifecycle barrier closed — event not acknowledged " + f"task_id={self._task_id} event_type={event_type} " + f"phase={lifecycle.diagnostic_snapshot()['phase']}", + flush=True, + ) def _put_event_best_effort(self, event_type: str, metadata: dict) -> bool: """Write a single progress event item to DynamoDB. diff --git a/agent/src/repo.py b/agent/src/repo.py index 19f2aa7ec..6b8010202 100644 --- a/agent/src/repo.py +++ b/agent/src/repo.py @@ -672,6 +672,20 @@ def _merge_predecessor_branch(repo_dir: str, pred_branch: str, notes: list[str]) log("SETUP", f"Predecessor merge conflicted, aborted: {pred_branch}") +def prepare_restored_repo(repo_dir: str) -> None: + """Recreate trusted tools/hooks without cloning or replacing saved build baselines.""" + run_cmd( + ["git", "config", "--global", "--add", "safe.directory", repo_dir], + label="safe-directory", + ) + for cfg in [repo_dir, *_find_mise_configs(repo_dir)]: + run_cmd(["mise", "trust", cfg], label="mise-trust-restored", cwd=repo_dir, check=False) + result = run_cmd(["mise", "install"], label="mise-install-restored", cwd=repo_dir, check=False) + if result.returncode != 0: + log("WARN", f"Restored workspace mise install failed (exit {result.returncode})") + _install_commit_hook(repo_dir) + + def _install_commit_hook(repo_dir: str) -> None: """Install the prepare-commit-msg git hook for Task-Id/Prompt-Version trailers.""" try: diff --git a/agent/src/runner.py b/agent/src/runner.py index 1d765d630..bb86e897e 100644 --- a/agent/src/runner.py +++ b/agent/src/runner.py @@ -293,6 +293,11 @@ def _initialize_policy_engine_and_hooks( extra_policies=cedar_policies if cedar_policies else None, **engine_kwargs, ) + from continuation_runtime import current_runtime + + continuation = current_runtime() + if continuation is not None: + continuation.seed_policy(policy_engine) # Surface the resolved cap + its source so operators can distinguish a # blueprint-threaded value from the engine's compile-time default on a # container restart. Mirrors the ``approval_gate_cap_source`` field on the @@ -634,13 +639,38 @@ def _on_stderr(line: str) -> None: **({"mcp_servers": mcp_servers} if mcp_servers else {}), ) - result = AgentResult() + from continuation_runtime import current_runtime + + continuation = current_runtime() + if continuation is not None: + options.session_store = continuation.store + options.session_store_flush = "eager" + if continuation.restored is not None: + options.resume = continuation.restored["session_id"] + options.max_turns = config.max_turns - continuation.context.turns_used + if options.max_turns <= 0: + raise RuntimeError("Task turn limit was reached before the saved continuation") + if config.max_budget_usd is not None: + options.max_budget_usd = config.max_budget_usd - continuation.prior_cost_usd + if options.max_budget_usd <= 0: + raise RuntimeError( + "Task dollar budget was reached before the saved continuation" + ) + + prior_turns = ( + continuation.context.turns_used + if continuation is not None and continuation.restored is not None + else 0 + ) + result = AgentResult(turns=prior_turns) message_counts = {"system": 0, "assistant": 0, "result": 0, "other": 0} # Use ClaudeSDKClient (connect/query/receive_response) instead of the # standalone query() function. This matches the official AWS sample: # https://github.com/aws-samples/sample-deploy-ClaudeAgentSDK-based-agents-to-AgentCore-Runtime client = ClaudeSDKClient(options=options) + if continuation is not None: + continuation.client = client from microvm_credentials import ScopedCredentialBroker from microvm_lifecycle import get_context @@ -758,7 +788,9 @@ def _on_stderr(line: str) -> None: subtype = getattr(message, "subtype", "unknown") result.status = subtype result.cost_usd = getattr(message, "total_cost_usd", None) - result.num_turns = getattr(message, "num_turns", 0) + if result.cost_usd is not None and continuation is not None: + result.cost_usd += continuation.prior_cost_usd + result.num_turns = getattr(message, "num_turns", 0) + prior_turns result.duration_ms = getattr(message, "duration_ms", 0) result.duration_api_ms = getattr(message, "duration_api_ms", 0) result.session_id = getattr(message, "session_id", "") or "" @@ -790,6 +822,13 @@ def _on_stderr(line: str) -> None: if raw_usage is not None: # Handle both object (dataclass) and dict forms usage = _parse_token_usage(raw_usage) + if continuation is not None: + usage = TokenUsage( + **{ + key: count + continuation.prior_token_usage.get(key, 0) + for key, count in usage.model_dump().items() + } + ) result.usage = usage if all(v == 0 for v in usage.model_dump().values()): log( @@ -808,8 +847,8 @@ def _on_stderr(line: str) -> None: log( "DONE", - f"status={result.status} turns={message.num_turns} " - f"cost=${message.total_cost_usd or 0:.4f} " + f"status={result.status} turns={result.num_turns} " + f"cost=${result.cost_usd or 0:.4f} " f"duration={message.duration_ms / 1000:.1f}s", ) if message.is_error and message.result: @@ -822,8 +861,8 @@ def _on_stderr(line: str) -> None: # Write trajectory result summary (use effective status after is_error remap) trajectory.write_result( subtype=result.status, - num_turns=getattr(message, "num_turns", 0), - cost_usd=getattr(message, "total_cost_usd", None), + num_turns=result.num_turns, + cost_usd=result.cost_usd, duration_ms=getattr(message, "duration_ms", 0), duration_api_ms=getattr(message, "duration_api_ms", 0), session_id=getattr(message, "session_id", ""), @@ -834,10 +873,10 @@ def _on_stderr(line: str) -> None: input_toks = usage.input_tokens if usage else 0 output_toks = usage.output_tokens if usage else 0 progress.write_agent_cost_update( - cost_usd=getattr(message, "total_cost_usd", None), + cost_usd=result.cost_usd, input_tokens=input_toks, output_tokens=output_toks, - turn=getattr(message, "num_turns", 0), + turn=result.num_turns, ) elif isinstance(message, UserMessage): diff --git a/agent/src/server.py b/agent/src/server.py index 9077d5a09..4091f086b 100644 --- a/agent/src/server.py +++ b/agent/src/server.py @@ -428,6 +428,7 @@ def _run_task_background( attachments: list[dict] | None = None, resolved_assets: list[dict] | None = None, microvm_id: str = "", + attempt_id: str = "", ) -> None: """Run the agent task in a background thread.""" global _background_pipeline_failed @@ -467,7 +468,9 @@ def _run_task_background( task_id=task_id, ) - lifecycle = register_task(task_id, microvm_id) if microvm_id else None + lifecycle = ( + register_task(task_id, microvm_id, attempt_id=attempt_id or task_id) if microvm_id else None + ) stop_heartbeat = threading.Event() hb_thread: threading.Thread | None = None try: @@ -844,29 +847,11 @@ async def invoke_agent(request: Request, body: InvocationRequest): # -------------------------------------------------------------------------- # AWS Lambda MicroVMs lifecycle hooks (ADR-021 P1 through P3) # -------------------------------------------------------------------------- -# The MicroVM backend has NO orchestrator→agent HTTP path: the task payload -# arrives as the ``/run`` hook's request body and nothing else dials in. The -# service calls these routes on the port declared in the image's ``hooks.port`` -# (8080 — the same uvicorn process that serves /invocations and /ping), so the -# hooks live here rather than in a sidecar. -# -# Six hooks are served and declared in managed images. The supervisor checks the -# launched version's capability and rollout settings before requesting suspension: -# * ``/ready`` (build, P1) is MANDATORY. ``CreateMicrovmImage`` refuses an image -# that enables ANY lifecycle hook without it ("The ready (/ready) MicroVM -# image hook must be enabled when any MicroVM lifecycle hook … is enabled"), -# and an image with no hooks at all cannot receive a ``runHookPayload``. So -# ADR-021's original "declare /run in P1, serve it in P2" split was not a -# reachable service state. -# * ``/run`` (runtime, P1) is the payload-delivery channel — and, since P2, the -# platform-configuration channel (see ``platform_config`` below). -# * ``/validate`` (build, P2) is the snapshot self-check. It runs under the -# BUILD role and makes ZERO AWS calls — see ``microvm_validate``. -# * ``/terminate`` (runtime, P2) is a best-effort final flush. It never writes -# terminal task status — the orchestrator owns terminal state. -# ``/suspend`` and ``/resume`` use a drained, acknowledged checkpoint and mandatory -# credential/gate reconciliation. The image must declare them with the shared -# service budget before it can advertise lifecycle capability. +# The service calls six hooks on the image listener. Build hooks warm/check the +# snapshot without AWS access; /run authenticates configuration and starts work. +# /terminate closes the coding barrier and acknowledges teardown. /suspend and +# /resume checkpoint and reconcile credentials/gates. Automatic suspension also +# requires coordinator approval of the launched image version and live settings. MICROVM_HOOK_PREFIX = "/aws/lambda-microvms/runtime/v1" app.add_api_route(f"{MICROVM_HOOK_PREFIX}/suspend", microvm_suspend, methods=["POST"]) @@ -1791,35 +1776,13 @@ def microvm_validate(): @app.post(f"{MICROVM_HOOK_PREFIX}/terminate") async def microvm_terminate(request: Request): - """MicroVM ``/terminate`` runtime hook — best-effort flush, always 200. - - Called as the MicroVM is torn down. Three hard constraints: - - * **It must not write terminal task status.** The orchestrator owns terminal - state: it finalizes the task and THEN calls ``TerminateMicrovm``, so a - terminate hook that wrote ``FAILED``/``COMPLETED`` would race the - finalization it follows and could clobber the real outcome with a - substrate-shutdown artifact. The pipeline thread's own crash path - (``_run_task_background``) remains the only in-guest terminal writer. - * **It must return 200 inside the hook budget, even with nothing running.** - So it never joins the pipeline thread (a drain could take minutes — that is - ``lifespan``'s job on a graceful shutdown) and every best-effort step is - wrapped: a failure here must not turn a clean teardown into a hook failure. - * **It must return 200 for any BODY too.** That is why this handler takes the - raw ``Request`` instead of a Pydantic body model: FastAPI validates a typed - body BEFORE the handler runs, so malformed JSON, a wrong content-type, or a - missing body would produce a 422 this function never gets a chance to - prevent — a reported hook failure on a successful teardown. Parsing is - deferred to ``_parse_terminate_microvm_id``, which degrades to ``""``. - - ``async def`` (unlike ``/ready`` and ``/run``) because reading the raw body - requires awaiting it. Safe on the event loop: the work is a JSON parse, a - thread-count read and a fire-and-forget log — no blocking AWS call. - - There is no progress queue to drain. ``ProgressWriter`` writes synchronously - but catches and drops failures; a return from its event method is not proof - of durability. This hook closes the coding barrier, logs and acknowledges - teardown. ``/suspend`` uses a separate acknowledged checkpoint transaction. + """Close the coding barrier, log teardown and acknowledge any request body. + + This hook never joins the pipeline or writes terminal task status. Termination + can interrupt active work, so it cannot assume finalization already finished. + Raw Request parsing avoids FastAPI rejecting malformed bodies before entry. + Each best-effort step is guarded; acknowledged checkpointing belongs to + /suspend, not this hook. """ # Close the local barrier before reading the body or emitting diagnostics. # A slow checkpoint/refresh thread must not release coding during teardown. @@ -2005,9 +1968,17 @@ def microvm_run(request: Request, body: MicrovmRunHookRequest): }, ) - reseed_random() - _spawn_background({**params, "microvm_id": body.microvmId}) + # The attempt token comes only from the authenticated bootstrap payload. + # It identifies coordinator authority before RunMicrovm returns a VM ID. task_id = params["task_id"] + attempt_id = payload.get("attempt_id", task_id) + if not isinstance(attempt_id, str) or not re.fullmatch(r"[A-Za-z0-9_-]{1,128}", attempt_id): + return JSONResponse( + status_code=400, + content={"code": "MICROVM_ATTEMPT_ID_INVALID", "message": "Invalid worker attempt"}, + ) + reseed_random() + _spawn_background({**params, "microvm_id": body.microvmId, "attempt_id": attempt_id}) # Carries microvm_id as well as task_id: the "/run hook received" line that # used to correlate the two is stdout-only now (pre-install), so this is the # first line that reaches the task's log group and it has to join the CloudWatch diff --git a/agent/src/task_state.py b/agent/src/task_state.py index a9ac64bd9..22bdde2fe 100644 --- a/agent/src/task_state.py +++ b/agent/src/task_state.py @@ -8,7 +8,7 @@ import os import time -from typing import TypedDict +from typing import NotRequired, TypedDict from shell import log, log_error_cw @@ -35,7 +35,8 @@ class ApprovalRow(TypedDict): status: str # always 'PENDING' on initial write. created_at: str timeout_s: int - ttl: int + deadline_epoch: NotRequired[int] + ttl: NotRequired[int] user_id: str repo: str @@ -65,6 +66,66 @@ def _now_iso() -> str: return time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()) +def _lease_check(task_id: str, *, low_level: bool, identity=None) -> dict | None: + """Read coordinator authority without granting workers permission to rewrite it.""" + from microvm_lifecycle import get_context + from shared_constants import SHARED_CONSTANTS + + lifecycle = get_context(task_id) + if lifecycle is None or not os.environ.get("CONTINUATION_BUCKET_NAME"): + return None + values = {":attempt": lifecycle.attempt_id, ":active": "ACTIVE"} + condition = "lease_attempt_id = :attempt AND lease_state = :active" + if identity is not None: + values.update( + {":user": identity.user_id, ":repo": identity.repo, ":vm": lifecycle.microvm_id} + ) + condition += " AND lease_user_id = :user AND lease_repo = :repo AND lease_microvm_id = :vm" + task_table, _ = _require_tables() + key = {"task_id": SHARED_CONSTANTS["microvm_continuation"]["lease_key_prefix"] + task_id} + if low_level: + key = {k: _py_to_ddb_attr(v) for k, v in key.items()} + values = {k: _py_to_ddb_attr(v) for k, v in values.items()} + return { + "ConditionCheck": { + "TableName": task_table, + "Key": key, + "ConditionExpression": condition, + "ExpressionAttributeValues": values, + } + } + + +def _transact_task(client, task_id: str, *, TransactItems: list, identity=None): + lease = _lease_check(task_id, low_level=True, identity=identity) + return client.transact_write_items(TransactItems=[*TransactItems, *([lease] if lease else [])]) + + +def _update_task(table, task_id: str, *, low_level: bool = False, **operation): + lease = _lease_check(task_id, low_level=low_level) + if lease is None: + return table.update_item(**operation) + if not low_level: + operation["TableName"] = table.name + client = table if low_level else table.meta.client + return client.transact_write_items(TransactItems=[{"Update": operation}, lease]) + + +def _task_status_conflict(error: Exception) -> bool: + """An expected status race is benign only if worker ownership still passed.""" + from botocore.exceptions import ClientError + + if not isinstance(error, ClientError): + return False + code = error.response.get("Error", {}).get("Code") + if code == "ConditionalCheckFailedException": + return True + reasons = error.response.get("CancellationReasons") or [] + return code == "TransactionCanceledException" and [ + reason.get("Code") for reason in reasons + ] == ["ConditionalCheckFailed", "None"] + + def _build_logs_url(task_id: str) -> str | None: """Build a CloudWatch Logs console URL filtered to this task_id.""" region = os.environ.get("AWS_REGION") or os.environ.get("AWS_DEFAULT_REGION") @@ -80,13 +141,43 @@ def _build_logs_url(task_id: str) -> str | None: ) +def verify_worker_lease(task_id: str, *, client=None) -> None: + """Reject a superseded launch before repository code or SDK tools can execute.""" + from boto3.dynamodb.types import TypeDeserializer + + from microvm_lifecycle import get_context + from shared_constants import SHARED_CONSTANTS + + lifecycle = get_context(task_id) + if lifecycle is None or not os.environ.get("CONTINUATION_BUCKET_NAME"): + return + table, _ = _require_tables() + result = _get_ddb_client(client=client).get_item( + TableName=table, + Key={ + "task_id": {"S": SHARED_CONSTANTS["microvm_continuation"]["lease_key_prefix"] + task_id} + }, + ConsistentRead=True, + ) + raw = result.get("Item") or {} + deserialize = TypeDeserializer().deserialize + lease = {key: deserialize(value) for key, value in raw.items()} + if ( + lease.get("lease_state") != "ACTIVE" + or lease.get("lease_attempt_id") != lifecycle.attempt_id + ): + raise RuntimeError("MICROVM_LEASE_LOST: this launch no longer owns task execution") + + def write_heartbeat(task_id: str) -> None: """Update ``agent_heartbeat_at`` while the task is RUNNING (orchestrator crash detection).""" try: table = _get_table() if table is None: return - table.update_item( + _update_task( + table, + task_id, Key={"task_id": task_id}, UpdateExpression="SET agent_heartbeat_at = :t", ConditionExpression="#s = :running", @@ -94,12 +185,7 @@ def write_heartbeat(task_id: str) -> None: ExpressionAttributeValues={":t": _now_iso(), ":running": "RUNNING"}, ) except Exception as e: - from botocore.exceptions import ClientError - - if ( - isinstance(e, ClientError) - and e.response.get("Error", {}).get("Code") == "ConditionalCheckFailedException" - ): + if _task_status_conflict(e): return log("WARN", f"[task_state] write_heartbeat failed (best-effort): {type(e).__name__}: {e}") @@ -134,7 +220,9 @@ def write_running(task_id: str) -> None: update_parts.append("logs_url = :logs") expr_values[":logs"] = logs_url - table.update_item( + _update_task( + table, + task_id, Key={"task_id": task_id}, UpdateExpression="SET " + ", ".join(update_parts), ConditionExpression="#s IN (:submitted, :hydrating)", @@ -142,12 +230,7 @@ def write_running(task_id: str) -> None: ExpressionAttributeValues=expr_values, ) except Exception as e: - from botocore.exceptions import ClientError - - if ( - isinstance(e, ClientError) - and e.response.get("Error", {}).get("Code") == "ConditionalCheckFailedException" - ): + if _task_status_conflict(e): log("INFO", "[task_state] write_running skipped: status precondition not met") return log("WARN", f"[task_state] write_running failed (best-effort): {type(e).__name__}") @@ -264,7 +347,9 @@ def write_terminal(task_id: str, status: str, result: dict | None = None) -> Non update_parts.append("artifact_uri = :au") expr_values[":au"] = result["artifact_uri"] - table.update_item( + _update_task( + table, + task_id, Key={"task_id": task_id}, UpdateExpression="SET " + ", ".join(update_parts), ConditionExpression="#s IN (:running, :hydrating, :finalizing, :awaiting_approval)", @@ -341,7 +426,9 @@ def write_trace_uri_conditional(task_id: str, uri: str) -> bool: table = _get_table() if table is None: return False - table.update_item( + _update_task( + table, + task_id, Key={"task_id": task_id}, UpdateExpression="SET trace_s3_uri = :ts3", ConditionExpression=( @@ -389,7 +476,7 @@ class TaskFetchError(Exception): """ -def get_task(task_id: str) -> dict | None: +def get_task(task_id: str, *, consistent_read: bool = False) -> dict | None: """Fetch a task record by ID. Returns: @@ -412,7 +499,9 @@ def get_task(task_id: str) -> dict | None: if table is None: return None try: - resp = table.get_item(Key={"task_id": task_id}) + resp = table.get_item( + Key={"task_id": task_id}, **({"ConsistentRead": True} if consistent_read else {}) + ) except Exception as e: log("WARN", f"[task_state] get_task failed: {type(e).__name__}: {e}") raise TaskFetchError(f"{type(e).__name__}: {e}") from e @@ -608,7 +697,9 @@ def transact_write_approval_request( approval_item.setdefault("status", {"S": "PENDING"}) try: - ddb.transact_write_items( + _transact_task( + ddb, + task_id, TransactItems=[ { "Put": { @@ -633,7 +724,7 @@ def transact_write_approval_request( }, } }, - ] + ], ) except Exception as exc: # TransactionCanceledException carries per-item reasons. Keep the @@ -675,7 +766,9 @@ def transact_resume_from_approval( ddb = _get_ddb_client(client=client) try: - ddb.transact_write_items( + _transact_task( + ddb, + task_id, TransactItems=[ { "Update": { @@ -683,7 +776,7 @@ def transact_resume_from_approval( "Key": {"task_id": {"S": task_id}}, "UpdateExpression": ( "SET #s = :running, agent_heartbeat_at = :heartbeat " - "REMOVE awaiting_approval_request_id" + "REMOVE awaiting_approval_request_id, continuation" ), "ConditionExpression": ( "#s = :awaiting AND awaiting_approval_request_id = :rid" @@ -697,7 +790,7 @@ def transact_resume_from_approval( }, } } - ] + ], ) except Exception as exc: reasons = _extract_cancellation_reasons(exc) @@ -710,6 +803,187 @@ def transact_resume_from_approval( raise +def publish_continuation_checkpoint( + identity, + receipt, + *, + tool_input_sha256: str, + cost_usd: float | None = None, + turns_used: int | None = None, + client=None, +) -> dict: + """Publish only the same worker's still-pending, fully saved approval checkpoint.""" + from dataclasses import asdict + from decimal import Decimal + + from boto3.dynamodb.types import TypeSerializer + + from continuation_storage import StorageLimits + from continuation_usage import valid_cost + from shared_constants import SHARED_CONSTANTS + + serialize = TypeSerializer().serialize + receipt.validate(identity, StorageLimits()) + if receipt.kind != "manifest": + raise ValueError("Only a complete continuation manifest may be published") + task_table, approvals_table = _require_tables() + ddb = _get_ddb_client(client=client) + record = { + "version": SHARED_CONSTANTS["microvm_continuation"]["version"], + "state": "READY", + "identity": asdict(identity), + "manifest": asdict(receipt), + } + values = { + ":request": identity.request_id, + ":awaiting": _STATUS_AWAITING_APPROVAL, + ":record": record, + ":consumed": "CONSUMED", + } + update = "SET continuation = :record" + if cost_usd is not None: + if not valid_cost(cost_usd): + raise ValueError("Continuation cost must be finite and nonnegative") + values[":cost"] = Decimal(str(cost_usd)) + update += ", cost_usd = :cost" + if turns_used is not None: + if type(turns_used) is not int or turns_used < 0: + raise ValueError("Continuation turns must be a nonnegative integer") + values[":turns"] = turns_used + update += ", turns = :turns" + _transact_task( + ddb, + identity.task_id, + identity=identity, + TransactItems=[ + { + "Update": { + "TableName": task_table, + "Key": {"task_id": {"S": identity.task_id}}, + "UpdateExpression": update, + "ConditionExpression": ( + "#status = :awaiting AND " + "awaiting_approval_request_id = :request AND " + "(attribute_not_exists(continuation) OR continuation = :record OR " + "continuation.#state = :consumed)" + ), + "ExpressionAttributeNames": {"#status": "status", "#state": "state"}, + "ExpressionAttributeValues": {k: serialize(v) for k, v in values.items()}, + } + }, + { + "ConditionCheck": { + "TableName": approvals_table, + "Key": { + "task_id": {"S": identity.task_id}, + "request_id": {"S": identity.request_id}, + }, + "ConditionExpression": ( + "user_id = :user AND repo = :repo AND #status = :pending " + "AND tool_input_sha256 = :hash" + ), + "ExpressionAttributeNames": {"#status": "status"}, + "ExpressionAttributeValues": { + ":user": {"S": identity.user_id}, + ":repo": {"S": identity.repo}, + ":pending": {"S": "PENDING"}, + ":hash": {"S": tool_input_sha256}, + }, + } + }, + ], + ) + return record + + +def consume_restored_continuation( + task_id: str, worker_id: str, record: dict, *, client=None +) -> None: + """Claim RUNNING after restoration, without deciding or revalidating the human action.""" + from boto3.dynamodb.types import TypeSerializer + + serialize = TypeSerializer().serialize + from continuation_session import CheckpointIdentity + + task_table, approvals_table = _require_tables() + identity = record["identity"] + if record.get("state") != "RESTORING" or record.get("worker_id") != worker_id: + raise ValueError("Continuation is not owned by this replacement worker") + values = { + ":request": identity["request_id"], + ":record": record, + ":awaiting": _STATUS_AWAITING_APPROVAL, + ":running": _STATUS_RUNNING, + ":consumed": "CONSUMED", + ":heartbeat": _now_iso(), + } + ddb = _get_ddb_client(client=client) + items = [ + { + "Update": { + "TableName": task_table, + "Key": {"task_id": {"S": task_id}}, + "UpdateExpression": ( + "SET #status = :running, agent_heartbeat_at = :heartbeat, " + "continuation.#state = :consumed REMOVE awaiting_approval_request_id" + ), + "ConditionExpression": ( + "continuation = :record " + "AND #status = :awaiting AND awaiting_approval_request_id = :request" + ), + "ExpressionAttributeNames": {"#status": "status", "#state": "state"}, + "ExpressionAttributeValues": {k: serialize(v) for k, v in values.items()}, + } + }, + { + "ConditionCheck": { + "TableName": approvals_table, + "Key": { + "task_id": {"S": task_id}, + "request_id": {"S": identity["request_id"]}, + }, + "ConditionExpression": ( + "user_id = :user AND #status IN (:approved, :denied, :timedout)" + ), + "ExpressionAttributeNames": {"#status": "status"}, + "ExpressionAttributeValues": { + ":user": {"S": identity["user_id"]}, + ":approved": {"S": "APPROVED"}, + ":denied": {"S": "DENIED"}, + ":timedout": {"S": "TIMED_OUT"}, + }, + } + }, + ] + try: + _transact_task(ddb, task_id, identity=CheckpointIdentity(**identity), TransactItems=items) + except Exception: + # A transport error can lose the response after DynamoDB commits. Only + # acknowledge that exact assignment, still RUNNING under this lease. + # Cancellation or a newer assignment must never become a successful retry. + from boto3.dynamodb.types import TypeDeserializer + + deserialize = TypeDeserializer().deserialize + try: + response = ddb.get_item( + TableName=task_table, + Key={"task_id": {"S": task_id}}, + ConsistentRead=True, + ) + task = {key: deserialize(value) for key, value in response.get("Item", {}).items()} + expected = {**record, "state": "CONSUMED"} + if ( + task.get("status") != _STATUS_RUNNING + or task.get("session_id") != worker_id + or task.get("continuation") != expected + or task.get("awaiting_approval_request_id") is not None + ): + raise RuntimeError("Continuation claim was not acknowledged") + verify_worker_lease(task_id, client=ddb) + except Exception: + raise + + def best_effort_update_approval_status( task_id: str, request_id: str, @@ -827,7 +1101,10 @@ def increment_approval_gate_count_in_ddb( try: ddb = _get_ddb_client(client=client) - ddb.update_item( + _update_task( + ddb, + task_id, + low_level=True, TableName=task_table, Key={"task_id": {"S": task_id}}, UpdateExpression="ADD approval_gate_count :one", diff --git a/agent/src/workflow/runner.py b/agent/src/workflow/runner.py index ba7a7bc31..5fffc5b61 100644 --- a/agent/src/workflow/runner.py +++ b/agent/src/workflow/runner.py @@ -407,9 +407,11 @@ def _handle_hydrate_context(step: Step, ctx: StepContext) -> StepOutcome: Hydration is largely orchestrator-side today (WORKFLOWS.md open question #4 leans "orchestrator hydrates, the agent step only consumes"); this handler is - that consumer. It sets BOTH prompts the ``run_agent`` step needs: + that consumer. It fills missing prompts for the ``run_agent`` step: - - ``ctx.user_prompt`` — the hydrated ``user_prompt`` when present. + - ``ctx.user_prompt`` — the hydrated ``user_prompt`` when the caller has + not already prepared one. A restored human decision or attachment context + must survive this step. - ``ctx.system_prompt`` — built via the existing ``build_system_prompt`` so the workflow path produces the same system prompt as ``pipeline.run_task`` (repo_url/branch/workspace/max_turns/setup_notes/memory_context + overrides @@ -418,7 +420,7 @@ def _handle_hydrate_context(step: Step, ctx: StepContext) -> StepOutcome: the ``RepoSetup``; when absent (repo-less workflows) the system prompt is left to the caller, since ``build_system_prompt`` is repo-shaped today. """ - if ctx.hydrated is not None: + if ctx.hydrated is not None and not ctx.user_prompt: ctx.user_prompt = ctx.hydrated.user_prompt built_system_prompt = False @@ -476,6 +478,11 @@ def _handle_run_agent(step: Step, ctx: StepContext) -> StepOutcome: ) ) ctx.agent_result = result + from microvm_lifecycle import get_context + + lifecycle = get_context(ctx.config.task_id) + if lifecycle and lifecycle.diagnostic_snapshot()["phase"] in {"closed", "failed"}: + raise RuntimeError("Worker execution is closed; remaining workflow steps must not run") # The agent loop "failing" is not a step failure here: pipeline's # _resolve_overall_task_status owns success inference. The step succeeds if # the SDK ran; downstream steps and the terminal-outcome check decide done. diff --git a/agent/tests/test_approval_retention.py b/agent/tests/test_approval_retention.py new file mode 100644 index 000000000..ba872ab6f --- /dev/null +++ b/agent/tests/test_approval_retention.py @@ -0,0 +1,37 @@ +"""Approval deadlines are explicit; compute lifetime never creates one.""" + +import math + +from hooks import _ApprovalDeadline, _compute_effective_timeout +from policy import PolicyEngine + + +def test_default_builtin_gate_has_no_automatic_deadline(): + engine = PolicyEngine(task_type="new_task", repo="owner/repo") + decision = engine.evaluate_tool_use("Bash", {"command": "git push --force origin feature"}) + assert decision.outcome == "require_approval" + assert decision.timeout_s == 0 + + +def test_zero_deadline_survives_a_long_pause(monkeypatch): + deadline = _ApprovalDeadline.from_recorded("2026-01-01T00:00:00Z", 0) + monkeypatch.setattr("hooks.time.time", lambda: 9_999_999_999) + monkeypatch.setattr("hooks.time.monotonic", lambda: 9_999_999_999) + assert math.isinf(deadline.remaining_s()) + assert _compute_effective_timeout( + decision_timeout_s=0, task_default_timeout_s=0, remaining_lifetime_s=10 + ) == (0, None, 0) + + +def test_explicit_rule_deadline_survives_a_no_deadline_task_default(): + engine = PolicyEngine( + task_type="new_task", + repo="owner/repo", + blueprint_soft_policies=( + '@tier("soft") @rule_id("explicit") @approval_timeout_s("120") ' + 'forbid (principal, action == Agent::Action::"execute_bash", resource) ' + 'when { context.command like "*explicit-tool*" };' + ), + ) + decision = engine.evaluate_tool_use("Bash", {"command": "explicit-tool"}) + assert decision.timeout_s == 120 diff --git a/agent/tests/test_config.py b/agent/tests/test_config.py index 7e1b1331f..8cb904506 100644 --- a/agent/tests/test_config.py +++ b/agent/tests/test_config.py @@ -199,7 +199,7 @@ class TestResolveLinearApiToken: The orchestrator stamps `linear_oauth_secret_arn` into the task's channel_metadata at creation time. resolve_linear_api_token reads the secret JSON via boto3, refreshes it if expiring, and caches the - access_token in `LINEAR_API_TOKEN` for the Linear MCP placeholder. + access_token in `LINEAR_API_TOKEN` for direct Linear API calls. """ def test_returns_cached_value_without_calling_secrets_manager(self, monkeypatch): @@ -226,6 +226,44 @@ def test_returns_empty_when_region_missing(self, monkeypatch): assert resolve_linear_api_token({"linear_oauth_secret_arn": "arn:test"}) == "" mock_boto.assert_not_called() + @pytest.mark.parametrize( + "payload", + [ + {"workspace_id": "ws", "provider_name": "bgagent-linear-oauth-acme"}, + {"refresh_token": None, "client_id": "cid", "client_secret": "secret"}, + {"refresh_token": "rt", "client_id": "", "client_secret": "secret"}, + {"refresh_token": "rt", "client_id": "cid"}, + ], + ) + def test_incomplete_fallback_skips_refresh_without_crashing(self, monkeypatch, payload): + monkeypatch.delenv("LINEAR_API_TOKEN", raising=False) + monkeypatch.delenv("LINEAR_VAULT_ENABLED", raising=False) + monkeypatch.setenv("AWS_REGION", "us-east-1") + mock_sm = MagicMock() + mock_sm.get_secret_value.return_value = { + "SecretString": __import__("json").dumps(payload), + } + with ( + patch("boto3.client", return_value=mock_sm), + patch("urllib.request.urlopen") as post, + patch("config.log") as log, + ): + assert resolve_linear_api_token({"linear_oauth_secret_arn": "arn:test"}) == "" + post.assert_not_called() + assert any( + "linear_oauth_refresh_unavailable" in call.args[1] for call in log.call_args_list + ) + + @pytest.mark.parametrize("payload", ["null", "[]", '"text"']) + def test_non_object_fallback_returns_empty(self, monkeypatch, payload): + monkeypatch.delenv("LINEAR_API_TOKEN", raising=False) + monkeypatch.delenv("LINEAR_VAULT_ENABLED", raising=False) + monkeypatch.setenv("AWS_REGION", "us-east-1") + mock_sm = MagicMock() + mock_sm.get_secret_value.return_value = {"SecretString": payload} + with patch("boto3.client", return_value=mock_sm): + assert resolve_linear_api_token({"linear_oauth_secret_arn": "arn:test"}) == "" + def test_resolves_from_secrets_manager_and_caches_in_env(self, monkeypatch): """Happy path: channel_metadata carries the ARN, secret has access_token + future expiry.""" from datetime import datetime, timedelta diff --git a/agent/tests/test_continuation_runtime.py b/agent/tests/test_continuation_runtime.py new file mode 100644 index 000000000..a3f059516 --- /dev/null +++ b/agent/tests/test_continuation_runtime.py @@ -0,0 +1,476 @@ +# Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. +# SPDX-License-Identifier: MIT-0 + +"""A complete checkpoint must recover after losing both process and workspace.""" + +from __future__ import annotations + +import asyncio +import shutil +import subprocess +from dataclasses import asdict +from typing import Any +from unittest.mock import MagicMock + +import pytest + +from continuation_runtime import ContinuationRuntime, restore_for_task +from continuation_storage import ContinuationContext, S3ContinuationStorage +from microvm_lifecycle import ApprovalRecord, LifecycleUnavailable, MicrovmLifecycle +from models import RepoSetup, TaskConfig +from tests.test_continuation_storage import VersionedS3 + +SESSION = "11111111-1111-4111-8111-aaaaaaaaaaaa" + + +class Deadline: + def remaining_s(self) -> float: + return 5000 + + +def git(workspace, *args): + return subprocess.run( + ["git", "-C", str(workspace), *args], + check=True, + capture_output=True, + text=True, + ).stdout.strip() + + +@pytest.fixture +def ready(tmp_path, monkeypatch): + from continuation_usage import UsageSnapshot + + async def usage(_): + return UsageSnapshot(0.02, {"input_tokens": 100, "output_tokens": 10}) + + monkeypatch.setattr("continuation_runtime.read_usage", usage) + workspace = tmp_path / "task" + workspace.mkdir() + git(workspace, "init", "-b", "main") + git(workspace, "config", "user.name", "Test") + git(workspace, "config", "user.email", "test@example.test") + (workspace / "tracked.txt").write_text("committed\n") + git(workspace, "add", ".") + git(workspace, "commit", "-m", "initial") + original_head = git(workspace, "rev-parse", "HEAD") + (workspace / "tracked.txt").write_text("staged edit\n") + git(workspace, "add", ".") + (workspace / "tracked.txt").write_text("unstaged edit\n") + (workspace / "untracked.txt").write_text("keep this work\n") + context = ContinuationContext( + RepoSetup( + repo_dir=str(workspace), + branch="main", + build_before=False, + head_sha_before=original_head, + ), + "Original user prompt", + "Original system prompt", + "coding/new-task-v1", + "1", + ) + client = VersionedS3() + runtime = ContinuationRuntime(context, S3ContinuationStorage("owned-bucket", client=client)) + return workspace, runtime + + +async def park_runtime(runtime, repo="owner/repo"): + lifecycle = MicrovmLifecycle("task", "microvm-one") + await lifecycle.tool_started("toolu_pending") + park = lifecycle.park_approval( + "request", + "toolu_pending", + Deadline(), + record=ApprovalRecord("user", repo, "2026-09-17T00:00:00Z", 5000), + ) + assert park is not None + tool_input = {"file_path": runtime.context.setup.repo_dir + "/untracked.txt"} + entries: Any = [ + { + "type": "assistant", + "uuid": "entry-one", + "sessionId": SESSION, + "message": { + "content": [ + { + "type": "tool_use", + "id": park.tool_use_id, + "name": "Read", + "input": tool_input, + } + ] + }, + } + ] + await runtime.store.append( + {"project_key": runtime.store.project_key, "session_id": SESSION}, entries + ) + return lifecycle, park, tool_input + + +class TestCompleteRecovery: + def test_restore_preserves_git_baseline_conversation_and_pending_action(self, ready): + workspace, runtime = ready + before_status = git(workspace, "status", "--porcelain=v1") + before_index = git(workspace, "diff", "--cached") + + async def save(): + lifecycle, park, tool_input = await park_runtime(runtime) + identity, receipt = await runtime.capture( + lifecycle, park, session_id=SESSION, tool_name="Read", tool_input=tool_input + ) + assert lifecycle.diagnostic_snapshot()["phase"] == "checkpoint-ready" + lifecycle.close() + return identity, receipt + + identity, receipt = asyncio.run(save()) + shutil.rmtree(workspace) + restored = ContinuationRuntime.restore( + runtime.storage, + receipt, + identity, + expected_workspace=workspace, + workflow_id="coding/new-task-v1", + workflow_version="1", + ) + assert restored.context == runtime.context + assert git(workspace, "status", "--porcelain=v1") == before_status + assert git(workspace, "diff", "--cached") == before_index + assert (workspace / "untracked.txt").read_text() == "keep this work\n" + assert restored.restored is not None + assert restored.restored["session_id"] == SESSION + prompt = restored.decision_prompt(decision="DENIED", reason="Use another approach") + assert "DENIED" in prompt and "Use another approach" in prompt + assert str(workspace / "untracked.txt") in prompt + assert "normal permission checks" in prompt + + def test_later_parallel_tool_cannot_modify_a_published_workspace(self, ready): + _, runtime = ready + + async def scenario(): + lifecycle, park, tool_input = await park_runtime(runtime) + await runtime.capture( + lifecycle, park, session_id=SESSION, tool_name="Read", tool_input=tool_input + ) + later = asyncio.create_task(lifecycle.tool_started("toolu_later")) + await asyncio.sleep(0.03) + assert not later.done() + # Reads/feedback remain available while tools are held. + async with lifecycle.approval_poll(): + pass + with lifecycle.activity(): + pass + with pytest.raises(LifecycleUnavailable, match="claim task ownership"): + await lifecycle.leave_approval(park) + await lifecycle.release_continuation(park) + await asyncio.wait_for(later, 1) + assert lifecycle.diagnostic_snapshot()["phase"] == "active" + + asyncio.run(scenario()) + + def test_sleep_and_wake_do_not_release_continuation_tools(self, ready): + _, runtime = ready + + async def scenario(): + lifecycle, park, tool_input = await park_runtime(runtime) + await runtime.capture( + lifecycle, park, session_id=SESSION, tool_name="Read", tool_input=tool_input + ) + await lifecycle.suspend(lambda _: None, budget_s=1) + await lifecycle.resume(lambda _: None, budget_s=1) + assert lifecycle.diagnostic_snapshot()["phase"] == "checkpoint-ready" + await lifecycle.release_continuation(park) + assert lifecycle.diagnostic_snapshot()["phase"] == "active" + + asyncio.run(scenario()) + + def test_failed_capture_keeps_original_approval_usable(self, ready, monkeypatch): + _, runtime = ready + + def fail(*_): + raise RuntimeError("owned storage failure") + + monkeypatch.setattr(runtime, "_save", fail) + + async def scenario(): + lifecycle, park, tool_input = await park_runtime(runtime) + with pytest.raises(RuntimeError, match="storage failure"): + await runtime.capture( + lifecycle, park, session_id=SESSION, tool_name="Read", tool_input=tool_input + ) + assert lifecycle.diagnostic_snapshot()["phase"] == "parked" + await lifecycle.leave_approval(park) + + asyncio.run(scenario()) + + def test_cancellation_never_opens_tools_while_capture_thread_may_still_run(self, ready): + _, runtime = ready + + async def scenario(): + lifecycle, park, _ = await park_runtime(runtime) + entered = asyncio.Event() + + async def capture(): + async with lifecycle.continuation_checkpoint(park): + entered.set() + await asyncio.Event().wait() + + task = asyncio.create_task(capture()) + await entered.wait() + task.cancel() + with pytest.raises(asyncio.CancelledError): + await task + assert lifecycle.diagnostic_snapshot()["phase"] == "failed" + with pytest.raises(LifecycleUnavailable): + await lifecycle.tool_started("toolu_later") + + asyncio.run(scenario()) + + @pytest.mark.parametrize("mismatch", ["workspace", "workflow", "version"]) + def test_wrong_restore_context_is_rejected_before_creating_files( + self, ready, mismatch, tmp_path + ): + workspace, runtime = ready + + async def save(): + lifecycle, park, tool_input = await park_runtime(runtime) + return await runtime.capture( + lifecycle, park, session_id=SESSION, tool_name="Read", tool_input=tool_input + ) + + identity, receipt = asyncio.run(save()) + shutil.rmtree(workspace) + with pytest.raises(RuntimeError, match="workspace or workflow"): + ContinuationRuntime.restore( + runtime.storage, + receipt, + identity, + expected_workspace=tmp_path / "another" if mismatch == "workspace" else workspace, + workflow_id="another" if mismatch == "workflow" else "coding/new-task-v1", + workflow_version="2" if mismatch == "version" else "1", + ) + assert not workspace.exists() + assert not (tmp_path / "another").exists() + + +@pytest.fixture +def replacement(ready, monkeypatch): + from continuation_session import decode_checkpoint + + workspace, runtime = ready + + async def save(): + lifecycle, park, tool_input = await park_runtime(runtime) + result = await runtime.capture( + lifecycle, park, session_id=SESSION, tool_name="Read", tool_input=tool_input + ) + lifecycle.close() + return result + + identity, receipt = asyncio.run(save()) + manifest = runtime.storage.load_manifest(receipt, identity) + checkpoint = runtime.storage.conversations.load(manifest.conversation, identity) + action = decode_checkpoint(checkpoint, identity)["action"] + task = { + "task_id": "task", + "user_id": "user", + "repo": "owner/repo", + "status": "AWAITING_APPROVAL", + "session_id": "microvm-new", + "awaiting_approval_request_id": "request", + "approval_gate_count": 4, + "continuation": { + "version": 1, + "state": "RESTORING", + "worker_id": "microvm-new", + "attempt_id": "attempt-new", + "identity": asdict(identity), + "manifest": asdict(receipt), + }, + } + approval = { + "user_id": "user", + "repo": "owner/repo", + "status": "APPROVED", + "tool_input_sha256": action["tool_input_sha256"], + } + lifecycle = MicrovmLifecycle("task", "microvm-new") + monkeypatch.setenv("CONTINUATION_BUCKET_NAME", "owned-bucket") + monkeypatch.setattr("config.AGENT_WORKSPACE", str(workspace.parent)) + monkeypatch.setattr("microvm_lifecycle.get_context", lambda _: lifecycle) + monkeypatch.setattr("continuation_runtime.S3ContinuationStorage", lambda _: runtime.storage) + get_task = MagicMock(return_value=task) + consume = MagicMock() + monkeypatch.setattr("task_state.get_task", get_task) + monkeypatch.setattr("task_state.get_approval_row", lambda *_args, **_kw: approval) + monkeypatch.setattr("task_state.consume_restored_continuation", consume) + config = TaskConfig( + task_id="task", + user_id="user", + repo_url="owner/repo", + github_token="test", + aws_region="us-west-2", + resolved_workflow={"id": "coding/new-task-v1", "version": "1"}, + ) + shutil.rmtree(workspace) + yield config, task, approval, consume, get_task, workspace + lifecycle.close() + + +@pytest.mark.parametrize("decision", ["APPROVED", "DENIED", "TIMED_OUT"]) +def test_production_restore_loads_saved_files_then_claims_recorded_decision(replacement, decision): + config, task, approval, consume, get_task, workspace = replacement + approval["status"] = decision + approval["deny_reason"] = "Choose another approach" + runtime = restore_for_task(config) + assert runtime is not None and runtime.restored is not None + assert (workspace / "untracked.txt").read_text() == "keep this work\n" + assert runtime.human_decision == approval + assert runtime.resume_prompt is not None and decision in runtime.resume_prompt + assert runtime.prior_cost_usd == pytest.approx(0.02) + assert config.initial_approval_gate_count == 4 + get_task.assert_called_once_with("task", consistent_read=True) + consume.assert_called_once_with("task", "microvm-new", task["continuation"]) + + +@pytest.mark.parametrize( + "field,value", + [ + ("status", "PENDING"), + ("status", "CANCELLED"), + ("user_id", "other"), + ("repo", "other/repo"), + ("tool_input_sha256", "different"), + ], +) +def test_production_restore_never_claims_an_unavailable_or_mismatched_decision( + replacement, field, value +): + from continuation_session import ContinuationCheckpointError + + config, _, approval, consume, _, _ = replacement + approval[field] = value + with pytest.raises(ContinuationCheckpointError, match="human decision"): + restore_for_task(config) + consume.assert_not_called() + + +def test_production_restore_waits_for_start_registration_without_fresh_clone( + replacement, monkeypatch +): + from copy import deepcopy + + config, task, _, consume, get_task, _ = replacement + starting = deepcopy(task) + starting["continuation"]["state"] = "STARTING" + get_task.side_effect = [starting, task] + monkeypatch.setattr("continuation_runtime.time.sleep", lambda _: None) + assert restore_for_task(config) is not None + assert get_task.call_count == 2 + consume.assert_called_once() + + +def test_production_restore_rejects_a_different_physical_worker_before_writing_files(replacement): + from continuation_session import ContinuationCheckpointError + + config, task, _, consume, _, workspace = replacement + task["continuation"]["worker_id"] = "other-worker" + with pytest.raises(ContinuationCheckpointError, match="does not own"): + restore_for_task(config) + assert not workspace.exists() + consume.assert_not_called() + + +def test_repository_free_microvm_restores_its_private_scratch_files_without_a_remote( + ready, tmp_path, monkeypatch +): + from continuation_runtime import prepare_repoless_runtime + from continuation_session import decode_checkpoint + + _, reference = ready + base = tmp_path / "scratch-root" + monkeypatch.setattr("config.AGENT_WORKSPACE", str(base)) + monkeypatch.setenv("CONTINUATION_BUCKET_NAME", "owned-bucket") + monkeypatch.setattr("continuation_runtime.S3ContinuationStorage", lambda _: reference.storage) + lifecycle = MicrovmLifecycle("task", "microvm-one") + monkeypatch.setattr("microvm_lifecycle.get_context", lambda _: lifecycle) + task = {"task_id": "task", "user_id": "user", "status": "RUNNING"} + monkeypatch.setattr("task_state.get_task", lambda *_args, **_kw: task) + config = TaskConfig( + task_id="task", + user_id="user", + repo_url="", + github_token="", + requires_repo=False, + aws_region="us-west-2", + resolved_workflow={"id": "default/agent-v1", "version": "1"}, + ) + runtime = prepare_repoless_runtime( + config, + user_prompt="Keep these scratch files", + system_prompt="Saved instructions", + workflow_id="default/agent-v1", + workflow_version="1", + ) + assert runtime is not None + workspace = base / "task" + assert runtime.context.setup.repo_dir == str(workspace) + assert git(workspace, "remote") == "" + (workspace / "untracked.txt").write_text("private draft\n") + + async def capture(): + active, park, tool_input = await park_runtime(runtime, repo="") + result = await runtime.capture( + active, park, session_id=SESSION, tool_name="Read", tool_input=tool_input + ) + active.close() + return result + + identity, receipt = asyncio.run(capture()) + manifest = runtime.storage.load_manifest(receipt, identity) + action = decode_checkpoint( + runtime.storage.conversations.load(manifest.conversation, identity), identity + )["action"] + lifecycle.close() + lifecycle = MicrovmLifecycle("task", "microvm-new") + task.update( + status="AWAITING_APPROVAL", + session_id="microvm-new", + awaiting_approval_request_id="request", + continuation={ + "version": 1, + "state": "RESTORING", + "worker_id": "microvm-new", + "identity": asdict(identity), + "manifest": asdict(receipt), + }, + ) + monkeypatch.setattr( + "task_state.get_approval_row", + lambda *_args, **_kw: { + "user_id": "user", + "repo": "", + "status": "APPROVED", + "tool_input_sha256": action["tool_input_sha256"], + }, + ) + consume = MagicMock() + monkeypatch.setattr("task_state.consume_restored_continuation", consume) + shutil.rmtree(workspace) + try: + restored = prepare_repoless_runtime( + config, + user_prompt="replacement input", + system_prompt="replacement input", + workflow_id="default/agent-v1", + workflow_version="1", + ) + assert restored is not None and restored.restored is not None + assert restored.context.system_prompt == "Saved instructions" + assert (workspace / "untracked.txt").read_text() == "private draft\n" + assert git(workspace, "remote") == "" + assert "credential" not in (workspace / ".git/config").read_text() + consume.assert_called_once() + finally: + lifecycle.close() diff --git a/agent/tests/test_continuation_sdk_probe.py b/agent/tests/test_continuation_sdk_probe.py index dfae7b637..5e58c74b3 100644 --- a/agent/tests/test_continuation_sdk_probe.py +++ b/agent/tests/test_continuation_sdk_probe.py @@ -30,6 +30,7 @@ sys.path.insert(0, str(AGENT_ROOT / "src")) from continuation_session import CheckpointIdentity, CheckpointSessionStore, decode_checkpoint +from continuation_usage import read_usage from continuation_workspace import capture_workspace, restore_workspace from scripts.verify_microvm_credentials import MODEL, event_frame, model_response @@ -107,6 +108,8 @@ async def pre(data, tool_id, context): assert data["tool_name"] == "Read" assert data["tool_input"] == {"file_path": str(target)} record("pre", session_id=data["session_id"], tool_id=tool_id) + usage = await read_usage(client) + record("usage", cost_usd=usage.cost_usd, tokens=usage.tokens) if original: body = await store.checkpoint_pending( IDENTITY, @@ -155,6 +158,16 @@ async def post(data, tool_id, context): options = ClaudeAgentOptions( model=MODEL, max_turns=3, + max_budget_usd=0.00006 + - ( + next( + item["cost_usd"] + for item in json.loads((directory / "original-audit.json").read_text()) + if item["kind"] == "usage" + ) + if not original + else 0 + ), cwd=str(workspace), tools=["Read"], permission_mode="bypassPermissions", @@ -182,7 +195,12 @@ async def post(data, tool_id, context): await client.query(prompt) async for message in client.receive_response(): if isinstance(message, ResultMessage): - record("result", session_id=message.session_id, is_error=message.is_error) + record( + "result", + session_id=message.session_id, + is_error=message.is_error, + cost_usd=message.total_cost_usd, + ) @pytest.mark.skipif( @@ -320,6 +338,11 @@ def start(name): assert posts == (["toolu_restored"] if decision == "approve" else []) result = next(event for event in audit if event["kind"] == "result") assert not result["is_error"] and result["session_id"] == saved["session_id"] + original_usage = next(event for event in original_audit if event["kind"] == "usage") + restored_usage = next(event for event in audit if event["kind"] == "usage") + assert original_usage["cost_usd"] == pytest.approx(0.000018) + assert restored_usage["cost_usd"] == pytest.approx(0.000018) + assert result["cost_usd"] + original_usage["cost_usd"] == pytest.approx(0.000054) assert target.read_text() == "OWNED_CONTINUATION_MARKER\n" assert (workspace / "untracked.txt").read_text() == "UNTRACKED_CONTINUATION_MARKER\n" restored_requests = [request for request in requests if request["phase"] == "restored"] @@ -340,6 +363,9 @@ def start(name): "untracked_file_preserved": True, "restored_posts": posts, "only_synthetic_loopback": True, + "prior_cost_usd": original_usage["cost_usd"], + "restored_budget_usd": 0.00006 - original_usage["cost_usd"], + "total_cost_usd": result["cost_usd"] + original_usage["cost_usd"], } (tmp_path / "verification.json").write_text(json.dumps(proof, indent=2)) finally: diff --git a/agent/tests/test_continuation_storage.py b/agent/tests/test_continuation_storage.py new file mode 100644 index 000000000..7cf38d134 --- /dev/null +++ b/agent/tests/test_continuation_storage.py @@ -0,0 +1,283 @@ +# Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. +# SPDX-License-Identifier: MIT-0 + +"""Storage acknowledgement must survive retries without accepting partial state.""" + +from __future__ import annotations + +import base64 +import hashlib +import io +import json +from dataclasses import asdict, replace +from decimal import Decimal +from types import SimpleNamespace + +import pytest + +import continuation_storage as storage +from continuation_session import CheckpointIdentity, CheckpointReceipt +from models import RepoSetup + +IDENTITY = CheckpointIdentity("task", "attempt", "request", "user", "owner/repo") + + +class RecordingStream(io.BytesIO): + def __init__(self, body, max_read=1024 * 1024): + super().__init__(body) + self.read_sizes = [] + self.max_read = max_read + + def read(self, size=-1): + assert 0 < size <= self.max_read + self.read_sizes.append(size) + return super().read(size) + + +class VersionedS3: + def __init__(self): + self.versions = {} + self.calls = [] + self.streams = [] + self.lost_reply = False + self.read_override = {} + + def put_object(self, **kwargs): + self.calls.append(("put", kwargs)) + body = kwargs["Body"] if isinstance(kwargs["Body"], bytes) else kwargs["Body"].read() + assert len(body) == kwargs.get("ContentLength", len(body)) + assert kwargs["IfNoneMatch"] == "*" + assert kwargs["ServerSideEncryption"] == "AES256" + assert base64.b64encode(hashlib.sha256(body).digest()).decode() == kwargs["ChecksumSHA256"] + key = kwargs["Key"] + if key in self.versions: + raise RuntimeError("PreconditionFailed") + self.versions[key] = [body] + if self.lost_reply: + raise TimeoutError("reply lost after commit") + return {"VersionId": "1"} + + def get_object(self, **kwargs): + self.calls.append(("get", kwargs)) + bodies = self.versions[kwargs["Key"]] + version = kwargs.get("VersionId", str(len(bodies))) + body = bodies[int(version) - 1] + stream = RecordingStream( + body, 16 * 1024 * 1024 + 1 if kwargs["Key"].endswith(".json") else 1024 * 1024 + ) + self.streams.append(stream) + return { + "Body": stream, + "VersionId": version, + "ContentLength": len(body), + "ChecksumSHA256": base64.b64encode(hashlib.sha256(body).digest()).decode(), + **self.read_override, + } + + +@pytest.fixture +def prepared(tmp_path): + source = tmp_path / "workspace.tar" + source.write_bytes(b"saved workspace\n" * 180_000) + client = VersionedS3() + adapter = storage.S3ContinuationStorage("owned-bucket", client=client) + return source, client, adapter + + +def manifest(receipt): + digest = "a" * 64 + return storage.ContinuationManifest( + IDENTITY, + CheckpointReceipt(IDENTITY.prefix + digest + ".json", "1", digest, 10), + receipt, + storage.ContinuationContext( + RepoSetup(repo_dir="/workspace/task", branch="agent/test", build_before=False), + "Original user prompt", + "Original system prompt", + "coding/new-task-v1", + "1", + ), + ) + + +class TestPinnedFiles: + @pytest.mark.parametrize("size", [Decimal("42.5"), Decimal("NaN"), "42", 42.0, True]) + def test_ddb_receipt_does_not_coerce_invalid_sizes(self, size): + digest = "a" * 64 + value = asdict( + storage.FileReceipt( + "manifest", IDENTITY.prefix + "manifest/" + digest + ".json", "1", digest, 42 + ) + ) + value["size_bytes"] = size + with pytest.raises(storage.ContinuationStorageError): + storage.FileReceipt.from_record(value, IDENTITY) + + def test_ddb_integral_receipt_roundtrip(self): + digest = "a" * 64 + receipt = storage.FileReceipt( + "manifest", IDENTITY.prefix + "manifest/" + digest + ".json", "1", digest, 42 + ) + assert ( + storage.FileReceipt.from_record( + {**asdict(receipt), "size_bytes": Decimal("42")}, IDENTITY + ) + == receipt + ) + + def test_streamed_roundtrip_pins_version_even_after_current_object_changes( + self, prepared, tmp_path + ): + source, client, adapter = prepared + receipt = adapter.save_workspace(source, IDENTITY) + client.versions[receipt.key].append(b"later unrelated contents") + destination = tmp_path / "restored.tar" + adapter.download_workspace(receipt, IDENTITY, destination) + assert destination.read_bytes() == source.read_bytes() + assert client.calls[-1][1]["VersionId"] == "1" + assert all(stream.closed for stream in client.streams) + assert len(client.streams[-1].read_sizes) > 2 + assert destination.stat().st_mode & 0o777 == 0o600 + assert not list(tmp_path.glob(".continuation-*")) + + def test_lost_put_reply_and_repeated_put_require_successful_readback(self, prepared): + source, client, adapter = prepared + client.lost_reply = True + first = adapter.save_workspace(source, IDENTITY) + assert adapter.save_workspace(source, IDENTITY) == first + assert len(client.versions[first.key]) == 1 + + @pytest.mark.parametrize( + "override", + [ + {"VersionId": "null"}, + {"VersionId": ""}, + {"ContentLength": 1}, + {"ContentLength": True}, + {"ChecksumSHA256": None}, + ], + ) + def test_bad_readback_never_acknowledges_worker_release(self, prepared, override): + source, client, adapter = prepared + client.read_override = override + with pytest.raises(storage.ContinuationStorageError, match="retain the worker"): + adapter.save_workspace(source, IDENTITY) + assert client.streams[-1].closed + + @pytest.mark.parametrize("payload", [b"short", b"x" * 2_880_000]) + def test_corrupt_or_truncated_stream_leaves_no_destination(self, prepared, tmp_path, payload): + source, client, adapter = prepared + receipt = adapter.save_workspace(source, IDENTITY) + bad_stream = RecordingStream(payload) + client.read_override["Body"] = bad_stream + destination = tmp_path / "restored.tar" + with pytest.raises(storage.ContinuationStorageError): + adapter.download_workspace(receipt, IDENTITY, destination) + assert not destination.exists() + assert not list(tmp_path.glob(".continuation-*")) + assert bad_stream.closed + + def test_existing_destination_is_never_overwritten(self, prepared, tmp_path): + source, _, adapter = prepared + receipt = adapter.save_workspace(source, IDENTITY) + destination = tmp_path / "restored.tar" + destination.write_text("keep me") + with pytest.raises(FileExistsError): + adapter.download_workspace(receipt, IDENTITY, destination) + assert destination.read_text() == "keep me" + assert not list(tmp_path.glob(".continuation-*")) + + def test_task_and_attempt_receipts_cannot_cross_boundaries(self, prepared, tmp_path): + source, client, adapter = prepared + receipt = adapter.save_workspace(source, IDENTITY) + count = len(client.calls) + for identity in (replace(IDENTITY, task_id="other"), replace(IDENTITY, attempt_id="other")): + with pytest.raises(storage.ContinuationStorageError, match="receipt"): + adapter.download_workspace(receipt, identity, tmp_path / "bad.tar") + assert len(client.calls) == count + + def test_disk_pressure_is_distinct_and_detected_before_network( + self, prepared, tmp_path, monkeypatch + ): + source, client, adapter = prepared + receipt = adapter.save_workspace(source, IDENTITY) + count = len(client.calls) + monkeypatch.setattr(storage.shutil, "disk_usage", lambda _: SimpleNamespace(free=0)) + with pytest.raises(storage.ContinuationStorageError) as raised: + adapter.download_workspace(receipt, IDENTITY, tmp_path / "bad.tar") + assert raised.value.code == "disk_pressure" + assert len(client.calls) == count + + def test_disk_pressure_during_download_removes_partial_file( + self, prepared, tmp_path, monkeypatch + ): + source, _, adapter = prepared + receipt = adapter.save_workspace(source, IDENTITY) + readings = iter([10**12, 10**12, 0]) + monkeypatch.setattr( + storage.shutil, "disk_usage", lambda _: SimpleNamespace(free=next(readings)) + ) + with pytest.raises(storage.ContinuationStorageError) as raised: + adapter.download_workspace(receipt, IDENTITY, tmp_path / "bad.tar") + assert raised.value.code == "disk_pressure" + assert not list(tmp_path.glob(".continuation-*")) + assert not (tmp_path / "bad.tar").exists() + + def test_upload_rejects_symlinks_and_hardlinks(self, prepared, tmp_path): + source, client, adapter = prepared + link = tmp_path / "link" + link.symlink_to(source) + with pytest.raises(OSError): + adapter.save_workspace(link, IDENTITY) + link.unlink() + link.hardlink_to(source) + with pytest.raises(storage.ContinuationStorageError, match="invalid"): + adapter.save_workspace(source, IDENTITY) + assert not client.calls + + def test_transfer_deadline_closes_stream_and_discards_partial_file( + self, prepared, tmp_path, monkeypatch + ): + source, _, adapter = prepared + receipt = adapter.save_workspace(source, IDENTITY) + times = iter([0, 0, 0, 301]) + monkeypatch.setattr(storage.time, "monotonic", lambda: next(times)) + with pytest.raises(storage.ContinuationStorageError) as raised: + adapter.download_workspace(receipt, IDENTITY, tmp_path / "bad.tar") + assert raised.value.code == "transfer_timeout" + assert not (tmp_path / "bad.tar").exists() + + +class TestManifest: + def test_complete_manifest_preserves_baseline_and_exact_receipts(self, prepared): + source, _, adapter = prepared + record = manifest(adapter.save_workspace(source, IDENTITY)) + receipt = adapter.save_manifest(record) + restored = adapter.load_manifest(receipt, IDENTITY) + assert restored == record + assert restored.context.setup.build_before is False + assert restored.workspace.version_id == "1" + + @pytest.mark.parametrize( + "change", + [ + lambda value: value.update(version=True), + lambda value: value["identity"].update(request_id="another"), + lambda value: value["context"].update(github_token="must not be serialized"), + lambda value: value["context"]["setup"].update(build_before="false"), + lambda value: value["context"]["setup"].update(repo_dir="relative"), + lambda value: value["workspace"].update(kind="manifest"), + lambda value: value["conversation"].update(version_id="null"), + ], + ) + def test_manifest_rejects_mixed_identity_unknown_fields_and_invalid_context( + self, prepared, change + ): + source, _, adapter = prepared + body = manifest(adapter.save_workspace(source, IDENTITY)).encode(adapter.limits) + value = json.loads(body) + change(value) + with pytest.raises(storage.ContinuationCheckpointError): + storage.ContinuationManifest.decode( + json.dumps(value).encode(), IDENTITY, adapter.limits + ) diff --git a/agent/tests/test_continuation_usage.py b/agent/tests/test_continuation_usage.py new file mode 100644 index 000000000..1c49a3f03 --- /dev/null +++ b/agent/tests/test_continuation_usage.py @@ -0,0 +1,80 @@ +# Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. +# SPDX-License-Identifier: MIT-0 + +"""Accounting cannot silently reset or estimate spend after replacement.""" + +import asyncio +from copy import deepcopy +from types import SimpleNamespace +from typing import Any +from unittest.mock import AsyncMock + +import pytest + +from continuation_session import ContinuationCheckpointError +from continuation_usage import read_usage + +RESPONSE: dict[str, Any] = { + "session": { + "total_cost_usd": 0.03, + "model_usage": { + "sonnet": { + "costUSD": 0.01, + "inputTokens": 100, + "outputTokens": 20, + "cacheReadInputTokens": 30, + "cacheCreationInputTokens": 40, + }, + "haiku": { + "costUSD": 0.02, + "inputTokens": 5, + "outputTokens": 6, + "cacheReadInputTokens": 7, + "cacheCreationInputTokens": 8, + }, + }, + } +} + + +def test_reads_exact_current_process_spending_across_models(): + send = AsyncMock(return_value=RESPONSE) + snapshot = asyncio.run( + read_usage(SimpleNamespace(_query=SimpleNamespace(_send_control_request=send))) + ) + assert snapshot.cost_usd == 0.03 + assert snapshot.tokens == { + "input_tokens": 105, + "output_tokens": 26, + "cache_read_input_tokens": 37, + "cache_creation_input_tokens": 48, + } + send.assert_awaited_once_with({"subtype": "get_usage"}) + + +@pytest.mark.parametrize( + "mutation", ["negative", "nan", "missing", "empty_models", "bad_tokens", "different_cost"] +) +def test_incomplete_or_invalid_accounting_cannot_publish_a_checkpoint(mutation): + response = deepcopy(RESPONSE) + if mutation in {"negative", "nan"}: + response["session"]["total_cost_usd"] = -1 if mutation == "negative" else float("nan") + elif mutation == "missing": + response = {} + elif mutation == "empty_models": + response["session"]["model_usage"] = {} + elif mutation == "bad_tokens": + response["session"]["model_usage"]["haiku"]["inputTokens"] = "5" + else: + response["session"]["model_usage"]["haiku"]["costUSD"] = 0.5 + client = SimpleNamespace( + _query=SimpleNamespace(_send_control_request=AsyncMock(return_value=response)) + ) + with pytest.raises(ContinuationCheckpointError): + asyncio.run(read_usage(client)) + + +def test_unverified_sdk_upgrade_requires_explicit_accounting_validation(monkeypatch): + monkeypatch.setattr("continuation_usage.importlib.metadata.version", lambda _: "0.3.0") + with pytest.raises(ContinuationCheckpointError, match="verified SDK"): + asyncio.run(read_usage(None)) diff --git a/agent/tests/test_continuation_workspace.py b/agent/tests/test_continuation_workspace.py index 4ec9a8879..24bcf723d 100644 --- a/agent/tests/test_continuation_workspace.py +++ b/agent/tests/test_continuation_workspace.py @@ -231,9 +231,7 @@ def test_git_deadline_stops_the_process_group(self, repo): assert error.value.code == "git_timeout" assert time.monotonic() - started < 5 - @pytest.mark.parametrize( - "repo_name", ["", "../repo", "owner/..", "https://github.com/owner/repo"] - ) + @pytest.mark.parametrize("repo_name", ["../repo", "owner/..", "https://github.com/owner/repo"]) def test_repository_identity_must_also_be_restorable(self, repo, repo_name): destination = repo.parent / "workspace.tar" with pytest.raises(workspace.WorkspaceCheckpointError) as error: diff --git a/agent/tests/test_hooks.py b/agent/tests/test_hooks.py index 2ce4faed0..9bd85ecc3 100644 --- a/agent/tests/test_hooks.py +++ b/agent/tests/test_hooks.py @@ -11,6 +11,7 @@ from hooks import ( _is_self_reclone, _reset_blocker_reason_for_tests, + _sha256_tool_input_for_row, _stuck_guard_between_turns_hook, build_hook_matchers, detect_egress_denial, @@ -20,7 +21,7 @@ pre_tool_use_hook, reset_stuck_summary, ) -from policy import PolicyEngine +from policy import Outcome, PolicyEngine @pytest.fixture(autouse=True) @@ -975,6 +976,152 @@ def _hook_input(tool_name: str = "Bash", command: str = "echo foo") -> dict: } +@pytest.fixture() +def restored_approval_runtime(): + from unittest.mock import MagicMock + + from continuation_runtime import ContinuationRuntime + from continuation_storage import ContinuationContext + from models import RepoSetup + + runtime = ContinuationRuntime( + ContinuationContext( + RepoSetup(repo_dir="/workspace/task", branch="main", build_before=False), + "Original request", + "System prompt", + "coding/new-task-v1", + "1", + ), + MagicMock(), + restored={ + "action": { + "tool_name": "Bash", + "tool_input_sha256": _sha256_tool_input_for_row({"command": "echo foo"}), + } + }, + ) + runtime.human_decision = { + "request_id": "original-request", + "status": "APPROVED", + "scope": "this_call", + "decided_at": "2026-09-17T17:00:00Z", + "created_at": "2026-09-16T17:00:00Z", + "matching_rule_ids": ["test_bash_foo"], + } + return runtime + + +class TestRestoredApproval: + def test_exact_saved_action_is_approved_once_without_another_request( + self, restored_approval_runtime, engine_with_soft_gate, fake_task_state, progress + ): + from continuation_runtime import bind_runtime + + async def call(command="echo foo"): + return await pre_tool_use_hook( + _hook_input(command=command), + "new-tool-id", + {}, + engine=engine_with_soft_gate, + progress=progress, + task_state_module=fake_task_state, + ) + + with bind_runtime(restored_approval_runtime): + # A different proposal does not spend the single-action grant. + changed = asyncio.run(call("echo other-foo")) + assert changed["hookSpecificOutput"]["permissionDecision"] == "deny" + approved = asyncio.run(call()) + assert approved["hookSpecificOutput"]["permissionDecision"] == "allow" + repeated = asyncio.run(call()) + assert repeated["hookSpecificOutput"]["permissionDecision"] == "deny" + assert fake_task_state.write_calls == [] + assert fake_task_state.gate_count_calls == [] + assert engine_with_soft_gate.approval_gate_count == 0 + grants = [kwargs for name, kwargs in progress.calls if name == "write_approval_granted"] + assert len(grants) == 1 + assert grants[0]["request_id"] == "original-request" + assert grants[0]["decided_at"] == "2026-09-17T17:00:00Z" + + def test_hard_denial_wins_without_consuming_saved_approval( + self, restored_approval_runtime, fake_task_state + ): + from continuation_runtime import bind_runtime + + engine = PolicyEngine( + task_type="new_task", + repo="owner/repo", + blueprint_hard_policies=( + '@tier("hard") @rule_id("never_foo") ' + 'forbid (principal, action == Agent::Action::"execute_bash", resource) ' + 'when { context.command like "*foo*" };' + ), + ) + with bind_runtime(restored_approval_runtime): + result = asyncio.run( + pre_tool_use_hook( + _hook_input(), + "new-tool-id", + {}, + engine=engine, + task_state_module=fake_task_state, + ) + ) + assert result["hookSpecificOutput"]["permissionDecision"] == "deny" + assert ( + restored_approval_runtime.consume_approved_action( + "Bash", _sha256_tool_input_for_row({"command": "echo foo"}) + ) + is not None + ) + + @pytest.mark.parametrize("status", ["DENIED", "TIMED_OUT"]) + def test_recorded_denial_seeds_normal_policy_cache( + self, restored_approval_runtime, engine_with_soft_gate, status + ): + restored_approval_runtime.human_decision.update( + status=status, deny_reason="Use another approach" + ) + restored_approval_runtime.seed_policy(engine_with_soft_gate) + decision = engine_with_soft_gate.evaluate_tool_use("Bash", {"command": "echo foo"}) + assert decision.outcome == Outcome.DENY + assert decision.cache_hit_metadata["original_decision_ts"] == "2026-09-17T17:00:00Z" + # Explicit denials also cover cosmetic changes under the same rule. + changed = engine_with_soft_gate.evaluate_tool_use("Bash", {"command": "echo other-foo"}) + assert changed.outcome == (Outcome.DENY if status == "DENIED" else Outcome.REQUIRE_APPROVAL) + + def test_grants_survive_a_replacement(self, restored_approval_runtime, engine_with_soft_gate): + from dataclasses import replace + + from policy import ApprovalAllowlist + + scopes = [ + "tool_type: Read", + "tool_group:file_write", + "rule:other_rule", + "bash_pattern:git status*", + "write_path:/workspace/docs/*", + ] + original = ApprovalAllowlist(scopes) + restored_approval_runtime.context = replace( + restored_approval_runtime.context, approval_scopes=original.snapshot_scopes() + ) + restored_approval_runtime.human_decision["scope"] = "rule:test_bash_foo" + restored_approval_runtime.seed_policy(engine_with_soft_gate) + for tool, tool_input in [ + ("Read", {}), + ("Write", {"file_path": "/workspace/a"}), + ("Bash", {"command": "git status --short"}), + ("Bash", {"command": "echo foo"}), + ]: + assert ( + engine_with_soft_gate.evaluate_tool_use(tool, tool_input).outcome == Outcome.ALLOW + ) + assert engine_with_soft_gate.allowlist.rule_ids == {"other_rule", "test_bash_foo"} + original.add("all_session") + assert ApprovalAllowlist(list(original.snapshot_scopes())).matches("AnyTool", {}) + + def _prime_approval(fake: _FakeTaskState, terminal_row: dict) -> None: """Queue an ``APPROVED``/``DENIED`` row on the second poll iteration. @@ -1240,6 +1387,83 @@ async def cancelled_sleep(_seconds): class TestApprovedPath: + @pytest.mark.parametrize("transferred", [False, True]) + @pytest.mark.parametrize("publish_failed", [False, True]) + def test_checkpoint_holds_tools_until_same_worker_claims_decision( + self, + fake_task_state, + progress, + engine_with_soft_gate, + monkeypatch, + transferred, + publish_failed, + ): + from types import SimpleNamespace + from typing import Any + + from continuation_runtime import bind_runtime + from continuation_session import CheckpointIdentity + from microvm_lifecycle import register_task, unregister_task + + _fast_poll(monkeypatch) + _prime_approval(fake_task_state, {"status": "APPROVED", "scope": "this_call"}) + lifecycle = register_task("checkpoint-hook-task", "microvm-hook") + phases = [] + original_resume = fake_task_state.transact_resume_from_approval + + async def capture(context, park, **kwargs): + assert kwargs["session_id"] == "sdk-session" + async with context.continuation_checkpoint(park): + phases.append(context.diagnostic_snapshot()["phase"]) + return CheckpointIdentity( + lifecycle.task_id, lifecycle.microvm_id, park.request_id, "user", "owner/repo" + ), "owned-receipt" + + def publish(*args, **kwargs): + phases.append(lifecycle.diagnostic_snapshot()["phase"]) + if publish_failed: + raise TimeoutError("publication acknowledgement lost") + + def resume(*args, **kwargs): + phases.append(lifecycle.diagnostic_snapshot()["phase"]) + if transferred: + raise _FakeApprovalResumeError("coordinator now owns the checkpoint") + return original_resume(*args, **kwargs) + + runtime: Any = SimpleNamespace( + capture=capture, + consume_approved_action=lambda *_: None, + context=SimpleNamespace(cost_usd=0.02, turns_used=1), + ) + monkeypatch.setattr( + fake_task_state, "publish_continuation_checkpoint", publish, raising=False + ) + monkeypatch.setattr(fake_task_state, "transact_resume_from_approval", resume) + monkeypatch.setattr(hooks, "task_state", fake_task_state) + matchers = build_hook_matchers( + engine=engine_with_soft_gate, + task_id=lifecycle.task_id, + user_id="user", + progress=progress, + ) + try: + with bind_runtime(runtime): + result = _run( + matchers["PreToolUse"][0].hooks[0]( + {**_hook_input(), "session_id": "sdk-session"}, + "tu-1", + {}, + ) + ) + assert phases == ["checkpointing", "checkpoint-ready", "checkpoint-ready"] + expected = "deny" if transferred else "allow" + assert result["hookSpecificOutput"]["permissionDecision"] == expected + assert lifecycle.diagnostic_snapshot()["phase"] == ( + "closed" if transferred else "active" + ) + finally: + unregister_task(lifecycle) + def test_approved_returns_allow_and_propagates_scope( self, fake_task_state, progress, engine_with_soft_gate, monkeypatch ): @@ -1332,7 +1556,7 @@ def test_denied_returns_deny_queues_injection_and_caches( # Recent-decision cache populated — identical next call auto-denies. follow_up = engine_with_soft_gate.evaluate_tool_use("Bash", {"command": "echo foo"}) assert follow_up.outcome.value == "deny" - assert "Recent DENIED" in follow_up.reason + assert "Recorded DENIED at t1" in follow_up.reason assert "write_approval_denied" in progress.milestones() @@ -1903,6 +2127,7 @@ def test_rule_annotation_clip_emits_milestone(self, fake_task_state, progress, m task_type="new_task", repo="owner/repo", blueprint_soft_policies=blueprint_soft, + task_default_timeout_s=300, ) _prime_approval( fake_task_state, diff --git a/agent/tests/test_microvm_checkpoint.py b/agent/tests/test_microvm_checkpoint.py index 694eeac3e..7f1c3550d 100644 --- a/agent/tests/test_microvm_checkpoint.py +++ b/agent/tests/test_microvm_checkpoint.py @@ -9,7 +9,7 @@ import copy import os import uuid -from dataclasses import dataclass +from dataclasses import dataclass, replace from datetime import UTC, datetime from unittest.mock import MagicMock from urllib.parse import urlsplit @@ -105,6 +105,56 @@ def state(monkeypatch): class TestCheckpoint: + @pytest.mark.parametrize("repo", ["owner/repo", ""]) + def test_retained_request_can_suspend_and_resume_with_explicit_null_deadline(self, state, repo): + from hooks import _ApprovalDeadline + + original, task, approval, client = state + assert original.record is not None + record = replace(original.record, timeout_s=0, repo=repo) + park = replace( + original, + record=record, + deadline=_ApprovalDeadline.from_recorded(record.created_at, 0), + ) + new_task, new_approval = make_rows(park) + task.update(new_task) + if not repo: + task.pop("repo") + approval.update(new_approval) + checkpoint.checkpoint_before_suspend(park) + operations = client.transact_write_items.call_args.kwargs["TransactItems"] + marker = { + key: TypeDeserializer().deserialize(value) + for key, value in operations[2]["Put"]["Item"].items() + } + assert marker["metadata"]["approval_deadline_ms"] is None + assert task["microvm_lifecycle"]["deadline_ms"] is None + task["microvm_lifecycle"].update(action="resume", generation="resume-generation") + approval["status"] = "APPROVED" + checkpoint.refresh_and_reconcile_after_resume(park) + assert park.deadline.remaining_s() == float("inf") + assert approval["status"] == "APPROVED" + assert len(client.transact_write_items.call_args.kwargs["TransactItems"]) == 2 + + @pytest.mark.parametrize("deadline_value", [0, True, "null", "missing"]) + def test_retained_request_rejects_non_null_or_missing_intent_deadline( + self, state, deadline_value + ): + original, task, approval, client = state + assert original.record is not None + park = replace(original, record=replace(original.record, timeout_s=0)) + new_task, new_approval = make_rows(park) + task.update(new_task) + approval.update(new_approval) + if deadline_value == "missing": + task["microvm_lifecycle"].pop("deadline_ms") + else: + task["microvm_lifecycle"]["deadline_ms"] = deadline_value + with pytest.raises(LifecycleUnavailable, match="intent"): + checkpoint.checkpoint_before_suspend(park) + client.transact_write_items.assert_not_called() + def test_checkpoint_is_an_acknowledged_cross_table_transaction(self, state): park, task, approval, client = state checkpoint.checkpoint_before_suspend(park) @@ -148,6 +198,7 @@ def test_checkpoint_is_an_acknowledged_cross_table_transaction(self, state): ("intent", "request_id", "other-request"), ("intent", "action", "resume"), ("intent", "deadline_ms", 1), + ("intent", "deadline_ms", None), ("intent", "requested_at_ms", -1), ("approval", "task_id", "other"), ("approval", "request_id", "other-request"), diff --git a/agent/tests/test_microvm_lifecycle.py b/agent/tests/test_microvm_lifecycle.py index e0cf73c33..7f913c9e1 100644 --- a/agent/tests/test_microvm_lifecycle.py +++ b/agent/tests/test_microvm_lifecycle.py @@ -353,6 +353,102 @@ async def test_real_progress_writer_failure_cannot_acknowledge_suspend(monkeypat unregister_task(context) +@pytest.mark.anyio +async def test_progress_during_continuation_capture_is_acknowledged_before_suspend(monkeypatch): + from progress_writer import _ProgressWriter + + context = register_task("capture-progress-task", "microvm") + try: + _, park = await parked(context) + monkeypatch.setenv("TASK_EVENTS_TABLE_NAME", "events") + writer = _ProgressWriter(context.task_id) + writer._table = Mock() + async with context.continuation_checkpoint(park): + # SDK progress can arrive while the workspace upload awaits a thread. + writer._put_event("agent_turn", {"message": "waiting for approval"}) + writer._table.put_item.assert_called_once() + assert not context.diagnostic_snapshot()["progress_failed"] + with pytest.raises(LifecycleUnavailable), context.activity(): + pytest.fail("General activity must remain paused during capture") + await context.suspend(Mock(), budget_s=1) + # The progress exception is limited to capture, never actual suspension. + writer._put_event("agent_turn", {"message": "cannot write while suspended"}) + writer._table.put_item.assert_called_once() + assert context.diagnostic_snapshot()["progress_failed"] + finally: + unregister_task(context) + + +@pytest.mark.anyio +async def test_capture_waits_for_progress_started_during_upload(monkeypatch): + from progress_writer import _ProgressWriter + + context = register_task("capture-progress-drain-task", "microvm") + entered = threading.Event() + release = threading.Event() + write_task = None + capture_task = None + try: + _, park = await parked(context) + monkeypatch.setenv("TASK_EVENTS_TABLE_NAME", "events") + writer = _ProgressWriter(context.task_id) + writer._table = Mock() + + def slow_write(**_): + entered.set() + assert release.wait(2) + + writer._table.put_item.side_effect = slow_write + + async def capture(): + nonlocal write_task + async with context.continuation_checkpoint(park, drain_budget_s=1): + write_task = asyncio.create_task( + asyncio.to_thread(writer._put_event, "agent_turn", {"message": "in flight"}) + ) + assert await asyncio.to_thread(entered.wait, 1) + + capture_task = asyncio.create_task(capture()) + assert await asyncio.to_thread(entered.wait, 1) + await asyncio.sleep(0.04) + assert not capture_task.done() + assert context.diagnostic_snapshot()["phase"] == "checkpointing" + release.set() + await asyncio.wait_for(capture_task, 1) + assert write_task is not None + await write_task + assert context.diagnostic_snapshot()["phase"] == "checkpoint-ready" + await context.suspend(Mock(), budget_s=1) + finally: + release.set() + if capture_task: + await asyncio.gather(capture_task, return_exceptions=True) + if write_task: + await asyncio.gather(write_task, return_exceptions=True) + unregister_task(context) + + +@pytest.mark.anyio +async def test_progress_failure_during_capture_prevents_checkpoint_publication(monkeypatch): + from progress_writer import _ProgressWriter + + context = register_task("capture-progress-error-task", "microvm") + try: + _, park = await parked(context) + monkeypatch.setenv("TASK_EVENTS_TABLE_NAME", "events") + writer = _ProgressWriter(context.task_id) + writer._table = Mock() + writer._table.put_item.side_effect = OSError("write reply lost") + with pytest.raises(LifecycleUnavailable, match="Progress was not acknowledged"): + async with context.continuation_checkpoint(park): + writer._put_event("agent_turn", {"message": "uncertain"}) + assert context.diagnostic_snapshot()["phase"] == "parked" + with pytest.raises(LifecycleUnavailable): + await context.suspend(Mock(), budget_s=1) + finally: + unregister_task(context) + + @pytest.mark.anyio async def test_resume_reseeds_from_fresh_os_entropy_after_refresh(monkeypatch): from microvm_lifecycle import reseed_random diff --git a/agent/tests/test_payload_bootstrap.py b/agent/tests/test_payload_bootstrap.py index f3b842eee..b87fda334 100644 --- a/agent/tests/test_payload_bootstrap.py +++ b/agent/tests/test_payload_bootstrap.py @@ -122,6 +122,26 @@ def test_authenticates_manifest_before_downloading_and_preserves_cross_region_se assert transport["build"].call_args.args[0].proxies == {} +@pytest.mark.parametrize("mutation", ["none", "url", "payload", "traversal"]) +def test_replacement_reference_binds_task_attempt_and_exact_object(transport, mutation): + reference = transport["reference"] + reference["attempt_id"] = "replacement-2" + reference["payload_url"] = signed_url("task-1/replacement-2") + transport["payload"]["attempt_id"] = "replacement-2" + if mutation == "url": + reference["payload_url"] = signed_url("task-1/replacement-other") + elif mutation == "payload": + transport["payload"]["attempt_id"] = "replacement-other" + elif mutation == "traversal": + reference["attempt_id"] = "../replacement-other" + if mutation == "none": + payload, _ = bootstrap.resolve_payload_reference(reference, "lambda-microvm") + assert payload["attempt_id"] == "replacement-2" + else: + with pytest.raises(ValueError): + bootstrap.resolve_payload_reference(reference, "lambda-microvm") + + def test_large_registry_bundle_reaches_run_mapper_and_mcp_loader(transport, monkeypatch, tmp_path): """Resolve real v2 bytes and carry the bundle from /run into the real local loader.""" from registry.loader import apply_resolved_assets diff --git a/agent/tests/test_policy_three_outcome.py b/agent/tests/test_policy_three_outcome.py index 244977d4e..63b5748ef 100644 --- a/agent/tests/test_policy_three_outcome.py +++ b/agent/tests/test_policy_three_outcome.py @@ -443,19 +443,18 @@ def test_soft_deny_force_push_returns_require_approval(self): d = engine.evaluate_tool_use("Bash", {"command": "git push --force origin feature"}) assert d.outcome == Outcome.REQUIRE_APPROVAL assert "force_push_any" in d.matching_rule_ids - assert d.timeout_s == 300 + assert d.timeout_s == 0 assert d.severity == "medium" def test_soft_deny_multi_match_merges_annotations(self): - # force_push_any (300s, medium) + force_push_main (600s, high) both - # match "git push --force origin main". Merge picks min(300, 600)=300s - # and max(medium, high)=high. §6.3. + # Both built-in rules have no automatic expiry. Their merged severity + # is max(medium, high)=high. engine = PolicyEngine(task_type="new_task", repo="owner/repo") d = engine.evaluate_tool_use("Bash", {"command": "git push --force origin main"}) assert d.outcome == Outcome.REQUIRE_APPROVAL assert "force_push_any" in d.matching_rule_ids assert "force_push_main" in d.matching_rule_ids - assert d.timeout_s == 300 # min across rules + task default + assert d.timeout_s == 0 # neither rule supplies a positive deadline assert d.severity == "high" # max across rules def test_default_allow_on_no_match(self): @@ -543,7 +542,7 @@ def test_denied_then_retry_hits_cache(self): engine.recent_decisions.record("Bash", sha, "DENIED", "user said force-push is too risky") d = engine.evaluate_tool_use("Bash", tool_input) assert d.outcome == Outcome.DENY - assert "Recent DENIED" in d.reason + assert "Recorded DENIED" in d.reason def test_cache_does_not_shadow_hard_deny(self): engine = PolicyEngine(task_type="new_task", repo="owner/repo") @@ -605,7 +604,7 @@ def test_semantic_retry_hits_rule_cache(self): "Bash", {"command": "git push --force origin some-other-branch"} ) assert d.outcome == Outcome.DENY - assert "Recent DENIED on rule 'force_push_any'" in d.reason + assert "Recorded DENIED on rule 'force_push_any'" in d.reason assert d.cache_hit_metadata is not None assert d.cache_hit_metadata["matched_rule_id"] == "force_push_any" assert d.cache_hit_metadata["cached_decision"] == "DENIED" diff --git a/agent/tests/test_poll_for_decision.py b/agent/tests/test_poll_for_decision.py index d0c45a818..8845d7564 100644 --- a/agent/tests/test_poll_for_decision.py +++ b/agent/tests/test_poll_for_decision.py @@ -149,12 +149,12 @@ def test_deadline_beats_failures_when_timeout_short(self, monkeypatch): ts.get_approval_row.return_value = {"status": "PENDING"} progress = MagicMock() - # 0 timeout → loop returns immediately at the deadline check. + # An already elapsed explicit deadline returns immediately. outcome = _run( hooks._poll_for_decision( task_id="01KTASK", request_id="01KREQ", - deadline=hooks._ApprovalDeadline.from_recorded(hooks._iso_now(), 0), + deadline=hooks._ApprovalDeadline(0, 0), progress=progress, ts=ts, ) diff --git a/agent/tests/test_runner.py b/agent/tests/test_runner.py index 7bc96bac8..ad0b9385c 100644 --- a/agent/tests/test_runner.py +++ b/agent/tests/test_runner.py @@ -121,6 +121,73 @@ async def messages(): microvm_lifecycle.unregister_task(context) +@pytest.mark.parametrize("exhausted", [None, "dollars", "turns"]) +def test_replacement_runner_preserves_total_limits_and_reports_cumulative_usage( + monkeypatch, exhausted +): + from types import SimpleNamespace + + import claude_agent_sdk + + from continuation_runtime import bind_runtime + + config = _config(max_turns=10, max_budget_usd=1.0) + runtime: Any = SimpleNamespace( + store=MagicMock(), + restored={"session_id": "saved-session"}, + context=SimpleNamespace(turns_used=10 if exhausted == "turns" else 4), + prior_cost_usd=1.0 if exhausted == "dollars" else 0.25, + prior_token_usage={"input_tokens": 100, "output_tokens": 10}, + client=None, + ) + client = MagicMock() + client.connect = AsyncMock() + client.query = AsyncMock() + client.disconnect = AsyncMock() + + async def messages(): + yield claude_agent_sdk.ResultMessage( + subtype="success", + duration_ms=1, + duration_api_ms=1, + is_error=False, + num_turns=2, + session_id="saved-session", + total_cost_usd=0.1, + usage={"input_tokens": 50, "output_tokens": 5}, + ) + + client.receive_response = messages + make_client = MagicMock(return_value=client) + monkeypatch.setattr(claude_agent_sdk, "ClaudeSDKClient", make_client) + monkeypatch.setattr(runner, "_setup_agent_env", lambda _: None) + monkeypatch.setattr(runner, "_log_claude_cli_version", lambda: None) + monkeypatch.setattr(runner, "_initialize_policy_engine_and_hooks", lambda **_: (None, {})) + monkeypatch.setattr(runner, "_register_gateway_server", lambda _: None) + monkeypatch.setattr(runner, "build_clarification_server", lambda: None) + monkeypatch.setattr(runner, "_ProgressWriter", MagicMock()) + with bind_runtime(runtime): + if exhausted: + with pytest.raises(RuntimeError, match="before the saved continuation"): + asyncio.run( + runner.run_agent("continue", "saved system", config, trajectory=MagicMock()) + ) + make_client.assert_not_called() + return + result = asyncio.run( + runner.run_agent("continue", "saved system", config, trajectory=MagicMock()) + ) + options = make_client.call_args.kwargs["options"] + assert options.resume == "saved-session" + assert options.max_turns == 6 + assert options.max_budget_usd == pytest.approx(0.75) + assert result.cost_usd == pytest.approx(0.35) + assert result.num_turns == 6 + assert result.usage is not None + assert result.usage.input_tokens == 150 + assert result.usage.output_tokens == 15 + + class TestInitializePolicyEngineAndHooks: """Bootstrap the per-task PolicyEngine + hooks without the SDK loop. diff --git a/agent/tests/test_server.py b/agent/tests/test_server.py index ceda0147d..31b83d9f7 100644 --- a/agent/tests/test_server.py +++ b/agent/tests/test_server.py @@ -10,7 +10,7 @@ import time from pathlib import Path from typing import Any -from unittest.mock import MagicMock +from unittest.mock import MagicMock, patch import pytest from fastapi.testclient import TestClient @@ -1350,6 +1350,34 @@ class TestMicrovmRunHookVerifiedPayload: authentication and transport checks have their own regression suite. """ + @pytest.mark.parametrize("attempt", ["replacement-2", "../other", "", 123]) + def test_attempt_identity_is_validated_before_starting( + self, client, monkeypatch, cached_github_token, attempt + ): + spawn = MagicMock() + monkeypatch.setattr(server, "_spawn_background", spawn) + response = client.post( + RUN_HOOK, + json=_run_hook_body( + { + "agent_payload": { + "task_id": "task-attempt", + "repo_url": "org/repo", + "prompt": "Continue", + "github_token": "ghp_x", + "attempt_id": attempt, + }, + } + ), + ) + if attempt == "replacement-2": + assert response.status_code == 200 + assert spawn.call_args.args[0]["attempt_id"] == "replacement-2" + else: + assert response.status_code == 400 + assert response.json()["code"] == "MICROVM_ATTEMPT_ID_INVALID" + spawn.assert_not_called() + def test_accepts_the_payload_and_starts_the_pipeline_asynchronously( self, client, monkeypatch, cached_github_token ): @@ -1752,8 +1780,11 @@ def test_wire_contract_is_exactly_the_documented_key_set(self): "log_group_name": "LOG_GROUP_NAME", "artifacts_bucket_name": "ARTIFACTS_BUCKET_NAME", "trace_artifacts_bucket_name": "TRACE_ARTIFACTS_BUCKET_NAME", + "continuation_bucket_name": "CONTINUATION_BUCKET_NAME", "github_token_secret_arn": "GITHUB_TOKEN_SECRET_ARN", "linear_oauth_secret_arn": "LINEAR_OAUTH_SECRET_ARN", + "linear_vault_enabled": "LINEAR_VAULT_ENABLED", + "linear_workload_identity_name": "LINEAR_WORKLOAD_IDENTITY_NAME", "jira_oauth_secret_arn": "JIRA_OAUTH_SECRET_ARN", "agent_session_role_arn": "AGENT_SESSION_ROLE_ARN", "aws_sdk_ua_app_id": "AWS_SDK_UA_APP_ID", @@ -1943,6 +1974,44 @@ def test_every_allowlisted_key_is_installable(self, env_guard): for key, env_name in server.MICROVM_PLATFORM_CONFIG_ENV_BY_KEY.items(): assert os.environ[env_name] == _platform_config_value(key) + def test_installed_vault_config_resolves_token_without_fallback(self, env_guard): + from config import resolve_linear_api_token + + os.environ.pop("LINEAR_API_TOKEN", None) + os.environ.pop("LINEAR_VAULT_ENABLED", None) + os.environ.pop("LINEAR_WORKLOAD_IDENTITY_NAME", None) + os.environ["AWS_REGION"] = "us-east-1" + server._install_platform_config( + _platform_config( + linear_vault_enabled="true", + linear_workload_identity_name="abca_linear_oauth", + ) + ) + client = MagicMock() + client.get_workload_access_token_for_user_id.return_value = { + "workloadAccessToken": "workload-token", + } + client.get_resource_oauth2_token.return_value = {"accessToken": "linear-vault-token"} + with patch("aws_session.platform_client", return_value=client) as make_client: + assert ( + resolve_linear_api_token( + { + "linear_provider_name": "bgagent-linear-oauth-acme", + "linear_workspace_id": "workspace-id", + "linear_vault_user_id": "linear-ws-acme", + "linear_oauth_secret_arn": _platform_config_value( + "linear_oauth_secret_arn" + ), + } + ) + == "linear-vault-token" + ) + make_client.assert_called_once_with("bedrock-agentcore", region_name="us-east-1") + client.get_workload_access_token_for_user_id.assert_called_once_with( + workloadName="abca_linear_oauth", userId="linear-ws-acme" + ) + client.get_secret_value.assert_not_called() + def test_payload_wins_over_a_pre_existing_image_env_value(self, env_guard): # The load-bearing precedence rule: image env is frozen at snapshot time, # the payload describes the live deployment. @@ -2957,7 +3026,10 @@ def test_ignores_unrelated_fields(self): @pytest.mark.parametrize("crash", [False, True]) -def test_microvm_pipeline_registers_identity_and_always_removes_lifecycle(monkeypatch, crash): +@pytest.mark.parametrize("attempt", ["", "replacement-2"]) +def test_microvm_pipeline_registers_identity_and_always_removes_lifecycle( + monkeypatch, crash, attempt +): from microvm_lifecycle import get_context observed = [] @@ -2966,6 +3038,7 @@ def run_task(**kwargs): context = get_context("lifecycle-server-task") assert context is not None assert context.microvm_id == "microvm-server" + assert context.attempt_id == (attempt or "lifecycle-server-task") assert "microvm_id" not in kwargs observed.append(context) if crash: @@ -2985,6 +3058,7 @@ def run_task(**kwargs): aws_region="us-west-2", task_id="lifecycle-server-task", microvm_id="microvm-server", + attempt_id=attempt, ) assert len(observed) == 1 assert get_context("lifecycle-server-task") is None diff --git a/agent/tests/test_task_state.py b/agent/tests/test_task_state.py index 8ce542240..6461d9a58 100644 --- a/agent/tests/test_task_state.py +++ b/agent/tests/test_task_state.py @@ -12,6 +12,44 @@ from task_state import TaskFetchError, _build_logs_url, _now_iso +@pytest.mark.parametrize("writer", [task_state.write_running, task_state.write_heartbeat]) +@pytest.mark.parametrize( + ("reasons", "expected_warning"), + [ + ([{"Code": "ConditionalCheckFailed"}, {"Code": "None"}], False), + ([{"Code": "None"}, {"Code": "ConditionalCheckFailed"}], True), + ([{"Code": "ConditionalCheckFailed"}, {"Code": "ConditionalCheckFailed"}], True), + ([], True), + ], +) +def test_status_race_logging_preserves_worker_fence_failures( + monkeypatch, writer, reasons, expected_warning +): + from botocore.exceptions import ClientError + + monkeypatch.setattr(task_state, "_get_table", MagicMock()) + monkeypatch.setattr( + task_state, + "_update_task", + MagicMock( + side_effect=ClientError( + { + "Error": { + "Code": "TransactionCanceledException", + "Message": "transaction failed", + }, + "CancellationReasons": reasons, + }, + "TransactWriteItems", + ) + ), + ) + log = MagicMock() + monkeypatch.setattr(task_state, "log", log) + writer("owned-task") + assert any(call.args[0] == "WARN" for call in log.call_args_list) is expected_warning + + class TestAgentWriteContract: def test_current_task_writers_fit_the_deployed_attribute_allowlist(self, monkeypatch): """Exercise real writers; detect a new field before IAM rejects it live. @@ -118,19 +156,29 @@ def test_write_inventory_requires_review_when_a_new_writer_is_added(self): if isinstance(node, ast.FunctionDef) and any( isinstance(child, ast.Call) - and isinstance(child.func, ast.Attribute) - and child.func.attr - in { - "update_item", - "put_item", - "delete_item", - "transact_write_items", - "batch_writer", - } + and ( + ( + isinstance(child.func, ast.Attribute) + and child.func.attr + in { + "update_item", + "put_item", + "delete_item", + "transact_write_items", + "batch_writer", + } + ) + or ( + isinstance(child.func, ast.Name) + and child.func.id in {"_update_task", "_transact_task"} + ) + ) for child in ast.walk(node) ) } assert writers == { + "_update_task", + "_transact_task", "write_running", "write_heartbeat", "write_terminal", @@ -138,6 +186,8 @@ def test_write_inventory_requires_review_when_a_new_writer_is_added(self): "transact_write_approval_request", "transact_resume_from_approval", "increment_approval_gate_count_in_ddb", + "publish_continuation_checkpoint", + "consume_restored_continuation", "best_effort_update_approval_status", # Writes only the supporting approvals table. } @@ -1125,3 +1175,52 @@ def test_extract_cancellation_reasons(self): def test_extract_cancellation_reasons_none_on_plain_exception(self): assert task_state._extract_cancellation_reasons(RuntimeError()) == [] + + +class TestContinuationClaimReadback: + @pytest.mark.parametrize( + "changed", ["none", "cancelled", "worker", "record", "request", "lease"] + ) + def test_lost_response_requires_exact_consumed_assignment( + self, approval_tables_env, monkeypatch, changed + ): + from boto3.dynamodb.types import TypeSerializer + + record = { + "version": 1, + "state": "RESTORING", + "worker_id": "microvm-new", + "identity": { + "task_id": "task", + "attempt_id": "microvm-old", + "request_id": "request", + "user_id": "user", + "repo": "owner/repo", + }, + "manifest": {"key": "exact-saved-key"}, + } + consumed = {**record, "state": "CONSUMED"} + task = {"status": "RUNNING", "session_id": "microvm-new", "continuation": consumed} + if changed == "cancelled": + task["status"] = "CANCELLED" + elif changed == "worker": + task["session_id"] = "microvm-other" + elif changed == "record": + task["continuation"] = {**consumed, "manifest": {"key": "different"}} + elif changed == "request": + task["awaiting_approval_request_id"] = "another-request" + client = MagicMock() + client.transact_write_items.side_effect = TimeoutError("response lost after commit") + serialize = TypeSerializer().serialize + client.get_item.return_value = {"Item": {k: serialize(v) for k, v in task.items()}} + lease = MagicMock(side_effect=RuntimeError("lease lost") if changed == "lease" else None) + monkeypatch.setattr(task_state, "verify_worker_lease", lease) + if changed == "none": + task_state.consume_restored_continuation("task", "microvm-new", record, client=client) + lease.assert_called_once_with("task", client=client) + else: + with pytest.raises(RuntimeError): + task_state.consume_restored_continuation( + "task", "microvm-new", record, client=client + ) + assert client.get_item.call_args.kwargs["ConsistentRead"] is True diff --git a/agent/tests/test_workflow_runner.py b/agent/tests/test_workflow_runner.py index d71494dde..116e55946 100644 --- a/agent/tests/test_workflow_runner.py +++ b/agent/tests/test_workflow_runner.py @@ -487,6 +487,73 @@ def _real_ctx(workflow: Workflow, **kw): return StepContext(workflow=workflow, config=config, **kw) +@pytest.mark.parametrize( + "prepared_prompt", + [ + "Recorded APPROVED. Saved proposal: " + '{"command": "cat marker", "description": "Read marker"}', + "Recorded DENIED. Do not work around this decision.", + "Recorded TIMED_OUT. Continue from the saved decision.", + "Original task with local attachment references.", + "", + ], +) +def test_hydration_preserves_the_prepared_prompt_sent_to_the_agent(monkeypatch, prepared_prompt): + from models import AgentResult, HydratedContext + from workflow.runner import _handle_hydrate_context, _handle_run_agent + + wf = _workflow([{"kind": "hydrate_context"}, {"kind": "run_agent"}]) + ctx = _real_ctx( + wf, + hydrated=HydratedContext(user_prompt="Original task before the human decision"), + user_prompt=prepared_prompt, + system_prompt="Saved system prompt", + ) + received = [] + + async def capture_prompt(prompt, system_prompt, *_args, **_kwargs): + received.append((prompt, system_prompt)) + return AgentResult(status="success") + + monkeypatch.setattr("runner.run_agent", capture_prompt) + _handle_hydrate_context(wf.steps[0], ctx) + _handle_run_agent(wf.steps[1], ctx) + + assert received == [ + (prepared_prompt or "Original task before the human decision", "Saved system prompt") + ] + + +def test_closed_microvm_cannot_run_delivery_steps_after_its_agent_unwinds(monkeypatch): + from microvm_lifecycle import register_task, unregister_task + from models import AgentResult + from workflow.runner import _handle_run_agent + + wf = _workflow([{"kind": "run_agent"}, {"kind": "ensure_pr", "strategy": "create"}]) + ctx = _real_ctx(wf, user_prompt="test", system_prompt="test") + lifecycle = register_task(ctx.config.task_id, "microvm-test") + + async def finish(*_args, **_kwargs): + lifecycle.close() + return AgentResult(status="success") + + def must_not_deliver(*_args): + raise AssertionError("closed worker reached its delivery step") + + monkeypatch.setattr("runner.run_agent", finish) + try: + result = run_workflow( + wf, ctx, handlers={"run_agent": _handle_run_agent, "ensure_pr": must_not_deliver} + ) + assert not result.succeeded + assert result.failed_step is not None + assert result.failed_step.error is not None + assert "Worker execution is closed" in result.failed_step.error + assert len(result.outcomes) == 1 + finally: + unregister_task(lifecycle) + + class TestVerifyHandlers: def test_verify_build_regression_only_passes_when_broken_before(self, monkeypatch): from models import RepoSetup diff --git a/cdk/src/constructs/agent-session-role.ts b/cdk/src/constructs/agent-session-role.ts index 83992a046..030139e7a 100644 --- a/cdk/src/constructs/agent-session-role.ts +++ b/cdk/src/constructs/agent-session-role.ts @@ -25,6 +25,7 @@ import * as s3 from 'aws-cdk-lib/aws-s3'; import { NagSuppressions } from 'cdk-nag'; import { Construct } from 'constructs'; import agentTaskWriteAttributes from './agent-task-write-attributes.json'; +import constants from '../../../contracts/constants.json'; /** * Task reporting may update only the attributes written by task_state.py. @@ -44,12 +45,19 @@ export function grantAgentTaskTableAccess( const leadingKeys = taskScoped ? { 'dynamodb:LeadingKeys': ['${aws:PrincipalTag/task_id}'] } : {}; + const readLeadingKeys = taskScoped + ? { + 'dynamodb:LeadingKeys': [ + '${aws:PrincipalTag/task_id}', `${constants.microvm_continuation.lease_key_prefix}\${aws:PrincipalTag/task_id}`, + ], + } + : {}; grantee.grantPrincipal.addToPrincipalPolicy(new iam.PolicyStatement({ actions: ['dynamodb:GetItem', 'dynamodb:BatchGetItem', 'dynamodb:Query', 'dynamodb:ConditionCheckItem'], resources: [table.tableArn], ...(taskScoped ? { conditions: { - 'ForAllValues:StringEquals': leadingKeys, + 'ForAllValues:StringEquals': readLeadingKeys, 'Null': { 'dynamodb:LeadingKeys': 'false' }, }, } : {}), @@ -302,6 +310,7 @@ export class AgentSessionRole extends Construct { 'Resource wildcards are the per-object suffix under a tenant-scoped ' + 'prefix (traces/${aws:PrincipalTag/user_id}/*, ' + 'attachments/${aws:PrincipalTag/user_id}/*, ' + + 'continuations/${aws:PrincipalTag/task_id}/*, ' + 'artifacts/${aws:PrincipalTag/task_id}/*) and the DynamoDB item ' + 'set gated by a dynamodb:LeadingKeys = ${aws:PrincipalTag/task_id} ' + 'condition — narrower than the compute role this replaces. Bedrock ' diff --git a/cdk/src/constructs/agent-task-write-attributes.json b/cdk/src/constructs/agent-task-write-attributes.json index ace3330da..49e4c9ff6 100644 --- a/cdk/src/constructs/agent-task-write-attributes.json +++ b/cdk/src/constructs/agent-task-write-attributes.json @@ -8,6 +8,7 @@ "agent_heartbeat_at", "awaiting_approval_request_id", "approval_gate_count", + "continuation", "pr_url", "error_message", "cost_usd", diff --git a/cdk/src/constructs/continuation-bucket.ts b/cdk/src/constructs/continuation-bucket.ts new file mode 100644 index 000000000..777852d9a --- /dev/null +++ b/cdk/src/constructs/continuation-bucket.ts @@ -0,0 +1,73 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import { Duration, RemovalPolicy } from 'aws-cdk-lib'; +import * as iam from 'aws-cdk-lib/aws-iam'; +import * as s3 from 'aws-cdk-lib/aws-s3'; +import { NagSuppressions } from 'cdk-nag'; +import { Construct } from 'constructs'; +import constants from '../../../contracts/constants.json'; + +/** Durable task recovery data. Active checkpoints must not expire with debug traces. */ +export class ContinuationBucket extends Construct { + public readonly bucket: s3.Bucket; + + constructor(scope: Construct, id: string) { + super(scope, id); + this.bucket = new s3.Bucket(this, 'Bucket', { + versioned: true, + blockPublicAccess: s3.BlockPublicAccess.BLOCK_ALL, + encryption: s3.BucketEncryption.S3_MANAGED, + enforceSSL: true, + removalPolicy: RemovalPolicy.RETAIN, + lifecycleRules: [{ + id: 'abandoned-multipart-uploads', + abortIncompleteMultipartUploadAfter: Duration.days(1), + }], + }); + NagSuppressions.addResourceSuppressions(this.bucket, [{ + id: 'AwsSolutions-S1', + reason: 'Recovery data is private and task-scoped through IAM; coordinator cleanup runs after task closure. No public or anonymous access path. Server access logging is not enabled for task storage.', + }]); + } + + /** A worker can save/read its own task versions, never list or remove data. */ + public grantWorker(grantee: iam.IGrantable): void { + grantee.grantPrincipal.addToPrincipalPolicy(new iam.PolicyStatement({ + actions: ['s3:GetObject', 's3:GetObjectVersion', 's3:PutObject'], + resources: [this.bucket.arnForObjects( + `${constants.microvm_continuation.object_key_prefix}\${aws:PrincipalTag/task_id}/*`, + )], + })); + } + + /** Coordinator owns launch inputs and cleanup after a task closes. */ + public grantCoordinator(grantee: iam.IGrantable): void { + const prefix = constants.microvm_continuation.object_key_prefix; + grantee.grantPrincipal.addToPrincipalPolicy(new iam.PolicyStatement({ + actions: ['s3:GetObject', 's3:GetObjectVersion', 's3:PutObject', 's3:DeleteObject', 's3:DeleteObjectVersion'], + resources: [this.bucket.arnForObjects(`${prefix}*`)], + })); + grantee.grantPrincipal.addToPrincipalPolicy(new iam.PolicyStatement({ + actions: ['s3:ListBucketVersions'], + resources: [this.bucket.bucketArn], + conditions: { StringLike: { 's3:prefix': `${prefix}*` } }, + })); + } +} diff --git a/cdk/src/constructs/lambda-microvm-compute.ts b/cdk/src/constructs/lambda-microvm-compute.ts index 294e0d6dc..2aee952c0 100644 --- a/cdk/src/constructs/lambda-microvm-compute.ts +++ b/cdk/src/constructs/lambda-microvm-compute.ts @@ -26,18 +26,7 @@ import * as s3 from 'aws-cdk-lib/aws-s3'; import * as secretsmanager from 'aws-cdk-lib/aws-secretsmanager'; import { NagSuppressions } from 'cdk-nag'; import { Construct } from 'constructs'; -// Cross-language contract (S9): `microvm_hook_budgets` couples THIS construct's -// `/ready` hook timeout to the agent's own warm-up ceiling in -// `agent/src/server.py`. Imported (not copied) so `tsc` fails on a renamed field, -// and `scripts/check-constants-sync.ts` enforces the `warmup_total < ready_hook` -// invariant plus the no-literal-redeclaration rule on both sides. See -// `contracts/constants.md`. -// Single source of truth for the supported-Region list. ADR-021's -// `microvm-regions.ts` header is explicit that the list "is the ONLY place the -// list is declared — do not copy it", so the synth-time gate IMPORTS it rather -// than duplicating it. The module is a dependency-free pair of pure constants -// (no AWS SDK, no Lambda-runtime code), so pulling it into the CDK app tree -// costs nothing and cannot drift. +// Shared hook budgets and Region constants prevent cross-package drift. import { AgentMemory } from './agent-memory'; import { AgentSessionRole } from './agent-session-role'; import { resolveBedrockModelIds } from './bedrock-models'; @@ -46,22 +35,13 @@ import sharedConstants from '../../../contracts/constants.json'; import { LAMBDA_MICROVM_SUPPORTED_REGIONS, isLambdaMicrovmRegionSupported } from '../handlers/shared/microvm-regions'; /** - * Lifecycle expiry for MicroVM `/run` hook payloads, in days. - * - * Mirrors {@link ECS_PAYLOAD_TTL_DAYS}. Finalization deletes payload.json and - * the private launch.json; shared manifests expire through lifecycle cleanup. - * Lifecycle expiry also removes task objects when finalization fails. S3 processes expiry asynchronously, not exactly 24 hours after - * upload. Payloads carry hydrated prompt context and are read once at `/run`. + * Fallback expiry for /run payloads and private launch references. Finalization + * deletes task objects; S3 lifecycle also cleans abandoned objects asynchronously. */ export const MICROVM_PAYLOAD_TTL_DAYS = 1; /** - * Cost-allocation tag applied to every resource this construct creates - * (ADR-021 sub-decision 4, "Cost attribution"). - * - * The stack-level `compute_type` tag in `main.ts` carries a single value and is - * already imprecise with two backends; these per-resource tags make MicroVM - * spend attributable regardless of what the stack-level tag says. + * Per-resource cost tag identifying MicroVM infrastructure in mixed-backend stacks. */ export const MICROVM_BACKEND_TAG_KEY = 'abca:compute-backend'; @@ -90,28 +70,8 @@ const AGENT_HOOK_PORT = sharedConstants.microvm_lifecycle.hook_port; const LIFECYCLE_HOOK_TIMEOUT_SECONDS = sharedConstants.microvm_hook_budgets.lifecycle_hook_timeout_seconds; /** - * Value every hook field on `AWS::Lambda::MicrovmImage` takes to turn a hook ON. - * - * **The field is an ENUM, not a path** — `[DISABLED, ENABLED]`. The CDK L1 types - * `hooks.microvmHooks.run` and its three siblings as plain `string` and documents - * no allowed values, which is why this construct originally sent the agent's - * route there. CloudFormation **rejected all four at change-set early validation** - * (live 2026-08-06, ADR-021 P2-F2 — the stack was never touched, so there was no - * rollback to read): - * - * > /aws/lambda-microvms/runtime/v1/run is not a valid enum value. Supported - * > values: [DISABLED, ENABLED] (at - * > /Resources/…/Properties/Hooks/MicrovmHooks/Run) - * - * So the CloudFormation surface is IDENTICAL to the `CreateMicrovmImage` API - * surface (`--hooks '{"microvmHooks":{"run":"ENABLED",…}}'`), not different from - * it as the previous comment here claimed. There is no hook-path field on either - * surface: the service calls fixed well-known routes, which - * {@link MICROVM_AGENT_HOOK_ROUTES} records and the live build/run logs confirm. - * - * `DISABLED` is never emitted: hooks outside the image's enabled capability are - * omitted. All six guest hooks are now declared for managed images. The separate - * supervisor enable flag still controls whether new automatic suspends are allowed. + * Hook fields accept ENABLED/DISABLED, not paths. Managed images enable all six + * served hooks; the coordinator separately controls automatic suspension. */ const HOOK_ENABLED = 'ENABLED'; @@ -122,28 +82,8 @@ const HOOK_ENABLED = 'ENABLED'; const MICROVM_HOOK_ROUTE_PREFIX = '/aws/lambda-microvms/runtime/v1'; /** - * The service's fixed routes for the hooks this image currently enables. - * `agent/src/server.py` must serve each; additional guest routes alone do not - * enable an image capability. - * - * ## ⚠️ These are AGENT ROUTE CONSTANTS ONLY. Never send them to an AWS API. - * - * They were previously passed as the `hooks.*` property VALUES on the L1, on the - * reasoning that the generated CloudFormation type accepts strings and documents - * no allowed-value constraint. CloudFormation refused every one of them - * (P2-F2 — see {@link HOOK_ENABLED} for the verbatim early-validation output): - * the hook fields are `ENABLED`/`DISABLED` enums on BOTH the CloudFormation and - * the API surface, and neither surface accepts a path at all. - * - * The routes are still worth declaring here because they are a CROSS-PACKAGE - * CONTRACT that nothing else in the CDK tree records: the service POSTs to these - * exact paths (live 2026-08-06 — `"POST /aws/lambda-microvms/runtime/v1/ready - * HTTP/1.1" 200 OK`, and the same for `/validate`, `/run` and `/terminate`), so a - * prefix drift between this map and the agent's `MICROVM_HOOK_PREFIX` surfaces as - * a failed image build (`/ready`, `/validate`) or a failed lifecycle transition on - * a real task (`/run`, `/terminate`). The construct test compares THIS map — not - * the rendered template, which no longer contains a path — against the routes the - * agent serves. + * Fixed service routes served by agent/src/server.py. Contract tests compare + * these paths with the guest routes; AWS hook properties take {@link HOOK_ENABLED}. */ export const MICROVM_AGENT_HOOK_ROUTES = { ready: `${MICROVM_HOOK_ROUTE_PREFIX}/ready`, @@ -155,136 +95,35 @@ export const MICROVM_AGENT_HOOK_ROUTES = { } as const; /** - * `/run` runtime-hook budget (seconds). - * - * `/run` is how the task payload reaches the agent (`runHookPayload`, ADR-021 - * sub-decision 3) — there is no other orchestrator→agent channel on this - * backend. The hook only validates + starts the pipeline asynchronously, so it - * stays well inside the service's 1–60 s runtime-hook window. + * /run validates the launch payload and starts the pipeline asynchronously. + * It must return within the service runtime-hook window. */ const RUN_HOOK_TIMEOUT_SECONDS = 60; /** - * `/ready` build-hook budget (seconds). - * - * `/ready` is **mandatory, not optional**: `CreateMicrovmImage` rejects an image - * that enables ANY lifecycle hook without it (live 2026-07-31): - * - * > The ready (/ready) MicroVM image hook must be enabled when any MicroVM - * > lifecycle hook (run, resume, suspend, or terminate) is enabled. The ready - * > hook signals when the application has finished initializing so the snapshot - * > is taken in a ready state. - * - * So ADR-021's original "declare `/run` in P1, serve it in P2" plan was not a - * reachable service state: `/ready` + `/run` land together in P1. - * - * **The budget is 300 s, not 60 s, because as of the P2-F5 fix `/ready` does real - * work.** It no longer just reports that uvicorn is bound: it warms the 225 MiB - * `claude` binary (`agent/src/server.py` → `_warm_snapshot_binaries`) so the - * binary's pages are resident when the snapshot is taken, instead of being faulted - * in lazily on the first task and blowing a timeout there (the defect that failed - * every P2 smoke task at turn 0). A cold 225 MiB `exec` is the one thing on this - * path that can plausibly take tens of seconds, and the build hook window allows - * up to 3600 s, so a 60 s budget would trade the runtime failure for a build - * failure. It stays far below the service ceiling: the cost of a too-generous - * budget is only how long the service waits before calling a permanently-wedged - * snapshot broken. - * - * The number is chosen against the agent's own warm-up ceiling, not guessed — and - * it is not declared here either. Both this budget and the agent's - * `_READY_WARMUP_TOTAL_BUDGET_SECONDS` come from `contracts/constants.json` → - * `microvm_hook_budgets`, because the relationship between them is the invariant - * that matters and a relationship cannot be enforced from one side. The agent's - * ceiling bounds the WHOLE warm-up (required command + every best-effort one, - * which share the remainder) at 240 s, leaving ~60 s here for uvicorn scheduling - * and the request itself. Per-command timeouts deliberately do NOT compose on the - * agent side — three commands at 120 s each would be 360 s and would blow this - * budget, turning a fix for a runtime failure into a build failure. - * `scripts/check-constants-sync.ts` fails the build if the contract ever stops - * satisfying `warmup_total < ready_hook`, so the two numbers cannot drift apart in - * a single-sided edit. + * /ready warms the required agent binary before the image snapshot is taken. + * The shared contract keeps the total guest warm-up budget below this hook timeout; + * scripts/check-constants-sync.ts enforces that relationship. */ const READY_HOOK_TIMEOUT_SECONDS = sharedConstants.microvm_hook_budgets.ready_hook_timeout_seconds; /** - * `/validate` build-hook budget (seconds). - * - * `/validate` is declared as of P2, when the agent started serving it. What it - * asserts is narrower than ADR-021 first sketched, and the narrowing is a - * consequence of THIS construct's IAM: it runs during the image build under - * {@link LambdaMicrovmCompute.buildRole}, which holds only `s3:GetObject` on the - * artifact plus log writes. So the "deeper warm-up assertions" (Bedrock - * reachability, Memory access, tool availability) are not implementable here — - * each would `AccessDenied` and fail every build. The agent's hook is therefore an - * in-process self-check (server alive, every declared hook route registered, - * interpreter floor, cross-package `platform_config` contract loaded), which is - * exactly the class of failure a build hook CAN catch: a typo'd hook prefix would - * otherwise surface as a failed lifecycle transition on the first real task - * instead of as a failed build. - * - * The checks themselves are sub-millisecond (no AWS calls, no I/O beyond a stdout - * line), so the budget is not sized for the work: it is sized for the - * still-initialising path, where the agent answers **503** until module import - * completes. 60 s covers that with orders of magnitude to spare. This used to be - * an alias for {@link READY_HOOK_TIMEOUT_SECONDS} on the argument that one number - * should cover both build hooks; the two DECOUPLED when `/ready` gained the - * binary warm-up (P2-F5) and `/validate` did not, so sharing a number would now - * mean sizing `/validate` for work it does not do. Set explicitly rather than - * relying on the service's 30 s default: a permanently failing check SHOULD fail - * the image build, and the budget is what decides how long the service waits - * before calling it that. + * /validate checks local readiness, hook registration and configuration contracts. + * It makes no AWS calls: the build role lacks runtime data and model permissions. */ const VALIDATE_HOOK_TIMEOUT_SECONDS = 60; /** - * `/terminate` runtime-hook budget (seconds). - * - * `/terminate` is declared as of P2. It is a log-and-acknowledge breadcrumb, NOT - * a shutdown mechanism: the orchestrator finalizes the task and *then* calls - * `TerminateMicrovm`, so the hook must not write terminal task status (it would - * race the finalization it follows) and must not join the pipeline thread. Its - * value is the last structured line in the task's log group from inside the guest. - * - * This is the one hook where a GENEROUS budget buys nothing and costs something. - * There is nothing to drain — `_ProgressWriter` does a synchronous `put_item` per - * event, so every progress write is already durable when this hook is called — - * and the handler never joins the pipeline thread, so it completes in - * milliseconds by construction. Meanwhile the budget bounds how long teardown - * waits on a guest that is WEDGED, and a MicroVM that has not finished - * terminating is still holding the account memory quota that gates admission for - * everyone else. - * - * So this is set near the bottom of the service's 1–60 s window rather than at - * it: 15 s is ~three orders of magnitude above the measured work, which absorbs - * a scheduling delay on a guest still saturated by a build (the realistic reason - * a fast handler answers slowly), while keeping teardown prompt. Exceeding it - * costs only a reported hook failure — the task is already finalized and - * `TerminateMicrovm` removes the VM regardless — which is why erring tight is - * the safe direction here and erring generous is not. + * /terminate logs and acknowledges teardown without joining the pipeline or + * writing task status. Termination can interrupt a task; coordinator recovery owns + * its resulting state. Keep the hook budget short so a stuck guest cannot delay + * teardown for the full runtime-hook window. */ const TERMINATE_HOOK_TIMEOUT_SECONDS = 15; /** - * BASELINE memory sizes (MiB) the service accepts for a MicroVM image. - * - * NOT a range and NOT a per-VM ceiling: the service enumerates the allowed - * baseline values per base image and rejects anything else. Live 2026-07-31 - * against `arn:aws:lambda:us-east-1:aws:microvm-image:al2023-1`: - * - * > The requested memory size of 32768 MiB is not supported by base MicroVM - * > image …al2023-1. Supported memory sizes in MiB are: - * > [512, 1024, 2048, 4096, 8192]. - * - * The service then scales a running MicroVM VERTICALLY on demand, up to a - * 32 GiB / 16 vCPU peak (developer-guide sizing table) — so 8 GiB is the top of - * the *configurable baseline*, not the amount of memory a task can use. The - * boundary probe above establishes what the FIELD accepts; only the guide - * establishes what the field MEANS (ADR-021's source hierarchy). - * - * Kept as an exported constant so {@link LambdaMicrovmComputeProps.minimumMemoryInMiB} - * can fail at synth with the real list instead of at image-create time. The - * individually named sizes exist only to keep the list out of `no-magic-numbers` - * territory — the list itself is the contract. + * Baseline values accepted by the al2023-1 image in live validation. This list + * does not establish guest-visible launch memory or capacity-change timing. */ const MEMORY_512_MIB = 512; const MEMORY_1_GIB_IN_MIB = 1024; @@ -300,15 +139,8 @@ export const MICROVM_SUPPORTED_MEMORY_MIB: readonly number[] = [ ]; /** - * Baseline memory the image declares, in MiB — the largest baseline the service - * accepts (see {@link MICROVM_SUPPORTED_MEMORY_MIB}). - * - * The ABCA agent is a build-heavy workload, so P1 asks for the top of the - * accepted baseline list rather than a smaller one it would spend the whole task - * scaling up from. Automatic vertical scaling then supplies burst capacity to a - * 32 GiB peak; nothing here requests that, and nothing can. - * - * This was `32768` until live verification refuted it as a *baseline* value. + * Baseline used by the verified ABCA workloads. Lower values require workload + * measurements; the accepted-value probe did not measure a performance advantage. */ export const DEFAULT_MINIMUM_MEMORY_MIB = MEMORY_8_GIB_IN_MIB; @@ -319,60 +151,26 @@ const LOG_RETENTION = logs.RetentionDays.THREE_MONTHS; const HTTPS_PORT = 443; /** - * HTTP port. Allowed on the **build-time** connector only. - * - * `agent/Dockerfile` installs Debian packages, and `apt-get` fetches over plain - * HTTP. With a 443-only egress path every snapshot build failed (live - * 2026-07-31): `Could not connect to deb.debian.org:80 … E: Unable to locate - * package curl` → `exit code: 100`. DNS resolved fine — the port was the sole - * cause. Opening 80 for the build path (and only the build path) made the build - * succeed immediately; see {@link LambdaMicrovmCompute.buildSecurityGroup}. + * Build-only HTTP egress for apt-get; runtime egress remains HTTPS-only. */ const HTTP_PORT = 80; /** - * Graviton/ARM64: the agent image is ARM64 on every backend. - * - * The value is the service's **enum member spelling**, `ARM_64` — not the - * lowercase `arm64` Docker/CDK use elsewhere. The CDK L1 types - * `cpuConfigurations[].architecture` as a plain `string` and documents no allowed - * values, and `arm64` was rejected at change-set early validation (live - * 2026-08-06, ADR-021 P2-F2): *"arm64 is not a valid enum value. Supported - * values: [ARM_64]"*. Matches `--cpu-configurations '[{"architecture":"ARM_64"}]'` - * in `cdk/scripts/package-microvm-artifact.sh`, which had it right all along. + * The MicroVM API spells its ARM64 enum ARM_64; Docker uses arm64. */ const CPU_ARCHITECTURE = 'ARM_64'; +const MAX_MANAGED_IMAGE_VERSION_LENGTH = 64; /** - * Resource-name half of the Lambda-managed **`NO_INGRESS`** network connector - * ARN, i.e. everything after `…:aws:network-connector:`. - * - * Why this exists at all: `RunMicrovm` does **not** default to "no ingress". A - * launch that omits `ingressNetworkConnectors` entirely came back (live - * 2026-07-31) with a service-attached PUBLIC connector and a public endpoint: - * - * ``` - * "ingressNetworkConnectors": ["arn:aws:lambda:us-east-1:aws:network-connector:aws-network-connector:HTTP_INGRESS"] - * ``` - * - * ADR-021's "no inbound exposure" posture therefore has to be an **explicit - * control**, not an omission: the strategy passes the `NO_INGRESS` connector on - * every `RunMicrovm`. The ARN shape is taken verbatim from that observed - * `HTTP_INGRESS` ARN, with the connector name swapped. - * - * A service endpoint URL is still returned with `NO_INGRESS` (verified - * 2026-09-17). Its existence does not prove that requests reach the guest: - * endpoint requests require a MicroVM auth token. An unauthenticated 403 tests - * that authentication boundary, not the connector's handling of valid tokens. + * Explicit no-ingress connector. Omitting ingress connectors let the service + * attach HTTP_INGRESS in live validation. NO_INGRESS can still return an endpoint + * URL; an unauthenticated 403 tests authentication, not valid-token reachability. */ export const MICROVM_NO_INGRESS_CONNECTOR_RESOURCE = 'aws-network-connector:NO_INGRESS'; /** - * ARN of the Lambda-managed `NO_INGRESS` connector in {@link scope}'s - * partition/Region (see {@link MICROVM_NO_INGRESS_CONNECTOR_RESOURCE}). - * - * Account segment is the literal `aws` — these connectors are service-owned, not - * account-owned, exactly like the AWS-managed policy ARNs. + * ARN of the service-owned NO_INGRESS connector in this partition and Region. + * Its account segment is the literal aws. */ export function microvmNoIngressConnectorArn(scope: Construct): string { return Stack.of(scope).formatArn({ @@ -385,36 +183,14 @@ export function microvmNoIngressConnectorArn(scope: Construct): string { } /** - * CDK context flag that bypasses the synth-time Region gate. - * - * Exists because {@link LAMBDA_MICROVM_SUPPORTED_REGIONS} rots by design: when - * AWS launches Lambda MicroVMs in a new Region, an operator there must not have - * to wait for an ABCA release. The live probes (CLI onboarding, `platform - * doctor`) already accept the new Region, so this flag only unblocks synth. + * Escape hatch for Regions launched after the static support list was updated. + * This bypasses only synth validation; live service checks still apply. */ export const MICROVM_REGION_OVERRIDE_CONTEXT = 'microvm_region_override'; /** - * Fail synth when the `lambda-microvm` backend is enabled in a Region that is - * not in the statically documented support list (ADR-021 sub-decision 4, - * "Regional availability enforcement" — the synth/deploy row). - * - * Three behaviours worth knowing: - * - * - **Unresolved (token) Region → check SKIPPED.** A region-agnostic app - * (`new Stack(app, 'X')` with no `env`, or `env.region` left to the CLI) - * resolves `Stack.region` to the `AWS::Region` pseudo-parameter, whose value - * is unknowable at synth. Comparing a token against a Region list would - * reject every region-agnostic synth — including `cdk synth` on a developer - * box with no `CDK_DEFAULT_REGION` — so the static layer stands down and the - * live probes (onboarding / doctor / orchestration classification) carry the - * enforcement. This is a deliberate hole in the *static* layer only. - * - **Escape hatch.** `--context microvm_region_override=true` skips the check; - * the error message names the flag so an operator in a just-launched Region - * is never blocked on a code change. - * - **Failure is a synth-time throw, not a warning.** ADR-021 requires "synth - * fails when ComputeTypes includes lambda-microvm in an unlisted Region"; a - * warning would let a broken deploy through to a runtime AccessDenied. + * Reject a concrete unsupported Region at synth. Unresolved Regions defer to + * runtime checks; microvm_region_override permits newly supported Regions. */ export function assertLambdaMicrovmRegionSupported(scope: Construct): void { const region = Stack.of(scope).region; @@ -450,33 +226,21 @@ export function assertLambdaMicrovmRegionSupported(scope: Construct): void { } /** - * Operator-supplied image inputs, read from CDK context by the stack. - * - * Extracted into a type so the stack can resolve them ONCE, before `TaskApi` is - * constructed, and hand the same object to this construct — see - * {@link isLambdaMicrovmImageConfigured} for why that ordering matters. + * Shared image inputs resolved before TaskApi so lifecycle IAM and this construct + * use the same image-availability decision. */ export interface LambdaMicrovmImageInputs { readonly baseImageArn?: string; readonly baseImageVersion?: string; readonly artifactSha256?: string; + readonly managedImageVersion?: string; readonly externalImageIdentifier?: string; readonly externalImageVersion?: string; } /** - * True when {@link inputs} selects one of the two image-provisioning states - * (managed-base-image build, or an out-of-band image), i.e. the deployment will - * have a MicroVM image and therefore an `imageArn` to scope IAM against. - * - * Exists so the three-state decision documented on {@link LambdaMicrovmCompute} - * is made in exactly ONE place. `TaskApi` is constructed before this construct - * (the cancel Lambda's ARN is needed earlier, hence the `Lazy.string` holders in - * `stacks/agent.ts`), so the stack must know whether an image will exist *before* - * the construct that creates it runs. Without a shared predicate the stack and - * the construct would each re-derive that answer and could drift — and a drift - * here means either a missing cancel grant or a grant scoped to an image that - * does not exist. + * Whether managed or external image inputs select an image. The stack uses this + * before constructing TaskApi to decide whether to grant lifecycle permissions. */ export function isLambdaMicrovmImageConfigured(inputs: LambdaMicrovmImageInputs): boolean { return Boolean((inputs.baseImageArn && inputs.baseImageVersion) || inputs.externalImageIdentifier); @@ -501,99 +265,43 @@ export interface LambdaMicrovmComputeProps extends LambdaMicrovmImageInputs { readonly connectorOperatorRoleName?: string; /** - * Platform VPC. Egress leaves the MicroVM through a `AWS::Lambda::NetworkConnector` - * bound to this VPC's private-with-egress subnets, so the DNS Firewall / - * security-group / flow-log stack applies to MicroVM traffic unchanged - * (ADR-021 security table: "Egress ... None" delta). + * Platform VPC for build and runtime connectors, using private-with-egress + * subnets and the existing DNS Firewall, NAT and flow logs. */ readonly vpc: ec2.IVpc; /** - * Per-task SessionRole (#209). When provided, the MicroVM **execution role** - * is admitted to the SessionRole's trust via - * {@link AgentSessionRole.admitComputeRole} — the mechanism was designed for - * exactly this (ADR-021 sub-decision 4) — so tenant-data access stays on the - * tag-scoped SessionRole instead of the execution role. Mirrors how - * `EcsAgentCluster` delegates the Fargate task role. Omitted in isolated - * construct tests, in which case NO tenant-data access is granted at all - * (this construct never grants DynamoDB directly: unlike the ECS backend - * there is no legacy direct-grant path to preserve). + * Per-task role providing tag-scoped tenant access. The execution role can assume + * it when supplied; omitting it grants no direct DynamoDB or artifact access. */ readonly agentSessionRole?: AgentSessionRole; /** - * GitHub PAT secret. When provided, the MicroVM **execution role** gets - * `grantRead` on it. - * - * This grant stays on the execution role rather than moving to the SessionRole - * because of WHEN it is used: the agent resolves the token at startup, before - * it has assumed the SessionRole — the same ordering that keeps the grant on the - * ECS task role and the AgentCore runtime role. Without it the MicroVM cannot - * clone, push, or open a PR, which is the whole task. - * - * Omitted in isolated construct tests → no grant. + * GitHub credential read by the execution role before the task assumes its + * SessionRole. Omit only when that startup credential is unnecessary. */ readonly githubTokenSecret?: secretsmanager.ISecret; /** - * AgentCore Memory for cross-task learning. When provided, the execution role - * gets read+write so the agent's `write_task_episode` / `write_repo_learnings` - * (`bedrock-agentcore:CreateEvent`) succeed on this substrate. - * - * Exactly the prop `EcsAgentCluster` takes, for exactly the same reason: the - * `MEMORY_ID` the agent receives (in `agent_payload`, unchanged by ADR-021 P2) - * makes it ATTEMPT the write, and without the grant that attempt fails closed on - * AccessDenied. `memory.py` treats that as an infra failure — logged, - * non-fatal — so learning would silently never persist on a MicroVM-only - * deployment. Omitted in isolated construct tests / memory-less deployments. + * Platform Memory used for cross-task learning. Grants execution-role read/write + * access; omit for deployments without Memory. */ readonly agentMemory?: AgentMemory; /** - * The platform's APPLICATION_LOGS group — the same log group whose NAME travels - * to the guest as `platform_config.log_group_name` (`stacks/agent.ts` → - * `TaskOrchestrator.agentPlatformConfig` → `LOG_GROUP_NAME`). When provided, the - * MicroVM **execution role** gets `logs:CreateLogStream` + `logs:PutLogEvents` - * on it. - * - * Not optional in spirit — omitted only in isolated construct tests. P2 wired - * the name into `platform_config`, which makes the agent ATTEMPT the write, and - * shipped without the matching grant, so every structured per-task log line was - * denied (live 2026-08-07, ADR-021 P2-F4): - * - * > User: …:assumed-role/…LambdaMicrovmComputeExecutionRo…/Lambda-microvmsExecutor-… - * > is not authorized to perform: logs:CreateLogStream on resource: - * > …:log-group:/aws/vendedlogs/bedrock-agentcore/runtime/APPLICATION_LOGS/… - * - * The role's OTHER logs grant ({@link LambdaMicrovmCompute.grantMicrovmLogWrites}) - * is scoped to the service's own `/aws/lambda-microvms/*` namespace and cannot - * cover this group — the two namespaces are unrelated. Non-fatal (the agent - * degrades to stdout, which the MicroVM log group captures) but it empties the - * platform's canonical per-task observability streams, `METRICS_REPORT` - * included, on this backend only. Exactly the omission class the P2 Bedrock / - * Secrets Manager / Memory grants exist to close. + * Platform APPLICATION_LOGS group named in platform_config. Its write grant is + * separate from the service-owned /aws/lambda-microvms log namespace. */ readonly applicationLogGroup?: logs.ILogGroup; /** - * ARN of the Lambda-managed base MicroVM image to build on - * (`aws lambda-microvms list-managed-microvm-images`), e.g. - * `arn:aws:lambda:us-east-1:aws:microvm-image:al2023-1`. - * - * Supplying this **and** {@link baseImageVersion} switches the construct into - * its primary mode: it synthesizes an `AWS::Lambda::MicrovmImage` (L1) whose - * `codeArtifact.uri` points at {@link artifactObjectKey} in the artifact - * bucket this construct creates. There is no default — base-image ARNs are - * account/Region-scoped service data that is only discoverable through a live - * API call, so hardcoding one would be a guess that rots. + * Service-managed base image ARN, discovered with list-managed-microvm-images. + * With baseImageVersion and artifactSha256, creates a CloudFormation-managed image. */ readonly baseImageArn?: string; /** - * Version of {@link baseImageArn} - * (`aws lambda-microvms list-managed-microvm-image-versions`). Required - * alongside `baseImageArn` because CloudFormation marks it required on - * `AWS::Lambda::MicrovmImage` even though the API treats it as optional. + * Base image version, required alongside baseImageArn by CloudFormation. */ readonly baseImageVersion?: string; @@ -604,14 +312,15 @@ export interface LambdaMicrovmComputeProps extends LambdaMicrovmImageInputs { readonly artifactSha256?: string; /** - * Identifier (name or ARN) of a MicroVM image built **out of band** — i.e. by - * running `cdk/scripts/package-microvm-artifact.sh` and then - * `aws lambda-microvms create-microvm-image` by hand. - * - * Only consulted when {@link baseImageArn} is absent. It exists so an - * operator can iterate on the snapshot (which takes minutes and often several - * attempts) without a stack update per attempt, then hand the finished image - * to the orchestrator. + * Optional version to RUN from the managed image (for example `7.0`). + * This does not alter the image build or transfer its CloudFormation ownership. + * Omit to run the latest active version; pin a verified version for rollback. + */ + readonly managedImageVersion?: string; + + /** + * Name or ARN of an image built outside this construct. Used only when managed + * base-image inputs are absent; enables image iteration without a stack build. */ readonly externalImageIdentifier?: string; @@ -637,137 +346,41 @@ export interface LambdaMicrovmComputeProps extends LambdaMicrovmImageInputs { readonly artifactObjectKey?: string; /** - * **Baseline** memory the image declares, in MiB. - * - * This is the size the MicroVM STARTS at, not a cap on what it may use: the - * service scales a running MicroVM vertically on demand, up to a 32 GiB / - * 16 vCPU peak, with no field to request that. So the practical effect of this - * prop is where the VM begins (and what it is baseline-priced at), not whether - * a build will fit. - * - * Must be one of {@link MICROVM_SUPPORTED_MEMORY_MIB} — the construct throws at - * synth otherwise, because the service rejects any other baseline at - * image-create time with a `ValidationException` an operator would only see - * minutes into a build. - * - * @default 8192 — the largest accepted baseline (see DEFAULT_MINIMUM_MEMORY_MIB) + * Service baseline in MiB; must be one of {@link MICROVM_SUPPORTED_MEMORY_MIB}. + * This setting does not guarantee workload fit or capacity-change timing. + * @default 8192 — the baseline used in live ABCA verification */ readonly minimumMemoryInMiB?: number; /** - * Non-secret environment variables baked into the snapshot at build time. - * - * Empty by default; the construct adds only its invariant lifecycle protocol marker. ADR-021 - * sub-decision 3 forbids secrets, tokens, and per-task identity in the snapshot - * — and P2 resolved the remaining question (where the agent's non-secret - * configuration parity with the ECS container comes from) in favour of the - * `/run` payload's `platform_config` block, NOT this prop. A snapshot is shared - * across every task and every deployment that reuses it, so a table or bucket - * name baked in here would be a deploy-time value frozen at image-build time — - * stale the moment the stack is redeployed. Reach for this only for genuinely - * image-invariant settings (a locale, a toolchain path). - * @default {} — no baked configuration + * Image-invariant, non-secret settings only. The construct also adds its lifecycle + * protocol marker. Deployment configuration, credentials and task identity arrive + * through /run and must not be baked into a shared snapshot. + * @default {} — no caller-supplied settings */ readonly imageEnvironmentVariables?: Record; } /** - * AWS Lambda MicroVMs compute backend — infrastructure half (ADR-021, - * sub-decision 4; P1 of the phased rollout). - * - * Provisions, in dependency order: + * Lambda MicroVM infrastructure: build/runtime VPC connectors, image artifacts, + * bootstrap payloads, logs, roles and an optional managed image (ADR-021). * - * 1. **Egress network connectors** (`AWS::Lambda::NetworkConnector`) on the - * platform VPC's private-with-egress subnets — TWO of them, sharing one - * operator role: - * - the **runtime** connector with a 443-only security group. This is what - * keeps the ADR's "Egress: no delta vs AgentCore/ECS" claim true — MicroVM - * traffic traverses the same NAT / DNS Firewall / flow logs as the other - * two backends. - * - the **build-time** connector with a 443 **and 80** security group, - * referenced only by the image resource. `agent/Dockerfile` runs - * `apt-get`, which is plain HTTP; a 443-only build path fails every - * snapshot build (see {@link HTTP_PORT}). Runtime egress stays 443-only. - * 2. **Artifact bucket** for the zip + Dockerfile the service builds the - * snapshot from, and a **payload bucket** for deployment manifests, task instructions and private - * launch references. Every task uses the authenticated v2 transport. - * 3. **Build role** — assumed by Lambda during image creation: `s3:GetObject` - * on the artifact object and CloudWatch Logs writes. Without it Lambda - * cannot emit build logs, which makes a failed snapshot build undebuggable. - * 4. **Execution role** — assumed by the running MicroVM: CloudWatch Logs (both - * the service's own `/aws/lambda-microvms/*` namespace and the platform - * APPLICATION_LOGS group whose name `platform_config` delivers), read access only to - * the payload bucket's bootstrap manifests, the P2 runtime-parity grants (GitHub PAT + - * channel-OAuth secret reads, scoped Bedrock invocation, AgentCore Memory, - * `ec2:DescribeAvailabilityZones` for a CDK repo's synth gate), and — when a - * SessionRole is wired — admission to the per-task SessionRole, which is the - * ONLY path to tenant data. - * 5. **MicroVM image** (`AWS::Lambda::MicrovmImage`) — see {@link baseImageArn} - * for why this is conditional. + * Runtime egress permits HTTPS; the separate build connector also permits HTTP + * for apt-get. Every launch explicitly selects NO_INGRESS. The execution role + * reads bootstrap manifests and startup credentials, invokes models, writes logs + * and uses Memory. Tenant data remains on the per-task SessionRole; lifecycle + * control remains on coordinator and decision-handler roles. * - * ## Image provisioning: three states, one construct + * Image provisioning has three states: + * - baseImageArn + baseImageVersion + artifactSha256: build a managed image. + * - externalImageIdentifier: use an image built outside this construct. + * - neither: create infrastructure for the initial artifact upload, with a warning; + * tasks cannot run until an image is configured. * - * | Props supplied | What happens | When to use it | - * |---|---|---| - * | `baseImageArn` + `baseImageVersion` + `artifactSha256` | `AWS::Lambda::MicrovmImage` L1 uses the immutable hash-suffixed artifact key; {@link imageIdentifier} is its ARN | steady state | - * | `externalImageIdentifier` | no image resource; the supplied identifier is resolved to its exact ARN and handed to the orchestrator | iterating on the snapshot out of band | - * | neither | roles + buckets + connectors only; a synth-time **warning**, no image, and no `MICROVM_IMAGE_IDENTIFIER` for the orchestrator | first deploy — you cannot upload the artifact before the bucket that holds it exists | - * - * That third state is not an oversight: the artifact bucket is created by this - * stack, so the very first `--context compute_type=lambda-microvm` deploy has - * nowhere to have put the zip yet. It is a **warning rather than a throw** - * precisely so the bootstrap sequence (deploy → run the packaging script - * against the now-existing bucket → redeploy with the base image and printed - * `microvm_artifact_sha256`) is - * possible at all. A `lambda-microvm` task submitted in that interim window - * fails fast with the strategy's own "stack deployed without the MicroVM - * substrate" error, which names the remedy. - * - * ## P2 clean smoke passed; broader acceptance remains open - * - * Reaching state 1 or 2 provisions a complete substrate, a buildable image, and - * a payload-deliverable `/run` path: P1 declares AND the agent serves `/ready` - * and `/run` (`agent/src/server.py`), because live verification proved the - * original "declare in P1, serve in P2" split was not a reachable service state - * — `CreateMicrovmImage` refuses any lifecycle hook without `/ready` (see - * {@link READY_HOOK_TIMEOUT_SECONDS}), and an image with no hooks at all cannot - * receive a `runHookPayload`. - * - * P2 adds the two halves an agent needs to actually finish a task: the runtime - * IAM parity on the execution role (see item 4 above) and non-secret - * configuration delivery through the `/run` payload's `platform_config` block - * (`handlers/shared/strategies/lambda-microvm-strategy.ts`) — the substitute for - * the env block the other two backends get at deploy time, since the snapshot must - * not bake it in. It also declares the two hooks the agent gained in the same - * phase: `/validate` (build-time self-check — see - * {@link VALIDATE_HOOK_TIMEOUT_SECONDS} for why the build role's permissions bound - * what it can assert) and `/terminate` (in-guest teardown breadcrumb — - * {@link TERMINATE_HOOK_TIMEOUT_SECONDS}). - * - * The 2026-09-14 clean deployment with bootstrap bundle 1.7.0 and subsequent - * coding, iteration and cancellation runs passed without manual IAM changes, - * including heartbeat, runtime logs, Memory writes and cleanup. The full - * failure/recovery, effective IAM and network matrix remains open; see - * docs/verification/645-p3-implementation-plan.md. The stable - * `abca:microvm-image-p1-smoke-unverified` warning below records that scope. - * P3 also declares the served `/suspend` and `/resume` hooks and bakes a non-secret - * protocol marker into the image. The coordinator verifies the actual launched - * version before allowing suspension. Supervisor integration is implemented; - * live acceptance remains a separate rollout gate and automatic sleep defaults off. - * - * ## Deliberately NOT here - * - * The execution role gets no artifacts-bucket grant and no DynamoDB grant: an - * artifact delivery write goes through the SessionRole's - * `artifacts/${task_id}/*` statement (the AgentCore runtime role has no direct - * grant either) and every table the agent touches is `task_id`-partitioned - * SessionRole territory. It also has no UserConcurrencyTable grant — that counter - * is orchestrator/reconciler-owned and the agent path never writes it. - * The coordinator receives scoped `lambda:SuspendMicrovm` / `lambda:ResumeMicrovm` - * grants, and decision handlers also receive scoped Resume permission. Neither - * action is granted to this construct's build or execution role. - * `lambda:CreateMicrovmAuthToken` is granted to no role in any phase — no JWE - * consumer exists (sub-decision 3). + * Managed images enable all six served hooks. Automatic approval sleep requires + * a compatible image and coordinator plus the deployment enable switch. P3 live + * acceptance is recorded in docs/verification/645-p3-implementation-plan.md; new + * installations must verify their own configuration before enabling sleep. */ export class LambdaMicrovmCompute extends Construct { /** S3 bucket holding the zip + Dockerfile the snapshot is built from. */ @@ -788,10 +401,7 @@ export class LambdaMicrovmCompute extends Construct { public readonly executionRole: iam.Role; /** - * Role Lambda assumes to manage the connectors' ENIs in the platform VPC. - * - * REQUIRED for `VPC_EGRESS` connectors, despite the generated L1 typing - * `operatorRole` as optional — see the comment at the connector below. + * Service role managing connector ENIs; required for VPC_EGRESS. */ public readonly connectorOperatorRole: iam.Role; @@ -811,11 +421,7 @@ export class LambdaMicrovmCompute extends Construct { public readonly buildEgressConnectorArns: string[]; /** - * Ingress connectors passed on every `RunMicrovm` - * (`MICROVM_INGRESS_CONNECTOR_ARNS`). In P1–P3 this is exactly the - * Lambda-managed `NO_INGRESS` connector: the service's default is a PUBLIC - * `HTTP_INGRESS`, so "no inbound" has to be requested explicitly (see - * {@link MICROVM_NO_INGRESS_CONNECTOR_RESOURCE}). + * Explicit NO_INGRESS connector passed on every RunMicrovm request. */ public readonly ingressConnectorArns: string[]; @@ -834,16 +440,14 @@ export class LambdaMicrovmCompute extends Construct { /** Image name used for the image resource and the log group. */ public readonly imageName: string; - /** BASELINE memory (MiB) declared on the image; the service bursts above it. */ + /** + * Configured image memory baseline in MiB. + */ public readonly minimumMemoryInMiB: number; /** - * Value for `MICROVM_IMAGE_IDENTIFIER`. **Always a full image ARN**, never a - * bare name — `RunMicrovm` rejects bare names outright - * (`ValidationException: Malformed ARN - doesn't start with 'arn:'`, live - * 2026-07-31), and so does `list-microvm-image-builds`. `undefined` in the - * neither-input-supplied bootstrap state, in which case the stack must not - * inject the MicroVM env block at all. + * Full image ARN for RunMicrovm. Undefined during the no-image bootstrap phase; + * bare external names are resolved to ARNs for both launch and IAM. */ public readonly imageIdentifier?: string; @@ -851,13 +455,8 @@ export class LambdaMicrovmCompute extends Construct { public readonly imageVersion?: string; /** - * The image's IAM resource ARN. Always set whenever {@link imageIdentifier} - * is — a bare image name is resolved to its full - * `…:microvm-image:` ARN — so the orchestrator's `lambda:RunMicrovm` / - * `GetMicrovm` / `TerminateMicrovm` grant is *always* scoped to this one - * platform-created image and never widens to an account-level wildcard. - * `undefined` only in the no-image bootstrap state, where no grant is issued - * at all. + * Exact image ARN used to scope lifecycle IAM; undefined only before an image + * is configured. Image versions are separate request fields, not ARN suffixes. */ public readonly imageArn?: string; @@ -896,10 +495,8 @@ export class LambdaMicrovmCompute extends Construct { throw new Error( `minimumMemoryInMiB=${this.minimumMemoryInMiB} is not a BASELINE memory size AWS Lambda ` + `MicroVMs accepts. Supported baselines (MiB): ${MICROVM_SUPPORTED_MEMORY_MIB.join(', ')}. ` - + `The default is ${DEFAULT_MINIMUM_MEMORY_MIB}, the largest accepted baseline. Note this is ` - + 'a BASELINE, not a cap: the service scales a running MicroVM vertically to a 32 GiB / ' - + '16 vCPU peak on its own, so asking for more here is neither possible nor necessary. A ' - + 'repo needing more than that SUSTAINED belongs on compute_type=ecs (16 vCPU / 120 GB).', + + `The default is ${DEFAULT_MINIMUM_MEMORY_MIB}. Choose a supported baseline and verify ` + + 'that the workload fits; this setting alone does not establish available peak capacity.', ); } @@ -943,49 +540,10 @@ export class LambdaMicrovmCompute extends Construct { 'Allow HTTP egress for apt-get during the snapshot build (build path only)', ); - // OPERATOR ROLE — required, despite the generated L1 typing it optional. - // - // `CfnNetworkConnector.operatorRole?: string` reads like an opt-in, and this - // construct originally left it unset "so Lambda manages the ENIs with its own - // service-linked role". Live deploy 2026-07-31 refuted that outright — the - // connector never reaches CREATE_COMPLETE without it: - // - // NetworkConnectorOperatorRole is required for VPC_EGRESS connector type - // (Service: Lambda, Status Code: 400) HandlerErrorCode: InvalidRequest - // - // The permission set below is the minimal recipe validated standalone in that - // run: `AWSLambdaVPCAccessExecutionRole` plus the ENI/tag/private-IP actions - // the managed policy omits. - // - // ONE role for BOTH connectors: they differ only in security group, both are - // created and owned by this construct in the same VPC, and a second identical - // role would double the IAM surface a reviewer has to check for no isolation - // gain (the role manages ENIs, not traffic). - // - // --- TRUST POLICY: bare service principal, NO source conditions (P2-F1/F3) --- - // - // Shared by all three MicroVM-facing roles (this one, `buildRole`, - // `executionRole`) because the decision is one decision. `lambda.amazonaws.com` - // is the correct principal — there is no `microvms.lambda.amazonaws.com`, and - // using one is rejected at role-creation time with MalformedPolicyDocument. - // - // ⚠️ **`aws:SourceAccount`/`aws:SourceArn` are deliberately absent. Adding one - // back re-breaks the deploy.** The Lambda MicroVMs service populates NO source - // condition key when it assumes these roles, so a trust policy carrying one is - // unassumable: both network connectors CREATE_FAILED deterministically, and - // `RunMicrovm` surfaced the same root cause as a misleading caller-side - // `iam:PassRole` denial on the orchestrator. Removing the conditions fixed both - // within seconds. This looks like a regression to anyone applying the standard - // confused-deputy pattern — and it already WAS one in the other direction: P1's - // working probe had no conditions, and the P1 F2 fix added them "to mirror the - // build/execution roles". `sts:TagSession` stays; it was never implicated. - // - // ADR-021 §4 is authoritative for the rest: the live evidence (P2-F1 / P2-F3, - // verbatim failures, the two-arm PassRole experiment, the contaminated control - // that produced run 1's false negative), the per-role compensating-controls - // table, and the conditions under which the condition could be restored. The - // trust shape here is asserted by `test/constructs/lambda-microvm-compute.test.ts` - // ("NO source-key condition on any MicroVM-facing role trust"). + // VPC_EGRESS requires an operator role shared by the two connectors. + // Live P2 checks rejected source-conditioned trust on all three roles and + // conditioned PassRole on the build/execution paths. Keep lambda.amazonaws.com + // without source conditions; ADR-021 §4 records evidence and compensating scopes. const microvmAssumedBy = new iam.ServicePrincipal('lambda.amazonaws.com'); this.connectorOperatorRole = new iam.Role(this, 'ConnectorOperatorRole', { roleName: props.connectorOperatorRoleName, @@ -1011,20 +569,12 @@ export class LambdaMicrovmCompute extends Construct { 'ec2:UnassignPrivateIpAddresses', 'ec2:CreateTags', ], - // EC2 network-interface APIs are largely non-resource-scopable (the ENI - // does not exist when CreateNetworkInterface is authorized, and the - // Describe* calls take no resource), which is exactly why the AWS-managed - // VPC-access policy uses `*` too. See the cdk-nag suppression below. + // Describe actions need wildcard resources. ENI mutations share this tested + // wildcard statement; their scope is a separate IAM-hardening consideration. resources: ['*'], })); - // The connector, not the MicroVM, owns the ENIs — which is why - // `lambda:PassNetworkConnector` is required on the orchestrator even for - // AWS-managed connectors (ADR-021 sub-decision 4). - // - // `associatedComputeResourceTypes: ['MicroVm']` is the only value the - // service accepts today (CloudFormation: "Currently, only MicroVm is - // supported"). + // Connectors own the VPC interfaces. The service accepts MicroVm here. const vpcEgressSubnetIds = props.vpc.selectSubnets({ subnetType: ec2.SubnetType.PRIVATE_WITH_EGRESS, }).subnetIds; @@ -1117,17 +667,8 @@ export class LambdaMicrovmCompute extends Construct { autoDeleteObjects: true, }); - // --- Roles --- - // - // TRUST POLICY: all three MicroVM-facing roles share `microvmAssumedBy` — - // the BARE `lambda.amazonaws.com` service principal, with NO source - // conditions. The warning and the pointer live at that constant's - // declaration, above the connector operator role that needs it first; ADR-021 - // §4 carries the evidence (P2-F1 / P2-F3) and the per-role compensating - // controls. The - // build and execution roles additionally need `sts:TagSession` alongside - // `sts:AssumeRole` (developer guide, "Trust policies"), which - // {@link grantTagSession} adds. + // Build and execution roles also require sts:TagSession. Trust constraints + // are documented beside microvmAssumedBy above. this.buildRole = new iam.Role(this, 'BuildRole', { roleName: props.buildRoleName, @@ -1154,15 +695,7 @@ export class LambdaMicrovmCompute extends Construct { // `grantMicrovmLogWrites` for the runbook citations and the re-verify note. this.grantMicrovmLogWrites(this.executionRole, { allowCreateLogGroup: false }); - // The APPLICATION_LOGS group the agent is TOLD to write to (P2-F4). Separate - // from `grantMicrovmLogWrites` above and not reachable from it: that grant - // covers the service-owned `/aws/lambda-microvms/*` namespace, while - // `platform_config.log_group_name` points at the platform's vended - // `/aws/vendedlogs/bedrock-agentcore/runtime/APPLICATION_LOGS/` group — - // the one the dashboard and every per-task log query read. Scoped to that one - // group (CDK `grantWrite` → `logs:CreateLogStream` + `logs:PutLogEvents` on the - // group's ARN, stream wildcard only), so this adds no cross-log-group reach. - // See `applicationLogGroup` for the denial this fixes. + // Platform task logs use a separate namespace from MicroVM service logs. props.applicationLogGroup?.grantWrite(this.executionRole); // Authenticate only this deployment's manifests. Payload and launch reads @@ -1170,51 +703,20 @@ export class LambdaMicrovmCompute extends Construct { // carry the coordinator's authorization instead. grantWorkerBootstrap(this.payloadBucket, this.executionRole); - // Tenant-data access is delegated to the per-task SessionRole, exactly as - // EcsAgentCluster does for the Fargate task role. NOTE the asymmetry with - // that construct: there is no `else` branch granting DynamoDB directly — - // this backend has no legacy deployments to keep working, so a missing - // SessionRole means no tenant-data access rather than broad access. - // - // This is ALSO why nothing below grants DynamoDB: every table the agent - // touches is `task_id`-partitioned and reachable only through the - // SessionRole's `dynamodb:LeadingKeys` condition. `admitComputeRole` wires - // both halves that needs (trust on the SessionRole + `sts:AssumeRole` / - // `sts:TagSession` here), so P2 adds nothing to this seam — asserted by a - // unit test, because a "convenience" direct grant is exactly how per-tenant - // isolation gets lost. + // Tenant access requires the task-scoped role; there is no direct-grant fallback. if (props.agentSessionRole) { props.agentSessionRole.admitComputeRole(this.executionRole); } - // --- P2 runtime parity on the EXECUTION role (ADR-021 "smoke parity") --- - // - // Feature-derived, not copied from `ecs-agent-cluster`: each grant below - // exists because a specific agent code path fails without it on THIS - // substrate. The ECS task role's remaining grants are deliberately absent — - // the UserConcurrencyTable (orchestrator/reconciler-owned; the agent path - // never writes it) and any artifacts-bucket access (delivery writes go - // through the SessionRole's `artifacts/${task_id}/*` statement, so the - // AgentCore runtime role has no direct grant either and neither does this). - + // Runtime startup grants complement task-scoped tenant permissions. // Secrets Manager, part 1: the GitHub PAT, read at startup before the agent // assumes the SessionRole. if (props.githubTokenSecret) { props.githubTokenSecret.grantRead(this.executionRole); } - // Secrets Manager, part 2: per-workspace Linear/Jira OAuth tokens. Same shape - // and same reason as `ecs-agent-cluster`'s grant (ABCA-488): the CLI creates - // `bgagent-linear-oauth-` / `bgagent-jira-oauth-` at setup, so - // the name is unknown at synth and a PREFIX grant is the only expressible - // scope. For a Linear/Jira-channel task the agent resolves that token at - // startup (`config.resolve_linear_api_token` / - // `resolve_jira_oauth_token`) to fire the 👀→✅ reaction and drive the channel - // MCP; without the grant the fetch hits AccessDenied and both silently no-op - // (logged by the token resolver, but invisible to the user in the channel). - // - // `GetSecretValue` ONLY — the agent reads; the orchestrator owns refresh / - // PutSecretValue. + // Workspace OAuth secrets are created after synth. Keep reads prefix-scoped; + // refresh and secret writes belong to control-plane resolvers. this.executionRole.addToPrincipalPolicy(new iam.PolicyStatement({ actions: ['secretsmanager:GetSecretValue'], resources: [ @@ -1233,17 +735,8 @@ export class LambdaMicrovmCompute extends Construct { ], })); - // Bedrock model invocation — scoped to explicit foundation-model and - // cross-Region inference-profile ARNs (parity with the AgentCore runtime and - // the ECS task role), NEVER `Resource: '*'`. The model set comes from the - // shared, context-overridable list (`constructs/bedrock-models.ts`) so no - // backend can drift from the others. - // - // Required on the COMPUTE role even though the SessionRole carries a - // session-tagged Bedrock grant for cost attribution (#215): that attribution - // is designed to FAIL OPEN — Claude Code's credential helper falls back to - // ambient compute-role credentials when the assume-role fails — so without - // this grant the fallback path AccessDenies and the task dies at turn 0. + // Use the shared model allowlist. The compute role supports the credential + // helper fallback when task-tagged model attribution cannot assume its role. const bedrockResources: string[] = []; for (const modelId of resolveBedrockModelIds(this.node)) { bedrockResources.push( @@ -1271,30 +764,26 @@ export class LambdaMicrovmCompute extends Construct { resources: bedrockResources, })); - // AgentCore Memory read+write, so cross-task learning actually persists on - // this substrate. `MEMORY_ID` already reaches the agent inside - // `agent_payload` (unchanged by P2 — it is task data, not platform config), - // which means the agent ATTEMPTS the write regardless; the grant is what - // decides whether it lands or fails closed. + // Memory is a standalone service shared by all compute backends. if (props.agentMemory) { props.agentMemory.grantReadWrite(this.executionRole); } - // A CDK-based target repo's build gate runs `cdk synth`, and a stack wired to - // a concrete env ({account, region}) does a synth-time availability-zone - // context lookup. On a developer box the gitignored cdk.context.json caches - // the answer; the agent clones fresh, so there is no cache and synth fires the - // live lookup. Without this grant the role hits AccessDenied → "Synthesis - // finished with errors" → a FALSE build-gate failure on code that builds fine - // everywhere else (the exact regression the ECS task role hit). Read-only - // describe with no resource-level scoping in IAM, so `Resource: '*'` is - // mandatory (suppressed below); it grants no mutation and no data access. + // Fresh CDK clones can need the read-only availability-zone context lookup. this.executionRole.addToPrincipalPolicy(new iam.PolicyStatement({ actions: ['ec2:DescribeAvailabilityZones'], resources: ['*'], })); // --- Image --- + if (props.managedImageVersion !== undefined + && (!props.baseImageArn || !props.baseImageVersion + || typeof props.managedImageVersion !== 'string' + || Token.isUnresolved(props.managedImageVersion) + || props.managedImageVersion.length > MAX_MANAGED_IMAGE_VERSION_LENGTH + || !/^[1-9]\d*(?:\.\d+)?$/.test(props.managedImageVersion))) { + throw new Error('microvm_managed_image_version requires managed base-image inputs and a literal positive image version'); + } // Branch through the shared predicate's two components rather than an // ad-hoc condition, so this construct and the stack's pre-TaskApi decision // (isLambdaMicrovmImageConfigured) can never disagree. @@ -1335,24 +824,8 @@ export class LambdaMicrovmCompute extends Construct { hooks: { port: AGENT_HOOK_PORT, microvmHooks: { - // Values are the `ENABLED` enum, NOT the hook route: CloudFormation - // rejects a path on every one of these four fields (P2-F2 — see - // {@link HOOK_ENABLED}). The routes are fixed and service-owned; - // {@link MICROVM_AGENT_HOOK_ROUTES} records them for the agent's - // benefit and must never be sent here again. - // - // `/run` is the payload-delivery channel (and, since P2, the - // platform-configuration channel); `/terminate` is the in-guest - // teardown breadcrumb. The agent serves BOTH — enabling a runtime - // hook nothing answers fails the corresponding lifecycle transition, - // so each is enabled only once it is served. - // - // The served P3 hooks use the shared service budget. Declaring them - // enables service callbacks; automatic suspension remains supervisor-gated. - // Note that termination does NOT depend on this hook: - // `TerminateMicrovm` removes the VM with or without in-guest - // cooperation, which is what makes a best-effort `/terminate` safe to - // declare. + // Fixed service routes use enum switches. Automatic sleep remains + // coordinator-gated even though the image serves both lifecycle hooks. run: HOOK_ENABLED, runTimeoutInSeconds: RUN_HOOK_TIMEOUT_SECONDS, terminate: HOOK_ENABLED, @@ -1363,13 +836,8 @@ export class LambdaMicrovmCompute extends Construct { resumeTimeoutInSeconds: LIFECYCLE_HOOK_TIMEOUT_SECONDS, }, microvmImageHooks: { - // `/ready` is MANDATORY whenever any lifecycle hook is enabled — the - // service refuses the create otherwise (see - // READY_HOOK_TIMEOUT_SECONDS), which is why it moved from P2 to P1. - // `/validate` joins it in P2, now that the agent serves a real - // (AWS-call-free — see VALIDATE_HOOK_TIMEOUT_SECONDS) self-check: a - // hook that 404s or reports failure fails every image build, so it - // could not be enabled before there was something behind it. + // Ready is required when runtime hooks are enabled; validate checks + // the local image contract without runtime AWS credentials. ready: HOOK_ENABLED, readyTimeoutInSeconds: READY_HOOK_TIMEOUT_SECONDS, validate: HOOK_ENABLED, @@ -1383,33 +851,15 @@ export class LambdaMicrovmCompute extends Construct { this.imageIdentifier = this.image.attrImageArn; this.imageArn = this.image.attrImageArn; - // Version intentionally unpinned: the service resolves the latest ACTIVE - // version, which is what a redeploy-after-rebuild flow wants. Pinning to + // Without an explicit runtime pin, the service resolves the latest ACTIVE + // version for the redeploy-after-rebuild flow. Pinning automatically to // `attrLatestActiveImageVersion` would be empty on the very first create - // (the build has not finished) and would force a stack update per rebuild. - this.imageVersion = undefined; + // (the build has not finished). An operator can instead select a known + // version without changing the managed image resource itself. + this.imageVersion = props.managedImageVersion; } else if (props.externalImageIdentifier) { this.imageVersion = props.externalImageVersion; - // An operator may pass a bare image NAME (that is what - // `create-microvm-image --name` takes), but a bare name is useless to BOTH - // consumers: it is not an IAM resource, and — refuting this construct's - // former comment — `RunMicrovm` rejects it outright - // (`ValidationException: Malformed ARN - doesn't start with 'arn:'`, live - // 2026-07-31), as does `list-microvm-image-builds`. So the name is resolved - // to its exact ARN ONCE here and that single value feeds both - // `MICROVM_IMAGE_IDENTIFIER` and the lifecycle IAM scope. - // - // The Service Authorization Reference gives the `microvmImage` resource an - // unambiguous shape — - // `arn:${Partition}:lambda:${Region}:${Account}:microvm-image:${MicrovmImageName}` - // — matching the ARN the live run observed, so the ARN is derivable from - // the name plus this stack's partition/Region/account. `formatArn` emits - // the Aws.PARTITION / Aws.REGION / Aws.ACCOUNT_ID pseudo-parameters, so it - // is correct in a region-agnostic app too. - // - // Note the version is NOT part of the resource ARN: the SAR pattern ends - // at the image name, and `RunMicrovm` carries `imageVersion` as a separate - // request field, so pinning a version must never change the IAM scope. + // Launch and IAM require the full image ARN; version is a separate field. this.imageArn = props.externalImageIdentifier.startsWith('arn:') ? props.externalImageIdentifier : stack.formatArn({ @@ -1433,25 +883,17 @@ export class LambdaMicrovmCompute extends Construct { } if (this.imageIdentifier) { - // Emitted on EVERY deploy that configures an image, in both image states. - // Prior image acceptance does not verify a newly configured image or turn - // on automatic suspension. Keep the warning scoped to deployment gates. - // - // The id is deliberately UNCHANGED across P1→P2 (operators grep for it, and a - // rename would read as "the old warning is gone, so it must be fine"). + // Keep operational guidance independent of one installation's image + // numbers and rollout dates. Retain the ID for existing operator filters. Annotations.of(this).addWarningV2( 'abca:microvm-image-p1-smoke-unverified', - 'A MicroVM image is configured. Clean P2 deployment with bootstrap bundle 1.7.0 and ' - + 'coding, iteration and cancellation runs passed on 2026-09-14 without manual IAM changes. ' - + 'The agent serves /ready, /validate, /run, /terminate, /suspend and /resume; managed images declare all six. ' - + 'Image 6.0 has nine successful isolated P3 approval workflows, including expired-credential renewal ' - + 'and repository/network checks after the Connection: close lifecycle fix. This does not verify ' - + 'a different image. P3 checks the actual launched image version; ' - + 'P3 requires bootstrap bundle 1.8.0 and defaults new suspension off; supervisor integration is implemented. ' + 'A MicroVM image is configured. Before enabling automatic suspension, verify the configured ' + + 'image and coordinator together using the P3 acceptance procedure. The coordinator checks the ' + + 'actual launched image version before allowing sleep. The agent serves /ready, /validate, ' + + '/run, /terminate, /suspend and /resume; managed images declare all six. ' + 'Nested deployments require bundle 1.9.0 and a reviewed migration from existing flat stacks. ' - + 'Normal automatic-suspension activation remains open; follow docs/verification/645-p3-implementation-plan.md ' - + 'and docs/verification/645-p3-nested-stack.md. The warning ID is retained ' - + 'across phases for existing operator filters.', + + 'Preserve a compatible coordinator and explicit image version for rollback. Follow ' + + 'docs/verification/645-p3-implementation-plan.md and docs/verification/645-p3-nested-stack.md.', ); } @@ -1474,7 +916,7 @@ export class LambdaMicrovmCompute extends Construct { + `${MICROVM_LOG_GROUP_PREFIX}/* namespace (log stream names are minted per MicroVM, so no ` + 'synth-time ARN exists); worker S3 GetObject is limited to bootstrap/* in its payload bucket, ' + 'with explicit denial outside that prefix and for bucket listing. The build ' - + 'role\'s s3:GetObject is scoped to a single object key, not a wildcard. On the execution ' + + 'role\'s s3:GetObject names the selected artifact and manual-build key. On the execution ' + 'role (ADR-021 P2 runtime parity, mirroring the ECS task role): the second Logs grant is ' + 'CDK grantWrite (CreateLogStream + PutLogEvents only) on the SINGLE platform ' + 'APPLICATION_LOGS group whose name platform_config delivers to the guest, whose ARN ends ' @@ -1510,46 +952,20 @@ export class LambdaMicrovmCompute extends Construct { }, { id: 'AwsSolutions-IAM5', - reason: 'EC2 network-interface APIs are not meaningfully resource-scopable here: ' - + 'CreateNetworkInterface is authorized before the ENI exists, and the Describe* calls ' - + 'take no resource at all. The role is assumable ONLY by lambda.amazonaws.com, holds no ' - + 'data-plane permission, and is used solely to attach the two platform-owned connectors ' - + 'to the platform VPC. An aws:SourceAccount confused-deputy condition is NOT available ' - + 'on this trust: the Lambda MicroVMs service presents no source key when it assumes the ' - + 'role, and adding one makes the connector un-creatable (live-verified, ADR-021 P2-F1 — ' - + 'see ADR-021 section 4 for the per-role compensating controls). The ' - + 'AWS-managed VPC-access policy uses the same wildcard for the same reason.', + reason: 'EC2 Describe actions require wildcard resources. ENI mutations retain the ' + + 'wildcard scope validated with the MicroVM service; that does not establish that ' + + 'narrower mutation permissions are impossible. The role manages the two platform ' + + 'connectors, trusts lambda.amazonaws.com and has no tenant-data grant. Source-conditioned ' + + 'trust failed the recorded live checks; ADR-021 section 4 documents the limitation.', }, ], true); } /** - * CloudWatch Logs writes scoped to the MicroVM log namespace. - * - * `CreateLogStream` + `PutLogEvents` go to both MicroVM-facing roles; the - * `logs:CreateLogGroup` half is **build-role only**, and that asymmetry is - * evidence-based rather than tidiness: - * - * - The service documents `CreateLogGroup` for the BUILD role, and losing build - * logs costs you the one artifact you need when a snapshot build fails — a - * failure mode we have actually hit (ADR-021 P1 4.3: the 443-only SG made the - * image unbuildable, and the root cause was only readable from this group). - * So the build role keeps it. - * - The EXECUTION role does not get it. CloudFormation pre-creates - * `/aws/lambda-microvms/`; build and guest-runtime lines use that - * group. The 2026-09-16 verification enumerated this prefix while an image - * 3.0 worker was RUNNING before and after the query. It found only the - * stack-managed image group. Image 4.0 runtime logs also arrived there. - * The live execution role had only CreateLogStream/PutLogEvents grants - * and no attached managed policies. - * - * See docs/verification/645-p3-callback-live-20260915.md for the inventory and - * policy evidence. This records the tested service behavior, not a guarantee - * about future log-group naming. Investigate a specific CreateLogGroup denial - * before changing runtime permissions. - * - * @param role - the role to grant. - * @param options - `allowCreateLogGroup` gates the build-role-only half. + * Scope log writes to the MicroVM namespace. Only the build role can create log + * groups; runtime logs use the pre-created image group. See + * docs/verification/645-p3-callback-live-20260915.md for the verified policy and + * log inventory. Investigate a specific runtime denial before widening the grant. */ private grantMicrovmLogWrites( role: iam.IRole, @@ -1594,16 +1010,8 @@ export function createMicrovmExecutionRole(scope: Construct, id: string): iam.Ro } /** - * Add `sts:TagSession` alongside the `sts:AssumeRole` CDK's `assumedBy` emits. - * - * The MicroVM service needs BOTH actions (developer guide, "Trust policies"), - * but `iam.Role`'s `assumedBy` only renders `sts:AssumeRole`. Passing a second - * statement through `assumeRolePolicy` keeps the two halves identical — which - * since P2-F1/F3 means "identical and unconditioned": `principal.policyFragment. - * conditions` is now empty, and the pass-through is kept deliberately rather - * than hardcoding `{}` so that if a source-condition key ever becomes usable on - * this path (see the trust-policy block in the constructor), adding it to the - * principal fixes BOTH actions instead of half-closing the hole. + * Add the service-required sts:TagSession action with the same principal and + * conditions as sts:AssumeRole. */ function grantTagSession(role: iam.Role, principal: iam.ServicePrincipal): void { role.assumeRolePolicy?.addStatements(new iam.PolicyStatement({ @@ -1614,11 +1022,7 @@ function grantTagSession(role: iam.Role, principal: iam.ServicePrincipal): void } /** - * Reduce a candidate name to the character set MicroVM image / network - * connector names accept (alphanumerics, `-`, `_`) and cap its length. - * - * Stack names can contain characters these APIs reject, and an unresolved - * (token) stack name would otherwise produce a name containing `${Token[...]}`. + * Restrict image/connector names to the service character set and length limit. */ function sanitizeImageName(candidate: string): string { const MAX_NAME_LENGTH = 64; diff --git a/cdk/src/constructs/lambda-microvm-stack.ts b/cdk/src/constructs/lambda-microvm-stack.ts index 76359069e..a53855632 100644 --- a/cdk/src/constructs/lambda-microvm-stack.ts +++ b/cdk/src/constructs/lambda-microvm-stack.ts @@ -32,6 +32,8 @@ export interface LambdaMicrovmStackProps extends Omit< > { readonly deploymentName: string; readonly executionRole: iam.Role; + /** Distinct image/connector/log names for an overlapping flat-to-nested migration. */ + readonly resourceNamePrefix?: string; } /** Child deployment for MicroVM resources; shared runtime trust stays in the parent. */ @@ -45,6 +47,11 @@ export class LambdaMicrovmStack extends NestedStack { if (Stack.of(props.executionRole) !== this.nestedStackParent) { throw new Error('LambdaMicrovmStack executionRole must be owned by its parent stack'); } + if (props.resourceNamePrefix !== undefined + && (Token.isUnresolved(props.resourceNamePrefix) + || !/^[A-Za-z0-9][A-Za-z0-9-]{0,39}$/.test(props.resourceNamePrefix))) { + throw new Error('microvm_resource_name_prefix must be 1–40 letters, digits or hyphens and start with a letter or digit'); + } // CDK creates bucket-cleanup providers at stack scope, outside Compute. // Those providers are also MicroVM-specific in this child. Tags.of(this).add(MICROVM_BACKEND_TAG_KEY, MICROVM_BACKEND_TAG_VALUE); @@ -60,6 +67,10 @@ export class LambdaMicrovmStack extends NestedStack { }; this.compute = new LambdaMicrovmCompute(this, 'Compute', { ...props, + // Keep bootstrap-authorized IAM role names tied to the parent deployment. + // Only service names change; the old and new image may coexist during a + // reviewed migration without moving the shared execution role. + deploymentName: props.resourceNamePrefix ?? props.deploymentName, buildRoleName: roleName('MicrovmBuildRole'), connectorOperatorRoleName: roleName('MicrovmConnectorRole'), }); diff --git a/cdk/src/constructs/linear-identity-vault.ts b/cdk/src/constructs/linear-identity-vault.ts index a176945fe..765f837cb 100644 --- a/cdk/src/constructs/linear-identity-vault.ts +++ b/cdk/src/constructs/linear-identity-vault.ts @@ -70,7 +70,7 @@ export interface LinearIdentityVaultProps { /** * The Linear identity vault's workload identity. Grant helpers wire the token * data-plane permissions onto whichever principal resolves Linear tokens - * (webhook processor, orchestrator, agent session role). + * (webhook processor, orchestrator, compute execution role). */ export class LinearIdentityVault extends Construct { /** The provisioned workload identity name (stable natural id). */ diff --git a/cdk/src/constructs/linear-integration.ts b/cdk/src/constructs/linear-integration.ts index f535365c0..5c5d73a74 100644 --- a/cdk/src/constructs/linear-integration.ts +++ b/cdk/src/constructs/linear-integration.ts @@ -152,7 +152,7 @@ export interface LinearIntegrationProps { * provider name; Phase 2.0b OAuth migration). Webhook processor and * orchestrator use this to look up which credential provider holds the * workspace's OAuth token. - * - LinearWebhookDedupTable (60s TTL dedup for webhook retries) + * - LinearWebhookDedupTable (8-hour TTL dedup for webhook retries) * - Lambda handlers for the webhook receiver, async processor, and account linking * - API Gateway routes under /linear/* * - Two Secrets Manager secrets (webhook signing secret + personal API token) @@ -171,7 +171,7 @@ export class LinearIntegration extends Construct { */ public readonly workspaceRegistryTable: dynamodb.Table; - /** Webhook dedup table — (issue_id, action) keys with 60s TTL. */ + /** Webhook dedup table — data.id/action/webhookTimestamp keys with 8-hour TTL. */ public readonly webhookDedupTable: dynamodb.Table; /** Linear webhook signing secret (placeholder — populated by `bgagent linear setup`). */ @@ -199,8 +199,7 @@ export class LinearIntegration extends Construct { this.userMappingTable = userMapping.table; this.workspaceRegistryTable = workspaceRegistry.table; - // Dedup table: linear webhook retries collapse to a single processor invoke - // within the 60s TTL window. Keyed on `{issue_id}#{action}`. + // The receiver deduplicates data.id/action/webhookTimestamp for 8 hours. this.webhookDedupTable = new dynamodb.Table(this, 'WebhookDedupTable', { partitionKey: { name: 'dedup_key', type: dynamodb.AttributeType.STRING }, billingMode: dynamodb.BillingMode.PAY_PER_REQUEST, diff --git a/cdk/src/constructs/microvm-continuation-manager.ts b/cdk/src/constructs/microvm-continuation-manager.ts new file mode 100644 index 000000000..62594c7e6 --- /dev/null +++ b/cdk/src/constructs/microvm-continuation-manager.ts @@ -0,0 +1,99 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +const RECONCILER_TIMEOUT_MINUTES = 5; +const RECONCILER_SCHEDULE_MINUTES = 5; +import * as path from 'path'; +import { Duration } from 'aws-cdk-lib'; +import type * as dynamodb from 'aws-cdk-lib/aws-dynamodb'; +import * as events from 'aws-cdk-lib/aws-events'; +import * as targets from 'aws-cdk-lib/aws-events-targets'; +import * as iam from 'aws-cdk-lib/aws-iam'; +import { Architecture, Runtime } from 'aws-cdk-lib/aws-lambda'; +import * as lambda from 'aws-cdk-lib/aws-lambda-nodejs'; +import { NagSuppressions } from 'cdk-nag'; +import { Construct } from 'constructs'; +import type { ContinuationBucket } from './continuation-bucket'; + +export interface MicrovmContinuationManagerProps { + readonly taskTable: dynamodb.ITable; + readonly approvalsTable: dynamodb.ITable; + readonly userConcurrencyTable: dynamodb.ITable; + readonly continuationBucket: ContinuationBucket; + /** Unqualified function ARN; saved tasks select their original published version. */ + readonly orchestratorFunctionArn: string; + readonly imageArn: string; + readonly maxConcurrentTasksPerUser?: number; +} + +/** Recover lost continuation signals and clean saved objects after confirmed shutdown. */ +export class MicrovmContinuationManager extends Construct { + public readonly fn: lambda.NodejsFunction; + + constructor(scope: Construct, id: string, props: MicrovmContinuationManagerProps) { + super(scope, id); + this.fn = new lambda.NodejsFunction(this, 'ReconcilerFn', { + entry: path.join(__dirname, '..', 'handlers', 'reconcile-microvm-continuations.ts'), + handler: 'handler', + runtime: Runtime.NODEJS_24_X, + architecture: Architecture.ARM_64, + timeout: Duration.minutes(RECONCILER_TIMEOUT_MINUTES), + memorySize: 256, + environment: { + ABCA_COMPONENT: 'orchestr', + TASK_TABLE_NAME: props.taskTable.tableName, + TASK_APPROVALS_TABLE_NAME: props.approvalsTable.tableName, + USER_CONCURRENCY_TABLE_NAME: props.userConcurrencyTable.tableName, + CONTINUATION_BUCKET_NAME: props.continuationBucket.bucket.bucketName, + ORCHESTRATOR_FUNCTION_ARN: props.orchestratorFunctionArn, + MAX_CONCURRENT_TASKS_PER_USER: String(props.maxConcurrentTasksPerUser ?? 10), + }, + // Bundle the pinned Lambda serializer: DurableExecutionName is required + // for deduplication and may be absent from the runtime's older SDK. + bundling: { + externalModules: ['@aws-sdk/client-dynamodb', '@aws-sdk/lib-dynamodb'], + // Shared supervisor imports reach attachment screening; pdf-parse needs + // its packaged worker assets, just as in the main orchestrator bundle. + nodeModules: ['pdf-parse'], + }, + }); + props.taskTable.grantReadWriteData(this.fn); + props.approvalsTable.grantReadWriteData(this.fn); + props.userConcurrencyTable.grantReadWriteData(this.fn); + props.continuationBucket.grantCoordinator(this.fn); + this.fn.addToRolePolicy(new iam.PolicyStatement({ + actions: ['lambda:GetMicrovm', 'lambda:TerminateMicrovm'], + resources: [props.imageArn, `${props.imageArn}:*`], + })); + this.fn.addToRolePolicy(new iam.PolicyStatement({ + actions: ['lambda:InvokeFunction'], resources: [`${props.orchestratorFunctionArn}:*`], + })); + new events.Rule(this, 'Schedule', { + schedule: events.Schedule.rate(Duration.minutes(RECONCILER_SCHEDULE_MINUTES)), + targets: [new targets.LambdaFunction(this.fn)], + }); + NagSuppressions.addResourceSuppressions(this.fn, [ + { id: 'AwsSolutions-IAM4', reason: 'AWSLambdaBasicExecutionRole supplies CloudWatch runtime logs.' }, + { + id: 'AwsSolutions-IAM5', + reason: 'DynamoDB index/* grants accompany the task tables; S3 is restricted to the continuation prefix; MicroVM state/termination uses one image and its versions; invoke uses only retained versions of the one coordinator.', + }, + ], true); + } +} diff --git a/cdk/src/constructs/stranded-task-reconciler.ts b/cdk/src/constructs/stranded-task-reconciler.ts index ac6bbc70b..1cfd54ef8 100644 --- a/cdk/src/constructs/stranded-task-reconciler.ts +++ b/cdk/src/constructs/stranded-task-reconciler.ts @@ -30,8 +30,8 @@ import { Construct } from 'constructs'; /** Default stranded-timeout (seconds; 20 minutes). */ const DEFAULT_STRANDED_TIMEOUT_SECONDS = 1200; -/** Default approval-stranded timeout (seconds; 2 hours). */ -const DEFAULT_APPROVAL_STRANDED_TIMEOUT_SECONDS = 7200; +/** Backstop for approval waits without a checkpoint (eight hours plus 30 minutes). */ +const DEFAULT_APPROVAL_STRANDED_TIMEOUT_SECONDS = 30600; /** Default task-record retention used for event TTL (days). */ const DEFAULT_TASK_RETENTION_DAYS = 90; @@ -54,6 +54,8 @@ export interface StrandedTaskReconcilerProps { /** TaskEventsTable (handler writes task_stranded + task_failed events). */ readonly taskEventsTable: dynamodb.ITable; + /** Close unanswered requests when the owning task is declared stranded. */ + readonly taskApprovalsTable?: dynamodb.ITable; /** UserConcurrencyTable (handler atomically releases held task reservations). */ readonly userConcurrencyTable: dynamodb.ITable; @@ -77,13 +79,12 @@ export interface StrandedTaskReconcilerProps { readonly strandedTimeoutSeconds?: number; /** - * Cedar HITL approval-stranded timeout (seconds). Tasks in - * AWAITING_APPROVAL older than this are transitioned to FAILED. - * Longer than the stranded-timeout because approvals legitimately - * sit for up to an hour (§7.3). Set via + * Backstop for AWAITING_APPROVAL tasks without a saved continuation. + * Allows the worker's full service lifetime and coordinator cleanup. + * Saved MicroVM requests are handled by the continuation manager. Set via * ``APPROVAL_STRANDED_TIMEOUT_SECONDS``. * - * @default 7200 (2 hours — double §7.3's 1-hour ceiling + an hour grace) + * @default 30600 (8.5 hours) */ readonly approvalStrandedTimeoutSeconds?: number; @@ -126,6 +127,7 @@ export class StrandedTaskReconciler extends Construct { // Solution-attribution component label (#319): orchestration plane. ABCA_COMPONENT: 'orchestr', TASK_TABLE_NAME: props.taskTable.tableName, + ...(props.taskApprovalsTable && { TASK_APPROVALS_TABLE_NAME: props.taskApprovalsTable.tableName }), TASK_EVENTS_TABLE_NAME: props.taskEventsTable.tableName, USER_CONCURRENCY_TABLE_NAME: props.userConcurrencyTable.tableName, STRANDED_TIMEOUT_SECONDS: String(strandedTimeout), @@ -140,6 +142,7 @@ export class StrandedTaskReconciler extends Construct { // TaskTable: read (query by StatusIndex) + conditional UpdateItem to // transition stranded rows to FAILED. props.taskTable.grantReadWriteData(this.fn); + props.taskApprovalsTable?.grantReadWriteData(this.fn); // TaskEvents: write task_stranded + task_failed events. props.taskEventsTable.grantWriteData(this.fn); // Concurrency: read/release a task-owned reservation transactionally. diff --git a/cdk/src/constructs/task-api.ts b/cdk/src/constructs/task-api.ts index 8c538bdaf..5a47f3aa4 100644 --- a/cdk/src/constructs/task-api.ts +++ b/cdk/src/constructs/task-api.ts @@ -259,6 +259,8 @@ export interface TaskApiProps { * - DELETE /api-keys/{key_id} → deleteApiKey (Cognito) */ export class TaskApi extends Construct { + private readonly approvalDecisionFunctions: lambda.NodejsFunction[] = []; + /** * The API Gateway REST API. */ @@ -998,6 +1000,10 @@ export class TaskApi extends Construct { ...commonEnv, TASK_APPROVALS_TABLE_NAME: props.taskApprovalsTable.tableName, }; + const decisionBundling = { + ...commonBundling, + externalModules: commonBundling.externalModules?.filter(name => name !== '@aws-sdk/client-lambda'), + }; // ApproveTaskFn — POST /tasks/{task_id}/approve const approveTaskFn = new lambda.NodejsFunction(this, 'ApproveTaskFn', { @@ -1006,7 +1012,7 @@ export class TaskApi extends Construct { runtime: Runtime.NODEJS_24_X, architecture: Architecture.ARM_64, environment: approvalEnv, - bundling: commonBundling, + bundling: decisionBundling, timeout: Duration.seconds(API_HANDLER_TIMEOUT_SECONDS), memorySize: API_HANDLER_MEMORY_MB, }); @@ -1021,13 +1027,14 @@ export class TaskApi extends Construct { runtime: Runtime.NODEJS_24_X, architecture: Architecture.ARM_64, environment: approvalEnv, - bundling: commonBundling, + bundling: decisionBundling, timeout: Duration.seconds(API_HANDLER_TIMEOUT_SECONDS), memorySize: API_HANDLER_MEMORY_MB, }); props.taskTable.grantReadWriteData(denyTaskFn); props.taskApprovalsTable.grantReadWriteData(denyTaskFn); props.taskEventsTable.grantReadWriteData(denyTaskFn); + this.approvalDecisionFunctions.push(approveTaskFn, denyTaskFn); if (props.lambdaMicrovmImageArn) { for (const decisionFn of [approveTaskFn, denyTaskFn]) { decisionFn.addToRolePolicy(new iam.PolicyStatement({ @@ -1448,4 +1455,18 @@ export class TaskApi extends Construct { }, ], true); } + + /** Wire after the coordinator and bucket exist, avoiding a construction-order cycle. */ + public enableMicrovmContinuations( + bucketName: string, coordinatorArn: string, concurrencyTable: dynamodb.ITable, + ): void { + for (const fn of this.approvalDecisionFunctions) { + fn.addEnvironment('CONTINUATION_BUCKET_NAME', bucketName); + fn.addEnvironment('ORCHESTRATOR_FUNCTION_ARN', coordinatorArn); + concurrencyTable.grantReadWriteData(fn); + fn.addToRolePolicy(new iam.PolicyStatement({ + actions: ['lambda:InvokeFunction'], resources: [`${coordinatorArn}:*`], + })); + } + } } diff --git a/cdk/src/constructs/task-approvals-table.ts b/cdk/src/constructs/task-approvals-table.ts index 20bb4e6ef..68d4683d3 100644 --- a/cdk/src/constructs/task-approvals-table.ts +++ b/cdk/src/constructs/task-approvals-table.ts @@ -75,8 +75,9 @@ export interface TaskApprovalsTableProps { * streams wired into the fan-out Lambda. Enabling streams here would * create duplicate fan-out paths. * - * TTL is sized by the agent as `created_at_epoch + timeout_s + 120s` - * so rows never expire during the decision window (§10.1). + * Pending rows have no TTL, including requests with an explicit deadline. + * Task closure applies retention cleanup; capacity delays must never erase + * the recorded decision before a replacement worker consumes it. */ export class TaskApprovalsTable extends Construct { /** diff --git a/cdk/src/constructs/task-orchestrator.ts b/cdk/src/constructs/task-orchestrator.ts index 5a7eb5db9..c620214e1 100644 --- a/cdk/src/constructs/task-orchestrator.ts +++ b/cdk/src/constructs/task-orchestrator.ts @@ -52,12 +52,17 @@ const ORCHESTRATOR_TIMEOUT_SECONDS = 60; /** Orchestrator Lambda memory (MB). */ const ORCHESTRATOR_MEMORY_MB = 1024; +import type { ContinuationBucket } from './continuation-bucket'; import { grantCoordinatorPayloads } from './payload-bootstrap-permissions'; /** * Properties for TaskOrchestrator construct. */ export interface TaskOrchestratorProps { + /** Close outstanding requests and apply retention after task completion. */ + readonly taskApprovalsTable?: dynamodb.ITable; + /** Versioned durable checkpoints and launch inputs for MicroVM worker replacement. */ + readonly continuationBucket?: ContinuationBucket; /** * The DynamoDB task table. */ @@ -224,8 +229,8 @@ export interface TaskOrchestratorProps { * only ever fire for a hand-edited Lambda environment — never because a * deploy-time gate and a per-repo `compute_type` disagreed. * - * Optional as a prop only so isolated construct tests can omit it. Four of the - * thirteen `platform_config` keys come from env vars the orchestrator already + * Optional as a prop only so isolated construct tests can omit it. Some + * `platform_config` keys come from env vars the orchestrator already * carries for its own work (`TASK_TABLE_NAME`, `TASK_EVENTS_TABLE_NAME`, * `GITHUB_TOKEN_SECRET_ARN`) or from the stack-wide `SolutionUaAspect` * (`AWS_SDK_UA_APP_ID`), so they are deliberately NOT repeated here. @@ -444,6 +449,7 @@ export class TaskOrchestrator extends Construct { // Solution-attribution component label (#319): orchestration plane. ABCA_COMPONENT: 'orchestr', TASK_TABLE_NAME: props.taskTable.tableName, + ...(props.taskApprovalsTable && { TASK_APPROVALS_TABLE_NAME: props.taskApprovalsTable.tableName }), TASK_EVENTS_TABLE_NAME: props.taskEventsTable.tableName, USER_CONCURRENCY_TABLE_NAME: props.userConcurrencyTable.tableName, RUNTIME_ARN: props.runtimeArn, @@ -488,6 +494,9 @@ export class TaskOrchestrator extends Construct { // unconditional; there is no "no ingress configured" state to express. MICROVM_INGRESS_CONNECTOR_ARNS: props.microvmConfig.ingressConnectorArns.join(','), MICROVM_PAYLOAD_BUCKET: props.microvmConfig.payloadBucket.bucketName, + ...(props.continuationBucket && { + CONTINUATION_BUCKET_NAME: props.continuationBucket.bucket.bucketName, + }), TASK_APPROVALS_TABLE_NAME: props.microvmConfig.approvalsTable.tableName, MICROVM_APPROVAL_SUSPEND_ENABLED: String(props.microvmConfig.approvalSuspendEnabled ?? false), MICROVM_APPROVAL_SUSPEND_PARAMETER_NAME: suspendParameter!.parameterName, @@ -518,6 +527,7 @@ export class TaskOrchestrator extends Construct { // DynamoDB grants props.taskTable.grantReadWriteData(this.fn); + props.taskApprovalsTable?.grantReadWriteData(this.fn); props.taskEventsTable.grantReadWriteData(this.fn); props.userConcurrencyTable.grantReadWriteData(this.fn); if (props.repoTable) { @@ -533,6 +543,7 @@ export class TaskOrchestrator extends Construct { // launch references for replay. Workers cannot read the launch records. if (props.ecsPayloadBucket) grantCoordinatorPayloads(props.ecsPayloadBucket, this.fn); if (props.microvmConfig) grantCoordinatorPayloads(props.microvmConfig.payloadBucket, this.fn); + props.continuationBucket?.grantCoordinator(this.fn); // Durable execution managed policy this.fn.role!.addManagedPolicy( @@ -686,14 +697,8 @@ export class TaskOrchestrator extends Construct { resources: [suspendParameter!.parameterArn], })); - // `lambda:PassNetworkConnector` supports NO resource-level permissions - // (the Service Authorization Reference lists no resource type for it), so - // `Resource: '*'` is mandatory — a narrowed ARN would simply never match - // and RunMicrovm would fail with AccessDenied. It is also why the ADR - // notes the action is needed "even for the default connectors": the - // AWS-managed connectors live in the `aws` account, outside any ARN we - // could enumerate. The action only permits *passing* a connector to a - // service, not creating or reading one. + // PassNetworkConnector has no resource-level authorization support. Its + // wildcard permits attaching connectors, not creating or inspecting them. this.fn.addToRolePolicy(new iam.PolicyStatement({ sid: 'MicrovmPassNetworkConnector', actions: ['lambda:PassNetworkConnector'], diff --git a/cdk/src/constructs/tool-gateway.ts b/cdk/src/constructs/tool-gateway.ts index 0ffa87beb..db83ead1f 100644 --- a/cdk/src/constructs/tool-gateway.ts +++ b/cdk/src/constructs/tool-gateway.ts @@ -18,7 +18,7 @@ */ import * as path from 'path'; -import { Duration } from 'aws-cdk-lib'; +import { CfnResource, Duration, Stack } from 'aws-cdk-lib'; import * as agentcore from 'aws-cdk-lib/aws-bedrockagentcore'; import * as dynamodb from 'aws-cdk-lib/aws-dynamodb'; import type { IGrantable } from 'aws-cdk-lib/aws-iam'; @@ -157,6 +157,13 @@ export class ToolGateway extends Construct { reason: 'DynamoDB index/* ARN wildcard generated by CDK grantReadData on the RepoTable', }, ], true); + + const targetId = Stack.of(this).getLogicalId(this.repoConfigFn.node.defaultChild as CfnResource); + NagSuppressions.addResourceSuppressions(this.gateway, [{ + id: 'AwsSolutions-IAM5', + reason: 'CDK grants the gateway invocation of its one target Lambda, including that function\'s versions and aliases.', + appliesTo: [`Resource::<${targetId}.Arn>:*`], + }], true); } /** diff --git a/cdk/src/handlers/approve-task.ts b/cdk/src/handlers/approve-task.ts index 981ae09c8..0d3bdd860 100644 --- a/cdk/src/handlers/approve-task.ts +++ b/cdk/src/handlers/approve-task.ts @@ -175,7 +175,8 @@ export async function handler( UpdateExpression: 'SET #status = :approved, decided_at = :now, #scope = :scope', ConditionExpression: - 'attribute_exists(request_id) AND #status = :pending AND user_id = :caller', + 'attribute_exists(request_id) AND #status = :pending AND user_id = :caller ' + + 'AND (attribute_not_exists(deadline_epoch) OR deadline_epoch > :epoch)', ExpressionAttributeNames: { '#status': 'status', '#scope': 'scope', @@ -186,6 +187,7 @@ export async function handler( ':now': nowIso, ':scope': scope, ':caller': callerUserId, + ':epoch': nowEpoch, }, }, }, diff --git a/cdk/src/handlers/deny-task.ts b/cdk/src/handlers/deny-task.ts index 27ac48ea1..eed55e5df 100644 --- a/cdk/src/handlers/deny-task.ts +++ b/cdk/src/handlers/deny-task.ts @@ -156,7 +156,8 @@ export async function handler( UpdateExpression: 'SET #status = :denied, decided_at = :now, deny_reason = :reason', ConditionExpression: - 'attribute_exists(request_id) AND #status = :pending AND user_id = :caller', + 'attribute_exists(request_id) AND #status = :pending AND user_id = :caller ' + + 'AND (attribute_not_exists(deadline_epoch) OR deadline_epoch > :epoch)', ExpressionAttributeNames: { '#status': 'status' }, ExpressionAttributeValues: { ':denied': 'DENIED', @@ -164,6 +165,7 @@ export async function handler( ':now': nowIso, ':reason': sanitizedReason, ':caller': callerUserId, + ':epoch': nowEpoch, }, }, }, diff --git a/cdk/src/handlers/get-pending.ts b/cdk/src/handlers/get-pending.ts index 08455ae58..909c235f5 100644 --- a/cdk/src/handlers/get-pending.ts +++ b/cdk/src/handlers/get-pending.ts @@ -224,7 +224,8 @@ function coerceStringList(value: unknown): readonly string[] { return value.filter((v): v is string => typeof v === 'string'); } -function computeExpiresAt(createdAt: string, timeoutS: number): string { +function computeExpiresAt(createdAt: string, timeoutS: number): string | null { + if (timeoutS === 0) return null; if (!createdAt || !Number.isFinite(timeoutS) || timeoutS <= 0) { return createdAt; } diff --git a/cdk/src/handlers/linear-webhook-processor.ts b/cdk/src/handlers/linear-webhook-processor.ts index 66a770acb..7e159bc70 100644 --- a/cdk/src/handlers/linear-webhook-processor.ts +++ b/cdk/src/handlers/linear-webhook-processor.ts @@ -478,24 +478,6 @@ function patchChildOwnAttachments( }; } -/** - * Post a Linear comment + ❌ reaction without ever propagating an error. - * - * Phase 2.0b-O2: feedback is workspace-scoped — the resolver looks up - * the per-workspace OAuth token via `LinearWorkspaceRegistryTable` and - * issues a Bearer token. If the workspace isn't registered (drop-on-the-floor - * for unmapped orgs) the feedback path no-ops cleanly. - * - * Two failure modes handled here: - * - `LINEAR_WORKSPACE_REGISTRY_TABLE_NAME` env var unset (deploy misconfig) — - * skip with a clear diagnostic instead of letting the resolver fail - * per-call. - * - `reportIssueFailure` throws synchronously (today impossible thanks to the - * helper's internal `Promise.allSettled`, but a future refactor could - * break that contract). Catching here means a synchronous throw can't - * bubble up and fail the Lambda — which would trigger SQS retries on a - * poison message. - */ /** * Iteration-UX: post the IMMEDIATE threaded "👀 On it" reply under the trigger * comment, synchronously at trigger time. This is what kills the multi-minute @@ -533,6 +515,7 @@ async function postIterationAck( } } +/** Report workspace-scoped failure feedback best-effort; skip missing OAuth routing. */ async function safeReportIssueFailure( issueId: string, linearWorkspaceId: string | undefined, diff --git a/cdk/src/handlers/linear-webhook.ts b/cdk/src/handlers/linear-webhook.ts index 0cba78c70..a9878edf1 100644 --- a/cdk/src/handlers/linear-webhook.ts +++ b/cdk/src/handlers/linear-webhook.ts @@ -74,8 +74,8 @@ interface LinearWebhookEnvelope { * * Verifies the `Linear-Signature` HMAC over the raw body, rejects stale * `webhookTimestamp` values (replay protection), dedups on - * `(issue_id, action)` with a 60s TTL, and async-invokes the processor - * Lambda so we can ack within Linear's 5s timeout. + * `(data.id, action, webhookTimestamp)` with an eight-hour TTL, and invokes + * the processor asynchronously to acknowledge the delivery promptly. */ export async function handler(event: APIGatewayProxyEvent): Promise { try { diff --git a/cdk/src/handlers/orchestrate-task.ts b/cdk/src/handlers/orchestrate-task.ts index c73dfb9d0..2d8ed913f 100644 --- a/cdk/src/handlers/orchestrate-task.ts +++ b/cdk/src/handlers/orchestrate-task.ts @@ -18,13 +18,17 @@ */ import { withDurableExecution, type DurableExecutionHandler } from '@aws/durable-execution-sdk-js'; +import { CONTINUATION_RETRY_POLL_SECONDS, CONTINUATION_TRANSITION_POLL_SECONDS } from './shared/microvm-continuation-timing'; import { TaskStatus, TERMINAL_STATUSES } from '../constructs/task-status'; import { resolveComputeStrategy } from './shared/compute-strategy'; -import { formatMicrovmTerminalFailure, MicrovmStartUncertainError } from './shared/error-classifier'; +import { MicrovmStartUncertainError } from './shared/error-classifier'; import { reportIssueFailure as reportJiraIssueFailure } from './shared/jira-feedback'; import { reportIssueFailure } from './shared/linear-feedback'; import { logger } from './shared/logger'; -import { stopMicrovmWithDiagnostics, superviseMicrovm } from './shared/microvm-supervisor'; +import { runMicrovmContinuation } from './shared/microvm-continuation-runner'; +import { saveContinuationLaunch } from './shared/microvm-continuation-storage'; +import { stopMicrovmWithDiagnostics } from './shared/microvm-supervisor'; +import { pollMicrovmTask } from './shared/microvm-task-poll'; import { admissionControl, buildComputeMetadata, @@ -58,6 +62,8 @@ interface OrchestrateTaskEvent { * durable-execution idempotency) can mistake it for a replay. */ readonly queue_pickup_id?: string; + readonly continuation_request_id?: string; + readonly continuation_attempt_id?: string; } const MAX_POLL_ATTEMPTS = 1020; // ~8.5h at 30s intervals @@ -68,6 +74,13 @@ const MAX_CONSECUTIVE_ECS_COMPLETED_POLLS = 5; const DEFAULT_POLL_INTERVAL_SECONDS = 30; const durableHandler: DurableExecutionHandler = async (event, context) => { + if (event.continuation_request_id !== undefined || event.continuation_attempt_id !== undefined) { + return runMicrovmContinuation({ + task_id: event.task_id, + continuation_request_id: event.continuation_request_id ?? '', + continuation_attempt_id: event.continuation_attempt_id ?? '', + }, context); + } const { task_id: taskId } = event; // Step 1: Load task record @@ -194,6 +207,9 @@ const durableHandler: DurableExecutionHandler = asyn let strategy: ReturnType | undefined; let startedHandle: Awaited>['handle'] | undefined; try { + if (blueprintConfig.compute_type === 'lambda-microvm') { + await saveContinuationLaunch(taskId, task.user_id, payload, blueprintConfig); + } strategy = resolveComputeStrategy(blueprintConfig); const startInput = { taskId, @@ -392,29 +408,15 @@ const durableHandler: DurableExecutionHandler = asyn 'await-agent-completion', async (state) => { if (microvmStrategy && sessionHandle.strategyType === 'lambda-microvm') { - const supervised = await superviseMicrovm({ + return pollMicrovmTask({ taskId, userId: task.user_id, handle: sessionHandle, strategy: microvmStrategy, - previous: state.microvmSupervisor, pollIntervalMs: blueprintConfig.poll_interval_ms ?? DEFAULT_POLL_INTERVAL_SECONDS * 1000, suspendEnabled: process.env.MICROVM_APPROVAL_SUSPEND_ENABLED === 'true', emitEvent: (type, metadata, options) => emitTaskEvent(taskId, type, metadata, correlation, options), - }); - const failure = supervised.kind === 'failure' ? supervised.reason - : supervised.kind === 'substrate-terminal' ? 'substrate-terminal' : undefined; - return { - attempts: state.attempts + 1, - lastStatus: supervised.snapshot?.status ?? state.lastStatus, - sessionUnhealthy: supervised.heartbeatUnhealthy, - microvmSupervisor: supervised.state, - microvmFailureReason: failure, - microvmFailureMessage: supervised.kind === 'substrate-terminal' ? formatMicrovmTerminalFailure( - `substrate state ${supervised.substrate!.status}`, supervised.substrate!.reason, - ) : undefined, - microvmOwnershipLost: supervised.kind === 'ownership-lost', - }; + }, state); } const ddbState = await pollTaskStatus(taskId, state, blueprintConfig.compute_type); let consecutiveEcsPollFailures = 0; @@ -472,6 +474,13 @@ const durableHandler: DurableExecutionHandler = asyn { initialState: { attempts: 0 }, waitStrategy: (state: PollState) => { + if (state.microvmParked || state.microvmOwnershipLost) return { shouldContinue: false }; + if (state.microvmRetiring) { + return { + shouldContinue: true, + delay: { seconds: state.microvmRetirementError ? CONTINUATION_RETRY_POLL_SECONDS : CONTINUATION_TRANSITION_POLL_SECONDS }, + }; + } if (state.lastStatus && TERMINAL_STATUSES.includes(state.lastStatus)) { return { shouldContinue: false }; } @@ -504,6 +513,14 @@ const durableHandler: DurableExecutionHandler = asyn // Step 6: Finalize — update terminal status, emit events, release concurrency await context.step('finalize', async () => { + if (finalPollState.microvmParked) { + await emitTaskEvent(taskId, 'continuation_parked', { + microvm_id: sessionHandle.sessionId, + detail: 'Your approval request is still available. The saved task will continue on another worker after your answer.', + }, correlation); + await deleteMicrovmPayload(taskId); + return; + } let finalized = false; try { if (!finalPollState.microvmOwnershipLost) { diff --git a/cdk/src/handlers/reconcile-concurrency.ts b/cdk/src/handlers/reconcile-concurrency.ts index 4e0828aee..a8c97daef 100644 --- a/cdk/src/handlers/reconcile-concurrency.ts +++ b/cdk/src/handlers/reconcile-concurrency.ts @@ -18,9 +18,10 @@ */ import { randomUUID } from 'node:crypto'; -import { ScanCommand, type ScanCommandOutput, UpdateCommand } from '@aws-sdk/lib-dynamodb'; +import { GetCommand, ScanCommand, type ScanCommandOutput, UpdateCommand } from '@aws-sdk/lib-dynamodb'; import { ACTIVE_STATUSES, TERMINAL_STATUSES } from '../constructs/task-status'; import { logger } from './shared/logger'; +import { workerLeaseKey } from './shared/microvm-continuation-types'; import { releaseTaskSlot, type ReservationTask } from './shared/task-concurrency'; import { makeDocClient } from './shared/ua'; @@ -68,7 +69,7 @@ export async function handler(): Promise { const page: ScanCommandOutput = await ddb.send(new ScanCommand({ TableName: TASK_TABLE, ConsistentRead: true, - ProjectionExpression: 'task_id, user_id, #status, concurrency_slot', + ProjectionExpression: 'task_id, user_id, #status, concurrency_slot, continuation, microvm_start, session_id, repo', ExpressionAttributeNames: { '#status': 'status' }, ExclusiveStartKey: lastKey, })); @@ -78,13 +79,24 @@ export async function handler(): Promise { const active = ACTIVE_STATUSES.some(status => status === task.status); if (!task.concurrency_slot && !active) continue; const owned = reservations.get(task.user_id) ?? { held: 0, ambiguous: false, terminal: [] }; + let parked = false; + if (task.status === 'AWAITING_APPROVAL' && task.concurrency_slot?.state === 'released' + && row.continuation?.state === 'PARKED') { + const saved = await ddb.send(new GetCommand({ + TableName: TASK_TABLE, Key: workerLeaseKey(task.task_id), ConsistentRead: true, + })); + parked = saved.Item?.lease_state === 'PARKED' + && saved.Item.lease_attempt_id === row.microvm_start?.clientToken + && saved.Item.lease_user_id === task.user_id && saved.Item.lease_repo === (row.repo ?? '') + && saved.Item.lease_microvm_id === row.session_id; + } if (task.concurrency_slot?.state === 'held') { // Terminal tasks still own their seat until release commits. Repair // that total first, then release terminal reservations through the // same transaction used by the orchestrator and stranded-task cleaner. owned.held++; if (TERMINAL_STATUSES.some(status => status === task.status)) owned.terminal.push(task.task_id); - } else if (active || (task.concurrency_slot && task.concurrency_slot.state !== 'released')) { + } else if ((active && !parked) || (task.concurrency_slot && task.concurrency_slot.state !== 'released')) { // Older active tasks and not-yet-admitted SUBMITTED tasks cannot be // distinguished by status. Wait for them to settle; never guess a count. owned.ambiguous = true; diff --git a/cdk/src/handlers/reconcile-microvm-continuations.ts b/cdk/src/handlers/reconcile-microvm-continuations.ts new file mode 100644 index 000000000..def3d5022 --- /dev/null +++ b/cdk/src/handlers/reconcile-microvm-continuations.ts @@ -0,0 +1,167 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import { GetCommand, ScanCommand, UpdateCommand, type ScanCommandOutput } from '@aws-sdk/lib-dynamodb'; +import type { Context } from 'aws-lambda'; +import { TaskStatus, TERMINAL_STATUSES } from '../constructs/task-status'; +import { closeTaskApprovals } from './shared/close-task-approvals'; +import { logger } from './shared/logger'; +import { dispatchMicrovmContinuation } from './shared/microvm-continuation-dispatch'; +import { retireCheckpointedMicrovm, type ContinuableTask } from './shared/microvm-continuation-retirement'; +import { deleteClosedTaskContinuations } from './shared/microvm-continuation-storage'; +import { CONTINUATION_IO_TIMEOUT_MS, CONTINUATION_RETIREMENT_TIMEOUT_MS } from './shared/microvm-continuation-timing'; +import { workerLeaseKey, type MicrovmHandle } from './shared/microvm-continuation-types'; +import { microvmErrorIdentity } from './shared/microvm-control'; +import { LambdaMicrovmComputeStrategy, MICROVM_MAX_DURATION_SECONDS } from './shared/strategies/lambda-microvm-strategy'; +import { releaseTaskSlot } from './shared/task-concurrency'; +import { makeDocClient } from './shared/ua'; + +const ddb = makeDocClient(); +const strategy = new LambdaMicrovmComputeStrategy(); +const TABLE = process.env.TASK_TABLE_NAME!; +const CURSOR_KEY = { task_id: 'continuation-manager#cursor' }; +const CURSOR_WRITE_TIMEOUT_MS = 5000; +const MIN_REMAINING_MS = 45_000; +const BATCH_SIZE = 4; + +/** Every mutation below rechecks the current owner/attempt; scan rows are only hints. */ +export async function reconcileMicrovmContinuation(task: ContinuableTask): Promise { + const options = { abortSignal: AbortSignal.timeout(CONTINUATION_RETIREMENT_TIMEOUT_MS) }; + // The scan can predate a replacement or cancellation. Resolve physical identity + // again before issuing a stop against a terminal task. + if (TERMINAL_STATUSES.includes(task.status)) { + const latest = await ddb.send(new GetCommand({ + TableName: TABLE, Key: { task_id: task.task_id }, ConsistentRead: true, + }), options); + if (latest.Item?.user_id !== task.user_id || !TERMINAL_STATUSES.includes(latest.Item.status)) return; + task = latest.Item as ContinuableTask; + const handle = task.microvm_start?.handle as MicrovmHandle | undefined; + await closeTaskApprovals(task.task_id, task.user_id, options); + if (handle) { + await strategy.stopSession(handle, options); + const observed = await strategy.pollSession(handle, options); + if (!['TERMINATED', 'NOT_FOUND'].includes(observed.microvmState ?? '')) return; + } else { + // An unanswered RunMicrovm response has no handle to inspect. Its fixed + // service lifetime is the only safe bound on a potentially created worker. + const started = Date.parse(task.microvm_start?.createdAt ?? ''); + if (task.microvm_start && !Number.isFinite(started)) { + throw new Error('MICROVM_CONTINUATION_START_TIME_INVALID: cannot confirm the unknown worker lifetime'); + } + if (Number.isFinite(started) && Date.now() < started + MICROVM_MAX_DURATION_SECONDS * 1000) return; + } + await ddb.send(new UpdateCommand({ + TableName: TABLE, + Key: workerLeaseKey(task.task_id), + UpdateExpression: 'SET #ttl = :ttl, lease_state = :closed, lease_user_id = :user, lease_attempt_id = :attempt', + ConditionExpression: 'attribute_not_exists(task_id) OR (lease_user_id = :user AND lease_attempt_id = :attempt)', + ExpressionAttributeNames: { '#ttl': 'ttl' }, + ExpressionAttributeValues: { + ':user': task.user_id, + ':closed': 'CLOSED', + ':attempt': task.microvm_start?.clientToken ?? task.task_id, + ':ttl': Math.floor(Date.now() / 1000) + Number(process.env.TASK_RETENTION_DAYS ?? '90') * 86400, + }, + }), options); + await releaseTaskSlot(task.task_id, task.user_id); + await deleteClosedTaskContinuations(task.task_id, task.user_id, options); + return; + } + if (task.status !== TaskStatus.AWAITING_APPROVAL || !task.awaiting_approval_request_id) return; + const handle = task.microvm_start?.handle as MicrovmHandle | undefined; + if (task.continuation?.state === 'READY' || task.continuation?.state === 'FENCED') { + if (!handle) throw new Error('MICROVM_CONTINUATION_HANDLE_MISSING: cannot confirm source shutdown'); + const observed = await strategy.pollSession(handle, options); + const terminal = ['TERMINATED', 'NOT_FOUND'].includes(observed.microvmState ?? ''); + const sourceStarted = observed.microvmStartedAtMs ?? Date.parse(task.microvm_start?.createdAt ?? ''); + await retireCheckpointedMicrovm({ + taskId: task.task_id, + userId: task.user_id, + handle, + strategy, + force: terminal, + abortSignal: options.abortSignal, + sessionDeadlineMs: Number.isFinite(sourceStarted) + ? sourceStarted + (observed.microvmMaximumDurationSeconds ?? MICROVM_MAX_DURATION_SECONDS) * 1000 + : Infinity, + }); + } + await dispatchMicrovmContinuation(task.task_id, task.user_id, task.awaiting_approval_request_id, options); +} + +/** + * Backstop for lost approval invokes, unfinished retirement, and closed-task + * object cleanup. The task table is scanned with the same bounded-page pattern + * as the concurrency reconciler; reserved lease items have no launch receipt. + */ +export async function handler(_event: unknown, context: Pick): Promise { + const saved = await ddb.send(new GetCommand({ + TableName: TABLE, Key: CURSOR_KEY, ConsistentRead: true, + }), { abortSignal: AbortSignal.timeout(CONTINUATION_IO_TIMEOUT_MS) }); + let lastKey: Record | undefined = saved.Item?.cursor; + const saveCursor = async (cursor?: Record) => { + await ddb.send(new UpdateCommand({ + TableName: TABLE, + Key: CURSOR_KEY, + UpdateExpression: cursor ? 'SET #cursor = :cursor' : 'REMOVE #cursor', + ExpressionAttributeNames: { '#cursor': 'cursor' }, + ...(cursor && { ExpressionAttributeValues: { ':cursor': cursor } }), + }), { abortSignal: AbortSignal.timeout(CURSOR_WRITE_TIMEOUT_MS) }); + }; + let processed = 0; + let failures = 0; + do { + if (context.getRemainingTimeInMillis() < MIN_REMAINING_MS) { + await saveCursor(lastKey); + return; + } + const page: ScanCommandOutput = await ddb.send(new ScanCommand({ + TableName: TABLE, + ConsistentRead: true, + Limit: 100, + FilterExpression: 'attribute_exists(continuation_launch)', + ExclusiveStartKey: lastKey, + }), { abortSignal: AbortSignal.timeout(CONTINUATION_IO_TIMEOUT_MS) }); + // Small batches bound both downstream concurrency and the invocation budget. + const tasks = page.Items ?? []; + let lastProcessedKey = lastKey; + for (let index = 0; index < tasks.length; index += BATCH_SIZE) { + if (context.getRemainingTimeInMillis() < MIN_REMAINING_MS) { + await saveCursor(lastProcessedKey); + logger.warn('Continuation reconciliation reached its invocation budget', { processed, failures }); + return; + } + // The slice bounds work to BATCH_SIZE. + // eslint-disable-next-line @cdklabs/promiseall-no-unbounded-parallelism + await Promise.all(tasks.slice(index, index + BATCH_SIZE).map(async row => { + try { + await reconcileMicrovmContinuation(row as ContinuableTask); + processed++; + } catch (error) { + failures++; + logger.warn('Continuation reconciliation will retry', { task_id: row.task_id, ...microvmErrorIdentity(error) }); + } + })); + lastProcessedKey = { task_id: tasks[Math.min(index + BATCH_SIZE - 1, tasks.length - 1)].task_id }; + } + lastKey = page.LastEvaluatedKey; + } while (lastKey); + await saveCursor(); + logger.info('Continuation reconciliation finished', { processed, failures }); +} diff --git a/cdk/src/handlers/reconcile-stranded-tasks.ts b/cdk/src/handlers/reconcile-stranded-tasks.ts index d3ebc8c5b..8a1098294 100644 --- a/cdk/src/handlers/reconcile-stranded-tasks.ts +++ b/cdk/src/handlers/reconcile-stranded-tasks.ts @@ -30,14 +30,10 @@ * in `orchestrator.ts` via the `agent_heartbeat_at` timeout path — this * reconciler targets `SUBMITTED`, `HYDRATING`, and `AWAITING_APPROVAL`. * - * AWAITING_APPROVAL reconciliation (§13.6): - * If the agent container evicts mid-approval, neither the poll loop - * nor the resume transaction ever fires; the task sits in - * AWAITING_APPROVAL indefinitely while its approval row's TTL - * eventually reaps the row. This reconciler sweeps any - * AWAITING_APPROVAL task whose age exceeds the stranded timeout and - * transitions it to FAILED with a specific reason so the user sees - * a clear failure rather than a silent hang. + * AWAITING_APPROVAL tasks with a saved MicroVM checkpoint belong to the + * continuation manager and can remain open across workers. For other tasks, + * the backstop allows the full eight-hour worker lifetime plus cleanup grace; + * it detects a lost worker/coordinator, not an unanswered human deadline. */ import { @@ -47,6 +43,7 @@ import { PutItemCommand, } from '@aws-sdk/client-dynamodb'; import { ulid } from 'ulid'; +import { closeTaskApprovals } from './shared/close-task-approvals'; import { logger } from './shared/logger'; import { releaseTaskSlot } from './shared/task-concurrency'; import { makeClient } from './shared/ua'; @@ -65,17 +62,12 @@ const STRANDED_TIMEOUT_SECONDS = Number( const TASK_RETENTION_DAYS = Number(process.env.TASK_RETENTION_DAYS ?? '90'); /** - * Separate (longer) timeout for AWAITING_APPROVAL tasks — approvals - * legitimately sit for an hour (the §7.3 ceiling for per-task approval - * timeout). A task's approval row carries its own per-row TTL; this - * sweep is the backstop for the case where the row gets reaped by DDB - * TTL but the TaskTable row never gets unstuck. - * - * Default: 7200s (2 hours) — double the §7.3 1-hour ceiling + an hour - * grace so this reconciler never races the happy-path timer. + * Backstop for approval waits without a recoverable checkpoint: the worker's + * eight-hour lifetime plus 30 minutes for its coordinator to close the task. + * Approval records do not expire merely because their worker stopped. */ const APPROVAL_STRANDED_TIMEOUT_SECONDS = Number( - process.env.APPROVAL_STRANDED_TIMEOUT_SECONDS ?? '7200', + process.env.APPROVAL_STRANDED_TIMEOUT_SECONDS ?? '30600', ); interface StrandedCandidate { @@ -123,6 +115,9 @@ async function findStrandedCandidates( const userId = item.user_id?.S; const createdAt = item.created_at?.S; if (!taskId || !userId || !createdAt) continue; + const continuationState = item.continuation?.M?.state?.S; + if (status === 'AWAITING_APPROVAL' + && ['READY', 'FENCED', 'PARKED', 'STARTING', 'RESTORING'].includes(continuationState ?? '')) continue; // Age by time-in-CURRENT-status, not creation time (#441). A task // that waited in the admission queue longer than the stranded @@ -180,14 +175,23 @@ async function failStrandedTask(task: StrandedCandidate): Promise { UpdateExpression: 'SET #s = :failed, updated_at = :now, completed_at = :now, ' + 'error_message = :err, status_created_at = :sca', - ConditionExpression: '#s = :expected', - ExpressionAttributeNames: { '#s': 'status' }, + ConditionExpression: '#s = :expected' + (task.status === 'AWAITING_APPROVAL' + ? ' AND (attribute_not_exists(continuation.#state) OR NOT (continuation.#state IN (:ready, :fenced, :parked, :starting, :restoring)))' + : ''), + ExpressionAttributeNames: { '#s': 'status', ...(task.status === 'AWAITING_APPROVAL' && { '#state': 'state' }) }, ExpressionAttributeValues: { ':failed': { S: 'FAILED' }, ':expected': { S: task.status }, ':now': { S: now }, ':err': { S: errorMessage }, ':sca': { S: `FAILED#${now}` }, + ...(task.status === 'AWAITING_APPROVAL' && { + ':ready': { S: 'READY' }, + ':fenced': { S: 'FENCED' }, + ':parked': { S: 'PARKED' }, + ':starting': { S: 'STARTING' }, + ':restoring': { S: 'RESTORING' }, + }), }, })); } catch (err: unknown) { @@ -290,6 +294,7 @@ async function failStrandedTask(task: StrandedCandidate): Promise { // If this fails after the terminal write, the capacity reconciler retries // release for terminal held reservations on its next sweep. await releaseTaskSlot(task.task_id, task.user_id); + await closeTaskApprovals(task.task_id, task.user_id); return true; } diff --git a/cdk/src/handlers/shared/builtin-policies.ts b/cdk/src/handlers/shared/builtin-policies.ts index e4c983f3f..e2dd5d9fc 100644 --- a/cdk/src/handlers/shared/builtin-policies.ts +++ b/cdk/src/handlers/shared/builtin-policies.ts @@ -126,20 +126,20 @@ permit (principal, action, resource); // Every rule in this file MUST carry: // @tier("soft") // @rule_id("...") — stable ID for --pre-approve rule:X -// @approval_timeout_s — integer seconds >= 30 (<120 emits WARN per IMPL-25) // @severity — "low" | "medium" | "high" // @category — optional free-form UX grouping +// An optional @approval_timeout_s sets a positive explicit deadline (>= 30s). +// Omitting it uses the task setting, whose default is no automatic expiry. // // Blueprints may OPT OUT of specific rules here via // \`security.cedarPolicies.disable: [rule_id]\`. They may NOT disable any // rule in hard_deny.cedar (blueprint loader rejects those at task start). -// Gate any git --force / -f push. 300s default approval window, medium severity. +// Gate any git --force / -f push, with medium severity. // Covers both long-form (--force) and short-form (-f) variants, including // the bare \`git push -f\` invocation with no branch argument. @tier("soft") @rule_id("force_push_any") -@approval_timeout_s("300") @severity("medium") @category("destructive") forbid (principal, action == Agent::Action::"execute_bash", resource) @@ -147,12 +147,11 @@ when { context.command like "*git push --force*" || context.command like "*git push -f *" || context.command like "*git push -f" }; -// Force-push to main/prod specifically — longer window, higher severity. +// Force-push to main/prod specifically — higher severity. // Multi-match with force_push_any is expected: the engine's annotation -// merging picks min(300, 600)=300s and max(medium, high)=high. +// merging picks max(medium, high)=high. There is no implicit approval expiry. @tier("soft") @rule_id("force_push_main") -@approval_timeout_s("600") @severity("high") @category("destructive") forbid (principal, action == Agent::Action::"execute_bash", resource) @@ -165,7 +164,6 @@ when { context.command like "*git push --force origin main*" // agent bypasses PR workflow by pushing directly. @tier("soft") @rule_id("push_to_protected_branch") -@approval_timeout_s("300") @severity("medium") @category("destructive") forbid (principal, action == Agent::Action::"execute_bash", resource) @@ -174,20 +172,18 @@ when { context.command like "*git push origin main*" || context.command like "*git push origin prod*" || context.command like "*git push origin release/*" }; -// Writes to \`.env\` files typically contain secrets. 600s window, high severity. +// Writes to \`.env\` files typically contain secrets. High severity. @tier("soft") @rule_id("write_env_files") -@approval_timeout_s("600") @severity("high") @category("filesystem") forbid (principal, action == Agent::Action::"write_file", resource) when { context.file_path like "*.env" }; // Writes to any path containing "credentials" — SSH keys, AWS creds, -// service-account JSON, etc. 300s window, high severity. +// service-account JSON, etc. High severity. @tier("soft") @rule_id("write_credentials") -@approval_timeout_s("300") @severity("high") @category("auth") forbid (principal, action == Agent::Action::"write_file", resource) diff --git a/cdk/src/handlers/shared/close-task-approvals.ts b/cdk/src/handlers/shared/close-task-approvals.ts new file mode 100644 index 000000000..96a092208 --- /dev/null +++ b/cdk/src/handlers/shared/close-task-approvals.ts @@ -0,0 +1,96 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import { GetCommand, QueryCommand, UpdateCommand, type QueryCommandOutput } from '@aws-sdk/lib-dynamodb'; +import type { SessionControlOptions } from './compute-strategy'; +import { makeDocClient } from './ua'; +import { TERMINAL_STATUSES } from '../../constructs/task-status'; + +const ddb = makeDocClient(); +const CLEANUP_TIMEOUT_MS = 10_000; +const UPDATE_BATCH_SIZE = 4; + +/** Task closure closes its unanswered requests; no action-content recheck is involved. */ +export async function closeTaskApprovals( + taskId: string, userId: string, options: SessionControlOptions = { abortSignal: AbortSignal.timeout(CLEANUP_TIMEOUT_MS) }, +): Promise { + if (!process.env.TASK_APPROVALS_TABLE_NAME) return; + const current = await ddb.send(new GetCommand({ + TableName: process.env.TASK_TABLE_NAME!, Key: { task_id: taskId }, ConsistentRead: true, + }), options); + if (current.Item?.user_id !== userId || !TERMINAL_STATUSES.includes(current.Item.status)) return; + const terminalStatus: string = current.Item.status; + let key: Record | undefined; + const now = new Date().toISOString(); + const ttl = Math.floor(Date.now() / 1000) + Number(process.env.TASK_RETENTION_DAYS ?? '90') * 86400; + do { + const page: QueryCommandOutput = await ddb.send(new QueryCommand({ + TableName: process.env.TASK_APPROVALS_TABLE_NAME, + ConsistentRead: true, + KeyConditionExpression: 'task_id = :task', + FilterExpression: 'attribute_not_exists(#ttl) OR #status = :pending', + ExpressionAttributeNames: { '#ttl': 'ttl', '#status': 'status' }, + ExpressionAttributeValues: { ':task': taskId, ':pending': 'PENDING' }, + ExclusiveStartKey: key, + }), options); + const rows = (page.Items ?? []).filter(row => + row.user_id === userId && typeof row.request_id === 'string' && typeof row.status === 'string'); + for (let index = 0; index < rows.length; index += UPDATE_BATCH_SIZE) { + // Each slice bounds concurrent writes; completed rows are filtered on retry. + // eslint-disable-next-line @cdklabs/promiseall-no-unbounded-parallelism + await Promise.all(rows.slice(index, index + UPDATE_BATCH_SIZE).map(async row => { + const pending = row.status === 'PENDING'; + try { + await ddb.send(new UpdateCommand({ + TableName: process.env.TASK_APPROVALS_TABLE_NAME, + Key: { task_id: taskId, request_id: row.request_id }, + UpdateExpression: 'SET #ttl = if_not_exists(#ttl, :ttl)' + (pending + ? ', #status = :cancelled, decided_at = :now, cancellation_reason = :reason' : ''), + ConditionExpression: 'user_id = :user AND #status = :observed', + ExpressionAttributeNames: { '#ttl': 'ttl', '#status': 'status' }, + ExpressionAttributeValues: { + ':ttl': ttl, + ':user': userId, + ':observed': row.status, + ...(pending && { + ':cancelled': 'CANCELLED', + ':now': now, + ':reason': `Owning task is ${terminalStatus.toLowerCase()}.`, + }), + }, + }), options); + } catch (error) { + if ((error as { name?: string }).name !== 'ConditionalCheckFailedException') throw error; + // An answer can win after the query. Preserve it and add only retention. + await ddb.send(new UpdateCommand({ + TableName: process.env.TASK_APPROVALS_TABLE_NAME, + Key: { task_id: taskId, request_id: row.request_id }, + UpdateExpression: 'SET #ttl = if_not_exists(#ttl, :ttl)', + ConditionExpression: 'user_id = :user AND #status <> :pending', + ExpressionAttributeNames: { '#ttl': 'ttl', '#status': 'status' }, + ExpressionAttributeValues: { ':ttl': ttl, ':user': userId, ':pending': 'PENDING' }, + }), options).catch(race => { + if ((race as { name?: string }).name !== 'ConditionalCheckFailedException') throw race; + }); + } + })); + } + key = page.LastEvaluatedKey; + } while (key); +} diff --git a/cdk/src/handlers/shared/compute-strategy.ts b/cdk/src/handlers/shared/compute-strategy.ts index 3a8ffc3ba..cf6ff4e97 100644 --- a/cdk/src/handlers/shared/compute-strategy.ts +++ b/cdk/src/handlers/shared/compute-strategy.ts @@ -147,6 +147,8 @@ export interface ComputeStrategy { * the build def (never worse than today). */ readOnly?: boolean; + /** Coordinator-owned checkpoint recovery pins the original worker image. */ + microvmImage?: { readonly imageArn: string; readonly imageVersion: string }; }): Promise; pollSession(handle: SessionHandle, options?: SessionControlOptions): Promise; stopSession(handle: SessionHandle, options?: SessionControlOptions): Promise; diff --git a/cdk/src/handlers/shared/create-task-core.ts b/cdk/src/handlers/shared/create-task-core.ts index 6eccf203d..e08e23ab6 100644 --- a/cdk/src/handlers/shared/create-task-core.ts +++ b/cdk/src/handlers/shared/create-task-core.ts @@ -312,9 +312,8 @@ export async function createTaskCore( } // Cedar HITL — validate approval_timeout_s if supplied (§7.3 step 5). - // maxLifetime-based ceiling clip is applied at orchestrator - // invocation time; at submit time we only enforce the `[floor, cap]` - // envelope. + // Zero retains unanswered requests. Positive explicit deadlines use the + // supported range; the worker's resource lifetime is managed separately. let approvalTimeoutS: number | undefined; if (body.approval_timeout_s !== undefined) { if (typeof body.approval_timeout_s !== 'number' @@ -326,12 +325,12 @@ export async function createTaskCore( requestId, ); } - if (body.approval_timeout_s < APPROVAL_TIMEOUT_S_MIN + if ((body.approval_timeout_s !== 0 && body.approval_timeout_s < APPROVAL_TIMEOUT_S_MIN) || body.approval_timeout_s > APPROVAL_TIMEOUT_S_MAX) { return errorResponse( 400, ErrorCode.VALIDATION_ERROR, - `Invalid approval_timeout_s. Must be between ${APPROVAL_TIMEOUT_S_MIN}s ` + `Invalid approval_timeout_s. Must be 0 (no automatic expiry), or between ${APPROVAL_TIMEOUT_S_MIN}s ` + `and ${APPROVAL_TIMEOUT_S_MAX}s.`, requestId, ); diff --git a/cdk/src/handlers/shared/microvm-approval-wake.ts b/cdk/src/handlers/shared/microvm-approval-wake.ts index 60b937662..1f5350e3d 100644 --- a/cdk/src/handlers/shared/microvm-approval-wake.ts +++ b/cdk/src/handlers/shared/microvm-approval-wake.ts @@ -23,6 +23,7 @@ import { GetMicrovmCommand, LambdaMicrovmsClient, ResumeMicrovmCommand } from '@ import type { Context } from 'aws-lambda'; import type { SessionControlOptions } from './compute-strategy'; import { logger } from './logger'; +import { dispatchMicrovmContinuation } from './microvm-continuation-dispatch'; import { microvmErrorIdentity, microvmRequestIdentity } from './microvm-control'; import { readMicrovmLifecycleSnapshot, saveMicrovmLifecycleIntent, type MicrovmLifecycleSnapshot } from './microvm-lifecycle'; import { makeClient } from './ua'; @@ -69,7 +70,8 @@ function relevant(snapshot: MicrovmLifecycleSnapshot, input: ApprovalWakeInput): /** * Optional latency improvement after the approval transaction commits. The * durable supervisor remains responsible for retries and observed RUNNING. - * This path cannot suspend, terminate, alter task status or rewrite a decision. + * A retired checkpoint is dispatched to its original coordinator version. + * This path cannot suspend, terminate or rewrite the human decision. */ export async function wakeMicrovmAfterApproval(input: ApprovalWakeInput): Promise { const options = { abortSignal: input.options.abortSignal ?? AbortSignal.timeout(APPROVAL_POST_COMMIT_TIMEOUT_MS) }; @@ -95,6 +97,7 @@ export async function wakeMicrovmAfterApproval(input: ApprovalWakeInput): Promis }; try { options.abortSignal?.throwIfAborted(); + if (await dispatchMicrovmContinuation(input.taskId, input.userId, input.requestId, options)) return; const snapshot = await readMicrovmLifecycleSnapshot(input.taskId, input.userId, options); options.abortSignal.throwIfAborted(); if (!snapshot) return; diff --git a/cdk/src/handlers/shared/microvm-continuation-dispatch.ts b/cdk/src/handlers/shared/microvm-continuation-dispatch.ts new file mode 100644 index 000000000..689a845fa --- /dev/null +++ b/cdk/src/handlers/shared/microvm-continuation-dispatch.ts @@ -0,0 +1,110 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import { createHash } from 'node:crypto'; +import { InvokeCommand, LambdaClient } from '@aws-sdk/client-lambda'; +import { GetCommand } from '@aws-sdk/lib-dynamodb'; +import type { SessionControlOptions } from './compute-strategy'; +import { logger } from './logger'; +import type { ContinuableTask } from './microvm-continuation-retirement'; +import type { MicrovmContinuationEvent } from './microvm-continuation-runner'; +import { admitContinuation } from './microvm-continuation-start'; +import { CONTINUATION_IO_TIMEOUT_MS } from './microvm-continuation-timing'; +import { validAttemptId } from './microvm-continuation-types'; +import { microvmErrorIdentity } from './microvm-control'; +import { continuationEnabled } from './microvm-worker-lease'; +import { makeClient, makeDocClient } from './ua'; +import { TaskStatus } from '../../constructs/task-status'; + +const ddb = makeDocClient(); +const MAX_VALIDATION_DETAIL_LENGTH = 512; +let lambda: LambdaClient | undefined; + +/** + * True means this is a retired/restoring worker; callers must not /resume its + * source handle. Admission and invocation can be retried by the periodic scan. + */ +export async function dispatchMicrovmContinuation( + taskId: string, userId: string, requestId: string, + options: SessionControlOptions = { abortSignal: AbortSignal.timeout(CONTINUATION_IO_TIMEOUT_MS) }, +): Promise { + if (!continuationEnabled()) return false; + const response = await ddb.send(new GetCommand({ + TableName: process.env.TASK_TABLE_NAME!, Key: { task_id: taskId }, ConsistentRead: true, + }), options); + const task = response.Item as ContinuableTask | undefined; + if (!task || task.user_id !== userId || task.compute_type !== 'lambda-microvm' + || task.status !== TaskStatus.AWAITING_APPROVAL || task.awaiting_approval_request_id !== requestId) return false; + if (!['FENCED', 'PARKED', 'STARTING', 'RESTORING'].includes(task.continuation?.state ?? '')) return false; + // Retirement still owns termination and capacity release. + if (task.continuation?.state === 'FENCED') return true; + const version = task.continuation_launch?.orchestrator_version; + const configuredArn = process.env.ORCHESTRATOR_FUNCTION_ARN ?? ''; + const match = /^(arn:[^:]+:lambda:[^:]+:\d+:function:[^:]+)(?::[^:]+)?$/.exec(configuredArn); + if (!version || !/^\d+$/.test(version) || !match) { + throw new Error('MICROVM_CONTINUATION_COORDINATOR_INVALID: original published function is unavailable'); + } + const admission = await admitContinuation( + taskId, userId, requestId, Number(process.env.MAX_CONCURRENT_TASKS_PER_USER ?? '10'), options, + ); + if (admission.kind !== 'ready') return true; + const attemptId = admission.task.continuation?.attempt_id; + if (!validAttemptId(attemptId)) throw new Error('MICROVM_CONTINUATION_ASSIGNMENT_INVALID'); + const event: MicrovmContinuationEvent = { + task_id: taskId, continuation_request_id: requestId, continuation_attempt_id: attemptId, + }; + const payload = JSON.stringify(event); + // Lambda accepts at most 64 characters. Keep the complete 256-bit identity; + // a descriptive prefix would exceed the service limit and strand admission. + const name = createHash('sha256').update(payload).digest('hex'); + lambda ??= makeClient(LambdaClient); + try { + const result = await lambda.send(new InvokeCommand({ + FunctionName: `${match[1]}:${version}`, + InvocationType: 'Event', + DurableExecutionName: name, + Payload: Buffer.from(payload), + }), options); + if (result.StatusCode !== 202) throw new Error('MICROVM_CONTINUATION_DISPATCH_FAILED: invocation was not accepted'); + logger.info('Saved task continuation dispatched', { + task_id: taskId, + request_id: requestId, + attempt_id: attemptId, + coordinator_version: version, + durable_execution_name: name, + }); + } catch (error) { + logger.warn('Saved task continuation dispatch needs reconciliation', { + task_id: taskId, + request_id: requestId, + attempt_id: attemptId, + operation: 'Invoke', + coordinator_version: version, + durable_execution_name_length: name.length, + // This Invoke carries only task/request/attempt identifiers, never a + // prompt, credential or signed URL. Its validation message is safe and + // necessary to diagnose errors that a class name alone cannot explain. + ...(error instanceof Error && error.name === 'ValidationException' + ? { validation_detail: error.message.slice(0, MAX_VALIDATION_DETAIL_LENGTH) } : {}), + ...microvmErrorIdentity(error), + }); + throw error; + } + return true; +} diff --git a/cdk/src/handlers/shared/microvm-continuation-retirement.ts b/cdk/src/handlers/shared/microvm-continuation-retirement.ts new file mode 100644 index 000000000..8192b3efa --- /dev/null +++ b/cdk/src/handlers/shared/microvm-continuation-retirement.ts @@ -0,0 +1,303 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import { randomUUID } from 'node:crypto'; +import { GetCommand, TransactWriteCommand } from '@aws-sdk/lib-dynamodb'; +import type { ComputeStrategy, SessionControlOptions } from './compute-strategy'; +import { logger } from './logger'; +import { verifyContinuationCheckpoint } from './microvm-continuation-storage'; +import { CONTINUATION_RETIREMENT_TIMEOUT_MS, CONTINUATION_STOP_TIMEOUT_MS } from './microvm-continuation-timing'; +import { CONTINUATION, type ContinuationRecord, type MicrovmHandle, type WorkerLease, validateContinuation, workerLeaseKey } from './microvm-continuation-types'; +import { continuationEnabled } from './microvm-worker-lease'; +import { MICROVM_SLEEP_AFTER_S_DEFAULT, type TaskRecord } from './types'; +import { makeDocClient } from './ua'; +import { TaskStatus } from '../../constructs/task-status'; + +const ddb = makeDocClient(); +const TABLE = process.env.TASK_TABLE_NAME!; +const APPROVALS = process.env.TASK_APPROVALS_TABLE_NAME!; +const COUNTERS = process.env.USER_CONCURRENCY_TABLE_NAME!; + +interface HeldSlot { + readonly state: 'held' | 'released'; + readonly acquired_at: string; + readonly released_at?: string; + readonly attempt_id?: string; +} + +export type ContinuableTask = TaskRecord & { + readonly concurrency_slot?: HeldSlot; + readonly microvm_start?: { readonly clientToken: string; readonly handle?: MicrovmHandle; readonly createdAt?: string }; +}; + +export type RetirementResult = 'not-due' | 'stopping' | 'parked' | 'ownership-lost'; + +async function readTask(taskId: string, options: SessionControlOptions): Promise { + const result = await ddb.send(new GetCommand({ + TableName: TABLE, Key: { task_id: taskId }, ConsistentRead: true, + }), options); + return result.Item as ContinuableTask | undefined; +} + +function sameRecord(left: unknown, right: unknown): boolean { + const canonical = (value: unknown) => JSON.stringify(value, (_key, item: unknown) => + item && typeof item === 'object' && !Array.isArray(item) + ? Object.fromEntries(Object.entries(item).sort(([a], [b]) => a.localeCompare(b))) + : item); + return canonical(left) === canonical(right); +} + +async function leaseMatches( + task: ContinuableTask, handle: MicrovmHandle, state: 'FENCED' | 'PARKED', options: SessionControlOptions, +): Promise { + if (!task.microvm_start?.clientToken) return false; + const result = await ddb.send(new GetCommand({ + TableName: TABLE, Key: workerLeaseKey(task.task_id), ConsistentRead: true, + }), options); + const lease = result.Item as WorkerLease | undefined; + return lease?.lease_state === state && lease.lease_attempt_id === task.microvm_start.clientToken + && lease.lease_microvm_id === handle.microvmId && lease.lease_user_id === task.user_id + && lease.lease_repo === (task.repo ?? ''); +} + +async function fence(task: ContinuableTask, handle: MicrovmHandle, options: SessionControlOptions): Promise { + const record = task.continuation!; + const fenced: ContinuationRecord = { ...record, state: 'FENCED', source_handle: handle }; + const attempt = task.microvm_start?.clientToken; + if (!attempt || record.identity.attempt_id !== handle.microvmId) { + throw new Error('MICROVM_CONTINUATION_INVALID: source worker has no matching launch receipt'); + } + await verifyContinuationCheckpoint(record, options); + try { + await ddb.send(new TransactWriteCommand({ + TransactItems: [ + { + Update: { + TableName: TABLE, + Key: { task_id: task.task_id }, + UpdateExpression: 'SET continuation = :fenced', + ConditionExpression: 'user_id = :user AND #status = :awaiting AND session_id = :vm ' + + 'AND awaiting_approval_request_id = :request AND continuation = :record', + ExpressionAttributeNames: { '#status': 'status' }, + ExpressionAttributeValues: { + ':user': task.user_id, + ':awaiting': TaskStatus.AWAITING_APPROVAL, + ':vm': handle.microvmId, + ':request': record.identity.request_id, + ':record': record, + ':fenced': fenced, + }, + }, + }, + { + Update: { + TableName: TABLE, + Key: workerLeaseKey(task.task_id), + UpdateExpression: 'SET lease_state = :fenced', + ConditionExpression: 'lease_state = :active AND lease_attempt_id = :attempt ' + + 'AND lease_microvm_id = :vm AND lease_user_id = :user', + ExpressionAttributeValues: { + ':fenced': 'FENCED', + ':active': 'ACTIVE', + ':attempt': attempt, + ':vm': handle.microvmId, + ':user': task.user_id, + }, + }, + }, + { + ConditionCheck: { + TableName: APPROVALS, + Key: { task_id: task.task_id, request_id: record.identity.request_id }, + ConditionExpression: 'user_id = :user AND #status IN (:pending, :approved, :denied, :timedout)', + ExpressionAttributeNames: { '#status': 'status' }, + ExpressionAttributeValues: { + ':user': task.user_id, + ':pending': 'PENDING', + ':approved': 'APPROVED', + ':denied': 'DENIED', + ':timedout': 'TIMED_OUT', + }, + }, + }, + ], + }), options); + return fenced; + } catch (error) { + const latest = await readTask(task.task_id, options); + if (latest?.user_id === task.user_id && latest.status === TaskStatus.AWAITING_APPROVAL + && sameRecord(latest.continuation, fenced) + && await leaseMatches(latest, handle, 'FENCED', options)) return fenced; + // Original worker resumed or cancellation won before the fence. + if (!latest || latest.status !== TaskStatus.AWAITING_APPROVAL + || !sameRecord(latest.continuation, record)) return undefined; + throw error; + } +} + +/** Close the old reservation only after GetMicrovm confirmed terminal/not-found. */ +async function parkAfterTermination(task: ContinuableTask, record: ContinuationRecord, options: SessionControlOptions): Promise { + if (task.concurrency_slot?.state !== 'held' || !task.microvm_start?.clientToken) { + throw new Error('MICROVM_CONTINUATION_RESERVATION_INVALID: fenced worker has no held capacity'); + } + let emptyCounter = false; + for (let attempt = 0; attempt < 2; attempt++) { + const now = new Date().toISOString(); + const revision = randomUUID(); + const parked: ContinuationRecord = { ...record, state: 'PARKED', parked_at: now }; + try { + await ddb.send(new TransactWriteCommand({ + ClientRequestToken: revision, + TransactItems: [ + { + Update: { + TableName: TABLE, + Key: { task_id: task.task_id }, + UpdateExpression: 'SET continuation = :parked, concurrency_slot.#state = :released, ' + + 'concurrency_slot.released_at = :now REMOVE #ttl', + ConditionExpression: 'user_id = :user AND #status = :awaiting AND continuation = :record ' + + 'AND concurrency_slot = :slot', + ExpressionAttributeNames: { '#status': 'status', '#state': 'state', '#ttl': 'ttl' }, + ExpressionAttributeValues: { + ':user': task.user_id, + ':awaiting': TaskStatus.AWAITING_APPROVAL, + ':record': record, + ':slot': task.concurrency_slot, + ':parked': parked, + ':released': 'released', + ':now': now, + }, + }, + }, + { + Update: { + TableName: TABLE, + Key: workerLeaseKey(task.task_id), + UpdateExpression: 'SET lease_state = :parked', + ConditionExpression: 'lease_state = :fenced AND lease_attempt_id = :attempt AND lease_microvm_id = :vm', + ExpressionAttributeValues: { + ':parked': 'PARKED', + ':fenced': 'FENCED', + ':attempt': task.microvm_start.clientToken, + ':vm': record.source_handle!.microvmId, + }, + }, + }, + { + Update: { + TableName: COUNTERS, + Key: { user_id: task.user_id }, + UpdateExpression: emptyCounter + ? 'SET active_count = if_not_exists(active_count, :zero), updated_at = :now, reservation_version = :revision' + : 'SET active_count = active_count - :one, updated_at = :now, reservation_version = :revision', + ConditionExpression: emptyCounter + ? 'attribute_not_exists(active_count) OR active_count >= :zero' + : 'active_count > :zero', + ExpressionAttributeValues: { + ':zero': 0, ...(!emptyCounter && { ':one': 1 }), ':now': now, ':revision': revision, + }, + }, + }, + ], + }), options); + if (emptyCounter) { + logger.warn('Parked continuation whose capacity counter was already empty', { + task_id: task.task_id, error_id: 'CONCURRENCY_EMPTY_COUNTER', + }); + } + return true; + } catch (error) { + const latest = await readTask(task.task_id, options); + if (latest?.user_id === task.user_id && latest.continuation?.state === 'PARKED' + && latest.concurrency_slot?.state === 'released' + && sameRecord(latest.continuation.identity, record.identity) + && await leaseMatches(latest, record.source_handle!, 'PARKED', options)) return true; + if (!latest || latest.status !== TaskStatus.AWAITING_APPROVAL + || !sameRecord(latest.continuation, record)) return false; + const failure = error as { name?: string; CancellationReasons?: { Code?: string }[] }; + if (!emptyCounter && failure.name === 'TransactionCanceledException' + && failure.CancellationReasons?.[2]?.Code === 'ConditionalCheckFailed') { + emptyCounter = true; + continue; + } + throw error; + } + } + throw new Error('MICROVM_CONTINUATION_RESERVATION_INVALID: capacity release was not acknowledged'); +} + +/** + * One bounded retirement cycle. A worker is fenced before termination; its + * reservation remains held until the service confirms it cannot execute. + */ +export async function retireCheckpointedMicrovm(input: { + taskId: string; + userId: string; + handle: MicrovmHandle; + strategy: ComputeStrategy; + sessionDeadlineMs: number; + force?: boolean; + abortSignal?: AbortSignal; +}): Promise { + if (!continuationEnabled()) return 'not-due'; + const timeout = AbortSignal.timeout(CONTINUATION_RETIREMENT_TIMEOUT_MS); + const options = { abortSignal: input.abortSignal ? AbortSignal.any([input.abortSignal, timeout]) : timeout }; + const task = await readTask(input.taskId, options); + if (!task || task.user_id !== input.userId || task.session_id !== input.handle.microvmId) { + return 'ownership-lost'; + } + if (task.status !== TaskStatus.AWAITING_APPROVAL || !task.continuation) return 'not-due'; + const record = task.continuation; + validateContinuation(record, task); + if (record.state === 'PARKED' || record.state === 'FENCED') { + // Workers can publish the continuation attribute, but only the coordinator + // can advance the separate lease. Never treat a worker-written state label + // as proof that shutdown or capacity release actually happened. + if (!await leaseMatches(task, input.handle, record.state, options) + || (record.state === 'PARKED' && task.concurrency_slot?.state !== 'released')) { + throw new Error('MICROVM_CONTINUATION_LEASE_INVALID: retirement state has no coordinator authority'); + } + if (record.state === 'PARKED') return 'parked'; + } + if (record.state !== 'READY' && record.state !== 'FENCED') return 'not-due'; + if (record.state === 'READY') { + const approval = await ddb.send(new GetCommand({ + TableName: APPROVALS, + Key: { task_id: task.task_id, request_id: record.identity.request_id }, + ConsistentRead: true, + }), options); + const now = Date.now(); + const created = Date.parse(approval.Item?.created_at ?? ''); + const sleep = task.microvm_sleep_after_s ?? MICROVM_SLEEP_AFTER_S_DEFAULT; + const ageDue = sleep > 0 && Number.isFinite(created) && now - created >= CONTINUATION.park_after_seconds * 1000; + const lifetimeDue = Number.isFinite(input.sessionDeadlineMs) + && now >= input.sessionDeadlineMs - CONTINUATION.retirement_margin_seconds * 1000; + if (!input.force && !ageDue && !lifetimeDue) return 'not-due'; + } + const fenced = record.state === 'FENCED' ? record : await fence(task, input.handle, options); + if (!fenced) return 'not-due'; + if (fenced.source_handle?.microvmId !== input.handle.microvmId) { + throw new Error('MICROVM_CONTINUATION_INVALID: retirement handle does not match the fence'); + } + const signal = AbortSignal.any([options.abortSignal, AbortSignal.timeout(CONTINUATION_STOP_TIMEOUT_MS)]); + await input.strategy.stopSession(input.handle, { abortSignal: signal }); + const state = await input.strategy.pollSession(input.handle, { abortSignal: signal }); + if (state.microvmState !== 'TERMINATED' && state.microvmState !== 'NOT_FOUND') return 'stopping'; + return await parkAfterTermination(task, fenced, options) ? 'parked' : 'not-due'; +} diff --git a/cdk/src/handlers/shared/microvm-continuation-runner.ts b/cdk/src/handlers/shared/microvm-continuation-runner.ts new file mode 100644 index 000000000..8e53c1382 --- /dev/null +++ b/cdk/src/handlers/shared/microvm-continuation-runner.ts @@ -0,0 +1,283 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import type { DurableContext, WaitForConditionDecision } from '@aws/durable-execution-sdk-js'; +import { TransactWriteCommand } from '@aws-sdk/lib-dynamodb'; +import { resolveComputeStrategy, type ComputeStrategy } from './compute-strategy'; +import type { ContinuableTask } from './microvm-continuation-retirement'; +import { admitContinuation } from './microvm-continuation-start'; +import { loadContinuationLaunch } from './microvm-continuation-storage'; +import { CONTINUATION_IO_TIMEOUT_MS, CONTINUATION_POLL_INTERVAL_MS, CONTINUATION_RETRY_POLL_SECONDS, CONTINUATION_START_ATTEMPTS, CONTINUATION_TRANSITION_POLL_SECONDS } from './microvm-continuation-timing'; +import { type MicrovmHandle, validAttemptId, workerLeaseKey } from './microvm-continuation-types'; +import { microvmErrorIdentity } from './microvm-control'; +import { MICROVM_MAX_POLL_FAILURES, stopMicrovmWithDiagnostics } from './microvm-supervisor'; +import { pollMicrovmTask } from './microvm-task-poll'; +import { emitTaskEvent, envelopeFor, finalizeTask, loadTask, type PollState } from './orchestrator'; +import { deleteMicrovmPayload } from './strategies/lambda-microvm-strategy'; +import { makeDocClient } from './ua'; +import { TaskStatus, TERMINAL_STATUSES } from '../../constructs/task-status'; + +const ddb = makeDocClient(); +const TABLE = process.env.TASK_TABLE_NAME!; +const RESTORE_TIMEOUT_MS = 900_000; + +export interface MicrovmContinuationEvent { + readonly task_id: string; + readonly continuation_request_id: string; + readonly continuation_attempt_id: string; +} + +function sameAttempt(task: ContinuableTask, event: MicrovmContinuationEvent): boolean { + return task.microvm_start?.clientToken === event.continuation_attempt_id + || task.continuation?.attempt_id === event.continuation_attempt_id; +} + +/** + * Fence the assigned process while closing a failed restoration. A lost reply + * is resolved by a strong read; a different attempt is never changed. + */ +export async function failContinuationAttempt( + event: MicrovmContinuationEvent, userId: string, detail: string, +): Promise { + const task = await loadTask(event.task_id, true) as ContinuableTask; + if (task.user_id !== userId || !sameAttempt(task, event) || TERMINAL_STATUSES.includes(task.status)) return; + try { + await ddb.send(new TransactWriteCommand({ + TransactItems: [ + { + Update: { + TableName: TABLE, + Key: { task_id: event.task_id }, + UpdateExpression: 'SET #status = :failed, error_message = :detail, completed_at = :now, ' + + 'updated_at = :now, status_created_at = :statusTime', + ConditionExpression: 'user_id = :user AND #status = :status ' + + 'AND (continuation.attempt_id = :attempt OR microvm_start.clientToken = :attempt)', + ExpressionAttributeNames: { '#status': 'status' }, + ExpressionAttributeValues: { + ':failed': TaskStatus.FAILED, + ':detail': detail, + ':now': new Date().toISOString(), + ':statusTime': `FAILED#${new Date().toISOString()}`, + ':user': userId, + ':status': task.status, + ':attempt': event.continuation_attempt_id, + }, + }, + }, + { + Update: { + TableName: TABLE, + Key: workerLeaseKey(event.task_id), + UpdateExpression: 'SET lease_state = :fenced', + ConditionExpression: 'lease_user_id = :user AND lease_attempt_id = :attempt AND lease_state = :active', + ExpressionAttributeValues: { + ':user': userId, ':attempt': event.continuation_attempt_id, ':active': 'ACTIVE', ':fenced': 'FENCED', + }, + }, + }, + ], + })); + } catch (error) { + const latest = await loadTask(event.task_id, true) as ContinuableTask; + if (latest.user_id === userId && sameAttempt(latest, event) && !TERMINAL_STATUSES.includes(latest.status)) throw error; + } +} + +export interface RestoreState { + readonly deadlineMs: number; + readonly ready?: boolean; + readonly closed?: boolean; + readonly ownershipLost?: boolean; + readonly failure?: string; + readonly consecutivePollFailures?: number; +} + +/** Restoration has its own bounded startup window; it is not a /resume hook. */ +export async function pollContinuationRestore( + event: MicrovmContinuationEvent, userId: string, handle: MicrovmHandle, + strategy: ComputeStrategy, previous: RestoreState, +): Promise { + const task = await loadTask(event.task_id, true) as ContinuableTask; + if (task.user_id !== userId || task.session_id !== handle.microvmId + || task.compute_metadata?.microvmId !== handle.microvmId) return { ...previous, ownershipLost: true }; + if (TERMINAL_STATUSES.includes(task.status)) return { ...previous, closed: true }; + if (task.status === TaskStatus.RUNNING || task.status === TaskStatus.FINALIZING + || (task.status === TaskStatus.AWAITING_APPROVAL + && task.awaiting_approval_request_id !== event.continuation_request_id)) { + return { ...previous, ready: true }; + } + if (!sameAttempt(task, event) || task.continuation?.state !== 'RESTORING') { + return { ...previous, failure: 'MICROVM_CONTINUATION_ASSIGNMENT_CHANGED: restoration no longer owns its saved request' }; + } + if (Date.now() >= previous.deadlineMs) { + return { ...previous, failure: 'MICROVM_CONTINUATION_RESTORE_TIMEOUT: saved workspace and conversation were not restored within 15 minutes' }; + } + try { + const observed = await strategy.pollSession(handle, { abortSignal: AbortSignal.timeout(CONTINUATION_IO_TIMEOUT_MS) }); + if (observed.status === 'completed' || observed.status === 'failed') { + return { ...previous, failure: `MICROVM_CONTINUATION_WORKER_STOPPED: ${observed.reason ?? ('error' in observed ? observed.error : observed.status)}` }; + } + return { deadlineMs: previous.deadlineMs, consecutivePollFailures: 0 }; + } catch (error) { + const failures = (previous.consecutivePollFailures ?? 0) + 1; + return { + deadlineMs: previous.deadlineMs, + consecutivePollFailures: failures, + ...(failures >= MICROVM_MAX_POLL_FAILURES && { + failure: `MICROVM_CONTINUATION_POLL_FAILED: ${microvmErrorIdentity(error).error_type}`, + }), + }; + } +} + +export function continuationWaitStrategy(state: PollState): WaitForConditionDecision { + if (state.microvmParked || state.microvmOwnershipLost + || (state.lastStatus && TERMINAL_STATUSES.includes(state.lastStatus))) return { shouldContinue: false }; + if (state.microvmRetiring) { + return { + shouldContinue: true, delay: { seconds: state.microvmRetirementError ? CONTINUATION_RETRY_POLL_SECONDS : CONTINUATION_TRANSITION_POLL_SECONDS }, + }; + } + if (state.microvmFailureReason || state.sessionUnhealthy) return { shouldContinue: false }; + return { + shouldContinue: true, + delay: { seconds: Math.max(1, Math.ceil((state.microvmSupervisor?.nextPollInMs ?? CONTINUATION_POLL_INTERVAL_MS) / 1000)) }, + }; +} + +/** One durable execution for one assigned replacement; dispatch supplies a stable execution name. */ +export async function runMicrovmContinuation(event: MicrovmContinuationEvent, context: DurableContext): Promise { + if (![event.task_id, event.continuation_request_id, event.continuation_attempt_id].every(validAttemptId)) { + throw new Error('MICROVM_CONTINUATION_EVENT_INVALID: invalid task, request or worker attempt'); + } + const task = await context.step('continuation-task', async () => loadTask(event.task_id, true) as Promise); + if (!sameAttempt(task, event) || TERMINAL_STATUSES.includes(task.status)) return; + const { correlation, log } = envelopeFor(task); + const emit = (type: string, metadata: Record, options?: { abortSignal?: AbortSignal }) => + emitTaskEvent(task.task_id, type, metadata, correlation, options); + let handle: MicrovmHandle | undefined; + let strategy: ComputeStrategy | undefined; + try { + const launch = await context.step('continuation-launch-inputs', async () => { + if (!task.continuation_launch) throw new Error('MICROVM_CONTINUATION_INPUT_INVALID: saved launch is missing'); + const saved = await loadContinuationLaunch(task.task_id, task.user_id, task.continuation_launch); + if (saved.orchestrator_version !== process.env.AWS_LAMBDA_FUNCTION_VERSION) { + throw new Error('MICROVM_CONTINUATION_VERSION_CHANGED: recovery requires the original published coordinator'); + } + return saved; + }); + strategy = resolveComputeStrategy(launch.blueprint); + const assigned = await context.step('continuation-assignment', async () => { + // Dispatch already admitted this attempt. This call verifies its readonly + // lease and held slot; it cannot re-admit a different request. + const current = await loadTask(task.task_id, true) as ContinuableTask; + if (!sameAttempt(current, event)) return null; + const admission = await admitContinuation(task.task_id, task.user_id, event.continuation_request_id, 1); + return admission.kind === 'ready' && sameAttempt(admission.task, event) ? admission.task : null; + }); + if (!assigned) return; + const source = assigned.continuation?.source_handle; + if (!source?.imageArn || !source.imageVersion || !assigned.continuation?.started_at) { + throw new Error('MICROVM_CONTINUATION_IMAGE_INVALID: saved worker image or assignment time is missing'); + } + const startInput = { + taskId: task.task_id, + userId: task.user_id, + blueprintConfig: launch.blueprint, + payload: { + ...launch.payload, + attempt_id: event.continuation_attempt_id, + task_started_at: assigned.continuation.started_at, + }, + microvmImage: { imageArn: source.imageArn, imageVersion: source.imageVersion }, + }; + const started = await context.step('continuation-start', () => strategy!.startSession(startInput), { + // All retries use the immutable start receipt/token, including lost replies. + retryStrategy: (_error, attempt) => ({ shouldRetry: attempt < CONTINUATION_START_ATTEMPTS, delay: { seconds: 10 } }), + }); + if (started.strategyType !== 'lambda-microvm') throw new Error('MICROVM_CONTINUATION_BACKEND_CHANGED'); + handle = started; + const restoring = await context.waitForCondition('continuation-restore', state => + pollContinuationRestore(event, task.user_id, started, strategy!, state), { + initialState: { deadlineMs: Date.parse(assigned.continuation.started_at) + RESTORE_TIMEOUT_MS }, + waitStrategy: state => ({ + shouldContinue: !state.ready && !state.closed && !state.ownershipLost && !state.failure, + delay: { seconds: 5 }, + }), + }); + if (restoring.failure) throw new Error(restoring.failure); + if (restoring.ownershipLost) { + await context.step('continuation-stop-old-worker', () => stopMicrovmWithDiagnostics({ + taskId: task.task_id, handle: started, strategy: strategy!, emitEvent: emit, + })); + return; + } + const final = restoring.closed ? { attempts: 0 } : await context.waitForCondition( + 'continuation-agent-completion', state => pollMicrovmTask({ + taskId: task.task_id, + userId: task.user_id, + handle: started, + strategy: strategy!, + pollIntervalMs: launch.blueprint.poll_interval_ms ?? CONTINUATION_POLL_INTERVAL_MS, + suspendEnabled: process.env.MICROVM_APPROVAL_SUSPEND_ENABLED === 'true', + emitEvent: emit, + }, state), { initialState: { attempts: 0 }, waitStrategy: continuationWaitStrategy }, + ); + await context.step('continuation-finalize', async () => { + if (final.microvmParked) { + await emit('continuation_parked', { + microvm_id: started.microvmId, + detail: 'Your approval request is still available. The saved task will continue on another worker after your answer.', + }); + } else { + try { + const current = await loadTask(task.task_id, true); + if (!final.microvmOwnershipLost && current.session_id === started.microvmId) { + await finalizeTask(task.task_id, final, task.user_id); + } + } finally { + await stopMicrovmWithDiagnostics({ + taskId: task.task_id, handle: started, strategy: strategy!, emitEvent: emit, + }); + } + } + await deleteMicrovmPayload(task.task_id, event.continuation_attempt_id); + }); + } catch (error) { + // Recover a handle saved before the start step's reply was lost. + await context.step('continuation-failed', async () => { + const current = await loadTask(task.task_id, true) as ContinuableTask; + if (!sameAttempt(current, event)) return; + const savedHandle = current.microvm_start?.handle as MicrovmHandle | undefined; + const ownedHandle = handle ?? savedHandle; + await failContinuationAttempt(event, task.user_id, `Saved-task continuation failed: ${String(error)}`); + if (ownedHandle) { + const cleanup = strategy ?? resolveComputeStrategy({ compute_type: 'lambda-microvm' } as Parameters[0]); + await stopMicrovmWithDiagnostics({ taskId: task.task_id, handle: ownedHandle, strategy: cleanup, emitEvent: emit }); + } + await finalizeTask(task.task_id, { attempts: 0 }, task.user_id); + await deleteMicrovmPayload(task.task_id, event.continuation_attempt_id); + log.error('Saved-task continuation failed', { + request_id: event.continuation_request_id, + attempt_id: event.continuation_attempt_id, + ...microvmErrorIdentity(error), + }); + }); + } +} diff --git a/cdk/src/handlers/shared/microvm-continuation-start.ts b/cdk/src/handlers/shared/microvm-continuation-start.ts new file mode 100644 index 000000000..f889f90a5 --- /dev/null +++ b/cdk/src/handlers/shared/microvm-continuation-start.ts @@ -0,0 +1,207 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import { randomUUID } from 'node:crypto'; +import { GetCommand, TransactWriteCommand } from '@aws-sdk/lib-dynamodb'; +import type { SessionControlOptions } from './compute-strategy'; +import type { ContinuableTask } from './microvm-continuation-retirement'; +import { CONTINUATION_IO_TIMEOUT_MS } from './microvm-continuation-timing'; +import { type ContinuationRecord, type WorkerLease, validateContinuation, workerLeaseKey } from './microvm-continuation-types'; +import { makeDocClient } from './ua'; +import { TaskStatus } from '../../constructs/task-status'; + +const ddb = makeDocClient(); +const TABLE = process.env.TASK_TABLE_NAME!; +const APPROVALS = process.env.TASK_APPROVALS_TABLE_NAME!; +const COUNTERS = process.env.USER_CONCURRENCY_TABLE_NAME!; + +export type ContinuationAdmission = + | { readonly kind: 'ready'; readonly task: ContinuableTask } + | { readonly kind: 'waiting' | 'closed' | 'capacity' }; + +async function readTask(taskId: string, options: SessionControlOptions): Promise { + const result = await ddb.send(new GetCommand({ + TableName: TABLE, Key: { task_id: taskId }, ConsistentRead: true, + }), options); + return result.Item as ContinuableTask | undefined; +} + +/** A replay can use only the same active assignment, never revive a retired lease. */ +async function readActiveAssignment(task: ContinuableTask, options: SessionControlOptions): Promise { + const record = task.continuation; + if (!record?.attempt_id || task.concurrency_slot?.state !== 'held' + || task.concurrency_slot.attempt_id !== record.attempt_id) return false; + const result = await ddb.send(new GetCommand({ + TableName: TABLE, Key: workerLeaseKey(task.task_id), ConsistentRead: true, + }), options); + const lease = result.Item as WorkerLease | undefined; + if (record.state === 'RESTORING' && (!task.session_id || record.worker_id !== task.session_id + || lease?.lease_microvm_id !== task.session_id)) return false; + return lease?.lease_state === 'ACTIVE' && lease.lease_attempt_id === record.attempt_id + && lease.lease_user_id === task.user_id && lease.lease_repo === (task.repo ?? ''); +} + +/** + * Claim capacity and assign a fresh worker token only for a resolved PARKED + * request. The old worker was already confirmed stopped before PARKED. + */ +export async function admitContinuation( + taskId: string, userId: string, requestId: string, limit: number, + options: SessionControlOptions = { abortSignal: AbortSignal.timeout(CONTINUATION_IO_TIMEOUT_MS) }, +): Promise { + const task = await readTask(taskId, options); + if (!task || task.user_id !== userId || task.status !== TaskStatus.AWAITING_APPROVAL + || task.awaiting_approval_request_id !== requestId || !task.continuation) return { kind: 'closed' }; + const record = task.continuation; + validateContinuation(record, task); + if (record.state === 'STARTING' || record.state === 'RESTORING') { + if (!await readActiveAssignment(task, options)) { + throw new Error('MICROVM_CONTINUATION_LEASE_INVALID: replacement assignment is not active'); + } + return { kind: 'ready', task }; + } + if (record.state !== 'PARKED') return { kind: 'waiting' }; + const approval = await ddb.send(new GetCommand({ + TableName: APPROVALS, Key: { task_id: taskId, request_id: requestId }, ConsistentRead: true, + }), options); + if (approval.Item?.user_id !== userId || approval.Item.status === 'CANCELLED') return { kind: 'closed' }; + const finiteDeadline = Date.parse(approval.Item?.created_at ?? '') + Number(approval.Item?.timeout_s) * 1000; + const expired = approval.Item?.status === 'PENDING' && Number(approval.Item.timeout_s) > 0 + && Number.isFinite(finiteDeadline) && Date.now() >= finiteDeadline; + if (!expired && !['APPROVED', 'DENIED', 'TIMED_OUT'].includes(approval.Item?.status)) return { kind: 'waiting' }; + if (task.concurrency_slot?.state !== 'released' || !task.microvm_start?.clientToken + || !record.source_handle || record.source_handle.microvmId !== task.session_id + || !Number.isSafeInteger(limit) || limit <= 0) { + throw new Error('MICROVM_CONTINUATION_RESERVATION_INVALID: parked task has no released source reservation'); + } + const attempt = randomUUID(); + const now = new Date().toISOString(); + const starting: ContinuationRecord = { + ...record, state: 'STARTING', attempt_id: attempt, started_at: now, + }; + const slot = { state: 'held' as const, acquired_at: now, attempt_id: attempt }; + const lease: WorkerLease = { + ...workerLeaseKey(taskId), + lease_state: 'ACTIVE', + lease_attempt_id: attempt, + lease_user_id: userId, + lease_repo: record.identity.repo, + }; + try { + await ddb.send(new TransactWriteCommand({ + ClientRequestToken: attempt, + TransactItems: [ + { + Update: { + TableName: TABLE, + Key: { task_id: taskId }, + UpdateExpression: 'SET continuation = :starting, concurrency_slot = :slot ' + + 'REMOVE session_id, compute_metadata, microvm_start, microvm_lifecycle, agent_heartbeat_at', + ConditionExpression: 'user_id = :user AND #status = :awaiting AND awaiting_approval_request_id = :request ' + + 'AND continuation = :record AND concurrency_slot.#state = :released', + ExpressionAttributeNames: { '#status': 'status', '#state': 'state' }, + ExpressionAttributeValues: { + ':user': userId, + ':awaiting': TaskStatus.AWAITING_APPROVAL, + ':request': requestId, + ':record': record, + ':released': 'released', + ':starting': starting, + ':slot': slot, + }, + }, + }, + { + Update: { + TableName: COUNTERS, + Key: { user_id: userId }, + UpdateExpression: 'SET active_count = if_not_exists(active_count, :zero) + :one, ' + + 'updated_at = :now, reservation_version = :revision', + ConditionExpression: 'attribute_not_exists(active_count) OR active_count < :limit', + ExpressionAttributeValues: { + ':zero': 0, ':one': 1, ':now': now, ':revision': attempt, ':limit': limit, + }, + }, + }, + { + Put: { + TableName: TABLE, + Item: lease, + ConditionExpression: 'lease_state = :parked AND lease_attempt_id = :source ' + + 'AND lease_microvm_id = :vm AND lease_user_id = :user', + ExpressionAttributeValues: { + ':parked': 'PARKED', + ':source': task.microvm_start.clientToken, + ':vm': record.source_handle.microvmId, + ':user': userId, + }, + }, + }, + expired ? { + Update: { + TableName: APPROVALS, + Key: { task_id: taskId, request_id: requestId }, + UpdateExpression: 'SET #status = :timedout, decided_at = :now', + ConditionExpression: 'user_id = :user AND #status = :pending AND created_at = :created AND timeout_s = :timeout', + ExpressionAttributeNames: { '#status': 'status' }, + ExpressionAttributeValues: { + ':user': userId, + ':pending': 'PENDING', + ':timedout': 'TIMED_OUT', + ':now': now, + ':created': approval.Item!.created_at, + ':timeout': approval.Item!.timeout_s, + }, + }, + } : { + ConditionCheck: { + TableName: APPROVALS, + Key: { task_id: taskId, request_id: requestId }, + ConditionExpression: 'user_id = :user AND #status IN (:approved, :denied, :timedout)', + ExpressionAttributeNames: { '#status': 'status' }, + ExpressionAttributeValues: { + ':user': userId, ':approved': 'APPROVED', ':denied': 'DENIED', ':timedout': 'TIMED_OUT', + }, + }, + }, + ], + }), options); + const updated: ContinuableTask = { + ...task, + continuation: starting, + concurrency_slot: slot, + session_id: undefined, + compute_metadata: undefined, + microvm_start: undefined, + agent_heartbeat_at: undefined, + }; + return { kind: 'ready', task: updated }; + } catch (error) { + const latest = await readTask(taskId, options); + if (!latest || latest.user_id !== userId || latest.status !== TaskStatus.AWAITING_APPROVAL + || latest.awaiting_approval_request_id !== requestId) return { kind: 'closed' }; + if (['STARTING', 'RESTORING'].includes(latest.continuation?.state ?? '') + && await readActiveAssignment(latest, options)) return { kind: 'ready', task: latest }; + const failure = error as { name?: string; CancellationReasons?: { Code?: string }[] }; + if (failure.name === 'TransactionCanceledException' + && failure.CancellationReasons?.[1]?.Code === 'ConditionalCheckFailed' + && latest.continuation?.state === 'PARKED') return { kind: 'capacity' }; + throw error; + } +} diff --git a/cdk/src/handlers/shared/microvm-continuation-storage.ts b/cdk/src/handlers/shared/microvm-continuation-storage.ts new file mode 100644 index 000000000..d12333f24 --- /dev/null +++ b/cdk/src/handlers/shared/microvm-continuation-storage.ts @@ -0,0 +1,287 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import { createHash } from 'node:crypto'; +import { Readable } from 'node:stream'; +import { DeleteObjectsCommand, GetObjectCommand, HeadObjectCommand, ListObjectVersionsCommand, PutObjectCommand, S3Client } from '@aws-sdk/client-s3'; +import { GetCommand, UpdateCommand } from '@aws-sdk/lib-dynamodb'; +import type { SessionControlOptions } from './compute-strategy'; +import { CONTINUATION_IO_TIMEOUT_MS } from './microvm-continuation-timing'; +import { + CONTINUATION, type ContinuationLaunchReceipt, type ContinuationRecord, validAttemptId, +} from './microvm-continuation-types'; +import { continuationEnabled } from './microvm-worker-lease'; +import type { BlueprintConfig } from './repo-config'; +import { makeClient, makeDocClient } from './ua'; +import constants from '../../../../contracts/constants.json'; +import { TERMINAL_STATUSES } from '../../constructs/task-status'; + +const ddb = makeDocClient(); +let s3: S3Client | undefined; +function storage(): S3Client { + return s3 ??= makeClient(S3Client); +} +const TABLE = process.env.TASK_TABLE_NAME!; +const MAX_BYTES = constants.payload_bootstrap.max_payload_bytes; + +interface LaunchInputs { + readonly version: number; + readonly task_id: string; + readonly user_id: string; + readonly payload: Record; + readonly blueprint: BlueprintConfig; + readonly orchestrator_version: string; +} + +function canonical(value: unknown): Buffer { + return Buffer.from(JSON.stringify(value, (_key, item: unknown) => + item && typeof item === 'object' && !Array.isArray(item) + ? Object.fromEntries(Object.entries(item).sort(([a], [b]) => a.localeCompare(b))) + : item)); +} + +function hash(value: Buffer): string { + return createHash('sha256').update(value).digest('hex'); +} + +async function readObject(key: string, versionId?: string, options?: SessionControlOptions) { + const timeout = AbortSignal.timeout(CONTINUATION_IO_TIMEOUT_MS); + const signal = options?.abortSignal ? AbortSignal.any([options.abortSignal, timeout]) : timeout; + const response = await storage().send(new GetObjectCommand({ + Bucket: process.env.CONTINUATION_BUCKET_NAME!, + Key: key, + ...(versionId && { VersionId: versionId }), + }), { abortSignal: signal }); + if (!(response.Body instanceof Readable) || !response.ContentLength || response.ContentLength > MAX_BYTES + || !response.VersionId || response.VersionId === 'null' + || (versionId && response.VersionId !== versionId)) { + if (response.Body instanceof Readable) response.Body.destroy(); + throw new Error('MICROVM_CONTINUATION_STORAGE_INVALID: saved object is incomplete or unversioned'); + } + const stream = response.Body; + const abort = () => { stream.destroy(new Error('MICROVM_CONTINUATION_STORAGE_TIMEOUT: download did not finish')); }; + signal.addEventListener('abort', abort, { once: true }); + const chunks: Buffer[] = []; + let size = 0; + try { + signal.throwIfAborted(); + for await (const chunk of stream) { + signal.throwIfAborted(); + const bytes = Buffer.from(chunk); + size += bytes.length; + if (size > response.ContentLength || size > MAX_BYTES) { + throw new Error('MICROVM_CONTINUATION_STORAGE_INVALID: download exceeded its declared size'); + } + chunks.push(bytes); + } + } finally { + signal.removeEventListener('abort', abort); + stream.destroy(); + } + const bytes = Buffer.concat(chunks, size); + if (bytes.length !== response.ContentLength) { + throw new Error('MICROVM_CONTINUATION_STORAGE_INVALID: launch object length changed'); + } + return { bytes, versionId: response.VersionId }; +} + +/** Save exact hydrated inputs independently of the short-lived bootstrap URL. */ +export async function saveContinuationLaunch( + taskId: string, userId: string, payload: Record, blueprint: BlueprintConfig, +): Promise { + if (!continuationEnabled()) return; + const version = process.env.AWS_LAMBDA_FUNCTION_VERSION ?? ''; + if (!validAttemptId(taskId) || payload.task_id !== taskId || payload.user_id !== userId || !/^\d+$/.test(version)) { + throw new Error('MICROVM_CONTINUATION_INPUT_INVALID: task identity or published coordinator version is missing'); + } + const inputs: LaunchInputs = { + version: CONTINUATION.version, + task_id: taskId, + user_id: userId, + payload, + blueprint, + orchestrator_version: version, + }; + const bytes = canonical(inputs); + if (!bytes.length || bytes.length > MAX_BYTES) { + throw new Error('MICROVM_CONTINUATION_INPUT_INVALID: launch inputs exceed the storage bound'); + } + const sha256 = hash(bytes); + const key = `${CONTINUATION.object_key_prefix}${taskId}/launch/${sha256}.json`; + let putError: unknown; + try { + await storage().send(new PutObjectCommand({ + Bucket: process.env.CONTINUATION_BUCKET_NAME!, + Key: key, + Body: bytes, + ContentType: 'application/json', + ServerSideEncryption: 'AES256', + ChecksumSHA256: createHash('sha256').update(bytes).digest('base64'), + IfNoneMatch: '*', + }), { abortSignal: AbortSignal.timeout(CONTINUATION_IO_TIMEOUT_MS) }); + } catch (error) { + putError = error; + } + // Read back even after a successful Put. This also recovers a lost reply or + // an identical publication by a replay, without changing the version pointer. + let stored: Awaited>; + try { + stored = await readObject(key); + } catch (error) { + throw putError ?? error; + } + if (!stored.bytes.equals(bytes)) { + throw new Error('MICROVM_CONTINUATION_STORAGE_INVALID: launch readback does not match published bytes'); + } + const receipt: ContinuationLaunchReceipt = { + version: CONTINUATION.version, + key, + version_id: stored.versionId, + sha256, + size_bytes: bytes.length, + orchestrator_version: version, + }; + try { + await ddb.send(new UpdateCommand({ + TableName: TABLE, + Key: { task_id: taskId }, + UpdateExpression: 'SET continuation_launch = :receipt REMOVE #ttl', + ConditionExpression: 'user_id = :user AND #status = :hydrating ' + + 'AND (attribute_not_exists(continuation_launch) OR continuation_launch = :receipt)', + ExpressionAttributeNames: { '#status': 'status', '#ttl': 'ttl' }, + ExpressionAttributeValues: { ':receipt': receipt, ':user': userId, ':hydrating': 'HYDRATING' }, + })); + } catch (error) { + const current = await ddb.send(new GetCommand({ + TableName: TABLE, Key: { task_id: taskId }, ConsistentRead: true, + })); + if (current.Item?.user_id !== userId + || !canonical(current.Item.continuation_launch ?? null).equals(canonical(receipt))) throw error; + } +} + +/** The task record pins the version; workers cannot choose or overwrite this pointer. */ +export async function loadContinuationLaunch( + taskId: string, userId: string, receipt: ContinuationLaunchReceipt, +): Promise { + if (!validAttemptId(taskId) || receipt?.version !== CONTINUATION.version + || !/^[a-f0-9]{64}$/.test(receipt.sha256) + || receipt.key !== `${CONTINUATION.object_key_prefix}${taskId}/launch/${receipt.sha256}.json` + || !receipt.version_id || receipt.version_id === 'null' + || !Number.isSafeInteger(receipt.size_bytes) || receipt.size_bytes <= 0 || receipt.size_bytes > MAX_BYTES + || !/^\d+$/.test(receipt.orchestrator_version)) { + throw new Error('MICROVM_CONTINUATION_INPUT_INVALID: saved launch receipt is invalid'); + } + const stored = await readObject(receipt.key, receipt.version_id); + if (stored.bytes.length !== receipt.size_bytes || hash(stored.bytes) !== receipt.sha256) { + throw new Error('MICROVM_CONTINUATION_STORAGE_INVALID: saved launch checksum does not match'); + } + const inputs = JSON.parse(stored.bytes.toString('utf8')) as LaunchInputs; + if (!inputs || inputs.version !== CONTINUATION.version || inputs.task_id !== taskId || inputs.user_id !== userId + || inputs.payload?.task_id !== taskId || inputs.payload.user_id !== userId + || inputs.blueprint?.compute_type !== 'lambda-microvm' + || inputs.orchestrator_version !== receipt.orchestrator_version) { + throw new Error('MICROVM_CONTINUATION_INPUT_INVALID: saved launch belongs to a different task'); + } + return inputs; +} + +/** Confirm all acknowledged object versions before retiring the only worker. */ +export async function verifyContinuationCheckpoint(record: ContinuationRecord, options?: SessionControlOptions): Promise { + const stored = await readObject(record.manifest.key, record.manifest.version_id, options); + if (stored.bytes.length !== record.manifest.size_bytes || hash(stored.bytes) !== record.manifest.sha256) { + throw new Error('MICROVM_CONTINUATION_STORAGE_INVALID: checkpoint manifest checksum does not match'); + } + const manifest = JSON.parse(stored.bytes.toString('utf8')); + if (!manifest || manifest.version !== CONTINUATION.version + || !canonical(manifest.identity ?? null).equals(canonical(record.identity))) { + throw new Error('MICROVM_CONTINUATION_STORAGE_INVALID: checkpoint manifest identity does not match'); + } + const identity = record.identity; + const prefix = `${CONTINUATION.object_key_prefix}${identity.task_id}/${identity.attempt_id}/${identity.request_id}/`; + for (const object of [ + { receipt: manifest.conversation, prefix, suffix: '.json', limit: CONTINUATION.max_conversation_bytes }, + { receipt: manifest.workspace, prefix: `${prefix}workspace/`, suffix: '.tar', limit: CONTINUATION.max_workspace_bytes }, + ]) { + const receipt = object.receipt; + if (!receipt || !/^[a-f0-9]{64}$/.test(receipt.sha256) + || receipt.key !== `${object.prefix}${receipt.sha256}${object.suffix}` + || typeof receipt.version_id !== 'string' || !receipt.version_id || receipt.version_id === 'null' + || !Number.isSafeInteger(receipt.size_bytes) || receipt.size_bytes <= 0 || receipt.size_bytes > object.limit) { + throw new Error('MICROVM_CONTINUATION_STORAGE_INVALID: checkpoint contains an invalid object receipt'); + } + const response = await storage().send(new HeadObjectCommand({ + Bucket: process.env.CONTINUATION_BUCKET_NAME!, + Key: receipt.key, + VersionId: receipt.version_id, + ChecksumMode: 'ENABLED', + }), { + abortSignal: options?.abortSignal + ? AbortSignal.any([options.abortSignal, AbortSignal.timeout(CONTINUATION_IO_TIMEOUT_MS)]) + : AbortSignal.timeout(CONTINUATION_IO_TIMEOUT_MS), + }); + if (response.VersionId !== receipt.version_id || response.ContentLength !== receipt.size_bytes + || response.ChecksumSHA256 !== Buffer.from(receipt.sha256, 'hex').toString('base64')) { + throw new Error('MICROVM_CONTINUATION_STORAGE_INVALID: checkpoint object version, length or checksum does not match'); + } + } +} + +/** + * Called only after the last worker is confirmed stopped. Remove every version, + * including superseded checkpoints, before clearing the task's cleanup marker. + * Active records have no object expiry: a pending answer can outlive a worker. + */ +export async function deleteClosedTaskContinuations( + taskId: string, userId: string, options: SessionControlOptions, +): Promise { + if (!validAttemptId(taskId)) throw new Error('MICROVM_CONTINUATION_CLEANUP_INVALID'); + const current = await ddb.send(new GetCommand({ + TableName: TABLE, Key: { task_id: taskId }, ConsistentRead: true, + }), options); + if (current.Item?.user_id !== userId || !TERMINAL_STATUSES.includes(current.Item.status)) return; + const prefix = `${CONTINUATION.object_key_prefix}${taskId}/`; + // Always list from the start after deleting a page. No cursor can skip a + // version whose neighbour was removed in the preceding batch. + while (true) { + options.abortSignal?.throwIfAborted(); + const page = await storage().send(new ListObjectVersionsCommand({ + Bucket: process.env.CONTINUATION_BUCKET_NAME!, Prefix: prefix, MaxKeys: 1000, + }), options); + const objects = [...(page.Versions ?? []), ...(page.DeleteMarkers ?? [])].map(object => { + if (!object.Key?.startsWith(prefix) || !object.VersionId) { + throw new Error('MICROVM_CONTINUATION_CLEANUP_INVALID: unexpected object identity'); + } + return { Key: object.Key, VersionId: object.VersionId }; + }); + if (!objects.length) break; + const removed = await storage().send(new DeleteObjectsCommand({ + Bucket: process.env.CONTINUATION_BUCKET_NAME!, Delete: { Objects: objects, Quiet: true }, + }), options); + if (removed.Errors?.length) throw new Error('MICROVM_CONTINUATION_CLEANUP_FAILED: object versions remain'); + } + await ddb.send(new UpdateCommand({ + TableName: TABLE, + Key: { task_id: taskId }, + UpdateExpression: 'SET continuation_cleanup_at = :now REMOVE continuation, continuation_launch', + ConditionExpression: 'user_id = :user AND #status = :status', + ExpressionAttributeNames: { '#status': 'status' }, + ExpressionAttributeValues: { ':user': userId, ':status': current.Item.status, ':now': new Date().toISOString() }, + }), options); +} diff --git a/cdk/src/handlers/shared/microvm-continuation-timing.ts b/cdk/src/handlers/shared/microvm-continuation-timing.ts new file mode 100644 index 000000000..e075fd41e --- /dev/null +++ b/cdk/src/handlers/shared/microvm-continuation-timing.ts @@ -0,0 +1,27 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +/** Transport limits fit inside a durable Lambda's 60-second invocation budget. */ +export const CONTINUATION_IO_TIMEOUT_MS = 10_000; +export const CONTINUATION_RETIREMENT_TIMEOUT_MS = 40_000; +export const CONTINUATION_STOP_TIMEOUT_MS = 25_000; +export const CONTINUATION_POLL_INTERVAL_MS = 30_000; +export const CONTINUATION_TRANSITION_POLL_SECONDS = 5; +export const CONTINUATION_RETRY_POLL_SECONDS = 30; +export const CONTINUATION_START_ATTEMPTS = 4; diff --git a/cdk/src/handlers/shared/microvm-continuation-types.ts b/cdk/src/handlers/shared/microvm-continuation-types.ts new file mode 100644 index 000000000..30426761b --- /dev/null +++ b/cdk/src/handlers/shared/microvm-continuation-types.ts @@ -0,0 +1,98 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import type { SessionHandle } from './compute-strategy'; +import constants from '../../../../contracts/constants.json'; + +export const CONTINUATION = constants.microvm_continuation; +export type MicrovmHandle = Extract; + +export interface ContinuationIdentity { + readonly task_id: string; + /** Physical worker that captured the source checkpoint. */ + readonly attempt_id: string; + readonly request_id: string; + readonly user_id: string; + readonly repo: string; +} + +export interface ContinuationReceipt { + readonly kind: 'manifest'; + readonly key: string; + readonly version_id: string; + readonly sha256: string; + readonly size_bytes: number; +} + +export interface ContinuationRecord { + readonly version: number; + readonly state: 'READY' | 'FENCED' | 'PARKED' | 'STARTING' | 'RESTORING' | 'CONSUMED'; + readonly identity: ContinuationIdentity; + readonly manifest: ContinuationReceipt; + readonly source_handle?: MicrovmHandle; + readonly parked_at?: string; + /** Logical launch token, assigned before calling RunMicrovm. */ + readonly attempt_id?: string; + readonly worker_id?: string; + readonly started_at?: string; +} + +export interface ContinuationLaunchReceipt { + readonly version: number; + readonly key: string; + readonly version_id: string; + readonly sha256: string; + readonly size_bytes: number; + readonly orchestrator_version: string; +} + +export interface WorkerLease { + readonly task_id: string; + readonly lease_attempt_id: string; + readonly lease_state: 'ACTIVE' | 'FENCED' | 'PARKED' | 'CLOSED'; + readonly lease_user_id: string; + readonly lease_repo: string; + readonly lease_microvm_id?: string; +} + +export function workerLeaseKey(taskId: string): { task_id: string } { + return { task_id: CONTINUATION.lease_key_prefix + taskId }; +} + +export function validAttemptId(value: unknown): value is string { + return typeof value === 'string' && /^[A-Za-z0-9][A-Za-z0-9_-]{0,127}$/.test(value); +} + +/** A worker may publish a pointer only within its own complete checkpoint prefix. */ +export function validateContinuation(record: ContinuationRecord, task: { + task_id: string; user_id: string; repo?: string; awaiting_approval_request_id?: string; +}): void { + const identity = record?.identity; + const receipt = record?.manifest; + if (record?.version !== CONTINUATION.version || !identity || !receipt + || identity.task_id !== task.task_id || identity.user_id !== task.user_id || identity.repo !== (task.repo ?? '') + || identity.request_id !== task.awaiting_approval_request_id + || !validAttemptId(identity.task_id) || !validAttemptId(identity.attempt_id) || !validAttemptId(identity.request_id) + || receipt.kind !== 'manifest' || !/^[a-f0-9]{64}$/.test(receipt.sha256) + || receipt.key !== `${CONTINUATION.object_key_prefix}${identity.task_id}/${identity.attempt_id}/${identity.request_id}/manifest/${receipt.sha256}.json` + || typeof receipt.version_id !== 'string' || !receipt.version_id || receipt.version_id === 'null' + || !Number.isSafeInteger(receipt.size_bytes) || receipt.size_bytes <= 0 || receipt.size_bytes > CONTINUATION.max_manifest_bytes) { + throw new Error('MICROVM_CONTINUATION_INVALID: checkpoint identity or receipt does not match the pending request'); + } +} diff --git a/cdk/src/handlers/shared/microvm-lifecycle-policy.ts b/cdk/src/handlers/shared/microvm-lifecycle-policy.ts index bdea8716a..e60aaef07 100644 --- a/cdk/src/handlers/shared/microvm-lifecycle-policy.ts +++ b/cdk/src/handlers/shared/microvm-lifecycle-policy.ts @@ -89,7 +89,7 @@ export function decideMicrovmLifecycle(input: MicrovmLifecyclePolicyInput): Micr return sleeping ? wake('wake-intent') : { action: 'wait', reason: 'wake-intent', nextPollInMs: transitionPoll }; } if (approval.createdAtMs > nowMs) return wake('approval-time-invalid'); - const wakeAt = Math.min(approval.deadlineMs, sessionDeadlineMs) - MICROVM_WAKE_MARGIN_MS; + const wakeAt = Math.min(approval.deadlineMs ?? sessionDeadlineMs, sessionDeadlineMs) - MICROVM_WAKE_MARGIN_MS; if (nowMs >= wakeAt) return wake('wake-deadline'); const sleepAfterSeconds = snapshot.sleepAfterSeconds === undefined diff --git a/cdk/src/handlers/shared/microvm-lifecycle.ts b/cdk/src/handlers/shared/microvm-lifecycle.ts index b8a88e5ce..e8e73d1bf 100644 --- a/cdk/src/handlers/shared/microvm-lifecycle.ts +++ b/cdk/src/handlers/shared/microvm-lifecycle.ts @@ -46,7 +46,7 @@ export type LifecycleApproval = readonly created_at: string; readonly timeout_s: number; readonly createdAtMs: number; - readonly deadlineMs: number; + readonly deadlineMs: number | null; } | { readonly kind: 'none' | 'missing' | 'invalid' } | { readonly kind: 'unavailable'; readonly errorType: string }; @@ -100,7 +100,7 @@ function validIntent(value: unknown): value is MicrovmLifecycleIntent { && (item.request_id === null || nonblank(item.request_id)) && (item.action === 'suspend' || item.action === 'resume') && timestamp(item.requested_at_ms) && (item.deadline_ms === null || timestamp(item.deadline_ms)) - && (item.action !== 'suspend' || (item.request_id !== null && item.deadline_ms !== null)); + && (item.action !== 'suspend' || item.request_id !== null); } function parseApproval(row: Record | undefined, taskId: string, userId: string, requestId: string): LifecycleApproval { @@ -108,11 +108,12 @@ function parseApproval(row: Record | undefined, taskId: string, if (row.task_id !== taskId || row.user_id !== userId || row.request_id !== requestId || !APPROVAL_STATUSES.includes(row.status as ApprovalStatus) || typeof row.created_at !== 'string' || !/^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(\.\d{3})?Z$/.test(row.created_at) - || !timestamp(row.timeout_s) || row.timeout_s === 0) return { kind: 'invalid' }; + || !timestamp(row.timeout_s)) return { kind: 'invalid' }; const createdAtMs = Date.parse(row.created_at); const canonical = row.created_at.includes('.') ? row.created_at : row.created_at.replace('Z', '.000Z'); - const deadlineMs = createdAtMs + row.timeout_s * 1000; - if (!timestamp(createdAtMs) || new Date(createdAtMs).toISOString() !== canonical || !timestamp(deadlineMs)) return { kind: 'invalid' }; + const deadlineMs = row.timeout_s === 0 ? null : createdAtMs + row.timeout_s * 1000; + if (!timestamp(createdAtMs) || new Date(createdAtMs).toISOString() !== canonical + || (deadlineMs !== null && !timestamp(deadlineMs))) return { kind: 'invalid' }; return { kind: 'present', status: row.status as ApprovalStatus, @@ -197,7 +198,8 @@ function eligible(snapshot: MicrovmLifecycleSnapshot, action: LifecycleAction, n return supportsMicrovmLifecycle(snapshot.handle) && snapshot.status === TaskStatus.AWAITING_APPROVAL && snapshot.requestId !== null && snapshot.approval.kind === 'present' && snapshot.approval.status === 'PENDING' - && nowMs >= snapshot.approval.createdAtMs && nowMs < snapshot.approval.deadlineMs + && nowMs >= snapshot.approval.createdAtMs + && (snapshot.approval.deadlineMs === null || nowMs < snapshot.approval.deadlineMs) && !(intentMatchesGate(snapshot) && (snapshot.intent?.action === 'resume' || snapshot.intent?.deadline_ms !== snapshot.approval.deadlineMs)); } diff --git a/cdk/src/handlers/shared/microvm-start.ts b/cdk/src/handlers/shared/microvm-start.ts index 8aaa2bffc..9553dc704 100644 --- a/cdk/src/handlers/shared/microvm-start.ts +++ b/cdk/src/handlers/shared/microvm-start.ts @@ -18,9 +18,11 @@ */ import { createHash } from 'node:crypto'; -import { GetCommand, UpdateCommand } from '@aws-sdk/lib-dynamodb'; +import { GetCommand, TransactWriteCommand, UpdateCommand } from '@aws-sdk/lib-dynamodb'; import type { SessionHandle } from './compute-strategy'; +import type { ContinuationRecord } from './microvm-continuation-types'; import { MICROVM_IMAGE_CAPABILITY_REQUEST_TIMEOUT_MS, readMicrovmImageMetadata, supportsMicrovmLifecycle } from './microvm-image-capability'; +import { continuationEnabled, ensureWorkerLease, leaseHandleUpdate } from './microvm-worker-lease'; import { makeDocClient } from './ua'; import { TaskStatus, TERMINAL_STATUSES } from '../../constructs/task-status'; @@ -28,8 +30,8 @@ type MicrovmHandle = Extract; /** * Internal TaskTable attribute, deliberately not part of the task API. - * One task owns one logical start. Retrying a failed task creates a new task ID; - * neither an application retry nor durable replay may mint a replacement token. + * Each authorized worker attempt owns one logical start. Only a coordinator + * continuation assignment may replace it; retries reuse the existing token. */ interface StartReceipt { readonly clientToken: string; @@ -46,6 +48,8 @@ interface StartRecord { readonly compute_type?: string; readonly compute_metadata?: Record; readonly microvm_start?: StartReceipt; + readonly repo?: string; + readonly continuation?: ContinuationRecord; } export interface MicrovmStartClaim { @@ -111,20 +115,27 @@ export async function claimMicrovmStart( taskId: string, userId: string, requestHash: string, + attemptId: string = taskId, ): Promise { for (let attempt = 0; attempt < 2; attempt++) { const record = await readStartRecord(taskId, userId); const handle = recordedHandle(record); if (TERMINAL_STATUSES.some(status => status === record.status)) { - return { clientToken: taskId, handle, closed: true }; + return { clientToken: attemptId, handle, closed: true }; } if (!ACTIVE.has(record.status)) { throw new Error(`MICROVM_START_STATE_INVALID: cannot start a task in ${record.status}`); } - if (handle) return { clientToken: taskId, handle, closed: false }; + if (attemptId !== taskId && record.continuation?.attempt_id !== attemptId) { + throw new Error('MICROVM_START_STATE_INVALID: replacement has no coordinator assignment'); + } + if (record.microvm_start && record.microvm_start.clientToken !== attemptId) { + throw new Error('MICROVM_START_STATE_INVALID: start belongs to another worker attempt'); + } + if (handle) return { clientToken: attemptId, handle, closed: false }; const receipt = record.microvm_start; if (receipt) { - if (receipt.clientToken !== taskId || !Number.isFinite(receipt.expiresAt)) { + if (receipt.clientToken !== attemptId || !Number.isFinite(receipt.expiresAt)) { throw new Error('MICROVM_START_STATE_INVALID: invalid saved start receipt'); } if (receipt.requestHash !== requestHash) { @@ -133,14 +144,17 @@ export async function claimMicrovmStart( if (Date.now() >= receipt.expiresAt) { throw new Error('MICROVM_START_OUTCOME_UNKNOWN: replay window expired; inspect the original start before retrying'); } + await ensureWorkerLease({ taskId, userId, repo: record.repo ?? '', attemptId, requestHash }); return { clientToken: receipt.clientToken, closed: false }; } - if (record.status !== TaskStatus.HYDRATING) { + const replacement = attemptId !== taskId + && record.status === TaskStatus.AWAITING_APPROVAL && record.continuation?.state === 'STARTING'; + if (record.status !== TaskStatus.HYDRATING && !replacement) { throw new Error('MICROVM_START_STATE_INVALID: active task has no recoverable start receipt or handle'); } const now = Date.now(); const receiptToSave: StartReceipt = { - clientToken: taskId, + clientToken: attemptId, requestHash, createdAt: new Date(now).toISOString(), expiresAt: now + MICROVM_START_REPLAY_WINDOW_MS, @@ -150,15 +164,18 @@ export async function claimMicrovmStart( TableName: TABLE_NAME, Key: { task_id: taskId }, UpdateExpression: 'SET microvm_start = :receipt', - ConditionExpression: '#status = :hydrating AND user_id = :user AND attribute_not_exists(microvm_start)', - ExpressionAttributeNames: { '#status': 'status' }, + ConditionExpression: '#status = :startingStatus AND user_id = :user AND attribute_not_exists(microvm_start)' + + (replacement ? ' AND continuation.#state = :starting AND continuation.attempt_id = :attempt' : ''), + ExpressionAttributeNames: { '#status': 'status', ...(replacement && { '#state': 'state' }) }, ExpressionAttributeValues: { ':receipt': receiptToSave, - ':hydrating': TaskStatus.HYDRATING, + ':startingStatus': replacement ? TaskStatus.AWAITING_APPROVAL : TaskStatus.HYDRATING, ':user': userId, + ...(replacement && { ':starting': 'STARTING', ':attempt': attemptId }), }, })); - return { clientToken: taskId, closed: false }; + await ensureWorkerLease({ taskId, userId, repo: record.repo ?? '', attemptId, requestHash }); + return { clientToken: attemptId, closed: false }; } catch (err) { if ((err as { name?: string }).name !== 'ConditionalCheckFailedException') throw err; // Re-read the winner, or observe cancellation, before any side effect. @@ -173,15 +190,18 @@ export async function saveMicrovmStartHandle( clientToken: string, handle: MicrovmHandle, ): Promise { - await ddb.send(new UpdateCommand({ + const replacement = clientToken !== taskId; + const update = { TableName: TABLE_NAME, Key: { task_id: taskId }, UpdateExpression: 'SET microvm_start.#handle = :handle, session_id = :id, ' - + 'compute_type = :type, compute_metadata = :metadata', + + 'compute_type = :type, compute_metadata = :metadata' + + (replacement ? ', continuation.worker_id = :id, continuation.#state = :restoring' : ''), ConditionExpression: 'microvm_start.clientToken = :token AND ' + '(attribute_not_exists(microvm_start.#handle) OR microvm_start.#handle.microvmId = :id) AND ' - + '(attribute_not_exists(session_id) OR session_id = :id)', - ExpressionAttributeNames: { '#handle': 'handle' }, + + '(attribute_not_exists(session_id) OR session_id = :id)' + + (replacement ? ' AND continuation.attempt_id = :token' : ''), + ExpressionAttributeNames: { '#handle': 'handle', ...(replacement && { '#state': 'state' }) }, ExpressionAttributeValues: { ':token': clientToken, ':id': handle.microvmId, @@ -190,8 +210,18 @@ export async function saveMicrovmStartHandle( ':metadata': { microvmId: handle.microvmId, endpoint: handle.endpoint, ...readMicrovmImageMetadata(handle), }, + ...(replacement && { ':restoring': 'RESTORING' }), }, - })); + }; + if (continuationEnabled()) { + await ddb.send(new TransactWriteCommand({ + TransactItems: [ + { Update: update }, leaseHandleUpdate(taskId, clientToken, handle.microvmId), + ], + })); + } else { + await ddb.send(new UpdateCommand(update)); + } } /** Enrich only the same durably saved launch; never replace its identity or task state. */ diff --git a/cdk/src/handlers/shared/microvm-supervisor.ts b/cdk/src/handlers/shared/microvm-supervisor.ts index 919232130..b59f18616 100644 --- a/cdk/src/handlers/shared/microvm-supervisor.ts +++ b/cdk/src/handlers/shared/microvm-supervisor.ts @@ -71,6 +71,8 @@ export interface MicrovmSupervisorInput { readonly previous?: MicrovmSupervisorState; readonly pollIntervalMs: number; readonly suspendEnabled: boolean; + /** Optional shared cycle deadline when retirement precedes ordinary supervision. */ + readonly abortSignal?: AbortSignal; /** Implementations must respect the supplied signal; event failure is best-effort. */ readonly emitEvent?: (eventType: string, metadata: Record, options: SessionControlOptions) => Promise; } @@ -129,7 +131,10 @@ export async function superviseMicrovm(input: MicrovmSupervisorInput): Promise= snapshot.approval.deadlineMs); + || (snapshot.approval.deadlineMs !== null && Date.now() >= snapshot.approval.deadlineMs)); if (observed === 'RUNNING') { // AWS RUNNING does not prove the guest consumed a decided/expired gate. // Recovery ends after fresh guest liveness, or an intentional early wake diff --git a/cdk/src/handlers/shared/microvm-task-poll.ts b/cdk/src/handlers/shared/microvm-task-poll.ts new file mode 100644 index 000000000..c4cbf04a2 --- /dev/null +++ b/cdk/src/handlers/shared/microvm-task-poll.ts @@ -0,0 +1,106 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +const RETIREMENT_EVENT_TIMEOUT_MS = 3000; +import { formatMicrovmTerminalFailure } from './error-classifier'; +import { logger } from './logger'; +import { retireCheckpointedMicrovm, type RetirementResult } from './microvm-continuation-retirement'; +import { microvmErrorIdentity } from './microvm-control'; +import { MICROVM_SUPERVISOR_CYCLE_MS, superviseMicrovm, type MicrovmSupervisorInput } from './microvm-supervisor'; +import type { PollState } from './orchestrator'; +import { TaskStatus } from '../../constructs/task-status'; + +function retirementState(state: PollState, result: RetirementResult): PollState | undefined { + if (result === 'not-due') return undefined; + return { + attempts: state.attempts + 1, + lastStatus: TaskStatus.AWAITING_APPROVAL, + microvmSupervisor: state.microvmSupervisor, + microvmParked: result === 'parked', + microvmRetiring: result === 'stopping', + microvmOwnershipLost: result === 'ownership-lost', + }; +} + +async function retirementCycle( + input: Omit, state: PollState, force = false, +): Promise { + try { + const retirement = await retireCheckpointedMicrovm({ + ...input, sessionDeadlineMs: state.microvmSupervisor?.sessionDeadlineMs ?? Infinity, force, + }); + return retirementState(state, retirement); + } catch (error) { + // Fencing/termination may have committed before a lost reply. Keep the + // reservation held and retry the persisted transition; never resume tools + // or mark the task complete based on an uncertain control outcome. + const identity = microvmErrorIdentity(error); + logger.warn('MicroVM continuation retirement needs reconciliation', { + task_id: input.taskId, microvm_id: input.handle.microvmId, ...identity, + }); + if (state.microvmRetirementError !== identity.error_type) { + try { + await input.emitEvent?.('continuation_retirement_delayed', { + microvm_id: input.handle.microvmId, + error_id: identity.error_type, + detail: 'The saved task is waiting for worker shutdown or storage confirmation. Its capacity reservation remains held.', + }, { abortSignal: AbortSignal.timeout(RETIREMENT_EVENT_TIMEOUT_MS) }); + } catch (eventError) { + logger.warn('Could not publish continuation reconciliation feedback', { + task_id: input.taskId, ...microvmErrorIdentity(eventError), + }); + } + } + return { + attempts: state.attempts + 1, + lastStatus: state.lastStatus, + microvmSupervisor: state.microvmSupervisor, + microvmRetiring: true, + microvmRetirementError: identity.error_type, + }; + } +} + +/** Shared by initial and replacement durable executions. Never follow another handle. */ +export async function pollMicrovmTask(input: Omit, state: PollState): Promise { + const timeout = AbortSignal.timeout(MICROVM_SUPERVISOR_CYCLE_MS); + input = { ...input, abortSignal: input.abortSignal ? AbortSignal.any([input.abortSignal, timeout]) : timeout }; + const retiring = await retirementCycle(input, state); + if (retiring) return retiring; + const supervised = await superviseMicrovm({ ...input, previous: state.microvmSupervisor }); + if (supervised.kind === 'substrate-terminal' || supervised.kind === 'failure') { + // A complete pending checkpoint can outlive a dead worker. No tool ran + // beyond that barrier, and the original human request/decision is retained. + const recovered = await retirementCycle(input, { ...state, microvmSupervisor: supervised.state }, true); + if (recovered) return recovered; + } + const failure = supervised.kind === 'failure' ? supervised.reason + : supervised.kind === 'substrate-terminal' ? 'substrate-terminal' : undefined; + return { + attempts: state.attempts + 1, + lastStatus: supervised.snapshot?.status ?? state.lastStatus, + sessionUnhealthy: supervised.heartbeatUnhealthy, + microvmSupervisor: supervised.state, + microvmFailureReason: failure, + microvmFailureMessage: supervised.kind === 'substrate-terminal' ? formatMicrovmTerminalFailure( + `substrate state ${supervised.substrate!.status}`, supervised.substrate!.reason, + ) : undefined, + microvmOwnershipLost: supervised.kind === 'ownership-lost', + }; +} diff --git a/cdk/src/handlers/shared/microvm-worker-lease.ts b/cdk/src/handlers/shared/microvm-worker-lease.ts new file mode 100644 index 000000000..75c138b47 --- /dev/null +++ b/cdk/src/handlers/shared/microvm-worker-lease.ts @@ -0,0 +1,93 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import { GetCommand, TransactWriteCommand } from '@aws-sdk/lib-dynamodb'; +import { workerLeaseKey, type WorkerLease } from './microvm-continuation-types'; +import { makeDocClient } from './ua'; + +const ddb = makeDocClient(); +const TABLE = process.env.TASK_TABLE_NAME!; + +export function continuationEnabled(): boolean { + return Boolean(process.env.CONTINUATION_BUCKET_NAME); +} + +/** Create authority once. Replays must never reactivate a fenced worker. */ +export async function ensureWorkerLease(input: { + taskId: string; userId: string; repo: string; attemptId: string; requestHash: string; +}): Promise { + if (!continuationEnabled()) return; + const lease: WorkerLease = { + ...workerLeaseKey(input.taskId), + lease_attempt_id: input.attemptId, + lease_state: 'ACTIVE', + lease_user_id: input.userId, + lease_repo: input.repo, + }; + try { + await ddb.send(new TransactWriteCommand({ + TransactItems: [ + { + ConditionCheck: { + TableName: TABLE, + Key: { task_id: input.taskId }, + ConditionExpression: 'user_id = :user AND microvm_start.clientToken = :attempt ' + + 'AND microvm_start.requestHash = :hash AND #status IN (:hydrating, :awaiting)', + ExpressionAttributeNames: { '#status': 'status' }, + ExpressionAttributeValues: { + ':user': input.userId, + ':attempt': input.attemptId, + ':hash': input.requestHash, + ':hydrating': 'HYDRATING', + ':awaiting': 'AWAITING_APPROVAL', + }, + }, + }, + { + Put: { + TableName: TABLE, + Item: lease, + ConditionExpression: 'attribute_not_exists(task_id)', + }, + }, + ], + })); + } catch (error) { + const result = await ddb.send(new GetCommand({ + TableName: TABLE, Key: workerLeaseKey(input.taskId), ConsistentRead: true, + })); + const current = result.Item as WorkerLease | undefined; + if (current?.lease_state !== 'ACTIVE' || current.lease_attempt_id !== input.attemptId + || current.lease_user_id !== input.userId || current.lease_repo !== input.repo) throw error; + } +} + +/** Bind a returned physical handle without altering whether the lease is fenced. */ +export function leaseHandleUpdate(taskId: string, attemptId: string, microvmId: string) { + return { + Update: { + TableName: TABLE, + Key: workerLeaseKey(taskId), + UpdateExpression: 'SET lease_microvm_id = :id', + ConditionExpression: 'lease_attempt_id = :attempt AND ' + + '(attribute_not_exists(lease_microvm_id) OR lease_microvm_id = :id)', + ExpressionAttributeValues: { ':attempt': attemptId, ':id': microvmId }, + }, + }; +} diff --git a/cdk/src/handlers/shared/orchestrator.ts b/cdk/src/handlers/shared/orchestrator.ts index e3efe35b3..0f0fd4afd 100644 --- a/cdk/src/handlers/shared/orchestrator.ts +++ b/cdk/src/handlers/shared/orchestrator.ts @@ -21,6 +21,7 @@ import { S3Client } from '@aws-sdk/client-s3'; import { GetCommand, PutCommand, UpdateCommand } from '@aws-sdk/lib-dynamodb'; import { ulid } from 'ulid'; import { evaluateAgentHeartbeat } from './agent-heartbeat'; +import { closeTaskApprovals } from './close-task-approvals'; import type { SessionControlOptions, SessionHandle } from './compute-strategy'; import { AttachmentBudgetExceededError, AttachmentConfigurationError, AttachmentResolutionError, hydrateContext, resolveGitHubToken } from './context-hydration'; import { logger, type Logger } from './logger'; @@ -70,6 +71,10 @@ export interface PollState { readonly microvmFailureMessage?: string; /** A stale execution must clean up only its own handle, leaving the replacement alone. */ readonly microvmOwnershipLost?: boolean; + /** The worker is stopped and its reservation released; the approval remains open. */ + readonly microvmParked?: boolean; + readonly microvmRetiring?: boolean; + readonly microvmRetirementError?: string; } /** @@ -962,6 +967,7 @@ export async function finalizeTask( // The marker makes this safe after a crash, an event failure, or a competing // cleaner. A still-active task keeps its reservation. await releaseTaskSlot(taskId, userId); + await closeTaskApprovals(taskId, userId); } } diff --git a/cdk/src/handlers/shared/payload-bootstrap.ts b/cdk/src/handlers/shared/payload-bootstrap.ts index 8134d3835..fbb88c84c 100644 --- a/cdk/src/handlers/shared/payload-bootstrap.ts +++ b/cdk/src/handlers/shared/payload-bootstrap.ts @@ -30,6 +30,7 @@ type Backend = 'ecs' | 'lambda-microvm'; export interface PayloadReference { version: number; task_id: string; + attempt_id?: string; bootstrap_s3_uri: string; payload_url: string; expires_at: number; @@ -96,6 +97,7 @@ async function createOnce(bucket: string, key: string, body: string): Promise; platformConfig?: Record; @@ -104,6 +106,11 @@ export async function preparePayloadReference(input: { if (!/^[A-Za-z0-9_-]{1,128}$/.test(taskId) || payload.task_id !== taskId) { throw new Error('PAYLOAD_BOOTSTRAP_INVALID: task identity does not match the payload'); } + if (input.attemptId !== undefined && ( + backend !== 'lambda-microvm' || !/^[A-Za-z0-9_-]{1,128}$/.test(input.attemptId) + || payload.attempt_id !== input.attemptId + )) throw new Error('PAYLOAD_BOOTSTRAP_INVALID: worker attempt does not match the payload'); + const objectPrefix = input.attemptId ? `${taskId}/${input.attemptId}` : taskId; const manifest = canonical({ version: PAYLOAD_BOOTSTRAP.version, backend, platform_config: input.platformConfig ?? {}, }); @@ -119,10 +126,11 @@ export async function preparePayloadReference(input: { throw new Error('PAYLOAD_BOOTSTRAP_TOO_LARGE: bootstrap manifest or task payload exceeds its byte limit'); } const fingerprint = sha256(canonical({ bucket, backend, manifest, payloadBody })); - const launchKey = `${taskId}/${PAYLOAD_BOOTSTRAP.launch_filename}`; + const launchKey = `${objectPrefix}/${PAYLOAD_BOOTSTRAP.launch_filename}`; const accept = (saved: string): PayloadReference => { const record = JSON.parse(saved) as LaunchRecord; - if (record.fingerprint !== fingerprint || record.reference?.task_id !== taskId) { + if (record.fingerprint !== fingerprint || record.reference?.task_id !== taskId + || record.reference.attempt_id !== input.attemptId) { throw new Error('PAYLOAD_BOOTSTRAP_CONFLICT: task already has different launch instructions'); } if (record.reference.expires_at <= Date.now()) { @@ -139,7 +147,7 @@ export async function preparePayloadReference(input: { const existing = await readObject(bucket, launchKey); if (existing !== undefined) return accept(existing); - const payloadKey = `${taskId}/payload.json`; + const payloadKey = `${objectPrefix}/payload.json`; const savedPayload = await createOnce(bucket, payloadKey, payloadBody); if (savedPayload !== payloadBody) { throw new Error('PAYLOAD_BOOTSTRAP_CONFLICT: task already has different stored instructions'); @@ -161,6 +169,7 @@ export async function preparePayloadReference(input: { const reference: PayloadReference = { version: PAYLOAD_BOOTSTRAP.version, task_id: taskId, + ...(input.attemptId && { attempt_id: input.attemptId }), bootstrap_s3_uri: `s3://${bucket}/${manifestKey}`, payload_url: url, expires_at: now + lifetime * 1000, @@ -170,10 +179,15 @@ export async function preparePayloadReference(input: { } /** Delete both task instructions and their saved capability; shared manifests expire by lifecycle. */ -export async function deletePayloadReference(bucket: string, taskId: string): Promise { +export async function deletePayloadReference(bucket: string, taskId: string, attemptId?: string): Promise { + if (!/^[A-Za-z0-9_-]{1,128}$/.test(taskId) + || (attemptId !== undefined && !/^[A-Za-z0-9_-]{1,128}$/.test(attemptId))) { + throw new Error('PAYLOAD_BOOTSTRAP_INVALID: cleanup identity is invalid'); + } + const objectPrefix = attemptId ? `${taskId}/${attemptId}` : taskId; for (const filename of ['payload.json', PAYLOAD_BOOTSTRAP.launch_filename]) { try { - await s3().send(new DeleteObjectCommand({ Bucket: bucket, Key: `${taskId}/${filename}` })); + await s3().send(new DeleteObjectCommand({ Bucket: bucket, Key: `${objectPrefix}/${filename}` })); } catch (error) { logger.warn('Payload bootstrap cleanup failed (non-fatal)', { task_id: taskId, filename, error: (error as { name?: string }).name ?? 'UnknownError', diff --git a/cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts b/cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts index 9d4ab2e4b..1fb66ca57 100644 --- a/cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts +++ b/cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts @@ -34,6 +34,7 @@ import sharedConstants from '../../../../../contracts/constants.json'; import type { ComputeStrategy, SessionControlOptions, SessionHandle, SessionLifecycleResult, SessionStatus, SessionStopResult } from '../compute-strategy'; import { MicrovmStartUncertainError } from '../error-classifier'; import { logger } from '../logger'; +import { validAttemptId } from '../microvm-continuation-types'; import { microvmErrorIdentity, microvmRequestIdentity } from '../microvm-control'; import { MICROVM_IMAGE_CAPABILITY_REQUEST_TIMEOUT_MS, MICROVM_LIFECYCLE_PROTOCOL, @@ -384,9 +385,9 @@ function wrapMicrovmError(operation: string, err: unknown): Error { /** Remove task instructions and their saved download capability after finalization. * Best-effort; bucket lifecycle reaps leftovers. Deployment manifests are shared. */ -export async function deleteMicrovmPayload(taskId: string): Promise { +export async function deleteMicrovmPayload(taskId: string, attemptId?: string): Promise { if (!MICROVM_PAYLOAD_BUCKET) return; - await deletePayloadReference(MICROVM_PAYLOAD_BUCKET, taskId); + await deletePayloadReference(MICROVM_PAYLOAD_BUCKET, taskId, attemptId); } /** Split a comma-separated env-var list into trimmed, non-empty entries. */ @@ -474,6 +475,7 @@ export class LambdaMicrovmComputeStrategy implements ComputeStrategy { userId: string; payload: Record; blueprintConfig: BlueprintConfig; + microvmImage?: { readonly imageArn: string; readonly imageVersion: string }; }): Promise { if (!MICROVM_IMAGE_IDENTIFIER || !MICROVM_EXECUTION_ROLE_ARN || !MICROVM_EGRESS_CONNECTOR_ARNS || !MICROVM_PAYLOAD_BUCKET) { // Config/deploy mismatch: this repo is compute_type=lambda-microvm but the @@ -493,10 +495,17 @@ export class LambdaMicrovmComputeStrategy implements ComputeStrategy { } const { taskId, payload } = input; + const attemptId = payload.attempt_id ?? taskId; + if (!validAttemptId(attemptId)) throw new Error('MICROVM_START_STATE_INVALID: invalid worker attempt'); // An identifier that is not an ARN cannot launch anything — check before the // payload upload so a misconfiguration never leaves an orphan S3 object. assertImageArn(MICROVM_IMAGE_IDENTIFIER); + if (input.microvmImage && (input.microvmImage.imageArn !== MICROVM_IMAGE_IDENTIFIER + || !/^\d+\.\d+$/.test(input.microvmImage.imageVersion))) { + throw new Error('MICROVM_CONTINUATION_IMAGE_INVALID: recovery must use a version of the configured image'); + } + const imageVersion = input.microvmImage?.imageVersion ?? MICROVM_IMAGE_VERSION; // The manifest authenticates deployment settings through the worker's IAM // grant. Payload access uses a single-object URL, saved outside TaskTable. @@ -521,7 +530,7 @@ export class LambdaMicrovmComputeStrategy implements ComputeStrategy { const request = { imageIdentifier: MICROVM_IMAGE_IDENTIFIER, - ...(MICROVM_IMAGE_VERSION && { imageVersion: MICROVM_IMAGE_VERSION }), + ...(imageVersion && { imageVersion }), executionRoleArn: MICROVM_EXECUTION_ROLE_ARN, // Egress rides the platform VPC through an egress network connector so the // DNS Firewall / security-group / flow-log stack applies unchanged @@ -547,14 +556,19 @@ export class LambdaMicrovmComputeStrategy implements ComputeStrategy { const requestHash = microvmStartRequestHash( { ...request, payloadBucket: MICROVM_PAYLOAD_BUCKET }, { ...payload, platform_config: platformConfig }, ); - const claim = await claimMicrovmStart(taskId, input.userId, requestHash); + const claim = await claimMicrovmStart(taskId, input.userId, requestHash, attemptId); if (claim.closed) { if (claim.handle) await this.stopSession(claim.handle); throw new Error('MICROVM_START_TASK_CLOSED: task became terminal before session start'); } if (claim.handle) return claim.handle; const reference = await preparePayloadReference({ - bucket: MICROVM_PAYLOAD_BUCKET, taskId, backend: 'lambda-microvm', payload, platformConfig, + bucket: MICROVM_PAYLOAD_BUCKET, + taskId, + backend: 'lambda-microvm', + payload, + platformConfig, + ...(attemptId !== taskId && { attemptId }), }).catch((error: unknown) => { throw wrapMicrovmError('payload bootstrap', error); }); const runHookPayload = JSON.stringify(reference); if (Buffer.byteLength(runHookPayload, 'utf8') > RUN_HOOK_PAYLOAD_LIMIT_BYTES) { @@ -562,7 +576,7 @@ export class LambdaMicrovmComputeStrategy implements ComputeStrategy { } // Uploads can take time. Observe cancellation/another saved handle again // immediately before the service call, using the same immutable request. - const latest = await claimMicrovmStart(taskId, input.userId, requestHash); + const latest = await claimMicrovmStart(taskId, input.userId, requestHash, attemptId); if (latest.closed) { if (latest.handle) await this.stopSession(latest.handle); throw new Error('MICROVM_START_TASK_CLOSED: task became terminal before session start'); @@ -633,7 +647,7 @@ export class LambdaMicrovmComputeStrategy implements ComputeStrategy { // The write may have committed before its response was lost. Recover that // receipt before destroying a computer whose handle is already durable. try { - const saved = await claimMicrovmStart(taskId, input.userId, requestHash); + const saved = await claimMicrovmStart(taskId, input.userId, requestHash, attemptId); if (!saved.closed && saved.handle?.microvmId === microvmId) return saved.handle; } catch (readErr) { logger.warn('Could not reconcile the MicroVM start receipt', { task_id: taskId, error: String(readErr) }); diff --git a/cdk/src/handlers/shared/task-concurrency.ts b/cdk/src/handlers/shared/task-concurrency.ts index d6f979d20..fcde85569 100644 --- a/cdk/src/handlers/shared/task-concurrency.ts +++ b/cdk/src/handlers/shared/task-concurrency.ts @@ -20,6 +20,7 @@ import { randomUUID } from 'node:crypto'; import { GetCommand, TransactWriteCommand } from '@aws-sdk/lib-dynamodb'; import { logger } from './logger'; +import { workerLeaseKey } from './microvm-continuation-types'; import { makeDocClient } from './ua'; import { ACTIVE_STATUSES, TaskStatus, TERMINAL_STATUSES } from '../../constructs/task-status'; @@ -28,6 +29,8 @@ export interface ReservationTask { readonly task_id: string; readonly user_id: string; readonly status: string; + readonly continuation_launch?: unknown; + readonly microvm_start?: { readonly clientToken: string }; readonly concurrency_slot?: { readonly state: 'held' | 'released'; readonly acquired_at: string; @@ -123,6 +126,17 @@ export async function releaseTaskSlot(taskId: string, userId: string): Promise { // Main task read/update grants plus three supporting tables. expect(ddbStatements).toHaveLength(5); for (const s of ddbStatements) { + const actions = Array.isArray(s.Action) ? s.Action : [s.Action]; + const readonlyTask = actions.includes('dynamodb:GetItem') && !actions.includes('dynamodb:PutItem'); expect(s.Condition['ForAllValues:StringEquals']['dynamodb:LeadingKeys']) - .toEqual(['${aws:PrincipalTag/task_id}']); + .toEqual(readonlyTask + ? ['${aws:PrincipalTag/task_id}', 'worker-lease#${aws:PrincipalTag/task_id}'] + : ['${aws:PrincipalTag/task_id}']); expect(s.Condition.Null['dynamodb:LeadingKeys']).toBe('false'); - const actions = Array.isArray(s.Action) ? s.Action : [s.Action]; + if (readonlyTask) expect(actions).not.toContain('dynamodb:UpdateItem'); // Scan must NOT be granted — it ignores leading-keys. expect(actions).not.toContain('dynamodb:Scan'); } diff --git a/cdk/test/constructs/continuation-bucket.test.ts b/cdk/test/constructs/continuation-bucket.test.ts new file mode 100644 index 000000000..2a99f19f2 --- /dev/null +++ b/cdk/test/constructs/continuation-bucket.test.ts @@ -0,0 +1,56 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import { App, Stack } from 'aws-cdk-lib'; +import { Template, Match } from 'aws-cdk-lib/assertions'; +import * as iam from 'aws-cdk-lib/aws-iam'; +import { ContinuationBucket } from '../../src/constructs/continuation-bucket'; + +test('active version-pinned checkpoints survive time and stack removal; workers cannot list/delete', () => { + const stack = new Stack(new App(), 'Test'); + const storage = new ContinuationBucket(stack, 'Continuation'); + const worker = new iam.Role(stack, 'Worker', { assumedBy: new iam.ServicePrincipal('lambda.amazonaws.com') }); + storage.grantWorker(worker); + const template = Template.fromStack(stack); + template.hasResource('AWS::S3::Bucket', { + DeletionPolicy: 'Retain', + Properties: { + VersioningConfiguration: { Status: 'Enabled' }, + LifecycleConfiguration: { + Rules: [{ + Id: 'abandoned-multipart-uploads', + Status: 'Enabled', + AbortIncompleteMultipartUpload: { DaysAfterInitiation: 1 }, + }], + }, + }, + }); + template.hasResourceProperties('AWS::IAM::Policy', { + PolicyDocument: { + Statement: [Match.objectLike({ + Action: ['s3:GetObject', 's3:GetObjectVersion', 's3:PutObject'], + Resource: Match.anyValue(), + })], + }, + }); + const policies = JSON.stringify(template.findResources('AWS::IAM::Policy')); + expect(policies).toContain('continuations/${aws:PrincipalTag/task_id}/*'); + expect(policies).not.toContain('s3:List'); + expect(policies).not.toContain('s3:Delete'); +}); diff --git a/cdk/test/constructs/lambda-microvm-compute.test.ts b/cdk/test/constructs/lambda-microvm-compute.test.ts index 4d92364eb..7617ce303 100644 --- a/cdk/test/constructs/lambda-microvm-compute.test.ts +++ b/cdk/test/constructs/lambda-microvm-compute.test.ts @@ -70,6 +70,7 @@ interface BuildOptions { readonly withImage?: boolean; readonly imageEnvironmentVariables?: Record; readonly artifactSha256?: string | null; + readonly managedImageVersion?: string; readonly externalImageIdentifier?: string; readonly externalImageVersion?: string; readonly withSessionRole?: boolean; @@ -147,6 +148,7 @@ function instantiate(options: BuildOptions = {}): Omit { }), externalImageIdentifier: options.externalImageIdentifier, externalImageVersion: options.externalImageVersion, + managedImageVersion: options.managedImageVersion, imageEnvironmentVariables: options.imageEnvironmentVariables, minimumMemoryInMiB: options.minimumMemoryInMiB, }); @@ -169,6 +171,33 @@ describe('LambdaMicrovmCompute — image provisioned from a managed base image', template = built.template; }); + describe('explicit managed runtime version', () => { + let pinned: Built; + + beforeAll(() => { + pinned = build({ + withImage: true, withSessionRole: true, withRuntimeParity: true, managedImageVersion: '7.0', + }); + }); + + test('pins new workers without changing the managed image build or ownership', () => { + expect(pinned.construct.imageVersion).toBe('7.0'); + expect(pinned.template.findResources('AWS::Lambda::MicrovmImage')) + .toEqual(template.findResources('AWS::Lambda::MicrovmImage')); + }); + + // Constructor-validation cases deliberately have no successful template to cache. + test.each(['', 'latest', '0', '1.a'])('rejects invalid runtime version %p', managedImageVersion => { + expect(() => build({ withImage: true, managedImageVersion })) + .toThrow('microvm_managed_image_version'); + }); + + test('requires managed image inputs', () => { + expect(() => build({ managedImageVersion: '7.0' })) + .toThrow('microvm_managed_image_version'); + }); + }); + test('synthesizes exactly one MicroVM image from the artifact bucket object', () => { template.resourceCountIs('AWS::Lambda::MicrovmImage', 1); template.hasResourceProperties('AWS::Lambda::MicrovmImage', { @@ -207,18 +236,8 @@ describe('LambdaMicrovmCompute — image provisioned from a managed base image', ); test('builds an ARM64 image at the largest ACCEPTED BASELINE (8 GiB)', () => { - // 32768 was rejected live: "The requested memory size of 32768 MiB is not - // supported by base MicroVM image …al2023-1. Supported memory sizes in MiB - // are: [512, 1024, 2048, 4096, 8192]." Note this configures the BASELINE — - // the service scales vertically to a 32 GiB / 16 vCPU peak on its own, which - // is why nothing here asks for the peak. - // - // `ARM_64`, not `arm64`: the CDK L1 types Architecture as a plain string and - // documents no allowed values, and CloudFormation rejected the lowercase - // spelling at change-set early validation — "arm64 is not a valid enum value. - // Supported values: [ARM_64]" (ADR-021 P2-F2). The literal is spelled out here - // rather than imported from the construct so the test fails if the constant is - // "corrected" back to Docker's spelling. + // Assert the accepted baseline and API enum spelling independently of source + // constants; this test does not measure runtime memory or scaling. template.hasResourceProperties('AWS::Lambda::MicrovmImage', { CpuConfigurations: [{ Architecture: 'ARM_64' }], Resources: [{ MinimumMemoryInMiB: 8192 }], @@ -989,8 +1008,7 @@ describe('LambdaMicrovmCompute — image provisioned from a managed base image', expect(JSON.stringify(template.toJSON())).not.toContain('CreateMicrovmAuthToken'); }); - test('distinguishes clean P2 smoke evidence from remaining acceptance and P3 hooks', () => { - // Prior image evidence does not establish new-image or normal rollout acceptance. + test('describes configured-image verification and rollback without installation-specific history', () => { const warnings = built.construct.node.metadata.filter(m => m.type === 'aws:cdk:warning'); const message = warnings.map(w => String(w.data)).join('\n'); expect(JSON.stringify(built.construct.node.metadata)) @@ -998,20 +1016,14 @@ describe('LambdaMicrovmCompute — image provisioned from a managed base image', // The superseded id must be gone, not merely reworded — operators grep for it. expect(JSON.stringify(built.construct.node.metadata)) .not.toContain('abca:microvm-image-p1-not-runnable'); - expect(message).toContain('Clean P2 deployment'); - expect(message).toContain('2026-09-14 without manual IAM changes'); - expect(message).toContain('This does not verify a different image'); - expect(message).toContain('P2'); - // It must state what IS true now, or it reads as the old (wrong) claim — and - // the hook list here is what an operator compares against a failed build or a - // failed lifecycle transition, so all six have to be named. + expect(message).toContain('verify the configured image and coordinator together'); for (const hook of ['/ready', '/validate', '/run', '/terminate', '/suspend', '/resume']) { expect(message).toContain(hook); } - expect(message).toContain('supervisor integration is implemented'); - expect(message).toContain('P3 requires bootstrap bundle 1.8.0 and defaults new suspension off'); + expect(message).toContain('actual launched image version before allowing sleep'); expect(message).toContain('Nested deployments require bundle 1.9.0'); - expect(message).toContain('Normal automatic-suspension activation remains open'); + expect(message).toContain('explicit image version for rollback'); + expect(message).not.toContain('Image 6.0'); }); test('enables every hook the agent serves, and only those (rendered form)', () => { @@ -1223,12 +1235,8 @@ describe('LambdaMicrovmCompute — memory sizing', () => { expect(error).toBeDefined(); expect(error!.message).toContain('32768'); expect(error!.message).toContain('512, 1024, 2048, 4096, 8192'); - // The message must say BASELINE, or an operator reads the rejection as "this - // backend caps at 8 GiB" and moves a repo to ECS it did not need to. + // Distinguish baseline validation from runtime capacity. expect(error!.message).toContain('BASELINE'); - expect(error!.message).toContain('32 GiB'); - // ...and points at the backend that DOES have the SUSTAINED capacity. - expect(error!.message).toContain('compute_type=ecs'); }); test.each([0, 256, 6144, 16384, 8193])('rejects the unsupported baseline %i MiB', (mib) => { diff --git a/cdk/test/constructs/lambda-microvm-stack.test.ts b/cdk/test/constructs/lambda-microvm-stack.test.ts index 52ed743af..5db13e15c 100644 --- a/cdk/test/constructs/lambda-microvm-stack.test.ts +++ b/cdk/test/constructs/lambda-microvm-stack.test.ts @@ -168,3 +168,36 @@ test('rejects a deployment name that would exceed IAM role-name limits', () => { executionRole: createMicrovmExecutionRole(parent, 'ExecutionRole'), })).toThrow(/at most 64 characters/); }); + +test('allows overlapping service names while preserving parent role identity and bootstrap role names', () => { + const parent = new Stack(new App(), 'backgroundagent-dev', { env: ENV }); + const executionRole = createMicrovmExecutionRole( + new Construct(parent, 'LambdaMicrovmCompute'), 'ExecutionRole', + ); + const nested = new LambdaMicrovmStack(parent, 'Microvm', { + ...IMAGE_INPUTS[2][1], + vpc: new ec2.Vpc(parent, 'Vpc', { maxAzs: 2 }), + deploymentName: parent.stackName, + resourceNamePrefix: 'backgroundagent-dev-p3', + executionRole, + }); + const child = Template.fromStack(nested); + child.hasResourceProperties('AWS::Lambda::MicrovmImage', { Name: 'backgroundagent-dev-p3-abca-agent' }); + child.hasResourceProperties('AWS::Lambda::NetworkConnector', { Name: 'backgroundagent-dev-p3-microvm-egress' }); + child.hasResourceProperties('AWS::Lambda::NetworkConnector', { Name: 'backgroundagent-dev-p3-microvm-build-egress' }); + child.hasResourceProperties('AWS::Logs::LogGroup', { LogGroupName: '/aws/lambda-microvms/backgroundagent-dev-p3-abca-agent' }); + child.hasResourceProperties('AWS::IAM::Role', { RoleName: 'backgroundagent-dev-MicrovmBuildRole' }); + child.hasResourceProperties('AWS::IAM::Role', { RoleName: 'backgroundagent-dev-MicrovmConnectorRole' }); + expect(Object.keys(Template.fromStack(parent).findResources('AWS::IAM::Role'))) + .toContain('LambdaMicrovmComputeExecutionRoleAA0C4A0D'); +}); + +test.each(['', '-invalid', 'has/slash', 'a'.repeat(41)])('rejects an unsafe migration name prefix %p', resourceNamePrefix => { + const parent = new Stack(new App(), 'backgroundagent-dev', { env: ENV }); + expect(() => new LambdaMicrovmStack(parent, 'Microvm', { + vpc: new ec2.Vpc(parent, 'Vpc', { maxAzs: 2 }), + deploymentName: parent.stackName, + resourceNamePrefix, + executionRole: createMicrovmExecutionRole(parent, 'ExecutionRole'), + })).toThrow(/microvm_resource_name_prefix/); +}); diff --git a/cdk/test/constructs/microvm-continuation-manager.test.ts b/cdk/test/constructs/microvm-continuation-manager.test.ts new file mode 100644 index 000000000..a98688bc5 --- /dev/null +++ b/cdk/test/constructs/microvm-continuation-manager.test.ts @@ -0,0 +1,64 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import { App, Stack } from 'aws-cdk-lib'; +import { Template } from 'aws-cdk-lib/assertions'; +import * as dynamodb from 'aws-cdk-lib/aws-dynamodb'; +import { ContinuationBucket } from '../../src/constructs/continuation-bucket'; +import { MicrovmContinuationManager } from '../../src/constructs/microvm-continuation-manager'; +import { TaskApi } from '../../src/constructs/task-api'; + +test('scheduled recovery and decisions invoke retained versions of only the original coordinator', () => { + const stack = new Stack(new App(), 'Continuations'); + const table = (id: string) => new dynamodb.Table(stack, id, { + partitionKey: { name: 'task_id', type: dynamodb.AttributeType.STRING }, + }); + const taskTable = table('Tasks'); + const approvalsTable = table('Approvals'); + const userConcurrencyTable = table('Counters'); + const bucket = new ContinuationBucket(stack, 'Saved'); + const coordinator = 'arn:aws:lambda:us-west-2:123456789012:function:coordinator'; + const imageArn = 'arn:aws:lambda:us-west-2:123456789012:microvm-image:agent'; + new MicrovmContinuationManager(stack, 'Manager', { + taskTable, + approvalsTable, + userConcurrencyTable, + continuationBucket: bucket, + orchestratorFunctionArn: coordinator, + imageArn, + }); + const api = new TaskApi(stack, 'Api', { taskTable, taskEventsTable: table('Events'), taskApprovalsTable: approvalsTable }); + api.enableMicrovmContinuations(bucket.bucket.bucketName, coordinator, userConcurrencyTable); + const template = Template.fromStack(stack); + const policies = Object.entries(template.findResources('AWS::IAM::Policy')); + for (const name of ['ManagerReconcilerFn', 'ApproveTaskFn', 'DenyTaskFn']) { + const statements = policies.filter(([id]) => id.includes(name)) + .flatMap(([, policy]) => policy.Properties.PolicyDocument.Statement); + expect(statements).toContainEqual(expect.objectContaining({ + Action: 'lambda:InvokeFunction', Resource: `${coordinator}:*`, + })); + expect(statements.some(statement => JSON.stringify(statement.Action).includes('RunMicrovm'))).toBe(false); + } + const functions = Object.entries(template.findResources('AWS::Lambda::Function')); + for (const name of ['ManagerReconcilerFn', 'ApproveTaskFn', 'DenyTaskFn']) { + const fn = functions.find(([id]) => id.includes(name))![1]; + expect(fn.Properties.Environment.Variables.ORCHESTRATOR_FUNCTION_ARN).toBe(coordinator); + expect(fn.Properties.Environment.Variables.CONTINUATION_BUCKET_NAME).toBeDefined(); + } +}); diff --git a/cdk/test/handlers/get-pending.test.ts b/cdk/test/handlers/get-pending.test.ts index 66ee88f36..cb19b6908 100644 --- a/cdk/test/handlers/get-pending.test.ts +++ b/cdk/test/handlers/get-pending.test.ts @@ -293,7 +293,7 @@ describe('get-pending', () => { expect(body.data.pending[0].severity).toBe('medium'); }); - test('expires_at falls back to created_at when timeout is missing', async () => { + test('expires_at is null when no timeout is configured', async () => { setupPendingMocks([ { task_id: 't', @@ -308,7 +308,7 @@ describe('get-pending', () => { ]); const res = await handler(makeEvent()); const body = JSON.parse(res.body); - expect(body.data.pending[0].expires_at).toBe('2026-05-07T00:00:00Z'); + expect(body.data.pending[0].expires_at).toBeNull(); }); test('500 on DDB error after rate-limit passes', async () => { diff --git a/cdk/test/handlers/get-policies.test.ts b/cdk/test/handlers/get-policies.test.ts index 7b0157f01..e4f008960 100644 --- a/cdk/test/handlers/get-policies.test.ts +++ b/cdk/test/handlers/get-policies.test.ts @@ -249,7 +249,7 @@ describe('get-policies', () => { expect(hardRule.summary).toBeDefined(); }); - test('soft rules carry severity + approval_timeout_s', async () => { + test('built-in soft rules carry severity without an implicit approval deadline', async () => { mockSend.mockResolvedValue({}); mockLoadRepoConfig.mockResolvedValue(null); const res = await handler(makeEvent('soft%2Fshape')); @@ -258,6 +258,6 @@ describe('get-policies', () => { (r: { rule_id: string }) => r.rule_id === 'force_push_any', ); expect(soft.severity).toBe('medium'); - expect(soft.approval_timeout_s).toBe(300); + expect(soft.approval_timeout_s).toBeUndefined(); }); }); diff --git a/cdk/test/handlers/reconcile-microvm-continuations.test.ts b/cdk/test/handlers/reconcile-microvm-continuations.test.ts new file mode 100644 index 000000000..53c22b50d --- /dev/null +++ b/cdk/test/handlers/reconcile-microvm-continuations.test.ts @@ -0,0 +1,142 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +const mockSend = jest.fn(); +const mockStop = jest.fn(); +const mockPoll = jest.fn(); +const mockClose = jest.fn(); +const mockDelete = jest.fn(); +const mockRelease = jest.fn(); +const mockRetire = jest.fn(); +const mockDispatch = jest.fn(); +jest.mock('../../src/handlers/shared/ua', () => ({ makeDocClient: () => ({ send: mockSend }) })); +jest.mock('../../src/handlers/shared/strategies/lambda-microvm-strategy', () => ({ + LambdaMicrovmComputeStrategy: jest.fn(() => ({ stopSession: mockStop, pollSession: mockPoll })), + MICROVM_MAX_DURATION_SECONDS: 28800, +})); +jest.mock('../../src/handlers/shared/close-task-approvals', () => ({ closeTaskApprovals: (...args: unknown[]) => mockClose(...args) })); +jest.mock('../../src/handlers/shared/task-concurrency', () => ({ releaseTaskSlot: (...args: unknown[]) => mockRelease(...args) })); +jest.mock('../../src/handlers/shared/microvm-continuation-storage', () => ({ + deleteClosedTaskContinuations: (...args: unknown[]) => mockDelete(...args), +})); +jest.mock('../../src/handlers/shared/microvm-continuation-retirement', () => ({ + retireCheckpointedMicrovm: (...args: unknown[]) => mockRetire(...args), +})); +jest.mock('../../src/handlers/shared/microvm-continuation-dispatch', () => ({ + dispatchMicrovmContinuation: (...args: unknown[]) => mockDispatch(...args), +})); +jest.mock('../../src/handlers/shared/logger', () => ({ logger: { warn: jest.fn(), info: jest.fn() } })); + +import { handler, reconcileMicrovmContinuation } from '../../src/handlers/reconcile-microvm-continuations'; + +let task: any; +beforeEach(() => { + jest.resetAllMocks(); + task = { + task_id: 'task', + user_id: 'user', + status: 'AWAITING_APPROVAL', + awaiting_approval_request_id: 'request', + continuation: { state: 'PARKED' }, + continuation_launch: {}, + microvm_start: { clientToken: 'attempt', createdAt: new Date().toISOString() }, + }; + mockSend.mockResolvedValue({}); +}); + +test('a parked unanswered task is offered for admission without waking its retired worker', async () => { + await reconcileMicrovmContinuation(task); + expect(mockDispatch).toHaveBeenCalledWith('task', 'user', 'request', expect.any(Object)); + expect(mockStop).not.toHaveBeenCalled(); + expect(mockDelete).not.toHaveBeenCalled(); +}); + +test('a terminated source triggers verified retirement before dispatch', async () => { + task.continuation.state = 'READY'; + task.microvm_start.handle = { microvmId: 'old' }; + mockPoll.mockResolvedValue({ microvmState: 'TERMINATED' }); + await reconcileMicrovmContinuation(task); + expect(mockRetire).toHaveBeenCalledWith(expect.objectContaining({ force: true, handle: { microvmId: 'old' } })); + expect(mockRetire.mock.invocationCallOrder[0]).toBeLessThan(mockDispatch.mock.invocationCallOrder[0]); +}); + +test('unknown start keeps artifacts and capacity until its maximum service lifetime', async () => { + task.status = 'FAILED'; + mockSend.mockResolvedValue({ Item: task }); + await reconcileMicrovmContinuation(task); + expect(mockClose).toHaveBeenCalled(); + expect(mockRelease).not.toHaveBeenCalled(); + expect(mockDelete).not.toHaveBeenCalled(); + task.microvm_start.createdAt = new Date(Date.now() - 29_000_000).toISOString(); + await reconcileMicrovmContinuation(task); + expect(mockRelease).toHaveBeenCalledWith('task', 'user'); + expect(mockDelete).toHaveBeenCalled(); + const closed = mockSend.mock.calls.find(([command]) => command.constructor.name === 'UpdateCommand')![0].input; + expect(closed.ExpressionAttributeValues).toMatchObject({ ':closed': 'CLOSED', ':attempt': 'attempt' }); +}); + +test('unconfirmed termination retains artifacts and reservation', async () => { + task.status = 'CANCELLED'; + task.microvm_start.handle = { microvmId: 'vm' }; + mockSend.mockResolvedValue({ Item: task }); + mockPoll.mockResolvedValue({ microvmState: 'TERMINATING' }); + await reconcileMicrovmContinuation(task); + expect(mockStop).toHaveBeenCalledWith({ microvmId: 'vm' }, expect.any(Object)); + expect(mockRelease).not.toHaveBeenCalled(); + expect(mockDelete).not.toHaveBeenCalled(); +}); + +test('terminal scan rows cannot stop a task whose current record is active', async () => { + mockSend.mockResolvedValue({ Item: task }); + await reconcileMicrovmContinuation({ ...task, status: 'FAILED' }); + expect(mockStop).not.toHaveBeenCalled(); + expect(mockClose).not.toHaveBeenCalled(); +}); + +test('invalid unknown-start timestamp cannot be treated as proof of shutdown', async () => { + task.status = 'FAILED'; + task.microvm_start.createdAt = 'invalid'; + mockSend.mockResolvedValue({ Item: task }); + await expect(reconcileMicrovmContinuation(task)).rejects.toThrow('START_TIME_INVALID'); + expect(mockRelease).not.toHaveBeenCalled(); +}); + +test('persists the last completed batch when invocation time is low, then resumes from that key', async () => { + const rows = Array.from({ length: 6 }, (_, index) => ({ ...task, task_id: `task-${index}` })); + mockSend.mockImplementation(async command => { + if (command.constructor.name === 'GetCommand') return { Item: { cursor: { task_id: 'previous' } } }; + if (command.constructor.name === 'ScanCommand') return { Items: rows, LastEvaluatedKey: { task_id: 'page-end' } }; + return {}; + }); + const remaining = jest.fn().mockReturnValueOnce(100000).mockReturnValueOnce(100000).mockReturnValue(40000); + await handler({}, { getRemainingTimeInMillis: remaining }); + expect(mockDispatch).toHaveBeenCalledTimes(4); + expect(mockSend.mock.calls.find(([command]) => command.constructor.name === 'ScanCommand')![0].input.ExclusiveStartKey) + .toEqual({ task_id: 'previous' }); + expect(mockSend.mock.calls.at(-1)![0].input.ExpressionAttributeValues[':cursor']).toEqual({ task_id: 'task-3' }); +}); + +test('a failed row does not prevent later rows or clearing the cursor after a complete scan', async () => { + mockSend.mockImplementation(async command => command.constructor.name === 'ScanCommand' + ? { Items: [task, { ...task, task_id: 'next' }] } : {}); + mockDispatch.mockRejectedValueOnce(new Error('temporary')).mockResolvedValueOnce(true); + await handler({}, { getRemainingTimeInMillis: () => 100000 }); + expect(mockDispatch).toHaveBeenCalledTimes(2); + expect(mockSend.mock.calls.at(-1)![0].input.UpdateExpression).toBe('REMOVE #cursor'); +}); diff --git a/cdk/test/handlers/reconcile-stranded-tasks.test.ts b/cdk/test/handlers/reconcile-stranded-tasks.test.ts index 0e1c1f1b5..f841017c7 100644 --- a/cdk/test/handlers/reconcile-stranded-tasks.test.ts +++ b/cdk/test/handlers/reconcile-stranded-tasks.test.ts @@ -20,6 +20,10 @@ // --- Mocks --- const mockDdbSend = jest.fn(); const mockRelease = jest.fn(); +const mockCloseApprovals = jest.fn(); +jest.mock('../../src/handlers/shared/close-task-approvals', () => ({ + closeTaskApprovals: (...args: unknown[]) => mockCloseApprovals(...args), +})); jest.mock('../../src/handlers/shared/task-concurrency', () => ({ releaseTaskSlot: (...args: unknown[]) => mockRelease(...args), })); @@ -263,7 +267,7 @@ describe('reconcile-stranded-tasks', () => { // (``event_type`` == their own names) never reach it. This test // pins the extra ``agent_milestone`` / ``approval_stranded`` emit // that makes the stranded case visible on §11.3 widgets. - const ancient = new Date(Date.now() - 2 * 3600 * 1000).toISOString(); + const ancient = new Date(Date.now() - 10 * 3600 * 1000).toISOString(); primeResponses([ { Items: [] }, // SUBMITTED { Items: [] }, // HYDRATING diff --git a/cdk/test/handlers/shared/close-task-approvals.test.ts b/cdk/test/handlers/shared/close-task-approvals.test.ts new file mode 100644 index 000000000..f6cf11e55 --- /dev/null +++ b/cdk/test/handlers/shared/close-task-approvals.test.ts @@ -0,0 +1,72 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +const mockSend = jest.fn(); +jest.mock('../../../src/handlers/shared/ua', () => ({ makeDocClient: () => ({ send: mockSend }) })); +import { closeTaskApprovals } from '../../../src/handlers/shared/close-task-approvals'; + +beforeEach(() => { + jest.clearAllMocks(); + process.env.TASK_APPROVALS_TABLE_NAME = 'approvals'; +}); +afterEach(() => { delete process.env.TASK_APPROVALS_TABLE_NAME; }); + +test('a parked task keeps its unanswered requests and has no retention timer', async () => { + mockSend.mockResolvedValue({ Item: { user_id: 'user', status: 'AWAITING_APPROVAL' } }); + await closeTaskApprovals('task', 'user'); + expect(mockSend).toHaveBeenCalledTimes(1); +}); + +test('task closure cancels unanswered requests, preserves answers, and stamps history retention', async () => { + mockSend.mockResolvedValueOnce({ Item: { user_id: 'user', status: 'FAILED' } }) + .mockResolvedValueOnce({ + Items: [ + { user_id: 'user', request_id: 'pending', status: 'PENDING' }, + { user_id: 'user', request_id: 'approved', status: 'APPROVED' }, + { user_id: 'other', request_id: 'wrong-owner', status: 'PENDING' }, + ], + }).mockResolvedValue({}); + await closeTaskApprovals('task', 'user'); + const updates = mockSend.mock.calls.slice(2).map(([command]) => command.input); + expect(updates).toHaveLength(2); + expect(updates[0].ExpressionAttributeValues).toMatchObject({ + ':cancelled': 'CANCELLED', ':observed': 'PENDING', ':reason': 'Owning task is failed.', + }); + expect(updates[0].ExpressionAttributeValues[':ttl']).toBeGreaterThan(Date.now() / 1000 + 86400); + expect(updates[1].UpdateExpression).toBe('SET #ttl = if_not_exists(#ttl, :ttl)'); + expect(updates[1].ExpressionAttributeValues[':cancelled']).toBeUndefined(); +}); + +test('a mismatched task owner cannot close another user’s approvals', async () => { + mockSend.mockResolvedValue({ Item: { user_id: 'other', status: 'CANCELLED' } }); + await closeTaskApprovals('task', 'user'); + expect(mockSend).toHaveBeenCalledTimes(1); +}); + +test('an answer racing task closure keeps its decision and still receives retention', async () => { + mockSend.mockResolvedValueOnce({ Item: { user_id: 'user', status: 'CANCELLED' } }) + .mockResolvedValueOnce({ Items: [{ user_id: 'user', request_id: 'request', status: 'PENDING' }] }) + .mockRejectedValueOnce(Object.assign(new Error('answered'), { name: 'ConditionalCheckFailedException' })) + .mockResolvedValue({}); + await closeTaskApprovals('task', 'user'); + expect(mockSend.mock.calls[3][0].input).toMatchObject({ + UpdateExpression: 'SET #ttl = if_not_exists(#ttl, :ttl)', + ConditionExpression: 'user_id = :user AND #status <> :pending', + }); +}); diff --git a/cdk/test/handlers/shared/microvm-continuation-dispatch.test.ts b/cdk/test/handlers/shared/microvm-continuation-dispatch.test.ts new file mode 100644 index 000000000..217eaf739 --- /dev/null +++ b/cdk/test/handlers/shared/microvm-continuation-dispatch.test.ts @@ -0,0 +1,110 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +const mockSend = jest.fn(); +const mockInvoke = jest.fn(); +const mockAdmit = jest.fn(); +const mockWarn = jest.fn(); +jest.mock('../../../src/handlers/shared/ua', () => ({ + makeDocClient: () => ({ send: mockSend }), makeClient: () => ({ send: mockInvoke }), +})); +jest.mock('../../../src/handlers/shared/microvm-continuation-start', () => ({ + admitContinuation: (...args: unknown[]) => mockAdmit(...args), +})); +jest.mock('../../../src/handlers/shared/logger', () => ({ logger: { info: jest.fn(), warn: mockWarn } })); + +import { dispatchMicrovmContinuation } from '../../../src/handlers/shared/microvm-continuation-dispatch'; + +let task: any; +beforeEach(() => { + jest.clearAllMocks(); + process.env.CONTINUATION_BUCKET_NAME = 'saved'; + process.env.ORCHESTRATOR_FUNCTION_ARN = 'arn:aws:lambda:us-west-2:123456789012:function:coordinator:live'; + task = { + task_id: 'task', + user_id: 'user', + compute_type: 'lambda-microvm', + status: 'AWAITING_APPROVAL', + awaiting_approval_request_id: 'request', + continuation: { state: 'PARKED', attempt_id: 'new-worker' }, + continuation_launch: { orchestrator_version: '42' }, + }; + mockSend.mockImplementation(async () => ({ Item: structuredClone(task) })); + mockAdmit.mockImplementation(async () => ({ kind: 'ready', task: structuredClone(task) })); + mockInvoke.mockResolvedValue({ StatusCode: 202 }); +}); +afterEach(() => { + delete process.env.CONTINUATION_BUCKET_NAME; + delete process.env.ORCHESTRATOR_FUNCTION_ARN; +}); + +test('uses the original published version and identical durable identity on repeated dispatch', async () => { + expect(await dispatchMicrovmContinuation('task', 'user', 'request')).toBe(true); + expect(await dispatchMicrovmContinuation('task', 'user', 'request')).toBe(true); + const [first, second] = mockInvoke.mock.calls.map(([command]) => command.input); + expect(first).toEqual(second); + expect(first.FunctionName).toBe('arn:aws:lambda:us-west-2:123456789012:function:coordinator:42'); + expect(first.DurableExecutionName).toMatch(/^[a-f0-9]{64}$/); + expect(first.DurableExecutionName.length).toBeLessThanOrEqual(64); + expect(JSON.parse(first.Payload.toString())).toEqual({ + task_id: 'task', continuation_request_id: 'request', continuation_attempt_id: 'new-worker', + }); +}); + +test('does not invoke a worker while capacity is unavailable', async () => { + mockAdmit.mockResolvedValue({ kind: 'capacity' }); + expect(await dispatchMicrovmContinuation('task', 'user', 'request')).toBe(true); + expect(mockInvoke).not.toHaveBeenCalled(); +}); + +test('fenced source must not be resumed while retirement is finishing', async () => { + task.continuation.state = 'FENCED'; + expect(await dispatchMicrovmContinuation('task', 'user', 'request')).toBe(true); + expect(mockAdmit).not.toHaveBeenCalled(); + expect(mockInvoke).not.toHaveBeenCalled(); +}); + +test.each(['READY', 'CONSUMED'])('leaves ordinary %s worker handling to the wake path', async state => { + task.continuation.state = state; + expect(await dispatchMicrovmContinuation('task', 'user', 'request')).toBe(false); + expect(mockInvoke).not.toHaveBeenCalled(); +}); + +test('an invocation failure preserves the assigned token for the scheduled retry', async () => { + mockInvoke.mockRejectedValue(new Error('lost invocation response')); + await expect(dispatchMicrovmContinuation('task', 'user', 'request')).rejects.toThrow('lost invocation'); + expect(task.continuation.attempt_id).toBe('new-worker'); +}); + +test('reports bounded validation detail and operation identity for dispatch diagnosis', async () => { + const error = Object.assign(new Error('durableExecutionName must have length less than or equal to 64'), { + name: 'ValidationException', + }); + mockInvoke.mockRejectedValue(error); + await expect(dispatchMicrovmContinuation('task', 'user', 'request')).rejects.toBe(error); + expect(mockWarn).toHaveBeenCalledWith( + 'Saved task continuation dispatch needs reconciliation', + expect.objectContaining({ + operation: 'Invoke', + coordinator_version: '42', + durable_execution_name_length: 64, + validation_detail: error.message, + }), + ); +}); diff --git a/cdk/test/handlers/shared/microvm-continuation-retirement.test.ts b/cdk/test/handlers/shared/microvm-continuation-retirement.test.ts new file mode 100644 index 000000000..c06e8b772 --- /dev/null +++ b/cdk/test/handlers/shared/microvm-continuation-retirement.test.ts @@ -0,0 +1,195 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +const mockSend = jest.fn(); +const mockVerify = jest.fn(); +const mockStopSession = jest.fn(); +jest.mock('../../../src/handlers/shared/ua', () => ({ makeDocClient: () => ({ send: mockSend }) })); +jest.mock('../../../src/handlers/shared/microvm-continuation-storage', () => ({ + verifyContinuationCheckpoint: (...args: unknown[]) => mockVerify(...args), +})); +jest.mock('@aws-sdk/lib-dynamodb', () => ({ + GetCommand: jest.fn(input => ({ kind: 'get', input })), + TransactWriteCommand: jest.fn(input => ({ kind: 'transact', input })), +})); +jest.mock('../../../src/handlers/shared/logger', () => ({ logger: { warn: jest.fn() } })); + +import type { ComputeStrategy } from '../../../src/handlers/shared/compute-strategy'; +import { retireCheckpointedMicrovm } from '../../../src/handlers/shared/microvm-continuation-retirement'; + +const handle = { + strategyType: 'lambda-microvm' as const, + microvmId: 'microvm-one', + sessionId: 'microvm-one', + endpoint: 'https://example.invalid', +}; +const identity = { task_id: 'task', attempt_id: 'microvm-one', request_id: 'request', user_id: 'user', repo: 'owner/repo' }; +const record = { + version: 1, + state: 'READY', + identity, + manifest: { + kind: 'manifest', + key: `continuations/task/microvm-one/request/manifest/${'a'.repeat(64)}.json`, + sha256: 'a'.repeat(64), + version_id: 'version-1', + size_bytes: 100, + }, +}; +let task: Record; +let strategy: ComputeStrategy; +let transactions: any[]; +let leaseState: string; + +beforeEach(() => { + jest.clearAllMocks(); + process.env.CONTINUATION_BUCKET_NAME = 'continuations'; + task = { + task_id: 'task', + user_id: 'user', + repo: 'owner/repo', + status: 'AWAITING_APPROVAL', + session_id: handle.microvmId, + awaiting_approval_request_id: 'request', + continuation: structuredClone(record), + microvm_start: { clientToken: 'task' }, + concurrency_slot: { state: 'held', acquired_at: '2026-09-17T00:00:00Z' }, + }; + transactions = []; + leaseState = 'ACTIVE'; + mockVerify.mockResolvedValue(undefined); + mockSend.mockImplementation(async ({ kind, input }) => { + if (kind === 'get' && input.Key.task_id === 'worker-lease#task') { + return { + Item: { + lease_state: leaseState, + lease_attempt_id: 'task', + lease_microvm_id: 'microvm-one', + lease_user_id: 'user', + lease_repo: 'owner/repo', + }, + }; + } + if (kind === 'get') { + return input.Key.request_id + ? { Item: { user_id: 'user', status: 'PENDING', created_at: new Date(Date.now() - 7200000).toISOString() } } + : { Item: structuredClone(task) }; + } + transactions.push(input.TransactItems); + const values = input.TransactItems[0].Update.ExpressionAttributeValues; + task.continuation = structuredClone(values[':fenced'] ?? values[':parked']); + leaseState = task.continuation.state; + if (leaseState === 'PARKED') task.concurrency_slot = { ...task.concurrency_slot, state: 'released' }; + return {}; + }); + strategy = { + type: 'lambda-microvm', + startSession: jest.fn(), + suspendSession: jest.fn(), + resumeSession: jest.fn(), + stopSession: mockStopSession.mockResolvedValue({ outcome: 'requested' }), + pollSession: jest.fn().mockResolvedValue({ status: 'running', microvmState: 'TERMINATING' }), + }; +}); +afterEach(() => { delete process.env.CONTINUATION_BUCKET_NAME; }); + +function run(force = false) { + return retireCheckpointedMicrovm({ + taskId: 'task', userId: 'user', handle, strategy, sessionDeadlineMs: Date.now() + 28800000, force, + }); +} + +test('fences before termination, holds capacity until terminal observation, then parks once', async () => { + expect(await run()).toBe('stopping'); + expect(mockVerify).toHaveBeenCalledWith(record, expect.objectContaining({ abortSignal: expect.any(AbortSignal) })); + expect(transactions).toHaveLength(1); + expect(transactions[0][1].Update.ExpressionAttributeValues).toMatchObject({ + ':fenced': 'FENCED', ':active': 'ACTIVE', ':attempt': 'task', ':vm': 'microvm-one', + }); + expect(transactions[0][0].Update.ConditionExpression).toContain('continuation = :record'); + expect(task.status).toBe('AWAITING_APPROVAL'); + (strategy.pollSession as jest.Mock).mockResolvedValue({ status: 'completed', microvmState: 'TERMINATED' }); + expect(await run()).toBe('parked'); + expect(transactions).toHaveLength(2); + expect(transactions[1][2].Update.UpdateExpression).toContain('active_count = active_count - :one'); + expect(transactions[1][0].Update.ExpressionAttributeValues[':slot'].state).toBe('held'); + expect(task.concurrency_slot.state).toBe('released'); + expect(task.awaiting_approval_request_id).toBe('request'); + expect(await run()).toBe('parked'); + expect(transactions).toHaveLength(2); +}); + +test('a human answer winning the fence race leaves the original worker running', async () => { + mockSend.mockImplementation(async ({ kind, input }) => { + if (kind === 'get') { + return input.Key.request_id + ? { Item: { created_at: new Date(Date.now() - 7200000).toISOString() } } + : { Item: structuredClone(task) }; + } + task.status = 'RUNNING'; + delete task.continuation; + throw new Error('fence condition lost'); + }); + expect(await run()).toBe('not-due'); + expect(mockStopSession).not.toHaveBeenCalled(); +}); + +test('missing durable objects cannot retire the only copy of the workspace', async () => { + mockVerify.mockRejectedValue(new Error('missing pinned workspace version')); + await expect(run()).rejects.toThrow('missing pinned workspace'); + expect(transactions).toHaveLength(0); + expect(mockStopSession).not.toHaveBeenCalled(); +}); + +test('sleep off retains the worker until its lifetime margin', async () => { + task.microvm_sleep_after_s = 0; + expect(await run()).toBe('not-due'); + expect(mockStopSession).not.toHaveBeenCalled(); + expect(await retireCheckpointedMicrovm({ + taskId: 'task', userId: 'user', handle, strategy, sessionDeadlineMs: Date.now() + 100000, + })).toBe('stopping'); +}); + +test('a stale coordinator never follows or stops a replacement worker', async () => { + task.session_id = 'microvm-new'; + expect(await run(true)).toBe('ownership-lost'); + expect(transactions).toHaveLength(0); + expect(mockStopSession).not.toHaveBeenCalled(); +}); + +test.each(['FENCED', 'PARKED'])('a worker-written %s label cannot replace coordinator lease authority', async state => { + task.continuation = { ...task.continuation, state, source_handle: handle }; + await expect(run()).rejects.toThrow('no coordinator authority'); + expect(mockStopSession).not.toHaveBeenCalled(); + expect(transactions).toHaveLength(0); +}); + +test('lost fence reply cannot accept a worker-written label while its readonly lease remains active', async () => { + const normal = mockSend.getMockImplementation()!; + mockSend.mockImplementation(async (command, options) => { + if (command.kind === 'transact') { + task.continuation = { ...task.continuation, state: 'FENCED', source_handle: handle }; + throw new Error('fence did not commit'); + } + return normal(command, options); + }); + expect(await run()).toBe('not-due'); + expect(mockStopSession).not.toHaveBeenCalled(); + expect(leaseState).toBe('ACTIVE'); +}); diff --git a/cdk/test/handlers/shared/microvm-continuation-runner.test.ts b/cdk/test/handlers/shared/microvm-continuation-runner.test.ts new file mode 100644 index 000000000..7f2fecb50 --- /dev/null +++ b/cdk/test/handlers/shared/microvm-continuation-runner.test.ts @@ -0,0 +1,186 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +const mockLoad = jest.fn(); +const mockLaunch = jest.fn(); +const mockAdmit = jest.fn(); +const mockStart = jest.fn(); +const mockPoll = jest.fn(); +const mockWorkerPoll = jest.fn(); +const mockStop = jest.fn(); +const mockFinalize = jest.fn(); +const mockDelete = jest.fn(); +const mockSend = jest.fn(); +jest.mock('../../../src/handlers/shared/ua', () => ({ makeDocClient: () => ({ send: mockSend }) })); +jest.mock('../../../src/handlers/shared/orchestrator', () => ({ + loadTask: (...args: unknown[]) => mockLoad(...args), + emitTaskEvent: jest.fn(), + envelopeFor: () => ({ correlation: {}, log: { error: jest.fn() } }), + finalizeTask: (...args: unknown[]) => mockFinalize(...args), +})); +jest.mock('../../../src/handlers/shared/compute-strategy', () => ({ + resolveComputeStrategy: () => ({ + startSession: mockStart, pollSession: mockWorkerPoll, + }), +})); +jest.mock('../../../src/handlers/shared/microvm-continuation-storage', () => ({ + loadContinuationLaunch: (...args: unknown[]) => mockLaunch(...args), +})); +jest.mock('../../../src/handlers/shared/microvm-continuation-start', () => ({ + admitContinuation: (...args: unknown[]) => mockAdmit(...args), +})); +jest.mock('../../../src/handlers/shared/microvm-task-poll', () => ({ + pollMicrovmTask: (...args: unknown[]) => mockPoll(...args), +})); +jest.mock('../../../src/handlers/shared/microvm-supervisor', () => ({ + stopMicrovmWithDiagnostics: (...args: unknown[]) => mockStop(...args), +})); +jest.mock('../../../src/handlers/shared/strategies/lambda-microvm-strategy', () => ({ + deleteMicrovmPayload: (...args: unknown[]) => mockDelete(...args), +})); + +import type { DurableContext } from '@aws/durable-execution-sdk-js'; +import type { ComputeStrategy } from '../../../src/handlers/shared/compute-strategy'; +import { + continuationWaitStrategy, failContinuationAttempt, pollContinuationRestore, runMicrovmContinuation, +} from '../../../src/handlers/shared/microvm-continuation-runner'; + +const event = { task_id: 'task', continuation_request_id: 'request', continuation_attempt_id: 'attempt-new' }; +const handle = { strategyType: 'lambda-microvm' as const, microvmId: 'vm-new', sessionId: 'vm-new', endpoint: 'https://worker.invalid' }; +let task: any; +const context = { + step: async (_name: string, action: () => Promise) => action(), + waitForCondition: async (_name: string, check: (state: any) => Promise, options: any) => { + let state = options.initialState; + for (let i = 0; i < 3; i++) { + state = await check(state); + if (!options.waitStrategy(state).shouldContinue) return state; + } + throw new Error('fixture wait did not complete'); + }, +} as unknown as DurableContext; + +beforeEach(() => { + jest.clearAllMocks(); + process.env.AWS_LAMBDA_FUNCTION_VERSION = '42'; + task = { + task_id: 'task', + user_id: 'user', + status: 'AWAITING_APPROVAL', + compute_type: 'lambda-microvm', + awaiting_approval_request_id: 'request', + continuation_launch: { orchestrator_version: '42' }, + continuation: { + state: 'STARTING', + attempt_id: 'attempt-new', + started_at: new Date().toISOString(), + source_handle: { imageArn: 'arn:image:original', imageVersion: '7.0' }, + }, + }; + mockLoad.mockImplementation(async () => structuredClone(task)); + mockLaunch.mockResolvedValue({ + orchestrator_version: '42', + payload: { task_id: 'task', user_id: 'user', message: 'original instructions' }, + blueprint: { compute_type: 'lambda-microvm' }, + }); + mockAdmit.mockImplementation(async () => ({ kind: 'ready', task: structuredClone(task) })); + mockStart.mockImplementation(async () => { + task.session_id = 'vm-new'; + task.compute_metadata = { microvmId: 'vm-new' }; + task.microvm_start = { clientToken: 'attempt-new', handle }; + task.continuation.state = 'CONSUMED'; + task.status = 'RUNNING'; + return handle; + }); + mockPoll.mockImplementation(async () => { + task.status = 'COMPLETED'; + return { attempts: 1, lastStatus: 'COMPLETED' }; + }); + mockWorkerPoll.mockResolvedValue({ status: 'running', microvmState: 'RUNNING' }); + mockSend.mockImplementation(async () => { task.status = 'FAILED'; return {}; }); +}); +afterEach(() => { delete process.env.AWS_LAMBDA_FUNCTION_VERSION; }); + +test('runs saved inputs on the pinned original image with a fresh worker lifetime', async () => { + await runMicrovmContinuation(event, context); + expect(mockStart).toHaveBeenCalledWith(expect.objectContaining({ + microvmImage: { imageArn: 'arn:image:original', imageVersion: '7.0' }, + payload: { + task_id: 'task', + user_id: 'user', + message: 'original instructions', + attempt_id: 'attempt-new', + task_started_at: task.continuation.started_at, + }, + })); + expect(mockFinalize).toHaveBeenCalledTimes(1); + expect(mockDelete).toHaveBeenCalledWith('task', 'attempt-new'); + expect(mockStop).toHaveBeenCalledWith(expect.objectContaining({ handle })); +}); + +test('a new parked approval ends this execution without finalizing the task', async () => { + mockPoll.mockResolvedValue({ attempts: 1, microvmParked: true }); + await runMicrovmContinuation(event, context); + expect(mockFinalize).not.toHaveBeenCalled(); + expect(mockStop).not.toHaveBeenCalled(); + expect(mockDelete).toHaveBeenCalledWith('task', 'attempt-new'); +}); + +test('a stale continuation event cannot launch or finalize a different assigned worker', async () => { + task.continuation.attempt_id = 'another'; + await runMicrovmContinuation(event, context); + expect(mockStart).not.toHaveBeenCalled(); + expect(mockFinalize).not.toHaveBeenCalled(); +}); + +test('a different coordinator version fails before launching a new worker', async () => { + process.env.AWS_LAMBDA_FUNCTION_VERSION = '43'; + await runMicrovmContinuation(event, context); + expect(mockStart).not.toHaveBeenCalled(); + const transaction = mockSend.mock.calls[0][0].input.TransactItems; + expect(transaction[0].Update.ExpressionAttributeValues[':detail']).toContain('VERSION_CHANGED'); + expect(transaction[1].Update.ExpressionAttributeValues[':fenced']).toBe('FENCED'); +}); + +test('restoration uses its own deadline and never issues /resume', async () => { + task.session_id = 'vm-new'; + task.compute_metadata = { microvmId: 'vm-new' }; + task.continuation.state = 'RESTORING'; + const resume = jest.fn(); + const strategy = { pollSession: mockWorkerPoll, resumeSession: resume } as unknown as ComputeStrategy; + const deadlineMs = Date.now() + 600_000; + const waiting = await pollContinuationRestore(event, 'user', handle, strategy, { deadlineMs }); + expect(waiting).toEqual({ deadlineMs, consecutivePollFailures: 0 }); + expect(resume).not.toHaveBeenCalled(); + const failed = await pollContinuationRestore(event, 'user', handle, strategy, { deadlineMs: Date.now() - 1 }); + expect(failed.failure).toContain('RESTORE_TIMEOUT'); +}); + +test('failed recovery fences the current attempt even after a new checkpoint replaced the old record', async () => { + task.microvm_start = { clientToken: 'attempt-new', handle }; + task.continuation = { state: 'READY', identity: { attempt_id: 'vm-new' } }; + await failContinuationAttempt(event, 'user', 'restore failed'); + expect(mockSend).toHaveBeenCalledTimes(1); + expect(mockSend.mock.calls[0][0].input.TransactItems[0].Update.ConditionExpression).toContain('microvm_start.clientToken = :attempt'); +}); + +test('retirement reconciliation delays do not trigger ordinary failure finalization', () => { + expect(continuationWaitStrategy({ attempts: 99, microvmRetiring: true, microvmRetirementError: 'Timeout' })) + .toEqual({ shouldContinue: true, delay: { seconds: 30 } }); +}); diff --git a/cdk/test/handlers/shared/microvm-continuation-start.test.ts b/cdk/test/handlers/shared/microvm-continuation-start.test.ts new file mode 100644 index 000000000..ad2a0893c --- /dev/null +++ b/cdk/test/handlers/shared/microvm-continuation-start.test.ts @@ -0,0 +1,145 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +const mockSend = jest.fn(); +jest.mock('../../../src/handlers/shared/ua', () => ({ makeDocClient: () => ({ send: mockSend }) })); +jest.mock('@aws-sdk/lib-dynamodb', () => ({ + GetCommand: jest.fn(input => ({ kind: 'get', input })), + TransactWriteCommand: jest.fn(input => ({ kind: 'transact', input })), +})); + +import { admitContinuation } from '../../../src/handlers/shared/microvm-continuation-start'; + +const identity = { task_id: 'task', user_id: 'user', repo: 'owner/repo', attempt_id: 'vm-old', request_id: 'request' }; +let task: Record; +let lease: Record; +let approval: Record; +let transactions: any[]; + +beforeEach(() => { + jest.clearAllMocks(); + task = { + task_id: 'task', + user_id: 'user', + repo: 'owner/repo', + status: 'AWAITING_APPROVAL', + session_id: 'vm-old', + awaiting_approval_request_id: 'request', + concurrency_slot: { state: 'released' }, + microvm_start: { clientToken: 'task' }, + continuation: { + version: 1, + state: 'PARKED', + identity, + source_handle: { microvmId: 'vm-old' }, + manifest: { + kind: 'manifest', + key: `continuations/task/vm-old/request/manifest/${'a'.repeat(64)}.json`, + sha256: 'a'.repeat(64), + version_id: 'v1', + size_bytes: 100, + }, + }, + }; + lease = { lease_state: 'PARKED', lease_attempt_id: 'task', lease_microvm_id: 'vm-old', lease_user_id: 'user', lease_repo: 'owner/repo' }; + approval = { user_id: 'user', status: 'APPROVED' }; + transactions = []; + mockSend.mockImplementation(async ({ kind, input }) => { + if (kind === 'get') { + return { + Item: structuredClone( + input.Key.request_id ? approval : input.Key.task_id.startsWith('worker-lease#') ? lease : task, + ), + }; + } + transactions.push(input.TransactItems); + const values = input.TransactItems[0].Update.ExpressionAttributeValues; + task.continuation = values[':starting']; + task.concurrency_slot = values[':slot']; + lease = input.TransactItems[2].Put.Item; + return {}; + }); +}); + +test('claims exactly one seat and worker token; a replay verifies the same readonly lease', async () => { + const first = await admitContinuation('task', 'user', 'request', 3); + const second = await admitContinuation('task', 'user', 'request', 3); + expect(first.kind).toBe('ready'); + expect(second.kind).toBe('ready'); + expect(transactions).toHaveLength(1); + expect(lease.lease_state).toBe('ACTIVE'); + expect(task.concurrency_slot.attempt_id).toBe(lease.lease_attempt_id); + expect(transactions[0][1].Update.ConditionExpression).toContain('active_count < :limit'); + expect(transactions[0][2].Put.ConditionExpression).toContain('lease_state = :parked'); + expect(transactions[0][3].ConditionCheck.ExpressionAttributeValues[':approved']).toBe('APPROVED'); +}); + +test.each(['PENDING', 'CANCELLED'])('does not allocate capacity for %s approval', async status => { + approval.status = status; + expect((await admitContinuation('task', 'user', 'request', 3)).kind).toBe(status === 'PENDING' ? 'waiting' : 'closed'); + expect(transactions).toHaveLength(0); +}); + +test('rejects worker-writable STARTING state without a matching active readonly lease', async () => { + task.continuation = { ...task.continuation, state: 'STARTING', attempt_id: 'new' }; + task.concurrency_slot = { state: 'held', attempt_id: 'new' }; + await expect(admitContinuation('task', 'user', 'request', 3)).rejects.toThrow('LEASE_INVALID'); + expect(transactions).toHaveLength(0); +}); + +test('recovers a transaction that committed but lost its reply without allocating again', async () => { + const normal = mockSend.getMockImplementation()!; + mockSend.mockImplementation(async (command, options) => { + const result = await normal(command, options); + if (command.kind === 'transact') throw new Error('reply lost'); + return result; + }); + expect((await admitContinuation('task', 'user', 'request', 3)).kind).toBe('ready'); + expect(transactions).toHaveLength(1); +}); + +test('capacity denial leaves the same saved request parked', async () => { + const normal = mockSend.getMockImplementation()!; + mockSend.mockImplementation(async (command, options) => { + if (command.kind === 'transact') { + throw Object.assign(new Error('capacity'), { + name: 'TransactionCanceledException', CancellationReasons: [{ Code: 'None' }, { Code: 'ConditionalCheckFailed' }], + }); + } + return normal(command, options); + }); + expect((await admitContinuation('task', 'user', 'request', 3)).kind).toBe('capacity'); + expect(task.continuation.state).toBe('PARKED'); + expect(lease.lease_state).toBe('PARKED'); +}); + +test('an explicit elapsed deadline is resolved atomically with replacement admission', async () => { + approval = { ...approval, status: 'PENDING', timeout_s: 30, created_at: new Date(Date.now() - 60_000).toISOString() }; + expect((await admitContinuation('task', 'user', 'request', 3)).kind).toBe('ready'); + expect(transactions[0][3].Update).toMatchObject({ + ExpressionAttributeValues: { ':timeout': 30, ':created': approval.created_at }, + }); + expect(transactions[0][3].Update.UpdateExpression).toContain('#status'); +}); + +test('timeout zero remains unanswered regardless of request age', async () => { + approval = { ...approval, status: 'PENDING', timeout_s: 0, created_at: '2020-01-01T00:00:00Z' }; + expect((await admitContinuation('task', 'user', 'request', 3)).kind).toBe('waiting'); + expect(transactions).toHaveLength(0); +}); diff --git a/cdk/test/handlers/shared/microvm-continuation-storage.test.ts b/cdk/test/handlers/shared/microvm-continuation-storage.test.ts new file mode 100644 index 000000000..1cd47252a --- /dev/null +++ b/cdk/test/handlers/shared/microvm-continuation-storage.test.ts @@ -0,0 +1,176 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import { createHash } from 'node:crypto'; +import { Readable } from 'node:stream'; + +const mockS3 = jest.fn(); +const mockDdb = jest.fn(); +jest.mock('../../../src/handlers/shared/ua', () => ({ + makeClient: () => ({ send: mockS3 }), + makeDocClient: () => ({ send: mockDdb }), +})); +import { + deleteClosedTaskContinuations, loadContinuationLaunch, saveContinuationLaunch, verifyContinuationCheckpoint, +} from '../../../src/handlers/shared/microvm-continuation-storage'; +import type { ContinuationRecord } from '../../../src/handlers/shared/microvm-continuation-types'; + +const digest = (bytes: Buffer) => createHash('sha256').update(bytes).digest('hex'); +const payload = { task_id: 'task', user_id: 'user', prompt: 'saved instruction' }; +const blueprint = { compute_type: 'lambda-microvm' as const }; +const inputs = { version: 1, task_id: 'task', user_id: 'user', payload, blueprint, orchestrator_version: '42' }; +function launch(value: unknown = inputs) { + const bytes = Buffer.from(JSON.stringify(value)); + const sha256 = digest(bytes); + return { + bytes, + receipt: { + version: 1, + key: `continuations/task/launch/${sha256}.json`, + sha256, + version_id: 'version-1', + size_bytes: bytes.length, + orchestrator_version: '42', + }, + }; +} +function object(bytes: Buffer, overrides: Record = {}) { + return { Body: Readable.from([bytes]), ContentLength: bytes.length, VersionId: 'version-1', ...overrides }; +} +beforeEach(() => { + mockS3.mockReset(); + mockDdb.mockReset(); + process.env.CONTINUATION_BUCKET_NAME = 'bucket'; + process.env.AWS_LAMBDA_FUNCTION_VERSION = '42'; +}); +afterEach(() => { + delete process.env.CONTINUATION_BUCKET_NAME; + delete process.env.AWS_LAMBDA_FUNCTION_VERSION; +}); + +test('loads the exact published launch version and rejects changed bytes', async () => { + const saved = launch(); + mockS3.mockResolvedValueOnce(object(saved.bytes)); + expect(await loadContinuationLaunch('task', 'user', saved.receipt)).toEqual(inputs); + expect(mockS3.mock.calls[0][0].input.VersionId).toBe('version-1'); + mockS3.mockResolvedValueOnce(object(Buffer.from('x'.repeat(saved.bytes.length)))); + await expect(loadContinuationLaunch('task', 'user', saved.receipt)).rejects.toThrow('checksum'); +}); + +test.each([ + { VersionId: 'null' }, { VersionId: 'different' }, { ContentLength: 0 }, { ContentLength: 100_000_000 }, +])('rejects incomplete or unpinned object metadata %j and closes the stream', async overrides => { + const saved = launch(); + const response = object(saved.bytes, overrides); + mockS3.mockResolvedValue(response); + await expect(loadContinuationLaunch('task', 'user', saved.receipt)).rejects.toThrow('STORAGE_INVALID'); + expect(response.Body.destroyed).toBe(true); +}); + +test('stops a response that sends more bytes than declared', async () => { + const saved = launch(); + const response = object(Buffer.concat([saved.bytes, Buffer.from('extra')]), { ContentLength: saved.bytes.length }); + mockS3.mockResolvedValue(response); + await expect(loadContinuationLaunch('task', 'user', saved.receipt)).rejects.toThrow('exceeded'); + expect(response.Body.destroyed).toBe(true); +}); + +test('aborting checkpoint verification closes a stalled response body', async () => { + const controller = new AbortController(); + const body = new Readable({ read() { /* Deliberately stalled transport. */ } }); + mockS3.mockImplementation(async () => { + setImmediate(() => controller.abort()); + return { Body: body, ContentLength: 10, VersionId: 'v1' }; + }); + const record = { manifest: { key: 'manifest', version_id: 'v1' } } as ContinuationRecord; + await expect(verifyContinuationCheckpoint(record, { abortSignal: controller.signal })).rejects.toThrow('TIMEOUT'); + expect(body.destroyed).toBe(true); +}); + +test.each([null, { ...inputs, user_id: 'other' }])('rejects saved input identity %j', async value => { + const saved = launch(value); + mockS3.mockResolvedValue(object(saved.bytes)); + await expect(loadContinuationLaunch('task', 'user', saved.receipt)).rejects.toThrow('INPUT_INVALID'); +}); + +test('recovers a lost Put reply only after exact versioned readback', async () => { + let published: Buffer; + mockS3.mockImplementation(async command => { + if (command.constructor.name === 'PutObjectCommand') { + published = command.input.Body; + throw new Error('reply lost'); + } + return object(published); + }); + mockDdb.mockResolvedValue({}); + await saveContinuationLaunch('task', 'user', payload, blueprint); + expect(mockS3.mock.calls[0][0].input).toMatchObject({ IfNoneMatch: '*', ServerSideEncryption: 'AES256' }); + const committed = mockDdb.mock.calls[0][0].input; + expect(committed.ExpressionAttributeValues[':receipt']).toMatchObject({ version_id: 'version-1', orchestrator_version: '42' }); + expect(committed.UpdateExpression).toContain('REMOVE #ttl'); +}); + +test('verifies both archive versions and checksums before permitting retirement', async () => { + const identity = { task_id: 'task', user_id: 'user', repo: 'owner/repo', attempt_id: 'vm', request_id: 'request' }; + const prefix = 'continuations/task/vm/request/'; + const conversation = { key: `${prefix}${'a'.repeat(64)}.json`, sha256: 'a'.repeat(64), size_bytes: 100, version_id: 'conversation-v' }; + const workspace = { key: `${prefix}workspace/${'b'.repeat(64)}.tar`, sha256: 'b'.repeat(64), size_bytes: 512, version_id: 'workspace-v' }; + const bytes = Buffer.from(JSON.stringify({ version: 1, identity, conversation, workspace })); + const record: ContinuationRecord = { + version: 1, + state: 'READY', + identity, + manifest: { kind: 'manifest', key: 'manifest', sha256: digest(bytes), size_bytes: bytes.length, version_id: 'version-1' }, + }; + mockS3.mockResolvedValueOnce(object(bytes)); + for (const receipt of [conversation, workspace]) { + mockS3.mockResolvedValueOnce({ + VersionId: receipt.version_id, + ContentLength: receipt.size_bytes, + ChecksumSHA256: Buffer.from(receipt.sha256, 'hex').toString('base64'), + }); + } + await verifyContinuationCheckpoint(record); + expect(mockS3.mock.calls.slice(1).map(([command]) => command.input.VersionId)).toEqual(['conversation-v', 'workspace-v']); + mockS3.mockResolvedValueOnce(object(bytes)).mockResolvedValueOnce({ VersionId: 'wrong' }); + await expect(verifyContinuationCheckpoint(record)).rejects.toThrow('version, length or checksum'); +}); + +test('cleanup preserves pending data and removes every closed-task version before clearing its pointer', async () => { + const options = { abortSignal: AbortSignal.timeout(1000) }; + mockDdb.mockResolvedValueOnce({ Item: { user_id: 'user', status: 'AWAITING_APPROVAL' } }); + await deleteClosedTaskContinuations('task', 'user', options); + expect(mockS3).not.toHaveBeenCalled(); + mockDdb.mockResolvedValueOnce({ Item: { user_id: 'user', status: 'CANCELLED' } }).mockResolvedValue({}); + mockS3.mockResolvedValueOnce({ + Versions: [{ Key: 'continuations/task/object', VersionId: 'v1' }], + DeleteMarkers: [{ Key: 'continuations/task/object', VersionId: 'v2' }], + }).mockResolvedValueOnce({}).mockResolvedValueOnce({}); + await deleteClosedTaskContinuations('task', 'user', options); + expect(mockS3.mock.calls[1][0].input.Delete.Objects).toHaveLength(2); + expect(mockDdb.mock.calls.at(-1)![0].input.UpdateExpression).toContain('REMOVE continuation, continuation_launch'); +}); + +test('partial delete failure retains the cleanup marker for a later retry', async () => { + mockDdb.mockResolvedValue({ Item: { user_id: 'user', status: 'FAILED' } }); + mockS3.mockResolvedValueOnce({ Versions: [{ Key: 'continuations/task/object', VersionId: 'v1' }] }) + .mockResolvedValueOnce({ Errors: [{ Code: 'AccessDenied' }] }); + await expect(deleteClosedTaskContinuations('task', 'user', {})).rejects.toThrow('CLEANUP_FAILED'); + expect(mockDdb).toHaveBeenCalledTimes(1); +}); diff --git a/cdk/test/handlers/shared/microvm-worker-lease.test.ts b/cdk/test/handlers/shared/microvm-worker-lease.test.ts new file mode 100644 index 000000000..100d009f0 --- /dev/null +++ b/cdk/test/handlers/shared/microvm-worker-lease.test.ts @@ -0,0 +1,83 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +const mockSend = jest.fn(); +jest.mock('../../../src/handlers/shared/ua', () => ({ makeDocClient: () => ({ send: mockSend }) })); +jest.mock('@aws-sdk/lib-dynamodb', () => ({ + GetCommand: jest.fn(input => ({ kind: 'get', input })), + TransactWriteCommand: jest.fn(input => ({ kind: 'transact', input })), +})); + +import { ensureWorkerLease, leaseHandleUpdate } from '../../../src/handlers/shared/microvm-worker-lease'; + +const input = { taskId: 'task', userId: 'user', repo: 'owner/repo', attemptId: 'attempt-1', requestHash: 'hash' }; +const lease = { + task_id: 'worker-lease#task', + lease_attempt_id: 'attempt-1', + lease_state: 'ACTIVE', + lease_user_id: 'user', + lease_repo: 'owner/repo', +}; + +beforeEach(() => { + jest.clearAllMocks(); + process.env.CONTINUATION_BUCKET_NAME = 'continuations'; + mockSend.mockResolvedValue({}); +}); +afterEach(() => { delete process.env.CONTINUATION_BUCKET_NAME; }); + +test('creates authority only alongside the same task owner and immutable start receipt', async () => { + await ensureWorkerLease(input); + const transaction = mockSend.mock.calls[0][0].input.TransactItems; + expect(transaction[0].ConditionCheck.ExpressionAttributeValues).toMatchObject({ + ':user': 'user', ':attempt': 'attempt-1', ':hash': 'hash', + }); + expect(transaction[1].Put).toMatchObject({ + Item: lease, ConditionExpression: 'attribute_not_exists(task_id)', + }); + // Reserved rows cannot enter ordinary task status/user indexes. + expect(transaction[1].Put.Item).not.toHaveProperty('status'); + expect(transaction[1].Put.Item).not.toHaveProperty('user_id'); +}); + +test('lost successful acknowledgement accepts the same active lease with a registered handle', async () => { + mockSend.mockRejectedValueOnce(new Error('reply lost')).mockResolvedValueOnce({ + Item: { ...lease, lease_microvm_id: 'microvm-one' }, + }); + await expect(ensureWorkerLease(input)).resolves.toBeUndefined(); + expect(mockSend.mock.calls[1][0].input.ConsistentRead).toBe(true); +}); + +test.each([ + { lease_state: 'FENCED' }, { lease_state: 'PARKED' }, { lease_attempt_id: 'new-attempt' }, + { lease_user_id: 'other' }, { lease_repo: 'other/repo' }, +])('a replay cannot replace or reactivate changed authority: %j', async changed => { + mockSend.mockRejectedValueOnce(new Error('conditional failure')).mockResolvedValueOnce({ + Item: { ...lease, ...changed }, + }); + await expect(ensureWorkerLease(input)).rejects.toThrow('conditional failure'); + expect(mockSend).toHaveBeenCalledTimes(2); +}); + +test('handle registration never changes the lease state', () => { + const update = leaseHandleUpdate('task', 'attempt-1', 'microvm-one').Update; + expect(update.Key).toEqual({ task_id: 'worker-lease#task' }); + expect(update.UpdateExpression).toBe('SET lease_microvm_id = :id'); + expect(update.ConditionExpression).toContain('lease_attempt_id = :attempt'); +}); diff --git a/cdk/test/handlers/shared/payload-bootstrap.test.ts b/cdk/test/handlers/shared/payload-bootstrap.test.ts index 091c58f07..c14f5ec8d 100644 --- a/cdk/test/handlers/shared/payload-bootstrap.test.ts +++ b/cdk/test/handlers/shared/payload-bootstrap.test.ts @@ -111,6 +111,31 @@ test('changed instructions cannot overwrite an existing payload or launch refere expect(objects.get('task-1/payload.json')).toBe(before); }); +test('replacement attempts have independent immutable objects and scoped cleanup', async () => { + await preparePayloadReference(input); + const original = objects.get('task-1/payload.json'); + const replacement = { + ...input, attemptId: 'replacement-2', payload: { ...input.payload, attempt_id: 'replacement-2' }, + }; + const reference = await preparePayloadReference(replacement); + expect(reference.attempt_id).toBe('replacement-2'); + expect(objects.has('task-1/replacement-2/launch.json')).toBe(true); + expect(await preparePayloadReference(replacement)).toEqual(reference); + await expect(preparePayloadReference({ + ...replacement, payload: { ...replacement.payload, prompt: 'changed' }, + })).rejects.toThrow('CONFLICT'); + expect(objects.get('task-1/payload.json')).toBe(original); + await deletePayloadReference(input.bucket, input.taskId, replacement.attemptId); + expect(objects.has('task-1/replacement-2/payload.json')).toBe(false); + expect(objects.get('task-1/payload.json')).toBe(original); +}); + +test.each(['../other', 'different-attempt'])('rejects invalid or mismatched attempt %s before S3', async attemptId => { + await expect(preparePayloadReference({ ...input, attemptId, payload: { ...input.payload, attempt_id: 'replacement-2' } })) + .rejects.toThrow('worker attempt'); + expect(mockSend).not.toHaveBeenCalled(); +}); + test('orphaned payload after a crash cannot be overwritten with changed instructions', async () => { await preparePayloadReference(input); objects.delete('task-1/launch.json'); diff --git a/cdk/test/handlers/shared/strategies/lambda-microvm-strategy.test.ts b/cdk/test/handlers/shared/strategies/lambda-microvm-strategy.test.ts index a3d76da3c..792bb3790 100644 --- a/cdk/test/handlers/shared/strategies/lambda-microvm-strategy.test.ts +++ b/cdk/test/handlers/shared/strategies/lambda-microvm-strategy.test.ts @@ -662,7 +662,11 @@ describe('LambdaMicrovmComputeStrategy', () => { describe('deleteMicrovmPayload', () => { test('delegates cleanup of payload and private launch record', async () => { await deleteMicrovmPayload('TASK001'); - expect(mockDelete).toHaveBeenCalledWith(PAYLOAD_BUCKET, 'TASK001'); + expect(mockDelete).toHaveBeenCalledWith(PAYLOAD_BUCKET, 'TASK001', undefined); + }); + test('scopes replacement cleanup to its own attempt', async () => { + await deleteMicrovmPayload('TASK001', 'attempt-two'); + expect(mockDelete).toHaveBeenCalledWith(PAYLOAD_BUCKET, 'TASK001', 'attempt-two'); }); }); @@ -1106,7 +1110,7 @@ describe('LambdaMicrovmComputeStrategy image-identifier validation', () => { }); describe('buildMicrovmPlatformConfig — the MicroVM substitute for a deploy-time env block', () => { - /** A fully-populated orchestrator environment: all thirteen keys present. */ + /** A fully-populated orchestrator environment. */ const FULL_ENV: NodeJS.ProcessEnv = { TASK_TABLE_NAME: 'tasks', TASK_EVENTS_TABLE_NAME: 'events', @@ -1115,8 +1119,11 @@ describe('buildMicrovmPlatformConfig — the MicroVM substitute for a deploy-tim LOG_GROUP_NAME: '/aws/abca/application', ARTIFACTS_BUCKET_NAME: 'artifacts-bucket', TRACE_ARTIFACTS_BUCKET_NAME: 'trace-bucket', + CONTINUATION_BUCKET_NAME: 'continuation-bucket', GITHUB_TOKEN_SECRET_ARN: 'arn:aws:secretsmanager:us-east-1:123456789012:secret:gh-AbCdEf', LINEAR_OAUTH_SECRET_ARN: 'arn:aws:secretsmanager:us-east-1:123456789012:secret:bgagent-linear-oauth-acme-XyZ', + LINEAR_VAULT_ENABLED: 'true', + LINEAR_WORKLOAD_IDENTITY_NAME: 'abca_linear_oauth', JIRA_OAUTH_SECRET_ARN: 'arn:aws:secretsmanager:us-east-1:123456789012:secret:bgagent-jira-oauth-cloud1-XyZ', AGENT_SESSION_ROLE_ARN: 'arn:aws:iam::123456789012:role/SessionRole', AWS_SDK_UA_APP_ID: 'uksb-wt64nei4u6#backgroundagent-dev', @@ -1156,14 +1163,17 @@ describe('buildMicrovmPlatformConfig — the MicroVM substitute for a deploy-tim 'log_group_name', 'artifacts_bucket_name', 'trace_artifacts_bucket_name', + 'continuation_bucket_name', 'github_token_secret_arn', 'linear_oauth_secret_arn', + 'linear_vault_enabled', + 'linear_workload_identity_name', 'jira_oauth_secret_arn', 'agent_session_role_arn', 'aws_sdk_ua_app_id', 'anthropic_default_haiku_model', ]); - expect(MICROVM_PLATFORM_CONFIG_KEYS).toHaveLength(13); + expect(MICROVM_PLATFORM_CONFIG_KEYS).toHaveLength(16); // snake_case on the wire, matching every other key in the /run envelope. for (const key of MICROVM_PLATFORM_CONFIG_KEYS) { expect(key).toMatch(/^[a-z][a-z0-9_]*$/); @@ -1184,11 +1194,14 @@ describe('buildMicrovmPlatformConfig — the MicroVM substitute for a deploy-tim } }); - test('emits all thirteen keys, in declaration order, from a full environment', () => { + test('emits every configured key in declaration order from a full environment', () => { const config = buildMicrovmPlatformConfig(FULL_ENV); expect(Object.keys(config)).toEqual([...MICROVM_PLATFORM_CONFIG_KEYS]); expect(config.task_table_name).toBe('tasks'); expect(config.nudges_table_name).toBe('nudges'); + expect(config.continuation_bucket_name).toBe('continuation-bucket'); + expect(config.linear_vault_enabled).toBe('true'); + expect(config.linear_workload_identity_name).toBe('abca_linear_oauth'); expect(config.agent_session_role_arn).toBe('arn:aws:iam::123456789012:role/SessionRole'); expect(config.anthropic_default_haiku_model).toBe('us.anthropic.claude-haiku-4-5-20251001-v1:0'); }); @@ -1324,6 +1337,8 @@ describe('buildMicrovmPlatformConfig — the MicroVM substitute for a deploy-tim for (const optional of [ 'TASK_APPROVALS_TABLE_NAME', 'NUDGES_TABLE_NAME', 'LOG_GROUP_NAME', 'ARTIFACTS_BUCKET_NAME', 'TRACE_ARTIFACTS_BUCKET_NAME', 'LINEAR_OAUTH_SECRET_ARN', + 'LINEAR_VAULT_ENABLED', 'LINEAR_WORKLOAD_IDENTITY_NAME', + 'CONTINUATION_BUCKET_NAME', 'JIRA_OAUTH_SECRET_ARN', 'AWS_SDK_UA_APP_ID', 'ANTHROPIC_DEFAULT_HAIKU_MODEL', ]) { delete env[optional]; diff --git a/cdk/test/handlers/shared/task-concurrency.test.ts b/cdk/test/handlers/shared/task-concurrency.test.ts index 2663db132..f687c2f2f 100644 --- a/cdk/test/handlers/shared/task-concurrency.test.ts +++ b/cdk/test/handlers/shared/task-concurrency.test.ts @@ -40,6 +40,31 @@ function cancelled(index: number) { } beforeEach(() => mockSend.mockReset()); +test.each([undefined, 'ACTIVE', 'FENCED', 'PARKED'])( + 'a terminal saved task keeps capacity until shutdown is confirmed (lease %s)', async leaseState => { + mockSend.mockResolvedValueOnce({ + Item: { ...base, status: 'FAILED', concurrency_slot: held, continuation_launch: {}, microvm_start: { clientToken: 'attempt' } }, + }).mockResolvedValueOnce({ + Item: { lease_user_id: 'user', lease_attempt_id: 'attempt', lease_state: leaseState }, + }); + expect(await releaseTaskSlot('task', 'user')).toBe(false); + expect(mockSend).toHaveBeenCalledTimes(2); + }, +); + +test('confirmed shutdown is checked atomically when returning a saved task’s seat', async () => { + mockSend.mockResolvedValueOnce({ + Item: { ...base, status: 'FAILED', concurrency_slot: held, continuation_launch: {}, microvm_start: { clientToken: 'attempt' } }, + }).mockResolvedValueOnce({ + Item: { lease_user_id: 'user', lease_attempt_id: 'attempt', lease_state: 'CLOSED' }, + }).mockResolvedValueOnce({}); + expect(await releaseTaskSlot('task', 'user')).toBe(true); + expect(mockSend.mock.calls[2][0].input.TransactItems[2].ConditionCheck).toMatchObject({ + Key: { task_id: 'worker-lease#task' }, + ExpressionAttributeValues: { ':user': 'user', ':attempt': 'attempt', ':closed': 'CLOSED' }, + }); +}); + test('admission atomically ties a counter increment to an owned SUBMITTED task', async () => { mockSend.mockResolvedValueOnce({ Item: base }).mockResolvedValueOnce({}); expect(await acquireTaskSlot('task', 'user', 3)).toBe(true); diff --git a/cdk/test/stacks/agent.test.ts b/cdk/test/stacks/agent.test.ts index b51d5a366..193ce5f07 100644 --- a/cdk/test/stacks/agent.test.ts +++ b/cdk/test/stacks/agent.test.ts @@ -1132,6 +1132,7 @@ describe('AgentStack with the Lambda MicroVMs substrate gate (--context compute_ microvm_base_image_arn: BASE_IMAGE_ARN, microvm_base_image_version: '1', microvm_artifact_sha256: 'a'.repeat(64), + microvm_managed_image_version: '7.0', }, }); const stack = new AgentStack(app, 'TestAgentStackMicrovm', { @@ -1202,11 +1203,12 @@ describe('AgentStack with the Lambda MicroVMs substrate gate (--context compute_ 'MICROVM_EGRESS_CONNECTOR_ARNS', 'MICROVM_EXECUTION_ROLE_ARN', 'MICROVM_IMAGE_IDENTIFIER', + 'MICROVM_IMAGE_VERSION', 'MICROVM_INGRESS_CONNECTOR_ARNS', 'MICROVM_PAYLOAD_BUCKET', ]); - // Image version is deliberately unpinned. - expect(env.MICROVM_IMAGE_VERSION).toBeUndefined(); + // A runtime pin leaves the child-owned managed image in place. + expect(env.MICROVM_IMAGE_VERSION).toBe('7.0'); expect(env.MICROVM_APPROVAL_SUSPEND_ENABLED).toBe('false'); // Ingress is NOT empty and NOT omitted: RunMicrovm attaches a PUBLIC // HTTP_INGRESS connector (with a public endpoint) when the field is absent, @@ -1862,20 +1864,104 @@ describe('AgentStack Linear identity vault gate (#809)', () => { expect([...workloadNames]).toEqual(['abca_linear_oauth_LinearVaultWritersStack']); }); - test('MicroVM + vault is REFUSED by name, not left to the resource counter', () => { - // Preserve the current #857 gate until bundled feature-matrix and deployment - // verification support removing it. The original 505-resource count is stale; - // current measurements and a cycle-free nested-stack prototype are documented - // in docs/verification/645-p3-readiness-review.md. Once the combination is - // supported, replace this refusal assertion with real parity/size assertions. - const app = new App({ - context: { enableLinearIdentityVault: true, compute_type: 'lambda-microvm' }, - }); - expect(() => Template.fromStack( - new AgentStack(app, 'LinearVaultMicrovmStack', { + describe('MicroVM vault configuration and execution-role grants (#857)', () => { + let template: Template; + let childTemplates: Template[]; + const workloadName = 'abca_linear_oauth_LinearVaultMicrovmStack'; + + beforeAll(() => { + const app = new App({ + context: { + enableLinearIdentityVault: true, + compute_type: 'lambda-microvm', + microvm_base_image_arn: 'arn:aws:lambda:us-east-1:aws:microvm-image:al2023-1', + microvm_base_image_version: '1', + microvm_artifact_sha256: 'a'.repeat(64), + }, + }); + const stack = new AgentStack(app, 'LinearVaultMicrovmStack', { env: { account: '123456789012', region: 'us-east-1' }, - }), - )).toThrow(/enableLinearIdentityVault cannot be combined with compute_type=lambda-microvm/); + }); + template = Template.fromStack(stack); + childTemplates = stack.node.findAll() + .filter((child): child is NestedStack => NestedStack.isNestedStack(child)) + .map(child => Template.fromStack(child)); + }); + + test('delivers the same workload identity through the coordinator and runtime configuration', () => { + template.hasResourceProperties('Custom::LinearWorkloadIdentity', { + WorkloadName: workloadName, + }); + template.hasOutput('LinearVaultWorkloadName', { Value: workloadName }); + const orchestrator = Object.entries(template.findResources('AWS::Lambda::Function')) + .find(([id]) => id.includes('TaskOrchestratorOrchestratorFn'))!; + expect(orchestrator[1].Properties.Environment.Variables).toMatchObject({ + LINEAR_VAULT_ENABLED: 'true', + LINEAR_WORKLOAD_IDENTITY_NAME: workloadName, + }); + template.hasResourceProperties('AWS::BedrockAgentCore::Runtime', { + EnvironmentVariables: Match.objectLike({ + LINEAR_VAULT_ENABLED: 'true', + LINEAR_WORKLOAD_IDENTITY_NAME: workloadName, + }), + }); + }); + + test('grants the MicroVM role both token operations on the Linear vault resources', () => { + const roles = template.findResources('AWS::IAM::Role'); + const executionRole = Object.keys(roles) + .find(id => id.startsWith('LambdaMicrovmComputeExecutionRole'))!; + expect(executionRole).toBeDefined(); + const policies = Object.values(template.findResources('AWS::IAM::Policy')) + .filter(policy => policy.Properties.Roles?.some( + (role: { Ref?: string }) => role.Ref === executionRole, + )); + const statements = policies.flatMap(policy => policy.Properties.PolicyDocument.Statement); + const arn = (suffix: string) => ({ + 'Fn::Join': ['', [ + 'arn:', { Ref: 'AWS::Partition' }, `:bedrock-agentcore:us-east-1:123456789012:${suffix}`, + ]], + }); + const directory = arn('workload-identity-directory/default'); + const identity = arn(`workload-identity-directory/default/workload-identity/${workloadName}`); + for (const action of [ + 'bedrock-agentcore:GetWorkloadAccessTokenForUserId', + 'bedrock-agentcore:GetResourceOauth2Token', + ]) { + const grants = statements.filter(statement => + [statement.Action].flat().includes(action)); + expect(grants).toHaveLength(1); + expect(grants[0].Effect).toBe('Allow'); + expect(grants[0].Resource).toEqual(expect.arrayContaining([directory, identity])); + expect([grants[0].Resource].flat()).not.toContain('*'); + } + const oauthGrant = statements.find(statement => + [statement.Action].flat().includes('bedrock-agentcore:GetResourceOauth2Token')); + expect(oauthGrant.Resource).toEqual(expect.arrayContaining([ + arn('token-vault/default'), + arn('token-vault/default/oauth2credentialprovider/bgagent-linear-oauth-*'), + ])); + const policyJson = JSON.stringify(policies); + expect(policyJson).toContain('bedrock-agentcore-identity!default/oauth2/bgagent-linear-oauth-*'); + expect(policyJson).not.toContain('bedrock-agentcore:CreateWorkloadIdentity'); + expect(policyJson).not.toContain('bedrock-agentcore:DeleteWorkloadIdentity'); + }); + + test('keeps task vault configuration and credentials out of the reusable image', () => { + const images = childTemplates.flatMap(child => + Object.values(child.findResources('AWS::Lambda::MicrovmImage'))); + expect(images).toHaveLength(1); + const imageJson = JSON.stringify(images); + for (const field of [ + 'LINEAR_VAULT_ENABLED', + 'LINEAR_WORKLOAD_IDENTITY_NAME', + 'access_token', + 'refresh_token', + workloadName, + ]) { + expect(imageJson).not.toContain(field); + } + }); }); test('the source graph names no Linear-minting handler that is unwired', () => { @@ -1973,17 +2059,19 @@ describe('AgentStack CloudFormation resource budget 500 with cushion', () => { }, ]; const CELLS = CONFIGURATIONS.flatMap(configuration => - [false, true].map(enableToolGateway => ({ ...configuration, enableToolGateway })), - ); + [false, true].flatMap(enableToolGateway => + [false, true].map(enableLinearIdentityVault => ({ + ...configuration, enableToolGateway, enableLinearIdentityVault, + })))); describe.each(CELLS)( - '$name enableToolGateway=$enableToolGateway', - ({ context, enableToolGateway }) => { + '$name enableToolGateway=$enableToolGateway enableLinearIdentityVault=$enableLinearIdentityVault', + ({ context, enableToolGateway, enableLinearIdentityVault }) => { let template: Template; let templates: Template[]; beforeAll(() => { - const app = new App({ context: { ...context, enableToolGateway } }); + const app = new App({ context: { ...context, enableToolGateway, enableLinearIdentityVault } }); const stack = new AgentStack(app, 'BudgetStack', { env: { account: '123456789012', region: 'us-east-1' }, }); diff --git a/cdk/test/stacks/microvm-managed-image-nag.test.ts b/cdk/test/stacks/microvm-managed-image-nag.test.ts index 9d0944f79..127575248 100644 --- a/cdk/test/stacks/microvm-managed-image-nag.test.ts +++ b/cdk/test/stacks/microvm-managed-image-nag.test.ts @@ -19,10 +19,14 @@ import { ManagedPolicy, PolicyStatement, Role } from 'aws-cdk-lib/aws-iam'; import { AGENTCORE_AZS_CONTEXT_KEY } from '../../src/constructs/agentcore-azs'; +import { OrchestrationReconciler } from '../../src/constructs/orchestration-reconciler'; import { TaskOrchestrator } from '../../src/constructs/task-orchestrator'; import { buildApp } from '../../src/main'; -describe.each([false, true])('managed MicroVM image security checks (extra wildcard: %s)', extraWildcard => { +const configurations = [false, true].flatMap(enableLinearIdentityVault => + [false, true].map(extraWildcard => ({ enableLinearIdentityVault, extraWildcard }))); + +describe.each(configurations)('managed MicroVM security checks (vault=$enableLinearIdentityVault, extra wildcard=$extraWildcard)', ({ enableLinearIdentityVault, extraWildcard }) => { let errors: string[]; beforeAll(async () => { @@ -32,6 +36,8 @@ describe.each([false, true])('managed MicroVM image security checks (extra wildc appProps: { context: { compute_type: 'lambda-microvm', + enableLinearIdentityVault, + enableToolGateway: true, microvm_base_image_arn: 'arn:aws:lambda:us-west-2:aws:microvm-image:al2023-1', microvm_base_image_version: '1', microvm_artifact_sha256: 'a'.repeat(64), @@ -40,8 +46,27 @@ describe.each([false, true])('managed MicroVM image security checks (extra wildc }, }); const stack = app.node.findChild('backgroundagent-dev'); + const orchestrator = stack.node.findChild('TaskOrchestrator') as TaskOrchestrator; + if (enableLinearIdentityVault) { + const reconciler = stack.node.findChild('OrchestrationReconciler') as OrchestrationReconciler; + // Bundled policies can overflow at different boundaries from unit-test + // policies. Exercise the documented exceptions on both owning roles. + for (const role of [orchestrator.fn.role, reconciler.fn.role]) { + new ManagedPolicy(role as Role, 'OverflowPolicyVaultProbe', { + statements: [ + new PolicyStatement({ + actions: ['bedrock-agentcore:GetResourceOauth2Token'], + resources: ['arn:aws:bedrock-agentcore:us-west-2:123456789012:token-vault/default/oauth2credentialprovider/bgagent-linear-oauth-*'], + }), + new PolicyStatement({ + actions: ['secretsmanager:GetSecretValue'], + resources: ['arn:aws:secretsmanager:us-west-2:123456789012:secret:bedrock-agentcore-identity!default/oauth2/bgagent-linear-oauth-*'], + }), + ], + }); + } + } if (extraWildcard) { - const orchestrator = stack.node.findChild('TaskOrchestrator') as TaskOrchestrator; // Model a future overflow document without relying on the policy // splitter placing an extra statement in a particular document. new ManagedPolicy(orchestrator.fn.role as Role, 'OverflowPolicy999', { diff --git a/cli/src/commands/pending.ts b/cli/src/commands/pending.ts index 2f8266fed..dce05ea87 100644 --- a/cli/src/commands/pending.ts +++ b/cli/src/commands/pending.ts @@ -65,7 +65,7 @@ function renderText(pending: readonly PendingApprovalSummary[]): void { } console.log(` preview: ${p.tool_input_preview}`); console.log(` created: ${p.created_at}`); - console.log(` expires: ${p.expires_at} (timeout_s=${p.timeout_s})`); + console.log(` expires: ${p.expires_at ?? 'no automatic expiry'}`); console.log( ` approve: bgagent approve ${p.task_id} ${p.request_id}`, ); diff --git a/cli/src/commands/submit.ts b/cli/src/commands/submit.ts index 947b29922..0c41f1710 100644 --- a/cli/src/commands/submit.ts +++ b/cli/src/commands/submit.ts @@ -88,7 +88,7 @@ export function makeSubmitCommand(): Command { .option( '--approval-timeout ', `Cedar HITL per-task default approval timeout (${APPROVAL_TIMEOUT_S_MIN}-${APPROVAL_TIMEOUT_S_MAX}s). ` - + 'Overrides the platform default of 300s. Per-rule @approval_timeout_s still min-wins at gate-firing.', + + 'Default: no automatic expiry (0). Explicit per-rule timeouts still apply.', parseInt, ) .option( @@ -149,11 +149,11 @@ export function makeSubmitCommand(): Command { if ( isNaN(opts.approvalTimeout) || !Number.isInteger(opts.approvalTimeout) - || opts.approvalTimeout < APPROVAL_TIMEOUT_S_MIN + || (opts.approvalTimeout !== 0 && opts.approvalTimeout < APPROVAL_TIMEOUT_S_MIN) || opts.approvalTimeout > APPROVAL_TIMEOUT_S_MAX ) { throw new CliError( - `--approval-timeout must be an integer between ${APPROVAL_TIMEOUT_S_MIN} ` + `--approval-timeout must be 0 (no expiry), or an integer between ${APPROVAL_TIMEOUT_S_MIN} ` + `and ${APPROVAL_TIMEOUT_S_MAX} seconds.`, ); } diff --git a/cli/src/commands/watch.ts b/cli/src/commands/watch.ts index 6f5612f03..caaaa4c09 100644 --- a/cli/src/commands/watch.ts +++ b/cli/src/commands/watch.ts @@ -157,7 +157,7 @@ function renderMilestoneSuffix(meta: Record): string { if (meta.request_id != null) parts.push(`request_id=${String(meta.request_id)}`); if (meta.scope != null) parts.push(`scope=${String(meta.scope)}`); if (meta.status != null) parts.push(`status=${String(meta.status)}`); - if (meta.timeout_s != null) parts.push(`timeout=${String(meta.timeout_s)}s`); + if (meta.timeout_s != null) parts.push(meta.timeout_s === 0 ? 'no automatic expiry' : `timeout=${String(meta.timeout_s)}s`); if (meta.reason != null) parts.push(`reason=${String(meta.reason)}`); const ruleIds = meta.matching_rule_ids; if (Array.isArray(ruleIds) && ruleIds.length > 0) { diff --git a/cli/src/types.ts b/cli/src/types.ts index 8b033768a..50d215273 100644 --- a/cli/src/types.ts +++ b/cli/src/types.ts @@ -475,7 +475,8 @@ export interface CreateTaskRequest { */ readonly trace?: boolean; /** Cedar HITL per-task default approval timeout (design §7.3 step 5). - * Valid range ``[APPROVAL_TIMEOUT_S_MIN, APPROVAL_TIMEOUT_S_MAX]``. */ + * Zero retains unanswered requests; positive values use + * ``[APPROVAL_TIMEOUT_S_MIN, APPROVAL_TIMEOUT_S_MAX]``. */ readonly approval_timeout_s?: number; /** Cedar HITL pre-approval allowlist seeded at task start (§7.3 step 4). * Each entry must be a valid ``ApprovalScope``. */ @@ -759,7 +760,7 @@ export interface PendingApprovalSummary { readonly reason: string; readonly created_at: string; readonly timeout_s: number; - readonly expires_at: string; + readonly expires_at: string | null; /** Cedar rule ids that matched this request — shown by * ``bgagent pending`` so users can see which rule fired without * spelunking TaskEventsTable. */ @@ -805,7 +806,7 @@ export const APPROVAL_TIMEOUT_S_MIN = 30; export const APPROVAL_TIMEOUT_S_MAX = 3600; /** Default approval_timeout_s when the submit payload omits it. */ -export const APPROVAL_TIMEOUT_S_DEFAULT = 300; +export const APPROVAL_TIMEOUT_S_DEFAULT = 0; /** Per-task MicroVM sleep delay bounds; zero disables automatic sleep. */ export const MICROVM_SLEEP_AFTER_S_MIN = 0; diff --git a/contracts/constants.json b/contracts/constants.json index e25cb8bd2..3f13607a3 100644 --- a/contracts/constants.json +++ b/contracts/constants.json @@ -7,7 +7,7 @@ "approval_timeout_s": { "min": 30, "max": 3600, - "default": 300 + "default": 0 }, "microvm_sleep_after_s": { "min": 0, @@ -45,8 +45,11 @@ "log_group_name": "LOG_GROUP_NAME", "artifacts_bucket_name": "ARTIFACTS_BUCKET_NAME", "trace_artifacts_bucket_name": "TRACE_ARTIFACTS_BUCKET_NAME", + "continuation_bucket_name": "CONTINUATION_BUCKET_NAME", "github_token_secret_arn": "GITHUB_TOKEN_SECRET_ARN", "linear_oauth_secret_arn": "LINEAR_OAUTH_SECRET_ARN", + "linear_vault_enabled": "LINEAR_VAULT_ENABLED", + "linear_workload_identity_name": "LINEAR_WORKLOAD_IDENTITY_NAME", "jira_oauth_secret_arn": "JIRA_OAUTH_SECRET_ARN", "agent_session_role_arn": "AGENT_SESSION_ROLE_ARN", "aws_sdk_ua_app_id": "AWS_SDK_UA_APP_ID", @@ -88,6 +91,17 @@ "hook_port": 8080, "maximum_duration_seconds": 28800 }, + "microvm_continuation": { + "version": 1, + "lease_key_prefix": "worker-lease#", + "object_key_prefix": "continuations/", + "max_manifest_bytes": 2097152, + "max_workspace_bytes": 1073741824, + "max_conversation_bytes": 16777216, + "park_after_seconds": 3600, + "retirement_margin_seconds": 300, + "verified_sdk_version": "0.2.110" + }, "linear_vault": { "_why": "The vault caches a grant keyed by the WHOLE token request, customParameters included. A one-token divergence between any two of these copies makes every resolve a cache miss, which post-#812 is reported as consent-required and can latch a healthy workspace revoked.", "scopes": [ diff --git a/contracts/constants.md b/contracts/constants.md index 8fa28deaa..fb211fc02 100644 --- a/contracts/constants.md +++ b/contracts/constants.md @@ -50,7 +50,7 @@ JSON at TypeScript compile time via `resolveJsonModule`. "approval_timeout_s": { "min": 30, "max": 3600, - "default": 300 + "default": 0 }, "max_budget_usd": { "min": 0.01, @@ -116,14 +116,20 @@ JSON at TypeScript compile time via `resolveJsonModule`. - **`approval_gate_cap.default`** — value applied when a blueprint omits the field. 50 is the design-decision default (see `docs/design/CEDAR_HITL_GATES.md` decision #13). -- **`approval_timeout_s.min`** — floor for `approval_timeout_s` (§6 - decision #6). 30 seconds — below this, humans cannot realistically - respond to an approval prompt. -- **`approval_timeout_s.max`** — absolute ceiling for `approval_timeout_s` - before the `maxLifetime - 300` clip is applied (§7.3). 3600 seconds - (1 hour). +- **`approval_timeout_s.min`** — minimum positive explicit timeout: 30 seconds. + Zero is separately accepted and means no automatic deadline. +- **`approval_timeout_s.max`** — maximum positive explicit timeout: 3600 seconds + (1 hour). MicroVM continuation keeps this human deadline separate from a + worker's service lifetime. - **`approval_timeout_s.default`** — value applied when the submit payload - omits `approval_timeout_s`. 300 seconds (5 minutes) per §6 decision #6. + omits `approval_timeout_s`: zero, or no automatic expiry. Pending rows have + no DynamoDB TTL; task closure starts retention cleanup. Positive rule + annotations still apply, with the shortest positive deadline winning. +- **`microvm_continuation`** — the versioned checkpoint contract. It pins object + prefixes, maximum archive sizes and the SDK version verified for conversation + and exact budget recovery. A saved approval can release its worker after one + hour when sleep is enabled, or five minutes before the worker's eight-hour + limit when sleep is disabled. The request remains open in both cases. - **`microvm_sleep_after_s`** — per-task delay before sleeping during a pending human approval: whole seconds from 0 to 3600, default 600 (10 minutes). Zero disables sleep. Task creation persists the resolved @@ -151,8 +157,8 @@ JSON at TypeScript compile time via `resolveJsonModule`. platform env arrives through an authenticated v2 manifest and task payload instead; the values land in `os.environ`, which makes an unrecognised key an env-injection attempt. The consumer (`agent/src/server.py`) therefore **rejects** any `platform_config` - carrying a key that is not in this map. Values are non-secret identifiers - (table/bucket names, secret ARNs, role ARNs) only. + carrying a key that is not in this map. Values are non-secret configuration: + resource identifiers and the Linear vault enabled flag/workload name. - **`microvm_platform_config.required`** — the subset without which a task cannot run (task + event tables, GitHub secret ARN, session role ARN). A `/run` hook whose `platform_config` misses or blanks any of these is rejected with HTTP 400. diff --git a/docs/decisions/ADR-016-pluggable-identity-and-auth.md b/docs/decisions/ADR-016-pluggable-identity-and-auth.md index 757f2bcc6..5bf28b7d4 100644 --- a/docs/decisions/ADR-016-pluggable-identity-and-auth.md +++ b/docs/decisions/ADR-016-pluggable-identity-and-auth.md @@ -160,7 +160,7 @@ Per-user `McpCredential` selection requires the Gateway to know *which task-user | P6 | **Trusted task-user identity propagation** (prerequisite for per-user MCP on the general plane, P5): specify + validate a user-scoped inbound identity the Gateway authorizer trusts, replacing the M2M JWT for per-user credential selection. | **Blocks per-user `McpCredential`.** Until done, MCP credentials are workspace-scoped at best. | | P7 | Jira + Slack `ChannelCredential` (same shape as P1); GitHub `GithubOauth2` behind a flag, retire the shared PAT; OBO `act`-claim delegation feeding #237. | Flag-gated; per-surface. | -**Substrate exception — `lambda-microvm` (2026-09-02):** P1's vault cannot be enabled on the MicroVM substrate. The two together synthesize 505 resources against CloudFormation's hard 500-resource limit (MicroVM alone 496, the vault alone 488), so `AgentStack` refuses the combination at synth, naming both context flags. The MicroVM wiring itself is complete — `platform_config` carries the workload name and the guest execution role holds the mint grant — so this is a capacity limit, not a design gap, and it lifts as soon as a subsystem moves into a nested stack. +**Lambda MicroVMs:** The vault uses the guest's compute execution role, with its workload name delivered in authenticated `platform_config`. The former resource-count refusal was removed under #857 after stack reductions and nesting; backend/image-mode tests now include vault and gateway combinations, and check each template's resource and size budgets. See the [Linear setup guide](../guides/LINEAR_SETUP_GUIDE.md#using-the-vault-with-lambda-microvms). **Substrate independence (verified 2026-07-21, both proven live):** the vault path works on any compute. AgentCore Runtime injects the Workload Access Token as the `WorkloadAccessToken` header; ECS/Fargate/Lambda bootstrap it via `GetWorkloadAccessTokenForJWT(workloadName, userToken=)` against a **standalone** (non-service-linked) workload identity, then call `GetResourceOauth2Token`. Runtime-managed (service-linked) workload identities cannot self-vend, so the ECS path needs a manually-created workload identity. The runtime execution role today has `GetWorkloadAccessToken*` but **not** `GetResourceOauth2Token` — P1 adds it, plus `GetSecretValue` scoped to that surface's providers (`bedrock-agentcore-identity!default/oauth2/*`; see the implementation notes below for why this is narrower than the wildcard first anticipated here). diff --git a/docs/decisions/ADR-021-lambda-microvms-compute-backend.md b/docs/decisions/ADR-021-lambda-microvms-compute-backend.md index ea42ce55e..476a9a803 100644 --- a/docs/decisions/ADR-021-lambda-microvms-compute-backend.md +++ b/docs/decisions/ADR-021-lambda-microvms-compute-backend.md @@ -1,439 +1,220 @@ # ADR-021: AWS Lambda MicroVMs as a third ComputeStrategy backend -> **Implementation status (2026-09-17):** P1 ([#689](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/689)) and P2 ([#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733)) are merged. The takeover has a [clean deployment](../verification/645-p2-clean-deployment-20260913.md), [live repository workflow evidence](../verification/645-p2-live-task-20260914.md), and integrated P3 guest hooks, credentials, durable intent and approval/supervisor handling. The normal deployment uses agent image **7.0**, coordinator live alias **10** and bootstrap **1.9.0**. The [approval UX update](../verification/645-p3-approval-ux-20260917.md) deploys atomic cancellation, channel notification support and guest cancellation handling. The [final image tests](../verification/645-p3-final-image-and-ecs-20260917.md), [repository workflow](../verification/645-p3-repository-path-20260917.md), and [MCP/network checks](../verification/645-p3-mcp-network-20260917.md) provide nine successful image 6.0 approval workflows, including real credential renewal after expiry. -> -> **P3 remains incomplete; automatic suspension is disabled.** Lifecycle responses now explicitly close their HTTP connections before freezing, with [transport evidence and a deployed fix](../verification/645-p3-connection-close-rollout-20260916.md). Historical service-side dispatch traces remain unavailable; passing workflows do not supply those traces. Remaining work includes normal activation/off-switch acceptance, approval UX work tracked in the [current implementation plan](../verification/645-p3-implementation-plan.md), and deployment of the requested [nested MicroVM stack](../verification/645-p3-nested-stack.md). Bootstrap **1.9.0** is installed; moving the existing flat deployment still requires a reviewed resource migration. Follow the [service-team feedback tracker](../verification/645-lambda-microvm-service-feedback.md) for open service questions. This ADR defines P1–P3, not a P4. +> **Implementation status (2026-09-18): P1 and P2 are merged; P3 implementation and normal deployment acceptance are complete.** P3 adds approval sleep/wake, retained requests, conversation/workspace recovery, replacement workers and nested infrastructure. See the [completed plan](../verification/645-p3-implementation-plan.md) and [normal acceptance record](../verification/645-p3-normal-closure-20260918.md). This ADR defines P1–P3; it does not define an official P4. **Status:** proposed **Date:** 2026-07-29 ## Context -ABCA selects a per-repo compute backend through the Blueprint's `compute_type` field (`cdk/src/handlers/shared/repo-config.ts`). Before this ADR, two backends existed, resolved by `resolveComputeStrategy` (`cdk/src/handlers/shared/compute-strategy.ts`) behind a uniform `ComputeStrategy` interface (`startSession` / `pollSession` / `stopSession`): +ABCA selects compute per repository through Blueprint `compute_type`. AgentCore Runtime is the default; ECS Fargate supports larger workloads. [#645](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/645) adds Lambda MicroVMs for explicit lifecycle control and reduced compute usage during human approval waits. -- **AgentCore Runtime** (`agentcore`, default) — managed Firecracker MicroVM per session. Invoked via `InvokeAgentRuntime`; liveness is inferred from agent heartbeats in DynamoDB plus the FastAPI `/ping` endpoint (`agent/src/server.py`) — the strategy's `pollSession` is a stub that always reports `running`. Constraints: 2 GB image limit, no substrate-level suspend API exposed to the orchestrator. -- **ECS on Fargate** (`ecs`) — Fargate task (ARM64; current build default 4 vCPU / 16 GiB, configurable up to 16 vCPU / 120 GiB) for repos that exceed AgentCore's limits ([#596](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/596)). Invoked via `RunTask` in batch mode (bypassing the HTTP server); liveness via `DescribeTasks`. No suspend — an idle task blocked on an approval wait burns full compute the whole time. - -**AWS Lambda MicroVMs** (launched 2026-06-22) is a serverless Firecracker sandbox primitive that AWS positions explicitly for AI coding agents: VM-level isolation, snapshot-based near-instant launch, **suspend/resume with full memory + disk state preserved** (compute charges stop while suspended), a dedicated JWE-authenticated HTTPS endpoint per instance, lifecycle hooks (`/run`, `/suspend`, `/resume`, `/terminate`), and up to 8 hours per session. It is **not** classic Lambda: the 15-minute function cap does not apply, and [COMPUTE.md](../design/COMPUTE.md)'s "Lambda: poor fit" verdict refers to functions, not MicroVMs. - -[#645](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/645) proposes adding Lambda MicroVMs as a third `ComputeStrategy`. The fit is strong but not free — pre-implementation review of the service's lifecycle model surfaced real design tensions this ADR resolves: +Lambda MicroVMs are managed Firecracker virtual machines, separate from ordinary Lambda functions. A MicroVM can preserve memory and disk while suspended and can live for eight hours, including suspended time. The function service’s fifteen-minute limit does not apply. ### Capability comparison (delta rows only — full matrix in COMPUTE.md) -| | AgentCore Runtime | ECS on Fargate | **Lambda MicroVMs** | +| Capability | AgentCore Runtime | ECS Fargate | Lambda MicroVMs | |---|---|---|---| -| Isolation | MicroVM (managed) | Task-level (Firecracker) | MicroVM (Firecracker) | -| Max duration | 8 h | No cap | 8 h (running + suspended) — **verified** (`L-B430C318 = 8 hours`) | -| Suspend/resume | No orchestrator-visible API | No | **Yes** — explicit API + idle policy, state preserved, no compute charge while suspended. **Verified**: suspend reaches `SUSPENDED` in ~1 s with no `idlePolicy`; resume restores `RUNNING` in ~1 s with `microvmId` **and** `endpoint` byte-identical | -| Resources | AgentCore-managed | Build default 4 vCPU / 16 GiB; up to 16 vCPU / 120 GiB; 20–200 GB disk | **Baseline 8 GiB RAM / 4 vCPU, auto-scaling to a 32 GiB / 16 vCPU peak**; 32 GB disk. `minimumMemoryInMiB` configures the BASELINE (max 8,192 MiB); the service scales vertically on demand — baseline capacity and additional burst usage are billed separately | -| Packaging | ECR image ≤ 2 GB | ECR image, no hard cap | **Zip + Dockerfile in S3 → service-built snapshot image** (versioned, storage billed) | -| Invocation | `InvokeAgentRuntime` (SigV4) | `RunTask` + container overrides | `RunMicrovm` (image **ARN** required — a bare name is rejected) → dedicated HTTPS endpoint + JWE token (`CreateMicrovmAuthToken`, ≤ 60 min TTL) | -| Liveness | Agent heartbeat + `/ping` | `DescribeTasks` | MicroVM state (RUNNING / SUSPENDED / TERMINATED) via control-plane API **and** agent heartbeat (see sub-decision 1) | -| Session storage | `/mnt/workspace` FUSE (no `flock()`) | Ephemeral disk | Native disk in snapshot — **survives suspend/resume, `flock()` works** | -| Architecture | ARM64 | ARM64 | ARM64 (Graviton) | -| Regions (launch) | Broad | Broad | 5 (us-east-1/2, us-west-2, eu-west-1, ap-northeast-1) | - -> Rows marked **verified** were discharged empirically on 2026-07-31 (us-east-1); see `docs/verification/645-p1-lambda-microvm-runbook.md`. Two originally documented constants were **refuted** by that run and are corrected throughout this ADR: the `runHookPayload` cap (16 KB → 4 KB) and the claim that `RunMicrovm` accepts a bare image name (it does not). -> -> **Source hierarchy for service facts.** Where sources disagree, the higher tier wins, and the tier is recorded next to the claim: -> -> 1. **Live boundary probe** — a request the service actually accepted or rejected in this account/Region. Strongest, and the only tier that can refute the others. (Both corrections above came from here.) -> 2. **Modeled constraints** — the CLI/SDK request schema and enumerated allowed values (`--generate-cli-skeleton`, `ARM_64`, `ENABLED|DISABLED`). Authoritative about request *shape*; silent about runtime behaviour. -> 3. **Service developer guide** — including its sizing/scaling tables. Authoritative about *semantics* a probe cannot see, which is exactly how the memory row was fixed: a probe can only show that 32,768 MiB is rejected as a `minimumMemoryInMiB` value; only the guide explains that the field is a BASELINE and that the service scales to a 32 GiB peak on its own. A boundary probe is the strongest evidence about a boundary and says nothing about what the boundary *means*. -> 4. **SDK docstrings** — generated, and demonstrably stale here: `runHookPayload` documents "Maximum: 16,384 bytes" against an enforced 4,096. -> 5. **Launch blogs / skills / toolkit material** — orientation only; never load-bearing on its own. -> -> **Omitted API fields mean "service default", never "none".** Two live findings drove this rule: leaving `ingressNetworkConnectors` unset attaches a PUBLIC `HTTP_INGRESS` connector, and leaving `/ready` out of `hooks` makes the image un-creatable once any lifecycle hook is enabled. So for any security-relevant field, the desired posture must be **requested explicitly** and the test must assert the **outcome** (the ARN present in the request, the hook enabled) rather than the omission (`expect(field).toBeUndefined()`) — an omission assertion passes just as happily when the service is silently choosing something wider. -> -> On the **memory** row specifically: `CreateMicrovmImage` enumerates the baseline sizes a base image supports — `[512, 1024, 2048, 4096, 8192]` MiB for `al2023-1` — and rejects anything else, which is why the construct validates against that list at synth. The 32 GiB / 16 vCPU peak is reached by the service's own vertical scaling, not by asking for it. The account memory quota (`L-CD1C0CC4`, 1024 GB, "burst up to 4×") is an aggregate across MicroVMs, not a per-VM limit; note that concurrency arithmetic should be done against the PEAK, not the baseline, since that is what a busy fleet can actually consume. +| Packaging | ECR image, 2 GB limit | ECR image | ZIP + Dockerfile in S3 → versioned snapshot | +| Duration | Eight hours | No task duration cap | Eight hours, running + suspended | +| Explicit suspend/resume in ABCA | Unsupported | Unsupported | Control-plane APIs; memory and disk retained | +| Storage | Ephemeral disk plus preview persistent FUSE mount | Configurable ephemeral disk | 32 GB native disk; supports `flock()` | +| Sizing | Service-managed | Configurable; larger sustained workloads | ABCA baseline 8,192 MiB; service guide lists up to 32 GiB / 16 vCPU | +| Invocation/liveness | Invoke API + agent heartbeat | RunTask/DescribeTasks | RunMicrovm/GetMicrovm + agent heartbeat | + +See [COMPUTE.md](../design/COMPUTE.md) for the full comparison and costs. Suspending stops compute charges, but snapshot storage and save/restore charges remain. A shorter sleep delay does not guarantee lower total cost. + +The [P1 probes](../verification/645-p1-lambda-microvm-runbook.md) established a 4,096-byte `runHookPayload` limit, an image-ARN requirement and accepted baseline values of 512, 1,024, 2,048, 4,096 and 8,192 MiB. Those observations override conflicting generated SDK descriptions for the tested account/Region. They do not measure guest-visible launch memory, vertical-scaling latency or sustained workload fit. The service guide’s capacity figures and live observations must remain distinguishable. ### Design tensions the strategy must resolve -1. **Idle detection is inbound-traffic-based; the ABCA agent is outbound-only.** MicroVM idle policies suspend when no traffic arrives at the *endpoint*. A busy agent running a 40-minute build receives no inbound traffic and would be suspended mid-work by a naive idle policy. Conversely, "no inbound traffic" is the agent's *normal* state. -2. **No self-suspend.** The agent cannot suspend its own MicroVM from inside; only an external `SuspendMicrovm` call can. Suspend decisions must be owned by the orchestrator — which aligns with the unified liveness model proposed in [#491](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/491). -3. **Snapshots bake state.** The image snapshot is captured once at build time; every MicroVM resumes from it. Secrets, tokens, and per-task identity must arrive at run time (`runHookPayload`, ≤ 4 KB, or fetched in the `/run` hook), never at image build. Application PRNG reseeding is a consequence of the same property and is scoped to **P3** — see the amended note under sub-decision 2 and the risk bullet, which record why the exposure is negligible today. -4. **Auth tokens are short-lived.** JWE tokens max out at 60 minutes; any orchestrator→agent HTTP interaction over the endpoint needs token refresh, unlike AgentCore's SigV4 invoke or ECS's no-endpoint model. -5. **Identity delta — narrower than it looks.** Most AgentCore services ABCA uses are standalone and substrate-portable: Memory is already consumed from ECS via an IAM grant plus `MEMORY_ID` (`EcsAgentCluster`), and Gateway (ADR-019) is portable by design (SigV4 inbound). The genuinely Runtime-coupled piece is the workload-access-token **delivery mechanism** (`runtimeUserId` → `WorkloadAccessToken` request header → `BedrockAgentCoreContext`, used by `resolve_linear_api_token()`), which has no MicroVM equivalent. The ECS backend already lives with this delta (env-var token delivery); MicroVMs inherit the same posture until the pluggable identity work ([#249](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/249), ADR-016) redesigns the seam. -6. **The service's defaults are not our posture.** Two of them, both discovered live: `RunMicrovm` attaches a **public** `HTTP_INGRESS` connector (and mints a public `*.lambda-microvm..on.aws` endpoint) when `ingressNetworkConnectors` is omitted, and `CreateMicrovmImage` **requires** the `/ready` hook whenever any lifecycle hook is enabled. Neither posture can be reached by leaving a field out — each needs an explicit control (see sub-decisions 3 and 4). +1. Traffic-based idle policies observe inbound endpoint traffic. A busy coding agent mostly sends outbound requests, so absence of inbound traffic cannot establish that it is idle. +2. Suspension needs an external controller. The agent can prepare a safe checkpoint, but the coordinator owns the service call. +3. Image snapshots share build-time process state. Credentials, task identity and deployment configuration must arrive after launch. +4. A worker’s eight-hour lifetime is shorter than an unanswered approval may remain useful. Durable task state must outlive the worker. +5. New packaging, IAM and lifecycle behavior need live checks; a successful CDK synth cannot validate service semantics. ## Decision -Adopt **AWS Lambda MicroVMs as a third, opt-in `ComputeStrategy` backend** named `lambda-microvm`, selected per repo via Blueprint `compute_type`. AgentCore remains the default. Five sub-decisions: +Add `lambda-microvm` as an opt-in `ComputeStrategy`. AgentCore remains the default. ### 1. Strategy shape: extend the interface with mandatory suspend/resume -`ComputeType` widens to `'agentcore' | 'ecs' | 'lambda-microvm'` (mirrored in `cli/src/types.ts` and the CLI's inline unions). `SessionHandle` gains a `{ strategyType: 'lambda-microvm', microvmId, endpoint }` variant — `microvmId` because every lifecycle API (`suspend-microvm`, `resume-microvm`, `terminate-microvm`, `get-microvm`, and `create-microvm-auth-token` — the latter not called in P1–P3, see sub-decision 3) takes only the MicroVM identifier, and `endpoint` because it is per-session (minted by `RunMicrovm`) and required for any orchestrator→agent HTTP interaction. Note the naming seam: the handle field is `microvmId` (matching `RunMicrovmResponse.microvmId`), while the request key on every lifecycle command is `microvmIdentifier` — the strategy and inline approval-wake helper translate between the two at their control calls. **P3 refinement (2026-09-15):** the handle also retains the actual `imageArn` and `imageVersion` returned by Run, plus `lifecycleProtocol` after exact-version verification. The requested image remains deployment configuration, but the version that launched an existing worker is per-session evidence. Save the known handle before optional discovery; check that exact version with `GetMicrovmImageVersion`, then conditionally persist support in the receipt and `compute_metadata`. Current deployment settings cannot substitute for missing legacy evidence. - -**A second, sharper naming seam: `imageIdentifier` must be an ARN.** The name suggests a bare image name is acceptable — `create-microvm-image --name` takes one, and this ADR originally assumed `run-microvm --image-identifier` would too. It does not: a bare name is rejected with `ValidationException: Malformed ARN - doesn't start with 'arn:'`, and so is `list-microvm-image-builds --image-identifier ` (`Invalid ARN format`). The construct therefore resolves an operator-supplied name to its exact `arn:${Partition}:lambda:${Region}:${Account}:microvm-image:${Name}` ARN **once** — the same value it scopes the lifecycle IAM grant to — and injects THAT as `MICROVM_IMAGE_IDENTIFIER`. One derivation, two consumers, so a request field and an IAM resource can never disagree. The strategy validates the invariant and fails fast with the remedy, because the service's own error names neither the env var nor the fix. - -The `ComputeStrategy` interface now has **mandatory** `suspendSession(handle)` / `resumeSession(handle)` methods returning `SessionLifecycleResult`, defined as `{ supported: false } | { supported: true }`. This local P3 foundation changes all three strategies together: AgentCore and ECS explicitly return false without calling AWS; MicroVM submits a command with a 10-second request bound. A future backend must implement both methods. The durable supervisor uses this explicit result without treating acknowledgment as final state. - -`supported: true` means AWS acknowledged the command, not that the target state was reached. The SDK's suspend/resume response bodies are empty. All operational failures, including conflicts and not-found responses, propagate through sanitized errors; no generic conflict is normalized into success without observed-state evidence. A timed-out command may still have committed and requires reconciliation. The installed SDK has no `RESUMING` state. `SessionStatus.microvmState` now supplies explicit observations because the coarse `running` result also includes PENDING/unknown. The local policy/store uses retained generations and sticky wake intent to handle a delayed suspend after approval; the durable supervisor now connects that policy, retains fixed recovery/session deadlines and waits for the required compute and guest observations. See `docs/verification/645-lifecycle-intent.md`. - -**Poll semantics — the strategy reports, the orchestrator interprets.** `pollSession(handle)` receives only the session handle and cannot see task state, so the health rules must live where the DynamoDB status lives. `SessionStatus` gains a `'suspended'` variant; the strategy maps `GetMicrovm` state mechanically and the **orchestrator** cross-references against the task row — the same division of labor `finalPollState` already uses for ECS (substrate stopped + non-terminal DynamoDB status → failed) and `pollTaskStatus` uses for agentcore heartbeats: substrate `suspended` + task `AWAITING_APPROVAL` is healthy (orchestrator-intended suspend); `suspended` with any other task status is an anomaly to surface, not fail-fast; substrate terminal + non-terminal task status → classify failed. - -**Liveness on this backend combines substrate state and the agent heartbeat.** `GetMicrovm` reports whether the VM is running or terminal. The independent `_heartbeat_worker` in `agent/src/server.py` refreshes the task timestamp every 45 seconds while the task is `RUNNING`; a stale timestamp detects loss of that writer, including a process crash or inability to write DynamoDB. It does **not** prove pipeline progress: a hung coding thread can coexist with a healthy heartbeat thread. A failed `/run` hook can trigger service teardown; after an accepted hook, ABCA still needs explicit termination and the maximum-duration backstop. A general progress watchdog remains separate work. +All strategies implement `startSession`, `pollSession`, `stopSession`, `suspendSession` and `resumeSession`. The lifecycle methods return an explicit supported/unsupported result. AgentCore and ECS return unsupported without making suspension API calls; MicroVM bounds each control request. An accepted API request does not prove the transition completed. -AgentCore and MicroVM tasks start the same periodic heartbeat worker, so they use the same 120-second grace and 240-second stale thresholds. ECS starts the pipeline directly, bypassing `server.py`; its pipeline writes a timestamp once but does not start this worker. Enabling the same stale check for ECS would therefore fail healthy tasks after roughly four minutes. The check applies only to `RUNNING`, not `AWAITING_APPROVAL`. The transition back to `RUNNING` refreshes `agent_heartbeat_at` in the same conditional transaction that clears the matching approval request. A poll immediately after a long gate therefore sees a fresh heartbeat before the next worker tick. The local P3 supervisor additionally bounds recovery through suspension and decision consumption; live verification remains required. +The MicroVM handle records `microvmId`, `endpoint`, the actual launched `imageArn`/`imageVersion`, and verified `lifecycleProtocol` when available. Lifecycle request fields use `microvmIdentifier`. Save the known handle before optional image discovery so a failed capability check cannot lose cleanup ownership. Current deployment settings cannot establish an older worker’s capabilities. -The service's `MicrovmState` enum has **six** members, not three, so the mapping is stated exhaustively (one line of rationale each, mirrored in the strategy's doc comment): - -| `MicrovmState` | `SessionStatus` | Why | +| Service state | Strategy result | Coordinator responsibility | |---|---|---| -| `PENDING` | `running` | Still booting; the same way ECS's `PENDING`/`PROVISIONING` map to `running`. | -| `RUNNING` | `running` | — | -| `SUSPENDING` | `suspended` | Already on its way to frozen; reporting `running` would tell the orchestrator compute is still progressing when it is not. Both suspend states land on a report the orchestrator treats as benign-or-anomalous depending on task status, never as failure. **Not observed in the recorded probe** — suspend reached `SUSPENDED` in under 1 s. Other runs may expose this state, so handle it without requiring it to appear. | -| `SUSPENDED` | `suspended` | — | -| `TERMINATING` | `completed` | Terminal-bound and carries no exit code, so "the substrate is gone" is all the strategy can honestly say. | -| `TERMINATED` | `completed` | Success vs failure is the orchestrator's call — it cross-references the DynamoDB status. **This is the load-bearing terminal signal**, not `NotFound` (see below). | -| *unrecognized* | `running` | A future service enum addition must never fail a healthy task; the strategy warns and keeps polling. | - -**`GetMicrovm` `ResourceNotFoundException` → `completed`, but as a LATE fallback.** This deliberately diverges from `ecs-strategy`, where `DescribeTasks` returning no task maps to `failed`. ECS keeps stopped tasks describable for roughly an hour, so a missing task there really is anomalous; a MicroVM is eventually reaped from the control plane **by design**, so `failed` would fail tasks that finished cleanly. - -What the live run corrected is the *timing*, not the mapping. `NotFound` is **not the near-term terminal signal**: a terminated MicroVM reported `TERMINATING` at +1 s, `TERMINATED` at +3 s, and was **still `TERMINATED` ~10 minutes later** and at every subsequent checkpoint — `ResourceNotFoundException` was never observed in that window. So the branch that actually fires in practice is `TERMINATED → completed` in the table above; the `NotFound` rule covers only a VM reaped after a long gap (a poll resumed after a crash, a stranded-task reconciler sweep). Both are required and neither substitutes for the other: without the `TERMINATED` row the orchestrator would poll a finished VM until its safety net fired, and without the `NotFound` rule a late sweep would classify a cleanly-finished task as a poll error. - -The divergence is safe because it does not weaken detection: the orchestrator still fails the task when a terminal report lands while the DynamoDB status is non-terminal, so a genuine mid-run disappearance is caught — it simply receives the substrate-failure classification instead of a misleading poll error. Because that cross-check acts on a status read earlier in the same poll cycle, the orchestrator **re-reads the task row before failing** (the normal shutdown order is "agent writes terminal status → agent exits → VM terminates", which a stale read would otherwise turn into a spurious failure); ECS buys the same protection with a five-consecutive-poll patience counter instead. - -Neither the mapping nor the NotFound rule is a health decision: both are mechanical restatements of substrate state, which is what keeps the "strategy reports, orchestrator interprets" split intact. - -Normative requirements (EARS, per [ADR-020](./ADR-020-ears-requirements-syntax.md)): - -- When a task's Blueprint sets `compute_type: 'lambda-microvm'`, the orchestrator shall resolve the `LambdaMicrovmComputeStrategy` via `resolveComputeStrategy`. -- When `startSession` is invoked, the strategy shall call `RunMicrovm` with `maximumDurationInSeconds` set to 28 800 (the service maximum, matching AgentCore's 8-hour session cap and sitting inside the orchestrator's ~8.5 h safety-net poll window). -- When `startSession` is invoked, the strategy shall pass a fully-qualified MicroVM image ARN as `imageIdentifier`. -- If the configured image identifier is not an ARN, then the strategy shall fail the session start with an error naming the environment variable and the redeploy remedy, before performing any AWS call. -- When `startSession` returns, the orchestrator shall persist the MicroVM handle (`microvmId`, `endpoint`, actual `imageArn`/`imageVersion` when returned, and verified `lifecycleProtocol` when supported) in the task row's `compute_metadata` (the field `cancel-task.ts` already reads ECS handles from). -- The strategy shall omit `idlePolicy` on every `RunMicrovm` call, in every phase. -- The orchestrator shall be the sole initiator of suspension, via `suspendSession`. -- When `pollSession` observes MicroVM state `SUSPENDED` or `SUSPENDING`, the strategy shall report `suspended` without interpreting task state. -- When `pollSession` observes MicroVM state `TERMINATED` or `TERMINATING`, the strategy shall report `completed` (the observable terminal state persists for at least ~10 minutes, so `completed` shall not depend on the MicroVM being reaped). -- When `pollSession` observes a MicroVM state it does not recognize, the strategy shall report `running`. -- If `GetMicrovm` reports that the MicroVM does not exist, then the strategy shall report `completed`. -- If the strategy reports a terminal substrate state while the task's DynamoDB status is non-terminal, then the orchestrator shall re-read the task row and, if it is still non-terminal, classify the task as failed with a substrate-failure remedy. -- If the strategy reports `suspended` while the task's DynamoDB status is not `AWAITING_APPROVAL`, then the orchestrator shall surface an anomaly event and shall not fail-fast the task. -- While a `lambda-microvm` task's DynamoDB status is `RUNNING`, if the task's `agent_heartbeat_at` is stale (or absent past the grace window) by the same thresholds the orchestrator applies to `agentcore`, then the orchestrator shall treat the session as unhealthy and stop polling — the substrate `GetMicrovm` check shall remain the crash detector, and the heartbeat shall detect loss of the in-guest heartbeat writer, without claiming to detect every pipeline hang. -- The task-detail API response shall include `agent_heartbeat_at`, and the CLI shall surface it while the task is non-terminal (P2r2-F11: the field drove the orchestrator's heartbeat check but was never projected, so no operator could observe the signal — and its invisibility produced a wrong verification conclusion). -- The task-**summary** API response (`GET /v1/tasks`) shall also include `agent_heartbeat_at`, and `bgagent list` shall render it as an age column. Extending the field to the list response is a deliberate widening of the requirement above rather than an incidental one: the detail-only projection makes liveness a per-task question, and an operator checking a fleet of tasks one `bgagent status` at a time is exactly how the hung task P2r2-F11 describes went unnoticed. Same suppression rule as the detail view (terminal tasks and never-beaten tasks render a placeholder) so the two views cannot disagree. -- If `suspendSession` or `resumeSession` is invoked on a strategy that does not support suspension, then the strategy shall return an explicit unsupported result. -- When the agent process reaches a terminal state, the agent shall exit. -- When the orchestrator finalizes a `lambda-microvm` task, the orchestrator shall call `terminate-microvm` (termination shall not rely on any substrate timeout, and shall not rely on the MicroVM self-terminating — it does not). - -*On the omitted `idlePolicy`:* if the block is present all three fields are required, so omission is the unambiguous disabled state the invariant test asserts. This deliberately forgoes `suspendedDurationSeconds` — it lives *inside* `idlePolicy` and cannot be set without re-enabling the traffic-idle machinery — so the suspended-state bound is `maximumDurationInSeconds` plus orchestrator termination and the stranded-approval reconciler (see sub-decision 2). A tighter substrate-level suspended-TTL remains available later as an additive `idlePolicy` change if operators want it. *On the fixed `maximumDurationInSeconds`:* no wall-clock task budget exists in the platform (budgets are `max_turns` / `max_budget_usd`), so the value is parity with AgentCore's 8 h cap rather than derived policy; a Blueprint override can be added later if a real need appears. - -### 2. Lifecycle: suspend/resume reconciled with the agent-owned approval poll - -The headline economic win is suspend during **HITL approval waits** (Cedar approval gates, [CEDAR_HITL_GATES.md](../design/CEDAR_HITL_GATES.md)): while a task waits on a human decision, the MicroVM is suspended (compute charges stop; memory/disk state — cloned repo, warm build caches — is preserved) and resumed when the decision lands. Under Cedar decision #6 the approval window is bounded (default 300 s, ceiling 1 h, timeout → deny), so the saving per gate is bounded at ~1 h of compute — real at 16 vCPU, and it makes any future extension of gate ceilings (the off-hours posture §14.8 deliberately defers) cheap on this backend. - -The handshake must respect the existing approval mechanics: the agent **discovers decisions itself** by polling DynamoDB (`_poll_for_decision`, monotonic timeout), the approve/deny Lambda commits the decision transaction before writing separate coordinator wake intent, and `AWAITING_APPROVAL` holds the concurrency slot (Cedar decision #7). Nothing "delivers" an approval to the agent, and suspension freezes the agent's monotonic clock — so the design is: - -- **Suspend — orchestrator-owned.** The orchestrator's durable poll observes `AWAITING_APPROVAL` on a `lambda-microvm` task and calls `suspendSession` after a grace period, and only when the gate's remaining window exceeds grace + resume overhead (suspending a 30 s gate is pure loss). Suspend is a policy decision on a poll observation, not a user action. -- **Resume — inline in the approve/deny Lambdas, orchestrator poll as backstop.** After the transactional decision write commits, `ApproveTaskFn`/`DenyTaskFn` load the MicroVM handle from the task row's `compute_metadata` (persisted at session start — the same field `cancel-task.ts` reads ECS handles from) via a post-commit strongly-consistent `GetItem`, then save wake intent, observe the VM and request `ResumeMicrovm` **best-effort** only when SUSPENDED: on failure they log a warning and write a resume-orphan task event; the decision response never fails on a compute error (the decision row is already durable). The orchestrator poll reconciles: current approval row APPROVED/DENIED + MicroVM still `SUSPENDED` → retry resume; a PENDING row alone must not trigger wake-up. - - *Why inline rather than poll-only — codebase precedent:* resume-on-approve is structurally identical to task cancellation — a user-initiated, latency-sensitive action whose purpose is an immediate compute-lifecycle side effect. `cancel-task.ts` already resolves this exact tension: the API-plane handler invokes ECS `StopTask` / AgentCore `StopRuntimeSession` inline, best-effort (a failed stop logs a warning and the state transition stands; a `task_cancel_compute_orphan` event is written when no stoppable compute handle exists, `reason: missing_runtime_handle`) — with the conditional IAM wired in `task-api.ts`. The resume path goes one step further than the precedent by also writing the orphan event on *failed* resume calls, because a failed resume strands a suspended VM awaiting a decision — a stronger liveness consequence than a failed stop of an already-cancelled task. The alternative (orchestrator-poll-only resume) preserves single-owner lifecycle purity but pays up to a full poll interval (~30 s) of latency on every approval, and the purity argument was already litigated and declined for cancel. `approve-task.ts` is deliberately minimal today (security-critical ownership comparison, Cedar finding #6); the resume call is therefore added *after* the transaction commits, cannot alter the decision outcome, and carries one conditional `lambda:ResumeMicrovm` grant — the same blast-radius trade the cancel handler accepted in review. -- **Timeout under freeze — the agent re-bases on the wall clock it already owns.** The agent's monotonic gate timer freezes while suspended, so resuming near the deadline is not enough: the frozen timer would still hold its remaining budget and fire the deny minutes *after* the user-visible window — colliding with the approval row's TTL (`created_at + timeout_s + 120s`) and triggering the "row reaped → stranded" fallback on a healthy gate. Instead, the gate expires at **`min(monotonic budget, created_at + timeout_s)`**, evaluated on each poll iteration and on `/resume`. This is not a new principle: Cedar decision #6 is already "min wins" for timeouts, the wall-clock deadline is already durable in the approval row the agent itself writes (`created_at` is recorded by the agent; clock changes across suspend still need testing), and §13.12's late-approval race fix already establishes that the durable row is authoritative over the agent's local timer. Deny authority stays agent-side (the conditional `TIMED_OUT` write + ConsistentRead re-read race protection is untouched); the orchestrator's resume at `deadline − margin` is purely the wake-up mechanism, with no correctness role. -- **Backstops, not mechanisms.** `maximumDurationInSeconds` (mandatory on every `RunMicrovm`, pinned at 28 800 s — see sub-decision 1) is the substrate kill switch bounding running **and** suspended time; the orchestrator's finalization `terminate-microvm` is the active cleanup path; the stranded-approval reconciler retains its role for orphaned waits. No `idlePolicy`-based bound is used in any phase — see sub-decision 1's omit-`idlePolicy` invariant. - - **The active terminate is still mandatory on the SUCCESS path, and P2 sharpened why.** P1 concluded flatly that "nothing self-terminates": a hook-less MicroVM reached `RUNNING` in 12 s and stayed there with no `stateReason` through every checkpoint. P2 refuted that *for the failure path only* — with `run: ENABLED`, a run hook that answers 4xx makes the **service** terminate the VM within ~12 s, `stateReason: "Run lifecycle hook returned HTTP status 400. Please check your hook endpoint and application logs for more details."`, after which `suspend-microvm` correctly refuses it. That is a real improvement in cost posture and a direct benefit of declaring hooks (see also the failure-path row in the phasing table, sub-decision 3). +| RUNNING or starting | `running` | Read task state and applicable heartbeat | +| SUSPENDING / SUSPENDED | `suspended` | Reconcile the approval and lifecycle intent | +| TERMINATING / TERMINATED / not found | `completed` | Re-read task state; distinguish completion, recoverable checkpoint and failure | +| Unknown future state | `running`, with warning | Continue bounded observation | - It does **not** relieve the orchestrator of anything, because the two cases are disjoint. The service reaps a hook *result* it did not like; it has no view of the guest once the hook returned 200. So a task that starts normally — the overwhelming majority — has no service-side reaper at all, and a VM whose pipeline finished, crashed after `/run`, or hung is reaped by nobody but `TerminateMicrovm`. A leaked handle therefore remains a cost incident that bills until the 8 h cap; only the "the guest rejected its own payload" corner now cleans itself up. -- **Concurrency slot stays held** during suspend. Cedar decision #7's rationale ("container alive, consuming memory") weakens under suspend, and the harder replacement rationale — "AWS counts `SUSPENDED` MicroVMs toward the account memory quota, so releasing ABCA's slot would not free real capacity" — is **undischarged**: the suspended VM stayed in `list-microvms` at every checkpoint, but that only proves *listed*. `L-CD1C0CC4` (1024 GB, account-scoped) exposes no `UsageMetric`, `AWS/Usage` carries only `CallCount` per API, and no MicroVM memory metric exists in any namespace, so consumption is **not observable safely** — proving it would need a large concurrent fleet. The conclusion (hold the slot) stands as the conservative choice, not as a verified fact. Size the arithmetic against the 32 GiB **peak** rather than the 8 GiB baseline: a busy fleet scales up, so peak is what actually competes for the account quota. +A strategy result is not a task outcome. A terminal substrate can leave a recoverable pending-approval checkpoint; otherwise a non-terminal task needs failure classification. Capacity release separately requires confirmed physical shutdown, not merely the strategy’s `completed` result. -In P3, the agent's `/suspend` hook must acknowledge durable progress before returning 200 within its hook budget; `/resume` must refresh credentials and reseed application PRNG state from fresh OS entropy. These hooks are deployed in image 6.0 with atomic checkpoints, retained credential refresh and original-gate reconciliation, and have passed the isolated live workflows linked above. Automatic suspension on the normal task path remains disabled pending rollout acceptance. A non-cryptographic PRNG such as Python `random` remains unsuitable for secrets even after reseeding; security-sensitive randomness must use an OS-backed cryptographic source. - -**Amended (P2 review): Application PRNG reseeding is P3 scope, and the P2 exposure is negligible.** The original EARS requirement below implied a `/run`-time reseed had to land with P2; it did not, and shipping P2 with the requirement unmet-but-asserted was itself the defect. Measured exposure, which is what changed the scoping: - -- The only consumer of a non-cryptographic PRNG in `agent/src` is `progress_writer.py`'s `getrandbits(80)`, used for the random half of a **ULID**. `os.urandom` / `secrets` are not seeded from the snapshot at all, and nothing in the agent derives a key, token, or nonce from `random`. -- That ULID is a DynamoDB **sort key under a `task_id` partition**. A collision therefore needs two events in the *same task* at the *same millisecond* with the same 80 random bits — and identical PRNG state across two MicroVMs restored from one snapshot does not produce that, because the events are in different partitions. The worst case is a duplicate progress event within one task, not a security boundary. -- Credential safety is a separate concern from the application PRNG. Build hooks avoid initializing AWS clients and their credential caches; their credential-environment warning does not prove that every possible image is secret-free. `platform_config.agent_session_role_arn` delivers a role identifier, not credentials. The running agent obtains credentials through its runtime provider and assumes the task-scoped role. - -So the reseed moves to P3 alongside `/suspend` + `/resume`, where a *resumed* VM — which really does continue with the exact PRNG state it was frozen with, repeatedly — makes it load-bearing rather than theoretical. P3 must reseed on both `/run` and `/resume`, and must not treat the ULID as the only consumer: security-sensitive values must continue to use `secrets` or an equivalent cryptographic source, not `random`. +AgentCore and MicroVM start a heartbeat writer every 45 seconds. Their RUNNING tasks use the same 120-second startup grace and 240-second stale threshold. ECS’s batch entrypoint does not start that writer, so applying this check to ECS would fail healthy tasks. Heartbeats establish writer liveness, not progress of every coding thread. Returning from approval to RUNNING refreshes the heartbeat atomically. Task detail/list APIs and CLI views expose the timestamp. Normative requirements (EARS): -- While approval suspension is enabled for a verified compatible worker, a `lambda-microvm` task is in `AWAITING_APPROVAL`, and the gate's remaining window exceeds the configured grace period plus resume overhead, the orchestrator shall call `suspendSession` after the grace period. -- When the approve or deny Lambda commits a decision for a `lambda-microvm` task, the Lambda shall load the MicroVM handle from `compute_metadata`, persist wake intent for the current identity, and request `ResumeMicrovm` best-effort after observing SUSPENDED and rechecking the gate. -- If the inline resume fails, then the Lambda shall record a resume-orphan task event and shall still return the decision outcome. -- While the current approval row is APPROVED or DENIED and the MicroVM remains `SUSPENDED`, the orchestrator shall retry `resumeSession`. A PENDING row alone is not a wake-up condition. -- While a `lambda-microvm` task waits on an approval gate, the agent shall evaluate gate expiry as the earlier of its monotonic budget and the row's wall-clock deadline (`created_at + timeout_s`), on each poll iteration and on `/resume`. -- If gate expiry is reached without a decision, then the agent shall deny. -- If no decision arrives by the gate's wall-clock deadline minus the resume margin, then the orchestrator shall resume the MicroVM so the agent can evaluate expiry and fire the deny agent-side. - -### 3. Packaging: same agent image source, new build path - -The existing agent container (`agent/` Dockerfile, already ARM64) is repackaged as a zip + Dockerfile artifact in S3 and built into a versioned `MicrovmImage` via `CreateMicrovmImage`. The agent runs its existing FastAPI server (`agent/src/server.py`) — the MicroVM path uses the HTTP entrypoint like AgentCore, not ECS's batch bypass — plus the runtime lifecycle hooks (`/run`, `/suspend`, `/resume`, `/terminate`) and the `/ready` + `/validate` build hooks, all on the same port the server already listens on (8080, declared as the image's `hooks.port`). Runtime hooks are fast-notification only (1–60 s): `/run` validates the payload and starts the pipeline **asynchronously**, mirroring how the agent loop already runs in a background thread behind `/ping` on AgentCore. +- When a Blueprint selects `lambda-microvm`, the orchestrator shall use the MicroVM strategy and persist its handle in `compute_metadata`. +- Every launch shall use a full image ARN, `maximumDurationInSeconds=28800`, explicit `NO_INGRESS` and no `idlePolicy`. Invalid image configuration shall fail before launch. +- Before allowing suspension, the coordinator shall verify the actual launched image version and lifecycle protocol. +- When a suspended state conflicts with task state, the coordinator shall record and reconcile the anomaly within bounded recovery windows. +- When the task ends, the coordinator shall actively terminate its MicroVM. The service lifetime is a backstop, not the normal cleanup mechanism. +- Uncertain starts, lost replies and supervisor replay shall preserve the original worker identity, ownership and lifetime rather than launch duplicate workers. -**Hook phasing — corrected: `/ready` + `/run` are both P1.** The original plan split *declaring* a hook from *serving* it, putting `/run`'s declaration in P1 and its implementation in P2. Live verification proved that split is **not a reachable service state**, on two independent counts: +### 2. Lifecycle: suspend/resume reconciled with the agent-owned approval poll -- `CreateMicrovmImage` rejects an image that enables any lifecycle hook without `/ready`: *"The ready (/ready) MicroVM image hook must be enabled when any MicroVM lifecycle hook (run, resume, suspend, or terminate) is enabled."* So a P1 image declaring only `/run` is **not creatable**. -- With `/ready` added but unserved, both chipset builds fail: *"Ready hook check failed: the application returned a client error (HTTP 4xx) response."* So a declared hook must be served in the same phase. -- And an image with **no** hooks at all — the only other creatable shape — cannot receive a payload: *"The run hook must be enabled in the MicroVM image to pass the run hook payload."* So deferring hooks entirely also defers the whole payload-delivery channel. +Unanswered approvals have no deadline by default (`approval_timeout_s=0`). Explicit task deadlines range from 30 to 3,600 seconds; a positive policy-rule deadline can also apply. The independent sleep preference defaults to 600 seconds per approval wait. `microvm_sleep_after_s=0` keeps the task awake. -The phasing is therefore: +Automatic suspension also requires the deployment’s `microvm_approval_suspend_enabled` opt-in, which defaults false for new deployments. A live Parameter Store switch lets existing durable executions stop initiating new suspensions without changing their pinned Lambda environment. The verified normal deployment has this opt-in enabled. Turning it off does not abandon already-suspended workers. -| Hook | Declared by | Served by the agent | Notes | -|---|---|---|---| -| `/ready` | **P1** (construct enables `hooks.microvmImageHooks.ready`) | **P1** | MANDATORY, not a quality nicety — see above. A 200 proves uvicorn is bound and `server` imported cleanly (pulling in `pipeline` → `runner` → the policy engine), so a missing policy file fails the BUILD instead of the first task. **Since P2-F5 it also WARMS the snapshot** — the hook's 200 is what the service waits for before capturing the snapshot, making this the only place a warm page can be created, and the 225 MiB `claude` binary was cold in it (see the P2-F5 correction below). A required warm-up failure answers 503, so a snapshot that cannot exec the agent's own CLI fails the image build instead of every task. Still makes ZERO AWS calls, logging included (a `--version` exec is neither an AWS call nor a network call). | -| `/run` | **P1** (construct sets `hooks.microvmHooks.run`) | **P1** | The payload-delivery channel. Must be served in P1 because `/ready` forces hooks to exist at all, and a hook-less image cannot accept `runHookPayload`. Since P2 it is also the **platform-configuration** channel (see "Platform configuration delivery" below). | -| `/validate` | **P2** (construct sets `hooks.microvmImageHooks.validate`) | **P2** | An **image** (build-time) hook, and a **shallow self-check only**: server alive, hook routes registered, interpreter + contract sanity. It runs under the BUILD role, which deliberately holds no Bedrock / Secrets Manager / DynamoDB grants, so it must make **zero AWS API calls** and must not touch credential resolution — the "deeper warm-up assertions (Bedrock reachability, Memory access, tool availability)" this ADR originally assigned here are **not implementable**: every one of them would `AccessDenied` and fail every image build. They belong to the first task's own error handling. 200 when the checks pass, 503 while still initialising. | -| `/terminate` | **P2** (construct sets `hooks.microvmHooks.terminate`) | **P2** | Best-effort final flush: a final structured log line, then 200 — always, inside the hook budget, even with nothing running. It must **not** write terminal task status (the orchestrator normally finalizes before `TerminateMicrovm`, and also attempts cleanup if database finalization fails, so a terminate hook that wrote a status would race that finalization and could clobber the real outcome) and must not join the pipeline thread. `ProgressWriter` writes synchronously but drops failed events, so this hook cannot guarantee that every progress write succeeded. P3 supplies a separate acknowledged checkpoint for suspension. "Always 200" also covers the BODY: the handler reads the raw request rather than a typed model, because a typed body is validated before the handler runs and would answer 422 to malformed JSON — a reported hook failure on a successful teardown. Safe to declare because `TerminateMicrovm` removes the VM with or without in-guest cooperation. **Correction (P2-F8):** the service sends `microvmId: ""` on this hook, unlike `/run` where it is populated, so an empty id is expected-normal and this hook cannot join the guest record to the control-plane one — `/run`'s accepted line carries that correlation instead. | -| `/suspend`, `/resume` | **P3**, deployed in image 6.0 | **P3**, deployed in image 6.0 | Managed images declare the served checkpoint/credential/gate hooks with the shared 30-second service timeout and protocol marker. The coordinator verifies the actual launched version before allowing new suspension. Lifecycle responses close the connection before freezing. Isolated live workflows pass; normal automatic-suspension activation remains gated off. | +The coordinator observes the exact pending gate and waits for the sleep delay. It skips sleep when a timed gate has too little time left before its wake margin. The guest holds a coding barrier, drains acknowledged progress and commits a checkpoint before accepting `/suspend`. Lifecycle HTTP responses explicitly close their connections before freeze to avoid reuse of a stale pooled connection; see the [transport experiment](../verification/645-p3-wake-transport-20260916.md). -Consequence to state plainly, replacing the original "a P1-built MicroVM image is not runnable end to end": **a P1 image is creatable, launchable and payload-deliverable, but carries no smoke-parity guarantee.** P1 delivers the strategy, the construct, the roles/buckets/connectors, the image resource, the packaging script, and the `/ready` + `/run` endpoints — so a `lambda-microvm` task can start a MicroVM and hand it a payload. What P1 has **not** established is anything P2 owns: AgentCore Memory grants and `MEMORY_ID` delivery, the agent's non-secret env parity inside the snapshot, egress specifics from a running MicroVM, and heartbeat/progress behaviour end to end. At the P1 handoff, no clone → change → PR run had happened. P2 initially demonstrated that path with an IAM workaround; the 2026-09-14 clean deployment and coding/iteration/cancellation runs later passed without manual IAM changes. The broader failure/recovery, effective IAM and networking matrix remains open. The construct and the packaging script surface that remaining scope at synth/run time (`abca:microvm-image-p1-smoke-unverified`) so an operator cannot mistake a launchable substrate for a verified one. +Approval and denial handlers commit the decision first. They then read the current handle consistently, persist wake intent and request resume best-effort. A wake failure records diagnostics and does not undo the accepted decision. The durable supervisor retries and observes both service state and guest consumption of the decision; RUNNING alone does not prove the tool was released. -**The `AWS::Lambda::MicrovmImage` L1 enforces the API's enums (P2-F2, live 2026-08-06).** This closes the one item P1 left explicitly open, and it closes it against the construct's own stated reasoning. CloudFormation's generated types make `cpuConfigurations[].architecture` and all four `hooks.*` fields plain strings and document no allowed values, from which P1 concluded that the CloudFormation surface takes a *hook path* while the API takes an `ENABLED`/`DISABLED` flag, and that both were correct for their own surface. CloudFormation refused the change set at **early validation** — the stack was never touched, so there was no rollback and no runtime symptom to trace back — on five values: +A sleeping worker retains its concurrency reservation. For a longer wait, ABCA verifies a complete, version-pinned conversation/workspace checkpoint, fences the attempt, confirms shutdown and releases the reservation. A later decision can admit one replacement through the original published coordinator. It restores Git state, required workspace files, the actual SDK conversation, exact pending tool inputs and cumulative usage with fresh scoped credentials. See the [continuation protocol](../verification/645-p3-continuation-protocol-20260917.md). -``` -/aws/lambda-microvms/runtime/v1/run is not a valid enum value. Supported values: [DISABLED, ENABLED] - (at /Resources/…/Properties/Hooks/MicrovmHooks/Run) … and the same for Terminate, Ready, Validate -arm64 is not a valid enum value. Supported values: [ARM_64] - (at /Resources/…/Properties/CpuConfigurations/0/Architecture) -``` +Normative requirements (EARS): -Three consequences. First, the CloudFormation surface is **identical** to the API surface, and the packaging script (`--cpu-configurations '[{"architecture":"ARM_64"}]'`, `--hooks '{"microvmHooks":{"run":"ENABLED",…}}'`) had it right all along. Second, the "CDK-managed (recommended)" bootstrap path was **non-functional** for the whole of P1 and P2 — the out-of-band `--create-image` script was the only working path — and no unit test, `cdk synth` or cdk-nag rule could see it, because the types accept any string. Third, **hook paths are not configurable on either surface**: the service calls fixed well-known routes (proved by the build and run logs, which POST to exactly the `/aws/lambda-microvms/runtime/v1/*` paths the agent serves), so the route constants in the construct are an agent-side cross-package contract ONLY and must never be sent as property values again. Also discharged in passing: the `microvmImageHooks` property name and nesting are correct — CloudFormation resolved `…/Hooks/MicrovmImageHooks/Ready` and objected only to its value. +- While all sleep gates and guest safety checks hold, the coordinator shall suspend after the configured delay; it shall remain the sole initiator of suspension. +- When a decision is committed, the API shall preserve that outcome even if optional wake or replacement dispatch fails. +- A timed request shall retain its original wall-clock deadline and, within the original process, its monotonic cap. The earlier limit wins; wake/replacement shall not restart either decision window. +- Before an unanswered timed gate reaches its deadline, the coordinator shall wake the worker with the configured margin. For a retired attempt, it may resolve expiry atomically while admitting a replacement. +- Before releasing a worker reservation, the coordinator shall verify checkpoint integrity, fence ownership and confirm physical shutdown. An uncertain control request shall not release capacity. +- A replacement shall preserve the task/request/tool identity and usage totals, obtain fresh scoped credentials and use a new authenticated launch reference. +- Pending requests shall have no storage TTL. Cancellation or terminal cleanup shall close unanswered requests and preserve recorded decisions without extending an existing retention TTL. +- The guest shall reseed application PRNG state from OS entropy on `/run` and `/resume`; security-sensitive values shall continue to use cryptographic randomness. -**A snapshot is only as warm as the pages touched before it was captured (P2-F5, live 2026-08-07).** This is the defect that stopped the P2 smoke run one step short of a pull request, and it is a property of the substrate rather than a bug in any one file. Every task failed at turn 0, reproducibly: +The platform verifies ownership and the exact approved action. It does not implement a separate semantic relevance checker; the agent decides whether the proposed work still makes sense. -``` -TimeoutExpired: Command '['claude', '--version']' timed out after 10 seconds -``` +### 3. Packaging: same agent image source, new build path -The binary was fine — in the identical image, locally, `claude --version` answers `2.1.191 (Claude Code)` in under a second. It is a **225 MiB (236,305,136-byte) statically-linked ELF** that nothing had exec'd before the snapshot was taken, so on a guest restored ~50 s earlier the first `exec` had to fault all of those pages in from lazily-restored storage, and 10 s was not enough. `/ready` existed precisely so "the snapshot is taken with a warm server", and the snapshot was warm for uvicorn and stone cold for the binary that does all the work. +Package the existing ARM64 `agent/Dockerfile` and its local inputs into a deterministic ZIP. Managed builds use `microvm-images/agent-artifact-.zip`; deploy the digest with the managed base-image ARN/version. A changed object URI triggers CloudFormation to build a new image version. The first deployment may create only infrastructure so the artifact bucket exists before upload. An external image identifier supports out-of-band builds. -Both halves of the fix are kept, because they answer different questions. `/ready` now **exec's the heavyweight binaries before returning 200** (`claude` required, `git`/`node` best-effort), which is the only mechanism that can make the shipped snapshot warm — and its own budget rises to 300 s, well inside the 3600 s build-hook window, because it now does work whose duration is a cold `exec`. Two structural rules keep that honest, because **per-command timeouts do not compose**: the required command runs FIRST with its own budget so no best-effort warm-up can starve the one that decides whether the snapshot is usable, and the best-effort ones then SHARE the remainder of a total warm-up ceiling that sits inside the hook budget with margin (240 s against 300 s). Without them, three commands at 120 s each would be 360 s — a fix for a runtime failure that produces a build failure instead — and a single hung optional command could hold up a 200 that the required warm-up had already earned. Separately, the version probe's timeout goes from 10 s to 60 s: a probe that exists to print a version string into a log line gains nothing from a tight bound and loses the whole task when it trips. The general rule this generalises to, and the reason it belongs in the ADR rather than only in a comment: **on this backend, a first-touch cost that other substrates pay during container start is deferred to the first task instead**, so anything large and lazily-loaded is a turn-0 hazard unless it is touched in `/ready`. +All six hooks share the FastAPI listener on port 8080. AWS hook properties accept `ENABLED`/`DISABLED`, not route paths; the architecture enum is `ARM_64`. -**Payload delivery — v2 amendment (2026-09-13, #817/#700).** The local implementation now shares an authenticated bootstrap protocol between ECS and MicroVM. This replaces the original inline/S3-pointer protocol and broad worker payload-bucket read grant. The historical P1/P2 runs below predate this amendment; AWS authorization, networking, expiry and coordinated deployment validation remain pending. The repository runbook is `docs/verification/645-payload-bootstrap.md`. +| Hook | Contract | +|---|---| +| `/ready` | Required when runtime hooks are enabled; execute required binary warm-up before snapshot capture | +| `/validate` | Check local readiness, routes and configuration contracts without AWS calls | +| `/run` | Authenticate/install launch configuration and start the pipeline asynchronously | +| `/terminate` | Close the local coding barrier, log and acknowledge any request body; do not join the pipeline or write terminal task status | +| `/suspend` | Drain acknowledged progress and commit the current safe checkpoint within the hook budget | +| `/resume` | Renew credentials and reconcile the original gate before releasing coding | -Every task uses S3. The coordinator publishes a non-secret deployment manifest at `bootstrap/.json`, conditionally creates `/payload.json`, and sends a short-lived signed URL for that one task object. The worker reads the manifest with its own AWS role, whose policy explicitly denies object reads outside its deployment's `bootstrap/*` and denies payload-bucket listing. The explicit deny also prevents a foreign public bucket from authenticating a forged manifest. The task document must name the reference's task ID and contain the exact manifest configuration before the MicroVM installs any of it. +Warm-up budgets come from `contracts/constants.json`; the total guest budget must remain below the image hook timeout. The [P2 runbook](../verification/645-p2-smoke-runbook.md) records the cold-binary failure and the enum/trust/logging fixes. Historical binary sizes and timings are observations of those images, not sizing guarantees for later builds. -The `runHookPayload` cap remains **4,096 bytes**, measured against the service despite the SDK's documented 16,384. It now applies to the serialized reference rather than choosing an inline branch. Payloads are capped at 8 MiB and manifests at 16 KiB; the bounds and version are shared in `contracts/constants.json`. +**Authenticated v2 payload transport.** Every ECS and MicroVM task uses S3. The coordinator publishes a deployment manifest, conditionally creates the task payload and sends a short-lived signed URL for that one object. The serialized MicroVM reference must fit 4,096 bytes; payloads are bounded at 8 MiB and manifests at 16 KiB. -| Location | v2 shape | +| Location | Shape | |---|---| -| `runHookPayload` string / ECS `AGENT_PAYLOAD_REF` | `{version:2, task_id, bootstrap_s3_uri, payload_url, expires_at}` | +| Hook reference / ECS `AGENT_PAYLOAD_REF` | `{version:2, task_id, bootstrap_s3_uri, payload_url, expires_at}` | | `bootstrap/.json` | `{version:2, backend, platform_config}` | | `/payload.json` | `{version:2, task_id, agent_payload, platform_config}` | -| Private `/launch.json` | `{fingerprint, reference}`; coordinator-readable only, never on the agent-readable task row | - -The coordinator saves the exact reference for replay: re-signing would change a Run request with the same client token. Task objects use conditional creation and read-back recovery after a lost committed write. URL lifetime is at most 900 seconds, bounded by known credential expiry, and initial creation requires at least 300 seconds. Expired saved references fail without re-signing. The coordinator needs bucket-scoped `ListBucket` so an absent launch record returns `NoSuchKey` rather than `AccessDenied`. It deletes both task objects at finalization; one-day lifecycle expiry is the backstop. Manifests are refreshed with identical bytes at preparation and may coexist across configuration revisions until lifecycle cleanup. +| Private `/launch.json` | `{fingerprint, reference}` for coordinator replay | -The image accepts only v2. The old inline/pointer forms and baked-environment fallback are rejected. A coordinated drain and switch of coordinator, images and application IAM is required; there is no new KMS/SSM resource or bootstrap bundle version. Signed URLs are bearer capabilities and must never appear in task rows, ordinary logs or repository subprocess environments. ECS consumes/removes its reference before importing the task pipeline. The URL consumer restricts requests to the exact task object on regional S3 HTTPS, disables redirects/proxies and bounds response bytes. AWS verifies signatures and actual expiration. +The worker authenticates only its deployment’s `bootstrap/*` using ambient credentials. Other payload-bucket object reads and listing are explicitly denied; signed task downloads carry coordinator authorization. The task ID and configuration must exactly match the reference and authenticated manifest before installation. MicroVM `platform_config` contains allowlisted non-secret identifiers, including role/secret ARNs; it does not contain credentials. ECS already receives its deployment configuration through task settings, so its manifest config is empty. -**Platform configuration delivery (P2, strengthened by v2).** A MicroVM restores a snapshot, including its process environment, so current deployment identifiers must arrive at task launch. The authenticated manifest and downloaded document both contain `platform_config`: non-secret table/bucket names, secret ARNs and session-role ARN. Per-task fields such as `memory_id` remain in `agent_payload`. ECS's manifest config is empty because ECS already receives deployment settings through its task definition and coordinator overrides. +The coordinator saves the exact reference for idempotent replay. Signed URLs last at most 900 seconds, bounded by known credential expiry, and initial creation requires at least 300 seconds. An expired saved reference fails without re-signing the same launch request. Finalization deletes task payload/launch objects; one-day asynchronous S3 expiry is the backstop. -MicroVM installs only allowlisted keys and rejects unknown keys before installing any. Required identifiers must be nonblank; malformed/control-character values and inconsistent ARN partitions/accounts are rejected. Legitimate cross-region secrets remain allowed when present in the trusted manifest. Account agreement alone cannot authenticate a sibling ARN; v2's manifest/config comparison supplies the provenance check. No-config envelopes are rejected regardless of the image's existing environment. - -Manifest read and signed payload download precede configuration installation. Secrets, task-scoped credential setup, CloudWatch initialization and the pipeline follow installation; pre-install diagnostics use stdout without unsafe exception chains. Build hooks remain AWS-silent. The settings allowlist and bootstrap bounds are cross-language contracts with drift checks. These source guarantees do not establish complete hostile-worker isolation: roles still retain other platform grants and choose session tags, and a stolen signed URL remains usable until expiry or revocation. - -**No orchestrator→agent HTTP path exists in P1–P3**: payload arrives through the `/run` hook, all agent work is outbound, and therefore **no JWE auth tokens are minted at all** — token minting (and its ≤ 60 min TTL refresh problem) is deferred until a real consumer exists (e.g. operator shell access, [#391](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/391)). The `endpoint` stays in the `SessionHandle` because it is genuinely per-session state that becomes load-bearing the day such a consumer appears. But note the service does not agree by default: omitting `ingressNetworkConnectors` on `RunMicrovm` attaches a **public** `HTTP_INGRESS` connector, so the strategy passes the Lambda-managed `NO_INGRESS` connector explicitly on every launch (see sub-decision 4's security table). +Normative requirements (EARS): -**Constraint accepted:** the configured **baseline is 8 GiB RAM / 4 vCPU** and the service scales vertically to a **32 GiB / 16 vCPU peak** on its own, with 32 GB of disk. So baseline capacity and additional burst usage are billed separately — good for the bursty compile-and-test shape of an agent task — but the SUSTAINED ceiling is still 32 GiB, so repos that motivated the 120 GB ECS sizing stay on `ecs`. What the construct configures (and validates) is the baseline; the peak is not something a deployment asks for. +- Image builds shall not embed credentials, task identity or deployment-specific configuration. Build hooks shall avoid AWS clients and credential caches. +- The worker shall reject legacy envelopes, oversized/malformed data, mismatched provenance and unknown configuration keys before starting a pipeline or installing configuration. +- Before configuration installation, the worker shall make only the bootstrap manifest/payload reads and shall log diagnostics to stdout. +- Signed URLs shall not appear in ordinary logs, agent-readable task rows or repository subprocess environments. Downloads shall use the exact regional S3 HTTPS object without redirects/proxies and with bounded response sizes. +- Producers, images and IAM shall be upgraded together; incompatible workers must be drained before switching transport. -Normative requirements (EARS). **Each requirement's own `(Pn)` tag is authoritative**; there is no blanket phase for the list. The tags are per-requirement because the original list *was* split P1/P2 on the assumption that a hook could be declared in one phase and served in a later one — which the service does not permit (see the phasing table above), so the hook-serving requirements collapsed into P1 while the P2 items below arrived with the P2 hooks and `platform_config`: +The [payload contract](../verification/645-payload-bootstrap.md) and [live checks](../verification/645-p2-payload-live-20260914.md) record the implementation and validation. This transport does not establish complete hostile-worker isolation: other platform grants remain, and a stolen signed URL is usable until expiry or revocation. -- (P1) The image build shall not embed secrets, tokens, or per-task identity in the snapshot. -- (P1, amended by v2) Every task shall use the authenticated manifest and single-object signed payload reference; the serialized hook reference shall not exceed 4,096 bytes. -- (P1, amended by v2) The worker role shall read only deployment bootstrap manifests with ambient credentials; other object reads and payload-bucket listing shall be explicitly denied. The coordinator shall own task-object writes, signing, replay and deletion. -- (P1) Where no ingress is configured for a deployment, the strategy shall pass the Lambda-managed `NO_INGRESS` network connector on every `RunMicrovm` call (the field shall not be omitted). -- (P1) Where the image enables any MicroVM lifecycle hook, the image shall also enable the `/ready` hook and the agent shall serve it. -- (P1) When the `/run` hook receives the task payload, the agent shall validate it, start the pipeline asynchronously, and return HTTP 200 within the hook budget. -- (P1, clarified) Invalid hook references shall return a structured client error; unreadable manifest/payload bytes shall return a structured server error. Neither path shall install configuration or start a pipeline. -- (P1) The agent shall not execute the clone→verify→PR pipeline on the hook path. -- (P1) The agent shall resolve credentials at `/run` time. -- (P1) Where a deployment configures a MicroVM image before smoke parity is verified, the platform shall warn that the backend has no smoke-parity guarantee. -- (P2) Where the image declares the `/validate` build hook, the agent shall serve it. -- (P2) The `/validate` hook shall make no AWS API calls. -- (P2) When the `/run` hook receives `platform_config`, the agent shall install only allowlisted keys into the environment before pipeline initialization. -- (P2) If `platform_config` carries a key that is not on the allowlist, then the agent shall reject the run with a 400 and shall install none of the block's keys. -- (P2) If a required `platform_config` key is missing, then the agent shall reject the run with a 400. -- (P2) Where a `platform_config` value and an image-baked environment value disagree, the agent shall use the `platform_config` value. -- (P2) Until `platform_config` is installed, the `/run` hook shall make no AWS API call other than the payload fetch, and shall log to stdout only. -- (P2) Where the image declares the `/terminate` hook, the agent shall return 200 within the hook budget for any request body — including a malformed, empty or absent one — and shall not write terminal task status. -- (P2) When the `/ready` hook runs, the agent shall exec the agent CLI binary before returning 200, so that its pages are resident when the snapshot is captured. -- (P2) If a required `/ready` warm-up does not complete successfully, then the agent shall report not-ready (HTTP 503) rather than allow the snapshot to be taken. -- (P2) The `/ready` hook shall make no AWS API call, warm-up included. +No ABCA endpoint consumer exists in P1–P3. The platform grants no `CreateMicrovmAuthToken` permission and mints no JWE tokens. `NO_INGRESS` can still return an endpoint URL; an unauthenticated 403 verifies the authentication boundary, not valid-token reachability. ### 4. Infra and IAM: conditional resources behind bootstrap `ComputeTypes` -Mirroring the ECS pattern: a `compute-lambda-microvm` bootstrap policy (`cdk/src/bootstrap/policies/`) gated on the `ComputeTypes` CFN parameter; a CDK construct provisioning the build role, execution role (admitted to the per-session role via `AgentSessionRole.admitComputeRole`, which was designed for exactly this), the S3 artifact bucket wiring, and image build automation. Egress uses the platform VPC via egress network connectors so the DNS Firewall / security-group / flow-log stack in [COMPUTE.md](../design/COMPUTE.md) applies unchanged; ingress is suppressed with the Lambda-managed `NO_INGRESS` connector (no `SHELL_INGRESS` — it is noted as a candidate for [#391](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/391), operator session access, as a separate decision). - -Two networking facts the construct has to encode, both established live: +The backend adds build/runtime VPC connectors, build artifacts, launch payloads, logs, roles and a managed or external image. Its bootstrap policy is conditional on `ComputeTypes` including `lambda-microvm`. A VPC egress connector requires an operator role. Build egress permits ports 80/443 for package installation; runtime egress permits 443 through the platform VPC. -- **A `VPC_EGRESS` connector requires an operator role.** CloudFormation's generated L1 types `operatorRole` as optional and this ADR originally assumed Lambda would manage the ENIs with its own service-linked role. It does not: the connector fails to create with *"NetworkConnectorOperatorRole is required for VPC_EGRESS connector type"*. The construct creates one role — trusting the bare `lambda.amazonaws.com` service principal (see the trust-policy fact below), carrying `AWSLambdaVPCAccessExecutionRole` plus the ENI / tag / private-IP actions that policy omits — and shares it across both connectors, since it manages interfaces rather than traffic. -- **The MicroVM-facing roles cannot carry a confused-deputy source condition.** All three (build, execution, connector operator) trust the bare `lambda.amazonaws.com` service principal with **no** `aws:SourceAccount` / `aws:SourceArn`, and that is a forced choice, not an oversight: the Lambda MicroVMs service presents no source key when it assumes them, so a trust policy carrying one is unassumable. Two symptoms of the one cause, both live 2026-08-06/07 and both blocking: +**Trust and PassRole limitation.** Recorded live checks rejected `aws:SourceAccount`/`aws:SourceArn` conditions on the MicroVM-facing roles and `iam:PassedToService` on the MicroVM PassRole paths. The working roles trust `lambda.amazonaws.com` without those conditions; build/execution roles also allow `sts:TagSession`. IAM simulation with caller-supplied condition values did not prove that the service supplied those values. See the [P2 evidence](../verification/645-p2-smoke-runbook.md), [clean deployment](../verification/645-p2-clean-deployment-20260913.md) and [effective IAM checks](../verification/645-effective-iam-20260915.md). Reintroduce a condition only after verifying service support. - - Both `AWS::Lambda::NetworkConnector` resources `CREATE_FAILED` **deterministically** (on a freshly deleted stack, so not propagation lag — which matters, because *"The service is unable to assume the provided NetworkConnectorOperatorRole. Please verify the trust policy on the role."* is also the classic propagation symptom and a re-run is the obvious wrong guess). Removing the condition → both created within a second. - - `RunMicrovm` failed with a **misleading `iam:PassRole` AccessDenied on the caller**, with the orchestrator's grant present, `simulate-principal-policy` returning `allowed`, no permissions boundary, and a temporary *unconditioned* `iam:PassRole` **also** denied. The real cause was the execution role's trust; removing its conditions made the next submission reach `RUNNING` in 6 s. So the service reports a role it cannot pass-and-assume as an identity-policy denial on the principal passing it. - - Recorded plainly because the fix looks like a regression to anyone applying the standard service-principal pattern — and because it *was* a regression in the other direction: P1's standalone-validated operator-role probe had no conditions and worked, and the P1 F2 fix then added them "to mirror the build/execution roles". `sts:TagSession` stays: the service needs both actions and it was never implicated. - -- **Neither can the `iam:PassRole` grants carry an `iam:PassedToService` condition — same root cause, identity side (P2r2-F9 + P2r2-F10, live 2026-08-07 run 2).** An earlier revision of this ADR recorded the opposite, that the identity-side condition "was exonerated" by run 1's elimination. **That was a false negative**, and its cause is worth recording because it is a general trap: run 1 tested the conditioned grant by *adding* a temporary unconditioned `iam:PassRole` and watching the task still fail — but the temporary grant remained attached through the later submissions that succeeded, so the conditioned grant was never once tested against a working trust policy. A contaminated control. - - Run 2 ran the clean experiment — same exact-ARN resource, same ~5-minute IAM settle, one variable. It removed the run-1 workaround **first** (submission 4: denied) and only then added the unconditioned grant back on the same resource (submission 5: `RUNNING`), which is the ordering run 1 got wrong: - - | Orchestrator `iam:PassRole` on the execution role | Result | - |---|---| - | exact ARN **+ `iam:PassedToService: lambda.amazonaws.com`** | **DENIED** (two independent submissions) | - | exact ARN, **no condition** | **`RUNNING` in 9 s** | - - The denial lands on the **caller**, which is what makes it so misleading — the statement names that exact ARN and `simulate-principal-policy` answers `allowed`: - - ``` - User: arn:aws:sts:::assumed-role/backgroundagent-dev-TaskOrchestratorOrchestratorFn-… - is not authorized to perform: iam:PassRole on resource: - arn:aws:iam:::role/backgroundagent-dev-LambdaMicrovmComputeExecutionRo-… - because no identity-based policy allows the iam:PassRole action - ``` - - And the same key blocks the *other* PassRole path, which run 1 never reached because the enum defect (P2-F2) stopped it earlier: CloudFormation could not pass the **build role** at `CreateMicrovmImage` under the bootstrap `infrastructure` policy's allowlisted `IAMPassRole`. Verbatim, so the diagnosis does not have to be taken on trust: - - ``` - LambdaMicrovmComputeImage… CREATE_FAILED - User: arn:aws:sts:::assumed-role/cdk-hnb659fds-cfn-exec-role--us-east-1/AWSCloudFormation - is not authorized to perform: iam:PassRole on resource: - arn:aws:iam:::role/backgroundagent-dev-LambdaMicrovmComputeBuildRoleF0-… - because no identity-based policy allows the iam:PassRole action - (Service: LambdaMicrovms, Status Code: 403) - ``` - - Three pieces of evidence pin that to the *condition* rather than to a stale bootstrap or a wrong resource pattern: - - 1. the live `IaCRole-ABCA-Infrastructure` policy was byte-identical to this branch's `cdk/bootstrap/policies/infrastructure.json`, so `cdk bootstrap --force` would have changed nothing; - 2. `aws iam simulate-principal-policy --policy-source-arn --action-names iam:PassRole --resource-arns ` returned `allowed` **with** `--context-entries ContextKeyName=iam:PassedToService,ContextKeyValues=lambda.amazonaws.com,ContextKeyType=string` and `implicitDeny` with no context entry — so the resource pattern matches and the condition key is the only remaining variable; - 3. the **control**: the out-of-band `create-microvm-image` call passed the *same build role* to the *same service* successfully, using operator credentials that carry no such condition. The role's trust is therefore fine and the denial is genuinely caller-side. - - So: **the Lambda MicroVMs service presents no usable value for `iam:PassedToService` on either PassRole path** (CloudFormation → build role at `CreateMicrovmImage`; orchestrator → execution role at `RunMicrovm`), exactly as it presents no `aws:SourceAccount` on the assume-role path. One root cause, two more symptoms. Both statements therefore drop the condition, and the fix is deliberately asymmetric so it stays contained: - - - `task-orchestrator.ts` sid `MicrovmPassExecutionRole` — condition removed; the **exact execution-role ARN** is now the whole of the scoping, which is why that resource must never be relaxed to a prefix or `*`. - - a new sid `MicrovmPassRoles` in the **conditional** `compute-lambda-microvm` bootstrap policy — unconditioned `iam:PassRole` on the build- and connector-operator role **name prefixes only** (not the execution role, which CloudFormation never passes). The shared `infrastructure` `IAMPassRole` keeps its allowlist, so no other role in the stack loses that constraint, and an agentcore-only bootstrap never gains an unconditioned pass at all. **Operators must re-bootstrap** (bundle ≥ 1.6.0) for the CDK-managed image path to work. - - If AWS documents the value the service does present, adding it to both statements restores the condition. `microvms.lambda.amazonaws.com`, `lambda-microvms.amazonaws.com` and `microvms.amazonaws.com` were all `implicitDeny` against the conditioned policy, so any one of them would serve as the allowlist entry if it turns out to be right. Note that CloudTrail carries **no `lambda-microvms` management events at all** today, so the value cannot be read out of a log — only confirmed by AWS or found by a bounded sweep. - - Compensating controls, enumerated per role. The deployment-role's shared `IAMPassRole` grant is **name-prefix-scoped**, so it technically covers all three roles; in practice only the orchestrator actively invokes `iam:PassRole` on the execution role (CloudFormation never requests it). The other two roles are passed **to** the deployment role (not to themselves). - - | Role | Who can pass it, and how that grant is scoped | - |---|---| - | **Execution role** | The **orchestrator Lambda only**, at `RunMicrovm` — `iam:PassRole` scoped to this role's **exact ARN**, no condition (`constructs/task-orchestrator.ts`, sid `MicrovmPassExecutionRole`). The referenced comment contains the authoritative two-arm experiment evidence that the condition is the true blocker (not a permissions gap or stale bootstrap). | - | **Build role** | The **CloudFormation deployment role**, at `CreateMicrovmImage` (the L1's `buildRoleArn`) — via the new `MicrovmPassRoles` statement, scoped to `role/backgroundagent-dev-LambdaMicrovmComputeBuild*`, no condition. Also whoever runs `package-microvm-artifact.sh --create-image` out of band, using their own credentials. | - | **Connector operator role** | The **CloudFormation deployment role**, at `AWS::Lambda::NetworkConnector` create/update (`operatorRole`) — the same statement, scoped to `role/backgroundagent-dev-LambdaMicrovmComputeConnector*`. | - - The rest of the posture: every resource these roles can reach is account-scoped by ARN **except two deliberate `Resource: '*'` statements** — `ec2:DescribeAvailabilityZones` on the execution role (EC2 describe actions have no resource-level scoping; read-only, no mutation, no data access, needed so a CDK target repo's `cdk synth` build gate can resolve AZ context on a fresh clone) and the connector operator role's ENI/tag/private-IP statement (`CreateNetworkInterface` is authorized before the ENI exists and the `Describe*` calls take no resource, which is why the AWS-managed VPC-access policy uses `*` too). Both are justified in the construct's cdk-nag `AwsSolutions-IAM5` suppressions, which is where a reviewer should check them rather than here. The Logs grants are prefix-scoped (`/aws/lambda-microvms/*` plus one named log group), i.e. wildcards inside a namespace, not `*`. Separately, the **orchestrator's** `lambda:PassNetworkConnector` is also `Resource: '*'` and unavoidably so — the AWS-managed connectors live in the `aws` account, outside any ARN we could enumerate (justified in `task-orchestrator.ts`, sid `MicrovmPassNetworkConnector`). Finally: none of the three roles holds `iam:*`, none has cross-account trust, and the only `sts:AssumeRole` any of them has is the execution role's, scoped to the per-task SessionRole. - - If AWS later populates a source key on this path, adding it to the shared principal fixes all three roles and both `sts` actions at once. - -- **Build-time egress needs port 80; runtime does not.** `agent/Dockerfile` installs Debian packages and `apt-get` fetches over plain HTTP, so a 443-only egress path fails every snapshot build (`Could not connect to deb.debian.org:80 … exit code: 100`). Rather than widen the runtime posture, the construct provisions a **second, build-only** connector on the same private subnets with a 443 + 80 security group, referenced solely by the image resource and the packaging script. The agent at run time still has 443-only egress. - -- Where the bootstrap `ComputeTypes` parameter includes `lambda-microvm`, the generated template shall attach the `IaCRole-ABCA-Compute-LambdaMicrovms` policy to the CloudFormation execution role. -- The orchestrator role shall receive only the MicroVM lifecycle actions it calls (`lambda:RunMicrovm`, `lambda:SuspendMicrovm`, `lambda:ResumeMicrovm`, `lambda:TerminateMicrovm`, `lambda:GetMicrovm` for `pollSession`, and `lambda:PassNetworkConnector`, which is required even for the default connectors), scoped to platform-created images. -- Where the `lambda-microvm` backend is enabled, the approve and deny Lambdas shall receive `lambda:ResumeMicrovm` and `lambda:GetMicrovm` — conditionally, mirroring the cancel handler's conditional `RUNTIME_ARN` wiring in `task-api.ts`. -- The trust policy of every MicroVM-facing role shall name `lambda.amazonaws.com` and shall carry no source-condition key (the service presents none; see the trust-policy fact above). -- The `iam:PassRole` grant the orchestrator uses for the MicroVM execution role shall carry no `iam:PassedToService` condition and shall be scoped to that role's exact ARN. -- Where the bootstrap `ComputeTypes` parameter includes `lambda-microvm`, the `IaCRole-ABCA-Compute-LambdaMicrovms` policy shall grant `iam:PassRole` without an `iam:PassedToService` condition, scoped to the MicroVM build- and connector-operator role name prefixes, and shall not extend that grant to the MicroVM execution role. -- The shared `IaCRole-ABCA-Infrastructure` `iam:PassRole` statement shall retain its `iam:PassedToService` allowlist. -- The MicroVM execution role shall hold `logs:CreateLogStream` and `logs:PutLogEvents` on the application log group whose name is delivered in `platform_config`, scoped to that log group. - -`lambda:CreateMicrovmAuthToken` is granted to no role in P1–P3 (no JWE consumer exists; see sub-decision 3). +| Role/action | Scope and responsibility | +|---|---| +| Coordinator lifecycle APIs | Configured image ARN and its version-qualified sibling; includes actual-version capability lookup | +| Coordinator `iam:PassRole` | Exact execution-role ARN, without `iam:PassedToService` | +| Deployment `iam:PassRole` | Backend-specific build/operator role name patterns; shared infrastructure allowlist remains intact | +| Approval/denial handlers | Observe/resume the configured image after committing a decision; dispatch parked continuations | +| Build role | Selected immutable artifact plus manual-build key; MicroVM log writes | +| Execution role | Bootstrap manifests, startup secrets, allowlisted models, Memory and logs; tenant data through the per-task SessionRole | +| Connector operator | Tested ENI/tag/private-IP permissions plus AWSLambdaVPCAccessExecutionRole | -**Cost attribution.** `cdk/src/main.ts` currently tags the whole stack with a single `compute_type` context value (default `agentcore`) — already imprecise with two backends, wrong with three. P1 must add backend-identifying cost-allocation tags on the MicroVM-specific resources (images, payload/artifact bucket wiring, log groups) and revisit the stack-level tag semantics (e.g. a `compute_types` list), keeping attribution consistent with [#645](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/645)'s cost/attribution acceptance criterion. +`lambda:PassNetworkConnector` has no resource-level authorization support and therefore uses `Resource: *`. The operator role also has wildcard ENI permissions; some mutations can be scoped in IAM, so their current tested wildcard is not evidence that narrower permissions are impossible. DescribeAvailabilityZones needs a wildcard for fresh CDK repository lookups. These exceptions and namespace wildcards are documented in the construct’s cdk-nag suppressions. -- Where a deployment enables the `lambda-microvm` backend, MicroVM-specific resources shall carry backend-identifying cost-allocation tags. +**Nested infrastructure.** `microvm_nested_stack=true` puts MicroVM resources in a nested stack. The shared execution role stays in the parent to avoid a role-trust dependency cycle. Bootstrap 1.9.0 covers nested deployment roles. Existing flat installations need the reviewed overlap/drain migration, explicit image/coordinator pins and rollback checks in the [nested migration record](../verification/645-p3-nested-stack.md). Moving construct paths alone is not a safe migration. Preserve `microvm_resource_name_prefix` after a migrated deployment. #### Security bar vs existing backends ([#645](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/645) acceptance criterion) -| Control | AgentCore | ECS | Lambda MicroVMs | Delta | -|---|---|---|---|---| -| Egress, runtime (DNS Firewall, TCP 443 SG, flow logs) | Platform VPC | Platform VPC | Platform VPC via egress network connector | None | -| Egress, image build | ECR build outside the platform VPC | ECR build outside the platform VPC | Platform VPC via a **separate build-only connector, TCP 443 + 80** (`apt-get` is plain HTTP) | New surface: build-time egress is wider than runtime egress by one port, on a connector no running MicroVM can use | -| Tenant-data scoping | Per-session role (`admitComputeRole`) | Per-session role | Per-session role, execution role admitted identically | None | -| Secrets delivery | Runtime env + Identity injection | Task env vars | Fetched at `/run`; never in snapshot | New surface: snapshot must stay secret-free (EARS req., sub-decision 3) | -| Non-secret platform config (table/bucket names, secret + role ARNs) | Runtime env vars | Task env vars | `platform_config` in the `/run` payload, installed into the process env | New surface: the values are attacker-relevant *as env vars* (`LD_PRELOAD`, `AWS_ENDPOINT_URL`), so the agent installs a fixed **allowlist** and rejects the whole run on any other key (EARS req., sub-decision 3) | -| Inbound exposure | None (SigV4 invoke only) | None (no endpoint) | **None — but only because the strategy passes `NO_INGRESS` explicitly.** The service default is a PUBLIC `HTTP_INGRESS` connector plus a public `*.lambda-microvm..on.aws` endpoint; no tokens are minted in P1–P3 either way | New surface **and** a new failure mode: "no inbound" is an active control, not an absence. Drop the `NO_INGRESS` argument and every agent MicroVM gets a public endpoint (EARS req., sub-decision 3) | -| IAM condition keys on the compute-role trust **and** on the `iam:PassRole` grants that hand it over | Trust pinned with `aws:SourceAccount`; `PassRole` under the allowlisted bootstrap statement | Trust pinned per-service; `PassRole` under the allowlisted bootstrap statement | **Neither is possible.** All three MicroVM-facing roles trust the bare `lambda.amazonaws.com` with no `aws:SourceAccount`/`aws:SourceArn`, **and** both `iam:PassRole` grants (orchestrator → execution role at `RunMicrovm`; CloudFormation → build role at `CreateMicrovmImage`) carry no `iam:PassedToService` — the service presents no usable value for any of those keys, and each condition is a hard blocker while present (live-verified, blocking, four times across two runs) | **Real, evidenced gap that does not close from our side, and it is wider than the trust policy alone.** `lambda.amazonaws.com` is shared with every other Lambda feature, so neither the account pin nor the passed-to-service pin is available on this path. Compensated per role (table in sub-decision 4): the **execution** role is passable by the **orchestrator only** (at `RunMicrovm`), restricted to its **exact ARN**; the **build** and **connector-operator** roles are passable by the CloudFormation deployment role under a new **conditional, per-backend, name-prefix-scoped** statement (`MicrovmPassRoles`, bootstrap ≥ 1.6.0) that deliberately excludes the execution role. The shared allowlisted `IAMPassRole` (`role/backgroundagent-dev-*`) is left intact to avoid widening the grant for ~30 other roles, so while it technically matches the execution role, only the orchestrator actively reaches for it. Resources are account-scoped by ARN apart from two justified `Resource: \'*\'` statements (`ec2:DescribeAvailabilityZones`; the operator role\'s ENI management — both carry cdk-nag IAM5 suppressions). No `iam:*`, no cross-account trust. Revisit if AWS ever documents the values the service presents; CloudTrail records no `lambda-microvms` events, so they cannot be read from logs | -| Per-task observability writes | Runtime writes to the vended APPLICATION_LOGS group | Task role writes to the task log group | Execution role writes to the SAME APPLICATION_LOGS group, granted against the group `platform_config` names (P2-F4) | None — but only after P2-F4: the name was delivered a phase before the grant, so the agent attempted the write and every per-task line (and `METRICS_REPORT`) was `AccessDenied`, degrading silently to guest stdout | -| Session isolation | MicroVM | Task-level | MicroVM (Firecracker) | None (≥ ECS) | -| State reuse | None | None | Snapshot shared across MicroVMs | New surface: application PRNG reseed + credential refresh on `/run`/`/resume` — **P3 scope** (P2 exposure measured as negligible: sole `random` consumer is a ULID sort key under a `task_id` partition; `os.urandom`/`secrets` unaffected; no credential derives from `random`). Credential refresh IS in P2: per-task credentials arrive via `platform_config` at `/run` | -| Workload-token injection | Yes (Runtime-coupled) | No (env-var posture) | No (env-var posture) | Shared with ECS; deferred to [#249](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/249)/ADR-016 | -| Operator shell access | No | No | Not enabled (`SHELL_INGRESS` omitted; candidate for [#391](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/391)) | None by default | -| Auth-token minting | n/a | n/a | `CreateMicrovmAuthToken` granted to no role in any phase | Verified static-only: any principal holding the action *can* mint a working JWE, including against a `SUSPENDED` MicroVM, so the posture rests entirely on the grant being absent | +| Control | MicroVM posture | Difference to account for | +|---|---|---| +| Runtime egress | Platform VPC, DNS Firewall, HTTPS security group, flow logs | Separate build connector additionally allows HTTP | +| Tenant access | Task-scoped SessionRole | Shared compute-role permissions remain outside that boundary | +| Configuration/secrets | Authenticated runtime identifiers; credentials resolved after launch | Shared snapshot must not capture credentials or task identity | +| Ingress | Explicit NO_INGRESS; no platform token minting | Service defaults would select HTTP_INGRESS if the field were omitted | +| Trust/PassRole | Exact role/image scopes where supported | Source-condition limitations described above | +| Logs | Service image group plus platform APPLICATION_LOGS | Both namespaces need explicit grants | +| Retained work | Versioned, bounded, checksum-verified checkpoint | Protect stored conversation/files and clean terminal state | +| Workload identity | Runtime credentials and task-role refresh | Linear vault identifiers arrive through `platform_config`; the compute execution role mints tokens ([setup](../guides/LINEAR_SETUP_GUIDE.md#using-the-vault-with-lambda-microvms)) | + +MicroVM-specific resources carry `abca:compute-backend=lambda-microvm` cost tags. A stack-wide compute tag cannot accurately attribute a mixed-backend deployment by itself. #### Regional availability enforcement -Lambda MicroVMs launched in 5 regions (us-east-1/2, us-west-2, eu-west-1, ap-northeast-1) and will expand. ABCA is a single-region deployment, so the constraint is binary per stack: either the stack region supports the backend or the backend does not exist there. Enforcement is layered — one static check where offline determinism is required, live probes everywhere else so the platform self-heals as AWS adds regions (`list-managed-microvm-images` is the documented read-only availability probe): - -| Stage | Mechanism | Check | -|---|---|---| -| CDK synth/deploy | Static region constant (single exported list, documented update path) | Synth fails when `ComputeTypes` includes `lambda-microvm` in an unlisted region; context-flag escape hatch for newly launched regions ahead of the constant update | -| Repo onboarding | Live probe from the CLI | `bgagent repo onboard --compute-type lambda-microvm` calls `list-managed-microvm-images` in the stack region and rejects with a remedy (supported-region list + suggest `agentcore`/`ecs`) | -| `bgagent platform doctor` | Live probe (precedent: `checkBedrockModel`) | Reports backend availability for the stack region whenever any active blueprint selects `lambda-microvm` | -| Orchestration (defense in depth) | Error classification | `startSession` failures from a missing regional endpoint classify to a typed remedy in `error-classifier.ts`, never a cryptic SDK error on the task | +The launch-region list is us-east-1, us-east-2, us-west-2, eu-west-1 and ap-northeast-1. New Regions can be supported before the static list is updated: -- If an operator onboards a repo with `compute_type: 'lambda-microvm'` and the availability probe fails for the stack region, then the CLI shall reject the onboarding with the supported-region list and alternative backends as the remedy. -- If `startSession` fails because the MicroVM service is unavailable in the stack region, then the orchestrator shall classify the failure with a configuration remedy and shall not retry. -- When the platform doctor runs in a deployment where any active blueprint selects `lambda-microvm`, the doctor shall probe MicroVM availability in the stack region and report the result. +- Synth rejects a concrete unlisted Region unless `microvm_region_override` is set. An unresolved Region defers to live checks. +- CLI onboarding and platform doctor probe `list-managed-microvm-images`. +- Runtime regional failures receive a configuration remedy instead of an opaque SDK error. ### 5. Rollout: phased, default unchanged -- **P1 — strategy + infra + minimal hook serving:** `LambdaMicrovmComputeStrategy` (start/poll/stop), CDK construct, bootstrap policy, types sync, unit + CDK assertion tests, and the agent's `/ready` + `/run` endpoints. No suspend yet. The image IS creatable and launchable and the payload DOES reach the agent — but there is **no smoke-parity guarantee** (sub-decision 3's phasing table). -- **P2 — smoke parity:** the agent serves `/terminate` + `/validate` and installs its platform env from the `/run` payload (see sub-decision 3's "Platform configuration delivery"); agent completes clone → change → PR on the backend with progress visible to `bgagent watch`; failure classification entries in `error-classifier.ts`; **AgentCore Memory parity** (IAM grant + `MEMORY_ID` delivery, following the `EcsAgentCluster` pattern — Memory is a standalone service already consumed cross-substrate, and omitting the grant silently no-ops cross-session learning); the agent's remaining non-secret env parity inside the snapshot. -- **P3 — suspend/resume:** the interface widening from sub-decision 1 (mandatory methods, all three strategies in one commit), HITL-wait suspend policy, inline resume in the approve/deny Lambda with orchestrator-poll reconciliation (sub-decision 2), timeout-under-freeze wall-clock handling; coordinate with [#491](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/491)'s unified liveness model and update Cedar decision #7's rationale note. -- **Out of scope:** replacing AgentCore as default; classic Lambda functions as a runtime; GPU; the Runtime-coupled workload-access-token injection path (delivery mechanism exists only on AgentCore Runtime; MicroVMs adopt the ECS env-var posture until [#249](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/249)/ADR-016 redesign the seam). Gateway integration is orthogonal: ADR-019/[#641](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/641) is substrate-portable by design and applies to this backend when it lands. - -## Consequences - -- (+) **Suspend/resume economics.** Tasks idling on approval waits stop billing compute while preserving full state — bounded at ~1 h per gate under the current Cedar ceiling (decision #6), and the enabler for cheap off-hours gate-ceiling extensions later (§14.8). Verified end to end at the substrate level: suspend and resume each complete in ~1 s, and `microvmId` **and** `endpoint` survive the cycle byte-identical, so a stored `SessionHandle` remains valid. -- (+) **VM-level isolation without cluster ops.** Firecracker isolation with no ECS cluster, task definition, or capacity management; one-session-per-MicroVM maps 1:1 onto ABCA's task model. -- (+) **Escapes AgentCore's 2 GB image limit and FUSE `flock()` workaround** — native disk in the snapshot supports `uv`/`mise` without the split-storage scheme. Note the size comparison must say WHICH measure it means: the same agent tree is 1.799 GB as an OCI image (629.7 MB compressed, i.e. under AgentCore's limit) but reports `codeInstallSizeInBytes` of 2.17 GiB as a MicroVM snapshot (i.e. over it). The two straddle the limit and are not interchangeable; memory/disk snapshot sizes are a third thing again and must not be summed into the comparison. -- (+) **Liveness becomes explicit.** Unlike AgentCore's stub `pollSession`, the strategy can report real substrate state, strengthening the [#491](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/491) unification. -- (−) **32 GiB sustained ceiling (8 GiB baseline + automatic 4× vertical scaling), 32 GB disk.** Not a successor to the ECS backend for heavy CI-parity builds; the platform now maintains three backends. -- (−) **Capacity is baseline-priced with burst headroom, which is a narrower promise than "32 GiB".** The deployment configures an 8 GiB / 4 vCPU baseline and the service scales to 32 GiB / 16 vCPU on demand — well matched to an agent task, which is idle-ish while waiting on the model and spiky during builds, and cheaper than reserving the peak. But it is *burst*, not a reservation: a workload that needs 32 GiB **sustained** is relying on scaling behaviour this ADR has not measured, and against ECS's 120 GB the gap for sustained-memory workloads is unchanged. So the value proposition remains the suspend economics, the observable control-plane state machine, and the absence of cluster ops — with capacity now a fair-to-good fit rather than a hard blocker. Repos with genuinely sustained heavy builds still belong on `ecs`. -- (−) **New packaging pipeline.** Zip + Dockerfile + service-side image builds with versioned snapshots (storage billed per version) alongside the existing ECR flow; image versions need lifecycle cleanup — including versions left behind by FAILED builds, and noting the last version of an image cannot be deleted individually (delete the image, which reaps it). -- (−) **The payload bucket is on the hot path, not the overflow path.** With a 4 KB `runHookPayload` cap, virtually every real task delivers its payload via S3, so the bucket, its TTL rule and the execution role's read grant are load-bearing for normal operation rather than an edge case (sub-decision 3). -- (−) **8-hour hard cap includes suspended time**, and with `idlePolicy` omitted there is no tighter substrate-level suspended-TTL — the suspended-state bound is `maximumDurationInSeconds` plus orchestrator termination and the stranded reconciler. A manually suspended VM was observed alive at 1 h with no TTL in sight (observation truncated there), so nothing contradicts this bound, but nothing narrows it either. Under today's 1 h gate ceiling it is comfortably sufficient; any future extension of gate ceilings must revisit the bound (an additive `idlePolicy` change) and give the orchestrator a checkpoint-and-restart path (push branch, new session) beyond the cap. -- (!) **Idle-policy foot-gun.** Traffic-based auto-suspend would freeze a busy outbound-only agent; the decision to disable auto-suspend must be enforced in code and covered by tests, not left to configuration discipline. -- (!) **Service defaults are not the desired posture.** Two live-caught cases (public `HTTP_INGRESS` by default; `/ready` mandatory) mean an omitted field on this backend does not mean "off" — it can mean "the service picks, and it picks wider than we want". Every new `RunMicrovm` / `CreateMicrovmImage` field should be assumed to have an opinionated default until checked. -- (!) **Nothing self-terminates on the paths that matter** — superseding P1's unqualified version of this bullet. With `run: ENABLED` the service DOES reap a VM whose run hook returns 4xx (~12 s, `stateReason: "Run lifecycle hook returned HTTP status 400."`, live-verified), so a guest that rejects its own payload cleans itself up. That is the only self-cleaning case: the service reaps a hook *result*, and once `/run` has answered 200 it has no view of the guest. A VM whose task finished, crashed after `/run`, or hung stays `RUNNING` and billing until the 8 h cap, so the orchestrator's `TerminateMicrovm` on finalize remains the only cleanup for normal operation and a leaked handle is still a cost incident. -- (!) **Snapshot uniqueness.** Shared memory snapshots mean every MicroVM restored from one image starts with identical PRNG state. **Re-scoped to P3 in the P2 review, with the exposure measured rather than assumed** (see the amendment under sub-decision 2): the sole `random` consumer in `agent/src` is a ULID sort key under a `task_id` partition, `os.urandom`/`secrets` are unaffected, and no credential or token derives from `random` — so the P2 exposure is a possible duplicate progress event, not a security boundary. It becomes load-bearing at P3, where a *resumed* VM continues from frozen state repeatedly; the reseed must land on `/run` and `/resume` together with those hooks. Asserting the requirement while leaving it unimplemented was the real defect, and this amendment is the fix. -- (−) **Root-stack resource headroom needs ongoing measurement.** The historical 985,886-byte / 486-resource measurement predates [#854](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/854), which reduced template size and resource count. The 2026-09-13 [offline review](../verification/645-p3-readiness-review.md) measures 472 root resources with a managed MicroVM image, and 489 with gateway and vault enabled in a scratch probe. Nesting can reduce that fullest root count to 474 in the prototype. These are unbundled synth measurements, not deployment validation; the vault combination remains guarded on main ([#857](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/857)). -- (!) **Snapshot WARMTH is a first-class property, not an optimisation.** A snapshot inherits only the pages something touched before it was captured, so a large lazily-loaded artifact — the 225 MiB `claude` binary, and anything similar added later — pays its first-touch cost on the *first task* instead of at container start. That cost failed every task at turn 0 in the P2 smoke run (P2-F5). Anything heavyweight added to the image must be exec'd in `/ready`, and any timeout guarding a first touch must be sized for a cold page fault rather than for the work itself. -- (!) **Regional availability (5 regions at launch, expanding)** — enforced in layers (synth-time static check, onboarding + doctor live probes, orchestration-time classification; see sub-decision 4). The static CDK constant is the one piece that rots as AWS expands; its update path and context-flag escape hatch are deliberate. -- (!) **Workload-token injection delta persists** (shared with the ECS backend) until [#249](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/249)/ADR-016 land; document it in the security bar comparison rather than blocking on it. Memory and Gateway are explicitly *not* deltas — both are standalone services consumed via IAM from any substrate. +| Phase | Delivered behavior | +|---|---| +| P1 | Strategy, infrastructure, bootstrap/types, minimal `/ready` + `/run` serving; merged in [#689](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/689) | +| P2 | Clone → change → PR, progress/logs/Memory, runtime configuration, `/validate` + `/terminate`; merged in [#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733), with takeover follow-up validation | +| P3 | Approval-aware sleep/wake, credential renewal, original deadlines, retained requests, complete checkpoint/replacement recovery, nested deployment and live acceptance | -## Testing +Activate sleep only after verifying the deployed image and coordinator together. Keep a compatible published coordinator and explicit image pin for rollback. Normal acceptance includes the 600-second default, explicit expiry, new and existing task off-switch behavior, replacement, cleanup and preservation of unrelated infrastructure. -**P1 (start/poll/stop — no suspend):** +Changing the default backend, GPU support, native Slack approval buttons, approval-by-Linear-reply and operator shell access are outside this ADR. CLI responses remain the supported approval path. Future work needs its own scope; “P4” is not an approved phase here. -- Unit tests for the strategy: start/poll/stop mapping (including `SessionStatus` `'suspended'` reported mechanically, without task-state interpretation), v2 reference-size bounds (4,096 bytes), immutable signed-reference replay and scoped payload transport, the image-identifier-must-be-an-ARN guard, the explicit `NO_INGRESS` argument (including the blank-env-var fallback, which must never omit the field), error classification (`ServiceQuotaExceededException`, `ThrottlingException`, `ResourceNotFoundException`, regional-unavailability), the omit-`idlePolicy` invariant, and `maximumDurationInSeconds` fixed at 28 800. -- Agent tests: `/ready` returns 200 after required local warm-up succeeds, without starting a task or calling AWS; `/run` accepts authenticated v2 references and rejects old unsigned envelopes, starts the pipeline asynchronously through the same mapper `/invocations` uses, returns before the pipeline finishes, and rejects every unusable envelope with a named code before spawning; `/suspend` and `/resume` are NOT served (`/validate` and `/terminate` joined the served set in P2). -- Orchestrator tests: substrate-terminal + non-terminal task status → failed classification; `suspended` + non-`AWAITING_APPROVAL` status → anomaly event, no fail-fast; `compute_metadata` persisted with `microvmId`/`endpoint` after `startSession`. -- CDK assertions: MicroVM resources present only when `ComputeTypes` includes the backend; synth failure for unsupported regions (plus the context-flag escape hatch); memory size validated against the accepted list at synth; the connector operator role and its trust; two connectors with the build-only one carrying port 80 and the runtime one not; the current six-hook declaration, shared protocol marker and per-worker capability admission; IAM actions scoped as specified (orchestrator lifecycle set; no `CreateMicrovmAuthToken` anywhere); payload-bucket grants (execution role read-only); backend cost-allocation tags; types-sync check covers the widened `ComputeType`. -- CLI tests: onboarding rejection with remedy when the availability probe fails; doctor check present when a blueprint selects the backend. -- P1 verification items (external service facts) — **executed 2026-07-31, us-east-1**; see `docs/verification/645-p1-lambda-microvm-runbook.md` for the full evidence. Discharged: `runHookPayload` limit (**4 096**, not 16 KB), the accepted baseline memory sizes (`[512…8192]` MiB — note the developer guide, not the probe, is what establishes that this is a BASELINE with a 32 GiB peak), image-identifier ARN requirement, IAM action names and the observed image-ARN shape, region probe behaviour, manual suspend/resume without `idlePolicy`, terminate timing and the `TERMINATED`-persists-≥10-min finding, the default public `HTTP_INGRESS`, and the `/ready` requirement. **Not** discharged: account-quota treatment of `SUSPENDED` MicroVMs (not observable safely), suspended TTL beyond 1 h (truncated), the vertical-scaling behaviour itself (no workload here approached the baseline, so the 4× peak is documented rather than observed), and the `AWS::Lambda::MicrovmImage` CloudFormation value shapes (never exercised — the run used the out-of-band script path; **discharged, and REFUTED, by the P2 run — see P2-F2 in sub-decision 3**). Record the closed answers in COMPUTE.md. +## Consequences -**P2 (smoke parity):** +- Approval waits can stop consuming compute without discarding the question or the saved work. Snapshot and checkpoint costs still apply. +- Native disk supports build-tool locking, and the service exposes explicit worker state without a cluster to operate. +- ABCA now maintains three backends and an additional artifact/snapshot lifecycle. Failed or unused image versions also need cleanup. +- Eight hours remains a per-worker limit. Retained approvals rely on verified retirement and replacement rather than extending a worker indefinitely. +- Suspended workers retain ABCA capacity until confirmed retirement. The service’s account-quota treatment of suspended memory was not established by the recorded probes. +- Memory baseline validation and published peak capacity are not workload benchmarks. Sustained heavy builds require measured sizing and may fit ECS better. +- A healthy heartbeat does not prove coding progress. Hook failures may cause service termination, but successful hook acceptance does not remove ABCA’s cleanup responsibility. +- Service error wording can hide transport errors. The pooled-hook mitigation has local and live evidence; historical internal dispatch traces remain unavailable. Open questions are tracked in the [service feedback record](../verification/645-lambda-microvm-service-feedback.md). -- Agent tests: `/validate` returns 200 with its individual check results, 503 while initialising, reports a missing hook route / unsupported interpreter, starts nothing, and makes **zero AWS calls even with `LOG_GROUP_NAME` set** (asserted by poisoning the boto3 and CloudWatch-writer seams — the same assertion covers `/ready`); `/terminate` returns 200 with no body at all, with a malformed / non-object / wrong-content-type / whitespace-only body, when the body read itself fails, with a pipeline still running (without joining it), and when its own best-effort step raises — and never calls `task_state.write_terminal`; a structural assertion that the route carries no typed body param keeps the 422 from being reintroduced. -- `platform_config` tests: the allowlist and required subset are read from `contracts/constants.json` (the wire key set is additionally asserted literally, as the agent-side tripwire on a contract edit); an unknown key rejects the whole block with nothing installed; a non-object block and a non-string value are rejected; a blank/`null` optional value is skipped without clobbering an image value while a blank required value is rejected; a payload value beats a pre-existing env value; installation is observed to happen before the GitHub-token resolver runs and before any pipeline thread exists; the downloaded config must equal the IAM-authenticated manifest, including same-account workspace identifiers; missing config and legacy envelopes are rejected regardless of baked environment values. Transport tests cover signed URLs, exact task identity, bad bytes and rejection before installation. -- Snapshot credential hygiene: a subprocess probe asserts that importing `server` and serving `/ready` + `/validate` imports neither `boto3` nor `botocore`, caches no `aws_session` session, and spawns no CloudWatch writer thread — the property that keeps a build-role credential chain and the build-time region out of the snapshot. -- `/run` pre-install silence: with a **baked `LOG_GROUP_NAME`** (the hostile case — without it the assertions pass vacuously) every AWS/credential seam (`boto3.client`/`Session`, the `aws_session` factories, `_debug_cw`/`_warn_cw`) is armed to raise until the install succeeds. Asserted on the accepted path, on all three rejection paths (bad envelope, `platform_config` invalid, `platform_config` incomplete) and on the failed-fetch 500 — where the seams stay armed for the whole request, because a rejected run installed nothing and so earns no AWS call. The permitted exception is asserted POSITIVELY: exactly one client is built pre-install, for `s3`, through the attributed factory. -- `/ready` warm-up tests (P2-F5): the hook exec's each configured binary exactly once with a generous timeout; `claude` is the only REQUIRED entry; a timeout, a missing binary, a non-zero exit and an unexpected `OSError` each produce **503 with the reason logged to stdout** rather than a 200 or a 500; a best-effort failure still reports ready; the warm-up makes zero AWS calls with `LOG_GROUP_NAME` baked. Plus the backstop half: the `claude --version` probe's bound is asserted to be ≥ 60 s and to be applied to the *exec* rather than to the PATH lookup, and a missing CLI warns instead of raising. -- CDK assertions (P2-F1/F2/F4): no source-condition key on any of the three MicroVM-facing role trusts, and no `aws:SourceAccount`/`aws:SourceArn` string anywhere in them; hook properties are `ENABLED` and the architecture is `ARM_64`, with a negative assertion that **no** hook route string appears anywhere in the rendered image resource; the agent hook routes are asserted against their own dedicated constant (the template no longer carries a path to compare); the execution role holds `logs:CreateLogStream`/`PutLogEvents` on the application log group and the two logs grants stay separate; the stack wires the SAME log group it delivers as `platform_config.log_group_name`. -- Smoke (gated like the ECS backend): clone → change → PR with `bgagent watch` progress; Memory write parity (no AccessDenied no-op). **Run 1 (2026-08-06) FAILED at `implement`, turn 0 — no PR. Run 2 (2026-08-07) PASSED: two tasks clone → change → commit → push → PR, `COMPLETED`, 12 turns / $0.279 / 153 s** (`docs/verification/645-p2-smoke-runbook.md`), which also discharged P2-F1, P2-F2, P2-F4, P2-F5 and the dual-signal-liveness item (45 s heartbeat cadence observed across a 181 s `RUNNING` window). Run 2 needed one live IAM workaround, producing P2r2-F10 (the identity-side `iam:PassedToService`) and P2r2-F9 (its CloudFormation twin). **The 2026-09-14 takeover-branch rerun exercised both corrected paths:** CloudFormation created the managed image and the orchestrator launched real coding tasks using source-defined permissions after a fresh bootstrap, with no manual IAM workaround. See the [deployment](../verification/645-p2-clean-deployment-20260913.md) and [task evidence](../verification/645-p2-live-task-20260914.md). This closes that historical clean-rerun requirement; the broader failure/recovery and permission/network acceptance matrix remains open. +## Testing -**P3 (suspend/resume):** +The [completed plan](../verification/645-p3-implementation-plan.md) indexes exact suite results and dated evidence. Required coverage includes: -- Unit tests: suspend/resume mapping; agentcore/ecs `unsupported` stubs; approve/deny inline resume with handle loaded from `compute_metadata`. -- HITL lifecycle tests: inline resume failure leaves the decision outcome intact and records the orphan event; orchestrator backstop retries resume; gate expiry fires at `min(monotonic budget, created_at + timeout_s)` — including the suspend/resume case where the monotonic budget exceeds the wall-clock remainder — without disturbing the §13.12 late-approval race protection. -- Smoke: suspend/resume across a simulated approval wait preserving workspace state. +- Strategy state mapping, explicit unsupported results, ARN/Region validation, NO_INGRESS, omitted idlePolicy and bounded uncertain-start recovery. +- Hook readiness/warm-up, AWS-silent build hooks, authenticated payload installation, arbitrary terminate bodies and lifecycle connection closure. +- Decision/timeout/cancellation races, image capability, coding/progress barriers, original deadlines, durable replay and exact-attempt capacity ownership. +- Real S3 version/integrity/access checks, real DynamoDB transactions and actual SDK conversation/Git/workspace recovery after process and disk loss. +- Cloud approve/deny/expiry/cancellation, repeated sleep/wake, AWS credential renewal after actual expiry, retirement/replacement and final resource cleanup. +- Nested fresh deployment and overlapping migration, compatible image/coordinator rollback, normal CLI feedback and live off-switch acceptance. -**All phases:** docs sync for [COMPUTE.md](../design/COMPUTE.md) (new column distinguishing MicroVMs from classic Lambda) and [ORCHESTRATOR.md](../design/ORCHESTRATOR.md) (liveness + suspend lifecycle). +Live evidence must distinguish service acknowledgment from completed guest recovery and synthetic handler checks from actual external-channel submissions. ## References -- Issue [#645](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/645) — originating RFC proposal -- Issue [#491](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/491) — unified liveness decision model (soft dependency, P3) -- Issue [#641](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/641) / ADR-019 (PR [#663](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/663)) — substrate-portable tool plane -- PR [#596](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/596) — ECS Fargate backend (pattern source for conditional wiring) -- [AWS Lambda MicroVMs](https://docs.aws.amazon.com/lambda/latest/dg/lambda-microvms-guide.html) — developer guide; [Running and using MicroVMs](https://docs.aws.amazon.com/lambda/latest/dg/microvms-launching.html) — lifecycle APIs and hooks -- [Agent Toolkit for AWS — aws-lambda-microvms skill](https://github.com/aws/agent-toolkit-for-aws/blob/main/skills/specialized-skills/serverless-skills/aws-lambda-microvms/SKILL.md) — operational constraints (no self-suspend, idle-policy semantics, snapshot uniqueness, size limits) -- [ADR-020](./ADR-020-ears-requirements-syntax.md) — EARS syntax used for the normative requirements above -- [CEDAR_HITL_GATES.md](../design/CEDAR_HITL_GATES.md) — approval-gate mechanics (decisions #6, #7) the suspend/resume handshake preserves; `cancel-task.ts` / `task-api.ts` — the inline best-effort + reconciler-backstop pattern the resume path mirrors -- [COMPUTE.md](../design/COMPUTE.md), [ORCHESTRATOR.md](../design/ORCHESTRATOR.md) — design docs to be updated by the implementing PRs +- [Issue #645](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/645) — originating proposal +- [Issue #491](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/491) — unified liveness model +- [Issue #641](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/641) — substrate-portable tool plane +- [AWS Lambda MicroVMs guide](https://docs.aws.amazon.com/lambda/latest/dg/lambda-microvms-guide.html) and [lifecycle APIs/hooks](https://docs.aws.amazon.com/lambda/latest/dg/microvms-launching.html) +- [ADR-020](./ADR-020-ears-requirements-syntax.md) — requirement syntax +- [Compute](../design/COMPUTE.md), [orchestrator](../design/ORCHESTRATOR.md), [approval gates](../design/CEDAR_HITL_GATES.md) diff --git a/docs/design/CEDAR_HITL_GATES.md b/docs/design/CEDAR_HITL_GATES.md index 88a01e69b..1bff5676b 100644 --- a/docs/design/CEDAR_HITL_GATES.md +++ b/docs/design/CEDAR_HITL_GATES.md @@ -1,17 +1,26 @@ # Cedar HITL Approval Gates -> **Status:** Core implemented; this document remains the authoritative design reference. +> **Status:** Core implemented. The September update below supersedes the original bounded-wait assumptions in the historical design and examples. > **Companion:** [`INTERACTIVE_AGENTS.md`](./INTERACTIVE_AGENTS.md) §9.3 (pointing here), §7 (state machine). > **Design locked:** 2026-04-23 (Sam ↔ assistant discussion). > **Rev:** 5 (2026-05-06 — fold in parallel adversarial + advocate review of the timeout design: late-approval re-read on TIMED_OUT ConditionCheckFailed; user-visible timeout-cap milestones; ceiling-shrink milestone; Runtime JWT bound verified as auto-refreshed IAM; three new tuning metrics; explicit off-hours trade-off section; notification-delivery-failure boundary. IMPL-24 through IMPL-28 added.). > **Implementation:** Core shipped. The 3-outcome engine (`agent/src/policy.py`), default policy sets (`agent/policies/hard_deny.cedar`, `agent/policies/soft_deny.cedar`), approval Lambdas (`cdk/src/handlers/{approve-task,deny-task,get-pending,get-policies}.ts`) wired into `cdk/src/constructs/task-api.ts` (routes `/tasks/{id}/approve`, `/deny`, `/pending`, `/repos/{repo_id}/policies`), the cross-engine parity fixtures (`contracts/cedar-parity/`), and the exact engine pins are all on `main`. §15's task list is preserved as a historical implementation record; see the note at the top of §15 for what (if anything) remains unbuilt. > -> **Current behavior clarified (2026-09-17):** closed approval-row conditions return -> `404 REQUEST_NOT_FOUND`; task-only conflicts return 409. Cancellation atomically -> closes a pending request. The stranded-task reconciler currently fails the task -> but leaves its approval row pending; the pending endpoint filters such rows. +> **Current source behavior (2026-09-17):** the task default is `approval_timeout_s=0`, +> meaning no decision deadline. Explicit task settings are 30–3,600 seconds; +> positive policy-rule deadlines still apply. Pending rows have no DynamoDB TTL, +> and `expires_at` is nullable. Task closure cancels unanswered requests and adds +> retention TTL without changing already-recorded decisions. Closed approval-row +> conditions return `404 REQUEST_NOT_FOUND`; task-only conflicts return 409. +> MicroVM checkpoint/retirement/replacement separates human waiting from worker +> lifetime and capacity; other backends retain their existing runtime limits. > CLI response instructions are implemented for Slack/Linear notifications; -> native channel decisions and approvals without an expiry remain future work. +> native channel decisions remain separate work. See the current +> [user guide](../guides/USER_GUIDE.md#approval-gates-cedar-hitl) and +> [continuation protocol](../verification/645-p3-continuation-protocol-20260917.md). +> The normal deployment passed retained-request, ten-minute sleep, explicit-expiry +> and sleep-off/rollback acceptance; see the +> [deployment record](../verification/645-p3-normal-closure-20260918.md). --- @@ -19,7 +28,7 @@ 1. [What we are building, in one paragraph](#1-what-we-are-building-in-one-paragraph) 2. [The three-outcome model and why Cedar alone can't give it](#2-the-three-outcome-model) -3. [Design decisions (locked)](#3-design-decisions-locked) +3. [Design decisions](#3-design-decisions) 4. [End-to-end request flow](#4-end-to-end-request-flow) 5. [Cedar policy authoring guide](#5-cedar-policy-authoring-guide) 6. [Engine implementation](#6-engine-implementation) @@ -116,18 +125,18 @@ The winning property: **policy authors can put on their "security-review-approve --- -## 3. Design decisions (locked) +## 3. Design decisions -Settled during the 2026-04-23 design discussion and extended after the 2026-04-24 and 2026-05-06 reviews. Each has detailed rationale in those conversations; summary here for implementers. **23 decisions**, all locked unless an adversarial review finding explicitly reopened a concern. +Settled during the 2026-04-23 design discussion and extended after the 2026-04-24 and 2026-05-06 reviews. Each has detailed rationale in those conversations; summary here for implementers. The September retained-request update also amends deadline, recovery and capacity behavior. | # | Decision | Summary | |---|---|---| | 1 | **Cedar encoding: two policy sets** | Physical hard-deny vs soft-deny split, validated via `@tier(...)` annotation. | | 2 | **Hook point: extend `PreToolUse`, not `can_use_tool`** | PreToolUse is already async-compatible, already wired to Cedar, and already owns the tool-governance boundary. | | 3 | **Wait mechanism: DDB strongly-consistent polling, 2s → 5s backoff** | Initial 2s cadence for the first 30s, then 5s. `ConsistentRead=True` so the agent never misses an approval that already landed. | -| 4 | **Scope allowlist: in-process, seeded from persisted `initial_approvals`** | Runtime escalation lives in the `PolicyEngine` instance. Submit-time `--pre-approve` flags persist on TaskTable and seed the allowlist at container startup. Lost on restart (rare; reconciler fails stranded tasks). | +| 4 | **Scope allowlist: in-process, seeded from persisted `initial_approvals`** | Runtime grants live in the `PolicyEngine` instance. Submit-time grants seed it at startup; a verified MicroVM checkpoint also saves session grant scopes and denial-cache entries for replacement. | | 5 | **CLI UX: standalone `bgagent approve/deny` + `--pre-approve ` + `bgagent policies list` + `bgagent pending`** | No inline interactive prompt in the streaming CLI for v1. Discovery + listing commands solve the request_id/rule_id copy problem. | -| 6 | **Timeouts: per-task default + per-rule Cedar annotation override, min wins, bounded floor + ceiling, fail-closed** | Per-task default: **300s** (5 min), overridable via `--approval-timeout` on submit and bounded by `[30, min(3600, maxLifetime - 300)]`. Floor: 30s (engine-enforced on both task default and rule annotations). Ceiling: `min(1h, maxLifetime_remaining - cleanup_margin)` — sized so the TTL on the approval row always covers the decision window. On timeout → deny (never auto-approve). See §14.8 for the off-hours trade-off this posture deliberately accepts. | +| 6 | **Timeouts: per-task default + per-rule Cedar annotation override, min wins, bounded floor + ceiling, fail-closed** | Per-task default: **0**, meaning no deadline. Explicit task deadlines are 30–3,600 seconds; positive matching rule deadlines can shorten them. Rule annotations still require at least 30 seconds. Workers without checkpoint/replacement support retain their existing lifetime limit. Pending rows have no storage TTL. Explicit timeout means deny, never auto-approve. | | 7 | **Concurrency slots: AWAITING_APPROVAL holds the slot** | Bounds unfinished sessions and their eventual resume demand, including a suspended MicroVM. This is an ABCA admission policy; suspended AWS memory-quota consumption remains unverified. | | 8 | **Hard-deny is absolute** | No `--pre-approve` scope, and no blueprint `disable:` directive, can bypass it. CreateTaskFn validates and rejects `rule:`; blueprint loader rejects `disable:` entries that name built-in hard-deny rules. | | 9 | **Submit-time scope cap: 20 entries, ≤128 chars each** | Keeps audit trail legible, bounds allowlist check cost, limits abuse-vector damage. | @@ -428,7 +437,7 @@ Fail-on-error is the right posture for blueprint misconfiguration — silent-fal |---|---|---|---| | `@rule_id("...")` | **Yes on soft-deny**, recommended on hard-deny | Kebab-case or snake_case identifier, unique across both tiers | Stable ID for `--pre-approve rule:X`, for audit trail, and for the `bgagent policies` discovery endpoint. `PolicyEngine.__init__` raises on duplicates. | | `@tier("hard"\|"soft")` | **Yes** | Exactly one of "hard" or "soft" | Validates policy is in the correct file/section. Engine rejects mismatch at load time. | -| `@approval_timeout_s("N")` | No | Integer seconds ≥ 30 | Per-rule timeout. If absent, uses the task default (**300s** by default, overridable via submit-time `--approval-timeout`; see decision #6). Has no effect on hard-deny rules. Values below the floor are rejected at load time. Values below **120s** emit a blueprint-load WARN but are accepted down to the 30s floor — almost no human responds to an approval request in under 2 minutes, so sub-120s is usually a policy-authoring mistake (see IMPL-25). Loader policy: STRICT at the floor (30s, reject) and ADVISORY below 120s (warn, accept). | +| `@approval_timeout_s("N")` | No | Integer seconds ≥ 30 | Per-rule timeout. If absent, uses the task setting (**0/no deadline** by default, configurable via submit-time `--approval-timeout`; see decision #6). Has no effect on hard-deny rules. Values below the floor are rejected at load time. Values below **120s** emit a blueprint-load WARN but are accepted down to the 30s floor — almost no human responds to an approval request in under 2 minutes, so sub-120s is usually a policy-authoring mistake (see IMPL-25). Loader policy: STRICT at the floor (30s, reject) and ADVISORY below 120s (warn, accept). | | `@severity("low"\|"medium"\|"high")` | No | One of the three | Shown in CLI approval prompt, colored by severity. Default: "medium". | | `@category("...")` | No | "destructive", "network", "filesystem", "auth", or free-form | UX grouping. CLI could filter approvals by category. Not enforced. | @@ -1077,7 +1086,7 @@ New field reference: | Field | Type | Required? | Default | Description | |---|---|---|---|---| -| `approval_timeout_s` | integer seconds | No | **300** | Per-task default approval timeout. Bounded by `[30, min(3600, maxLifetime - 300)]`. Per-rule `@approval_timeout_s` annotations may clip this further (min-wins; see decision #6 and §6.3). Default matches the §10.2 `TaskTable.approval_timeout_s` default. | +| `approval_timeout_s` | integer seconds | No | **0** | Zero retains unanswered requests without a deadline. Positive values must be 30–3,600. The shortest positive task/rule deadline wins. | | `initial_approvals` | list of scope strings | No | `[]` | Pre-approval allowlist scopes (≤20 entries, ≤128 chars each). Validated per §7.3 rules below. | `CreateTaskFn` validations: @@ -1091,7 +1100,7 @@ New field reference: - `write_path:X` — same rules as bash_pattern - `rule:X` — X must exist in the (built-in + target repo's blueprint) soft-deny policy set per the shared policy-parsing library; hard-deny rule IDs rejected - `all_session` — rejected if `Blueprint.security.maxPreApprovalScope` forbids -5. `approval_timeout_s` within `[30, min(3600, maxLifetime - 300)]` — cap at 1 hour OR (maxLifetime - 5min), whichever is smaller. Prevents multi-hour slot-exhaustion attacks and keeps approval windows within the TTL budget. +5. `approval_timeout_s` is zero or an integer from 30 through 3,600. MicroVM worker lifetime and capacity are bounded separately through checkpointed retirement; requests are not deleted by a pending-row TTL. 6. Combined `hard + soft + disable + custom` Cedar text size ≤ 64 KB (§12.4); reject on overflow. ### 7.4 Degenerate-pattern detection @@ -1222,7 +1231,7 @@ bgagent submit --task "..." --pre-approve all_session --yes `--pre-approve-file` reads a YAML/JSON array of scope strings — supports the 20-entry cap without command-line bloat. -`--approval-timeout` default (CLI and server): **300 seconds** (5 min), matching decision #6, §7.3, and the `TaskTable.approval_timeout_s` default in §10.2. Accepted range `[30, min(3600, maxLifetime - 300)]` — CLI validates client-side and the server re-validates. `bgagent submit --help` surfaces the default explicitly. +`--approval-timeout` default (CLI and server): **0**, meaning no automatic decision deadline. A positive value must be 30–3,600 seconds. The CLI validates it and the server re-validates it. Matching positive rule deadlines still apply. ### 8.3 Streaming UX @@ -1443,7 +1452,7 @@ Attributes: | `user_id` | S | Yes | Cognito `sub` **verbatim**; used in ownership check `ConditionExpression` (§7.1 finding #6) | | `repo` | S | Yes | Denormalized for fan-out | -**TTL sizing**: the TTL is always `timeout_s + 120s`, so a 300s approval window has a 420s TTL, a 3600s window has a 3720s TTL. The row never expires during the decision window. After the decision + a short grace period, DDB's eventual-consistency TTL reaper cleans up. +**TTL and decision deadlines are separate.** Pending approval rows have no TTL, including explicitly timed requests. Task closure cancels unanswered requests and assigns retention TTL to its approval records. Already-recorded decisions are preserved, and retries do not extend an existing retention deadline. **Why a list, not a StringSet, for `matching_rule_ids`**: DDB string sets cannot be empty. Pathological no-match soft-deny hits would fail to persist. Lists handle empty gracefully. @@ -1455,7 +1464,7 @@ Five new attributes on the existing task row: | Name | Type | Required | Description | |---|---|---|---| -| `approval_timeout_s` | N | No | Default timeout for soft-deny gates. Default 300. | +| `approval_timeout_s` | N | No | Task setting for soft-deny gates. Default 0/no deadline; positive values are 30–3,600 seconds. | | `initial_approvals` | L | No | List of scope strings from submit time | | `awaiting_approval_request_id` | S | No | Set when status = AWAITING_APPROVAL; cleared on transition back (via joint `UpdateExpression`) | | `approval_gate_count` | N | No | Running counter of approval gates fired on this task; used to enforce `approval_gate_cap` (decision #13) | @@ -2119,16 +2128,20 @@ Setup: Bob submits with `--approval-timeout 600`. The blueprint has a `write_cre Bob sees both the pre-submit warning (`approval_timeout_capped_at_submit`) and the per-gate cap event (`approval_timeout_capped`) so he understands why his 600s didn't apply. Without these milestones, the user sees only `timeout: 300s` in the approval banner and may think the CLI dropped their setting. Both events are captured in the event stream and surface via `bgagent watch`. See §11.1, §11.3 (`ApprovalTimeoutClipRate`), and Fix 4 / IMPL-26 in §16. -### 14.8 Off-hours and unattended tasks (known trade-off) +### 14.8 Off-hours and unattended tasks -**Known trade-off: off-hours failure.** Because timeouts are fail-closed (decision #6), a task running overnight with pending approvals will fail if no approver responds in time. This is deliberate — auto-approve on timeout would make "wait the reviewer out" the attacker's winning strategy; see decision #6 and §13.15 fail-closed summary. +Unanswered requests now remain available by default. On MicroVMs, a verified +checkpoint and worker retirement bound resource use while the person is away; +their later answer can start a replacement worker. Other compute substrates keep +their existing runtime limits. This does not approve any action automatically. -**For overnight / unattended runs, choose one of:** -- `--pre-approve all_session --yes` to bypass gates entirely for that task (accept the broader trust grant; see §7.3). -- Configure escalation on the `approval_requested` event via the notification plane (see `docs/design/INTERACTIVE_AGENTS.md` for channel configuration). Route to whoever is on-call; escalation schedule is the tenant's responsibility, not the timeout engine's. -- Schedule the task during business hours. +For a time-sensitive request, set an explicit positive decision deadline. Its +expiry remains fail-closed: the tool is denied, and the agent decides what to do +next. Notification routing can still notify an on-call reviewer; scheduling that +escalation belongs to the notification layer. -The `approvalGateCap` (decision #13) will force-fail the task after approximately `cap × task_default_timeout_s` of unanswered gates — default worst case ~4h at cap=50 / timeout=300s. Plan accordingly. +`approvalGateCap` limits the number of gates reached by a task, not elapsed human +waiting time. It does not impose a deadline on one unanswered request. **Why the timer itself is timezone-unaware:** Business-hours logic belongs in the notification plane, not the authorization engine. Baking calendars or on-call rotations into the timer couples the security boundary to a scheduling system it doesn't own. Same rule evaluated at 9am and 3am because the security property (adversary cannot wait out review) is time-invariant. Delivery-availability-aware scheduling is the notification plane's job via subscribed-channel health and escalation policies. @@ -2300,7 +2313,9 @@ Rollout steps: ### 15.5 Backward compatibility -- Existing tasks without `initial_approvals` → empty list → no pre-approvals, default `approval_timeout_s = 300` +- Tasks without `initial_approvals` receive an empty list and no pre-approvals. + New tasks default to `approval_timeout_s = 0`. An already-persisted approval + retains its original timeout; an upgrade or replacement does not extend it. - Existing policies without `@rule_id` / `@tier` → engine fails to start (fail-closed). Blueprint authors must add annotations explicitly during migration. - `PolicyDecision.allowed` property provides backward compat for existing `if not decision.allowed` callers - Hook return shape unchanged — Phase 1a/1b tests continue to pass diff --git a/docs/design/COMPUTE.md b/docs/design/COMPUTE.md index 0f44807a4..c34ed2845 100644 --- a/docs/design/COMPUTE.md +++ b/docs/design/COMPUTE.md @@ -81,20 +81,23 @@ Lambda MicroVMs are an opt-in third backend, selected per repository with `compu For managed images, run `cdk/scripts/package-microvm-artifact.sh --stack-name ` after each agent change. It packages the Dockerfile's local inputs into a deterministic ZIP and uploads it under `microvm-images/agent-artifact-.zip`. The checksum is a fingerprint of the uploaded bytes: identical inputs reuse the same verified object, and changed inputs produce a new filename. Deploy with the printed `--context microvm_artifact_sha256=` alongside `microvm_base_image_arn` and `microvm_base_image_version`, and retain these inputs for later deployments. The changed S3 URI tells CloudFormation to update the existing image; overwriting the old fixed filename alone does not. Missing or malformed digests fail synthesis. An initial deployment without an image must create the bucket first. The script's explicit `--create-image` alternative retains the fixed base key and calls the image API directly. -Because a snapshot freezes its build-time environment, current deployment identifiers arrive through the v2 payload bootstrap. The coordinator publishes a non-secret deployment manifest and sends a single-object signed download URL for the task. The worker reads only its deployment's `bootstrap/*` with ambient credentials; other object reads and payload-bucket listing are explicitly denied. The downloaded task identity and configuration must match the authenticated manifest before configuration installation. The serialized reference fits the verified 4,096-byte hook limit; all payload sizes use S3. ECS shares this transport through `AGENT_PAYLOAD_REF`. [ADR-021 §3](../decisions/ADR-021-lambda-microvms-compute-backend.md#3-packaging-same-agent-image-source-new-build-path) defines the wire format, compatibility change and pending live gates. +Because a snapshot freezes its build-time environment, current deployment identifiers arrive through the v2 payload bootstrap. The coordinator publishes a non-secret deployment manifest and sends a single-object signed download URL for the task. The worker reads only its deployment's `bootstrap/*` with ambient credentials; other object reads and payload-bucket listing are explicitly denied. The downloaded task identity and configuration must match the authenticated manifest before configuration installation. The serialized reference fits the verified 4,096-byte hook limit; all payload sizes use S3. ECS shares this transport through `AGENT_PAYLOAD_REF`. [ADR-021 §3](../decisions/ADR-021-lambda-microvms-compute-backend.md#3-packaging-same-agent-image-source-new-build-path) defines the wire format and compatibility requirements; [live payload checks](../verification/645-p2-payload-live-20260914.md) record validation. -Networking separates image build from execution: the build-only connector permits TCP 80 and 443 because the Dockerfile uses `apt-get`, while running MicroVMs retain 443-only egress through the platform VPC. Every launch explicitly passes the Lambda-managed `NO_INGRESS` connector; omission would select the service's public-ingress default. The P2 image declares and serves `/ready` and `/validate` at build time and `/run` and `/terminate` at runtime. Local P3 images also declare the implemented `/suspend` and `/resume` hooks; automatic suspension separately defaults off pending live acceptance. Registry HTTP/SSE tools therefore need reachable HTTPS/443 endpoints; remote non-443 tools are unsupported under the default policies of AgentCore and ECS as well. Local `stdio` tools can run, with their outbound traffic subject to the same restriction. Asset resolution/loading does not probe connectivity; see [registry network support](./REGISTRY.md#2-asset-kinds-for-mvp). +Networking separates image build from execution: the build-only connector permits TCP 80 and 443 because the Dockerfile uses `apt-get`, while running MicroVMs retain 443-only egress through the platform VPC. Every launch explicitly passes the Lambda-managed `NO_INGRESS` connector; omission would select the service's public-ingress default. Images declare and serve `/ready` and `/validate` at build time and `/run`, `/terminate`, `/suspend` and `/resume` at runtime. Automatic suspension is a separate deployment opt-in. Registry HTTP/SSE tools therefore need reachable HTTPS/443 endpoints; remote non-443 tools are unsupported under the default policies of AgentCore and ECS as well. Local `stdio` tools can run, with their outbound traffic subject to the same restriction. Asset resolution/loading does not probe connectivity; see [registry network support](./REGISTRY.md#2-asset-kinds-for-mvp). -Local P3 now connects the guest checkpoint/credential hooks, actual image-version capability, durable supervisor and post-commit approval wake. The `microvm_approval_suspend_enabled` deployment context defaults false and controls both a static opt-in and a live Parameter Store switch. Existing durable executions reread the live switch before new suspension because their original Lambda environment is pinned. Recovery and the original service lifetime survive supervisor replay; the API preserves accepted decisions when optional wake fails. AgentCore/ECS retain explicit unsupported pause/wake results. Approval expiry still uses the original UTC/monotonic deadline. See the [supervisor runbook](../verification/645-p3-supervisor.md) for budgets, scoped permissions and remaining live gates. +P3 connects the guest checkpoint/credential hooks, actual image-version capability, durable supervisor and post-commit approval wake. The `microvm_approval_suspend_enabled` deployment context defaults false and controls both a static opt-in and a live Parameter Store switch. Existing durable executions reread the live switch before new suspension because their original Lambda environment is pinned. Recovery and the original service lifetime survive supervisor replay; the API preserves accepted decisions when optional wake fails. AgentCore/ECS retain explicit unsupported pause/wake results. Explicit approval deadlines use the original UTC/monotonic deadline. See the [supervisor runbook](../verification/645-p3-supervisor.md) for budgets and scoped permissions, and the [completion record](../verification/645-p3-implementation-plan.md) for deployment evidence. Tasks can set `microvm_sleep_after_s` (CLI: `--microvm-sleep-after `). The default is 600 seconds of waiting for each approval; zero disables sleep. Creation persists the resolved preference, while legacy rows use the default. -The global suspension switch still takes precedence. The five-minute default -approval window therefore stays awake; only longer windows can reach the -ten-minute sleep delay. Neither this setting nor suspension extends a gate's -original deadline. Snapshot storage and save/restore fees mean the delay is a -user preference, not a guarantee of savings for every pause. +The global suspension switch still takes precedence. Unanswered approvals have +no deadline by default (`approval_timeout_s=0`); an explicit finite deadline +remains available. A short timed request stays awake when there is too little +useful sleep time before its wake margin. Neither suspension nor replacement +extends that original deadline. For a longer wait, a verified conversation and +workspace checkpoint lets ABCA retire the worker, release capacity and start a +replacement after the answer. Snapshot storage and save/restore fees mean the +sleep delay is a user preference, not a guarantee of savings for every pause. ## ECS Fargate task sizing (build vs. planning) @@ -117,7 +120,7 @@ The agent harness is the layer around the LLM that manages the execution loop: c The platform uses the [Claude Agent SDK](https://github.com/anthropics/claude-agent-sdk-python) as the harness. It provides the agent loop, built-in tools (filesystem, shell), and streaming message reception for per-turn trajectory capture (token usage, cost, tool calls). -**Execution model:** Tasks are fully unattended and one-shot. The agent loop runs in a background thread so the FastAPI `/ping` endpoint stays responsive on the main thread. The agent thread uses `asyncio.run()` with the stdlib event loop (uvicorn is configured with `--loop asyncio` to avoid uvloop conflicts with subprocess SIGCHLD handling). +**Execution model:** The agent runs automatically until it completes, needs a human decision or is stopped. The HTTP entrypoint runs its loop in a background thread so FastAPI `/ping` stays responsive; ECS uses the batch entrypoint. The agent thread uses `asyncio.run()` with the stdlib event loop (uvicorn is configured with `--loop asyncio` to avoid uvloop conflicts with subprocess SIGCHLD handling). **System prompt:** Selected by workflow from a shared base template (`agent/src/prompts/base.py`) with per-workflow sections (`coding/new-task-v1`, `coding/pr-iteration-v1`, `coding/pr-review-v1`). The platform defines what the agent should do; the harness executes it. @@ -132,7 +135,7 @@ The platform uses the [Claude Agent SDK](https://github.com/anthropics/claude-ag | GitHub | AgentCore Gateway + Identity | Clone, push, PR, issues | | Web search | AgentCore Gateway | Documentation lookups | -Plugins, skills, and MCP servers are out of scope for MVP. Additional tools can be added via Gateway integration. +Agents can use configured skills and MCP tools through the [registry](./REGISTRY.md). Their runtime network access remains subject to the compute backend’s egress rules. ### Policy enforcement diff --git a/docs/design/DEPLOYMENT_ROLES.md b/docs/design/DEPLOYMENT_ROLES.md index fe905ca85..9c34c2349 100644 --- a/docs/design/DEPLOYMENT_ROLES.md +++ b/docs/design/DEPLOYMENT_ROLES.md @@ -854,6 +854,15 @@ stays in the parent and is still excluded. Re-bootstrap before deploying the child stack. Existing flat deployments must keep `microvm_nested_stack=false` until their resource migration is reviewed; changing ownership is not an ordinary in-place update. See the [nested-stack runbook](../verification/645-p3-nested-stack.md). + +For a reviewed migration that keeps old and new resources side by side, +`microvm_resource_name_prefix` gives the nested image, network connectors and log +group distinct names. It requires nested mode and a concrete 1–40 character +letter/digit/hyphen prefix. Build/operator IAM role names remain derived from the +parent deployment so the bootstrap's existing `PassRole` scope still applies. +Keep the selected prefix stable in later deployments. This option alone does +not preserve the old resources or their runtime permissions; those remain part +of the migration procedure. This statement lets CloudFormation manage and tag the live suspension setting at `//microvm-approval-suspend-enabled`. The coordinator gets only `GetParameter` on its exact parameter. Existing durable diff --git a/docs/design/ORCHESTRATOR.md b/docs/design/ORCHESTRATOR.md index a1ede8f3d..7036167c8 100644 --- a/docs/design/ORCHESTRATOR.md +++ b/docs/design/ORCHESTRATOR.md @@ -293,11 +293,19 @@ When the session is unhealthy, the task transitions to `FAILED` with "Agent sess **Lambda MicroVM state polling.** Liveness is a dual signal. The strategy maps `GetMicrovm` mechanically: `PENDING`/`RUNNING` report `running`, `SUSPENDING`/`SUSPENDED` report `suspended`, and `TERMINATING`/`TERMINATED` report terminal completion. The orchestrator supplies the health interpretation: - Intentional suspension requires the matching pending gate and saved suspend intent. Unexpected suspension emits one anomaly per episode and starts bounded wake recovery, preserving recoverable work. -- A terminal substrate report paired with a non-terminal task is a failure, but finalization first strongly re-reads the task row to confirm the agent did not write a terminal result between the original read and VM termination. +- A terminal substrate report paired with a non-terminal task first checks for a complete, acknowledged approval checkpoint. Such a checkpoint can retire the old attempt and retain the task for a replacement. Without one, finalization strongly re-reads the task row before classifying a substrate failure. - Substrate state detects a dead VM; heartbeat staleness detects loss of the heartbeat writer inside a VM that still reports `RUNNING`. The independent heartbeat thread can continue during a pipeline hang, so a fresh timestamp is not proof of progress. The P3 supervisor saves intent before control calls and rechecks the gate before and after them. Its durable state retains an absolute service lifetime, consecutive failures, recovery start time and next delay. Three failed cycles or 120 seconds of unconfirmed wake cannot become an indefinite wait. AWS RUNNING does not end recovery while the guest remains stuck on a decided/expired approval; fresh guest liveness is required. API approve/deny commit first, then attempt a bounded wake without changing the decision response. Automatic suspension defaults off via `microvm_approval_suspend_enabled`; disabling new sleep preserves wake and cleanup. See the [supervisor runbook](../verification/645-p3-supervisor.md). +Unanswered approvals have no deadline by default. A checkpointed MicroVM wait can +retire after an hour, or before that worker's lifetime ends. The coordinator +verifies immutable storage, fences the old attempt, confirms shutdown and then +releases capacity. A saved decision admits one replacement and invokes the +original published coordinator version. A scheduled manager retries lost signals +and performs terminal cleanup. See the [continuation protocol](../verification/645-p3-continuation-protocol-20260917.md) +for ownership, capacity, expiry and failure behavior. + `TERMINATED` is the normal terminal signal and remains observable for at least 10 minutes. `ResourceNotFoundException` maps to completion only as a late fallback after the control-plane record is eventually reaped; polling does not wait for `NotFound`. **`/ping` health endpoint (AgentCore only).** The agent's FastAPI server responds to AgentCore's `/ping` calls while the coding task runs in a separate thread. AgentCore sees `HealthyBusy` and keeps the session alive. diff --git a/docs/guides/CEDAR_POLICY_GUIDE.md b/docs/guides/CEDAR_POLICY_GUIDE.md index 15fd92322..0ad43f54b 100644 --- a/docs/guides/CEDAR_POLICY_GUIDE.md +++ b/docs/guides/CEDAR_POLICY_GUIDE.md @@ -63,11 +63,14 @@ Every tool call the agent makes is evaluated as a Cedar `(principal, action, res |---|---|---|---| | `@rule_id("...")` | **Yes on soft-deny** (recommended on hard-deny) | Unique kebab/snake-case identifier | Stable ID for `--pre-approve rule:X`, for audit events, and for `bgagent policies show --rule X`. Engine rejects duplicates at task start. | | `@tier("hard"\|"soft")` | **Yes** | Exactly one of `"hard"` or `"soft"` | Must match the file section. Mismatches fail task start. | -| `@approval_timeout_s("N")` | No | Integer seconds ≥ 30 | Per-rule timeout. Defaults to 300 s (overridable per-task via `--approval-timeout`). When multiple soft rules match, the engine picks the minimum. Values < 120 s emit a load-time warning; values < 30 s are rejected. Ignored on hard-deny. | +| `@approval_timeout_s("N")` | No | Integer seconds ≥ 30 | Optional per-rule decision deadline. If absent, uses the task setting, whose default `0` means no deadline. The shortest positive task/rule deadline wins. Values < 120 s emit a load-time warning; values < 30 s are rejected. Ignored on hard-deny. | | `@severity("low"\|"medium"\|"high")` | No | One of three | Displayed in the approval prompt. Default: `medium`. | | `@category("...")` | No | `destructive`, `network`, `filesystem`, `auth`, or free-form | Optional UX grouping. Not enforced. | -**Rule of thumb:** every soft-deny rule must have `@rule_id` and should set `@severity` + `@approval_timeout_s` explicitly. Users scanning `bgagent pending` lean on these fields to triage quickly. +**Rule of thumb:** every soft-deny rule must have `@rule_id` and should set +`@severity`. Add `@approval_timeout_s` only when the workflow needs a decision +deadline. Omit the annotation to let the task choose; unlike the task's `0` +setting, a zero-valued rule annotation is invalid. ## Common patterns diff --git a/docs/guides/LINEAR_SETUP_GUIDE.md b/docs/guides/LINEAR_SETUP_GUIDE.md index c76c862af..d2ed88834 100644 --- a/docs/guides/LINEAR_SETUP_GUIDE.md +++ b/docs/guides/LINEAR_SETUP_GUIDE.md @@ -11,7 +11,7 @@ Set up the ABCA Linear integration so that applying a label to a Linear issue tr ## How it works -You create a Linear OAuth app and authorize it on your workspace. When someone adds the trigger label to an issue in a mapped project, Linear fires a webhook at ABCA; the receiver verifies the HMAC signature, looks up the workspace, resolves a Linear API token, and creates a task. The agent clones the repo, makes the change, opens a PR, and comments back on the issue. +You create a Linear OAuth app and authorize it on your workspace. When someone adds the trigger label to an issue in a mapped project, Linear sends a webhook to ABCA. The receiver verifies its signature and invokes a processor, which resolves the workspace token and creates a task. The agent clones the repo, makes the change, opens a PR, and reports back on the issue. The app is installed with `actor=app`, so everything ABCA writes is attributed to the app rather than to whoever clicked Authorize. @@ -21,10 +21,10 @@ One of two places, chosen automatically at setup time: | | When it's used | What's stored | |---|---|---| -| **AgentCore Identity vault** | The stack was deployed with `--context enableLinearIdentityVault=true` | Nothing long-lived. AgentCore holds the refresh token and mints short-lived access tokens on demand. | +| **AgentCore Identity vault** | The stack was deployed with `--context enableLinearIdentityVault=true` | AgentCore manages the OAuth grant and token refresh. ABCA retains the OAuth client credentials and workspace/webhook metadata in Secrets Manager. | | **Secrets Manager** | Otherwise — including regions where AgentCore Identity isn't available | An OAuth token bundle in `bgagent-linear-oauth-`, refreshed and rotated by ABCA. | -The vault is unavailable on the `lambda-microvm` substrate — see [Not available with `compute_type=lambda-microvm`](#not-available-with-compute_typelambda-microvm) below. +The vault can be enabled with AgentCore, ECS or Lambda MicroVM compute. `bgagent linear setup` picks whichever the deployment supports and tells you which one it used. There is no flag. If the vault isn't available it prints one line and continues on Secrets Manager: @@ -34,11 +34,16 @@ AgentCore Identity not available in us-east-1 — using Secrets Manager. A workspace that started on Secrets Manager and later moves to the vault **keeps** its Secrets Manager token as a fallback. A workspace onboarded straight onto the vault has no such token by design — it needs the vault to be reachable. -When a workspace's authorization dies, ABCA records it on the registry row and publishes to the stack's operational alert topic. That topic has **no subscribers unless you deployed with `alertEmail`**, so set it if you want to hear about a dead workspace rather than discover it from `bgagent platform doctor`. +When a workspace's authorization dies, ABCA records it on the registry row and publishes to the stack's operational alert topic. A new topic starts without subscribers; configure `alertEmail` or subscribe another destination to receive those alerts. A revoked legacy Secrets Manager fallback does not by itself mean the active vault grant is revoked. -#### Not available with `compute_type=lambda-microvm` +#### Using the vault with Lambda MicroVMs -The vault and the Lambda MicroVMs substrate remain gated from being enabled together ([#857](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/857)). The original 505-resource measurement predates stack-size reductions; the [current offline review](../verification/645-p3-readiness-review.md) finds room in its tested configurations, but bundled feature-matrix and deployment verification are still required before removing the guard. Use the vault on `agentcore` or `ecs`; MicroVM deployments use Secrets Manager while this gate remains. +Deploy with both `compute_type=lambda-microvm` and `enableLinearIdentityVault=true`. The coordinator sends the workload identity name through authenticated `platform_config`; the guest uses its compute execution role to obtain a Linear token. Credentials are not baked into the MicroVM image. + +When upgrading an existing MicroVM deployment, rebuild the guest image too: +the coordinator and guest must both support the vault configuration fields. + +The runtime resolves the vault in its AWS Region. A grant in another Region or under another workload identity does not automatically carry over. The old resource-count guard ([#857](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/857)) has been replaced with configuration, permission and deployment-budget checks. #### One workload identity per stack diff --git a/docs/guides/USER_GUIDE.md b/docs/guides/USER_GUIDE.md index e414eb15d..043c9538c 100644 --- a/docs/guides/USER_GUIDE.md +++ b/docs/guides/USER_GUIDE.md @@ -638,7 +638,7 @@ Created: 2026-04-01T00:39:51.271Z | `--max-budget` | Maximum cost budget in USD (0.01–100). Overrides per-repo Blueprint default. No default limit. | | `--idempotency-key` | Idempotency key for deduplication. | | `--trace` | Enable detailed tracing: raises progress preview cap to 4 KB and uploads full NDJSON trajectory to S3 on completion. Download with `bgagent trace download`. | -| `--approval-timeout` | Cedar HITL per-task approval timeout in seconds (default 300). A matching rule with its own `@approval_timeout_s` annotation still takes the minimum. See [Approval gates](#approval-gates-cedar-hitl). | +| `--approval-timeout` | Cedar HITL decision window: `0` (default) keeps unanswered requests available; a positive value sets a 30–3,600 second deadline. A matching rule can require a shorter positive deadline. See [Approval gates](#approval-gates-cedar-hitl). | | `--microvm-sleep-after` | Seconds to wait for approval before putting a Lambda MicroVM to sleep (default 600 = 10 minutes; 0–3600 accepted). Use `off` to keep it awake. Requires the deployment's automatic-sleep feature to be enabled; does not change approval deadlines or affect other compute backends. | | `--pre-approve` | Cedar HITL scope to approve up-front (repeatable). Same scope forms as `bgagent approve --scope`. Hard-deny rules are always enforced. | | `--wait` | Poll until the task reaches a terminal status. | @@ -842,7 +842,7 @@ When a rule marked `@tier("soft")` matches a tool call: 1. The agent stops before invoking the tool. 2. A row is atomically written to the approvals table and the task status flips to `AWAITING_APPROVAL`. 3. A progress event (`approval_requested`) is emitted so `bgagent watch` shows the gate in real time. -4. The task waits for your decision up to the rule's timeout (default 300 s, configurable per-rule and per-task). +4. The task waits for your decision without an automatic deadline by default. An explicit per-task or policy-rule timeout can limit that window. 5. On approval, the agent proceeds; on denial, the deny reason is best-effort injected back into the agent's context so it can adapt; on timeout, the gate is treated as a denial with `timed_out` as the reason. A decision is recorded at most once per request. Replaying approve/deny on the same `(task_id, request_id)` is idempotent. @@ -853,7 +853,7 @@ A decision is recorded at most once per request. Replaying approve/deny on the s node lib/bin/bgagent.js pending ``` -Lists every approval across your tasks that is currently awaiting your decision. The default text output gives you the `request_id`, tool, severity, the reason the rule matched, the tool-input preview, the expiry time, and ready-to-run `approve` / `deny` command lines. Pipe through `--output json` for scripting. +Lists every approval across your tasks that is currently awaiting your decision. The default text output gives you the `request_id`, tool, severity, the reason the rule matched, the tool-input preview, the deadline or “no automatic expiry,” and ready-to-run `approve` / `deny` command lines. The JSON `expires_at` is `null` when there is no deadline. Cancelled or completed tasks no longer have answerable requests. ```text 1 pending approval(s): @@ -865,7 +865,7 @@ Lists every approval across your tasks that is currently awaiting your decision. rules: force_push_any preview: git push --force origin feature/xyz created: 2026-05-13T12:04:12Z - expires: 2026-05-13T12:09:12Z (timeout_s=300) + expires: no automatic expiry approve: bgagent approve 01KN37PZ77P1W19D71DTZ15X6X 01R... deny: bgagent deny 01KN37PZ77P1W19D71DTZ15X6X 01R... --reason "..." ``` @@ -925,13 +925,16 @@ node lib/bin/bgagent.js submit --repo owner/repo --issue 42 \ --pre-approve tool_type:Bash \ --pre-approve write_path:tests/** -# Per-task timeout override (platform default is 300s) +# Optional ten-minute decision deadline (the default has no deadline) node lib/bin/bgagent.js submit --repo owner/repo --issue 42 --approval-timeout 600 ``` `--pre-approve` can be repeated up to the platform limit (see `bgagent submit --help` for the current cap). Valid scope forms are the same as the `approve --scope` table above. Hard-deny rules are still enforced — `--pre-approve` only short-circuits soft-deny rules. -`--approval-timeout` sets the task-wide default; a rule with its own `@approval_timeout_s` annotation still takes the minimum of the two. +`--approval-timeout 0` keeps unanswered requests available. A positive setting +limits the decision window; the shortest positive deadline from the task and +matching policy rules wins. Zero does not disable a policy rule's explicit +deadline. Cancelling the task closes its requests. For Lambda MicroVM tasks, `--microvm-sleep-after 600` selects the default 10-minute delay; `--microvm-sleep-after 120` selects two minutes and @@ -939,11 +942,13 @@ For Lambda MicroVM tasks, `--microvm-sleep-after 600` selects the default approval request is created. Waking for approval, denial, or an approaching deadline remains automatic. Sleep never starts a new approval timer. -The default approval timeout is five minutes, so those requests stay awake -with the ten-minute sleep delay. A longer task timeout does not override a -shorter policy-rule timeout. Sleeping saves compute charges but adds snapshot -save/restore charges and wake-up time; short pauses can cost more than staying -awake. The API equivalent is `microvm_sleep_after_s` (zero means off); task +An unanswered request can outlive its MicroVM. After a longer wait, ABCA saves +the workspace and conversation, stops the worker, and releases its capacity. +Your answer can then start a replacement when capacity is available. A request +remaining open does not mean its old computer must stay alive. Sleeping saves +compute charges but adds snapshot save/restore charges and wake-up time; short +pauses can cost more than staying awake. The API equivalent is +`microvm_sleep_after_s` (zero means off); task details return the saved setting. Automatic suspension remains disabled by default pending the [P3 acceptance checks](../verification/645-p3-implementation-plan.md). diff --git a/docs/src/content/docs/architecture/Cedar-hitl-gates.md b/docs/src/content/docs/architecture/Cedar-hitl-gates.md index dcadae836..ce3ac49fe 100644 --- a/docs/src/content/docs/architecture/Cedar-hitl-gates.md +++ b/docs/src/content/docs/architecture/Cedar-hitl-gates.md @@ -4,18 +4,27 @@ title: Cedar hitl gates # Cedar HITL Approval Gates -> **Status:** Core implemented; this document remains the authoritative design reference. +> **Status:** Core implemented. The September update below supersedes the original bounded-wait assumptions in the historical design and examples. > **Companion:** [`INTERACTIVE_AGENTS.md`](/sample-autonomous-cloud-coding-agents/architecture/interactive-agents) §9.3 (pointing here), §7 (state machine). > **Design locked:** 2026-04-23 (Sam ↔ assistant discussion). > **Rev:** 5 (2026-05-06 — fold in parallel adversarial + advocate review of the timeout design: late-approval re-read on TIMED_OUT ConditionCheckFailed; user-visible timeout-cap milestones; ceiling-shrink milestone; Runtime JWT bound verified as auto-refreshed IAM; three new tuning metrics; explicit off-hours trade-off section; notification-delivery-failure boundary. IMPL-24 through IMPL-28 added.). > **Implementation:** Core shipped. The 3-outcome engine (`agent/src/policy.py`), default policy sets (`agent/policies/hard_deny.cedar`, `agent/policies/soft_deny.cedar`), approval Lambdas (`cdk/src/handlers/{approve-task,deny-task,get-pending,get-policies}.ts`) wired into `cdk/src/constructs/task-api.ts` (routes `/tasks/{id}/approve`, `/deny`, `/pending`, `/repos/{repo_id}/policies`), the cross-engine parity fixtures (`contracts/cedar-parity/`), and the exact engine pins are all on `main`. §15's task list is preserved as a historical implementation record; see the note at the top of §15 for what (if anything) remains unbuilt. > -> **Current behavior clarified (2026-09-17):** closed approval-row conditions return -> `404 REQUEST_NOT_FOUND`; task-only conflicts return 409. Cancellation atomically -> closes a pending request. The stranded-task reconciler currently fails the task -> but leaves its approval row pending; the pending endpoint filters such rows. +> **Current source behavior (2026-09-17):** the task default is `approval_timeout_s=0`, +> meaning no decision deadline. Explicit task settings are 30–3,600 seconds; +> positive policy-rule deadlines still apply. Pending rows have no DynamoDB TTL, +> and `expires_at` is nullable. Task closure cancels unanswered requests and adds +> retention TTL without changing already-recorded decisions. Closed approval-row +> conditions return `404 REQUEST_NOT_FOUND`; task-only conflicts return 409. +> MicroVM checkpoint/retirement/replacement separates human waiting from worker +> lifetime and capacity; other backends retain their existing runtime limits. > CLI response instructions are implemented for Slack/Linear notifications; -> native channel decisions and approvals without an expiry remain future work. +> native channel decisions remain separate work. See the current +> [user guide](/sample-autonomous-cloud-coding-agents/using/overview#approval-gates-cedar-hitl) and +> [continuation protocol](/sample-autonomous-cloud-coding-agents/architecture/645-p3-continuation-protocol-20260917). +> The normal deployment passed retained-request, ten-minute sleep, explicit-expiry +> and sleep-off/rollback acceptance; see the +> [deployment record](/sample-autonomous-cloud-coding-agents/architecture/645-p3-normal-closure-20260918). --- @@ -23,7 +32,7 @@ title: Cedar hitl gates 1. [What we are building, in one paragraph](#1-what-we-are-building-in-one-paragraph) 2. [The three-outcome model and why Cedar alone can't give it](#2-the-three-outcome-model) -3. [Design decisions (locked)](#3-design-decisions-locked) +3. [Design decisions](#3-design-decisions) 4. [End-to-end request flow](#4-end-to-end-request-flow) 5. [Cedar policy authoring guide](#5-cedar-policy-authoring-guide) 6. [Engine implementation](#6-engine-implementation) @@ -120,18 +129,18 @@ The winning property: **policy authors can put on their "security-review-approve --- -## 3. Design decisions (locked) +## 3. Design decisions -Settled during the 2026-04-23 design discussion and extended after the 2026-04-24 and 2026-05-06 reviews. Each has detailed rationale in those conversations; summary here for implementers. **23 decisions**, all locked unless an adversarial review finding explicitly reopened a concern. +Settled during the 2026-04-23 design discussion and extended after the 2026-04-24 and 2026-05-06 reviews. Each has detailed rationale in those conversations; summary here for implementers. The September retained-request update also amends deadline, recovery and capacity behavior. | # | Decision | Summary | |---|---|---| | 1 | **Cedar encoding: two policy sets** | Physical hard-deny vs soft-deny split, validated via `@tier(...)` annotation. | | 2 | **Hook point: extend `PreToolUse`, not `can_use_tool`** | PreToolUse is already async-compatible, already wired to Cedar, and already owns the tool-governance boundary. | | 3 | **Wait mechanism: DDB strongly-consistent polling, 2s → 5s backoff** | Initial 2s cadence for the first 30s, then 5s. `ConsistentRead=True` so the agent never misses an approval that already landed. | -| 4 | **Scope allowlist: in-process, seeded from persisted `initial_approvals`** | Runtime escalation lives in the `PolicyEngine` instance. Submit-time `--pre-approve` flags persist on TaskTable and seed the allowlist at container startup. Lost on restart (rare; reconciler fails stranded tasks). | +| 4 | **Scope allowlist: in-process, seeded from persisted `initial_approvals`** | Runtime grants live in the `PolicyEngine` instance. Submit-time grants seed it at startup; a verified MicroVM checkpoint also saves session grant scopes and denial-cache entries for replacement. | | 5 | **CLI UX: standalone `bgagent approve/deny` + `--pre-approve ` + `bgagent policies list` + `bgagent pending`** | No inline interactive prompt in the streaming CLI for v1. Discovery + listing commands solve the request_id/rule_id copy problem. | -| 6 | **Timeouts: per-task default + per-rule Cedar annotation override, min wins, bounded floor + ceiling, fail-closed** | Per-task default: **300s** (5 min), overridable via `--approval-timeout` on submit and bounded by `[30, min(3600, maxLifetime - 300)]`. Floor: 30s (engine-enforced on both task default and rule annotations). Ceiling: `min(1h, maxLifetime_remaining - cleanup_margin)` — sized so the TTL on the approval row always covers the decision window. On timeout → deny (never auto-approve). See §14.8 for the off-hours trade-off this posture deliberately accepts. | +| 6 | **Timeouts: per-task default + per-rule Cedar annotation override, min wins, bounded floor + ceiling, fail-closed** | Per-task default: **0**, meaning no deadline. Explicit task deadlines are 30–3,600 seconds; positive matching rule deadlines can shorten them. Rule annotations still require at least 30 seconds. Workers without checkpoint/replacement support retain their existing lifetime limit. Pending rows have no storage TTL. Explicit timeout means deny, never auto-approve. | | 7 | **Concurrency slots: AWAITING_APPROVAL holds the slot** | Bounds unfinished sessions and their eventual resume demand, including a suspended MicroVM. This is an ABCA admission policy; suspended AWS memory-quota consumption remains unverified. | | 8 | **Hard-deny is absolute** | No `--pre-approve` scope, and no blueprint `disable:` directive, can bypass it. CreateTaskFn validates and rejects `rule:`; blueprint loader rejects `disable:` entries that name built-in hard-deny rules. | | 9 | **Submit-time scope cap: 20 entries, ≤128 chars each** | Keeps audit trail legible, bounds allowlist check cost, limits abuse-vector damage. | @@ -432,7 +441,7 @@ Fail-on-error is the right posture for blueprint misconfiguration — silent-fal |---|---|---|---| | `@rule_id("...")` | **Yes on soft-deny**, recommended on hard-deny | Kebab-case or snake_case identifier, unique across both tiers | Stable ID for `--pre-approve rule:X`, for audit trail, and for the `bgagent policies` discovery endpoint. `PolicyEngine.__init__` raises on duplicates. | | `@tier("hard"\|"soft")` | **Yes** | Exactly one of "hard" or "soft" | Validates policy is in the correct file/section. Engine rejects mismatch at load time. | -| `@approval_timeout_s("N")` | No | Integer seconds ≥ 30 | Per-rule timeout. If absent, uses the task default (**300s** by default, overridable via submit-time `--approval-timeout`; see decision #6). Has no effect on hard-deny rules. Values below the floor are rejected at load time. Values below **120s** emit a blueprint-load WARN but are accepted down to the 30s floor — almost no human responds to an approval request in under 2 minutes, so sub-120s is usually a policy-authoring mistake (see IMPL-25). Loader policy: STRICT at the floor (30s, reject) and ADVISORY below 120s (warn, accept). | +| `@approval_timeout_s("N")` | No | Integer seconds ≥ 30 | Per-rule timeout. If absent, uses the task setting (**0/no deadline** by default, configurable via submit-time `--approval-timeout`; see decision #6). Has no effect on hard-deny rules. Values below the floor are rejected at load time. Values below **120s** emit a blueprint-load WARN but are accepted down to the 30s floor — almost no human responds to an approval request in under 2 minutes, so sub-120s is usually a policy-authoring mistake (see IMPL-25). Loader policy: STRICT at the floor (30s, reject) and ADVISORY below 120s (warn, accept). | | `@severity("low"\|"medium"\|"high")` | No | One of the three | Shown in CLI approval prompt, colored by severity. Default: "medium". | | `@category("...")` | No | "destructive", "network", "filesystem", "auth", or free-form | UX grouping. CLI could filter approvals by category. Not enforced. | @@ -1081,7 +1090,7 @@ New field reference: | Field | Type | Required? | Default | Description | |---|---|---|---|---| -| `approval_timeout_s` | integer seconds | No | **300** | Per-task default approval timeout. Bounded by `[30, min(3600, maxLifetime - 300)]`. Per-rule `@approval_timeout_s` annotations may clip this further (min-wins; see decision #6 and §6.3). Default matches the §10.2 `TaskTable.approval_timeout_s` default. | +| `approval_timeout_s` | integer seconds | No | **0** | Zero retains unanswered requests without a deadline. Positive values must be 30–3,600. The shortest positive task/rule deadline wins. | | `initial_approvals` | list of scope strings | No | `[]` | Pre-approval allowlist scopes (≤20 entries, ≤128 chars each). Validated per §7.3 rules below. | `CreateTaskFn` validations: @@ -1095,7 +1104,7 @@ New field reference: - `write_path:X` — same rules as bash_pattern - `rule:X` — X must exist in the (built-in + target repo's blueprint) soft-deny policy set per the shared policy-parsing library; hard-deny rule IDs rejected - `all_session` — rejected if `Blueprint.security.maxPreApprovalScope` forbids -5. `approval_timeout_s` within `[30, min(3600, maxLifetime - 300)]` — cap at 1 hour OR (maxLifetime - 5min), whichever is smaller. Prevents multi-hour slot-exhaustion attacks and keeps approval windows within the TTL budget. +5. `approval_timeout_s` is zero or an integer from 30 through 3,600. MicroVM worker lifetime and capacity are bounded separately through checkpointed retirement; requests are not deleted by a pending-row TTL. 6. Combined `hard + soft + disable + custom` Cedar text size ≤ 64 KB (§12.4); reject on overflow. ### 7.4 Degenerate-pattern detection @@ -1226,7 +1235,7 @@ bgagent submit --task "..." --pre-approve all_session --yes `--pre-approve-file` reads a YAML/JSON array of scope strings — supports the 20-entry cap without command-line bloat. -`--approval-timeout` default (CLI and server): **300 seconds** (5 min), matching decision #6, §7.3, and the `TaskTable.approval_timeout_s` default in §10.2. Accepted range `[30, min(3600, maxLifetime - 300)]` — CLI validates client-side and the server re-validates. `bgagent submit --help` surfaces the default explicitly. +`--approval-timeout` default (CLI and server): **0**, meaning no automatic decision deadline. A positive value must be 30–3,600 seconds. The CLI validates it and the server re-validates it. Matching positive rule deadlines still apply. ### 8.3 Streaming UX @@ -1447,7 +1456,7 @@ Attributes: | `user_id` | S | Yes | Cognito `sub` **verbatim**; used in ownership check `ConditionExpression` (§7.1 finding #6) | | `repo` | S | Yes | Denormalized for fan-out | -**TTL sizing**: the TTL is always `timeout_s + 120s`, so a 300s approval window has a 420s TTL, a 3600s window has a 3720s TTL. The row never expires during the decision window. After the decision + a short grace period, DDB's eventual-consistency TTL reaper cleans up. +**TTL and decision deadlines are separate.** Pending approval rows have no TTL, including explicitly timed requests. Task closure cancels unanswered requests and assigns retention TTL to its approval records. Already-recorded decisions are preserved, and retries do not extend an existing retention deadline. **Why a list, not a StringSet, for `matching_rule_ids`**: DDB string sets cannot be empty. Pathological no-match soft-deny hits would fail to persist. Lists handle empty gracefully. @@ -1459,7 +1468,7 @@ Five new attributes on the existing task row: | Name | Type | Required | Description | |---|---|---|---| -| `approval_timeout_s` | N | No | Default timeout for soft-deny gates. Default 300. | +| `approval_timeout_s` | N | No | Task setting for soft-deny gates. Default 0/no deadline; positive values are 30–3,600 seconds. | | `initial_approvals` | L | No | List of scope strings from submit time | | `awaiting_approval_request_id` | S | No | Set when status = AWAITING_APPROVAL; cleared on transition back (via joint `UpdateExpression`) | | `approval_gate_count` | N | No | Running counter of approval gates fired on this task; used to enforce `approval_gate_cap` (decision #13) | @@ -2123,16 +2132,20 @@ Setup: Bob submits with `--approval-timeout 600`. The blueprint has a `write_cre Bob sees both the pre-submit warning (`approval_timeout_capped_at_submit`) and the per-gate cap event (`approval_timeout_capped`) so he understands why his 600s didn't apply. Without these milestones, the user sees only `timeout: 300s` in the approval banner and may think the CLI dropped their setting. Both events are captured in the event stream and surface via `bgagent watch`. See §11.1, §11.3 (`ApprovalTimeoutClipRate`), and Fix 4 / IMPL-26 in §16. -### 14.8 Off-hours and unattended tasks (known trade-off) +### 14.8 Off-hours and unattended tasks -**Known trade-off: off-hours failure.** Because timeouts are fail-closed (decision #6), a task running overnight with pending approvals will fail if no approver responds in time. This is deliberate — auto-approve on timeout would make "wait the reviewer out" the attacker's winning strategy; see decision #6 and §13.15 fail-closed summary. +Unanswered requests now remain available by default. On MicroVMs, a verified +checkpoint and worker retirement bound resource use while the person is away; +their later answer can start a replacement worker. Other compute substrates keep +their existing runtime limits. This does not approve any action automatically. -**For overnight / unattended runs, choose one of:** -- `--pre-approve all_session --yes` to bypass gates entirely for that task (accept the broader trust grant; see §7.3). -- Configure escalation on the `approval_requested` event via the notification plane (see `docs/design/INTERACTIVE_AGENTS.md` for channel configuration). Route to whoever is on-call; escalation schedule is the tenant's responsibility, not the timeout engine's. -- Schedule the task during business hours. +For a time-sensitive request, set an explicit positive decision deadline. Its +expiry remains fail-closed: the tool is denied, and the agent decides what to do +next. Notification routing can still notify an on-call reviewer; scheduling that +escalation belongs to the notification layer. -The `approvalGateCap` (decision #13) will force-fail the task after approximately `cap × task_default_timeout_s` of unanswered gates — default worst case ~4h at cap=50 / timeout=300s. Plan accordingly. +`approvalGateCap` limits the number of gates reached by a task, not elapsed human +waiting time. It does not impose a deadline on one unanswered request. **Why the timer itself is timezone-unaware:** Business-hours logic belongs in the notification plane, not the authorization engine. Baking calendars or on-call rotations into the timer couples the security boundary to a scheduling system it doesn't own. Same rule evaluated at 9am and 3am because the security property (adversary cannot wait out review) is time-invariant. Delivery-availability-aware scheduling is the notification plane's job via subscribed-channel health and escalation policies. @@ -2304,7 +2317,9 @@ Rollout steps: ### 15.5 Backward compatibility -- Existing tasks without `initial_approvals` → empty list → no pre-approvals, default `approval_timeout_s = 300` +- Tasks without `initial_approvals` receive an empty list and no pre-approvals. + New tasks default to `approval_timeout_s = 0`. An already-persisted approval + retains its original timeout; an upgrade or replacement does not extend it. - Existing policies without `@rule_id` / `@tier` → engine fails to start (fail-closed). Blueprint authors must add annotations explicitly during migration. - `PolicyDecision.allowed` property provides backward compat for existing `if not decision.allowed` callers - Hook return shape unchanged — Phase 1a/1b tests continue to pass diff --git a/docs/src/content/docs/architecture/Compute.md b/docs/src/content/docs/architecture/Compute.md index 52495f52f..4f72a8a22 100644 --- a/docs/src/content/docs/architecture/Compute.md +++ b/docs/src/content/docs/architecture/Compute.md @@ -85,20 +85,23 @@ Lambda MicroVMs are an opt-in third backend, selected per repository with `compu For managed images, run `cdk/scripts/package-microvm-artifact.sh --stack-name ` after each agent change. It packages the Dockerfile's local inputs into a deterministic ZIP and uploads it under `microvm-images/agent-artifact-.zip`. The checksum is a fingerprint of the uploaded bytes: identical inputs reuse the same verified object, and changed inputs produce a new filename. Deploy with the printed `--context microvm_artifact_sha256=` alongside `microvm_base_image_arn` and `microvm_base_image_version`, and retain these inputs for later deployments. The changed S3 URI tells CloudFormation to update the existing image; overwriting the old fixed filename alone does not. Missing or malformed digests fail synthesis. An initial deployment without an image must create the bucket first. The script's explicit `--create-image` alternative retains the fixed base key and calls the image API directly. -Because a snapshot freezes its build-time environment, current deployment identifiers arrive through the v2 payload bootstrap. The coordinator publishes a non-secret deployment manifest and sends a single-object signed download URL for the task. The worker reads only its deployment's `bootstrap/*` with ambient credentials; other object reads and payload-bucket listing are explicitly denied. The downloaded task identity and configuration must match the authenticated manifest before configuration installation. The serialized reference fits the verified 4,096-byte hook limit; all payload sizes use S3. ECS shares this transport through `AGENT_PAYLOAD_REF`. [ADR-021 §3](/sample-autonomous-cloud-coding-agents/architecture/adr-021-lambda-microvms-compute-backend#3-packaging-same-agent-image-source-new-build-path) defines the wire format, compatibility change and pending live gates. +Because a snapshot freezes its build-time environment, current deployment identifiers arrive through the v2 payload bootstrap. The coordinator publishes a non-secret deployment manifest and sends a single-object signed download URL for the task. The worker reads only its deployment's `bootstrap/*` with ambient credentials; other object reads and payload-bucket listing are explicitly denied. The downloaded task identity and configuration must match the authenticated manifest before configuration installation. The serialized reference fits the verified 4,096-byte hook limit; all payload sizes use S3. ECS shares this transport through `AGENT_PAYLOAD_REF`. [ADR-021 §3](/sample-autonomous-cloud-coding-agents/architecture/adr-021-lambda-microvms-compute-backend#3-packaging-same-agent-image-source-new-build-path) defines the wire format and compatibility requirements; [live payload checks](/sample-autonomous-cloud-coding-agents/architecture/645-p2-payload-live-20260914) record validation. -Networking separates image build from execution: the build-only connector permits TCP 80 and 443 because the Dockerfile uses `apt-get`, while running MicroVMs retain 443-only egress through the platform VPC. Every launch explicitly passes the Lambda-managed `NO_INGRESS` connector; omission would select the service's public-ingress default. The P2 image declares and serves `/ready` and `/validate` at build time and `/run` and `/terminate` at runtime. Local P3 images also declare the implemented `/suspend` and `/resume` hooks; automatic suspension separately defaults off pending live acceptance. Registry HTTP/SSE tools therefore need reachable HTTPS/443 endpoints; remote non-443 tools are unsupported under the default policies of AgentCore and ECS as well. Local `stdio` tools can run, with their outbound traffic subject to the same restriction. Asset resolution/loading does not probe connectivity; see [registry network support](/sample-autonomous-cloud-coding-agents/architecture/registry#2-asset-kinds-for-mvp). +Networking separates image build from execution: the build-only connector permits TCP 80 and 443 because the Dockerfile uses `apt-get`, while running MicroVMs retain 443-only egress through the platform VPC. Every launch explicitly passes the Lambda-managed `NO_INGRESS` connector; omission would select the service's public-ingress default. Images declare and serve `/ready` and `/validate` at build time and `/run`, `/terminate`, `/suspend` and `/resume` at runtime. Automatic suspension is a separate deployment opt-in. Registry HTTP/SSE tools therefore need reachable HTTPS/443 endpoints; remote non-443 tools are unsupported under the default policies of AgentCore and ECS as well. Local `stdio` tools can run, with their outbound traffic subject to the same restriction. Asset resolution/loading does not probe connectivity; see [registry network support](/sample-autonomous-cloud-coding-agents/architecture/registry#2-asset-kinds-for-mvp). -Local P3 now connects the guest checkpoint/credential hooks, actual image-version capability, durable supervisor and post-commit approval wake. The `microvm_approval_suspend_enabled` deployment context defaults false and controls both a static opt-in and a live Parameter Store switch. Existing durable executions reread the live switch before new suspension because their original Lambda environment is pinned. Recovery and the original service lifetime survive supervisor replay; the API preserves accepted decisions when optional wake fails. AgentCore/ECS retain explicit unsupported pause/wake results. Approval expiry still uses the original UTC/monotonic deadline. See the [supervisor runbook](/sample-autonomous-cloud-coding-agents/architecture/645-p3-supervisor) for budgets, scoped permissions and remaining live gates. +P3 connects the guest checkpoint/credential hooks, actual image-version capability, durable supervisor and post-commit approval wake. The `microvm_approval_suspend_enabled` deployment context defaults false and controls both a static opt-in and a live Parameter Store switch. Existing durable executions reread the live switch before new suspension because their original Lambda environment is pinned. Recovery and the original service lifetime survive supervisor replay; the API preserves accepted decisions when optional wake fails. AgentCore/ECS retain explicit unsupported pause/wake results. Explicit approval deadlines use the original UTC/monotonic deadline. See the [supervisor runbook](/sample-autonomous-cloud-coding-agents/architecture/645-p3-supervisor) for budgets and scoped permissions, and the [completion record](/sample-autonomous-cloud-coding-agents/architecture/645-p3-implementation-plan) for deployment evidence. Tasks can set `microvm_sleep_after_s` (CLI: `--microvm-sleep-after `). The default is 600 seconds of waiting for each approval; zero disables sleep. Creation persists the resolved preference, while legacy rows use the default. -The global suspension switch still takes precedence. The five-minute default -approval window therefore stays awake; only longer windows can reach the -ten-minute sleep delay. Neither this setting nor suspension extends a gate's -original deadline. Snapshot storage and save/restore fees mean the delay is a -user preference, not a guarantee of savings for every pause. +The global suspension switch still takes precedence. Unanswered approvals have +no deadline by default (`approval_timeout_s=0`); an explicit finite deadline +remains available. A short timed request stays awake when there is too little +useful sleep time before its wake margin. Neither suspension nor replacement +extends that original deadline. For a longer wait, a verified conversation and +workspace checkpoint lets ABCA retire the worker, release capacity and start a +replacement after the answer. Snapshot storage and save/restore fees mean the +sleep delay is a user preference, not a guarantee of savings for every pause. ## ECS Fargate task sizing (build vs. planning) @@ -121,7 +124,7 @@ The agent harness is the layer around the LLM that manages the execution loop: c The platform uses the [Claude Agent SDK](https://github.com/anthropics/claude-agent-sdk-python) as the harness. It provides the agent loop, built-in tools (filesystem, shell), and streaming message reception for per-turn trajectory capture (token usage, cost, tool calls). -**Execution model:** Tasks are fully unattended and one-shot. The agent loop runs in a background thread so the FastAPI `/ping` endpoint stays responsive on the main thread. The agent thread uses `asyncio.run()` with the stdlib event loop (uvicorn is configured with `--loop asyncio` to avoid uvloop conflicts with subprocess SIGCHLD handling). +**Execution model:** The agent runs automatically until it completes, needs a human decision or is stopped. The HTTP entrypoint runs its loop in a background thread so FastAPI `/ping` stays responsive; ECS uses the batch entrypoint. The agent thread uses `asyncio.run()` with the stdlib event loop (uvicorn is configured with `--loop asyncio` to avoid uvloop conflicts with subprocess SIGCHLD handling). **System prompt:** Selected by workflow from a shared base template (`agent/src/prompts/base.py`) with per-workflow sections (`coding/new-task-v1`, `coding/pr-iteration-v1`, `coding/pr-review-v1`). The platform defines what the agent should do; the harness executes it. @@ -136,7 +139,7 @@ The platform uses the [Claude Agent SDK](https://github.com/anthropics/claude-ag | GitHub | AgentCore Gateway + Identity | Clone, push, PR, issues | | Web search | AgentCore Gateway | Documentation lookups | -Plugins, skills, and MCP servers are out of scope for MVP. Additional tools can be added via Gateway integration. +Agents can use configured skills and MCP tools through the [registry](/sample-autonomous-cloud-coding-agents/architecture/registry). Their runtime network access remains subject to the compute backend’s egress rules. ### Policy enforcement diff --git a/docs/src/content/docs/architecture/Deployment-roles.md b/docs/src/content/docs/architecture/Deployment-roles.md index c115a54ba..a0e4e714a 100644 --- a/docs/src/content/docs/architecture/Deployment-roles.md +++ b/docs/src/content/docs/architecture/Deployment-roles.md @@ -858,6 +858,15 @@ stays in the parent and is still excluded. Re-bootstrap before deploying the child stack. Existing flat deployments must keep `microvm_nested_stack=false` until their resource migration is reviewed; changing ownership is not an ordinary in-place update. See the [nested-stack runbook](/sample-autonomous-cloud-coding-agents/architecture/645-p3-nested-stack). + +For a reviewed migration that keeps old and new resources side by side, +`microvm_resource_name_prefix` gives the nested image, network connectors and log +group distinct names. It requires nested mode and a concrete 1–40 character +letter/digit/hyphen prefix. Build/operator IAM role names remain derived from the +parent deployment so the bootstrap's existing `PassRole` scope still applies. +Keep the selected prefix stable in later deployments. This option alone does +not preserve the old resources or their runtime permissions; those remain part +of the migration procedure. This statement lets CloudFormation manage and tag the live suspension setting at `//microvm-approval-suspend-enabled`. The coordinator gets only `GetParameter` on its exact parameter. Existing durable diff --git a/docs/src/content/docs/architecture/Orchestrator.md b/docs/src/content/docs/architecture/Orchestrator.md index 4a72a1022..ff67cd3c4 100644 --- a/docs/src/content/docs/architecture/Orchestrator.md +++ b/docs/src/content/docs/architecture/Orchestrator.md @@ -297,11 +297,19 @@ When the session is unhealthy, the task transitions to `FAILED` with "Agent sess **Lambda MicroVM state polling.** Liveness is a dual signal. The strategy maps `GetMicrovm` mechanically: `PENDING`/`RUNNING` report `running`, `SUSPENDING`/`SUSPENDED` report `suspended`, and `TERMINATING`/`TERMINATED` report terminal completion. The orchestrator supplies the health interpretation: - Intentional suspension requires the matching pending gate and saved suspend intent. Unexpected suspension emits one anomaly per episode and starts bounded wake recovery, preserving recoverable work. -- A terminal substrate report paired with a non-terminal task is a failure, but finalization first strongly re-reads the task row to confirm the agent did not write a terminal result between the original read and VM termination. +- A terminal substrate report paired with a non-terminal task first checks for a complete, acknowledged approval checkpoint. Such a checkpoint can retire the old attempt and retain the task for a replacement. Without one, finalization strongly re-reads the task row before classifying a substrate failure. - Substrate state detects a dead VM; heartbeat staleness detects loss of the heartbeat writer inside a VM that still reports `RUNNING`. The independent heartbeat thread can continue during a pipeline hang, so a fresh timestamp is not proof of progress. The P3 supervisor saves intent before control calls and rechecks the gate before and after them. Its durable state retains an absolute service lifetime, consecutive failures, recovery start time and next delay. Three failed cycles or 120 seconds of unconfirmed wake cannot become an indefinite wait. AWS RUNNING does not end recovery while the guest remains stuck on a decided/expired approval; fresh guest liveness is required. API approve/deny commit first, then attempt a bounded wake without changing the decision response. Automatic suspension defaults off via `microvm_approval_suspend_enabled`; disabling new sleep preserves wake and cleanup. See the [supervisor runbook](/sample-autonomous-cloud-coding-agents/architecture/645-p3-supervisor). +Unanswered approvals have no deadline by default. A checkpointed MicroVM wait can +retire after an hour, or before that worker's lifetime ends. The coordinator +verifies immutable storage, fences the old attempt, confirms shutdown and then +releases capacity. A saved decision admits one replacement and invokes the +original published coordinator version. A scheduled manager retries lost signals +and performs terminal cleanup. See the [continuation protocol](/sample-autonomous-cloud-coding-agents/architecture/645-p3-continuation-protocol-20260917) +for ownership, capacity, expiry and failure behavior. + `TERMINATED` is the normal terminal signal and remains observable for at least 10 minutes. `ResourceNotFoundException` maps to completion only as a late fallback after the control-plane record is eventually reaped; polling does not wait for `NotFound`. **`/ping` health endpoint (AgentCore only).** The agent's FastAPI server responds to AgentCore's `/ping` calls while the coding task runs in a separate thread. AgentCore sees `HealthyBusy` and keeps the session alive. diff --git a/docs/src/content/docs/customizing/Cedar-policies.md b/docs/src/content/docs/customizing/Cedar-policies.md index 6fe17e11a..73db677c7 100644 --- a/docs/src/content/docs/customizing/Cedar-policies.md +++ b/docs/src/content/docs/customizing/Cedar-policies.md @@ -67,11 +67,14 @@ Every tool call the agent makes is evaluated as a Cedar `(principal, action, res |---|---|---|---| | `@rule_id("...")` | **Yes on soft-deny** (recommended on hard-deny) | Unique kebab/snake-case identifier | Stable ID for `--pre-approve rule:X`, for audit events, and for `bgagent policies show --rule X`. Engine rejects duplicates at task start. | | `@tier("hard"\|"soft")` | **Yes** | Exactly one of `"hard"` or `"soft"` | Must match the file section. Mismatches fail task start. | -| `@approval_timeout_s("N")` | No | Integer seconds ≥ 30 | Per-rule timeout. Defaults to 300 s (overridable per-task via `--approval-timeout`). When multiple soft rules match, the engine picks the minimum. Values < 120 s emit a load-time warning; values < 30 s are rejected. Ignored on hard-deny. | +| `@approval_timeout_s("N")` | No | Integer seconds ≥ 30 | Optional per-rule decision deadline. If absent, uses the task setting, whose default `0` means no deadline. The shortest positive task/rule deadline wins. Values < 120 s emit a load-time warning; values < 30 s are rejected. Ignored on hard-deny. | | `@severity("low"\|"medium"\|"high")` | No | One of three | Displayed in the approval prompt. Default: `medium`. | | `@category("...")` | No | `destructive`, `network`, `filesystem`, `auth`, or free-form | Optional UX grouping. Not enforced. | -**Rule of thumb:** every soft-deny rule must have `@rule_id` and should set `@severity` + `@approval_timeout_s` explicitly. Users scanning `bgagent pending` lean on these fields to triage quickly. +**Rule of thumb:** every soft-deny rule must have `@rule_id` and should set +`@severity`. Add `@approval_timeout_s` only when the workflow needs a decision +deadline. Omit the annotation to let the task choose; unlike the task's `0` +setting, a zero-valued rule annotation is invalid. ## Common patterns diff --git a/docs/src/content/docs/decisions/Adr-016-pluggable-identity-and-auth.md b/docs/src/content/docs/decisions/Adr-016-pluggable-identity-and-auth.md index d30b13960..147a62e9d 100644 --- a/docs/src/content/docs/decisions/Adr-016-pluggable-identity-and-auth.md +++ b/docs/src/content/docs/decisions/Adr-016-pluggable-identity-and-auth.md @@ -164,7 +164,7 @@ Per-user `McpCredential` selection requires the Gateway to know *which task-user | P6 | **Trusted task-user identity propagation** (prerequisite for per-user MCP on the general plane, P5): specify + validate a user-scoped inbound identity the Gateway authorizer trusts, replacing the M2M JWT for per-user credential selection. | **Blocks per-user `McpCredential`.** Until done, MCP credentials are workspace-scoped at best. | | P7 | Jira + Slack `ChannelCredential` (same shape as P1); GitHub `GithubOauth2` behind a flag, retire the shared PAT; OBO `act`-claim delegation feeding #237. | Flag-gated; per-surface. | -**Substrate exception — `lambda-microvm` (2026-09-02):** P1's vault cannot be enabled on the MicroVM substrate. The two together synthesize 505 resources against CloudFormation's hard 500-resource limit (MicroVM alone 496, the vault alone 488), so `AgentStack` refuses the combination at synth, naming both context flags. The MicroVM wiring itself is complete — `platform_config` carries the workload name and the guest execution role holds the mint grant — so this is a capacity limit, not a design gap, and it lifts as soon as a subsystem moves into a nested stack. +**Lambda MicroVMs:** The vault uses the guest's compute execution role, with its workload name delivered in authenticated `platform_config`. The former resource-count refusal was removed under #857 after stack reductions and nesting; backend/image-mode tests now include vault and gateway combinations, and check each template's resource and size budgets. See the [Linear setup guide](/sample-autonomous-cloud-coding-agents/using/linear-setup-guide#using-the-vault-with-lambda-microvms). **Substrate independence (verified 2026-07-21, both proven live):** the vault path works on any compute. AgentCore Runtime injects the Workload Access Token as the `WorkloadAccessToken` header; ECS/Fargate/Lambda bootstrap it via `GetWorkloadAccessTokenForJWT(workloadName, userToken=)` against a **standalone** (non-service-linked) workload identity, then call `GetResourceOauth2Token`. Runtime-managed (service-linked) workload identities cannot self-vend, so the ECS path needs a manually-created workload identity. The runtime execution role today has `GetWorkloadAccessToken*` but **not** `GetResourceOauth2Token` — P1 adds it, plus `GetSecretValue` scoped to that surface's providers (`bedrock-agentcore-identity!default/oauth2/*`; see the implementation notes below for why this is narrower than the wildcard first anticipated here). diff --git a/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md b/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md index 7499a11c5..19ec3107c 100644 --- a/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md +++ b/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md @@ -4,440 +4,221 @@ title: Adr 021 lambda microvms compute backend # ADR-021: AWS Lambda MicroVMs as a third ComputeStrategy backend -> **Implementation status (2026-09-17):** P1 ([#689](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/689)) and P2 ([#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733)) are merged. The takeover has a [clean deployment](/sample-autonomous-cloud-coding-agents/architecture/645-p2-clean-deployment-20260913), [live repository workflow evidence](/sample-autonomous-cloud-coding-agents/architecture/645-p2-live-task-20260914), and integrated P3 guest hooks, credentials, durable intent and approval/supervisor handling. The normal deployment uses agent image **7.0**, coordinator live alias **10** and bootstrap **1.9.0**. The [approval UX update](/sample-autonomous-cloud-coding-agents/architecture/645-p3-approval-ux-20260917) deploys atomic cancellation, channel notification support and guest cancellation handling. The [final image tests](/sample-autonomous-cloud-coding-agents/architecture/645-p3-final-image-and-ecs-20260917), [repository workflow](/sample-autonomous-cloud-coding-agents/architecture/645-p3-repository-path-20260917), and [MCP/network checks](/sample-autonomous-cloud-coding-agents/architecture/645-p3-mcp-network-20260917) provide nine successful image 6.0 approval workflows, including real credential renewal after expiry. -> -> **P3 remains incomplete; automatic suspension is disabled.** Lifecycle responses now explicitly close their HTTP connections before freezing, with [transport evidence and a deployed fix](/sample-autonomous-cloud-coding-agents/architecture/645-p3-connection-close-rollout-20260916). Historical service-side dispatch traces remain unavailable; passing workflows do not supply those traces. Remaining work includes normal activation/off-switch acceptance, approval UX work tracked in the [current implementation plan](/sample-autonomous-cloud-coding-agents/architecture/645-p3-implementation-plan), and deployment of the requested [nested MicroVM stack](/sample-autonomous-cloud-coding-agents/architecture/645-p3-nested-stack). Bootstrap **1.9.0** is installed; moving the existing flat deployment still requires a reviewed resource migration. Follow the [service-team feedback tracker](/sample-autonomous-cloud-coding-agents/architecture/645-lambda-microvm-service-feedback) for open service questions. This ADR defines P1–P3, not a P4. +> **Implementation status (2026-09-18): P1 and P2 are merged; P3 implementation and normal deployment acceptance are complete.** P3 adds approval sleep/wake, retained requests, conversation/workspace recovery, replacement workers and nested infrastructure. See the [completed plan](/sample-autonomous-cloud-coding-agents/architecture/645-p3-implementation-plan) and [normal acceptance record](/sample-autonomous-cloud-coding-agents/architecture/645-p3-normal-closure-20260918). This ADR defines P1–P3; it does not define an official P4. **Status:** proposed **Date:** 2026-07-29 ## Context -ABCA selects a per-repo compute backend through the Blueprint's `compute_type` field (`cdk/src/handlers/shared/repo-config.ts`). Before this ADR, two backends existed, resolved by `resolveComputeStrategy` (`cdk/src/handlers/shared/compute-strategy.ts`) behind a uniform `ComputeStrategy` interface (`startSession` / `pollSession` / `stopSession`): +ABCA selects compute per repository through Blueprint `compute_type`. AgentCore Runtime is the default; ECS Fargate supports larger workloads. [#645](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/645) adds Lambda MicroVMs for explicit lifecycle control and reduced compute usage during human approval waits. -- **AgentCore Runtime** (`agentcore`, default) — managed Firecracker MicroVM per session. Invoked via `InvokeAgentRuntime`; liveness is inferred from agent heartbeats in DynamoDB plus the FastAPI `/ping` endpoint (`agent/src/server.py`) — the strategy's `pollSession` is a stub that always reports `running`. Constraints: 2 GB image limit, no substrate-level suspend API exposed to the orchestrator. -- **ECS on Fargate** (`ecs`) — Fargate task (ARM64; current build default 4 vCPU / 16 GiB, configurable up to 16 vCPU / 120 GiB) for repos that exceed AgentCore's limits ([#596](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/596)). Invoked via `RunTask` in batch mode (bypassing the HTTP server); liveness via `DescribeTasks`. No suspend — an idle task blocked on an approval wait burns full compute the whole time. - -**AWS Lambda MicroVMs** (launched 2026-06-22) is a serverless Firecracker sandbox primitive that AWS positions explicitly for AI coding agents: VM-level isolation, snapshot-based near-instant launch, **suspend/resume with full memory + disk state preserved** (compute charges stop while suspended), a dedicated JWE-authenticated HTTPS endpoint per instance, lifecycle hooks (`/run`, `/suspend`, `/resume`, `/terminate`), and up to 8 hours per session. It is **not** classic Lambda: the 15-minute function cap does not apply, and [COMPUTE.md](/sample-autonomous-cloud-coding-agents/architecture/compute)'s "Lambda: poor fit" verdict refers to functions, not MicroVMs. - -[#645](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/645) proposes adding Lambda MicroVMs as a third `ComputeStrategy`. The fit is strong but not free — pre-implementation review of the service's lifecycle model surfaced real design tensions this ADR resolves: +Lambda MicroVMs are managed Firecracker virtual machines, separate from ordinary Lambda functions. A MicroVM can preserve memory and disk while suspended and can live for eight hours, including suspended time. The function service’s fifteen-minute limit does not apply. ### Capability comparison (delta rows only — full matrix in COMPUTE.md) -| | AgentCore Runtime | ECS on Fargate | **Lambda MicroVMs** | +| Capability | AgentCore Runtime | ECS Fargate | Lambda MicroVMs | |---|---|---|---| -| Isolation | MicroVM (managed) | Task-level (Firecracker) | MicroVM (Firecracker) | -| Max duration | 8 h | No cap | 8 h (running + suspended) — **verified** (`L-B430C318 = 8 hours`) | -| Suspend/resume | No orchestrator-visible API | No | **Yes** — explicit API + idle policy, state preserved, no compute charge while suspended. **Verified**: suspend reaches `SUSPENDED` in ~1 s with no `idlePolicy`; resume restores `RUNNING` in ~1 s with `microvmId` **and** `endpoint` byte-identical | -| Resources | AgentCore-managed | Build default 4 vCPU / 16 GiB; up to 16 vCPU / 120 GiB; 20–200 GB disk | **Baseline 8 GiB RAM / 4 vCPU, auto-scaling to a 32 GiB / 16 vCPU peak**; 32 GB disk. `minimumMemoryInMiB` configures the BASELINE (max 8,192 MiB); the service scales vertically on demand — baseline capacity and additional burst usage are billed separately | -| Packaging | ECR image ≤ 2 GB | ECR image, no hard cap | **Zip + Dockerfile in S3 → service-built snapshot image** (versioned, storage billed) | -| Invocation | `InvokeAgentRuntime` (SigV4) | `RunTask` + container overrides | `RunMicrovm` (image **ARN** required — a bare name is rejected) → dedicated HTTPS endpoint + JWE token (`CreateMicrovmAuthToken`, ≤ 60 min TTL) | -| Liveness | Agent heartbeat + `/ping` | `DescribeTasks` | MicroVM state (RUNNING / SUSPENDED / TERMINATED) via control-plane API **and** agent heartbeat (see sub-decision 1) | -| Session storage | `/mnt/workspace` FUSE (no `flock()`) | Ephemeral disk | Native disk in snapshot — **survives suspend/resume, `flock()` works** | -| Architecture | ARM64 | ARM64 | ARM64 (Graviton) | -| Regions (launch) | Broad | Broad | 5 (us-east-1/2, us-west-2, eu-west-1, ap-northeast-1) | - -> Rows marked **verified** were discharged empirically on 2026-07-31 (us-east-1); see `docs/verification/645-p1-lambda-microvm-runbook.md`. Two originally documented constants were **refuted** by that run and are corrected throughout this ADR: the `runHookPayload` cap (16 KB → 4 KB) and the claim that `RunMicrovm` accepts a bare image name (it does not). -> -> **Source hierarchy for service facts.** Where sources disagree, the higher tier wins, and the tier is recorded next to the claim: -> -> 1. **Live boundary probe** — a request the service actually accepted or rejected in this account/Region. Strongest, and the only tier that can refute the others. (Both corrections above came from here.) -> 2. **Modeled constraints** — the CLI/SDK request schema and enumerated allowed values (`--generate-cli-skeleton`, `ARM_64`, `ENABLED|DISABLED`). Authoritative about request *shape*; silent about runtime behaviour. -> 3. **Service developer guide** — including its sizing/scaling tables. Authoritative about *semantics* a probe cannot see, which is exactly how the memory row was fixed: a probe can only show that 32,768 MiB is rejected as a `minimumMemoryInMiB` value; only the guide explains that the field is a BASELINE and that the service scales to a 32 GiB peak on its own. A boundary probe is the strongest evidence about a boundary and says nothing about what the boundary *means*. -> 4. **SDK docstrings** — generated, and demonstrably stale here: `runHookPayload` documents "Maximum: 16,384 bytes" against an enforced 4,096. -> 5. **Launch blogs / skills / toolkit material** — orientation only; never load-bearing on its own. -> -> **Omitted API fields mean "service default", never "none".** Two live findings drove this rule: leaving `ingressNetworkConnectors` unset attaches a PUBLIC `HTTP_INGRESS` connector, and leaving `/ready` out of `hooks` makes the image un-creatable once any lifecycle hook is enabled. So for any security-relevant field, the desired posture must be **requested explicitly** and the test must assert the **outcome** (the ARN present in the request, the hook enabled) rather than the omission (`expect(field).toBeUndefined()`) — an omission assertion passes just as happily when the service is silently choosing something wider. -> -> On the **memory** row specifically: `CreateMicrovmImage` enumerates the baseline sizes a base image supports — `[512, 1024, 2048, 4096, 8192]` MiB for `al2023-1` — and rejects anything else, which is why the construct validates against that list at synth. The 32 GiB / 16 vCPU peak is reached by the service's own vertical scaling, not by asking for it. The account memory quota (`L-CD1C0CC4`, 1024 GB, "burst up to 4×") is an aggregate across MicroVMs, not a per-VM limit; note that concurrency arithmetic should be done against the PEAK, not the baseline, since that is what a busy fleet can actually consume. +| Packaging | ECR image, 2 GB limit | ECR image | ZIP + Dockerfile in S3 → versioned snapshot | +| Duration | Eight hours | No task duration cap | Eight hours, running + suspended | +| Explicit suspend/resume in ABCA | Unsupported | Unsupported | Control-plane APIs; memory and disk retained | +| Storage | Ephemeral disk plus preview persistent FUSE mount | Configurable ephemeral disk | 32 GB native disk; supports `flock()` | +| Sizing | Service-managed | Configurable; larger sustained workloads | ABCA baseline 8,192 MiB; service guide lists up to 32 GiB / 16 vCPU | +| Invocation/liveness | Invoke API + agent heartbeat | RunTask/DescribeTasks | RunMicrovm/GetMicrovm + agent heartbeat | + +See [COMPUTE.md](/sample-autonomous-cloud-coding-agents/architecture/compute) for the full comparison and costs. Suspending stops compute charges, but snapshot storage and save/restore charges remain. A shorter sleep delay does not guarantee lower total cost. + +The [P1 probes](/sample-autonomous-cloud-coding-agents/architecture/645-p1-lambda-microvm-runbook) established a 4,096-byte `runHookPayload` limit, an image-ARN requirement and accepted baseline values of 512, 1,024, 2,048, 4,096 and 8,192 MiB. Those observations override conflicting generated SDK descriptions for the tested account/Region. They do not measure guest-visible launch memory, vertical-scaling latency or sustained workload fit. The service guide’s capacity figures and live observations must remain distinguishable. ### Design tensions the strategy must resolve -1. **Idle detection is inbound-traffic-based; the ABCA agent is outbound-only.** MicroVM idle policies suspend when no traffic arrives at the *endpoint*. A busy agent running a 40-minute build receives no inbound traffic and would be suspended mid-work by a naive idle policy. Conversely, "no inbound traffic" is the agent's *normal* state. -2. **No self-suspend.** The agent cannot suspend its own MicroVM from inside; only an external `SuspendMicrovm` call can. Suspend decisions must be owned by the orchestrator — which aligns with the unified liveness model proposed in [#491](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/491). -3. **Snapshots bake state.** The image snapshot is captured once at build time; every MicroVM resumes from it. Secrets, tokens, and per-task identity must arrive at run time (`runHookPayload`, ≤ 4 KB, or fetched in the `/run` hook), never at image build. Application PRNG reseeding is a consequence of the same property and is scoped to **P3** — see the amended note under sub-decision 2 and the risk bullet, which record why the exposure is negligible today. -4. **Auth tokens are short-lived.** JWE tokens max out at 60 minutes; any orchestrator→agent HTTP interaction over the endpoint needs token refresh, unlike AgentCore's SigV4 invoke or ECS's no-endpoint model. -5. **Identity delta — narrower than it looks.** Most AgentCore services ABCA uses are standalone and substrate-portable: Memory is already consumed from ECS via an IAM grant plus `MEMORY_ID` (`EcsAgentCluster`), and Gateway (ADR-019) is portable by design (SigV4 inbound). The genuinely Runtime-coupled piece is the workload-access-token **delivery mechanism** (`runtimeUserId` → `WorkloadAccessToken` request header → `BedrockAgentCoreContext`, used by `resolve_linear_api_token()`), which has no MicroVM equivalent. The ECS backend already lives with this delta (env-var token delivery); MicroVMs inherit the same posture until the pluggable identity work ([#249](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/249), ADR-016) redesigns the seam. -6. **The service's defaults are not our posture.** Two of them, both discovered live: `RunMicrovm` attaches a **public** `HTTP_INGRESS` connector (and mints a public `*.lambda-microvm..on.aws` endpoint) when `ingressNetworkConnectors` is omitted, and `CreateMicrovmImage` **requires** the `/ready` hook whenever any lifecycle hook is enabled. Neither posture can be reached by leaving a field out — each needs an explicit control (see sub-decisions 3 and 4). +1. Traffic-based idle policies observe inbound endpoint traffic. A busy coding agent mostly sends outbound requests, so absence of inbound traffic cannot establish that it is idle. +2. Suspension needs an external controller. The agent can prepare a safe checkpoint, but the coordinator owns the service call. +3. Image snapshots share build-time process state. Credentials, task identity and deployment configuration must arrive after launch. +4. A worker’s eight-hour lifetime is shorter than an unanswered approval may remain useful. Durable task state must outlive the worker. +5. New packaging, IAM and lifecycle behavior need live checks; a successful CDK synth cannot validate service semantics. ## Decision -Adopt **AWS Lambda MicroVMs as a third, opt-in `ComputeStrategy` backend** named `lambda-microvm`, selected per repo via Blueprint `compute_type`. AgentCore remains the default. Five sub-decisions: +Add `lambda-microvm` as an opt-in `ComputeStrategy`. AgentCore remains the default. ### 1. Strategy shape: extend the interface with mandatory suspend/resume -`ComputeType` widens to `'agentcore' | 'ecs' | 'lambda-microvm'` (mirrored in `cli/src/types.ts` and the CLI's inline unions). `SessionHandle` gains a `{ strategyType: 'lambda-microvm', microvmId, endpoint }` variant — `microvmId` because every lifecycle API (`suspend-microvm`, `resume-microvm`, `terminate-microvm`, `get-microvm`, and `create-microvm-auth-token` — the latter not called in P1–P3, see sub-decision 3) takes only the MicroVM identifier, and `endpoint` because it is per-session (minted by `RunMicrovm`) and required for any orchestrator→agent HTTP interaction. Note the naming seam: the handle field is `microvmId` (matching `RunMicrovmResponse.microvmId`), while the request key on every lifecycle command is `microvmIdentifier` — the strategy and inline approval-wake helper translate between the two at their control calls. **P3 refinement (2026-09-15):** the handle also retains the actual `imageArn` and `imageVersion` returned by Run, plus `lifecycleProtocol` after exact-version verification. The requested image remains deployment configuration, but the version that launched an existing worker is per-session evidence. Save the known handle before optional discovery; check that exact version with `GetMicrovmImageVersion`, then conditionally persist support in the receipt and `compute_metadata`. Current deployment settings cannot substitute for missing legacy evidence. - -**A second, sharper naming seam: `imageIdentifier` must be an ARN.** The name suggests a bare image name is acceptable — `create-microvm-image --name` takes one, and this ADR originally assumed `run-microvm --image-identifier` would too. It does not: a bare name is rejected with `ValidationException: Malformed ARN - doesn't start with 'arn:'`, and so is `list-microvm-image-builds --image-identifier ` (`Invalid ARN format`). The construct therefore resolves an operator-supplied name to its exact `arn:${Partition}:lambda:${Region}:${Account}:microvm-image:${Name}` ARN **once** — the same value it scopes the lifecycle IAM grant to — and injects THAT as `MICROVM_IMAGE_IDENTIFIER`. One derivation, two consumers, so a request field and an IAM resource can never disagree. The strategy validates the invariant and fails fast with the remedy, because the service's own error names neither the env var nor the fix. - -The `ComputeStrategy` interface now has **mandatory** `suspendSession(handle)` / `resumeSession(handle)` methods returning `SessionLifecycleResult`, defined as `{ supported: false } | { supported: true }`. This local P3 foundation changes all three strategies together: AgentCore and ECS explicitly return false without calling AWS; MicroVM submits a command with a 10-second request bound. A future backend must implement both methods. The durable supervisor uses this explicit result without treating acknowledgment as final state. - -`supported: true` means AWS acknowledged the command, not that the target state was reached. The SDK's suspend/resume response bodies are empty. All operational failures, including conflicts and not-found responses, propagate through sanitized errors; no generic conflict is normalized into success without observed-state evidence. A timed-out command may still have committed and requires reconciliation. The installed SDK has no `RESUMING` state. `SessionStatus.microvmState` now supplies explicit observations because the coarse `running` result also includes PENDING/unknown. The local policy/store uses retained generations and sticky wake intent to handle a delayed suspend after approval; the durable supervisor now connects that policy, retains fixed recovery/session deadlines and waits for the required compute and guest observations. See `docs/verification/645-lifecycle-intent.md`. - -**Poll semantics — the strategy reports, the orchestrator interprets.** `pollSession(handle)` receives only the session handle and cannot see task state, so the health rules must live where the DynamoDB status lives. `SessionStatus` gains a `'suspended'` variant; the strategy maps `GetMicrovm` state mechanically and the **orchestrator** cross-references against the task row — the same division of labor `finalPollState` already uses for ECS (substrate stopped + non-terminal DynamoDB status → failed) and `pollTaskStatus` uses for agentcore heartbeats: substrate `suspended` + task `AWAITING_APPROVAL` is healthy (orchestrator-intended suspend); `suspended` with any other task status is an anomaly to surface, not fail-fast; substrate terminal + non-terminal task status → classify failed. - -**Liveness on this backend combines substrate state and the agent heartbeat.** `GetMicrovm` reports whether the VM is running or terminal. The independent `_heartbeat_worker` in `agent/src/server.py` refreshes the task timestamp every 45 seconds while the task is `RUNNING`; a stale timestamp detects loss of that writer, including a process crash or inability to write DynamoDB. It does **not** prove pipeline progress: a hung coding thread can coexist with a healthy heartbeat thread. A failed `/run` hook can trigger service teardown; after an accepted hook, ABCA still needs explicit termination and the maximum-duration backstop. A general progress watchdog remains separate work. +All strategies implement `startSession`, `pollSession`, `stopSession`, `suspendSession` and `resumeSession`. The lifecycle methods return an explicit supported/unsupported result. AgentCore and ECS return unsupported without making suspension API calls; MicroVM bounds each control request. An accepted API request does not prove the transition completed. -AgentCore and MicroVM tasks start the same periodic heartbeat worker, so they use the same 120-second grace and 240-second stale thresholds. ECS starts the pipeline directly, bypassing `server.py`; its pipeline writes a timestamp once but does not start this worker. Enabling the same stale check for ECS would therefore fail healthy tasks after roughly four minutes. The check applies only to `RUNNING`, not `AWAITING_APPROVAL`. The transition back to `RUNNING` refreshes `agent_heartbeat_at` in the same conditional transaction that clears the matching approval request. A poll immediately after a long gate therefore sees a fresh heartbeat before the next worker tick. The local P3 supervisor additionally bounds recovery through suspension and decision consumption; live verification remains required. +The MicroVM handle records `microvmId`, `endpoint`, the actual launched `imageArn`/`imageVersion`, and verified `lifecycleProtocol` when available. Lifecycle request fields use `microvmIdentifier`. Save the known handle before optional image discovery so a failed capability check cannot lose cleanup ownership. Current deployment settings cannot establish an older worker’s capabilities. -The service's `MicrovmState` enum has **six** members, not three, so the mapping is stated exhaustively (one line of rationale each, mirrored in the strategy's doc comment): - -| `MicrovmState` | `SessionStatus` | Why | +| Service state | Strategy result | Coordinator responsibility | |---|---|---| -| `PENDING` | `running` | Still booting; the same way ECS's `PENDING`/`PROVISIONING` map to `running`. | -| `RUNNING` | `running` | — | -| `SUSPENDING` | `suspended` | Already on its way to frozen; reporting `running` would tell the orchestrator compute is still progressing when it is not. Both suspend states land on a report the orchestrator treats as benign-or-anomalous depending on task status, never as failure. **Not observed in the recorded probe** — suspend reached `SUSPENDED` in under 1 s. Other runs may expose this state, so handle it without requiring it to appear. | -| `SUSPENDED` | `suspended` | — | -| `TERMINATING` | `completed` | Terminal-bound and carries no exit code, so "the substrate is gone" is all the strategy can honestly say. | -| `TERMINATED` | `completed` | Success vs failure is the orchestrator's call — it cross-references the DynamoDB status. **This is the load-bearing terminal signal**, not `NotFound` (see below). | -| *unrecognized* | `running` | A future service enum addition must never fail a healthy task; the strategy warns and keeps polling. | - -**`GetMicrovm` `ResourceNotFoundException` → `completed`, but as a LATE fallback.** This deliberately diverges from `ecs-strategy`, where `DescribeTasks` returning no task maps to `failed`. ECS keeps stopped tasks describable for roughly an hour, so a missing task there really is anomalous; a MicroVM is eventually reaped from the control plane **by design**, so `failed` would fail tasks that finished cleanly. - -What the live run corrected is the *timing*, not the mapping. `NotFound` is **not the near-term terminal signal**: a terminated MicroVM reported `TERMINATING` at +1 s, `TERMINATED` at +3 s, and was **still `TERMINATED` ~10 minutes later** and at every subsequent checkpoint — `ResourceNotFoundException` was never observed in that window. So the branch that actually fires in practice is `TERMINATED → completed` in the table above; the `NotFound` rule covers only a VM reaped after a long gap (a poll resumed after a crash, a stranded-task reconciler sweep). Both are required and neither substitutes for the other: without the `TERMINATED` row the orchestrator would poll a finished VM until its safety net fired, and without the `NotFound` rule a late sweep would classify a cleanly-finished task as a poll error. - -The divergence is safe because it does not weaken detection: the orchestrator still fails the task when a terminal report lands while the DynamoDB status is non-terminal, so a genuine mid-run disappearance is caught — it simply receives the substrate-failure classification instead of a misleading poll error. Because that cross-check acts on a status read earlier in the same poll cycle, the orchestrator **re-reads the task row before failing** (the normal shutdown order is "agent writes terminal status → agent exits → VM terminates", which a stale read would otherwise turn into a spurious failure); ECS buys the same protection with a five-consecutive-poll patience counter instead. - -Neither the mapping nor the NotFound rule is a health decision: both are mechanical restatements of substrate state, which is what keeps the "strategy reports, orchestrator interprets" split intact. - -Normative requirements (EARS, per [ADR-020](/sample-autonomous-cloud-coding-agents/architecture/adr-020-ears-requirements-syntax)): - -- When a task's Blueprint sets `compute_type: 'lambda-microvm'`, the orchestrator shall resolve the `LambdaMicrovmComputeStrategy` via `resolveComputeStrategy`. -- When `startSession` is invoked, the strategy shall call `RunMicrovm` with `maximumDurationInSeconds` set to 28 800 (the service maximum, matching AgentCore's 8-hour session cap and sitting inside the orchestrator's ~8.5 h safety-net poll window). -- When `startSession` is invoked, the strategy shall pass a fully-qualified MicroVM image ARN as `imageIdentifier`. -- If the configured image identifier is not an ARN, then the strategy shall fail the session start with an error naming the environment variable and the redeploy remedy, before performing any AWS call. -- When `startSession` returns, the orchestrator shall persist the MicroVM handle (`microvmId`, `endpoint`, actual `imageArn`/`imageVersion` when returned, and verified `lifecycleProtocol` when supported) in the task row's `compute_metadata` (the field `cancel-task.ts` already reads ECS handles from). -- The strategy shall omit `idlePolicy` on every `RunMicrovm` call, in every phase. -- The orchestrator shall be the sole initiator of suspension, via `suspendSession`. -- When `pollSession` observes MicroVM state `SUSPENDED` or `SUSPENDING`, the strategy shall report `suspended` without interpreting task state. -- When `pollSession` observes MicroVM state `TERMINATED` or `TERMINATING`, the strategy shall report `completed` (the observable terminal state persists for at least ~10 minutes, so `completed` shall not depend on the MicroVM being reaped). -- When `pollSession` observes a MicroVM state it does not recognize, the strategy shall report `running`. -- If `GetMicrovm` reports that the MicroVM does not exist, then the strategy shall report `completed`. -- If the strategy reports a terminal substrate state while the task's DynamoDB status is non-terminal, then the orchestrator shall re-read the task row and, if it is still non-terminal, classify the task as failed with a substrate-failure remedy. -- If the strategy reports `suspended` while the task's DynamoDB status is not `AWAITING_APPROVAL`, then the orchestrator shall surface an anomaly event and shall not fail-fast the task. -- While a `lambda-microvm` task's DynamoDB status is `RUNNING`, if the task's `agent_heartbeat_at` is stale (or absent past the grace window) by the same thresholds the orchestrator applies to `agentcore`, then the orchestrator shall treat the session as unhealthy and stop polling — the substrate `GetMicrovm` check shall remain the crash detector, and the heartbeat shall detect loss of the in-guest heartbeat writer, without claiming to detect every pipeline hang. -- The task-detail API response shall include `agent_heartbeat_at`, and the CLI shall surface it while the task is non-terminal (P2r2-F11: the field drove the orchestrator's heartbeat check but was never projected, so no operator could observe the signal — and its invisibility produced a wrong verification conclusion). -- The task-**summary** API response (`GET /v1/tasks`) shall also include `agent_heartbeat_at`, and `bgagent list` shall render it as an age column. Extending the field to the list response is a deliberate widening of the requirement above rather than an incidental one: the detail-only projection makes liveness a per-task question, and an operator checking a fleet of tasks one `bgagent status` at a time is exactly how the hung task P2r2-F11 describes went unnoticed. Same suppression rule as the detail view (terminal tasks and never-beaten tasks render a placeholder) so the two views cannot disagree. -- If `suspendSession` or `resumeSession` is invoked on a strategy that does not support suspension, then the strategy shall return an explicit unsupported result. -- When the agent process reaches a terminal state, the agent shall exit. -- When the orchestrator finalizes a `lambda-microvm` task, the orchestrator shall call `terminate-microvm` (termination shall not rely on any substrate timeout, and shall not rely on the MicroVM self-terminating — it does not). - -*On the omitted `idlePolicy`:* if the block is present all three fields are required, so omission is the unambiguous disabled state the invariant test asserts. This deliberately forgoes `suspendedDurationSeconds` — it lives *inside* `idlePolicy` and cannot be set without re-enabling the traffic-idle machinery — so the suspended-state bound is `maximumDurationInSeconds` plus orchestrator termination and the stranded-approval reconciler (see sub-decision 2). A tighter substrate-level suspended-TTL remains available later as an additive `idlePolicy` change if operators want it. *On the fixed `maximumDurationInSeconds`:* no wall-clock task budget exists in the platform (budgets are `max_turns` / `max_budget_usd`), so the value is parity with AgentCore's 8 h cap rather than derived policy; a Blueprint override can be added later if a real need appears. - -### 2. Lifecycle: suspend/resume reconciled with the agent-owned approval poll - -The headline economic win is suspend during **HITL approval waits** (Cedar approval gates, [CEDAR_HITL_GATES.md](/sample-autonomous-cloud-coding-agents/architecture/cedar-hitl-gates)): while a task waits on a human decision, the MicroVM is suspended (compute charges stop; memory/disk state — cloned repo, warm build caches — is preserved) and resumed when the decision lands. Under Cedar decision #6 the approval window is bounded (default 300 s, ceiling 1 h, timeout → deny), so the saving per gate is bounded at ~1 h of compute — real at 16 vCPU, and it makes any future extension of gate ceilings (the off-hours posture §14.8 deliberately defers) cheap on this backend. - -The handshake must respect the existing approval mechanics: the agent **discovers decisions itself** by polling DynamoDB (`_poll_for_decision`, monotonic timeout), the approve/deny Lambda commits the decision transaction before writing separate coordinator wake intent, and `AWAITING_APPROVAL` holds the concurrency slot (Cedar decision #7). Nothing "delivers" an approval to the agent, and suspension freezes the agent's monotonic clock — so the design is: - -- **Suspend — orchestrator-owned.** The orchestrator's durable poll observes `AWAITING_APPROVAL` on a `lambda-microvm` task and calls `suspendSession` after a grace period, and only when the gate's remaining window exceeds grace + resume overhead (suspending a 30 s gate is pure loss). Suspend is a policy decision on a poll observation, not a user action. -- **Resume — inline in the approve/deny Lambdas, orchestrator poll as backstop.** After the transactional decision write commits, `ApproveTaskFn`/`DenyTaskFn` load the MicroVM handle from the task row's `compute_metadata` (persisted at session start — the same field `cancel-task.ts` reads ECS handles from) via a post-commit strongly-consistent `GetItem`, then save wake intent, observe the VM and request `ResumeMicrovm` **best-effort** only when SUSPENDED: on failure they log a warning and write a resume-orphan task event; the decision response never fails on a compute error (the decision row is already durable). The orchestrator poll reconciles: current approval row APPROVED/DENIED + MicroVM still `SUSPENDED` → retry resume; a PENDING row alone must not trigger wake-up. - - *Why inline rather than poll-only — codebase precedent:* resume-on-approve is structurally identical to task cancellation — a user-initiated, latency-sensitive action whose purpose is an immediate compute-lifecycle side effect. `cancel-task.ts` already resolves this exact tension: the API-plane handler invokes ECS `StopTask` / AgentCore `StopRuntimeSession` inline, best-effort (a failed stop logs a warning and the state transition stands; a `task_cancel_compute_orphan` event is written when no stoppable compute handle exists, `reason: missing_runtime_handle`) — with the conditional IAM wired in `task-api.ts`. The resume path goes one step further than the precedent by also writing the orphan event on *failed* resume calls, because a failed resume strands a suspended VM awaiting a decision — a stronger liveness consequence than a failed stop of an already-cancelled task. The alternative (orchestrator-poll-only resume) preserves single-owner lifecycle purity but pays up to a full poll interval (~30 s) of latency on every approval, and the purity argument was already litigated and declined for cancel. `approve-task.ts` is deliberately minimal today (security-critical ownership comparison, Cedar finding #6); the resume call is therefore added *after* the transaction commits, cannot alter the decision outcome, and carries one conditional `lambda:ResumeMicrovm` grant — the same blast-radius trade the cancel handler accepted in review. -- **Timeout under freeze — the agent re-bases on the wall clock it already owns.** The agent's monotonic gate timer freezes while suspended, so resuming near the deadline is not enough: the frozen timer would still hold its remaining budget and fire the deny minutes *after* the user-visible window — colliding with the approval row's TTL (`created_at + timeout_s + 120s`) and triggering the "row reaped → stranded" fallback on a healthy gate. Instead, the gate expires at **`min(monotonic budget, created_at + timeout_s)`**, evaluated on each poll iteration and on `/resume`. This is not a new principle: Cedar decision #6 is already "min wins" for timeouts, the wall-clock deadline is already durable in the approval row the agent itself writes (`created_at` is recorded by the agent; clock changes across suspend still need testing), and §13.12's late-approval race fix already establishes that the durable row is authoritative over the agent's local timer. Deny authority stays agent-side (the conditional `TIMED_OUT` write + ConsistentRead re-read race protection is untouched); the orchestrator's resume at `deadline − margin` is purely the wake-up mechanism, with no correctness role. -- **Backstops, not mechanisms.** `maximumDurationInSeconds` (mandatory on every `RunMicrovm`, pinned at 28 800 s — see sub-decision 1) is the substrate kill switch bounding running **and** suspended time; the orchestrator's finalization `terminate-microvm` is the active cleanup path; the stranded-approval reconciler retains its role for orphaned waits. No `idlePolicy`-based bound is used in any phase — see sub-decision 1's omit-`idlePolicy` invariant. - - **The active terminate is still mandatory on the SUCCESS path, and P2 sharpened why.** P1 concluded flatly that "nothing self-terminates": a hook-less MicroVM reached `RUNNING` in 12 s and stayed there with no `stateReason` through every checkpoint. P2 refuted that *for the failure path only* — with `run: ENABLED`, a run hook that answers 4xx makes the **service** terminate the VM within ~12 s, `stateReason: "Run lifecycle hook returned HTTP status 400. Please check your hook endpoint and application logs for more details."`, after which `suspend-microvm` correctly refuses it. That is a real improvement in cost posture and a direct benefit of declaring hooks (see also the failure-path row in the phasing table, sub-decision 3). +| RUNNING or starting | `running` | Read task state and applicable heartbeat | +| SUSPENDING / SUSPENDED | `suspended` | Reconcile the approval and lifecycle intent | +| TERMINATING / TERMINATED / not found | `completed` | Re-read task state; distinguish completion, recoverable checkpoint and failure | +| Unknown future state | `running`, with warning | Continue bounded observation | - It does **not** relieve the orchestrator of anything, because the two cases are disjoint. The service reaps a hook *result* it did not like; it has no view of the guest once the hook returned 200. So a task that starts normally — the overwhelming majority — has no service-side reaper at all, and a VM whose pipeline finished, crashed after `/run`, or hung is reaped by nobody but `TerminateMicrovm`. A leaked handle therefore remains a cost incident that bills until the 8 h cap; only the "the guest rejected its own payload" corner now cleans itself up. -- **Concurrency slot stays held** during suspend. Cedar decision #7's rationale ("container alive, consuming memory") weakens under suspend, and the harder replacement rationale — "AWS counts `SUSPENDED` MicroVMs toward the account memory quota, so releasing ABCA's slot would not free real capacity" — is **undischarged**: the suspended VM stayed in `list-microvms` at every checkpoint, but that only proves *listed*. `L-CD1C0CC4` (1024 GB, account-scoped) exposes no `UsageMetric`, `AWS/Usage` carries only `CallCount` per API, and no MicroVM memory metric exists in any namespace, so consumption is **not observable safely** — proving it would need a large concurrent fleet. The conclusion (hold the slot) stands as the conservative choice, not as a verified fact. Size the arithmetic against the 32 GiB **peak** rather than the 8 GiB baseline: a busy fleet scales up, so peak is what actually competes for the account quota. +A strategy result is not a task outcome. A terminal substrate can leave a recoverable pending-approval checkpoint; otherwise a non-terminal task needs failure classification. Capacity release separately requires confirmed physical shutdown, not merely the strategy’s `completed` result. -In P3, the agent's `/suspend` hook must acknowledge durable progress before returning 200 within its hook budget; `/resume` must refresh credentials and reseed application PRNG state from fresh OS entropy. These hooks are deployed in image 6.0 with atomic checkpoints, retained credential refresh and original-gate reconciliation, and have passed the isolated live workflows linked above. Automatic suspension on the normal task path remains disabled pending rollout acceptance. A non-cryptographic PRNG such as Python `random` remains unsuitable for secrets even after reseeding; security-sensitive randomness must use an OS-backed cryptographic source. - -**Amended (P2 review): Application PRNG reseeding is P3 scope, and the P2 exposure is negligible.** The original EARS requirement below implied a `/run`-time reseed had to land with P2; it did not, and shipping P2 with the requirement unmet-but-asserted was itself the defect. Measured exposure, which is what changed the scoping: - -- The only consumer of a non-cryptographic PRNG in `agent/src` is `progress_writer.py`'s `getrandbits(80)`, used for the random half of a **ULID**. `os.urandom` / `secrets` are not seeded from the snapshot at all, and nothing in the agent derives a key, token, or nonce from `random`. -- That ULID is a DynamoDB **sort key under a `task_id` partition**. A collision therefore needs two events in the *same task* at the *same millisecond* with the same 80 random bits — and identical PRNG state across two MicroVMs restored from one snapshot does not produce that, because the events are in different partitions. The worst case is a duplicate progress event within one task, not a security boundary. -- Credential safety is a separate concern from the application PRNG. Build hooks avoid initializing AWS clients and their credential caches; their credential-environment warning does not prove that every possible image is secret-free. `platform_config.agent_session_role_arn` delivers a role identifier, not credentials. The running agent obtains credentials through its runtime provider and assumes the task-scoped role. - -So the reseed moves to P3 alongside `/suspend` + `/resume`, where a *resumed* VM — which really does continue with the exact PRNG state it was frozen with, repeatedly — makes it load-bearing rather than theoretical. P3 must reseed on both `/run` and `/resume`, and must not treat the ULID as the only consumer: security-sensitive values must continue to use `secrets` or an equivalent cryptographic source, not `random`. +AgentCore and MicroVM start a heartbeat writer every 45 seconds. Their RUNNING tasks use the same 120-second startup grace and 240-second stale threshold. ECS’s batch entrypoint does not start that writer, so applying this check to ECS would fail healthy tasks. Heartbeats establish writer liveness, not progress of every coding thread. Returning from approval to RUNNING refreshes the heartbeat atomically. Task detail/list APIs and CLI views expose the timestamp. Normative requirements (EARS): -- While approval suspension is enabled for a verified compatible worker, a `lambda-microvm` task is in `AWAITING_APPROVAL`, and the gate's remaining window exceeds the configured grace period plus resume overhead, the orchestrator shall call `suspendSession` after the grace period. -- When the approve or deny Lambda commits a decision for a `lambda-microvm` task, the Lambda shall load the MicroVM handle from `compute_metadata`, persist wake intent for the current identity, and request `ResumeMicrovm` best-effort after observing SUSPENDED and rechecking the gate. -- If the inline resume fails, then the Lambda shall record a resume-orphan task event and shall still return the decision outcome. -- While the current approval row is APPROVED or DENIED and the MicroVM remains `SUSPENDED`, the orchestrator shall retry `resumeSession`. A PENDING row alone is not a wake-up condition. -- While a `lambda-microvm` task waits on an approval gate, the agent shall evaluate gate expiry as the earlier of its monotonic budget and the row's wall-clock deadline (`created_at + timeout_s`), on each poll iteration and on `/resume`. -- If gate expiry is reached without a decision, then the agent shall deny. -- If no decision arrives by the gate's wall-clock deadline minus the resume margin, then the orchestrator shall resume the MicroVM so the agent can evaluate expiry and fire the deny agent-side. - -### 3. Packaging: same agent image source, new build path - -The existing agent container (`agent/` Dockerfile, already ARM64) is repackaged as a zip + Dockerfile artifact in S3 and built into a versioned `MicrovmImage` via `CreateMicrovmImage`. The agent runs its existing FastAPI server (`agent/src/server.py`) — the MicroVM path uses the HTTP entrypoint like AgentCore, not ECS's batch bypass — plus the runtime lifecycle hooks (`/run`, `/suspend`, `/resume`, `/terminate`) and the `/ready` + `/validate` build hooks, all on the same port the server already listens on (8080, declared as the image's `hooks.port`). Runtime hooks are fast-notification only (1–60 s): `/run` validates the payload and starts the pipeline **asynchronously**, mirroring how the agent loop already runs in a background thread behind `/ping` on AgentCore. +- When a Blueprint selects `lambda-microvm`, the orchestrator shall use the MicroVM strategy and persist its handle in `compute_metadata`. +- Every launch shall use a full image ARN, `maximumDurationInSeconds=28800`, explicit `NO_INGRESS` and no `idlePolicy`. Invalid image configuration shall fail before launch. +- Before allowing suspension, the coordinator shall verify the actual launched image version and lifecycle protocol. +- When a suspended state conflicts with task state, the coordinator shall record and reconcile the anomaly within bounded recovery windows. +- When the task ends, the coordinator shall actively terminate its MicroVM. The service lifetime is a backstop, not the normal cleanup mechanism. +- Uncertain starts, lost replies and supervisor replay shall preserve the original worker identity, ownership and lifetime rather than launch duplicate workers. -**Hook phasing — corrected: `/ready` + `/run` are both P1.** The original plan split *declaring* a hook from *serving* it, putting `/run`'s declaration in P1 and its implementation in P2. Live verification proved that split is **not a reachable service state**, on two independent counts: +### 2. Lifecycle: suspend/resume reconciled with the agent-owned approval poll -- `CreateMicrovmImage` rejects an image that enables any lifecycle hook without `/ready`: *"The ready (/ready) MicroVM image hook must be enabled when any MicroVM lifecycle hook (run, resume, suspend, or terminate) is enabled."* So a P1 image declaring only `/run` is **not creatable**. -- With `/ready` added but unserved, both chipset builds fail: *"Ready hook check failed: the application returned a client error (HTTP 4xx) response."* So a declared hook must be served in the same phase. -- And an image with **no** hooks at all — the only other creatable shape — cannot receive a payload: *"The run hook must be enabled in the MicroVM image to pass the run hook payload."* So deferring hooks entirely also defers the whole payload-delivery channel. +Unanswered approvals have no deadline by default (`approval_timeout_s=0`). Explicit task deadlines range from 30 to 3,600 seconds; a positive policy-rule deadline can also apply. The independent sleep preference defaults to 600 seconds per approval wait. `microvm_sleep_after_s=0` keeps the task awake. -The phasing is therefore: +Automatic suspension also requires the deployment’s `microvm_approval_suspend_enabled` opt-in, which defaults false for new deployments. A live Parameter Store switch lets existing durable executions stop initiating new suspensions without changing their pinned Lambda environment. The verified normal deployment has this opt-in enabled. Turning it off does not abandon already-suspended workers. -| Hook | Declared by | Served by the agent | Notes | -|---|---|---|---| -| `/ready` | **P1** (construct enables `hooks.microvmImageHooks.ready`) | **P1** | MANDATORY, not a quality nicety — see above. A 200 proves uvicorn is bound and `server` imported cleanly (pulling in `pipeline` → `runner` → the policy engine), so a missing policy file fails the BUILD instead of the first task. **Since P2-F5 it also WARMS the snapshot** — the hook's 200 is what the service waits for before capturing the snapshot, making this the only place a warm page can be created, and the 225 MiB `claude` binary was cold in it (see the P2-F5 correction below). A required warm-up failure answers 503, so a snapshot that cannot exec the agent's own CLI fails the image build instead of every task. Still makes ZERO AWS calls, logging included (a `--version` exec is neither an AWS call nor a network call). | -| `/run` | **P1** (construct sets `hooks.microvmHooks.run`) | **P1** | The payload-delivery channel. Must be served in P1 because `/ready` forces hooks to exist at all, and a hook-less image cannot accept `runHookPayload`. Since P2 it is also the **platform-configuration** channel (see "Platform configuration delivery" below). | -| `/validate` | **P2** (construct sets `hooks.microvmImageHooks.validate`) | **P2** | An **image** (build-time) hook, and a **shallow self-check only**: server alive, hook routes registered, interpreter + contract sanity. It runs under the BUILD role, which deliberately holds no Bedrock / Secrets Manager / DynamoDB grants, so it must make **zero AWS API calls** and must not touch credential resolution — the "deeper warm-up assertions (Bedrock reachability, Memory access, tool availability)" this ADR originally assigned here are **not implementable**: every one of them would `AccessDenied` and fail every image build. They belong to the first task's own error handling. 200 when the checks pass, 503 while still initialising. | -| `/terminate` | **P2** (construct sets `hooks.microvmHooks.terminate`) | **P2** | Best-effort final flush: a final structured log line, then 200 — always, inside the hook budget, even with nothing running. It must **not** write terminal task status (the orchestrator normally finalizes before `TerminateMicrovm`, and also attempts cleanup if database finalization fails, so a terminate hook that wrote a status would race that finalization and could clobber the real outcome) and must not join the pipeline thread. `ProgressWriter` writes synchronously but drops failed events, so this hook cannot guarantee that every progress write succeeded. P3 supplies a separate acknowledged checkpoint for suspension. "Always 200" also covers the BODY: the handler reads the raw request rather than a typed model, because a typed body is validated before the handler runs and would answer 422 to malformed JSON — a reported hook failure on a successful teardown. Safe to declare because `TerminateMicrovm` removes the VM with or without in-guest cooperation. **Correction (P2-F8):** the service sends `microvmId: ""` on this hook, unlike `/run` where it is populated, so an empty id is expected-normal and this hook cannot join the guest record to the control-plane one — `/run`'s accepted line carries that correlation instead. | -| `/suspend`, `/resume` | **P3**, deployed in image 6.0 | **P3**, deployed in image 6.0 | Managed images declare the served checkpoint/credential/gate hooks with the shared 30-second service timeout and protocol marker. The coordinator verifies the actual launched version before allowing new suspension. Lifecycle responses close the connection before freezing. Isolated live workflows pass; normal automatic-suspension activation remains gated off. | +The coordinator observes the exact pending gate and waits for the sleep delay. It skips sleep when a timed gate has too little time left before its wake margin. The guest holds a coding barrier, drains acknowledged progress and commits a checkpoint before accepting `/suspend`. Lifecycle HTTP responses explicitly close their connections before freeze to avoid reuse of a stale pooled connection; see the [transport experiment](/sample-autonomous-cloud-coding-agents/architecture/645-p3-wake-transport-20260916). -Consequence to state plainly, replacing the original "a P1-built MicroVM image is not runnable end to end": **a P1 image is creatable, launchable and payload-deliverable, but carries no smoke-parity guarantee.** P1 delivers the strategy, the construct, the roles/buckets/connectors, the image resource, the packaging script, and the `/ready` + `/run` endpoints — so a `lambda-microvm` task can start a MicroVM and hand it a payload. What P1 has **not** established is anything P2 owns: AgentCore Memory grants and `MEMORY_ID` delivery, the agent's non-secret env parity inside the snapshot, egress specifics from a running MicroVM, and heartbeat/progress behaviour end to end. At the P1 handoff, no clone → change → PR run had happened. P2 initially demonstrated that path with an IAM workaround; the 2026-09-14 clean deployment and coding/iteration/cancellation runs later passed without manual IAM changes. The broader failure/recovery, effective IAM and networking matrix remains open. The construct and the packaging script surface that remaining scope at synth/run time (`abca:microvm-image-p1-smoke-unverified`) so an operator cannot mistake a launchable substrate for a verified one. +Approval and denial handlers commit the decision first. They then read the current handle consistently, persist wake intent and request resume best-effort. A wake failure records diagnostics and does not undo the accepted decision. The durable supervisor retries and observes both service state and guest consumption of the decision; RUNNING alone does not prove the tool was released. -**The `AWS::Lambda::MicrovmImage` L1 enforces the API's enums (P2-F2, live 2026-08-06).** This closes the one item P1 left explicitly open, and it closes it against the construct's own stated reasoning. CloudFormation's generated types make `cpuConfigurations[].architecture` and all four `hooks.*` fields plain strings and document no allowed values, from which P1 concluded that the CloudFormation surface takes a *hook path* while the API takes an `ENABLED`/`DISABLED` flag, and that both were correct for their own surface. CloudFormation refused the change set at **early validation** — the stack was never touched, so there was no rollback and no runtime symptom to trace back — on five values: +A sleeping worker retains its concurrency reservation. For a longer wait, ABCA verifies a complete, version-pinned conversation/workspace checkpoint, fences the attempt, confirms shutdown and releases the reservation. A later decision can admit one replacement through the original published coordinator. It restores Git state, required workspace files, the actual SDK conversation, exact pending tool inputs and cumulative usage with fresh scoped credentials. See the [continuation protocol](/sample-autonomous-cloud-coding-agents/architecture/645-p3-continuation-protocol-20260917). -``` -/aws/lambda-microvms/runtime/v1/run is not a valid enum value. Supported values: [DISABLED, ENABLED] - (at /Resources/…/Properties/Hooks/MicrovmHooks/Run) … and the same for Terminate, Ready, Validate -arm64 is not a valid enum value. Supported values: [ARM_64] - (at /Resources/…/Properties/CpuConfigurations/0/Architecture) -``` +Normative requirements (EARS): -Three consequences. First, the CloudFormation surface is **identical** to the API surface, and the packaging script (`--cpu-configurations '[{"architecture":"ARM_64"}]'`, `--hooks '{"microvmHooks":{"run":"ENABLED",…}}'`) had it right all along. Second, the "CDK-managed (recommended)" bootstrap path was **non-functional** for the whole of P1 and P2 — the out-of-band `--create-image` script was the only working path — and no unit test, `cdk synth` or cdk-nag rule could see it, because the types accept any string. Third, **hook paths are not configurable on either surface**: the service calls fixed well-known routes (proved by the build and run logs, which POST to exactly the `/aws/lambda-microvms/runtime/v1/*` paths the agent serves), so the route constants in the construct are an agent-side cross-package contract ONLY and must never be sent as property values again. Also discharged in passing: the `microvmImageHooks` property name and nesting are correct — CloudFormation resolved `…/Hooks/MicrovmImageHooks/Ready` and objected only to its value. +- While all sleep gates and guest safety checks hold, the coordinator shall suspend after the configured delay; it shall remain the sole initiator of suspension. +- When a decision is committed, the API shall preserve that outcome even if optional wake or replacement dispatch fails. +- A timed request shall retain its original wall-clock deadline and, within the original process, its monotonic cap. The earlier limit wins; wake/replacement shall not restart either decision window. +- Before an unanswered timed gate reaches its deadline, the coordinator shall wake the worker with the configured margin. For a retired attempt, it may resolve expiry atomically while admitting a replacement. +- Before releasing a worker reservation, the coordinator shall verify checkpoint integrity, fence ownership and confirm physical shutdown. An uncertain control request shall not release capacity. +- A replacement shall preserve the task/request/tool identity and usage totals, obtain fresh scoped credentials and use a new authenticated launch reference. +- Pending requests shall have no storage TTL. Cancellation or terminal cleanup shall close unanswered requests and preserve recorded decisions without extending an existing retention TTL. +- The guest shall reseed application PRNG state from OS entropy on `/run` and `/resume`; security-sensitive values shall continue to use cryptographic randomness. -**A snapshot is only as warm as the pages touched before it was captured (P2-F5, live 2026-08-07).** This is the defect that stopped the P2 smoke run one step short of a pull request, and it is a property of the substrate rather than a bug in any one file. Every task failed at turn 0, reproducibly: +The platform verifies ownership and the exact approved action. It does not implement a separate semantic relevance checker; the agent decides whether the proposed work still makes sense. -``` -TimeoutExpired: Command '['claude', '--version']' timed out after 10 seconds -``` +### 3. Packaging: same agent image source, new build path -The binary was fine — in the identical image, locally, `claude --version` answers `2.1.191 (Claude Code)` in under a second. It is a **225 MiB (236,305,136-byte) statically-linked ELF** that nothing had exec'd before the snapshot was taken, so on a guest restored ~50 s earlier the first `exec` had to fault all of those pages in from lazily-restored storage, and 10 s was not enough. `/ready` existed precisely so "the snapshot is taken with a warm server", and the snapshot was warm for uvicorn and stone cold for the binary that does all the work. +Package the existing ARM64 `agent/Dockerfile` and its local inputs into a deterministic ZIP. Managed builds use `microvm-images/agent-artifact-.zip`; deploy the digest with the managed base-image ARN/version. A changed object URI triggers CloudFormation to build a new image version. The first deployment may create only infrastructure so the artifact bucket exists before upload. An external image identifier supports out-of-band builds. -Both halves of the fix are kept, because they answer different questions. `/ready` now **exec's the heavyweight binaries before returning 200** (`claude` required, `git`/`node` best-effort), which is the only mechanism that can make the shipped snapshot warm — and its own budget rises to 300 s, well inside the 3600 s build-hook window, because it now does work whose duration is a cold `exec`. Two structural rules keep that honest, because **per-command timeouts do not compose**: the required command runs FIRST with its own budget so no best-effort warm-up can starve the one that decides whether the snapshot is usable, and the best-effort ones then SHARE the remainder of a total warm-up ceiling that sits inside the hook budget with margin (240 s against 300 s). Without them, three commands at 120 s each would be 360 s — a fix for a runtime failure that produces a build failure instead — and a single hung optional command could hold up a 200 that the required warm-up had already earned. Separately, the version probe's timeout goes from 10 s to 60 s: a probe that exists to print a version string into a log line gains nothing from a tight bound and loses the whole task when it trips. The general rule this generalises to, and the reason it belongs in the ADR rather than only in a comment: **on this backend, a first-touch cost that other substrates pay during container start is deferred to the first task instead**, so anything large and lazily-loaded is a turn-0 hazard unless it is touched in `/ready`. +All six hooks share the FastAPI listener on port 8080. AWS hook properties accept `ENABLED`/`DISABLED`, not route paths; the architecture enum is `ARM_64`. -**Payload delivery — v2 amendment (2026-09-13, #817/#700).** The local implementation now shares an authenticated bootstrap protocol between ECS and MicroVM. This replaces the original inline/S3-pointer protocol and broad worker payload-bucket read grant. The historical P1/P2 runs below predate this amendment; AWS authorization, networking, expiry and coordinated deployment validation remain pending. The repository runbook is `docs/verification/645-payload-bootstrap.md`. +| Hook | Contract | +|---|---| +| `/ready` | Required when runtime hooks are enabled; execute required binary warm-up before snapshot capture | +| `/validate` | Check local readiness, routes and configuration contracts without AWS calls | +| `/run` | Authenticate/install launch configuration and start the pipeline asynchronously | +| `/terminate` | Close the local coding barrier, log and acknowledge any request body; do not join the pipeline or write terminal task status | +| `/suspend` | Drain acknowledged progress and commit the current safe checkpoint within the hook budget | +| `/resume` | Renew credentials and reconcile the original gate before releasing coding | -Every task uses S3. The coordinator publishes a non-secret deployment manifest at `bootstrap/.json`, conditionally creates `/payload.json`, and sends a short-lived signed URL for that one task object. The worker reads the manifest with its own AWS role, whose policy explicitly denies object reads outside its deployment's `bootstrap/*` and denies payload-bucket listing. The explicit deny also prevents a foreign public bucket from authenticating a forged manifest. The task document must name the reference's task ID and contain the exact manifest configuration before the MicroVM installs any of it. +Warm-up budgets come from `contracts/constants.json`; the total guest budget must remain below the image hook timeout. The [P2 runbook](/sample-autonomous-cloud-coding-agents/architecture/645-p2-smoke-runbook) records the cold-binary failure and the enum/trust/logging fixes. Historical binary sizes and timings are observations of those images, not sizing guarantees for later builds. -The `runHookPayload` cap remains **4,096 bytes**, measured against the service despite the SDK's documented 16,384. It now applies to the serialized reference rather than choosing an inline branch. Payloads are capped at 8 MiB and manifests at 16 KiB; the bounds and version are shared in `contracts/constants.json`. +**Authenticated v2 payload transport.** Every ECS and MicroVM task uses S3. The coordinator publishes a deployment manifest, conditionally creates the task payload and sends a short-lived signed URL for that one object. The serialized MicroVM reference must fit 4,096 bytes; payloads are bounded at 8 MiB and manifests at 16 KiB. -| Location | v2 shape | +| Location | Shape | |---|---| -| `runHookPayload` string / ECS `AGENT_PAYLOAD_REF` | `{version:2, task_id, bootstrap_s3_uri, payload_url, expires_at}` | +| Hook reference / ECS `AGENT_PAYLOAD_REF` | `{version:2, task_id, bootstrap_s3_uri, payload_url, expires_at}` | | `bootstrap/.json` | `{version:2, backend, platform_config}` | | `/payload.json` | `{version:2, task_id, agent_payload, platform_config}` | -| Private `/launch.json` | `{fingerprint, reference}`; coordinator-readable only, never on the agent-readable task row | - -The coordinator saves the exact reference for replay: re-signing would change a Run request with the same client token. Task objects use conditional creation and read-back recovery after a lost committed write. URL lifetime is at most 900 seconds, bounded by known credential expiry, and initial creation requires at least 300 seconds. Expired saved references fail without re-signing. The coordinator needs bucket-scoped `ListBucket` so an absent launch record returns `NoSuchKey` rather than `AccessDenied`. It deletes both task objects at finalization; one-day lifecycle expiry is the backstop. Manifests are refreshed with identical bytes at preparation and may coexist across configuration revisions until lifecycle cleanup. +| Private `/launch.json` | `{fingerprint, reference}` for coordinator replay | -The image accepts only v2. The old inline/pointer forms and baked-environment fallback are rejected. A coordinated drain and switch of coordinator, images and application IAM is required; there is no new KMS/SSM resource or bootstrap bundle version. Signed URLs are bearer capabilities and must never appear in task rows, ordinary logs or repository subprocess environments. ECS consumes/removes its reference before importing the task pipeline. The URL consumer restricts requests to the exact task object on regional S3 HTTPS, disables redirects/proxies and bounds response bytes. AWS verifies signatures and actual expiration. +The worker authenticates only its deployment’s `bootstrap/*` using ambient credentials. Other payload-bucket object reads and listing are explicitly denied; signed task downloads carry coordinator authorization. The task ID and configuration must exactly match the reference and authenticated manifest before installation. MicroVM `platform_config` contains allowlisted non-secret identifiers, including role/secret ARNs; it does not contain credentials. ECS already receives its deployment configuration through task settings, so its manifest config is empty. -**Platform configuration delivery (P2, strengthened by v2).** A MicroVM restores a snapshot, including its process environment, so current deployment identifiers must arrive at task launch. The authenticated manifest and downloaded document both contain `platform_config`: non-secret table/bucket names, secret ARNs and session-role ARN. Per-task fields such as `memory_id` remain in `agent_payload`. ECS's manifest config is empty because ECS already receives deployment settings through its task definition and coordinator overrides. +The coordinator saves the exact reference for idempotent replay. Signed URLs last at most 900 seconds, bounded by known credential expiry, and initial creation requires at least 300 seconds. An expired saved reference fails without re-signing the same launch request. Finalization deletes task payload/launch objects; one-day asynchronous S3 expiry is the backstop. -MicroVM installs only allowlisted keys and rejects unknown keys before installing any. Required identifiers must be nonblank; malformed/control-character values and inconsistent ARN partitions/accounts are rejected. Legitimate cross-region secrets remain allowed when present in the trusted manifest. Account agreement alone cannot authenticate a sibling ARN; v2's manifest/config comparison supplies the provenance check. No-config envelopes are rejected regardless of the image's existing environment. - -Manifest read and signed payload download precede configuration installation. Secrets, task-scoped credential setup, CloudWatch initialization and the pipeline follow installation; pre-install diagnostics use stdout without unsafe exception chains. Build hooks remain AWS-silent. The settings allowlist and bootstrap bounds are cross-language contracts with drift checks. These source guarantees do not establish complete hostile-worker isolation: roles still retain other platform grants and choose session tags, and a stolen signed URL remains usable until expiry or revocation. - -**No orchestrator→agent HTTP path exists in P1–P3**: payload arrives through the `/run` hook, all agent work is outbound, and therefore **no JWE auth tokens are minted at all** — token minting (and its ≤ 60 min TTL refresh problem) is deferred until a real consumer exists (e.g. operator shell access, [#391](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/391)). The `endpoint` stays in the `SessionHandle` because it is genuinely per-session state that becomes load-bearing the day such a consumer appears. But note the service does not agree by default: omitting `ingressNetworkConnectors` on `RunMicrovm` attaches a **public** `HTTP_INGRESS` connector, so the strategy passes the Lambda-managed `NO_INGRESS` connector explicitly on every launch (see sub-decision 4's security table). +Normative requirements (EARS): -**Constraint accepted:** the configured **baseline is 8 GiB RAM / 4 vCPU** and the service scales vertically to a **32 GiB / 16 vCPU peak** on its own, with 32 GB of disk. So baseline capacity and additional burst usage are billed separately — good for the bursty compile-and-test shape of an agent task — but the SUSTAINED ceiling is still 32 GiB, so repos that motivated the 120 GB ECS sizing stay on `ecs`. What the construct configures (and validates) is the baseline; the peak is not something a deployment asks for. +- Image builds shall not embed credentials, task identity or deployment-specific configuration. Build hooks shall avoid AWS clients and credential caches. +- The worker shall reject legacy envelopes, oversized/malformed data, mismatched provenance and unknown configuration keys before starting a pipeline or installing configuration. +- Before configuration installation, the worker shall make only the bootstrap manifest/payload reads and shall log diagnostics to stdout. +- Signed URLs shall not appear in ordinary logs, agent-readable task rows or repository subprocess environments. Downloads shall use the exact regional S3 HTTPS object without redirects/proxies and with bounded response sizes. +- Producers, images and IAM shall be upgraded together; incompatible workers must be drained before switching transport. -Normative requirements (EARS). **Each requirement's own `(Pn)` tag is authoritative**; there is no blanket phase for the list. The tags are per-requirement because the original list *was* split P1/P2 on the assumption that a hook could be declared in one phase and served in a later one — which the service does not permit (see the phasing table above), so the hook-serving requirements collapsed into P1 while the P2 items below arrived with the P2 hooks and `platform_config`: +The [payload contract](/sample-autonomous-cloud-coding-agents/architecture/645-payload-bootstrap) and [live checks](/sample-autonomous-cloud-coding-agents/architecture/645-p2-payload-live-20260914) record the implementation and validation. This transport does not establish complete hostile-worker isolation: other platform grants remain, and a stolen signed URL is usable until expiry or revocation. -- (P1) The image build shall not embed secrets, tokens, or per-task identity in the snapshot. -- (P1, amended by v2) Every task shall use the authenticated manifest and single-object signed payload reference; the serialized hook reference shall not exceed 4,096 bytes. -- (P1, amended by v2) The worker role shall read only deployment bootstrap manifests with ambient credentials; other object reads and payload-bucket listing shall be explicitly denied. The coordinator shall own task-object writes, signing, replay and deletion. -- (P1) Where no ingress is configured for a deployment, the strategy shall pass the Lambda-managed `NO_INGRESS` network connector on every `RunMicrovm` call (the field shall not be omitted). -- (P1) Where the image enables any MicroVM lifecycle hook, the image shall also enable the `/ready` hook and the agent shall serve it. -- (P1) When the `/run` hook receives the task payload, the agent shall validate it, start the pipeline asynchronously, and return HTTP 200 within the hook budget. -- (P1, clarified) Invalid hook references shall return a structured client error; unreadable manifest/payload bytes shall return a structured server error. Neither path shall install configuration or start a pipeline. -- (P1) The agent shall not execute the clone→verify→PR pipeline on the hook path. -- (P1) The agent shall resolve credentials at `/run` time. -- (P1) Where a deployment configures a MicroVM image before smoke parity is verified, the platform shall warn that the backend has no smoke-parity guarantee. -- (P2) Where the image declares the `/validate` build hook, the agent shall serve it. -- (P2) The `/validate` hook shall make no AWS API calls. -- (P2) When the `/run` hook receives `platform_config`, the agent shall install only allowlisted keys into the environment before pipeline initialization. -- (P2) If `platform_config` carries a key that is not on the allowlist, then the agent shall reject the run with a 400 and shall install none of the block's keys. -- (P2) If a required `platform_config` key is missing, then the agent shall reject the run with a 400. -- (P2) Where a `platform_config` value and an image-baked environment value disagree, the agent shall use the `platform_config` value. -- (P2) Until `platform_config` is installed, the `/run` hook shall make no AWS API call other than the payload fetch, and shall log to stdout only. -- (P2) Where the image declares the `/terminate` hook, the agent shall return 200 within the hook budget for any request body — including a malformed, empty or absent one — and shall not write terminal task status. -- (P2) When the `/ready` hook runs, the agent shall exec the agent CLI binary before returning 200, so that its pages are resident when the snapshot is captured. -- (P2) If a required `/ready` warm-up does not complete successfully, then the agent shall report not-ready (HTTP 503) rather than allow the snapshot to be taken. -- (P2) The `/ready` hook shall make no AWS API call, warm-up included. +No ABCA endpoint consumer exists in P1–P3. The platform grants no `CreateMicrovmAuthToken` permission and mints no JWE tokens. `NO_INGRESS` can still return an endpoint URL; an unauthenticated 403 verifies the authentication boundary, not valid-token reachability. ### 4. Infra and IAM: conditional resources behind bootstrap `ComputeTypes` -Mirroring the ECS pattern: a `compute-lambda-microvm` bootstrap policy (`cdk/src/bootstrap/policies/`) gated on the `ComputeTypes` CFN parameter; a CDK construct provisioning the build role, execution role (admitted to the per-session role via `AgentSessionRole.admitComputeRole`, which was designed for exactly this), the S3 artifact bucket wiring, and image build automation. Egress uses the platform VPC via egress network connectors so the DNS Firewall / security-group / flow-log stack in [COMPUTE.md](/sample-autonomous-cloud-coding-agents/architecture/compute) applies unchanged; ingress is suppressed with the Lambda-managed `NO_INGRESS` connector (no `SHELL_INGRESS` — it is noted as a candidate for [#391](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/391), operator session access, as a separate decision). - -Two networking facts the construct has to encode, both established live: +The backend adds build/runtime VPC connectors, build artifacts, launch payloads, logs, roles and a managed or external image. Its bootstrap policy is conditional on `ComputeTypes` including `lambda-microvm`. A VPC egress connector requires an operator role. Build egress permits ports 80/443 for package installation; runtime egress permits 443 through the platform VPC. -- **A `VPC_EGRESS` connector requires an operator role.** CloudFormation's generated L1 types `operatorRole` as optional and this ADR originally assumed Lambda would manage the ENIs with its own service-linked role. It does not: the connector fails to create with *"NetworkConnectorOperatorRole is required for VPC_EGRESS connector type"*. The construct creates one role — trusting the bare `lambda.amazonaws.com` service principal (see the trust-policy fact below), carrying `AWSLambdaVPCAccessExecutionRole` plus the ENI / tag / private-IP actions that policy omits — and shares it across both connectors, since it manages interfaces rather than traffic. -- **The MicroVM-facing roles cannot carry a confused-deputy source condition.** All three (build, execution, connector operator) trust the bare `lambda.amazonaws.com` service principal with **no** `aws:SourceAccount` / `aws:SourceArn`, and that is a forced choice, not an oversight: the Lambda MicroVMs service presents no source key when it assumes them, so a trust policy carrying one is unassumable. Two symptoms of the one cause, both live 2026-08-06/07 and both blocking: +**Trust and PassRole limitation.** Recorded live checks rejected `aws:SourceAccount`/`aws:SourceArn` conditions on the MicroVM-facing roles and `iam:PassedToService` on the MicroVM PassRole paths. The working roles trust `lambda.amazonaws.com` without those conditions; build/execution roles also allow `sts:TagSession`. IAM simulation with caller-supplied condition values did not prove that the service supplied those values. See the [P2 evidence](/sample-autonomous-cloud-coding-agents/architecture/645-p2-smoke-runbook), [clean deployment](/sample-autonomous-cloud-coding-agents/architecture/645-p2-clean-deployment-20260913) and [effective IAM checks](/sample-autonomous-cloud-coding-agents/architecture/645-effective-iam-20260915). Reintroduce a condition only after verifying service support. - - Both `AWS::Lambda::NetworkConnector` resources `CREATE_FAILED` **deterministically** (on a freshly deleted stack, so not propagation lag — which matters, because *"The service is unable to assume the provided NetworkConnectorOperatorRole. Please verify the trust policy on the role."* is also the classic propagation symptom and a re-run is the obvious wrong guess). Removing the condition → both created within a second. - - `RunMicrovm` failed with a **misleading `iam:PassRole` AccessDenied on the caller**, with the orchestrator's grant present, `simulate-principal-policy` returning `allowed`, no permissions boundary, and a temporary *unconditioned* `iam:PassRole` **also** denied. The real cause was the execution role's trust; removing its conditions made the next submission reach `RUNNING` in 6 s. So the service reports a role it cannot pass-and-assume as an identity-policy denial on the principal passing it. - - Recorded plainly because the fix looks like a regression to anyone applying the standard service-principal pattern — and because it *was* a regression in the other direction: P1's standalone-validated operator-role probe had no conditions and worked, and the P1 F2 fix then added them "to mirror the build/execution roles". `sts:TagSession` stays: the service needs both actions and it was never implicated. - -- **Neither can the `iam:PassRole` grants carry an `iam:PassedToService` condition — same root cause, identity side (P2r2-F9 + P2r2-F10, live 2026-08-07 run 2).** An earlier revision of this ADR recorded the opposite, that the identity-side condition "was exonerated" by run 1's elimination. **That was a false negative**, and its cause is worth recording because it is a general trap: run 1 tested the conditioned grant by *adding* a temporary unconditioned `iam:PassRole` and watching the task still fail — but the temporary grant remained attached through the later submissions that succeeded, so the conditioned grant was never once tested against a working trust policy. A contaminated control. - - Run 2 ran the clean experiment — same exact-ARN resource, same ~5-minute IAM settle, one variable. It removed the run-1 workaround **first** (submission 4: denied) and only then added the unconditioned grant back on the same resource (submission 5: `RUNNING`), which is the ordering run 1 got wrong: - - | Orchestrator `iam:PassRole` on the execution role | Result | - |---|---| - | exact ARN **+ `iam:PassedToService: lambda.amazonaws.com`** | **DENIED** (two independent submissions) | - | exact ARN, **no condition** | **`RUNNING` in 9 s** | - - The denial lands on the **caller**, which is what makes it so misleading — the statement names that exact ARN and `simulate-principal-policy` answers `allowed`: - - ``` - User: arn:aws:sts:::assumed-role/backgroundagent-dev-TaskOrchestratorOrchestratorFn-… - is not authorized to perform: iam:PassRole on resource: - arn:aws:iam:::role/backgroundagent-dev-LambdaMicrovmComputeExecutionRo-… - because no identity-based policy allows the iam:PassRole action - ``` - - And the same key blocks the *other* PassRole path, which run 1 never reached because the enum defect (P2-F2) stopped it earlier: CloudFormation could not pass the **build role** at `CreateMicrovmImage` under the bootstrap `infrastructure` policy's allowlisted `IAMPassRole`. Verbatim, so the diagnosis does not have to be taken on trust: - - ``` - LambdaMicrovmComputeImage… CREATE_FAILED - User: arn:aws:sts:::assumed-role/cdk-hnb659fds-cfn-exec-role--us-east-1/AWSCloudFormation - is not authorized to perform: iam:PassRole on resource: - arn:aws:iam:::role/backgroundagent-dev-LambdaMicrovmComputeBuildRoleF0-… - because no identity-based policy allows the iam:PassRole action - (Service: LambdaMicrovms, Status Code: 403) - ``` - - Three pieces of evidence pin that to the *condition* rather than to a stale bootstrap or a wrong resource pattern: - - 1. the live `IaCRole-ABCA-Infrastructure` policy was byte-identical to this branch's `cdk/bootstrap/policies/infrastructure.json`, so `cdk bootstrap --force` would have changed nothing; - 2. `aws iam simulate-principal-policy --policy-source-arn --action-names iam:PassRole --resource-arns ` returned `allowed` **with** `--context-entries ContextKeyName=iam:PassedToService,ContextKeyValues=lambda.amazonaws.com,ContextKeyType=string` and `implicitDeny` with no context entry — so the resource pattern matches and the condition key is the only remaining variable; - 3. the **control**: the out-of-band `create-microvm-image` call passed the *same build role* to the *same service* successfully, using operator credentials that carry no such condition. The role's trust is therefore fine and the denial is genuinely caller-side. - - So: **the Lambda MicroVMs service presents no usable value for `iam:PassedToService` on either PassRole path** (CloudFormation → build role at `CreateMicrovmImage`; orchestrator → execution role at `RunMicrovm`), exactly as it presents no `aws:SourceAccount` on the assume-role path. One root cause, two more symptoms. Both statements therefore drop the condition, and the fix is deliberately asymmetric so it stays contained: - - - `task-orchestrator.ts` sid `MicrovmPassExecutionRole` — condition removed; the **exact execution-role ARN** is now the whole of the scoping, which is why that resource must never be relaxed to a prefix or `*`. - - a new sid `MicrovmPassRoles` in the **conditional** `compute-lambda-microvm` bootstrap policy — unconditioned `iam:PassRole` on the build- and connector-operator role **name prefixes only** (not the execution role, which CloudFormation never passes). The shared `infrastructure` `IAMPassRole` keeps its allowlist, so no other role in the stack loses that constraint, and an agentcore-only bootstrap never gains an unconditioned pass at all. **Operators must re-bootstrap** (bundle ≥ 1.6.0) for the CDK-managed image path to work. - - If AWS documents the value the service does present, adding it to both statements restores the condition. `microvms.lambda.amazonaws.com`, `lambda-microvms.amazonaws.com` and `microvms.amazonaws.com` were all `implicitDeny` against the conditioned policy, so any one of them would serve as the allowlist entry if it turns out to be right. Note that CloudTrail carries **no `lambda-microvms` management events at all** today, so the value cannot be read out of a log — only confirmed by AWS or found by a bounded sweep. - - Compensating controls, enumerated per role. The deployment-role's shared `IAMPassRole` grant is **name-prefix-scoped**, so it technically covers all three roles; in practice only the orchestrator actively invokes `iam:PassRole` on the execution role (CloudFormation never requests it). The other two roles are passed **to** the deployment role (not to themselves). - - | Role | Who can pass it, and how that grant is scoped | - |---|---| - | **Execution role** | The **orchestrator Lambda only**, at `RunMicrovm` — `iam:PassRole` scoped to this role's **exact ARN**, no condition (`constructs/task-orchestrator.ts`, sid `MicrovmPassExecutionRole`). The referenced comment contains the authoritative two-arm experiment evidence that the condition is the true blocker (not a permissions gap or stale bootstrap). | - | **Build role** | The **CloudFormation deployment role**, at `CreateMicrovmImage` (the L1's `buildRoleArn`) — via the new `MicrovmPassRoles` statement, scoped to `role/backgroundagent-dev-LambdaMicrovmComputeBuild*`, no condition. Also whoever runs `package-microvm-artifact.sh --create-image` out of band, using their own credentials. | - | **Connector operator role** | The **CloudFormation deployment role**, at `AWS::Lambda::NetworkConnector` create/update (`operatorRole`) — the same statement, scoped to `role/backgroundagent-dev-LambdaMicrovmComputeConnector*`. | - - The rest of the posture: every resource these roles can reach is account-scoped by ARN **except two deliberate `Resource: '*'` statements** — `ec2:DescribeAvailabilityZones` on the execution role (EC2 describe actions have no resource-level scoping; read-only, no mutation, no data access, needed so a CDK target repo's `cdk synth` build gate can resolve AZ context on a fresh clone) and the connector operator role's ENI/tag/private-IP statement (`CreateNetworkInterface` is authorized before the ENI exists and the `Describe*` calls take no resource, which is why the AWS-managed VPC-access policy uses `*` too). Both are justified in the construct's cdk-nag `AwsSolutions-IAM5` suppressions, which is where a reviewer should check them rather than here. The Logs grants are prefix-scoped (`/aws/lambda-microvms/*` plus one named log group), i.e. wildcards inside a namespace, not `*`. Separately, the **orchestrator's** `lambda:PassNetworkConnector` is also `Resource: '*'` and unavoidably so — the AWS-managed connectors live in the `aws` account, outside any ARN we could enumerate (justified in `task-orchestrator.ts`, sid `MicrovmPassNetworkConnector`). Finally: none of the three roles holds `iam:*`, none has cross-account trust, and the only `sts:AssumeRole` any of them has is the execution role's, scoped to the per-task SessionRole. - - If AWS later populates a source key on this path, adding it to the shared principal fixes all three roles and both `sts` actions at once. - -- **Build-time egress needs port 80; runtime does not.** `agent/Dockerfile` installs Debian packages and `apt-get` fetches over plain HTTP, so a 443-only egress path fails every snapshot build (`Could not connect to deb.debian.org:80 … exit code: 100`). Rather than widen the runtime posture, the construct provisions a **second, build-only** connector on the same private subnets with a 443 + 80 security group, referenced solely by the image resource and the packaging script. The agent at run time still has 443-only egress. - -- Where the bootstrap `ComputeTypes` parameter includes `lambda-microvm`, the generated template shall attach the `IaCRole-ABCA-Compute-LambdaMicrovms` policy to the CloudFormation execution role. -- The orchestrator role shall receive only the MicroVM lifecycle actions it calls (`lambda:RunMicrovm`, `lambda:SuspendMicrovm`, `lambda:ResumeMicrovm`, `lambda:TerminateMicrovm`, `lambda:GetMicrovm` for `pollSession`, and `lambda:PassNetworkConnector`, which is required even for the default connectors), scoped to platform-created images. -- Where the `lambda-microvm` backend is enabled, the approve and deny Lambdas shall receive `lambda:ResumeMicrovm` and `lambda:GetMicrovm` — conditionally, mirroring the cancel handler's conditional `RUNTIME_ARN` wiring in `task-api.ts`. -- The trust policy of every MicroVM-facing role shall name `lambda.amazonaws.com` and shall carry no source-condition key (the service presents none; see the trust-policy fact above). -- The `iam:PassRole` grant the orchestrator uses for the MicroVM execution role shall carry no `iam:PassedToService` condition and shall be scoped to that role's exact ARN. -- Where the bootstrap `ComputeTypes` parameter includes `lambda-microvm`, the `IaCRole-ABCA-Compute-LambdaMicrovms` policy shall grant `iam:PassRole` without an `iam:PassedToService` condition, scoped to the MicroVM build- and connector-operator role name prefixes, and shall not extend that grant to the MicroVM execution role. -- The shared `IaCRole-ABCA-Infrastructure` `iam:PassRole` statement shall retain its `iam:PassedToService` allowlist. -- The MicroVM execution role shall hold `logs:CreateLogStream` and `logs:PutLogEvents` on the application log group whose name is delivered in `platform_config`, scoped to that log group. - -`lambda:CreateMicrovmAuthToken` is granted to no role in P1–P3 (no JWE consumer exists; see sub-decision 3). +| Role/action | Scope and responsibility | +|---|---| +| Coordinator lifecycle APIs | Configured image ARN and its version-qualified sibling; includes actual-version capability lookup | +| Coordinator `iam:PassRole` | Exact execution-role ARN, without `iam:PassedToService` | +| Deployment `iam:PassRole` | Backend-specific build/operator role name patterns; shared infrastructure allowlist remains intact | +| Approval/denial handlers | Observe/resume the configured image after committing a decision; dispatch parked continuations | +| Build role | Selected immutable artifact plus manual-build key; MicroVM log writes | +| Execution role | Bootstrap manifests, startup secrets, allowlisted models, Memory and logs; tenant data through the per-task SessionRole | +| Connector operator | Tested ENI/tag/private-IP permissions plus AWSLambdaVPCAccessExecutionRole | -**Cost attribution.** `cdk/src/main.ts` currently tags the whole stack with a single `compute_type` context value (default `agentcore`) — already imprecise with two backends, wrong with three. P1 must add backend-identifying cost-allocation tags on the MicroVM-specific resources (images, payload/artifact bucket wiring, log groups) and revisit the stack-level tag semantics (e.g. a `compute_types` list), keeping attribution consistent with [#645](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/645)'s cost/attribution acceptance criterion. +`lambda:PassNetworkConnector` has no resource-level authorization support and therefore uses `Resource: *`. The operator role also has wildcard ENI permissions; some mutations can be scoped in IAM, so their current tested wildcard is not evidence that narrower permissions are impossible. DescribeAvailabilityZones needs a wildcard for fresh CDK repository lookups. These exceptions and namespace wildcards are documented in the construct’s cdk-nag suppressions. -- Where a deployment enables the `lambda-microvm` backend, MicroVM-specific resources shall carry backend-identifying cost-allocation tags. +**Nested infrastructure.** `microvm_nested_stack=true` puts MicroVM resources in a nested stack. The shared execution role stays in the parent to avoid a role-trust dependency cycle. Bootstrap 1.9.0 covers nested deployment roles. Existing flat installations need the reviewed overlap/drain migration, explicit image/coordinator pins and rollback checks in the [nested migration record](/sample-autonomous-cloud-coding-agents/architecture/645-p3-nested-stack). Moving construct paths alone is not a safe migration. Preserve `microvm_resource_name_prefix` after a migrated deployment. #### Security bar vs existing backends ([#645](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/645) acceptance criterion) -| Control | AgentCore | ECS | Lambda MicroVMs | Delta | -|---|---|---|---|---| -| Egress, runtime (DNS Firewall, TCP 443 SG, flow logs) | Platform VPC | Platform VPC | Platform VPC via egress network connector | None | -| Egress, image build | ECR build outside the platform VPC | ECR build outside the platform VPC | Platform VPC via a **separate build-only connector, TCP 443 + 80** (`apt-get` is plain HTTP) | New surface: build-time egress is wider than runtime egress by one port, on a connector no running MicroVM can use | -| Tenant-data scoping | Per-session role (`admitComputeRole`) | Per-session role | Per-session role, execution role admitted identically | None | -| Secrets delivery | Runtime env + Identity injection | Task env vars | Fetched at `/run`; never in snapshot | New surface: snapshot must stay secret-free (EARS req., sub-decision 3) | -| Non-secret platform config (table/bucket names, secret + role ARNs) | Runtime env vars | Task env vars | `platform_config` in the `/run` payload, installed into the process env | New surface: the values are attacker-relevant *as env vars* (`LD_PRELOAD`, `AWS_ENDPOINT_URL`), so the agent installs a fixed **allowlist** and rejects the whole run on any other key (EARS req., sub-decision 3) | -| Inbound exposure | None (SigV4 invoke only) | None (no endpoint) | **None — but only because the strategy passes `NO_INGRESS` explicitly.** The service default is a PUBLIC `HTTP_INGRESS` connector plus a public `*.lambda-microvm..on.aws` endpoint; no tokens are minted in P1–P3 either way | New surface **and** a new failure mode: "no inbound" is an active control, not an absence. Drop the `NO_INGRESS` argument and every agent MicroVM gets a public endpoint (EARS req., sub-decision 3) | -| IAM condition keys on the compute-role trust **and** on the `iam:PassRole` grants that hand it over | Trust pinned with `aws:SourceAccount`; `PassRole` under the allowlisted bootstrap statement | Trust pinned per-service; `PassRole` under the allowlisted bootstrap statement | **Neither is possible.** All three MicroVM-facing roles trust the bare `lambda.amazonaws.com` with no `aws:SourceAccount`/`aws:SourceArn`, **and** both `iam:PassRole` grants (orchestrator → execution role at `RunMicrovm`; CloudFormation → build role at `CreateMicrovmImage`) carry no `iam:PassedToService` — the service presents no usable value for any of those keys, and each condition is a hard blocker while present (live-verified, blocking, four times across two runs) | **Real, evidenced gap that does not close from our side, and it is wider than the trust policy alone.** `lambda.amazonaws.com` is shared with every other Lambda feature, so neither the account pin nor the passed-to-service pin is available on this path. Compensated per role (table in sub-decision 4): the **execution** role is passable by the **orchestrator only** (at `RunMicrovm`), restricted to its **exact ARN**; the **build** and **connector-operator** roles are passable by the CloudFormation deployment role under a new **conditional, per-backend, name-prefix-scoped** statement (`MicrovmPassRoles`, bootstrap ≥ 1.6.0) that deliberately excludes the execution role. The shared allowlisted `IAMPassRole` (`role/backgroundagent-dev-*`) is left intact to avoid widening the grant for ~30 other roles, so while it technically matches the execution role, only the orchestrator actively reaches for it. Resources are account-scoped by ARN apart from two justified `Resource: \'*\'` statements (`ec2:DescribeAvailabilityZones`; the operator role\'s ENI management — both carry cdk-nag IAM5 suppressions). No `iam:*`, no cross-account trust. Revisit if AWS ever documents the values the service presents; CloudTrail records no `lambda-microvms` events, so they cannot be read from logs | -| Per-task observability writes | Runtime writes to the vended APPLICATION_LOGS group | Task role writes to the task log group | Execution role writes to the SAME APPLICATION_LOGS group, granted against the group `platform_config` names (P2-F4) | None — but only after P2-F4: the name was delivered a phase before the grant, so the agent attempted the write and every per-task line (and `METRICS_REPORT`) was `AccessDenied`, degrading silently to guest stdout | -| Session isolation | MicroVM | Task-level | MicroVM (Firecracker) | None (≥ ECS) | -| State reuse | None | None | Snapshot shared across MicroVMs | New surface: application PRNG reseed + credential refresh on `/run`/`/resume` — **P3 scope** (P2 exposure measured as negligible: sole `random` consumer is a ULID sort key under a `task_id` partition; `os.urandom`/`secrets` unaffected; no credential derives from `random`). Credential refresh IS in P2: per-task credentials arrive via `platform_config` at `/run` | -| Workload-token injection | Yes (Runtime-coupled) | No (env-var posture) | No (env-var posture) | Shared with ECS; deferred to [#249](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/249)/ADR-016 | -| Operator shell access | No | No | Not enabled (`SHELL_INGRESS` omitted; candidate for [#391](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/391)) | None by default | -| Auth-token minting | n/a | n/a | `CreateMicrovmAuthToken` granted to no role in any phase | Verified static-only: any principal holding the action *can* mint a working JWE, including against a `SUSPENDED` MicroVM, so the posture rests entirely on the grant being absent | +| Control | MicroVM posture | Difference to account for | +|---|---|---| +| Runtime egress | Platform VPC, DNS Firewall, HTTPS security group, flow logs | Separate build connector additionally allows HTTP | +| Tenant access | Task-scoped SessionRole | Shared compute-role permissions remain outside that boundary | +| Configuration/secrets | Authenticated runtime identifiers; credentials resolved after launch | Shared snapshot must not capture credentials or task identity | +| Ingress | Explicit NO_INGRESS; no platform token minting | Service defaults would select HTTP_INGRESS if the field were omitted | +| Trust/PassRole | Exact role/image scopes where supported | Source-condition limitations described above | +| Logs | Service image group plus platform APPLICATION_LOGS | Both namespaces need explicit grants | +| Retained work | Versioned, bounded, checksum-verified checkpoint | Protect stored conversation/files and clean terminal state | +| Workload identity | Runtime credentials and task-role refresh | Linear vault identifiers arrive through `platform_config`; the compute execution role mints tokens ([setup](/sample-autonomous-cloud-coding-agents/using/linear-setup-guide#using-the-vault-with-lambda-microvms)) | + +MicroVM-specific resources carry `abca:compute-backend=lambda-microvm` cost tags. A stack-wide compute tag cannot accurately attribute a mixed-backend deployment by itself. #### Regional availability enforcement -Lambda MicroVMs launched in 5 regions (us-east-1/2, us-west-2, eu-west-1, ap-northeast-1) and will expand. ABCA is a single-region deployment, so the constraint is binary per stack: either the stack region supports the backend or the backend does not exist there. Enforcement is layered — one static check where offline determinism is required, live probes everywhere else so the platform self-heals as AWS adds regions (`list-managed-microvm-images` is the documented read-only availability probe): - -| Stage | Mechanism | Check | -|---|---|---| -| CDK synth/deploy | Static region constant (single exported list, documented update path) | Synth fails when `ComputeTypes` includes `lambda-microvm` in an unlisted region; context-flag escape hatch for newly launched regions ahead of the constant update | -| Repo onboarding | Live probe from the CLI | `bgagent repo onboard --compute-type lambda-microvm` calls `list-managed-microvm-images` in the stack region and rejects with a remedy (supported-region list + suggest `agentcore`/`ecs`) | -| `bgagent platform doctor` | Live probe (precedent: `checkBedrockModel`) | Reports backend availability for the stack region whenever any active blueprint selects `lambda-microvm` | -| Orchestration (defense in depth) | Error classification | `startSession` failures from a missing regional endpoint classify to a typed remedy in `error-classifier.ts`, never a cryptic SDK error on the task | +The launch-region list is us-east-1, us-east-2, us-west-2, eu-west-1 and ap-northeast-1. New Regions can be supported before the static list is updated: -- If an operator onboards a repo with `compute_type: 'lambda-microvm'` and the availability probe fails for the stack region, then the CLI shall reject the onboarding with the supported-region list and alternative backends as the remedy. -- If `startSession` fails because the MicroVM service is unavailable in the stack region, then the orchestrator shall classify the failure with a configuration remedy and shall not retry. -- When the platform doctor runs in a deployment where any active blueprint selects `lambda-microvm`, the doctor shall probe MicroVM availability in the stack region and report the result. +- Synth rejects a concrete unlisted Region unless `microvm_region_override` is set. An unresolved Region defers to live checks. +- CLI onboarding and platform doctor probe `list-managed-microvm-images`. +- Runtime regional failures receive a configuration remedy instead of an opaque SDK error. ### 5. Rollout: phased, default unchanged -- **P1 — strategy + infra + minimal hook serving:** `LambdaMicrovmComputeStrategy` (start/poll/stop), CDK construct, bootstrap policy, types sync, unit + CDK assertion tests, and the agent's `/ready` + `/run` endpoints. No suspend yet. The image IS creatable and launchable and the payload DOES reach the agent — but there is **no smoke-parity guarantee** (sub-decision 3's phasing table). -- **P2 — smoke parity:** the agent serves `/terminate` + `/validate` and installs its platform env from the `/run` payload (see sub-decision 3's "Platform configuration delivery"); agent completes clone → change → PR on the backend with progress visible to `bgagent watch`; failure classification entries in `error-classifier.ts`; **AgentCore Memory parity** (IAM grant + `MEMORY_ID` delivery, following the `EcsAgentCluster` pattern — Memory is a standalone service already consumed cross-substrate, and omitting the grant silently no-ops cross-session learning); the agent's remaining non-secret env parity inside the snapshot. -- **P3 — suspend/resume:** the interface widening from sub-decision 1 (mandatory methods, all three strategies in one commit), HITL-wait suspend policy, inline resume in the approve/deny Lambda with orchestrator-poll reconciliation (sub-decision 2), timeout-under-freeze wall-clock handling; coordinate with [#491](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/491)'s unified liveness model and update Cedar decision #7's rationale note. -- **Out of scope:** replacing AgentCore as default; classic Lambda functions as a runtime; GPU; the Runtime-coupled workload-access-token injection path (delivery mechanism exists only on AgentCore Runtime; MicroVMs adopt the ECS env-var posture until [#249](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/249)/ADR-016 redesign the seam). Gateway integration is orthogonal: ADR-019/[#641](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/641) is substrate-portable by design and applies to this backend when it lands. - -## Consequences - -- (+) **Suspend/resume economics.** Tasks idling on approval waits stop billing compute while preserving full state — bounded at ~1 h per gate under the current Cedar ceiling (decision #6), and the enabler for cheap off-hours gate-ceiling extensions later (§14.8). Verified end to end at the substrate level: suspend and resume each complete in ~1 s, and `microvmId` **and** `endpoint` survive the cycle byte-identical, so a stored `SessionHandle` remains valid. -- (+) **VM-level isolation without cluster ops.** Firecracker isolation with no ECS cluster, task definition, or capacity management; one-session-per-MicroVM maps 1:1 onto ABCA's task model. -- (+) **Escapes AgentCore's 2 GB image limit and FUSE `flock()` workaround** — native disk in the snapshot supports `uv`/`mise` without the split-storage scheme. Note the size comparison must say WHICH measure it means: the same agent tree is 1.799 GB as an OCI image (629.7 MB compressed, i.e. under AgentCore's limit) but reports `codeInstallSizeInBytes` of 2.17 GiB as a MicroVM snapshot (i.e. over it). The two straddle the limit and are not interchangeable; memory/disk snapshot sizes are a third thing again and must not be summed into the comparison. -- (+) **Liveness becomes explicit.** Unlike AgentCore's stub `pollSession`, the strategy can report real substrate state, strengthening the [#491](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/491) unification. -- (−) **32 GiB sustained ceiling (8 GiB baseline + automatic 4× vertical scaling), 32 GB disk.** Not a successor to the ECS backend for heavy CI-parity builds; the platform now maintains three backends. -- (−) **Capacity is baseline-priced with burst headroom, which is a narrower promise than "32 GiB".** The deployment configures an 8 GiB / 4 vCPU baseline and the service scales to 32 GiB / 16 vCPU on demand — well matched to an agent task, which is idle-ish while waiting on the model and spiky during builds, and cheaper than reserving the peak. But it is *burst*, not a reservation: a workload that needs 32 GiB **sustained** is relying on scaling behaviour this ADR has not measured, and against ECS's 120 GB the gap for sustained-memory workloads is unchanged. So the value proposition remains the suspend economics, the observable control-plane state machine, and the absence of cluster ops — with capacity now a fair-to-good fit rather than a hard blocker. Repos with genuinely sustained heavy builds still belong on `ecs`. -- (−) **New packaging pipeline.** Zip + Dockerfile + service-side image builds with versioned snapshots (storage billed per version) alongside the existing ECR flow; image versions need lifecycle cleanup — including versions left behind by FAILED builds, and noting the last version of an image cannot be deleted individually (delete the image, which reaps it). -- (−) **The payload bucket is on the hot path, not the overflow path.** With a 4 KB `runHookPayload` cap, virtually every real task delivers its payload via S3, so the bucket, its TTL rule and the execution role's read grant are load-bearing for normal operation rather than an edge case (sub-decision 3). -- (−) **8-hour hard cap includes suspended time**, and with `idlePolicy` omitted there is no tighter substrate-level suspended-TTL — the suspended-state bound is `maximumDurationInSeconds` plus orchestrator termination and the stranded reconciler. A manually suspended VM was observed alive at 1 h with no TTL in sight (observation truncated there), so nothing contradicts this bound, but nothing narrows it either. Under today's 1 h gate ceiling it is comfortably sufficient; any future extension of gate ceilings must revisit the bound (an additive `idlePolicy` change) and give the orchestrator a checkpoint-and-restart path (push branch, new session) beyond the cap. -- (!) **Idle-policy foot-gun.** Traffic-based auto-suspend would freeze a busy outbound-only agent; the decision to disable auto-suspend must be enforced in code and covered by tests, not left to configuration discipline. -- (!) **Service defaults are not the desired posture.** Two live-caught cases (public `HTTP_INGRESS` by default; `/ready` mandatory) mean an omitted field on this backend does not mean "off" — it can mean "the service picks, and it picks wider than we want". Every new `RunMicrovm` / `CreateMicrovmImage` field should be assumed to have an opinionated default until checked. -- (!) **Nothing self-terminates on the paths that matter** — superseding P1's unqualified version of this bullet. With `run: ENABLED` the service DOES reap a VM whose run hook returns 4xx (~12 s, `stateReason: "Run lifecycle hook returned HTTP status 400."`, live-verified), so a guest that rejects its own payload cleans itself up. That is the only self-cleaning case: the service reaps a hook *result*, and once `/run` has answered 200 it has no view of the guest. A VM whose task finished, crashed after `/run`, or hung stays `RUNNING` and billing until the 8 h cap, so the orchestrator's `TerminateMicrovm` on finalize remains the only cleanup for normal operation and a leaked handle is still a cost incident. -- (!) **Snapshot uniqueness.** Shared memory snapshots mean every MicroVM restored from one image starts with identical PRNG state. **Re-scoped to P3 in the P2 review, with the exposure measured rather than assumed** (see the amendment under sub-decision 2): the sole `random` consumer in `agent/src` is a ULID sort key under a `task_id` partition, `os.urandom`/`secrets` are unaffected, and no credential or token derives from `random` — so the P2 exposure is a possible duplicate progress event, not a security boundary. It becomes load-bearing at P3, where a *resumed* VM continues from frozen state repeatedly; the reseed must land on `/run` and `/resume` together with those hooks. Asserting the requirement while leaving it unimplemented was the real defect, and this amendment is the fix. -- (−) **Root-stack resource headroom needs ongoing measurement.** The historical 985,886-byte / 486-resource measurement predates [#854](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/854), which reduced template size and resource count. The 2026-09-13 [offline review](/sample-autonomous-cloud-coding-agents/architecture/645-p3-readiness-review) measures 472 root resources with a managed MicroVM image, and 489 with gateway and vault enabled in a scratch probe. Nesting can reduce that fullest root count to 474 in the prototype. These are unbundled synth measurements, not deployment validation; the vault combination remains guarded on main ([#857](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/857)). -- (!) **Snapshot WARMTH is a first-class property, not an optimisation.** A snapshot inherits only the pages something touched before it was captured, so a large lazily-loaded artifact — the 225 MiB `claude` binary, and anything similar added later — pays its first-touch cost on the *first task* instead of at container start. That cost failed every task at turn 0 in the P2 smoke run (P2-F5). Anything heavyweight added to the image must be exec'd in `/ready`, and any timeout guarding a first touch must be sized for a cold page fault rather than for the work itself. -- (!) **Regional availability (5 regions at launch, expanding)** — enforced in layers (synth-time static check, onboarding + doctor live probes, orchestration-time classification; see sub-decision 4). The static CDK constant is the one piece that rots as AWS expands; its update path and context-flag escape hatch are deliberate. -- (!) **Workload-token injection delta persists** (shared with the ECS backend) until [#249](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/249)/ADR-016 land; document it in the security bar comparison rather than blocking on it. Memory and Gateway are explicitly *not* deltas — both are standalone services consumed via IAM from any substrate. +| Phase | Delivered behavior | +|---|---| +| P1 | Strategy, infrastructure, bootstrap/types, minimal `/ready` + `/run` serving; merged in [#689](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/689) | +| P2 | Clone → change → PR, progress/logs/Memory, runtime configuration, `/validate` + `/terminate`; merged in [#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733), with takeover follow-up validation | +| P3 | Approval-aware sleep/wake, credential renewal, original deadlines, retained requests, complete checkpoint/replacement recovery, nested deployment and live acceptance | -## Testing +Activate sleep only after verifying the deployed image and coordinator together. Keep a compatible published coordinator and explicit image pin for rollback. Normal acceptance includes the 600-second default, explicit expiry, new and existing task off-switch behavior, replacement, cleanup and preservation of unrelated infrastructure. -**P1 (start/poll/stop — no suspend):** +Changing the default backend, GPU support, native Slack approval buttons, approval-by-Linear-reply and operator shell access are outside this ADR. CLI responses remain the supported approval path. Future work needs its own scope; “P4” is not an approved phase here. -- Unit tests for the strategy: start/poll/stop mapping (including `SessionStatus` `'suspended'` reported mechanically, without task-state interpretation), v2 reference-size bounds (4,096 bytes), immutable signed-reference replay and scoped payload transport, the image-identifier-must-be-an-ARN guard, the explicit `NO_INGRESS` argument (including the blank-env-var fallback, which must never omit the field), error classification (`ServiceQuotaExceededException`, `ThrottlingException`, `ResourceNotFoundException`, regional-unavailability), the omit-`idlePolicy` invariant, and `maximumDurationInSeconds` fixed at 28 800. -- Agent tests: `/ready` returns 200 after required local warm-up succeeds, without starting a task or calling AWS; `/run` accepts authenticated v2 references and rejects old unsigned envelopes, starts the pipeline asynchronously through the same mapper `/invocations` uses, returns before the pipeline finishes, and rejects every unusable envelope with a named code before spawning; `/suspend` and `/resume` are NOT served (`/validate` and `/terminate` joined the served set in P2). -- Orchestrator tests: substrate-terminal + non-terminal task status → failed classification; `suspended` + non-`AWAITING_APPROVAL` status → anomaly event, no fail-fast; `compute_metadata` persisted with `microvmId`/`endpoint` after `startSession`. -- CDK assertions: MicroVM resources present only when `ComputeTypes` includes the backend; synth failure for unsupported regions (plus the context-flag escape hatch); memory size validated against the accepted list at synth; the connector operator role and its trust; two connectors with the build-only one carrying port 80 and the runtime one not; the current six-hook declaration, shared protocol marker and per-worker capability admission; IAM actions scoped as specified (orchestrator lifecycle set; no `CreateMicrovmAuthToken` anywhere); payload-bucket grants (execution role read-only); backend cost-allocation tags; types-sync check covers the widened `ComputeType`. -- CLI tests: onboarding rejection with remedy when the availability probe fails; doctor check present when a blueprint selects the backend. -- P1 verification items (external service facts) — **executed 2026-07-31, us-east-1**; see `docs/verification/645-p1-lambda-microvm-runbook.md` for the full evidence. Discharged: `runHookPayload` limit (**4 096**, not 16 KB), the accepted baseline memory sizes (`[512…8192]` MiB — note the developer guide, not the probe, is what establishes that this is a BASELINE with a 32 GiB peak), image-identifier ARN requirement, IAM action names and the observed image-ARN shape, region probe behaviour, manual suspend/resume without `idlePolicy`, terminate timing and the `TERMINATED`-persists-≥10-min finding, the default public `HTTP_INGRESS`, and the `/ready` requirement. **Not** discharged: account-quota treatment of `SUSPENDED` MicroVMs (not observable safely), suspended TTL beyond 1 h (truncated), the vertical-scaling behaviour itself (no workload here approached the baseline, so the 4× peak is documented rather than observed), and the `AWS::Lambda::MicrovmImage` CloudFormation value shapes (never exercised — the run used the out-of-band script path; **discharged, and REFUTED, by the P2 run — see P2-F2 in sub-decision 3**). Record the closed answers in COMPUTE.md. +## Consequences -**P2 (smoke parity):** +- Approval waits can stop consuming compute without discarding the question or the saved work. Snapshot and checkpoint costs still apply. +- Native disk supports build-tool locking, and the service exposes explicit worker state without a cluster to operate. +- ABCA now maintains three backends and an additional artifact/snapshot lifecycle. Failed or unused image versions also need cleanup. +- Eight hours remains a per-worker limit. Retained approvals rely on verified retirement and replacement rather than extending a worker indefinitely. +- Suspended workers retain ABCA capacity until confirmed retirement. The service’s account-quota treatment of suspended memory was not established by the recorded probes. +- Memory baseline validation and published peak capacity are not workload benchmarks. Sustained heavy builds require measured sizing and may fit ECS better. +- A healthy heartbeat does not prove coding progress. Hook failures may cause service termination, but successful hook acceptance does not remove ABCA’s cleanup responsibility. +- Service error wording can hide transport errors. The pooled-hook mitigation has local and live evidence; historical internal dispatch traces remain unavailable. Open questions are tracked in the [service feedback record](/sample-autonomous-cloud-coding-agents/architecture/645-lambda-microvm-service-feedback). -- Agent tests: `/validate` returns 200 with its individual check results, 503 while initialising, reports a missing hook route / unsupported interpreter, starts nothing, and makes **zero AWS calls even with `LOG_GROUP_NAME` set** (asserted by poisoning the boto3 and CloudWatch-writer seams — the same assertion covers `/ready`); `/terminate` returns 200 with no body at all, with a malformed / non-object / wrong-content-type / whitespace-only body, when the body read itself fails, with a pipeline still running (without joining it), and when its own best-effort step raises — and never calls `task_state.write_terminal`; a structural assertion that the route carries no typed body param keeps the 422 from being reintroduced. -- `platform_config` tests: the allowlist and required subset are read from `contracts/constants.json` (the wire key set is additionally asserted literally, as the agent-side tripwire on a contract edit); an unknown key rejects the whole block with nothing installed; a non-object block and a non-string value are rejected; a blank/`null` optional value is skipped without clobbering an image value while a blank required value is rejected; a payload value beats a pre-existing env value; installation is observed to happen before the GitHub-token resolver runs and before any pipeline thread exists; the downloaded config must equal the IAM-authenticated manifest, including same-account workspace identifiers; missing config and legacy envelopes are rejected regardless of baked environment values. Transport tests cover signed URLs, exact task identity, bad bytes and rejection before installation. -- Snapshot credential hygiene: a subprocess probe asserts that importing `server` and serving `/ready` + `/validate` imports neither `boto3` nor `botocore`, caches no `aws_session` session, and spawns no CloudWatch writer thread — the property that keeps a build-role credential chain and the build-time region out of the snapshot. -- `/run` pre-install silence: with a **baked `LOG_GROUP_NAME`** (the hostile case — without it the assertions pass vacuously) every AWS/credential seam (`boto3.client`/`Session`, the `aws_session` factories, `_debug_cw`/`_warn_cw`) is armed to raise until the install succeeds. Asserted on the accepted path, on all three rejection paths (bad envelope, `platform_config` invalid, `platform_config` incomplete) and on the failed-fetch 500 — where the seams stay armed for the whole request, because a rejected run installed nothing and so earns no AWS call. The permitted exception is asserted POSITIVELY: exactly one client is built pre-install, for `s3`, through the attributed factory. -- `/ready` warm-up tests (P2-F5): the hook exec's each configured binary exactly once with a generous timeout; `claude` is the only REQUIRED entry; a timeout, a missing binary, a non-zero exit and an unexpected `OSError` each produce **503 with the reason logged to stdout** rather than a 200 or a 500; a best-effort failure still reports ready; the warm-up makes zero AWS calls with `LOG_GROUP_NAME` baked. Plus the backstop half: the `claude --version` probe's bound is asserted to be ≥ 60 s and to be applied to the *exec* rather than to the PATH lookup, and a missing CLI warns instead of raising. -- CDK assertions (P2-F1/F2/F4): no source-condition key on any of the three MicroVM-facing role trusts, and no `aws:SourceAccount`/`aws:SourceArn` string anywhere in them; hook properties are `ENABLED` and the architecture is `ARM_64`, with a negative assertion that **no** hook route string appears anywhere in the rendered image resource; the agent hook routes are asserted against their own dedicated constant (the template no longer carries a path to compare); the execution role holds `logs:CreateLogStream`/`PutLogEvents` on the application log group and the two logs grants stay separate; the stack wires the SAME log group it delivers as `platform_config.log_group_name`. -- Smoke (gated like the ECS backend): clone → change → PR with `bgagent watch` progress; Memory write parity (no AccessDenied no-op). **Run 1 (2026-08-06) FAILED at `implement`, turn 0 — no PR. Run 2 (2026-08-07) PASSED: two tasks clone → change → commit → push → PR, `COMPLETED`, 12 turns / $0.279 / 153 s** (`docs/verification/645-p2-smoke-runbook.md`), which also discharged P2-F1, P2-F2, P2-F4, P2-F5 and the dual-signal-liveness item (45 s heartbeat cadence observed across a 181 s `RUNNING` window). Run 2 needed one live IAM workaround, producing P2r2-F10 (the identity-side `iam:PassedToService`) and P2r2-F9 (its CloudFormation twin). **The 2026-09-14 takeover-branch rerun exercised both corrected paths:** CloudFormation created the managed image and the orchestrator launched real coding tasks using source-defined permissions after a fresh bootstrap, with no manual IAM workaround. See the [deployment](/sample-autonomous-cloud-coding-agents/architecture/645-p2-clean-deployment-20260913) and [task evidence](/sample-autonomous-cloud-coding-agents/architecture/645-p2-live-task-20260914). This closes that historical clean-rerun requirement; the broader failure/recovery and permission/network acceptance matrix remains open. +## Testing -**P3 (suspend/resume):** +The [completed plan](/sample-autonomous-cloud-coding-agents/architecture/645-p3-implementation-plan) indexes exact suite results and dated evidence. Required coverage includes: -- Unit tests: suspend/resume mapping; agentcore/ecs `unsupported` stubs; approve/deny inline resume with handle loaded from `compute_metadata`. -- HITL lifecycle tests: inline resume failure leaves the decision outcome intact and records the orphan event; orchestrator backstop retries resume; gate expiry fires at `min(monotonic budget, created_at + timeout_s)` — including the suspend/resume case where the monotonic budget exceeds the wall-clock remainder — without disturbing the §13.12 late-approval race protection. -- Smoke: suspend/resume across a simulated approval wait preserving workspace state. +- Strategy state mapping, explicit unsupported results, ARN/Region validation, NO_INGRESS, omitted idlePolicy and bounded uncertain-start recovery. +- Hook readiness/warm-up, AWS-silent build hooks, authenticated payload installation, arbitrary terminate bodies and lifecycle connection closure. +- Decision/timeout/cancellation races, image capability, coding/progress barriers, original deadlines, durable replay and exact-attempt capacity ownership. +- Real S3 version/integrity/access checks, real DynamoDB transactions and actual SDK conversation/Git/workspace recovery after process and disk loss. +- Cloud approve/deny/expiry/cancellation, repeated sleep/wake, AWS credential renewal after actual expiry, retirement/replacement and final resource cleanup. +- Nested fresh deployment and overlapping migration, compatible image/coordinator rollback, normal CLI feedback and live off-switch acceptance. -**All phases:** docs sync for [COMPUTE.md](/sample-autonomous-cloud-coding-agents/architecture/compute) (new column distinguishing MicroVMs from classic Lambda) and [ORCHESTRATOR.md](/sample-autonomous-cloud-coding-agents/architecture/orchestrator) (liveness + suspend lifecycle). +Live evidence must distinguish service acknowledgment from completed guest recovery and synthetic handler checks from actual external-channel submissions. ## References -- Issue [#645](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/645) — originating RFC proposal -- Issue [#491](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/491) — unified liveness decision model (soft dependency, P3) -- Issue [#641](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/641) / ADR-019 (PR [#663](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/663)) — substrate-portable tool plane -- PR [#596](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/596) — ECS Fargate backend (pattern source for conditional wiring) -- [AWS Lambda MicroVMs](https://docs.aws.amazon.com/lambda/latest/dg/lambda-microvms-guide.html) — developer guide; [Running and using MicroVMs](https://docs.aws.amazon.com/lambda/latest/dg/microvms-launching.html) — lifecycle APIs and hooks -- [Agent Toolkit for AWS — aws-lambda-microvms skill](https://github.com/aws/agent-toolkit-for-aws/blob/main/skills/specialized-skills/serverless-skills/aws-lambda-microvms/SKILL.md) — operational constraints (no self-suspend, idle-policy semantics, snapshot uniqueness, size limits) -- [ADR-020](/sample-autonomous-cloud-coding-agents/architecture/adr-020-ears-requirements-syntax) — EARS syntax used for the normative requirements above -- [CEDAR_HITL_GATES.md](/sample-autonomous-cloud-coding-agents/architecture/cedar-hitl-gates) — approval-gate mechanics (decisions #6, #7) the suspend/resume handshake preserves; `cancel-task.ts` / `task-api.ts` — the inline best-effort + reconciler-backstop pattern the resume path mirrors -- [COMPUTE.md](/sample-autonomous-cloud-coding-agents/architecture/compute), [ORCHESTRATOR.md](/sample-autonomous-cloud-coding-agents/architecture/orchestrator) — design docs to be updated by the implementing PRs +- [Issue #645](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/645) — originating proposal +- [Issue #491](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/491) — unified liveness model +- [Issue #641](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/641) — substrate-portable tool plane +- [AWS Lambda MicroVMs guide](https://docs.aws.amazon.com/lambda/latest/dg/lambda-microvms-guide.html) and [lifecycle APIs/hooks](https://docs.aws.amazon.com/lambda/latest/dg/microvms-launching.html) +- [ADR-020](/sample-autonomous-cloud-coding-agents/architecture/adr-020-ears-requirements-syntax) — requirement syntax +- [Compute](/sample-autonomous-cloud-coding-agents/architecture/compute), [orchestrator](/sample-autonomous-cloud-coding-agents/architecture/orchestrator), [approval gates](/sample-autonomous-cloud-coding-agents/architecture/cedar-hitl-gates) diff --git a/docs/src/content/docs/using/Approval-gates-cedar-hitl.md b/docs/src/content/docs/using/Approval-gates-cedar-hitl.md index 74580144a..a97543089 100644 --- a/docs/src/content/docs/using/Approval-gates-cedar-hitl.md +++ b/docs/src/content/docs/using/Approval-gates-cedar-hitl.md @@ -15,7 +15,7 @@ When a rule marked `@tier("soft")` matches a tool call: 1. The agent stops before invoking the tool. 2. A row is atomically written to the approvals table and the task status flips to `AWAITING_APPROVAL`. 3. A progress event (`approval_requested`) is emitted so `bgagent watch` shows the gate in real time. -4. The task waits for your decision up to the rule's timeout (default 300 s, configurable per-rule and per-task). +4. The task waits for your decision without an automatic deadline by default. An explicit per-task or policy-rule timeout can limit that window. 5. On approval, the agent proceeds; on denial, the deny reason is best-effort injected back into the agent's context so it can adapt; on timeout, the gate is treated as a denial with `timed_out` as the reason. A decision is recorded at most once per request. Replaying approve/deny on the same `(task_id, request_id)` is idempotent. @@ -26,7 +26,7 @@ A decision is recorded at most once per request. Replaying approve/deny on the s node lib/bin/bgagent.js pending ``` -Lists every approval across your tasks that is currently awaiting your decision. The default text output gives you the `request_id`, tool, severity, the reason the rule matched, the tool-input preview, the expiry time, and ready-to-run `approve` / `deny` command lines. Pipe through `--output json` for scripting. +Lists every approval across your tasks that is currently awaiting your decision. The default text output gives you the `request_id`, tool, severity, the reason the rule matched, the tool-input preview, the deadline or “no automatic expiry,” and ready-to-run `approve` / `deny` command lines. The JSON `expires_at` is `null` when there is no deadline. Cancelled or completed tasks no longer have answerable requests. ```text 1 pending approval(s): @@ -38,7 +38,7 @@ Lists every approval across your tasks that is currently awaiting your decision. rules: force_push_any preview: git push --force origin feature/xyz created: 2026-05-13T12:04:12Z - expires: 2026-05-13T12:09:12Z (timeout_s=300) + expires: no automatic expiry approve: bgagent approve 01KN37PZ77P1W19D71DTZ15X6X 01R... deny: bgagent deny 01KN37PZ77P1W19D71DTZ15X6X 01R... --reason "..." ``` @@ -98,13 +98,16 @@ node lib/bin/bgagent.js submit --repo owner/repo --issue 42 \ --pre-approve tool_type:Bash \ --pre-approve write_path:tests/** -# Per-task timeout override (platform default is 300s) +# Optional ten-minute decision deadline (the default has no deadline) node lib/bin/bgagent.js submit --repo owner/repo --issue 42 --approval-timeout 600 ``` `--pre-approve` can be repeated up to the platform limit (see `bgagent submit --help` for the current cap). Valid scope forms are the same as the `approve --scope` table above. Hard-deny rules are still enforced — `--pre-approve` only short-circuits soft-deny rules. -`--approval-timeout` sets the task-wide default; a rule with its own `@approval_timeout_s` annotation still takes the minimum of the two. +`--approval-timeout 0` keeps unanswered requests available. A positive setting +limits the decision window; the shortest positive deadline from the task and +matching policy rules wins. Zero does not disable a policy rule's explicit +deadline. Cancelling the task closes its requests. For Lambda MicroVM tasks, `--microvm-sleep-after 600` selects the default 10-minute delay; `--microvm-sleep-after 120` selects two minutes and @@ -112,10 +115,12 @@ For Lambda MicroVM tasks, `--microvm-sleep-after 600` selects the default approval request is created. Waking for approval, denial, or an approaching deadline remains automatic. Sleep never starts a new approval timer. -The default approval timeout is five minutes, so those requests stay awake -with the ten-minute sleep delay. A longer task timeout does not override a -shorter policy-rule timeout. Sleeping saves compute charges but adds snapshot -save/restore charges and wake-up time; short pauses can cost more than staying -awake. The API equivalent is `microvm_sleep_after_s` (zero means off); task +An unanswered request can outlive its MicroVM. After a longer wait, ABCA saves +the workspace and conversation, stops the worker, and releases its capacity. +Your answer can then start a replacement when capacity is available. A request +remaining open does not mean its old computer must stay alive. Sleeping saves +compute charges but adds snapshot save/restore charges and wake-up time; short +pauses can cost more than staying awake. The API equivalent is +`microvm_sleep_after_s` (zero means off); task details return the saved setting. Automatic suspension remains disabled by default pending the [P3 acceptance checks](/sample-autonomous-cloud-coding-agents/architecture/645-p3-implementation-plan). \ No newline at end of file diff --git a/docs/src/content/docs/using/Linear-setup-guide.md b/docs/src/content/docs/using/Linear-setup-guide.md index 9791d3243..5385771d1 100644 --- a/docs/src/content/docs/using/Linear-setup-guide.md +++ b/docs/src/content/docs/using/Linear-setup-guide.md @@ -15,7 +15,7 @@ Set up the ABCA Linear integration so that applying a label to a Linear issue tr ## How it works -You create a Linear OAuth app and authorize it on your workspace. When someone adds the trigger label to an issue in a mapped project, Linear fires a webhook at ABCA; the receiver verifies the HMAC signature, looks up the workspace, resolves a Linear API token, and creates a task. The agent clones the repo, makes the change, opens a PR, and comments back on the issue. +You create a Linear OAuth app and authorize it on your workspace. When someone adds the trigger label to an issue in a mapped project, Linear sends a webhook to ABCA. The receiver verifies its signature and invokes a processor, which resolves the workspace token and creates a task. The agent clones the repo, makes the change, opens a PR, and reports back on the issue. The app is installed with `actor=app`, so everything ABCA writes is attributed to the app rather than to whoever clicked Authorize. @@ -25,10 +25,10 @@ One of two places, chosen automatically at setup time: | | When it's used | What's stored | |---|---|---| -| **AgentCore Identity vault** | The stack was deployed with `--context enableLinearIdentityVault=true` | Nothing long-lived. AgentCore holds the refresh token and mints short-lived access tokens on demand. | +| **AgentCore Identity vault** | The stack was deployed with `--context enableLinearIdentityVault=true` | AgentCore manages the OAuth grant and token refresh. ABCA retains the OAuth client credentials and workspace/webhook metadata in Secrets Manager. | | **Secrets Manager** | Otherwise — including regions where AgentCore Identity isn't available | An OAuth token bundle in `bgagent-linear-oauth-`, refreshed and rotated by ABCA. | -The vault is unavailable on the `lambda-microvm` substrate — see [Not available with `compute_type=lambda-microvm`](#not-available-with-compute_typelambda-microvm) below. +The vault can be enabled with AgentCore, ECS or Lambda MicroVM compute. `bgagent linear setup` picks whichever the deployment supports and tells you which one it used. There is no flag. If the vault isn't available it prints one line and continues on Secrets Manager: @@ -38,11 +38,16 @@ AgentCore Identity not available in us-east-1 — using Secrets Manager. A workspace that started on Secrets Manager and later moves to the vault **keeps** its Secrets Manager token as a fallback. A workspace onboarded straight onto the vault has no such token by design — it needs the vault to be reachable. -When a workspace's authorization dies, ABCA records it on the registry row and publishes to the stack's operational alert topic. That topic has **no subscribers unless you deployed with `alertEmail`**, so set it if you want to hear about a dead workspace rather than discover it from `bgagent platform doctor`. +When a workspace's authorization dies, ABCA records it on the registry row and publishes to the stack's operational alert topic. A new topic starts without subscribers; configure `alertEmail` or subscribe another destination to receive those alerts. A revoked legacy Secrets Manager fallback does not by itself mean the active vault grant is revoked. -#### Not available with `compute_type=lambda-microvm` +#### Using the vault with Lambda MicroVMs -The vault and the Lambda MicroVMs substrate remain gated from being enabled together ([#857](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/857)). The original 505-resource measurement predates stack-size reductions; the [current offline review](/sample-autonomous-cloud-coding-agents/architecture/645-p3-readiness-review) finds room in its tested configurations, but bundled feature-matrix and deployment verification are still required before removing the guard. Use the vault on `agentcore` or `ecs`; MicroVM deployments use Secrets Manager while this gate remains. +Deploy with both `compute_type=lambda-microvm` and `enableLinearIdentityVault=true`. The coordinator sends the workload identity name through authenticated `platform_config`; the guest uses its compute execution role to obtain a Linear token. Credentials are not baked into the MicroVM image. + +When upgrading an existing MicroVM deployment, rebuild the guest image too: +the coordinator and guest must both support the vault configuration fields. + +The runtime resolves the vault in its AWS Region. A grant in another Region or under another workload identity does not automatically carry over. The old resource-count guard ([#857](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/857)) has been replaced with configuration, permission and deployment-budget checks. #### One workload identity per stack diff --git a/docs/src/content/docs/using/Using-the-cli.md b/docs/src/content/docs/using/Using-the-cli.md index f9785b6b1..0cb9fa8d6 100644 --- a/docs/src/content/docs/using/Using-the-cli.md +++ b/docs/src/content/docs/using/Using-the-cli.md @@ -114,7 +114,7 @@ Created: 2026-04-01T00:39:51.271Z | `--max-budget` | Maximum cost budget in USD (0.01–100). Overrides per-repo Blueprint default. No default limit. | | `--idempotency-key` | Idempotency key for deduplication. | | `--trace` | Enable detailed tracing: raises progress preview cap to 4 KB and uploads full NDJSON trajectory to S3 on completion. Download with `bgagent trace download`. | -| `--approval-timeout` | Cedar HITL per-task approval timeout in seconds (default 300). A matching rule with its own `@approval_timeout_s` annotation still takes the minimum. See [Approval gates](#approval-gates-cedar-hitl). | +| `--approval-timeout` | Cedar HITL decision window: `0` (default) keeps unanswered requests available; a positive value sets a 30–3,600 second deadline. A matching rule can require a shorter positive deadline. See [Approval gates](#approval-gates-cedar-hitl). | | `--microvm-sleep-after` | Seconds to wait for approval before putting a Lambda MicroVM to sleep (default 600 = 10 minutes; 0–3600 accepted). Use `off` to keep it awake. Requires the deployment's automatic-sleep feature to be enabled; does not change approval deadlines or affect other compute backends. | | `--pre-approve` | Cedar HITL scope to approve up-front (repeatable). Same scope forms as `bgagent approve --scope`. Hard-deny rules are always enforced. | | `--wait` | Poll until the task reaches a terminal status. | diff --git a/docs/verification/645-lambda-microvm-service-feedback.md b/docs/verification/645-lambda-microvm-service-feedback.md index ecf04662d..ab5a77c5a 100644 --- a/docs/verification/645-lambda-microvm-service-feedback.md +++ b/docs/verification/645-lambda-microvm-service-feedback.md @@ -1,9 +1,14 @@ # Lambda MicroVM service-team feedback tracker -Updated 2026-09-17. Working notes for the ADR-021 takeover. **Not submitted to the +Updated 2026-09-18. Working notes for the ADR-021 takeover. **Not submitted to the service team.** Keep each item's evidence, question, service response and next action here as verification continues. +P3 application acceptance is complete and automatic approval suspension is +enabled in the [normal deployment](./645-p3-normal-closure-20260918.md). +The dated investigations below preserve earlier deployment settings; the +service questions remain open. Account IDs are redacted from this public copy. + In plain language: a MicroVM is the worker's little computer. A lifecycle hook is the doorbell AWS rings to tell it to start, pause or wake. An AWS request ID is the receipt that lets the service team find a particular call. @@ -24,8 +29,8 @@ is the receipt that lets the service team find a particular call. **Impact:** the user approves an action, but the worker stops before continuing. The coordinator releases capacity correctly; the requested coding workflow -failed in these recorded cases. The application correction below is deployed; -automatic suspension remains disabled pending the remaining P3 acceptance gates. +failed in these recorded cases. The application correction below passed P3 +acceptance before automatic suspension was enabled. The [September 17 follow-up](./645-p3-final-image-and-ecs-20260917.md) adds two repository sleep/wake cycles, late approval winning, and wake after actual STS diff --git a/docs/verification/645-p3-cloud-continuation-20260917.md b/docs/verification/645-p3-cloud-continuation-20260917.md new file mode 100644 index 000000000..df70be4d7 --- /dev/null +++ b/docs/verification/645-p3-cloud-continuation-20260917.md @@ -0,0 +1,129 @@ +# P3 cloud continuation acceptance — September 17, 2026 + +The final private MicroVM matrix passes approval, denial, cancellation and explicit +expiry after planned worker retirement, plus approval and denial on the original +sleeping worker. These checks exercise the real guest, model, SDK, lifecycle hooks, +AWS storage, Durable coordinator and API decision handlers. + +This record covers the isolated acceptance deployment. Normal nested migration, +activation and the signed-in CLI path have their own completion gates. + +## Exact build and fixture boundaries + +- Account/Region: `` / `us-west-2`. +- Stack: `backgroundagent-dev-p3-recovery-20260917`. +- Image: `abca-645-p3-recovery-20260917`, version `3.0`, 8,192 MiB. +- Guest ZIP SHA-256: + `a4ab66250ab20aceeefab05d5d13eea8387f7fa4d6a91376bc9265a06cb68013`. +- The ZIP contains 116 files and is 518,981 bytes. +- Coordinator version `4` ran the retirement matrix and wake-deny; version `5` + refined the private command-specific gate for wake-approve. Both use the same + image and production lifecycle implementation. + +The private wrapper selects dedicated tables/buckets and a repository-free +workflow. It sets the real AWS worker lifetime to 600 seconds so production +retirement is reached in approximately five minutes. It does not mock the clock, +AWS responses, lifecycle transitions, checkpoint transfers or model responses. +The configured sleep delay is 30 seconds; production's default remains 600. + +The wrapper invokes the production scheduled manager every 20 seconds to keep the +test short. Production schedules it every five minutes. Approval and denial call +the production handlers with the fixture owner's identity; authenticated HTTP/CLI +acceptance is a separate normal-deployment check. + +## Passing cases + +| Case | Task ID | Verified outcome | +|---|---|---| +| Approve after retirement | `01M2RHVXTC3VZT63008V2D2BVX` | One replacement; the approved command reads the original marker and its exact SHA-256 from restored disk | +| Deny after retirement | `01M2RHVXTGPRGR01ATR0CEGE0N` | One replacement; the denied command does not execute and the agent acknowledges the saved denial | +| Cancel while parked | `01M2RHVXTGPSSCBRM42GD46ANG` | Request/task cancelled; no replacement or terminal delivery | +| Explicit expiry while parked | `01M2RHVXTGB9H7DFHATTF2VEVJ` | The 480-second deadline closes the request; one replacement acknowledges the timeout without executing the command | +| Approve on the original worker | `01M2RN4WWRK0BHJ0VCWDAAT125` | Same worker wakes, reads the saved marker/checksum and finishes | +| Deny on the original worker | `01M2RH8HN8GPGYW08R4HBXWCB6` | Same worker wakes, respects denial and finishes | + +For every retirement case the original worker was observed `SUSPENDED`, then +`TERMINATED` with `stateReason: "Success."`. The coordinator fenced the old worker, +parked the task and released its capacity reservation before the human decision. +An unanswered request had no DynamoDB TTL. + +The harness verifies each immutable manifest's object version, byte count and +SHA-256, saved workspace location and positive accumulated model cost. Replacement +tasks keep their original image version and final costs include both worker runs. +Successful cases finish with released reservations, zero user capacity and no +outstanding continuation launch. Duplicate decision submissions return 404. + +The agent's normal reasoning decides whether the saved action is still relevant. +The platform checks exact task/request ownership and approved tool inputs; this +does not add an action-relevance rechecking system. + +## Findings corrected during integration + +The real retirement tests found a workflow boundary missing from the earlier +process-level tests: hydration overwrote the prepared continuation prompt with +the original task prompt. The resumed model therefore never received the saved +human decision. `workflow/runner.py` now fills the user prompt only when it is +empty, matching the existing treatment of a prepared system prompt. Five tests +exercise the real hydration and agent-run handlers; four failed before the fix. + +The approved continuation instructions also explicitly preserve every saved tool +input field, including descriptions. The exact-input permission check remains +unchanged. Restored denial feedback reports its original timestamp instead of +incorrectly describing a decision many minutes old as a recent 60-second denial. + +Earlier integration corrected retained-request hook checkpoint validation and +AWS's 64-character Durable execution-name limit. Dispatch errors now include the +operation, coordinator version, name length and bounded validation detail. + +One first wake-approve fixture paused at an incidental `ls` command before the +intended marker command. That run was cancelled and excluded from marker +acceptance. The reproducible fixture now gates only the exact intended command, +validates the approval's input before answering, and correlates tool results with +their trace/turn. It fails if parallel calls make that correlation ambiguous. + +## Cleanup and reproducibility + +All 15 task records from final and diagnostic attempts were terminal, all 15 +worker leases were closed, reservations were released and continuation launch +records were absent before deletion. The stack and its 40 tracked resources were +removed. Cleanup also removed automatically created Lambda log groups and checked +absence of the exact owned functions, buckets, images and network connectors. + +Private scripts and raw evidence are outside the Git worktree: + +```text +/Users/sphias/.local/share/abca-verification/645-p3-integration/ +``` + +The final matrix is in `cloud-replacement/results/`; `cleanup.json`, +`cleanup-resources.json` and `terminal-task-and-lease-audit.json` preserve cleanup +proof. Failed and invalidated earlier cases remain archived separately. + +`run.py` provides local, SDK, real S3, DynamoDB and coordinator checks; `cloud.py` +creates a fresh isolated stack and image for the real worker matrix. It uses the +selected worktree's production implementations and records the artifact digest. +The scripts deliberately do not belong in the PR. + +Full agent verification after these corrections passes **2,159 tests**, with +13 explicitly opt-in cases skipped and **86.34%** coverage. The real SDK suite was +run separately. The full infrastructure suite passed 5,223 tests before the +managed runtime-pin addition; 245 focused infrastructure tests and compilation/ +lint subsequently passed for that addition and dispatch diagnostics. + +## Corrected-image follow-up — September 18 + +The later normal-workflow progress/capture race is documented in the +[dedicated record](./645-p3-progress-race-20260918.md). After its correction, a +fresh portable cloud run built image `abca-645-p3-it-09180207b210de:1.0` from +artifact `924a1b51fe6b9aa62f61191a6bde9b10df01d65873181a9385489191492cadbe`. +It passed both same-worker wake/approval and approval after confirmed retirement +and replacement, with real marker/checksum assertions. + +- approve: `01M2S4A6MB2CZJQRP0ZYMT7EYZ`. +- wake-approve: `01M2S4A6MB7WH4A781WFFMQCPK`. + +The owned stack and all 40 tracked resources were removed; cleanup finished at +`2026-09-18T02:30:30.707Z` with no leaked resources. Source hashes, +image receipt, full timelines and cleanup checks are in the private harness +`runs/20260918-progress-race-cloud`. The full agent suite with this fix passes +2,162 tests, with 13 opt-in skips and 86.37% coverage. diff --git a/docs/verification/645-p3-connection-close-rollout-20260916.md b/docs/verification/645-p3-connection-close-rollout-20260916.md index b00931ea8..254c917c7 100644 --- a/docs/verification/645-p3-connection-close-rollout-20260916.md +++ b/docs/verification/645-p3-connection-close-rollout-20260916.md @@ -184,7 +184,7 @@ instrumented comparison have their separate archive in the transport report. ## Remaining completion gates The application correction and these four normal-image workflows are complete. -The broader [P3 plan](./645-p3-implementation-plan.md#remaining-work-in-execution-order) +The broader [P3 plan](./645-p3-implementation-plan.md#implementation-progress) still requires the wider final-image workspace/credential/decision matrix, other-backend permission and network checks, and the normal deployment's coordinated capacity upgrade/drain and rollback validation. Service token diff --git a/docs/verification/645-p3-continuation-protocol-20260917.md b/docs/verification/645-p3-continuation-protocol-20260917.md new file mode 100644 index 000000000..9ee9d71d1 --- /dev/null +++ b/docs/verification/645-p3-continuation-protocol-20260917.md @@ -0,0 +1,77 @@ +# Retained approval continuation — September 17, 2026 + +The local implementation and isolated persistence checks are complete. Full +replacement-worker execution and the normal nested deployment remain open. +This record does not claim P3 deployment acceptance. + +An unanswered request now has no automatic deadline by default +(`approval_timeout_s: 0`). A positive, explicitly selected timeout remains +available. Pending requests have no DynamoDB deletion timer; closing their task +cancels unanswered requests and retains the decision history for 90 days. +The agent decides how to continue after an answer. There is no new system that +tries to determine whether the proposed action is still relevant. + +## Worker and task lifetimes + +At an approval boundary the worker saves the conversation, pending proposal, +workflow context, exact accumulated usage, and full Git/workspace archive. +Objects are versioned and checksummed. New tool execution stays blocked while +the checkpoint is ready. + +The coordinator verifies all saved versions before fencing the old worker. +Its authority is a separate, coordinator-owned `worker-lease#` record: +the worker can read and condition-check that record, but cannot change it. +After confirmed shutdown, the task becomes `PARKED` and returns its capacity +reservation while the approval stays available. + +An answer triggers admission of one replacement attempt when capacity permits. +Admission, its new lease and capacity reservation form one DynamoDB transaction. +A deterministic Durable execution name deduplicates invocations. The replacement +uses the original published coordinator version and exact source image version, +then restores the saved files/conversation and consumes the recorded answer. +Its budget is the original allowance minus the accumulated cost and turns. + +Repository-free MicroVM tasks use a private directory per task, with a local Git +baseline for the same archive format. Recovery preserves those scratch files +without creating a remote or installing a GitHub credential helper. A closed +worker cannot proceed to the workflow's artifact or PR delivery steps. + +The scheduled continuation manager retries unfinished retirement, missed +dispatches and terminal cleanup. Its persistent scan cursor advances across +invocations. It does not have permission to launch, suspend or resume workers; +launching remains with the pinned coordinator. + +A failed start request does not prove that AWS failed to create a worker. +Terminal saved tasks keep their capacity until the manager confirms shutdown, +or until the full service lifetime has elapsed for an unknown handle. The +coordinator records `CLOSED` in the exact-attempt lease before the atomic release. +`TERMINATING` alone is insufficient. + +## Verification + +- TypeScript compilation and ESLint passed. +- Focused coordinator/cleanup checks: **107 passed, 15 skipped**, 11 suites. +- Python recovery entry point and runner checks: **69 passed**. These include + actual Git/archive restoration through `restore_for_task`, registration races, + approve/deny/timeout handling, identity rejection and cumulative budget limits. +- Private real-SDK approve/deny recovery: **2 passed** using the pinned SDK/CLI. +- Full agent quality after repository-free recovery: **2,139 passed, 13 skipped**, + with 86.21% coverage. CLI compilation and **943 tests** passed. +- Private AWS coordinator suite: **12 passed**, including six competing + admissions, deliberately lost successful fence/park/admission replies, a + day-old unanswered request, repository-free admission, opt-in expiry, capacity + ownership and cleanup. + All three temporary tables and the temporary versioned bucket were deleted; + absence was verified. This suite launched no MicroVM worker. + +The standalone integration runner, README, source hashes and raw evidence are +outside Git at +`/Users/sphias/.local/share/abca-verification/645-p3-integration`. +The coordinator run is `runs/20260917T191209Z-6beab7bb`; the SDK run after the +retained-request default is `runs/20260917T184225Z-f4ac16e3`. + +The requested +[microvms-agentd reference](https://github.com/laithalsaadoon/microvms-agentd/tree/78304e361fbbe62e3a6b255b43c6f6c372b47510) +informed bounded transfers, disk-reserve checks, artifact assertions and cleanup +receipts. No reference executable was run and no source was copied. ABCA keeps +its task-scoped credential broker and complete Git/workspace preservation. diff --git a/docs/verification/645-p3-durable-live-20260915.md b/docs/verification/645-p3-durable-live-20260915.md index b51be4708..793779419 100644 --- a/docs/verification/645-p3-durable-live-20260915.md +++ b/docs/verification/645-p3-durable-live-20260915.md @@ -3,7 +3,7 @@ Date: 2026-09-15. This continues the [deployment and isolated guest checks](./645-p3-live-deployment-20260915.md). Production automatic suspension remains disabled. The full -[acceptance matrix](./645-p3-implementation-plan.md#7-acceptance-matrix-and-completion-gates) +[acceptance matrix](./645-p3-implementation-plan.md#acceptance-matrix-and-completion-gates) is not complete. ## Scope and isolation diff --git a/docs/verification/645-p3-final-image-and-ecs-20260917.md b/docs/verification/645-p3-final-image-and-ecs-20260917.md index 3d0707f72..cd5675c5b 100644 --- a/docs/verification/645-p3-final-image-and-ecs-20260917.md +++ b/docs/verification/645-p3-final-image-and-ecs-20260917.md @@ -26,10 +26,10 @@ The actual tool sequence was: Both lint runs and both Vitest runs passed; the repository has one test. The marker hashes were: -| Marker | SHA-256 | +| Marker | SHA-256 prefix (16 hex characters) | |---|---| -| First | `8437311a6e403b723febea0fbae6509e03b48c198b597aab9f04cd2cf7e30d5e` | -| Second | `cb1e409f22270af5 (SHA-256 prefix)` | +| First | `8437311a6e403b72` | +| Second | `cb1e409f22270af5` | Worker `microvm-3a114559-4c86-34a0-af36-0a04e6232a5f` recorded two suspend and two resume HTTP 200 results from PID 1. Independent normal approval-handler diff --git a/docs/verification/645-p3-implementation-plan.md b/docs/verification/645-p3-implementation-plan.md index 13e39cf21..5ef06713a 100644 --- a/docs/verification/645-p3-implementation-plan.md +++ b/docs/verification/645-p3-implementation-plan.md @@ -1,952 +1,209 @@ # ADR-021 P3 implementation and completion plan -Prepared 2026-09-13 from `main` `5e10038c7e28179b302ac4de78b709795aeba3ce`. Read the [review](./645-p3-readiness-review.md) for evidence and the beginner introduction. This document tracks implementation and validation; local completion does not mark P3 live acceptance complete. - -**Approval UX scope decision:** leave assessment of whether an action is still -relevant to the agent's ordinary reasoning. The user has excluded a separate -approval rechecking system, including automatic context/staleness detection, -from this plan. The platform continues to enforce authenticated ownership, -the task/request identity, approval or denial, and task cancellation. This scope -decision does not change the currently implemented approval deadlines. +Updated September 18, 2026 UTC. The [original review](./645-p3-readiness-review.md) +explains P1 and P2 and introduces the service. This is the current checklist; +dated verification records describe what was true at each earlier milestone. + +**Current status: P3 complete.** Implementation, cloud continuation, normal +activation, live off-switch/rollback checks, retirement of the old flat +infrastructure, final smoke and cleanup are verified. See the +[normal acceptance record](./645-p3-normal-closure-20260918.md). + +## The behavior being delivered + +When the agent needs permission, it saves a question for the person. After ten +minutes, its computer can sleep while the question stays available. A person can +approve or deny later. If the wait is long, ABCA saves the conversation and files, +confirms that the old computer has stopped, and releases its capacity. The answer +can then start a replacement computer with the saved work. + +- Unanswered approvals have no deadline by default: `approval_timeout_s=0`. +- An explicit decision timeout remains available, from 30 to 3,600 seconds. + A positive policy-rule deadline can also apply. +- The default sleep delay is 600 seconds. `--microvm-sleep-after off` keeps that + task awake; a deployment switch can disable new suspensions globally. +- An expired explicit deadline denies the proposed tool call. Waking or replacing + a worker does not restart the decision clock. +- Cancellation closes unanswered requests atomically. Already-recorded decisions + remain part of task history. +- The platform checks authenticated ownership, task/request identity, exact + approved tool inputs and cancellation. The agent decides whether the action + still makes sense. A separate relevance/staleness rechecking system is excluded + by the user's scope decision. + +“Durable” means saved outside the worker. A checkpoint is the saved conversation, +files and task context. A capacity reservation is the worker's place in the +deployment's concurrency limit. ## Implementation progress -Prerequisite work is tracked here on `fix/645-microvm-readiness`. “Completed” means implemented and checked locally; AWS deployment and live verification have separate completion gates below. - -**AgentCore effective permissions (2026-09-17):** the -[actual runtime probe](./645-p3-agentcore-permissions-20260917.md) passes 19 -ambient/scoped AWS permission checks on normal runtime version 5. The -independent audit verifies the exact tool command and AWS receipts, and -separately records a watcher snapshot race after normal finalization. -Its session, private infrastructure and harmless probe objects were removed. - -**Normal repository path (2026-09-17):** the -[repository acceptance record](./645-p3-repository-path-20260917.md) verifies -normal clone/setup, one approval sleep/wake, passing post-build/lint and -resolve-only delivery for existing PR #584 on image 6.0. Private event storage -and supported prompt configuration excluded publication; GitHub snapshots -were unchanged. The worker finalized without repair and all private -infrastructure was removed. The initial guardrail rejection is retained as -an excluded attempt. This adds the eighth successful image 6.0 Durable workflow. - -**Final-image and ECS follow-up (2026-09-17):** the -[new acceptance record](./645-p3-final-image-and-ecs-20260917.md) verifies -repository marker persistence through two sleeps on image 6.0, unchanged -repository commit and passing lint/tests before and after, and late approval -winning with the original deadline. Both private Durable cases finalized -without repair and their infrastructure was removed. The same follow-up found -and fixed missing ECS approval-table configuration in source `c14d2ba0`. -Two fresh ECS approval/cancellation cases and 19 real permission/port checks -passed; the isolated ECS deployment was removed. The full CDK suite passed -5,057 tests. The real image 6.0 credential-expiry case also passed: its worker -was observed asleep after expiry, CloudTrail confirmed a new session preserving -identity, and the original deadline and automatic finalization held. Normal -sleep gates remain off. - -**Wake transport correction (2026-09-16):** the -[instrumented comparison](./645-p3-wake-transport-20260916.md) captured a sixth -exact refusal: the restored event loop expired the old suspend connection while -PID 1 still owned its listener, with no fresh resume connection. Two unchanged -cases sleeping longer than 90 seconds passed. Three private candidate cases -with explicit `Connection: close` passed, each proving socket closure before -freeze, no idle timer on that socket, fresh resume connection, original timeout, -late approval rejection and complete cleanup. This supplies a concrete -application correction. The [normal rollout and acceptance](./645-p3-connection-close-rollout-20260916.md) -now verify image 6.0 and coordinator 10, 8,192 MiB, both sleep switches off, -and four real Durable workflows: approve, deny, timeout winning over a late -decision, and cancellation while asleep. All four finalized without watcher -repair; temporary infrastructure was removed. Six deployed task-handler checks -also verify wake feedback, including nonretryable service/admin guidance for -hook timeouts. Service dispatch traces and the wider final-image matrix remain -separate gates. - -**Adjustable sleep deployed (2026-09-16):** source `a81c565d` adds -`microvm_sleep_after_s` and CLI `--microvm-sleep-after `, with a -600-second default and zero to stay awake. The full build passed 7,939 tests; -56 real DynamoDB Local transaction tests passed separately. -The [settings and live record](./645-p3-user-sleep-20260916.md) verifies normal -coordinator version 8, retained version 7, unchanged image 5.0 and 475 resources, -both suspension gates off, and seven deployed API checks. Five of six worker -cases passed: default/off/custom timing, late approval winning its decision -race, and safe failure after actual credential-renewal denial. The timeout-wins -case reproduced the fifth resume connection refusal. Its failure remains open; -all six workers and temporary verification infrastructure were cleaned up. -The normal deployed capacity reconciler also passed an owned overcount and -terminal-reservation repair while preserving the real waiting worker. - -**AgentCore compatibility follow-up (2026-09-16):** the -[reviewed container update and live checks](./645-p3-agentcore-20260916.md) -advanced the normal AgentCore runtime and its default endpoint to version 5. -The actual change modified only its container URI; all 475 resource identities -and normal MicroVM settings stayed unchanged. Fresh production-handler fixtures -passed approval and cancellation, kept AgentCore awake beyond a requested -MicroVM sleep delay, and released both reservations. Both test sessions were -verified absent. An earlier wrapper-invalidated attempt is explicitly excluded. -ECS, the wider role/network matrix and the unexplained MicroVM wake failures -remain separate gates. - -**Independent PID 1 follow-up (2026-09-16):** the -[new diagnostic](./645-p3-pid1-observer-20260916.md) kept the original server as -PID 1 and observed it from a child process. One corrected timeout-winner case -passed with automatic cleanup. Another wake failed with the distinct generic -message `Resume lifecycle hook failed.` The observer saw PID 1 owning its -listener after restoration, 519 ms before AWS terminated the worker, but no -resume hook entry appeared. A first attempt's incorrect HTTP 409 expectation -is retained and excluded from automatic-finalization acceptance; the existing -late-decision contract is HTTP 404. The final planned attempt was not started. -The new failure is tracked separately as F08; neither it nor the five exact -connection refusals is resolved. - -**Generic wake feedback deployed (2026-09-16):** the -[reviewed code-only update](./645-p3-wake-feedback-20260916.md) advanced the normal -coordinator to version 9, retaining version 8 and changing no image, policy or -sleep setting. All 11 deployed classifier consumers matched the reviewed ZIPs. -Four normal task-API checks passed, with synthetic rows removed afterward. -The observed generic wake failure now receives specific service/admin guidance; -already-persisted stable codes remain unchanged. This corrects feedback, not -the unresolved wake failure. - -**Capacity upgrade rehearsal (2026-09-16):** the -[isolated AWS protocol check](./645-p3-capacity-upgrade-20260916.md) passed 12 -checks: admission fences, old/current task drains, safe repeated finalization, -drained rollback and re-upgrade. It used unchanged old function bodies, the -current reservation helper and the exact deployed reconciler artifact. -All five functions, two roles, two tables and five log groups were removed. -This verifies the bounded table protocol, not a completed migration of the -normal deployment's admission routes and retained durable executions. - -**Comment cleanup follow-up (2026-09-16):** ECS comments now describe the actual -4-vCPU/16-GiB/50-GiB build defaults and the scope of its existing build settings. -Shared role and runner documentation now distinguishes AgentCore/ECS credential -export from MicroVM's retained scoped provider. TypeScript emitted code and -Python executable syntax trees were unchanged. These corrections did not change -sizing, build settings or credential behavior; ECS live acceptance remains open. - -**Guest hook milestone (2026-09-14):** production -[worker suspend/resume hooks](./645-p3-lifecycle-hooks.md) now connect the guest -barrier to atomic checkpoint writes and retained-credential refresh followed by -task/gate reconciliation. Duplicate acknowledgments stay within one approval -generation; a new gate cannot reuse an old wake result. Shared handler/service -budgets are 20/30 seconds. At this milestone, image capability and supervisor -integration remained open. -No deployment or automatic suspension was enabled in this milestone. - -**Image milestone (2026-09-15):** [per-worker image capability](./645-p3-image-capability.md) -now declares all six hooks and the shared protocol marker. The coordinator saves -the worker handle first, verifies the exact returned image ARN/version, then -conditionally records support in both start receipt and compute metadata. Missing -or unreadable support permits normal coding and disables new suspension. Database -race tests reject changed identities and recover a committed capability after a -lost reply. No deployment or automatic sleep was enabled in that milestone. - -**Supervisor milestone (2026-09-15):** [production supervision and approval wake](./645-p3-supervisor.md) connect the policy/store to durable polling and post-commit approve/deny handlers. Recovery clocks and the original service lifetime survive replay; failed wake and cleanup remain visible. The rollout flag defaults off and uses a live Parameter Store switch for existing durable executions. That milestone passed full repository validation (5,028 CDK, 1,951 Python and 928 CLI tests). - -**P3 deployment follow-up (2026-09-15):** the [live record](./645-p3-live-deployment-20260915.md) -verifies bootstrap `1.8.0`, source `9a5f4606`, six-hook image `3.0`, 475 root resources -and both suspension settings off. Rollout review exposed missing retention for -pinned coordinator/guardrail versions; the fix passed another full build -(5,030 CDK tests), and existing versions were protected before replacement. -Initial isolated guest cases pass. These use a local production -supervisor and manual guarded Suspend; deployed durable entrypoint, automatic -suspension admission and the rest of the acceptance matrix remain open. - -**AWS durable follow-up (2026-09-15):** a -[temporary isolated durable supervisor](./645-p3-durable-live-20260915.md) -now exercises the production handler and automatic lifecycle policy in AWS. -Eleven complete cases pass, including original-deadline timeout, cancellation -during real process-crash recovery, worker death, live-switch rollback and -wake after an actual supervisor outage past the approval deadline. -Three approval wakes failed with a service-reported connection-refused error; -a fresh retry passed, but the cause remains open. The long case renewed actual -expired credentials, but its pending approval callback was abandoned before the -original deadline, so overall acceptance failed. Temporary AWS fixtures have been -removed. The [callback-timeout follow-up](./645-p3-callback-timeout.md) records the -reproduction and correction. Image `4.0` is now deployed; its -[fresh AWS acceptance run](./645-p3-callback-live-20260915.md) passed all nine -core callback cases, including renewal after real credential expiry and the -original approval timeout. Mutable files also survived two approval generations. -A fourth [connection refusal](./645-p3-resume-refusal-investigation.md) occurred -on this image and remains unresolved. -The [effective IAM follow-up](./645-effective-iam-20260915.md) adds -37 actual AWS metadata checks, 10 S3 checks, and failure after real signer -credential expiry. Public-object checks pair anonymous success with worker-signed -denial. These results supersede corresponding untested items -above without completing the full acceptance matrix. - -**Live infrastructure and image deployed (2026-09-14):** the -[clean P2 deployment record](./645-p2-clean-deployment-20260913.md) tracks the new -Oregon environment, four deployment fixes and actual verification results. -The clean deployment used bootstrap 1.7.0 and application source `29dcaa74`. -CloudFormation reached `UPDATE_COMPLETE`; managed image version `1.0` became active, its ready and -validate hooks passed, and authenticated API reads passed after the update. -The subsequent [image rebuild fix](./645-microvm-image-rebuild-20260914.md) -deployed `e1d5debe` through a normal reviewed update and activated version `2.0`. -Its ready/validate hooks passed; repeated packaging reused the artifact, and -redeploying the same cloud assembly reported no changes. The root had -474 resources at that milestone. -The [live task verification](./645-p2-live-task-20260914.md) on image `1.0` passed -normal coding, PR iteration and cancellation on `isadeks/vercel-abca-linear`, -including heartbeat, npm checks, Memory writes and automatic cleanup. The -repository's [temporary verification configuration](./645-p2-repository-config-20260914.md) -passed all four pre/post npm commands live, and a worker-reported failure was -cleaned up. The user subsequently requested removal of the CLI addition; its -live overrides were also removed. Repository mise tasks are still needed for -the restored default commands. -The [image 2.0 payload verification](./645-p2-payload-live-20260914.md) passed -11 transport/failure cases, including URL expiry/revocation, a foreign-manifest -denial and a payload over 1 MiB. Concurrent/repeated preparation and immediate -identical Run replay passed; all 12 disposable workers and 29 synthetic object -locations were cleaned up. These direct probes use operator credentials for -preparation and bypass coordinator admission/finalization. -The [start/recovery follow-up](./645-p2-start-recovery-live-20260914.md) -passed simultaneous/changed-request service checks, terminated-worker replay -through roughly five minutes, and 13 production-code cases against AWS storage -with local process/reply faults. Saved-handle recovery, changed-input refusal, -cancellation and the actual 120-second cutoff passed. These tests use operator -credentials and do not interrupt the deployed durable Lambda. -Deployed-coordinator recovery and the wider IAM/network matrix still remain. -Full P2 acceptance and the remaining P3 live gates remain open. The batch notes below -record what was verified at their original completion; their deployment status -is superseded by these records. - -- [x] Review current implementation, clean up verified stale comments, and prototype nesting. -- [x] Fix server-test thread isolation (#841). -- [x] Grant scoped coordinator payload deletion (#817). -- [x] Refresh heartbeat atomically after approval. -- [x] Stabilize terminal failure classification and user retry guidance (#817). -- [x] Exercise real S3 bad-byte paths and fix closed-stream error classification (#817). -- [x] Require new ARN fields to participate in validation; pin contract fields and anchor (#817). -- [x] Bind configuration to IAM-authenticated deployment manifests and use single-object payload links for ECS/MicroVM (#817 / #700). -- [x] Verify MicroVM manifest/download transport, malformed or mismatched inputs, URL expiry/revocation, a foreign private-bucket denial and >1 MiB transport in AWS; verify concurrent/repeated/conflicting S3 preparation with operator credentials. -- [x] Verify deployed MicroVM-role metadata/S3 permissions, public-object denial and actual signer-credential expiry in AWS; see the [effective IAM evidence](./645-effective-iam-20260915.md) for scope. -- [x] Fix ECS approval-table configuration and verify current-container approval/cancellation, v2 payload cleanup, ambient/scoped permissions and port 443/80 controls in a bounded deployment; remove its infrastructure. -- [x] Verify 19 actual ambient/scoped permission requests from normal AgentCore runtime 5 and remove the private fixture. -- [x] Verify actual remote MCP calls, runtime HTTPS/port-80 behavior and unauthenticated endpoint denial before/after sleep on image 6.0; retain observer limitations. -- [ ] Complete the installation-specific coordinated-rollout matrix in AWS. -- [x] Implement saved MicroVM start receipts, stable tokens, input fingerprints and handle recovery. -- [x] Verify immediate identical `RunMicrovm` replay returns the same worker ID in the live payload probes. -- [x] Verify simultaneous identical Run calls, changed-parameter rejection, and replay after termination through roughly five minutes against AWS; distinguish cached Run responses from fresh VM state. -- [x] Verify production start/receipt/payload code against AWS with lost replies and local process death, saved-handle recovery, changed-input/cancellation refusal and actual receipt expiry. -- [x] Verify real AWS durable replay after a saved worker receipt and process exit, including cancellation during recovery, in the isolated production-handler fixture. -- [x] Verify deployed durable recovery after a committed registration reply is lost, cancellation during registration, and operator recovery/termination of a worker whose ID was not saved. -- [ ] Establish AWS behavior after token retention expires and recovery when guest identity logs are missing or ambiguous. -- [x] Make capacity acquisition/release atomic per task across crash replay; unify counter writers and repair. -- [x] Verify normal deployed capacity repair while preserving a waiting worker; verify bounded 600-user AWS pagination, partial-scan safety and conservative legacy handling. -- [ ] Complete the capacity protocol's old-writer upgrade/drain and rollback procedure; validate volume against the intended production workload. -- [x] Restrict agent task updates to reporting fields; remove replacement/deletion and worker counter grants. -- [x] Verify metadata restrictions with real AWS task-tagged sessions and mixed transactions under the deployed MicroVM role; retain status/tag trust limits and the separate other-role gate. -- [x] Replace unused logging-failure bookkeeping with structured stdout diagnostics (#810); document shared runtime networking and verify large registry payload delivery (#818). -- [x] Deploy a fresh bootstrap, application and managed image from current source; verify build hooks and API reads. -- [x] Verify normal coding, PR iteration, Memory writes, live logging and successful/canceled-task cleanup in AWS; observe cancellation preserve another task's capacity. -- [x] Verify automatic pre/post npm checks under temporary overrides and worker-reported failure cleanup in AWS. -- [x] Give managed image builds immutable, checksum-verified artifacts and require their digest in deployment context; packaging, construct, stack and CDK-nag regressions and the full build pass. -- [x] Verify a normal CloudFormation update builds and activates image `2.0` from the changed artifact URI; repeat packaging reuses the verified object and a same-assembly redeploy reports no changes. -- Deferred at user request: publish mise tasks in the target repository and verify its default commands. The CLI addition and temporary overrides were withdrawn; this repository configuration work is outside the current P3 implementation. -- Optional: production nesting remains unimplemented; the P3 deployment uses 475 of the root stack's 500 resource slots. P3 does not inherently require nesting. Recheck the count for supported feature combinations and validate the split/migration if adopted. -- [x] Add mandatory pause/wake command methods across all three compute strategies, with explicit unsupported results and bounded MicroVM requests. -- [x] Keep the original approval deadline through database writes and polling, including frozen/backward clocks; preserve decision races and cancellation. -- [x] Save gate/VM-bound lifecycle intent with stale-writer protection; add explicit VM observations and a tested policy helper. -- [x] Add the guest pause controller, original-gate registration, parallel-tool tracking, progress acknowledgment tracking, heartbeat/read drain and generation-guarded wake completion; reseed the application PRNG at run and controller resume. -- [x] Add production agent hooks with acknowledged checkpoints, retained ambient/tenant credential renewal, a sole scoped Claude provider and atomic task/gate reconciliation; verify duplicates, original deadlines, timeout and teardown behavior locally. -- [x] Declare compatible image hooks using shared budgets and bind lifecycle capability to the actual image/version used by each worker; keep automatic sleep disabled until integration/live acceptance. -- [x] Persist bounded poll/recovery counters and connect lifecycle policy to the supervisor. -- [x] Connect post-commit approval wake, bounded diagnostics/cleanup, the default-off rollout flag and scoped IAM. -- [x] Complete full repository validation with the P3 supervisor and live switch. -- [x] Deploy the supervisor and six-hook image with suspension disabled; verify six isolated guest cases, repair the discovered cancellation stop omission, and prove API termination before test cleanup. -- [ ] Deploy and verify the complete P3 sleep/wake lifecycle in AWS. -- [x] Deploy the explicit SDK callback-timeout fix and repeat long sleep, late wake, approval, denial and cancellation; require final approval/tool evidence as well as cleanup. Image `4.0` passed these checks, including real renewal after credential expiry. -- [x] Deploy the explicit connection-close correction and verify approve, deny, timeout and cancel-asleep on normal image 6.0 through real Durable execution; retain all six exact refusals and the generic failure. The [transport comparison](./645-p3-wake-transport-20260916.md) proves closure before freeze; the [rollout record](./645-p3-connection-close-rollout-20260916.md) records normal-image acceptance and cleanup. -- [x] Complete the wider image 6.0 workspace, multiple-gate, late-decision-winner and expired-credential checks, with actual CloudTrail renewal proof and private cleanup. The repository case uses a temporary clone/artifact workflow; normal repository-bound delivery remains separate. - -First prerequisite batch completed locally on 2026-09-13: - -| Commit | Completed work | Proof | -|---|---|---| -| `b5155928` | #841: join test pipeline threads before restoring mocks/environment; retain and report timed-out handles; distinct task IDs | A deterministic two-test subprocess regression failed on the old fixture and now passes. All 234 server/isolation tests pass. | -| `564ccc19` | #817 deletion IAM: exact `s3:DeleteObject` on `*/payload.json` in the dedicated bucket, on the coordinator only | The IAM regression failed before the grant; construct and full-stack assertions now verify the action and resource scope. Worker read-only permissions are unchanged. | -| `5490d645` | Approval resume: refresh heartbeat in the same conditional write that restores RUNNING | Atomic-write regression failed before the change; immediate post-wait polling is tested for AgentCore and MicroVM. Cancellation/wrong-gate conditions remain intact. | - -Validation for this batch: `mise run quality` in `agent/` passed lint, formatting, type checks and **1,785 tests**, with **83.75%** coverage. Six relevant CDK suites passed **509 tests** via `mise run testf`; CDK ESLint and TypeScript compilation also passed. The original review/cleanup is commit `19904775`. - -These are source changes, not changes to deployed AWS resources. The IAM fix needs a normal stack deployment; the heartbeat fix needs an updated agent image. No bootstrap-policy change is required by this batch. The historical validation section at the end describes the earlier review commit only. - -Second prerequisite batch completed locally on 2026-09-13: - -| Commit | Completed work | Proof | -|---|---|---| -| `8c04dd34` | #817: exercise real S3 body failures; classify a closed stream as unreadable payload | Five route cases cover bad JSON/encoding, incomplete/closed streams and a non-object body. The closed-stream case failed with HTTP 400 before the fix and now returns structured HTTP 500 without installing config or starting work. | -| `2e20d54a` | #817: stable terminal failure codes, precise region/auth/config guidance and consistent task/reply classification | New regressions reproduced the prior misclassification. Tests assert saved message → task API → channel/panel guidance, legacy records, hook 4xx/5xx and start-retry decisions. | -| `cfe95c5e` | #817: exact contract assertions and reverse ARN-validation guards | Negative mutations failed in both Python and the real constants-checker subprocess before the fix. New ARN fields pass only when added to the validation set. | - -Final checks for the second batch: `mise run quality` passed **1,797 Python tests**, lint, formatting and type checks (**83.87%** coverage). Eight relevant CDK suites passed **476 tests**; CDK ESLint, TypeScript compilation, constants-sync and Markdown link checks passed. These counts describe the selected suites for each batch, not additional disjoint tests. No cloud resources were changed. Deploy the orchestrator update and an updated agent image to use these fixes. At the close of that batch, trusted configuration provenance, task-scoped payload reads, uncertain-start retries and P3 remained open. - -Third prerequisite batch completed locally on 2026-09-13: - -| Commit | Completed work | Proof | -|---|---|---| -| `e9944475` | Saved MicroVM start receipts, stable request tokens, immutable payload fingerprints, bounded replay and handle recovery; cancellation/registration reconciliation; one start-failure finalization path; consistent initial finalizer reads | Fault injection covers a lost successful response, process restart, saved-handle reuse, changed/expired requests, cancellation and failed/lost database writes. Three HTTP-timeout cases and two stale-finalization cases reproduced incorrect outcomes before their fixes. | - -Final checks for the third batch: CDK ESLint and TypeScript compilation passed. `mise run testf -- test/handlers/ --detectOpenHandles` passed **151 suites / 3,557 tests** and exited successfully. An earlier ordinary broad run reported a delayed-exit warning; the diagnostic run produced no open-handle trace, so its cause remains unidentified. The changed start/recovery integration suites also exited cleanly in isolation. Documentation sync, the **77-page** Astro build and Markdown link checks passed. No Python source changed in this batch. - -Deploy the orchestrator update to activate these changes. This batch needs no additional IAM/bootstrap change or agent-image update. The receipt is internal task-table data, not a new public task field. The service emulator proves our retry behavior; AWS token retention/conflicts and unknown-ID cleanup still need live evidence. At the close of the third batch, atomic capacity-slot release and the remaining P2/P3 work were still open. - -Fourth prerequisite batch completed locally on 2026-09-13: - -| Commit | Completed work | Proof | -|---|---|---| -| `0bb95243` | Task-owned atomic capacity reservations; one admission owner; shared finalizer/stranded release; guarded queue restoration; revision-checked repair including approval waits; scoped IAM and stale counter-doc cleanup | The old replay regression reduced two seats to zero. Fifteen tests against DynamoDB Local now verify real transaction conditions, competing writers, lost committed replies, empty-counter races and stale-revision rejection. Unit and construct tests cover handler wiring and permission scope. | - -Final checks for the fourth batch: CDK ESLint and compilation passed. **153 handler suites / 3,600 tests** passed with the local integration suite enabled; **15** of those tests used DynamoDB Local. The focused run passed **323 tests**, including the two relevant IAM construct suites; those counts overlap. Documentation sync, the **77-page** Astro build and Markdown link checks passed. The broad handler run exited successfully after the previously observed delayed-exit warning; the focused run exited normally. The temporary database container was stopped and removed. - -No AWS resources changed. Activating this batch requires the coordinated deployment/drain procedure in [capacity verification](./645-capacity-reservations.md), including the reconciler's task-update permission and upload confirmation's narrowed counter access. No agent source/image or bootstrap bundle changed. Coordinator metadata protection is newly tracked in 1G; internal reservation fields currently share an agent-writable row. - -Fifth prerequisite batch completed locally on 2026-09-13: - -- Main-task agent writes now use a reviewed attribute allowlist; whole-row replacement/deletion and coordinator field updates are excluded. Missing session/attribute context fails closed. ECS's legacy direct path shares the restriction. -- Removed unused AgentCore/ECS capacity-table grants and the uncalled Python submission/session-registration helpers. Corrected comments that overstated tenant isolation or described approval writes as best-effort. -- The old-policy regression failed on `PutItem`. **1,795 Python tests** pass with **84.47%** coverage, including actual writer-request/permission-contract checks; **268 CDK tests** pass across session-role, ECS, MicroVM and full-stack suites. Python quality, CDK lint/compilation and documentation checks pass. These counts overlap earlier batches; four obsolete helper tests were removed and two contract tests added. -- [Metadata verification](./645-coordinator-metadata.md) records the effective source-policy boundary, writer inventory, rollback constraints and pending real-AWS allowed/denied transaction matrix. No table migration or bootstrap-policy update is required. Application-role deployment and a matching agent image remain necessary; nothing was deployed. - -Sixth prerequisite batch completed locally (2026-09-13), commit `82a9df80`: - -- v2 payload bootstrap for #817/#700 now covers both ECS and MicroVM: IAM-authenticated deployment manifests, one-object signed downloads, immutable private retry references, bounded bytes/expiry and coordinator cleanup of both task objects. Worker reads outside the manifest prefix and payload-bucket listing are explicitly denied. -- Regressions reproduced and fixed bearer-URL leaks through Python and JavaScript exception chains. Removed the old unsigned transports, stale permission/compatibility comments and unused per-backend key helpers. -- Python quality passed **1,806 tests / 84.51% coverage**. The broad CDK run passed **159 suites / 3,935 tests** with `--detectOpenHandles` and exited normally; **15 existing DynamoDB Local tests were skipped** because this batch did not start that service or change its transaction protocol. The final five transport/strategy suites passed **187 tests** after the last cleanup (overlapping the broad run); CDK lint/compilation, Python lint/type checks, constants-sync, the **77-page** docs build and link checks also pass. -- The [bootstrap runbook](./645-payload-bootstrap.md) records the design, AWS documentation evidence, remaining trust boundaries, live allowed/denied matrix and coordinated deployment/rollback procedure. Nothing was deployed; effective IAM, conditional S3 writes, expiry, DNS/HTTPS, ingress negatives and clean launches remain gates. - -Seventh prerequisite batch completed locally (2026-09-13): - -- #810: removed the unread CloudWatch failure counter, lock and unused threshold. Debug/warn failures emit structured stdout records with writer, task ID and exception class; no AWS retry or failed-message contents. Six client/stream/event failure cases reproduced the old unstructured output and now pass. There is no metric or configured alarm; stdout collection remains a live gate. -- #818: documented that remote non-443 endpoints are unsupported under all three shipped runtime policies, with no new connectivity validator or port grants. A real-hydration test resolves a large MCP asset, checks its durable audit record and v2 S3 bytes, and verifies that the Run reference stays within 4,096 bytes. Python separately verifies the real download, hook mapping and `.mcp.json` loader. Live remote-tool connectivity is still unproven. -- Full agent quality passed **1,811 tests / 84.51% coverage**. CDK lint/compilation and **38 tests** in the two relevant orchestrator/registry suites passed. These counts overlap previous runs; no CDK runtime code or IAM changed in this batch. -- Source/generated registry/compute/deployment documentation, the **77-page** build and link checks pass. An offline coordinator bundle check includes the S3 client, presigner and shared constants; it does not replace a full deployed packaging check. Nothing was deployed or posted to the issue trackers. Package 3's clean P2 rerun and package 7's live P3 matrix remain required. - -First P3 foundation batch completed locally (2026-09-13): - -- The shared strategy contract now requires `suspendSession` and `resumeSession`. AgentCore/ECS return `{supported:false}` without calling AWS. MicroVM sends the exact identifier with a 10-second request bound, returns `{supported:true}` only on acknowledgement, and surfaces sanitized errors. Conflicts, missing VMs and uncertain timeouts still require state reconciliation; they are not treated as success. -- An immutable approval deadline is captured with the original row, before database writes/notifications. Polling uses the smaller UTC/monotonic remainder. Regressions reproduced a frozen-clock gate that waited after ten minutes and a 30-second window stretched to 42 seconds by a slow write. The fix also preserves late committed decisions, missing-row handling and cancellation. -- CDK lint/compilation and **5 suites / 169 tests** pass. Full agent quality passes **1,823 tests / 84.53% coverage**, including **12 new clock/race/cancellation cases**. Counts overlap earlier runs. -- No callers, IAM grants or image hooks enable automatic suspension yet. The same deadline must still be registered in the future lifecycle context and checked immediately by `/resume`. Durable intent, policy, credential refresh, acknowledged progress durability and live validation remain open. - -Second P3 foundation batch implemented locally (2026-09-13): - -- Added `microvm_lifecycle` coordinator records with unique generations, VM/gate identity, desired action, original request time and deadline. Writes check the current task/handle/gate/generation; suspend also transaction-checks the exact PENDING approval. Wake remains sticky within one gate, records are retained, and repeated saves preserve the recovery age. -- Added explicit `microvmState` observations without changing existing coarse status/reason semantics, plus a policy helper for grace, useful sleep, pre-deadline wake, image/enable guards, missing/unreadable data, cancellation and delayed suspend-after-wake races. -- DynamoDB Local verifies actual transaction conditions, rollback, competing writers, changed identities, lost committed replies, cancellation/decision during recovery and a fresh module/client loading saved intent. See the [lifecycle runbook](./645-lifecycle-intent.md) for protocol and remaining integration/deployment gates. -- CDK lint/compilation passed. The broad handler/session-role run passed **158 suites / 3,738 tests**, including **23 lifecycle** and **15 existing capacity** DynamoDB Local tests. Five relevant suites passed **218 overlapping tests** and exited normally. The broad run exited successfully after a delay (about 72 seconds total versus 17.5 seconds reported test execution), with no open-handle trace; its cause is not established. Documentation sync, the **77-page** build and link checks pass. No Python source changed. The temporary local database was removed. -- At that foundation milestone, no production caller used this policy/store. No IAM grants, image hooks or automatic suspension were enabled. Durable poll failure/recovery tracking, guest barriers, supervisor/decision-handler wiring and live AWS gates remain unfinished. - -## Current completion checklist - -The implemented P3 sleep/wake backend has nine successful image 6.0 Durable -workflows and two focused image 7.0 approval/cancellation workflows. -Normal automatic sleep remains disabled. -The recent approval UX discussion adds product work that those lifecycle tests -do not cover: - -- [x] Implement the requested [nested MicroVM stack](./645-p3-nested-stack.md), - preserve parent execution-role identity and outputs, generate bootstrap 1.9.0, - and validate the bundled managed-image templates. The parent has 460 resources; - the MicroVM child has 19. AWS structural template validation passed. -- [x] Deploy bootstrap 1.9.0 and verify its exact installed policy, bundle hash, - and allowed/denied role names under the CloudFormation execution role. -- [x] Deploy an isolated fresh nested stack through that execution role, build - the managed image, verify hooks/roles/network rules and delete its resources. - The parent has three resources and the test child has 18; image 1.0 reached - `ACTIVE` / `SUCCESSFUL`. All 14 cleanup checks passed and normal resource - identities were preserved. This does not migrate the existing flat stack. -- [ ] Rehearse and review the resource migration, then - migrate the existing flat deployment and verify its image build/task path. - The existing deployment remains flat; nesting is part of the requested scope. - The isolated native image refactor passed preview but execution rejected the - resource's unsupported tag schema. Automatic rollback preserved the image and - all identities. A retain/import or explicit replacement procedure now needs - its own supported, reversible rehearsal; see the nested-stack record and F09. -- [ ] Implement the agreed human waiting policy: keep unanswered - requests available, separate the human decision window from the configurable - 600-second worker sleep delay, and retain an explicit timeout option. Today, - requests still default to 300 seconds with a maximum submission setting of - 3,600 seconds. Waiting beyond a worker's lifetime needs a defined recovery - path; current same-worker suspend/resume does not provide that continuation. -- [x] Implement and verify the [conversation/action checkpoint prerequisite](./645-p3-session-recovery-20260917.md). - The public SDK store restores approve/deny in a new process after removing the - old configuration. Immutable S3 version reads and task-scoped access passed nine - live checks; all temporary resources were removed. Full agent quality passed - 2,004 tests at that milestone. Workspace storage and production - worker-replacement wiring remain required before changing approval deadlines - or releasing a waiting worker. -- [x] Add the [local workspace archive](./645-p3-workspace-recovery-20260917.md): - preserve local commits, staged/unstaged edits and untracked/ignored files, - validate restoration without original Git config or hooks, and connect it to - the actual SDK approve/deny recovery test. Local Git ignore rules also survive. - Final agent quality passes 2,054 tests, including 50 workspace cases and both - SDK decisions. Durable workspace upload, saved - workflow context and replacement cloud-worker integration remain open. -- [x] Implement [actionable Slack/Linear notifications and CLI response - instructions](./645-p3-approval-ux-20260917.md), including saved decisions and - closure reasons. Unwrap the actual agent approval milestones and keep approval - messages out of Linear's terminal-reply path. Verify retry and receipt behavior. -- [x] Deploy the notification router, channel renderers and required table access. - Verify the deployed function packages against the reviewed S3 assets. -- [ ] Verify live channel delivery/response. Native Slack buttons, - Linear approval replies and the proposed per-user notification throttle remain - separate unfinished pieces; the implemented response path uses the signed-in CLI. -- [x] Atomically close pending approvals when REST/CLI or Slack cancels the task, - filter the pending list against owning task state, and stop the guest wait without - restoring `RUNNING`. Real DynamoDB transaction/race and notification-receipt checks - passed; all temporary tables were removed. -- [x] Deploy the cancellation/API changes. Deployed handler checks confirm atomic - cancellation, closure events, pending-list filtering and late-approval rejection; - eight consistent reads verify removal of all owned verification records. -- [x] Deploy the guest cancellation handling in image 7.0; preserve image 6.0, - all 475 physical resources, coordinator 10 and both disabled sleep switches. -- [x] Complete the image 7.0 approval/cancellation acceptance and cleanup. - Both Durable executions finalized without repair. Approve resumed PID 1 and - allowed one Read; cancel closed the request and terminated the sleeping worker. -- [x] Fix and deploy pending-list pagination so 100 cancelled legacy requests - cannot hide a live request on the next page. The real 101-request check passed; - all 202 temporary task/approval records were removed and verified absent. -- [x] Correct and deploy stranded/timeout/terminal-task notification feedback - and the stream retry cursor. The actual legacy stranded event, sequence-number - response and missing-sequence failure passed against the deployed function. - No channel messages were sent; both owned fixture rows were removed. -- [ ] Perform controlled activation on the normal deployment with the compatible - image/coordinator and the configurable 600-second sleep default. Verify an - ordinary submission through notification, decision, wake, continuation and - cleanup. For the current timeout-based implementation, use an explicitly - longer approval window: a default five-minute approval cannot reach a - ten-minute sleep delay. -- [ ] Verify the normal deployment's live off switch prevents new suspensions - while existing sleeping workers can still wake and finish. Retain compatible - coordinator versions, image and permissions. -- [ ] Finish the relevant comment/error-feedback cleanup, rollout documentation - and ADR/issue handoff, distinguishing shipped behavior from proposed approval - changes. The upload-dispatch feedback correction is now deployed and its - function package matches the reviewed S3 asset. - -The human waiting policy remains unimplemented; notification and cancellation -API changes are deployed, with guest handling and channel acceptance tracked above. -A separate approval relevance/staleness rechecking system is excluded by the -user's scope decision at the top of this plan. - -### Unanswered approvals: implementation order - -The product direction is agreed: unanswered requests stay available; a short -expiry is an explicit option; the worker's sleep delay defaults to 600 seconds. -The following work is still required before changing the current deadline default: - -1. Prove recovery of a paused agent on a **replacement worker**. The current - MicroVM checkpoint records task/gate identity and lifecycle acknowledgements; - it does not export the workspace or the SDK conversation. `runner.py` records - the SDK session ID in its result but does not currently start a resumed SDK - session. First test the SDK's supported recovery behavior at a pending tool - hook and define how the saved decision reaches the agent's normal reasoning. - The [local pinned-SDK recovery probe](./645-p3-session-recovery-20260917.md) - now verifies conversation recovery through the public `SessionStore` after - abrupt process loss and deletion of the old configuration. The pending call - is omitted from the restored model context; the checkpoint carries its full - action separately. A continuation prompt transfers that action and decision, - and the newly proposed call receives a new ID and hook. Local approve/deny - pass; replacement cloud-worker recovery remains required. -2. Add a durable continuation checkpoint: workspace changes, conversation/session - state and exact pending request identity. Store it under task-scoped - permissions, exclude credentials, and confirm the write before releasing the - worker. A failed checkpoint must produce actionable feedback, never a claim - that the work was saved. Test restoration of uncommitted and untracked files. - `agent/src/continuation_session.py` now implements acknowledged conversation/ - action storage with immutable S3 version receipts; nine live AWS permission - and persistence checks passed. `continuation_workspace.py` now preserves the - complete supported working tree and Git state in a bounded local archive. - Its durable upload/read-back, saved workflow context, production IAM/retention, - lifecycle-barrier integration and conditional receipt publication remain open. -3. Separate the task from each worker attempt. Add an attempt generation to - launch tokens, saved handles, writes and reservations so a replaced worker - cannot overwrite its successor. Release capacity while the task waits and - reacquire it on continuation. Duplicate decisions, lost launch replies, - cancellation and coordinator replay must not launch two workers or release - another attempt's reservation. -4. Let an open request have no decision deadline. Keep the explicit timeout - setting, retain existing finite deadlines for already-running tasks and remove - retention TTLs from unanswered requests. Start retention when a request - closes. Update validation, types, pending responses, CLI displays, notifications - and metrics together. The 600-second sleep setting remains independent. -5. Replace worker-lifetime failure with checkpoint/release for this supported - waiting state. The eight-hour service limit still applies to each worker; - the task and unanswered request survive it. Update the stranded-task - reconciler's current two-hour approval backstop so it does not fail a - deliberately parked task. A reply starts continuation if the old worker is - gone; the existing worker resumes when it is still usable. -6. Verify short sleep, a reply after the old five-minute deadline, reply after - worker replacement, explicit timeout, cancel while parked, simultaneous - replies, lost checkpoint/launch replies and permission failures. Confirm - single decision consumption, preserved files, bounded worker cost, accurate - feedback and cleanup. Then change the default and enable it through a - reviewed compatible rollout. - -These are recovery and resource-accounting requirements. They do not introduce a -separate system to judge whether the proposed action is still relevant; the agent -continues to make that judgment. - -Service token-retention/unknown-worker recovery and installation-specific legacy -capacity migration remain broader backend follow-ups. They are not automatically -prerequisites for enabling the existing sleep implementation on this clean, -compatible installation. The user has separately requested nested infrastructure. - -## Verification evidence and broader follow-ups - -Keep service questions and evidence in the -[Lambda MicroVM service-team feedback tracker](./645-lambda-microvm-service-feedback.md). - -The image `4.0` long-sleep acceptance and verification-infrastructure cleanup are -complete. The original approval timed out correctly, real credentials renewed -after expiry, and coordinator cleanup passed without watcher repair. The -[live record](./645-p3-callback-live-20260915.md) records the evidence and verified -absence of all temporary infrastructure. - -The detailed batches below preserve the implementation history and the limits of -the recorded evidence. Use the current checklist above for the next work. - -1. The wider final-image lifecycle matrix on image 6.0 is now complete for its - recorded scope: a temporary repository clone with mutable files across two - approval gates, the late-decision winner, and wake after actual credential - expiry with CloudTrail renewal proof. See the - [final-image follow-up](./645-p3-final-image-and-ecs-20260917.md). - Earlier image 4.0/5.0 results remain evidence for their recorded scope. - The [connection-close correction](./645-p3-wake-transport-20260916.md) and - [normal rollout](./645-p3-connection-close-rollout-20260916.md) are complete. - Three instrumented candidates proved socket closure before freeze and a new - resume connection; four normal-image Durable workflows passed approval, - denial, the original timeout winning, and cancellation while asleep. - Coordinator 10 and image 6.0 are deployed with both sleep gates off. - Retain the six [exact refusals](./645-p3-resume-refusal-investigation.md) and - F08, and obtain service-side dispatch details separately. The guest evidence - does not reveal the actual service error in every historical case. - The [lifecycle diagnostics guide](./645-p3-lifecycle-diagnostics.md) describes - the deployed hook-stage, AWS request-ID and durable state-change logging, - including the distinction between an accepted Resume and a successful hook. -2. The [command-race checks](./645-p3-command-races-20260916.md) now pass cancellation - before/after Suspend, during observed `SUSPENDING`, and during restore - `PENDING`; approval during an accepted suspension and three consecutive - polling failures also pass. Six required cases used nine workers, with three - harness-invalidated attempts explicitly excluded and replaced. All temporary - resources were removed. The same follow-up deployed clearer status-read - failure guidance in coordinator version 7 and verified the normal task API. - Those cases used image 5.0; disabled suspension settings remain in place. - Complete the remaining live fault matrix: - service token-retention expiry and recovery - when guest identity evidence is unavailable. Keep each injected failure distinct from an - unrelated service failure. - The [adjustable-sleep follow-up](./645-p3-user-sleep-20260916.md) completed the - late-approval winner and actual credential-refresh denial checks, plus default, - off and custom delays. It deployed coordinator version 8 with both gates off. - The timeout-winner case was interrupted by the fifth unexplained wake refusal - and remained unaccepted in that run. A corrected timeout-winning case now - passes on the private PID 1 diagnostic image with unchanged application code, - the original deadline, late HTTP 404 rejection and automatic cleanup. - The diagnostic also reproduced a distinct generic wake failure, so this - individual passing case did not complete final-image enablement. The timeout - winner now also passes on normal image 6.0 through real Durable execution, - with a late HTTP 404 and no unapproved Read. The wider image 6.0 matrix in - step 1 has subsequently passed too. - The [durable registration follow-up](./645-p3-registration-20260916.md) - passed a lost reply after an actual registration commit, cancellation before - identity registration, and explicit operator recovery/termination of a live - worker whose ID never reached coordinator state. The latter required an - exact task/worker pair in the guest log; it does not establish post-retention - behavior or a recovery path when those logs are unavailable. -3. Runtime networking and remote-MCP connectivity now pass the - [September 17 independent audit](./645-p3-mcp-network-20260917.md): - real documentation-tool calls before/after a 60-second sleep, HTTPS success - versus port-80 timeout on the same IP, and unauthenticated endpoint 403s while - the worker was running. This adds the ninth successful image 6.0 Durable - workflow. The record retains the excluded discovery-policy attempt and - watcher bookkeeping failure; product finalization preceded its fallback. - Valid-token ingress and exact TCP socket reuse were not tested. - The [AgentCore permission probe](./645-p3-agentcore-permissions-20260917.md) - now passes 19 actual requests on normal runtime version 5, including ambient - denial, scoped task access and protected-field denials. The - [ECS follow-up](./645-p3-final-image-and-ecs-20260917.md) now verifies real - ambient/scoped permission requests, port 443 success versus port 80 denial, - approval/cancellation, payload deletion and stopped workers on the current - shared container. Its source fix supplies the missing approval-table name - to both ECS task definitions. Trace and nudge environment parity are separate - existing ECS gaps; they were not silently supplied by the fixture. - The final-image repository test proves mutable files across two sleeps - using an explicit temporary clone inside an artifact task. The subsequent - [normal repository check](./645-p3-repository-path-20260917.md) also passes - clone/setup, approval sleep/wake, build/lint and existing-PR resolution. - New-PR publication remains covered by the earlier P2 record for its stated - image/scope; this final-image fixture deliberately publishes nothing. - The [AgentCore follow-up](./645-p3-agentcore-20260916.md) now verifies its - current shared container, approval/cancellation, exclusion from MicroVM sleep, - reservation release and owned-session cleanup. The normal stack has no ECS - resources; the separate bounded ECS deployment has now been tested and removed. -4. Exercise the coordinated capacity upgrade/drain and rollback procedure under - deployed writer roles, including realistic scan volume. Retain the verified - local transaction and isolated-live results as evidence for their narrower - scope. - The [isolated protocol rehearsal](./645-p3-capacity-upgrade-20260916.md) now - passes enforced admission pauses, old/current drains, rollback and re-upgrade. - Its private helper entry points and fixture-managed task states do not replace - the normal deployment's complete admission-route/durable-execution drain. - The [AWS scan follow-up](./645-p3-capacity-scan-20260916.md) now passes a - 600-user fixture with real multi-page reads, exact normal Lambda artifact, - equivalent table permissions, zero writes after interrupted scans, and - conservative handling until an older task settles. Its temporary tables and - function were removed. This bounds the verified volume without claiming the - production migration or arbitrary retention scale. A subsequent measurement - found 86 task rows and 36 counter rows in the normal development deployment, - below the fixture's 600 rows per table. - The [September 17 normal-rollout review](./645-p3-normal-rollout-review-20260917.md) - now inventories all nine actual producers and retained coordinator versions - 2–10. It found 112 terminal tasks, no held reservations and 36 zero counters. - The clean installation already used the current reservation protocol. - Admission was not fenced: its async processors lack failure retention, so - setting their concurrency to zero could lose incoming work. A safe retained - input/replay procedure remains necessary. Coordinator rollback also requires - an explicit image plan because the managed runtime selects latest active. -5. Complete the compatible rollout and normal activation checks listed above, - retaining the required runtime versions and rollback options. Record shared - ECS/AgentCore changes for installations that use them. Keep the broader - service and legacy-migration follow-ups separately visible in the handoff. - -Nested CloudFormation stacks are now requested. The last deployed flat root had -475 resources; moving its existing resources requires a reviewed migration. - -## The result we want - -When the coding agent asks a human for permission, its computer may go to sleep. The human can approve or deny while it sleeps. The computer wakes, reads the saved answer, and continues or refuses the action. If nobody answers, it wakes before the deadline and applies the existing timeout-as-denial rule. Its files, task identity, permissions, progress and deadline remain correct. - -**HITL** means “human in the loop.” **Cedar** is the policy system that decides which actions need permission. **DynamoDB** is where ABCA stores task and approval records. A **transaction** changes/checks related records together so competing actions cannot half-win. A **strongly consistent read** asks DynamoDB for the latest committed value. **Idempotent** means repeating an operation has the same intended effect as doing it once. A **reconciler** repeatedly compares what should be happening with what is actually happening and repairs differences. - -P3 coordinates the supervisor, approval records and the sleeping computer. - -For example, with a thirty-minute approval window and the default ten-minute sleep delay: the question appears at **12:00**; after **12:10** the VM can sleep. If approval arrives at **12:15**, ABCA wakes it and reads the saved answer. If nobody answers, ABCA wakes it around **12:29** so the agent can deny at **12:30**. Waking must not start a new timer. Default five-minute approval windows stay awake. Users can choose a different delay or disable task sleep. - -A few implementation words used below: - -| Term | Meaning here | -|---|---| -| API / SDK | An API is a service's set of commands; an SDK is a library for sending them from code. | -| Bootstrap | Install the deployment's foundation permissions and storage before deploying the application. | -| ARN | An AWS resource's full address, such as the address of one permission role or image. | -| Monotonic clock / wall clock | A stopwatch measuring elapsed time versus a clock showing the date/time. Sleep can affect them differently, so the saved deadline must still count. | -| Durable | Saved outside the VM, so putting the VM to sleep does not lose it. A failed save is not durable. | -| Barrier | A controlled door: coding cannot proceed until the required saves or credential refresh have succeeded. | -| Orphan | A VM that still exists but whose normal supervisor/wake-up path has lost track of it. | -| Replay / race | Replay repeats earlier work after recovery. A race happens when two actions, such as approval and cancellation, arrive almost together. | -| Regression test / fault injection | A test that prevents a known bug returning / a test that deliberately simulates a failure. | -| Change set / rollback | AWS's preview of deployment changes / the procedure to restore the previous working deployment. | - -## Fixed boundaries - -- Keep the eight-hour `maximumDurationInSeconds = 28800`, including suspended time. -- Keep `idlePolicy` absent. Traffic-based idleness would mistake outbound-only coding work for inactivity. -- Keep explicit `NO_INGRESS` and no `CreateMicrovmAuthToken` grant. No public agent-control endpoint or JWE token refresh system is needed for P3. -- Keep the 4,096-byte serialized hook-reference boundary and v2 S3 transport for every task, including registry-resolved assets. -- Resume the restored task context; do not repeat bootstrap or reuse its short-lived launch URL after sleep. -- Only the orchestrator initiates suspension. Approval handlers may request resume after committing a decision; the orchestrator repairs missed resumes. -- The agent remains the authority that consumes a decision or times out a gate. Approval HTTP handlers must not simply mark the coding task RUNNING. -- Preserve tenant-scoped credentials, task/user/repository tags and fail-closed behavior. “Fail closed” means refusing an action when safe authorization cannot be established. -- Keep the ABCA concurrency slot while suspended; release it exactly once on terminal finalization. Suspended AWS memory-quota consumption remains unverified, not the justification for a new claim. -- General progress-based hang detection (#491), operator shell access and extended/off-hours approval windows are outside P3. Coordinate compatible seams without making those projects hard dependencies. - -## Delivery order - -| Work package | Depends on | Exit condition | -|---|---|---| -| 0. Freeze evidence and establish test baseline | Current review | Known baseline, reproducible checks and issue-to-test map | -| 1. Repair P2 correctness/security prerequisites | 0 | Critical defects have regression tests and are fixed | -| 2. Optional infrastructure nesting | 0; finish before final deployment validation if chosen | Safe boundary, current feature matrix, bootstrap/migration verified | -| 3. Prove clean P2 deployment | 1 and any chosen nesting changes | Managed image and task complete with no manual IAM workaround | -| 4. Define lifecycle contract and widen all strategies | 0; can develop alongside 1–3 | Typed interface, state/race contract and focused tests | -| 5. Implement safe agent hooks and clocks | 1's test/liveness fixes, 4 | Hooks, refresh barrier, deadlines and failure paths tested | -| 6. Add orchestrator and approval wake-up | 4–5 | Race-safe integration, bounded recovery, scoped grants | -| 7. Live P3 verification and completion | 3, 5–6 | Full acceptance matrix passes with retained evidence | - -Keep nesting in a separate change from lifecycle logic. Developing P3 locally need not wait for every live prerequisite, but claiming it complete does. - -## 0. Establish the baseline - -1. Record the source commit, dependency lockfiles, bootstrap bundle version, image version and enabled context flags. Track #645, #817, #700, #818, #841 and #857 with concrete remaining checkboxes. Do not reopen consolidated #813–816 as if they represented four separate finished fixes. -2. Land/review this comment cleanup independently. It deliberately does not repair runtime behavior. -3. **Completed locally — #841:** every tracked pipeline thread is joined before mocks/environment are restored; teardown reports surviving threads without discarding their handles. A deterministic subprocess regression verifies isolation across successive tests. -4. Run focused baseline suites for server hooks, task-state approvals, credentials, strategy, orchestrator, approval handlers, construct IAM and bootstrap coverage. Record existing unrelated failures rather than quietly weakening assertions or thresholds. -5. Make the phase vocabulary explicit: this is P3 of ADR-021. Use different names for PR/work-package sequencing so “P1 priority” on an issue is not confused with phase P1. - -## 1. Repair the prerequisites - -### 1A. Finish #817 - -- [x] **Deletion IAM:** the coordinator has exact `s3:DeleteObject` on `*/payload.json` and `*/launch.json` in the dedicated bucket. Construct and stack tests pin that scope. Worker ambient reads are now restricted to bootstrap manifests; deletion failure does not hide the task outcome. -- [x] **Classifier:** reconciliation persists `MICROVM_SUBSTRATE_TERMINATED` or `MICROVM_RUN_HOOK_REJECTED` separately from the descriptive AWS reason. The known leading run-hook 4xx response selects the latter; arbitrary appended text cannot override the code. Legacy task records remain readable. Tests cover task API classification, channel/panel retry guidance, host/capacity failures, regional faults, authorization/configuration, concurrency words, hook 400/500 and other backends. -- [x] **Trusted configuration (local):** v2 reads a deployment manifest with ambient IAM credentials restricted by an explicit deny outside that deployment's bootstrap prefix, then verifies downloaded task/config equality. Another workspace's secret in the same account is rejected; an exact trusted cross-region secret is preserved. Both languages share the version/caps. Live effective-IAM, public-bucket and ingress negatives remain open. -- [x] **Contract guards:** pin exact contract fields, ARN fields and the account anchor in both languages. Python import-time validation and the constants checker reject newly added `*_arn` / `*_ARN` fields omitted from `arn_keys`. This prevents accidental validation gaps; it does not establish deployment identity. -- [x] **S3 bad bytes:** real `StreamingBody` tests cover truncated JSON, invalid encoding, short and closed streams, and non-object JSON through fetch/decode/envelope/route. They assert a structured unreadable response, no configuration installation/environment mutation, and no pipeline thread. A closed-stream `ValueError` now reaches the unreadable-payload 500 branch. Malformed hook envelopes retain the separate 400 response. S3 writes are atomic; invalid stored bytes need replacement, not an assumption that another identical read repairs them. -- **Documentation:** verify all remaining contract/status changes update source docs and their generated copies through the sync script. - -### 1B. Narrow payload reads (#700) - -**Completed locally for ECS and MicroVM:** every task uses an IAM-authenticated deployment manifest plus a short-lived, single-object signed URL. Worker ambient roles explicitly deny other object reads and payload-bucket listing. The coordinator stores the exact URL privately in S3 for replay, conditionally creates task objects, rejects changed/expired launches, and deletes payload plus launch record at finalization. ECS consumes/removes its capability before the pipeline; errors and Python exception chains must not expose it. - -The [bootstrap runbook](./645-payload-bootstrap.md) records wire/storage shapes, caps, credential lifetime, the coordinator's `ListBucket` requirement for missing-object detection, coordinated drain/image/controller/policy upgrade and rollback, and the real AWS allow/deny matrix. Old unsigned envelopes are intentionally rejected; this is a coordinated contract change, not a rolling mixed-version deployment. - -**Live subset completed:** [image 2.0 probes](./645-p2-payload-live-20260914.md) -verify MicroVM manifest/download access through the runtime connector, invalid -task/config/path/bytes/signature rejection, URL expiry/revocation, foreign -private-bucket denial and >1 MiB transport. Concurrent/repeated preparation and -changed-input conflicts exercise real S3 with operator credentials. -The [start/recovery follow-up](./645-p2-start-recovery-live-20260914.md) -also verifies recovery after committed S3 replies are lost or the local process -exits after payload/launch writes. - -**Still required:** effective-role cross-task/list/write/public-bucket negatives, -expired signer credentials, recovery under the deployed coordinator role and -durable execution, and equivalent ECS/upgrade evidence. The role/tag and other -platform-grant limits in 1G remain; this boot-path fix does not establish complete -hostile-worker isolation. - -### 1C. Fix approval/heartbeat ordering - -**Completed locally:** `task_state.transact_resume_from_approval` refreshes `agent_heartbeat_at` in the **same conditional update** that restores RUNNING. The expected status and `awaiting_approval_request_id` conditions remain in place. - -Regression coverage includes a task that waits over 240 seconds and resumes before the next heartbeat tick; immediate polling remains healthy for AgentCore and MicroVM. Existing cancellation/wrong-request conditions and ECS behavior are preserved. This is approval-state coverage, not yet a live frozen-MicroVM test. - -### 1D. Make session-start retries honest - -**Implemented locally:** `microvm-start.ts` conditionally records one start per task, using the task ID as its stable token. The internal `microvm_start` attribute stores the request fingerprint, creation time, a 120-second local replay deadline and any recovered handle. The fingerprint includes the full S3 payload, so a changed retry cannot overwrite the first task's instructions. It is checked before uploads and again before `RunMicrovm`. A new attempt requires a new task ID. - -The receipt works like an order number: when the reply gets lost, the next call asks about the same order. - -Fault tests exercise a successful simulated service creation followed by a lost response, a second application call with the same token, a fresh strategy instance, saved-handle replay, changed input, expired recovery, confirmed rejection, cancellation before/during/after creation, and lost DynamoDB responses. The handler recovers committed registration, treats start-audit failures as non-fatal, and routes start failures through one finalization path. Finalization reads the latest committed task, avoiding a stale cancellation/failure report. An unknown first outcome stays unknown even when the second call gets a definite rejection; HTTP 408 and named service timeouts remain uncertain even with a 4xx status. - -**Live subset completed:** [service and application recovery probes](./645-p2-start-recovery-live-20260914.md) -extend immediate replay with simultaneous identical requests, changed-parameter -rejection and terminated-worker replay through roughly five minutes. Production -strategy/storage code recovered lost successful replies and local process death -using real AWS; it also reused saved handles without Run, refused changed input -and cancellation, and stopped replay after its actual 120-second deadline. -These use operator credentials and do not kill the deployed durable Lambda. - -**Still required:** the installed SDK documents `clientToken` idempotency but gives no retention period. Public AWS API documentation URLs did not provide a usable RunMicrovm reference during this review. The local 120-second limit is a conservative application cutoff; roughly five minutes of observed AWS replay does not establish maximum retention or post-expiry behavior. Verify deployed durable-Lambda interruption, registration/cancellation races and operator recovery of an unknown worker ID before accepting this prerequisite. Cached `RunMicrovm` replay responses can still say `PENDING` after the worker terminated; use `GetMicrovm` for current state. - -For an unknown outcome with no returned ID, the task error or cancellation event identifies the saved token for investigation. Do not automatically submit a replacement task. Verify how operators find and terminate that VM in the deployed service; if they cannot recover an ID, the eight-hour lifetime cap is the remaining bound. Keep this limitation explicit in live evidence. - -### 1E. Make verification observable - -**Completed locally:** #810 uses the removal option. Both CloudWatch writers emit `cloudwatch_write_failed` to stdout with `writer`, `task_id` and `error_type`; the unused counter and threshold are gone. Six injected failures cover client creation, stream creation and event submission without recursive logging or sensitive error contents. This is a structured log, not an alarm or metric. Verify guest stdout ingestion while the VM is running; AgentCore APPLICATION_LOGS does not automatically collect it. - -For #818, documentation explicitly limits remote MCP endpoints to reachable HTTPS/443 on all shipped backends. No synth/onboarding connectivity probe is added, and no ports are widened. Registry-specific integration tests exercise real resolution/hydration, audit writes, >4,096-byte asset data in S3, the bounded v2 reference, Python hook mapping and the real local loader. The retired inline branch is no longer the acceptance target. Successful config delivery does not prove remote-tool connectivity; test DNS/routing/TLS/auth live. - -### 1F. Make concurrency release safe across replay - -**Implemented locally:** `task-concurrency.ts` saves a per-task reservation together with the user counter change in one transaction. Repeated acquisition reuses the held reservation; release requires a terminal task and changes `held → released` atomically with the decrement. Missing/released markers never decrement another task's count. This covers cooperating platform writers and crash replay. - -The regression reproduced two finalizer executions reducing two occupied seats to zero, instead of leaving the other task's seat occupied. DynamoDB Local tests now exercise real transaction conditions for that replay, concurrent admissions/finalizers, lost committed responses, early failure, cancellation, approval waits, empty counters and a racing new admission. The normal finalizer, early failure path and stranded cleaner share release. Upload confirmation only submits and reads capacity; the orchestrator reserves once. Queue restoration cannot requeue a task that acquired a reservation after an uncertain invoke. - -Every counter change carries a fresh revision. Scheduled repair strongly scans the base counter/task tables, compares the saved revision before replacing a count, and completes abandoned terminal releases. It includes approval waits. An increment/decrement with no net count change still invalidates an old scan. Incomplete scans install no partial result. Ambiguous older active tasks prevent guessing a replacement count. - -**Deployment gate:** pause admissions and drain old executions, update all counter writers and the reconciler's scoped task-update permission, reconcile after legacy tasks settle, then reopen admissions. Rollback also requires draining. Verify scan duration/read capacity at deployment scale and the mixed-version/drain procedure in AWS. The local simulator does not prove deployed IAM or cloud-scale behavior. See [capacity verification](./645-capacity-reservations.md). - -### 1G. Protect coordinator-owned metadata - -**Implemented locally:** the main task table now permits own-task reads and only `UpdateItem` on an explicit reporting/approval attribute list. Replacement/deletion, start receipts, capacity markers, owner identity and compute handles are excluded. Supporting tables retain their task-scoped access. Missing IAM context keys fail closed. ECS's legacy direct-grant path uses the same attribute restriction; AgentCore/ECS no longer receive the unused shared-counter grant, and MicroVM never had it. - -The writer inventory found no production callers of Python `write_submitted` or `write_session_info`; those unused helpers and stale comments are removed. Contract tests exercise every current main-task writer, including complete terminal results and approval transactions, against the JSON attribute list used by CDK. Construct/stack tests inspect the actual grants across all three backends and prevent the main table from being supplied as an unrestricted supporting table. - -**Deployment gate:** run the allowed/denied request matrix in [metadata verification](./645-coordinator-metadata.md) using real scoped and ambient credentials, including aliased/nested updates, replacement/deletion and transactions that mix forbidden task writes with valid approval writes. Inspect effective policies and preserve coordinator start/finalize/cancel behavior. This requires no new table or bootstrap policy change, but it does require application-role deployment and matching image verification. - -**Remaining trust limits:** agent status/results remain reports from the agent. Compute roles choose their session tags; existing trust does not independently bind those choices to a task. This patch protects coordinator attributes from the resulting session permissions; it does not establish complete hostile-worker tenant isolation. The replay tests in 1D/1F also do not prove AWS authorization. - -## 2. Nest infrastructure (requested) - -1. Introduce `LambdaMicrovmStack` as a `NestedStack` wrapper and an explicit way to supply the MicroVM execution role from the parent. Keep shared session-role trust and runtime-role ownership in the parent. Preserve exact grant behavior. -2. Pass a stable deployment name for image and both connector names. Never sanitize unresolved CDK tokens. Preserve name/ARN resolution for imported images as well as managed images. -3. Return artifact/payload bucket, image, connector and build-role identifiers to existing parent consumers. Keep parent `Microvm*` outputs unchanged so packaging/CLI discovery still works. Check no-image bootstrap state and imported-image state, not just the managed-image case. -4. Inspect all generated role names against bootstrap allowlists. If permissions/names require a bundle update, regenerate artifacts, apply the repository's version rule, and run policy/golden/artifact-sync tests. Keep `PassRole` exact or narrowly prefixed; no broad all-roles shortcut or resurrected broken service condition. -5. Update assertions to inspect every child template. Assert no cycles with `Template.fromStack`, correct tags and solution user-agent propagation, supported region checks, hook values and runtime/build network separation. -6. Synthesize real bundled configurations: default, ECS, MicroVM without image, imported image and managed image; cross the relevant gateway/vault flags; validate the heaviest configuration and custom naming paths. Enforce each stack's resource/byte limits, parameter/output limits and the deployment's nested-operation limits. Do not require the numbers in this review to stay constant forever. -7. Remove or replace the #857 guard only when the supported combination actually passes the current matrix. Keep a size regression test; do not retain “505 resources” as a permanent error message. -8. Review the CloudFormation change set for replacement/deletion. For the first experimental rollout, prefer a dedicated test deployment. For migration of an existing one, stop new admissions, drain/terminate existing tasks, preserve required artifacts/state and use a service-supported migration procedure. Define rollback before applying the change set. Do not let `autoDeleteObjects` silently erase an active payload/artifact bucket. - -**Implementation status:** the wrapper, parent execution role, stable names, -preserved outputs, bootstrap 1.9.0 artifacts and child-template assertions are -implemented. Production managed-image synthesis and AWS structural template -validation passed. The unit budget matrix covers AgentCore, ECS and all three -MicroVM image modes with gateway on/off. The #857 guard remains. - -**Completion requires:** the remaining bundled deployment combinations, safe -migration/change-set evidence, installation of bootstrap 1.9.0, and a clean -deployed image build. The normal deployment has not been migrated. See the -[nested-stack verification record](./645-p3-nested-stack.md). - -## 3. Re-prove P2 on the final infrastructure - -Use a supported Region and an isolated development repository/account deployment. Record the actual deployed bootstrap bundle (at least 1.9.0 for the nested layout; 1.8.0 for the explicit flat compatibility layout). Compare effective policies as well as the displayed version, including exact-self CloudFormation PassRole, scoped live-suspension parameter permissions and the nested build/operator role names. Update bootstrap deliberately when required; a command that skips an already bootstrapped stack is not evidence of refresh. - -The [2026-09-13–14 clean deployment](./645-p2-clean-deployment-20260913.md) -completed the infrastructure, managed-image creation and build-hook checks in -steps 1–2 below. The [live task follow-up](./645-p2-live-task-20260914.md) -provides positive runtime evidence for steps 3–5 and success/cancellation cleanup -in step 6. The [configuration follow-up](./645-p2-repository-config-20260914.md) -also verifies automatic pre/post npm checks and cleanup after a worker-reported -delivery failure. The [image 2.0 payload follow-up](./645-p2-payload-live-20260914.md) -verifies direct startup-hook transport/rejections, immediate Run replay and -operator cleanup. The [recovery follow-up](./645-p2-start-recovery-live-20260914.md) -adds service replay through five minutes and real-storage application tests with -local process/reply faults. Deployed coordinator classification/finalization -after rejected hooks, durable restart/cleanup-error paths and the wider IAM/network matrix -remain unverified. -`isadeks/vercel-abca-linear` explicitly selects `lambda-microvm`; the original -seeded repository retains AgentCore. - -Follow `cdk/scripts/package-microvm-artifact.sh` and the P1/P2 runbooks: - -1. Deploy the no-image infrastructure if its artifact bucket does not exist. Package/upload the current source; create the managed image through CloudFormation with a pinned base image/version. Check `ARM_64`, hook `ENABLED` values and separate build/runtime connectors. Do not substitute the manual image path when validating the formerly broken managed path. -2. Verify `/ready` local warm-up and `/validate` self-check; no AWS credential initialization from build hooks. Review credential warnings without recording secret values. -3. Run a normal task through clone → change → tests → commit/push → PR. Observe `bgagent watch`, task details/list heartbeat, structured application logs and final task outcome. -4. Exercise the Memory path and any features claimed as parity. Absence of an error log alone is not proof a Memory write happened. -5. While the VM is **RUNNING**, verify logging with the narrowed runtime role, including whether removing CreateLogGroup still permits all required streams. Checking after teardown is not equivalent. -6. Verify payload deletion and active termination after success, failure and cancellation. Record service states, request IDs and timing. Negative-test public ingress and runtime port restrictions from suitable test locations. -7. No ad hoc IAM edits, console workarounds or privileged fallback. If one is needed, fix source/bootstrap and repeat the affected acceptance case on that final version. - -**Done:** the remaining P2 warning conditions are discharged by a dated runbook. Update warning text only to match the evidence; do not imply that P3 is finished. - -## 4. Define the P3 contract before wiring consumers - -### Strategy methods - -**Implemented locally:** mandatory `suspendSession(handle)` and `resumeSession(handle)` were added to `ComputeStrategy` together with all three implementations. `SessionLifecycleResult` is `{ supported: false } | { supported: true }`; true means the command was acknowledged, not that the VM reached its final state. Operational failures throw. - -AgentCore and ECS return explicit unsupported results without an AWS request. MicroVM issues `SuspendMicrovm`/`ResumeMicrovm` with `microvmIdentifier: handle.microvmId` and a 10-second abort bound. Local tests cover wrong/empty handles, repeated requests, simulated conflicts/not-found and retriable/permanent errors. **Still required:** verify actual AWS behavior for already-target-state/terminated VMs and races before normalizing any conflict into success. - -The installed SDK returns empty suspend/resume responses. Its observed states are `PENDING`, `RUNNING`, `SUSPENDING`, `SUSPENDED`, `TERMINATING` and `TERMINATED`; there is no `RESUMING` value. The coarse poll mapping groups PENDING/unknown with running and SUSPENDING with suspended. **Implemented locally:** `SessionStatus.microvmState` carries explicit state, including local `UNKNOWN`/`NOT_FOUND` observations, while preserving existing coarse status/reason semantics. The policy uses this field; the coarse `running` result alone cannot prove wake completion. `reason` remains diagnostic. - -### Durable intent and policy - -**Implemented locally:** `microvm-lifecycle.ts` stores a typed optional record on the existing task row, separate from `compute_metadata`. It includes format version, generation, VM/gate identity, desired action, original timestamp and deadline. Conditions guard owner/status/gate/handle/generation; suspend atomically checks the same PENDING approval row and unchanged deadline inputs. A wake cannot become a sleep for that gate, and retaining the record prevents an older absent-record snapshot from recreating a sleep intent. A new gate may establish a new generation. No credentials or bearer URLs are stored. - -**Implemented locally:** the durable supervisor persists failure counters, anomaly episodes, next-poll delay and fixed recovery/session deadlines. Lambda replay does not reset them. Repeated intent saves already retain the original timestamp/generation. The record is internal and has no public task API field. Database success is not a lock over a later AWS command: reread before suspend and reconcile after every command outcome. - -**Connected to the durable supervisor:** the policy combines task status, the **specific current approval row's status**, desired action, explicit VM state and current time. PENDING alone is not a reason to resume. All terminal approval states, deadline proximity, missing/unreadable data or unintended suspension can require wake. SUSPENDING records desired wake but returns `requestReady: false` until SUSPENDED is observed. - -Current local policy values: a per-task suspend delay of 600 seconds by default (`microvm_sleep_after_s`, 0–3600 seconds, zero disables sleep), 60-second pre-deadline wake margin, 30-second minimum available sleep window and at most 5-second transition polling. The earlier 30-second grace remains explicit in historical verification fixtures; it is no longer the application default. Long intervals are clamped to the relevant grace/wake/session deadline. Snapshot costs and actual wait distributions still need measurement; the 30-second available window does not promise financial savings. The supervisor escalates after three consecutive failed cycles and bounds the entire cycle to 45 seconds, including the store's 5-second request/read-sequence budgets and lost-reply recovery. Wake/unknown recovery is bounded to 120 seconds and startup to 300 seconds; see the supervisor runbook for the complete budgets. - -### State/action table - -| Task and gate | Observed VM | Action | -|---|---|---| -| RUNNING, doing work | RUNNING | Check liveness; never suspend based on lack of inbound traffic | -| AWAITING_APPROVAL, current gate PENDING, before grace or too near deadline | RUNNING | Keep awake | -| Same, grace passed and sufficient remaining time | RUNNING | Conditionally record suspend intent, recheck gate, request suspend | -| PENDING, intentionally suspended, ample time remains | SUSPENDING/SUSPENDED | Wait; do not mark the stopped heartbeat as a crash | -| Current gate APPROVED/DENIED | SUSPENDING/SUSPENDED | Request/reschedule resume; let agent consume the committed answer | -| PENDING at deadline minus margin | SUSPENDING/SUSPENDED | Resume for agent-side expiry evaluation | -| RUNNING or a different gate, unexpectedly suspended | SUSPENDING/SUSPENDED | Emit one anomaly per episode and perform bounded recovery; do not label suspension itself a crash | -| Task terminal/cancelled | Any live or suspended state | Terminate; never resurrect task state | -| VM terminal, task nonterminal | Terminal/not found | Re-read task consistently, apply substrate-failure reconciliation, finalize once | -| Gate row missing or unreadable while asleep | SUSPENDED | Wake conservatively when possible; preserve existing missing-row/timeout safeguards and bounded API-error handling | - -## 5. Implement the agent's suspend/resume safety - -Files: `agent/src/server.py`, `hooks.py`, `task_state.py`, `aws_session.py`, `progress_writer.py`, credential-helper/cache owners, `contracts/constants.json`, and focused tests. - -**Guest controller implemented locally (2026-09-14):** `microvm_lifecycle.py` -now registers the MicroVM task, tracks SDK tool completion (including failures), -parks the exact approval and original deadline, and drains approval reads, -heartbeat calls and progress writes before a supplied checkpoint callback. -Every progress writer shares the task's acknowledgment history; a dropped event -keeps suspension disabled even after later successful writes. Failed or timed-out -wake callbacks cannot release coding through a late thread completion. The -controller and `/run` reseed `random` using fresh OS entropy. - -The [guest barrier review](./645-p3-guest-barrier.md) records that controller -milestone. The subsequent [HTTP hook implementation](./645-p3-lifecycle-hooks.md) -now supplies production checkpoint and refresh/reconciliation callbacks. No image -capability, new IAM grants or automatic sleep were enabled at that guest-barrier -milestone. The later image, supervisor and live-verification milestones above -record their implementation and deployment. - -**Credential implementation added locally (2026-09-14):** the -[credential verification](./645-p3-credentials.md) records actual pinned-CLI -expiry/failure probes, the MicroVM-only scoped loopback provider, retained -ambient/tenant refresh and SDK/broker teardown. Static/unknown runtime providers -fail closed; their real renewal path must be verified before enabling suspension. -The production HTTP resume callback now invokes this refresh before any AWS reads. - -1. **Implemented locally:** a per-task context holds task/VM identity, active approval and the original deadline. MicroVM background startup registers it; pipeline exit/crash unregisters it. The SDK hook removes the approval safe point before changing task state or returning a permission result. HTTP handlers use this context, cache completed acknowledgments and reject conflicting transitions. A new gate clears the previous wake result. -2. **Implemented locally:** `/suspend` drains tracked activity, strongly reads the task/gate and atomically checks current coordinator intent plus the original PENDING approval while writing an acknowledged TaskEvents checkpoint. Failed/uncertain writes never acknowledge suspension. Real DynamoDB Local transactions cover approval, cancellation, deadline and intent changes between reads and commit. Existing best-effort event methods are not used for this barrier. -3. **Implemented/tested locally:** `/resume` invokes `refresh_microvm_credentials` behind the closed barrier, forcing ambient renewal before renewing the same tenant credential object **with the same task/user/repo tags**. Existing DynamoDB/S3/platform clients retain their references; session identity cannot change after construction. The Claude child uses only the scoped container provider and its managed export helper returns no cached keys. Actual pinned CLI probes establish first-request renewal and failure without fallback. **Remaining:** verify actual runtime provider, long sleep, Gateway signing and managed image behavior in AWS. Never use test-only `reset_session_cache()` to simulate renewal. -4. **Implemented locally:** coding remains blocked until refresh and atomic task/gate reconciliation finish. Completed duplicates acknowledge cached results; concurrent requests receive 409. The 20-second handler budget includes body reads and all lifecycle work. A timed-out or terminated callback cannot release work through a late thread completion. -5. **Implemented locally:** `/run` and successful HTTP/controller resume reseed the application PRNG from fresh OS entropy. Do not seed it with task IDs, timestamps or an image-fixed value. Continue using cryptographic randomness for secrets. Test resumed/sibling snapshot uniqueness where meaningful; do not claim that `random` becomes cryptographically safe. -6. **Implemented locally:** `_ApprovalDeadline` captures the original recorded UTC expiry and a monotonic cap before database writes. Remaining time is `min(monotonic_deadline - monotonic_now, created_at + timeout_s - wall_now)`, clamped at zero, including each sleep bound. Resume verifies the original recorded creation time/timeout and coordinator deadline, then releases the same approval loop with this exact object. Expired wake enters the existing timeout/late-decision path; it never creates a fresh window. -7. **Preserved/tested locally:** conditional TIMED_OUT write, strongly consistent reread when that write loses, and late-decision winner behavior. Forward/backward clocks, frozen monotonic time, slow writes/reads, missing rows and cancellation have regression coverage. **Still required:** exercise these through the actual resume barrier and live AWS lifecycle. TTL is asynchronous garbage collection, not a precise alarm clock. -8. **Implemented locally:** managed images and the packaging helper declare all six served hooks with shared budgets and the source's non-secret protocol marker. `/ready` and `/validate` remain AWS-silent; validation rejects a supplied incompatible marker. The coordinator first saves the known worker handle, then verifies the exact Run-returned image ARN/version and conditionally persists support. Both policy and store require this evidence for new suspend; unknown/legacy workers keep new suspends off. See [image capability verification](./645-p3-image-capability.md). Matching artifact/coordinator deployment and live acceptance remain required. - -## 6. Wire the supervisor and human decisions - -### Orchestrator - -**Implemented locally; live acceptance pending.** The production path implements the state/action table in a small testable policy/reconciliation helper called by the durable poll loop. Read current gate identity/status/deadline consistently. Record intent before requesting suspension; reread after uncertain outcomes and after suspend success to catch an approval that won concurrently. API acknowledgement is not final VM state: poll it. - -An approval can arrive before a pending suspend finishes. Even if an inline resume sees “already running,” the orchestrator must later notice that the machine became suspended and wake it. Do not clear durable wake intent merely because one API call appeared successful. - -When `ResumeMicrovm` is acknowledged but a subsequent observation does not yet confirm `RUNNING`, keep reconciling within a bounded recovery interval. Do not invent a `RESUMING` service state or treat an unknown/coarse state as confirmation. Defer the pre-suspend stale-heartbeat check only within that bound. When the agent restores task RUNNING, use the fresh timestamp from prerequisite 1C. Never exempt genuine crashed RUNNING tasks indefinitely. - -Consecutive MicroVM poll-error tracking resets after a complete successful observation cycle; classify permanent failures separately from transient ones. At the chosen threshold, perform a final consistent task read, record an explicit infrastructure failure and finalize/terminate through the existing single-owner path. Emit recovery/orphan diagnostics when termination itself fails; do not silently lose the handle. Keep suspend failures distinguishable from lost compute: failure to save money can leave a task safely awake, whereas failure to wake threatens correctness and needs bounded escalation. - -### Approve and deny handlers - -**Implemented locally; live acceptance pending.** After the existing authorization checks and decision transaction **commit**, use a shared helper to load `compute_metadata` with a strongly consistent task read. Validate compute type, complete handle and current task/gate identity. For MicroVM, request resume with a short bound. No HTTP call to the guest is necessary. - -Missing handle, read failure, wrong/terminal state or resume failure must produce a warning and a structured resume-orphan event (include task ID, gate ID, VM ID when known, stage, reason and safe AWS request ID). Audit-event failure is also best-effort. **None of these post-commit failures may turn a successful decision into a 500 or undo the transaction.** Preserve the current response/status and ownership/already-decided/wrong-gate protections. The poll loop is the repair path. The current API has no independent wall-clock expiry check: the agent owns TIMED_OUT, and the first committed decision wins. Strict API expiry would be a separate behavior change. - -### IAM and deployment - -**Implemented in CDK; effective AWS permissions pending.** Grant orchestrator SuspendMicrovm/ResumeMicrovm on the exact configured image ARN and required version suffix, alongside its existing lifecycle actions. Grant approve/deny ResumeMicrovm, and GetMicrovm only if the shared wake helper uses it, with the same image scope. Do not grant these actions to the agent execution role. Add no token-minting, broad role-passing or network ingress permission. - -Verify the lifecycle store's DynamoDB permissions too: task GetItem/UpdateItem and approval GetItem/ConditionCheckItem for the supervisor's cross-table suspend transaction. Confirm environment wiring for both table names and test effective permissions. Worker writes to `microvm_lifecycle` must remain excluded. - -Check `task-api.ts`'s lazy image-ARN wiring and no-image branch, bootstrap deployment-role coverage, tests/suppressions and CloudFormation resource counts. Image-hook changes and runtime hook serving must deploy together; automatic suspension remains off until the compatible image is ready. The `microvm_approval_suspend_enabled` context defaults false and sets both the static opt-in and a live SSM parameter. Durable executions pin their original environment, so rollback must verify the live parameter is false to stop new suspends in existing executions. Resume, timeout handling and termination remain available. See the [supervisor runbook](./645-p3-supervisor.md#deployment-configuration-and-permissions) for drift/rollback details. Deploy matching source/image with false before controlled opt-in. - -## 7. Acceptance matrix and completion gates - -| Case | Required result | +- [x] Finish P2 follow-ups for payload provenance, task-scoped permissions, + bounded uncertain-start recovery, atomic capacity ownership, worker cleanup, + error classification, thread isolation and logging. See the + [clean deployment](./645-p2-clean-deployment-20260913.md), + [payload checks](./645-p2-payload-live-20260914.md), + [registration recovery](./645-p3-registration-20260916.md) and + [effective permissions](./645-effective-iam-20260915.md). +- [x] Implement all six image/runtime hooks, guest activity barriers, original + deadlines, credential renewal, exact image capability and Durable supervision. + [Lifecycle hooks](./645-p3-lifecycle-hooks.md) and the + [supervisor](./645-p3-supervisor.md) document the contract. +- [x] Reproduce and correct stale pooled hook connections. Lifecycle responses + explicitly close their HTTP connections before freeze. The + [transport comparison](./645-p3-wake-transport-20260916.md) includes the + greater-than-90-second control and closure-before-freeze evidence. +- [x] Verify approval, denial, cancellation, deadline races, multiple sleeps, + repository work, networking and actual AWS credential renewal after expiry. + See [final-image/ECS checks](./645-p3-final-image-and-ecs-20260917.md), + [repository acceptance](./645-p3-repository-path-20260917.md), + [MCP/network checks](./645-p3-mcp-network-20260917.md) and + [AgentCore permissions](./645-p3-agentcore-permissions-20260917.md). +- [x] Implement retained approvals, versioned conversation/workspace storage, + budget accounting, attempt ownership, confirmed retirement, replacement + admission and scheduled cleanup for repository and repository-free tasks. + See the [continuation protocol](./645-p3-continuation-protocol-20260917.md). +- [x] Pass the real cloud continuation matrix: approve, deny and expiry after + retirement; cancellation without replacement; approval and denial on the + original sleeping worker. Verify real output/checksums and resource cleanup. + See [cloud acceptance](./645-p3-cloud-continuation-20260917.md). +- [x] Verify normal signed-in CLI pending/decision feedback, a request answered + after more than two hours, one replacement, exact Read execution and final + capacity release. The [progress-race record](./645-p3-progress-race-20260918.md) + distinguishes recovered cloud evidence from the interrupted local watcher. +- [x] Fix the normal-workflow progress/checkpoint race. Progress can finish + during upload; capture drains acknowledgments before publication. General + activity and actual suspension retain their barriers. Three regressions + cover successful writes, in-flight writes and failed writes. +- [x] Implement the requested nested MicroVM stack, install bootstrap 1.9.0, + verify fresh deployment and cleanup, and rehearse overlapping old/new + resources with a live permission probe and consumer rollback. +- [x] Add the normal nested stack, build its image, preserve the shared + execution-role identity, and switch compatible consumers while retaining old + resources for drain. The [nested record](./645-p3-nested-stack.md) also retains + the failed native-refactor attempt and verified automatic rollback. +- [x] Complete corrected-image normal default/expiry/off-switch acceptance, + rollback to its compatible sleep-off coordinator and restore activation. +- [x] Verify the old worker/execution drain, remove the 16 obsolete flat + resources, and check final resource identities, outputs and permissions. +- [x] Remove owned acceptance data/identity and temporary infrastructure; finish + documentation sync/check/build and the final evidence manifest. + +The full agent suite after the Linear vault follow-up passes **2,170 tests**, with +13 explicitly opt-in skips and **86.39%** coverage. Lint, formatting and types pass. +The CDK suite passed **5,223 tests**, 56 skips and its snapshot; the later +managed-image pin/dispatch checks passed **245 tests**. CLI compilation, lint and +**943 tests** passed. These are suite results, not additive independent counts. + +The subsequent real Linear submission also passed with AgentCore Identity vault: +the guest obtained its token from the existing vault grant, opened a one-file +test PR, posted completion feedback to Linear, terminated and released capacity. +Its fallback secret contained metadata only. This check caught and fixed missing +vault fields in the shared guest configuration and a metadata-only fallback +crash. The test PR was closed unmerged and the owned Linear fixture removed. +The private reproducible harness and raw receipts are maintained outside Git. + +## Unanswered approvals implementation order + +The prerequisites are implemented in this order: + +1. Persist exact task/request/tool identity and original decision deadlines. +2. Save and restore the actual SDK conversation, including the pending action. +3. Preserve Git commits, index, worktree, required ignored files and workflow + context in bounded, checksum-verified, version-pinned storage. +4. Keep tool execution behind an ownership barrier while capturing or restoring. +5. Confirm retirement before releasing the old reservation. Admit one replacement + through a conditional coordinator transaction. +6. Deliver the saved decision to the restored agent, preserve all approved input + fields, and account for cost and turns across both worker runs. +7. Keep pending rows free of retention TTL; close them on task cancellation or + completion. Clean terminal checkpoints, launch records and worker leases. +8. Deploy compatible producers, APIs, coordinator and image before activating + automatic suspension; verify the normal submission/response path. + +The implementation does not retain an old worker indefinitely. It also does not +claim recovery when a complete checkpoint could not be verified. Unsafe parallel +or detached work, failed storage and uncertain writes remain explicit failures +or prevent suspension. + +## Approval UX and scope + +The supported response path is the signed-in CLI. `bgagent pending` shows the +task, proposed tool, reason, creation time, deadline or “no automatic expiry,” +and exact approve/deny commands. `bgagent watch` displays the request and recorded +decision. Restarting the CLI does not remove the saved request. + +Slack and Linear notification renderers carry the same response instructions. +Their deployed packages, event routing, retry receipts and closure messages +have been verified; see [approval UX](./645-p3-approval-ux-20260917.md). +No external Slack/Linear messages were sent during this acceptance run. +Native Slack decision buttons, approval-by-Linear-reply and notification +throttling remain separate product follow-ups. + +Repository mise/build-command changes are excluded at the user's request. +The earlier CLI installer/configuration addition was withdrawn. The private +verification harness's `--mise` option only selects the local test runner tool. + +## Acceptance matrix and completion gates + +| Area | Evidence and required outcome | |---|---| -| Ordinary task, no approval | Same successful workflow as clean P2; no suspend calls | -| Short gate or decision during grace | No unnecessary freeze | -| Long gate + approve | VM sleeps, decision commits, same workspace and identity resume, allowed action runs once | -| Long gate + deny | VM wakes, denied action does not execute | -| No decision | Wake before deadline; agent-owned timeout/deny occurs on the original deadline | -| Deadline already passed while frozen | No restarted timeout; conditional late-decision race remains correct | -| Inline resume read/API/event failure | Decision API still reports its committed success; orphan signal and orchestrator repair | -| Approval races with suspend | No permanently stranded approved/denied task | -| Cancel during suspend/resume | Terminal task never becomes RUNNING again; VM terminated; slot released once | -| Duplicate/replayed lifecycle calls | No duplicate task/action and no changed approval deadline | -| Credentials expire during sleep | Resume refreshes all relevant credential consumers without losing tenant tags | -| Resume refresh/durability failure | Fail closed or remain safely awake; no unbounded hidden wait | -| VM dies or API polling repeatedly fails | Specific failure, bounded recovery, finalization/cleanup with retained handle | -| Unexpected suspension during RUNNING | Anomaly emitted once per episode; bounded wake/recovery rather than immediate false failure | -| Payload/network negatives | Wrong-task reads denied, truncated bytes rejected, no public ingress, documented port behavior | -| Other backends | AgentCore/ECS explicit unsupported responses; normal tasks/cancellation unchanged | -| Deploy/migrate/rollback | Compatible pinned image, valid bootstrap grants, every stack within limits, no accidental data deletion | - -Use unit tests for timers/state transitions and fault injection, integration tests for cross-record races and credential/client ownership, and real AWS runs for hook ordering, snapshot behavior, networking and permissions. Do not use a real hour-long sleep as the only clock test: simulate expired credentials locally, then run a controlled live long-suspend case when the service/session limits permit it. Record what was actually verified. - -For live evidence, retain redacted task IDs, commit/image/bootstrap versions, context flags, timeline, VM state transitions, approval timestamps, CloudWatch/progress evidence, workspace checksums or sentinel files, and proof of final termination/payload cleanup. Measure resume latency to tune grace/wake margin. Never include tokens, signed URLs or secret values in runbooks. - -**P3 is complete only when:** the full strategy interface and agent hooks ship; approval/deadline races and credential refresh pass; clean P2 and live P3 gates pass on the final deployment; behavior-changing follow-ups have passing regressions; generated docs, bootstrap artifacts and compatibility controls match; no test VM is left running/suspended; and #645/ADR/runbook status is updated with the actual evidence. An unverified gate must be called out explicitly, not converted to a checked box because unit tests passed. - -## Validation of the original review commit - -Completed locally on 2026-09-13. These checks concern this cleanup/review branch, not the future P3 acceptance tests. - -- **246 CDK tests passed** across `lambda-microvm-compute`, `task-orchestrator`, `orchestrate-task-microvm` and `lambda-microvm-strategy` suites (`npx jest --runInBand --coverage=false` with those four paths). -- **104 Python MicroVM server tests passed** (`pytest tests/test_server.py -k microvm --no-cov`); passing this focused run does not resolve #841's fixture isolation issue. -- **The vault/MicroVM guard regression passed** in `stacks/agent.test.ts`; only its wording changed, and it still rejects the same combination. -- **Code comparison passed:** TypeScript output with comments removed is unchanged after accounting for the cdk-nag explanation and guard error-text corrections. Python syntax trees are identical after removing docstrings. No lifecycle logic, IAM grants or deployment topology changed. -- **Python formatting, Markdown source link checks, new artifact relative links and `git diff --check` passed.** -- **Documentation sync and build passed: 77 pages.** The installed mise version could not expand the build task's `:sync` dependency (`':task' pattern should be expanded before matching`), so the declared steps were executed in order with `mise run sync` followed by `mise exec -- ./node_modules/.bin/astro build`. Existing Astro/Cedar highlighting/deprecation warnings remain; they did not fail the build. -- **Offline nesting probe:** measured five configurations in current, naive-nested and parent-role/stable-name modes. The naive mode failed cycle validation; the corrected prototype passed. Probe output is linked in the review. No AWS resources were created or modified. - -The live P2 rerun, production nested-stack migration and P3 implementation/acceptance matrix remain future work. This original review commit supplies the cleanup, evidence and plan; subsequent prerequisite fixes are tracked at the top of this document. +| Ordinary coding and other substrates | Clean P2 repository flow, ECS approval/cancellation and AgentCore permission checks pass | +| Same-worker sleep/wake | Approve, deny, original-deadline expiry and cancellation pass; files and identity survive | +| Credential expiry | Real expired AWS session is renewed with unchanged task/user/repo tags; no ambient fallback | +| Hook transport | Old pooled-connection race reproduced; explicit close verified before freeze | +| Retained decision | Pending request survives retirement without TTL; a later answer remains actionable | +| Replacement | One admitted worker restores exact work/action context and cumulative usage | +| Cancellation and races | Task cannot become RUNNING after cancellation; competing decisions and lost replies preserve ownership | +| Failure feedback | Bounded recovery and stable task/API guidance; hook stage, identity and timing diagnostics | +| Storage and permissions | Corrupt/interrupted transfers rejected; wrong-task reads and forbidden payload access denied | +| Capacity | Task-owned reservations, exact-attempt leases and confirmed shutdown before release | +| Nested migration | Rehearsed overlap, one merged payload-deny exception list, explicit image/coordinator pins and drain before deletion | +| Normal activation | Corrected image, actual 600-second default, timed request, off switch, compatible rollback and restored activation | +| Cleanup | No owned test worker remains alive; temporary resources, test identity and private credentials removed | + +Real AWS requests establish service behavior and effective permissions. Unit +tests cover deterministic races and clocks; the real SDK and private AWS suites +cover process loss, disk loss and transaction ownership. An accepted resume API +response alone is never counted as a successful wake. + +## Reproducible tests and handoff + +The independent harness and raw evidence live outside Git: + +`~/.local/share/abca-verification/645-p3-integration/README.md` + +Its local, real-SDK, S3, DynamoDB, coordinator and full-cloud suites record source +hashes and cleanup results. `all` runs the first five; `cloud` is an explicit +option because it builds images and uses real MicroVM/model capacity. Fresh +cloud runs use isolated tagged resources and the existing deployment's VPC. +The normal migration/CLI fixtures are installation-specific and are labelled +separately. + +The [microvms-agentd reference](https://github.com/laithalsaadoon/microvms-agentd/tree/78304e361fbbe62e3a6b255b43c6f6c372b47510) +informed reproducible commands, output-file assertions, disk-reserve checks and +cleanup receipts. No source was copied and no third-party executable was run. +ABCA keeps its task-scoped credential broker and complete Git/workspace recovery. + +## Service and later-phase follow-ups + +The [service feedback tracker](./645-lambda-microvm-service-feedback.md) preserves +requests for internal hook-dispatch traces, precise transport-error wording, +fresh-connection/retry behavior and the reproduced CloudFormation refactor tag +schema limitation. Historical worker traces remain unavailable. The application +fixes and migration do not claim that those service questions were answered. + +Unknown Run outcomes remain bounded: reuse the saved start token only within +the supported recovery window; do not blindly create a second worker. If no +unique worker can be recovered, preserve the diagnostic and the full service +lifetime bound. This is a documented operator limitation, not proof of an +unbounded AWS idempotency guarantee. + +The legacy/current capacity migration passed its isolated upgrade/rollback +rehearsal and a 600-user scan check. This installation already used task-owned +reservations, so its normal rollout preserves that protocol. Arbitrary +production scale is not established by these checks. Linear-vault integration +(#857) has its own validation, separate from this P3 acceptance record. + +ADR-021 defines P1, P2 and P3. It does not define an official P4. Native channel +approval controls, broader liveness detection (#491), operator shell access and +additional deployment combinations can be scoped as follow-up work. diff --git a/docs/verification/645-p3-live-deployment-20260915.md b/docs/verification/645-p3-live-deployment-20260915.md index 33452ede5..36bd0e33a 100644 --- a/docs/verification/645-p3-live-deployment-20260915.md +++ b/docs/verification/645-p3-live-deployment-20260915.md @@ -4,7 +4,7 @@ Date: 2026-09-15. The development stack is deployed with the P3 supervisor and six-hook image. Automatic suspension remains disabled. Six isolated guest cases pass; suspended cancellation exposed a stop-path bug that was repaired, deployed and verified with a stronger probe below. -The full [P3 acceptance matrix](./645-p3-implementation-plan.md#7-acceptance-matrix-and-completion-gates) +The full [P3 acceptance matrix](./645-p3-implementation-plan.md#acceptance-matrix-and-completion-gates) is not complete. ## Deployed configuration diff --git a/docs/verification/645-p3-nested-stack.md b/docs/verification/645-p3-nested-stack.md index 0b879b834..e95a8f6e5 100644 --- a/docs/verification/645-p3-nested-stack.md +++ b/docs/verification/645-p3-nested-stack.md @@ -1,10 +1,13 @@ # ADR-021 nested MicroVM stack -The user requested this split after the P3 approval UX review. The implementation, -local checks and an isolated fresh AWS deployment, image build and deletion are -complete. The normal `backgroundagent-dev` deployment still uses the flat layout; -migration of those existing resources remains open. This infrastructure split -does not nest virtual machines. +The user requested this split after the P3 approval UX review. Implementation, +fresh deployment and an overlapping-resource migration rehearsal are complete. +The normal `backgroundagent-dev` deployment now runs the nested image. Its old +flat resources have been removed after the verified drain and rollback/restore +acceptance. The parent has 471 resources and the MicroVM child has 18; +all 459 original resources outside the removal set preserved their identities. +See the [normal acceptance record](./645-p3-normal-closure-20260918.md). +This infrastructure split does not nest virtual machines. An isolated image-ownership refactor was also exercised on September 17. CloudFormation accepted the preview but rejected execution because @@ -37,6 +40,13 @@ For `compute_type=lambda-microvm`, nesting defaults to enabled. deployments that have not migrated. The setting accepts booleans or the strings `true` and `false`. +`microvm_managed_image_version` optionally pins new workers to a verified version +of the managed image, such as `7.0`. It changes the coordinator's runtime +selection without removing, replacing or rebuilding the image resource. Omit it +to keep the latest-active behavior. When pinned, building a newer image does not +switch new tasks to it until the operator updates the pin. Record the image ARN +and version alongside the coordinator version for rollback. + Nested deployments require bootstrap bundle **1.9.0**. It adds only the two exact child build/operator names to the backend-specific, unconditioned `iam:PassRole` statement. Legacy role prefixes remain for flat deployments; the runtime @@ -55,8 +65,8 @@ Installing this prerequisite does not move the normal stack's resources. Nesting does not change the 8,192 MiB memory baseline, hook configuration, runtime HTTPS-only egress, separate build HTTP/HTTPS egress, task-scoped -permissions or sleep gates. The MicroVM/Linear-vault combination remains gated -by #857 until its own deployment verification is complete. +permissions or sleep gates. Linear vault configuration is covered separately in +the [setup guide](../guides/LINEAR_SETUP_GUIDE.md#using-the-vault-with-lambda-microvms). ## Existing deployment migration @@ -73,9 +83,11 @@ preparing a concrete migration: 2. Rehearse the chosen resource-transfer or replacement procedure in an isolated deployment, including its rollback. Inspect each change set for deletions, replacements, custom-resource effects and named-resource conflicts. -3. Stop new admissions through a procedure that preserves accepted inputs and - drain active work before switching resource ownership. An idle inventory alone - does not prevent new tasks from arriving. +3. Either pause admission while preserving accepted inputs, or keep old and new + resources available during a compatible consumer switch. The latter requires + verified overlapping permissions and explicit image/coordinator pins. Drain + every old worker and retained execution before removing its resources; an idle + inventory alone is not an admission fence. 4. Deploy bootstrap 1.9.0 and apply only the reviewed migration. Confirm outputs, permission boundaries, managed image build and normal task lifecycle. 5. Keep both sleep gates off until the normal activation checks are complete; @@ -102,14 +114,56 @@ for inline policies and S3 auto-delete custom resources, preserving permissions and bucket contents throughout. Existing generated role names also differ from the new explicit names; moving ownership must not silently rename those roles. -The execution failure below means the normal migration must now choose and -rehearse another supported procedure. A retain/remove/import sequence is a -candidate only after an actual import of this resource type succeeds in isolation; -identifier discovery does not prove import support. An explicit replacement -procedure must preserve artifacts, pending payloads and compatible images while -using non-conflicting names. Neither alternative has been executed or accepted. -Do not treat the failed native refactor as a reason to apply the final nested -template directly to the existing stack. +The failed refactor led to the explicit replacement procedure below. Import was +not assumed to work. The replacement preserves artifacts, payload access and +compatible images during overlap, using non-conflicting resource names. + +## Overlapping replacement rehearsal and normal cutover + +The private `backgroundagent-dev-p3-migration-20260917` stack first deployed flat +resources, then added a separately named nested image and its infrastructure. +A Lambda using the exact shared MicroVM execution role verified that both old +and new bootstrap markers were readable, and that `tasks/forbidden.txt` was denied +in both payload buckets. Consumer outputs switched to the new image, rolled back +to the old image, and switched forward again while both resource sets existed. +The execution-role ARN and saved marker contents stayed unchanged. + +One permission detail matters: the payload policy has an explicit +`Deny`/`NotResource`. During overlap, its single exception list must contain both +bootstrap prefixes. Adding a second deny would make the two policies deny each +other's allowed bucket. The rehearsal tested the effective permissions. + +After the old resources were removed, the rehearsal had three parent resources +and 18 child resources. The whole private stack was then deleted. Its cleanup +manifest tracks 43 old/new resource identities, verifies absence and reports no +leaks; automatically created provider log groups were removed too. + +The normal cutover follows the same staged procedure: + +- Add the child without changing the original 475 physical resource identities. +- Build `backgroundagent-dev-p3-abca-agent` under the new child. +- Switch consumers with overlapping permissions and a compatible pinned image. + Deploy retained-request producers after the compatible coordinator alias. +- Verify normal CLI decisions, worker recovery and capacity release. +- Drain old executions/workers before deleting the 16 obsolete flat resources. + Preserve the old log group intentionally for historical diagnosis. + +The normal nested image is now version 2.0, with artifact SHA-256 +`924a1b51fe6b9aa62f61191a6bde9b10df01d65873181a9385489191492cadbe`. +Compatible coordinator 13 pins that image with automatic sleep disabled; +coordinator 14 pins the same image with sleep enabled. Final off-switch acceptance +and old-resource removal passed, as recorded in the +[normal acceptance record](./645-p3-normal-closure-20260918.md). +Future updates must retain `microvm_resource_name_prefix=backgroundagent-dev-p3`; +that migration prefix now identifies the installed service resources. + +No global admission pause or zero-concurrency setting was used. The +[earlier rollout review](./645-p3-normal-rollout-review-20260917.md) explains why +that shortcut could lose asynchronous inputs on this installation. + +Exact templates, change sets, probes and cleanup results are outside Git under +`~/.local/share/abca-verification/645-p3-integration/nested-overlap-rehearsal` +and `normal-nested-rollout`. The following sections preserve earlier milestones. ## Verification @@ -145,7 +199,8 @@ Every template fits the 500-resource, 1 MiB and 200-parameter/output limits. The hierarchy totals 535 resources, below the 2,500-resource nested-operation limit even if every resource changed. Unit synthesis additionally covers AgentCore, ECS and all three MicroVM image modes, each with the tool gateway -disabled and enabled. It preserves the separate #857 vault guard. +disabled and enabled. At that point it preserved the separate #857 vault guard; +the subsequent vault integration replaces that guard with parity and budget checks. The flat and nested templates retain the execution role logical ID `LambdaMicrovmComputeExecutionRoleAA0C4A0D`, the same trust document and the same @@ -204,14 +259,13 @@ of both buckets, connectors, security groups, image/version, roles, provider function and log groups. The implicitly created provider log group was archived and removed separately. -The normal deployment retains all 475 physical resource identities, image 7.0, +At that milestone, the normal deployment retained all 475 physical resource identities, image 7.0, coordinator alias 10, its disabled live sleep switch and the shared VPC. Both phases' exact templates, 36 evidence files and a SHA-256 manifest are archived at `/Users/sphias/.local/share/abca-verification/645-p3-20260916/nested-live`. -This proves fresh nested deployment, image build and deletion. Existing P3 worker -lifecycle evidence remains in its dated records. Moving the existing normal stack -and testing its task path after migration are still required. +This established fresh nested deployment, image build and deletion. The later +overlapping migration and normal task checks are recorded above. ## Image ownership refactor: execution rejected, rollback verified @@ -278,7 +332,7 @@ The three uploaded verification template versions were removed from their exact toolkit-bucket prefix, with no remaining versions or delete markers. Normal CDK assets were retained. -The normal deployment still has all 475 original resource identities, image 7.0, +At that milestone, the normal deployment still had all 475 original resource identities, image 7.0, coordinator alias 10, its disabled live sleep switch and the shared VPC. The rehearsal resources are fully deleted. Evidence and a SHA-256 manifest are archived at diff --git a/docs/verification/645-p3-normal-closure-20260918.md b/docs/verification/645-p3-normal-closure-20260918.md new file mode 100644 index 000000000..c5082f452 --- /dev/null +++ b/docs/verification/645-p3-normal-closure-20260918.md @@ -0,0 +1,187 @@ +# P3 normal deployment acceptance — September 18, 2026 + +The normal `backgroundagent-dev` deployment in account ``, +`us-west-2`, now uses the nested MicroVM infrastructure. P3 implementation and +deployment acceptance are complete: all four corrected-image normal cases, the +final test after removing old resources, and owned test-data cleanup passed. + +## Deployed configuration + +| Setting | Verified value | +|---|---| +| Managed image | `backgroundagent-dev-p3-abca-agent:2.0`, `ACTIVE` / `SUCCESSFUL` | +| Guest artifact SHA-256 | `924a1b51fe6b9aa62f61191a6bde9b10df01d65873181a9385489191492cadbe` | +| Artifact size | 519,287 bytes, 116 files | +| Memory baseline | 8,192 MiB | +| Lifecycle hooks | All six image/runtime hooks enabled | +| Active coordinator | Published version **14**, alias `live` | +| Compatible sleep-off rollback | Published version **13**, also pinned to image **2.0** | +| Live sleep switch | `/backgroundagent-dev/microvm-approval-suspend-enabled=true` | +| Default sleep delay | 600 seconds; per-task `0` disables sleep | +| Default approval deadline | `0`, no automatic decision expiry | +| Optional finite deadline | 30–3,600 seconds; positive policy-rule deadlines can also apply | +| Bootstrap bundle | 1.9.0 | + +The [progress/checkpoint correction](./645-p3-progress-race-20260918.md) is in +image 2.0. Reverting to image 1.0 would restore that known race; the tested +rollback therefore uses coordinator 13 with the corrected image. + +Subsequent updates to this installation must preserve the recorded CDK context: + +```json +{ + "microvm_nested_stack": true, + "microvm_resource_name_prefix": "backgroundagent-dev-p3", + "microvm_managed_image_version": "2.0", + "microvm_approval_suspend_enabled": true +} +``` + +The prefix remains part of this installation's configuration after migration. +Removing it would change the image, connector and log names again. The private +`normal-nested-rollout/progress-race-image/active-deployment-context.json` +contains the full context, including the exact base image and artifact hash. +Review the next generated change set; the archived migration templates preserve +unrelated deployed assets and are not substitutes for future source builds. + +## Actual signed-in CLI acceptance + +These cases used an owned Cognito identity, the normal API, normal Durable +coordinator, and `isadeks/vercel-abca-linear`. The fixture required approval for +one `Read` of that task's `README.md`, then used the existing clarification +workflow to finish without editing the repository. No PR, issue comment, email, +Slack message or Linear message was sent. + +The observer checked `bgagent pending`, `bgagent watch`, the actual CLI decision +receipt, stored approval identity, tool result, worker identity, terminal task +state and released capacity. A successful resume API response alone was +insufficient to pass. + +| Case | Task | Observed result | +|---|---|---| +| Omitted sleep/deadline overrides | `01M2S55A38R7V4FM532FF298KC` | Request remained pending without TTL; worker suspended after about 602 seconds; real CLI approval woke the same worker and exactly one Read succeeded | +| Explicit 300-second deadline, 30-second sleep delay | `01M2S55B567E0CZ613410VB2CW` | Same worker slept, woke before the original deadline and received `TIMED_OUT`; Read was denied; task completed and released capacity | +| New task while sleep disabled | `01M2S5W2NRDEVJVTM5VTQAHHHX` | Coordinator 13, static/live sleep flags false; worker stayed running for more than 75 seconds despite a 30-second delay, then accepted approval | +| Existing coordinator 14 task while live switch disabled | `01M2S5TMJK9JXPFF6SV7SHA5AK` | Version 14 execution was verified in its Lambda log stream; pending worker stayed running for 195 seconds despite a 120-second delay, then accepted approval | + +The default request was created at `02:24:19Z`; its worker first appeared +suspended at `02:34:21.491Z`. While it slept, the reviewed two-change rollback +switched the alias from 14 to 13 and the live SSM flag to false. The new-task +sleep-off case passed before the default request was approved at about +`02:37:49Z`. That same sleeping worker returned to `RUNNING` at `02:37:56Z`, +completed and terminated. Disabling new sleeps did not prevent an existing +sleeping worker from waking. + +All four cases ended `COMPLETED`; all workers terminated and all task-owned +capacity reservations were released. The scheduled manager released the last +two terminal reservations at `02:42:48Z`. The reviewed two-change restore +returned the alias to 14 and the SSM flag to true at `02:44:41Z`. + +An earlier normal request also survived more than two hours and retirement of +its original worker. Its later CLI approval admitted one replacement and +executed the exact Read. That evidence, including an interrupted local observer, +is described separately in the +[progress-race record](./645-p3-progress-race-20260918.md). + +## Final nested migration + +The [overlap rehearsal](./645-p3-nested-stack.md) established compatible +old/new permissions and rollback before normal migration. Native CloudFormation +image refactoring was rejected by the provider and was not used. + +Before final removal, a consistent inventory at `02:43:56Z` found: + +- No active tasks, held reservations, open worker leases or nonzero user counters. +- No live workers on the old image. +- No running Durable executions in retained coordinator versions 2–14. + +Change set `p3-normal-retire-flat-v2-20260918` removed the 16 obsolete flat +resource declarations and pruned old image/payload references from five IAM +policies. There were no replacements or unrelated modifications. The exact +final parent template hash is: + +`d497bcefaf81e217ce3b77b98cc36bc89bfcd7796c8714d2b4d63a2432848c04` + +The parent reached `UPDATE_COMPLETE` with **471 resources** in a 702,583-byte +template; its MicroVM child has **18 resources**. All **459** original resources outside the removal set +preserved their physical IDs, including the shared MicroVM execution role. +Parent output names continue to refer to the new child resources. + +Nine direct absence checks confirmed deletion of the old image, two buckets, +two connectors, two security groups and two roles. Their inline policies and +auto-delete resources completed removal. The old MicroVM log group was +intentionally retained with its 90-day retention for diagnosis; it is no longer +owned by the parent template. Seven old build-artifact versions were archived +before their bucket was removed. + +The resulting IAM simulation allows the new bootstrap-object read and explicitly +denies both the new forbidden payload path and the old bootstrap path. The final +normal smoke test exercises the actual worker and approval APIs after these +permissions were narrowed. + +## Final smoke and cleanup + +Final task `01M2S6S0YQ296KA52HM7HM0RJK` passed after the old resources and their +permissions were removed. Worker +`microvm-59f7debe-131d-3a8b-bb20-977d780f6df6` suspended at `02:52:52Z`, +received the actual CLI approval, resumed at `02:53:00Z`, completed at +`02:53:05Z` and terminated at `02:53:32Z`. Its exact Read succeeded once. +The scheduled manager released its reservation at `02:57:47.823Z`; +the observer recorded the complete pass at `02:57:53.216Z`. + +Normal test cleanup completed at `03:00:49.739Z`. All 11 synthetic-user tasks, +their closed leases, 341 approval/event/nudge rows, task object versions and +Memory episodes were removed. The zero counter, Cognito identity and private +CLI credentials were deleted and their absence verified. Repository settings +match their exact original values. + +Temporary cloud suites and the overlap rehearsal have already been deleted: +the corrected-image portable run verified removal of 40 resources, and the +overlap rehearsal verified removal of 43 recorded resources. Both report no +leaks. + +Normal fixture cleanup preserves real incurred-cost accounting and diagnostic +CloudWatch logs. It removes only the recorded synthetic user's task data, +approval/event rows, closed leases, task artifacts, conversation/workspace +objects, short-term Memory episodes, exact task memory namespaces, test identity +and private CLI credentials. Shared repository memory and shared bootstrap +configuration are outside that deletion boundary. + +An initial cleanup check observed a deleted Memory event briefly remaining in +`ListEvents`. The cleanup verifier now waits for confirmed absence with a bounded +deadline. The initial attempt and successful retry are both archived; no task +rows were deleted before that first visibility check stopped. + +## Evidence and reproduction + +The harness stays outside Git and the PR: + +`~/.local/share/abca-verification/645-p3-integration/README.md` + +The portable suites cover local restoration, the real pinned SDK with a +deterministic model, real S3, DynamoDB, the production coordinator, and fresh +cloud MicroVMs. Each run records source hashes, assertions and cleanup results. +The normal deployment scripts are explicitly installation-specific records, +not generic smoke commands. + +Key evidence directories: + +- `normal-acceptance-v2`: CLI cases, rollback/restore, final smoke and cleanup. +- `normal-nested-rollout/progress-race-image`: reviewed phase templates, final + resource identities, IAM simulation and old-resource absence. +- `runs/20260918-progress-race-cloud`: a fresh corrected image passing both + replacement approval and same-worker wake, followed by complete deletion. +- `cloud-replacement/results`: the full approve/deny/cancel/expiry and same-worker + approval/denial matrix. + +Full agent validation after the latest correction passed 2,162 tests with 13 +opt-in skips and 86.37% coverage, plus lint, formatting and types. CDK validation +passed 5,223 tests with 56 skips; the subsequent pin/dispatch checks passed 245. +CLI compilation, lint and 943 tests passed. +All 116 files in the deployed guest archive were compared with the final source +and matched exactly. + +Native Slack/Linear decision controls are separate product follow-ups; this +acceptance did not send external channel messages. Historical service-side +transport traces remain unavailable and are tracked in the +[service feedback record](./645-lambda-microvm-service-feedback.md). diff --git a/docs/verification/645-p3-normal-rollout-review-20260917.md b/docs/verification/645-p3-normal-rollout-review-20260917.md index 6f7e826ef..a0130a9c9 100644 --- a/docs/verification/645-p3-normal-rollout-review-20260917.md +++ b/docs/verification/645-p3-normal-rollout-review-20260917.md @@ -1,5 +1,11 @@ # P3 normal deployment rollout review — September 17, 2026 +Historical review. The later [overlapping nested cutover](./645-p3-nested-stack.md#overlapping-replacement-rehearsal-and-normal-cutover) +keeps both resource sets available, uses explicit compatible image/coordinator +pins, and drains old work before removal. That procedure supersedes the proposed +global pause below. The [completion plan](./645-p3-implementation-plan.md) tracks +current deployment status. + This was a read-only review of `backgroundagent-dev` in account ``, Region `us-west-2`. No admission settings, aliases, roles or task records were changed. It identifies the remaining operational work; it is not a completed @@ -83,7 +89,13 @@ older MicroVM image. Once a worker starts, its saved handle records the actual returned image version, but that does not select the version for future starts. Before rehearsing rollback, make image selection explicit in the reviewed -rollout procedure. The existing external-image path supports a version pin, +rollout procedure. The new `microvm_managed_image_version` context option pins +runtime selection while keeping the managed image under CloudFormation. Use a +version verified through `GetMicrovmImageVersion`, and preserve that coordinator +version together with its image ARN/version. This option is implemented locally; +normal deployment and rollback verification remain open. + +The external-image path also supports a version pin, but switching an existing managed image resource to that path is an infrastructure migration and must be reviewed for deletion/replacement. Do not remove a managed image merely to obtain a pin. diff --git a/docs/verification/645-p3-progress-race-20260918.md b/docs/verification/645-p3-progress-race-20260918.md new file mode 100644 index 000000000..3e2136d80 --- /dev/null +++ b/docs/verification/645-p3-progress-race-20260918.md @@ -0,0 +1,66 @@ +# P3 progress during checkpoint capture — September 18, 2026 + +The normal repository workflow exposed a progress-write race while saving a +continuation. This is separate from the earlier pooled HTTP connection problem. + +Task `01M2S3QH4RA02SMV7GKP0PSKR1` captured its conversation and workspace on +`backgroundagent-dev-p3-abca-agent:1.0`. An SDK progress event arrived during +the upload, while the lifecycle phase was `checkpointing`. The progress writer +treated the closed activity barrier as a lost write and permanently set +`progress_failed`. Capture nevertheless completed. Thirty seconds after the +approval request, the suspend hook correctly rejected this unsafe state with +HTTP 409; AWS terminated worker +`microvm-bfb43f74-bc32-37e5-860f-c0359bb6d94d`. + +The saved request and checkpoint survived. The coordinator confirmed termination, +parked the task and released capacity. A subsequent signed-in CLI approval +launched a replacement, which read the intended README and reached `COMPLETED`. +That recovery does not make the rejected suspend acceptable: the race is fixed +before final activation. + +## Correction and regression + +Progress events can write during continuation upload because these writes do not +modify the saved workspace. This exception applies only to the progress writer; +general activity and new tools remain paused. Capture drains progress a second +time after upload and refuses publication if a write failed or remains uncertain. +Actual suspension and credential renewal still close the activity barrier. + +The progress rejection log now includes task ID, event type and lifecycle phase. +It omits message contents and tool arguments. + +Two new regression tests failed before the correction. The corrected focused +suite passes 150 tests, including: + +- A progress event arriving during capture is acknowledged and permits suspend. +- A write started during upload must finish before capture is published. +- A real failed write prevents checkpoint publication and suspension. +- General activity remains blocked during upload; progress remains blocked during + actual suspension. + +The corrected guest artifact is 519,287 bytes with 116 files: + +`924a1b51fe6b9aa62f61191a6bde9b10df01d65873181a9385489191492cadbe` + +The normal deployment was returned to compatible coordinator 11 and the disabled +live sleep switch while rebuilding. Corrected image 2.0 is now active with +coordinator 14. Normal ten-minute sleep, explicit expiry, new and existing +sleep-off tasks, and compatible coordinator-13 rollback/restore all passed. +The [normal acceptance record](./645-p3-normal-closure-20260918.md) contains that +deployment evidence separately from the local regression results. + +## Independent evidence recovered after a watcher interruption + +The local watcher disconnected and later exceeded its deadline. The cloud task +continued without it. Full guest logs for retained request +`01M2RX085T9T5865XX1TBEVTYE` record a successful suspend acknowledgment +**601.45 seconds** after its creation. Its worker retired successfully after +about an hour, leaving the request pending without a TTL. + +The real CLI recorded approval at `2026-09-18T02:02:07.439Z`, more than two hours +after the request. A replacement worker consumed that decision and completed +the README task. This is cloud evidence recovered after the interruption, not +a passing result from the original local watcher. + +Private logs, task snapshots, exact approval receipts and regression results are +under `~/.local/share/abca-verification/645-p3-integration/normal-acceptance`. diff --git a/scripts/check-constants-sync.ts b/scripts/check-constants-sync.ts index 016051c6f..ff76aeef2 100644 --- a/scripts/check-constants-sync.ts +++ b/scripts/check-constants-sync.ts @@ -228,6 +228,17 @@ function main(): number { hook_port: number; maximum_duration_seconds: number; }; + microvm_continuation?: { + version: number; + lease_key_prefix: string; + object_key_prefix: string; + max_manifest_bytes: number; + max_workspace_bytes: number; + max_conversation_bytes: number; + park_after_seconds: number; + retirement_margin_seconds: number; + verified_sdk_version: string; + }; payload_bootstrap?: { version: number; manifest_prefix: string; @@ -275,7 +286,7 @@ function main(): number { if (agc.default < agc.min) invariantErrors.push('approval_gate_cap.default must be >= min'); if (agc.max < agc.default) invariantErrors.push('approval_gate_cap.max must be >= default'); if (ats.min <= 0) invariantErrors.push('approval_timeout_s.min must be > 0'); - if (ats.default < ats.min) invariantErrors.push('approval_timeout_s.default must be >= min'); + if (ats.default !== 0 && ats.default < ats.min) invariantErrors.push('approval_timeout_s.default must be 0 or >= min'); if (ats.max < ats.default) invariantErrors.push('approval_timeout_s.max must be >= default'); if (jiraAppActor.min_secret_length < 32) { invariantErrors.push('jira_app_actor.min_secret_length must be >= 32'); @@ -394,6 +405,18 @@ function main(): number { || typeof lifecycle.image_protocol_env !== 'string' || !/^ABCA_MICROVM_[A-Z0-9_]+$/.test(lifecycle.image_protocol_env)) { invariantErrors.push('microvm_lifecycle requires a positive protocol version, valid hook port, duration within 1–28800 seconds and ABCA_MICROVM_ marker name'); } + const continuation = json.microvm_continuation; + if (!continuation || !Number.isSafeInteger(continuation.version) || continuation.version <= 0 + || continuation.lease_key_prefix !== 'worker-lease#' || continuation.object_key_prefix !== 'continuations/' + || !Number.isSafeInteger(continuation.max_manifest_bytes) || continuation.max_manifest_bytes <= 0 + || !Number.isSafeInteger(continuation.max_workspace_bytes) || continuation.max_workspace_bytes <= 0 + || !Number.isSafeInteger(continuation.max_conversation_bytes) || continuation.max_conversation_bytes <= 0 + || !Number.isInteger(continuation.park_after_seconds) || continuation.park_after_seconds <= 0 + || !Number.isInteger(continuation.retirement_margin_seconds) || continuation.retirement_margin_seconds <= 0 + || !lifecycle || continuation.park_after_seconds + continuation.retirement_margin_seconds >= lifecycle.maximum_duration_seconds + || !/^\d+\.\d+\.\d+$/.test(continuation.verified_sdk_version)) { + invariantErrors.push('microvm_continuation requires stable prefixes, a positive version/size, verified SDK version and retirement within the worker lifetime'); + } const BUDGET_FIELDS = [ 'ready_hook_timeout_seconds', 'warmup_total_budget_seconds', From 2f82f96bd1c4f4c0c559db8de4b714b38fccae02 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Fri, 18 Sep 2026 09:23:11 -0400 Subject: [PATCH 076/149] feat(migration): validate staged CloudFormation templates Reject missing resource references and oversized intermediate templates before preparing a flat-to-nested MicroVM migration. Cover intrinsic references, template limits and baseline preservation with 13 tests. --- cdk/src/migration/template.ts | 138 ++++++++++++++++++++++++++++ cdk/test/migration/template.test.ts | 115 +++++++++++++++++++++++ 2 files changed, 253 insertions(+) create mode 100644 cdk/src/migration/template.ts create mode 100644 cdk/test/migration/template.test.ts diff --git a/cdk/src/migration/template.ts b/cdk/src/migration/template.ts new file mode 100644 index 000000000..1ec63bd31 --- /dev/null +++ b/cdk/src/migration/template.ts @@ -0,0 +1,138 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import { createHash } from 'node:crypto'; + +export interface TemplateResource { + Type: string; + Properties?: Record; + DependsOn?: string | string[]; + DeletionPolicy?: string; + UpdateReplacePolicy?: string; + [key: string]: any; +} + +export interface MigrationTemplate { + Resources: Record; + Parameters?: Record; + Conditions?: Record; + Outputs?: Record; + [key: string]: any; +} + +export function canonical(value: unknown): string { + const sort = (item: any): any => { + if (Array.isArray(item)) return item.map(sort); + if (item !== null && typeof item === 'object') { + return Object.fromEntries(Object.keys(item).sort().map(key => [key, sort(item[key])])); + } + return item; + }; + return JSON.stringify(sort(value)) ?? 'undefined'; +} + +export function digest(value: unknown): string { + return createHash('sha256').update(canonical(value)).digest('hex'); +} + +/** References in prose, asset names and metadata are not CloudFormation dependencies. */ +export function references(value: unknown): Set { + const result = new Set(); + const visit = (item: any): void => { + if (!item || typeof item !== 'object') return; + if (Array.isArray(item)) { + item.forEach(visit); + return; + } + if (typeof item.Ref === 'string') result.add(item.Ref); + const attribute = item['Fn::GetAtt']; + if (Array.isArray(attribute) && typeof attribute[0] === 'string') result.add(attribute[0]); + if (typeof attribute === 'string') result.add(attribute.split('.')[0]!); + const substitution = item['Fn::Sub']; + if (substitution !== undefined) { + const expression = typeof substitution === 'string' ? substitution : substitution[0]; + const bindings = typeof substitution === 'string' ? {} : substitution[1] ?? {}; + if (typeof expression !== 'string') throw new Error('Invalid Fn::Sub in migration template'); + for (const match of expression.matchAll(/\$\{([^}]+)\}/g)) { + const token = match[1]!; + if (token.startsWith('!') || Object.hasOwn(bindings, token)) continue; + result.add(token.split('.')[0]!); + } + visit(bindings); + } + for (const [key, child] of Object.entries(item)) { + if (key !== 'Fn::Sub' && key !== 'Metadata') visit(child); + } + }; + visit(value); + return result; +} + +export function referencesAny(value: unknown, ids: ReadonlySet): boolean { + return [...references(value)].some(id => ids.has(id)); +} + +/** Refuse incomplete staged templates before publishing any asset or change set. */ +export function validateTemplate(template: MigrationTemplate, label: string): void { + if (!template.Resources || typeof template.Resources !== 'object' + || Array.isArray(template.Resources) || !Object.keys(template.Resources).length) { + throw new Error(`${label}: template must contain resources`); + } + const count = Object.keys(template.Resources).length; + if (count > 500) throw new Error(`${label}: ${count} resources exceed the CloudFormation limit of 500`); + if (Buffer.byteLength(JSON.stringify(template)) > 1024 * 1024) { + throw new Error(`${label}: template exceeds the CloudFormation 1 MiB limit`); + } + for (const section of ['Parameters', 'Outputs'] as const) { + if (Object.keys(template[section] ?? {}).length > 200) { + throw new Error(`${label}: ${section} exceeds the CloudFormation limit of 200`); + } + } + const known = new Set([...Object.keys(template.Resources), ...Object.keys(template.Parameters ?? {})]); + for (const ref of references(template)) { + if (!known.has(ref) && !ref.startsWith('AWS::')) { + throw new Error(`${label}: unresolved reference ${ref}`); + } + } + for (const [id, resource] of Object.entries(template.Resources)) { + if (!resource || typeof resource.Type !== 'string') throw new Error(`${label}: invalid resource ${id}`); + const dependencies = resource.DependsOn + ? (Array.isArray(resource.DependsOn) ? resource.DependsOn : [resource.DependsOn]) + : []; + for (const dependency of dependencies) { + if (!template.Resources[dependency]) throw new Error(`${label}: ${id} depends on missing ${dependency}`); + } + } +} + +export interface TemplateDelta { + added: string[]; + removed: string[]; + modified: string[]; +} + +export function templateDelta(before: MigrationTemplate, after: MigrationTemplate): TemplateDelta { + return { + added: Object.keys(after.Resources).filter(id => !before.Resources[id]).sort(), + removed: Object.keys(before.Resources).filter(id => !after.Resources[id]).sort(), + modified: Object.keys(after.Resources).filter(id => + before.Resources[id] && canonical(before.Resources[id]) !== canonical(after.Resources[id]), + ).sort(), + }; +} diff --git a/cdk/test/migration/template.test.ts b/cdk/test/migration/template.test.ts new file mode 100644 index 000000000..b33cfdf95 --- /dev/null +++ b/cdk/test/migration/template.test.ts @@ -0,0 +1,115 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import { + canonical, digest, MigrationTemplate, references, referencesAny, templateDelta, validateTemplate, +} from '../../src/migration/template'; + +const fixture = (): MigrationTemplate => ({ + Parameters: { Environment: { Type: 'String' } }, + Resources: { + Bucket: { Type: 'AWS::S3::Bucket' }, + Consumer: { + Type: 'AWS::IAM::Policy', + DependsOn: 'Bucket', + Properties: { + Resource: { 'Fn::Sub': '${Bucket.Arn}/${Environment}/${AWS::Region}/*' }, + }, + }, + }, + Outputs: { Name: { Value: { Ref: 'Bucket' } } }, +}); + +describe('migration template boundaries', () => { + test('recognizes Ref, both GetAtt forms, Sub bindings and escaped variables', () => { + expect(references({ + one: { Ref: 'Role' }, + two: { 'Fn::GetAtt': ['Bucket', 'Arn'] }, + three: { 'Fn::GetAtt': 'Image.ImageArn' }, + four: { 'Fn::Sub': ['${Alias}/${!Literal}/${Target.Arn}/${AWS::Region}', { Alias: { Ref: 'Bound' } }] }, + prose: 'OldBucket', + Metadata: { Ref: 'NotARealReference' }, + })).toEqual(new Set(['Role', 'Bucket', 'Image', 'Target', 'AWS::Region', 'Bound'])); + expect(referencesAny({ 'Fn::Sub': '${OldBucket.Arn}/*' }, new Set(['OldBucket']))).toBe(true); + expect(referencesAny('OldBucket', new Set(['OldBucket']))).toBe(false); + }); + + test('hashes object key order consistently without changing ordered arrays or strings', () => { + expect(digest({ a: 1, b: 2 })).toBe(digest({ b: 2, a: 1 })); + expect(digest([1, 2])).not.toBe(digest([2, 1])); + expect(canonical({ policy: '{"Version":"2012-10-17"}' })).not.toBe( + canonical({ policy: { Version: '2012-10-17' } }), + ); + }); + + test('accepts parameters and AWS pseudo-parameters in nested resource references', () => { + expect(() => validateTemplate(fixture(), 'prepare')).not.toThrow(); + }); + + test.each([ + { Ref: 'OldBucket' }, + { 'Fn::GetAtt': ['OldBucket', 'Arn'] }, + { 'Fn::GetAtt': 'OldBucket.Arn' }, + { 'Fn::Sub': 'arn:${AWS::Partition}:s3:::${OldBucket}/*' }, + ])('rejects stale resource references in retirement: %p', (value) => { + const template = fixture(); + template.Outputs!.Stale = { Value: value }; + expect(() => validateTemplate(template, 'retire')).toThrow('retire: unresolved reference OldBucket'); + }); + + test('rejects dangling DependsOn separately from property references', () => { + const template = fixture(); + template.Resources.Consumer!.DependsOn = ['Deleted']; + expect(() => validateTemplate(template, 'retire')).toThrow('depends on missing Deleted'); + }); + + test.each(['Parameters', 'Outputs'] as const)('rejects oversized %s', (section) => { + const template = fixture(); + template[section] = Object.fromEntries(Array.from({ length: 201 }, (_, index) => [`P${index}`, {}])); + expect(() => validateTemplate(template, 'cutover')).toThrow(`${section} exceeds`); + }); + + test('rejects overlap exceeding 500 resources instead of dropping legacy resources to fit', () => { + const template: MigrationTemplate = { + Resources: Object.fromEntries(Array.from({ length: 501 }, (_, index) => [`R${index}`, { Type: 'AWS::S3::Bucket' }])), + }; + expect(() => validateTemplate(template, 'cutover')).toThrow('501 resources'); + }); + + test('rejects oversized templates and malformed resource collections', () => { + const template = fixture(); + template.Description = 'x'.repeat(1024 * 1024); + expect(() => validateTemplate(template, 'prepare')).toThrow('1 MiB'); + expect(() => validateTemplate({ Resources: {} }, 'prepare')).toThrow('must contain resources'); + expect(() => references({ 'Fn::Sub': [null, {}] })).toThrow('Invalid Fn::Sub'); + }); + + test('reports additions, removals and changes without mutating the baseline', () => { + const before = fixture(); + const snapshot = structuredClone(before); + const after = structuredClone(before); + after.Resources.Bucket!.DeletionPolicy = 'Retain'; + delete after.Resources.Consumer; + after.Resources.Child = { Type: 'AWS::CloudFormation::Stack' }; + expect(templateDelta(before, after)).toEqual({ + added: ['Child'], removed: ['Consumer'], modified: ['Bucket'], + }); + expect(before).toEqual(snapshot); + }); +}); From 189051b624db697653d7950b0536280a810c646b Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Fri, 18 Sep 2026 09:29:43 -0400 Subject: [PATCH 077/149] feat(migration): preserve legacy worker permissions during cutover --- cdk/src/migration/permissions.ts | 162 +++++++++++++++++++++++++ cdk/test/migration/permissions.test.ts | 117 ++++++++++++++++++ 2 files changed, 279 insertions(+) create mode 100644 cdk/src/migration/permissions.ts create mode 100644 cdk/test/migration/permissions.test.ts diff --git a/cdk/src/migration/permissions.ts b/cdk/src/migration/permissions.ts new file mode 100644 index 000000000..4440e74b7 --- /dev/null +++ b/cdk/src/migration/permissions.ts @@ -0,0 +1,162 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import { canonical, digest, MigrationTemplate, referencesAny } from './template'; + +interface Policy { + id: string; + roles: any[]; + statements: any[]; + save: () => void; +} + +const POLICY_ID_HASH_LENGTH = 12; +// Leave space below IAM's 6,144-character quota for resolved ARN values. +const POLICY_DOCUMENT_BUDGET = 5000; + +const list = (value: any): any[] => Array.isArray(value) ? value : [value]; +const unique = (values: any[]): any[] => [...new Map(values.map(value => [canonical(value), value])).values()]; +const roleKey = (roles: any[]): string => canonical(roles.map(canonical).sort()); + +/** Include CDK overflow policies attached through Role.ManagedPolicyArns. */ +function policies(template: MigrationTemplate): Policy[] { + const result: Policy[] = []; + const attached = new Map(); + for (const [id, resource] of Object.entries(template.Resources)) { + if (resource.Type !== 'AWS::IAM::Role') continue; + for (const arn of resource.Properties?.ManagedPolicyArns ?? []) { + if (arn?.Ref) attached.set(arn.Ref, [...(attached.get(arn.Ref) ?? []), { Ref: id }]); + } + } + const add = (id: string, roles: any[], holder: Record): void => { + if (!roles.length) return; + const encoded = typeof holder.PolicyDocument === 'string'; + const document = encoded ? JSON.parse(holder.PolicyDocument) : holder.PolicyDocument; + if (!document || !Array.isArray(document.Statement)) throw new Error(`Unsupported IAM policy document: ${id}`); + result.push({ + id, + roles, + statements: document.Statement, + save: () => { holder.PolicyDocument = encoded ? JSON.stringify(document) : document; }, + }); + }; + for (const [id, resource] of Object.entries(template.Resources)) { + const props = resource.Properties ?? {}; + if (['AWS::IAM::Policy', 'AWS::IAM::ManagedPolicy'].includes(resource.Type)) { + add(id, unique([...(props.Roles ?? []), ...(attached.get(id) ?? [])]), props); + } else if (resource.Type === 'AWS::IAM::Role') { + for (const policy of props.Policies ?? []) add(`${id}/${policy.PolicyName}`, [{ Ref: id }], policy); + } + } + return result; +} + +function shape(statement: Record): string { + const copy = structuredClone(statement); + delete copy.Resource; + delete copy.NotResource; + delete copy.Sid; + for (const field of ['Action', 'NotAction']) { + if (copy[field]) copy[field] = list(copy[field]).sort(); + } + return canonical(copy); +} + +function mergeScope(target: Record, source: Record): void { + for (const field of ['Resource', 'NotResource']) { + if (Object.hasOwn(target, field) !== Object.hasOwn(source, field)) { + throw new Error('IAM migration cannot change Resource into NotResource'); + } + if (target[field] !== undefined) target[field] = unique([...list(target[field]), ...list(source[field])]); + } +} + +/** + * Keep legacy access on its original roles while both resource sets exist. + * Separate managed policies avoid exceeding a role's aggregate inline-policy + * quota. They disappear from the final template after the retirement check. + */ +export function preserveLegacyPermissions( + baseline: MigrationTemplate, + target: MigrationTemplate, + legacyIds: ReadonlySet, +): { template: MigrationTemplate; bridgePolicyIds: string[] } { + const template = structuredClone(target); + const current = policies(template); + const bridges = new Map(); + for (const old of policies(baseline)) { + if (legacyIds.has(old.id)) continue; // The old build/operator policy is retained verbatim. + const affected = old.statements.filter(statement => referencesAny(statement, legacyIds)); + if (!affected.length) continue; + const matchingPolicies = current.filter(policy => roleKey(policy.roles) === roleKey(old.roles)); + if (!matchingPolicies.length) throw new Error(`Missing original IAM principal for migration policy ${old.id}`); + const statements = matchingPolicies.flatMap(policy => policy.statements); + for (const statement of affected) { + const matching = statements.filter(candidate => shape(candidate) === shape(statement)); + if (statement.Effect === 'Deny') { + if (matching.length !== 1) throw new Error(`Ambiguous legacy Deny in ${old.id}; refusing to append another deny`); + mergeScope(matching[0], statement); + continue; + } + if (statement.Effect !== 'Allow' || !statement.Resource || statement.NotResource || statement.Principal) { + throw new Error(`Unsupported legacy identity grant in ${old.id}`); + } + const key = roleKey(old.roles); + const bridge = bridges.get(key) ?? { roles: old.roles, statements: [] }; + const retained = structuredClone(statement); + delete retained.Sid; + bridge.statements.push(retained); + bridges.set(key, bridge); + + // P2 allowed direct reads of the old payload bucket and had no bootstrap + // deny. P3's new deny must also exempt those pre-existing read resources + // until old workers drain. This does not grant access to the NEW bucket. + const oldActions: string[] = list(statement.Action ?? []); + if (oldActions.some(action => /^s3:(GetObject\*?|\*)$/i.test(action))) { + const oldObjects = list(statement.Resource).filter(value => referencesAny(value, legacyIds)); + const denies = statements.filter(candidate => + candidate.Effect === 'Deny' && candidate.NotResource + && list(candidate.Action ?? []).some(action => /^s3:(GetObject\*?|\*)$/i.test(action)), + ); + if (denies.length > 1) throw new Error(`Multiple S3 bootstrap denies for ${old.id}; explicit review required`); + for (const deny of denies) deny.NotResource = unique([...list(deny.NotResource), ...oldObjects]); + } + } + } + for (const policy of current) policy.save(); + const bridgePolicyIds: string[] = []; + for (const bridge of bridges.values()) { + const id = `MicrovmMigrationAccess${digest(bridge.roles).slice(0, POLICY_ID_HASH_LENGTH)}`; + if (template.Resources[id]) throw new Error(`Migration policy identifier already exists: ${id}`); + const document = { Version: '2012-10-17', Statement: unique(bridge.statements) }; + if (Buffer.byteLength(JSON.stringify(document)) > POLICY_DOCUMENT_BUDGET) { + throw new Error(`Legacy grants for ${id} need more than one managed policy; explicit review required`); + } + template.Resources[id] = { + Type: 'AWS::IAM::ManagedPolicy', + Properties: { + Description: 'Temporary original-role access while the MicroVM migration drains old workers', + Roles: bridge.roles, + PolicyDocument: document, + }, + }; + bridgePolicyIds.push(id); + } + return { template, bridgePolicyIds }; +} diff --git a/cdk/test/migration/permissions.test.ts b/cdk/test/migration/permissions.test.ts new file mode 100644 index 000000000..2530a79b0 --- /dev/null +++ b/cdk/test/migration/permissions.test.ts @@ -0,0 +1,117 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import { preserveLegacyPermissions } from '../../src/migration/permissions'; +import { MigrationTemplate } from '../../src/migration/template'; + +const oldObject = { 'Fn::Sub': '${OldPayload.Arn}/*' }; +const newObject = { 'Fn::Sub': '${Child.Outputs.PayloadArn}/bootstrap/*' }; +const ids = new Set(['OldPayload', 'OldImage']); +const policy = (statements: any[], role = 'Worker') => ({ + Type: 'AWS::IAM::Policy', + Properties: { Roles: [{ Ref: role }], PolicyDocument: { Statement: statements } }, +}); +const allow = (resource: any) => ({ Effect: 'Allow', Action: ['s3:GetObject*'], Resource: resource }); +const deny = (resource: any) => ({ Effect: 'Deny', Action: ['s3:GetObject*'], NotResource: resource }); +const baseline = (): MigrationTemplate => ({ Resources: { WorkerPolicy: policy([allow(oldObject)]) } }); +const target = (): MigrationTemplate => ({ + Resources: { WorkerPolicy: policy([allow(newObject), deny(newObject)]) }, +}); + +describe('migration permission overlap', () => { + test('P2 direct reads remain possible on the original role without granting new-bucket payload reads', () => { + const original = baseline(); + const next = target(); + const { template, bridgePolicyIds } = preserveLegacyPermissions(original, next, ids); + const statements = template.Resources.WorkerPolicy!.Properties!.PolicyDocument.Statement; + expect(statements.filter((statement: any) => statement.Effect === 'Deny')).toEqual([ + { ...deny(newObject), NotResource: [newObject, oldObject] }, + ]); + expect(bridgePolicyIds).toHaveLength(1); + expect(template.Resources[bridgePolicyIds[0]!]!.Properties).toMatchObject({ + Roles: [{ Ref: 'Worker' }], + PolicyDocument: { Statement: [allow(oldObject)] }, + }); + expect(original).toEqual(baseline()); + expect(next).toEqual(target()); + }); + + test('P3 flat bootstrap deny merges exceptions into one deny', () => { + const original = baseline(); + original.Resources.WorkerPolicy = policy([allow(oldObject), deny(oldObject)]); + const { template } = preserveLegacyPermissions(original, target(), ids); + const denies = template.Resources.WorkerPolicy!.Properties!.PolicyDocument.Statement + .filter((statement: any) => statement.Effect === 'Deny'); + expect(denies).toEqual([{ ...deny(newObject), NotResource: [newObject, oldObject] }]); + }); + + test('rejects unmatched legacy NotResource deny rather than denying both buckets', () => { + const original = baseline(); + original.Resources.WorkerPolicy = policy([{ ...deny(oldObject), Condition: { Bool: { Example: true } } }]); + expect(() => preserveLegacyPermissions(original, target(), ids)).toThrow('Ambiguous legacy Deny'); + }); + + test('does not share a worker bridge with the coordinator or retain unrelated grants', () => { + const original = baseline(); + original.Resources.WorkerPolicy!.Properties!.PolicyDocument.Statement.push(allow({ 'Fn::Sub': '${Other.Arn}/*' })); + original.Resources.CoordinatorPolicy = policy([ + { Effect: 'Allow', Action: ['lambda:TerminateMicrovm'], Resource: { Ref: 'OldImage' } }, + ], 'Coordinator'); + const next = target(); + next.Resources.CoordinatorPolicy = policy([ + { Effect: 'Allow', Action: ['lambda:TerminateMicrovm'], Resource: { Ref: 'NewImage' } }, + ], 'Coordinator'); + const { template, bridgePolicyIds } = preserveLegacyPermissions(original, next, ids); + expect(bridgePolicyIds).toHaveLength(2); + const worker = bridgePolicyIds.map(id => template.Resources[id]!.Properties!) + .find(props => props.Roles[0].Ref === 'Worker'); + expect(worker.PolicyDocument.Statement).toEqual([allow(oldObject)]); + }); + + test('finds managed overflow policies attached from the role and preserves JSON-string documents', () => { + const next = target(); + next.Resources.Worker = { + Type: 'AWS::IAM::Role', Properties: { ManagedPolicyArns: [{ Ref: 'Overflow' }] }, + }; + next.Resources.Overflow = { + Type: 'AWS::IAM::ManagedPolicy', + Properties: { PolicyDocument: JSON.stringify({ Statement: [deny(newObject)] }) }, + }; + next.Resources.WorkerPolicy = policy([allow(newObject)]); + const { template } = preserveLegacyPermissions(baseline(), next, ids); + expect(typeof template.Resources.Overflow!.Properties!.PolicyDocument).toBe('string'); + expect(JSON.parse(template.Resources.Overflow!.Properties!.PolicyDocument).Statement[0].NotResource) + .toEqual([newObject, oldObject]); + }); + + test('keeps a removed role or unsupported grant from being silently reassigned', () => { + const next = target(); + next.Resources.WorkerPolicy!.Properties!.Roles = [{ Ref: 'OtherWorker' }]; + expect(() => preserveLegacyPermissions(baseline(), next, ids)).toThrow('Missing original IAM principal'); + const original = baseline(); + original.Resources.WorkerPolicy!.Properties!.PolicyDocument.Statement[0].Principal = '*'; + expect(() => preserveLegacyPermissions(original, target(), ids)).toThrow('Unsupported legacy identity grant'); + }); + + test('rejects duplicate bootstrap denies', () => { + const next = target(); + next.Resources.WorkerPolicy!.Properties!.PolicyDocument.Statement.push(deny({ Ref: 'Other' })); + expect(() => preserveLegacyPermissions(baseline(), next, ids)).toThrow('Multiple S3 bootstrap denies'); + }); +}); From 848539c3791c58bde36f2a517ba02395d82056b8 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Fri, 18 Sep 2026 09:39:09 -0400 Subject: [PATCH 078/149] docs(microvm): describe integrated continuation and recovery --- agent/README.md | 70 ++++++++----------- docs/design/ORCHESTRATOR.md | 38 ++++++++-- .../content/docs/architecture/Orchestrator.md | 38 ++++++++-- 3 files changed, 94 insertions(+), 52 deletions(-) diff --git a/agent/README.md b/agent/README.md index 88955f2b2..4a045ab9e 100644 --- a/agent/README.md +++ b/agent/README.md @@ -309,58 +309,48 @@ Rejections are structured so they are readable in the MicroVM log group: `400 MI `/suspend` and `/resume` are served by `microvm_http.py` and declared in managed images with 30-second service timeouts and a shared non-secret protocol marker. Pause requires an active original approval gate, matching coordinator intent, drained activity and an acknowledged checkpoint; wake renews credentials and atomically rechecks the original task/gate before allowing coding. Each handler has a 20-second total budget. `/validate` rejects a supplied incompatible image marker without contacting AWS. The coordinator checks the actual launched image version and persists support on that worker; missing support disables new suspension. See [image capability verification](../docs/verification/645-p3-image-capability.md) and [completed P3 verification](../docs/verification/645-p3-nested-stack.md). -#### Conversation continuation checkpoints (P3 prerequisite) +#### Conversation and workspace continuation -`src/continuation_session.py` provides an SDK conversation store and versioned S3 -checkpoint adapter. It is **not yet connected to the production runner or worker -release** and does not change approval deadlines. The existing MicroVM lifecycle -checkpoint resumes the same frozen process; this new component supports future -continuation after that process is gone. +`src/continuation_runtime.py` connects the SDK conversation store, workspace +archive, approval hooks and production runner. Before retiring a worker, it +saves the exact pending tool call, conversation, workflow state, cumulative usage +and workspace in the versioned continuation bucket. A replacement restores those +objects using the coordinator-owned assignment; task payloads cannot choose +arbitrary checkpoint keys. `CheckpointSessionStore` implements the pinned SDK's public `SessionStore` -contract. A caller supplies it with `session_store_flush="eager"` and waits for -`checkpoint_pending()` to acknowledge the exact assistant tool call. Enabling -mirroring alone is insufficient because the SDK copies the transcript -asynchronously. A missing or rejected transcript batch prevents acknowledgement. -The envelope retains the session, full proposed action and task/attempt/request -identity, with a 16 MiB / 50,000-entry limit. - -`S3ContinuationCheckpoints` saves under -`continuations////.json`, verifies a -read-back, and returns a receipt pinned to the S3 object version and checksum. -Its default client requires task-scoped credentials. A versioned bucket and -task-prefix `s3:PutObject`, `s3:GetObject` and `s3:GetObjectVersion` permissions -are required; the current production artifact grants do not supply these reads. -The adapter neither lists nor deletes objects. Retention must be arranged by the -future integration after the request closes. - -`src/continuation_workspace.py` now captures and restores the local workspace: -Git history, staged/unstaged changes, and untracked/ignored files, with a 1 GiB / -100,000-entry default limit. It rebuilds Git configuration and never replaces an -existing destination. Unsupported Git/filesystem states or detected concurrent -writes prevent capture. This archive still needs durable upload and integration; -see the [workspace recovery boundaries](../docs/verification/645-p3-workspace-recovery-20260917.md). - -Before releasing a worker, the integration must hold the lifecycle barrier and -conditionally publish verified conversation and workspace receipts for the same -task attempt. A failed save must leave the worker available and report -the failure. Checkpoints are private task data: transcripts and proposed actions -can contain sensitive content. The module does not read CLI authentication files -or the process environment, but does not redact conversation contents. +contract with `session_store_flush="eager"`. `checkpoint_pending()` must +acknowledge the exact assistant tool call: enabling transcript mirroring alone +is insufficient because SDK writes are asynchronous. The conversation envelope +has a 16 MiB / 50,000-entry limit. + +`src/continuation_workspace.py` preserves Git history, staged/unstaged changes, +and untracked/ignored files, with a 1 GiB / 100,000-entry default limit. Restore +rebuilds Git configuration and refuses to replace an existing destination. +Unsupported filesystem/Git states or detected concurrent writes prevent capture. +Repository-free tasks use a private workspace with a local Git baseline. + +`S3ContinuationStorage` uploads and verifies version-pinned, checksummed objects +using task-scoped credentials. The continuation bucket and SessionRole grants +are provisioned by CDK. A failed or incomplete save prevents planned retirement; +capacity is released only after the coordinator confirms the old worker stopped. +Checkpoints contain private task data, including unredacted conversation and +workspace contents. They must not be published as diagnostic attachments. + +The replacement consumes the recorded decision and keeps the original cost/turn +allowance minus accumulated usage. See the [retained approval protocol](../docs/design/ORCHESTRATOR.md#retained-microvm-approvals) +for ownership, admission and cleanup. The opt-in test uses the actual pinned SDK/CLI with a deterministic loopback model. It kills the original process, deletes its configuration and workspace, -restores from the conversation store and workspace archive, and verifies approve -and deny through a fresh tool hook: +restores the conversation and files, and verifies approve and deny through a +fresh tool hook: ```bash cd agent ABCA_TEST_SDK_CONTINUATION=1 uv run pytest tests/test_continuation_sdk_probe.py --no-cov ``` -See the [session recovery and storage evidence](../docs/verification/645-p3-session-recovery-20260917.md) -for the live S3 permission checks and remaining replacement-worker requirements. - ### Testing Server Mode Locally Use `run.sh --server` to build and start the server locally. It handles credentials, port mapping, and resource constraints automatically: diff --git a/docs/design/ORCHESTRATOR.md b/docs/design/ORCHESTRATOR.md index 7036167c8..e6a4fd27a 100644 --- a/docs/design/ORCHESTRATOR.md +++ b/docs/design/ORCHESTRATOR.md @@ -299,17 +299,43 @@ When the session is unhealthy, the task transitions to `FAILED` with "Agent sess The P3 supervisor saves intent before control calls and rechecks the gate before and after them. Its durable state retains an absolute service lifetime, consecutive failures, recovery start time and next delay. Three failed cycles or 120 seconds of unconfirmed wake cannot become an indefinite wait. AWS RUNNING does not end recovery while the guest remains stuck on a decided/expired approval; fresh guest liveness is required. API approve/deny commit first, then attempt a bounded wake without changing the decision response. Automatic suspension defaults off via `microvm_approval_suspend_enabled`; disabling new sleep preserves wake and cleanup. See the [supervisor runbook](../verification/645-p3-supervisor.md). Unanswered approvals have no deadline by default. A checkpointed MicroVM wait can -retire after an hour, or before that worker's lifetime ends. The coordinator -verifies immutable storage, fences the old attempt, confirms shutdown and then -releases capacity. A saved decision admits one replacement and invokes the -original published coordinator version. A scheduled manager retries lost signals -and performs terminal cleanup. See the [continuation protocol](../verification/645-p3-continuation-protocol-20260917.md) -for ownership, capacity, expiry and failure behavior. +retire after an hour, or before that worker's lifetime ends. Retirement and +replacement follow the [retained approval protocol](#retained-microvm-approvals). `TERMINATED` is the normal terminal signal and remains observable for at least 10 minutes. `ResourceNotFoundException` maps to completion only as a late fallback after the control-plane record is eventually reaped; polling does not wait for `NotFound`. **`/ping` health endpoint (AgentCore only).** The agent's FastAPI server responds to AgentCore's `/ping` calls while the coding task runs in a separate thread. AgentCore sees `HealthyBusy` and keeps the session alive. +### Retained MicroVM approvals + +A pending approval has no deletion timer by default. Closing its task cancels +unanswered requests and retains the decision history for 90 days. An explicit +positive approval deadline still applies; waking or replacing a worker does not +restart it. The agent decides whether the approved action remains relevant. + +At the approval barrier, the worker saves the conversation, exact pending tool +inputs, workflow context, cumulative usage and Git/workspace archive. The +coordinator verifies the checksummed S3 object versions before fencing the old +attempt through its coordinator-owned `worker-lease#` record. Workers may +read and condition-check that lease but cannot modify it. Only confirmed shutdown +allows the task to become `PARKED` and release its concurrency reservation. + +An answer admits one replacement when capacity permits. Admission, the new lease +and capacity reservation are one DynamoDB transaction; a deterministic Durable +execution name deduplicates invocation. The replacement uses the original +published coordinator and exact image version, restores the files/conversation, +and consumes the recorded answer. Remaining cost and turns are the original +allowance minus accumulated usage. Repository-free tasks preserve scratch files +using a private directory and local Git baseline. + +The scheduled continuation manager retries unfinished retirement, missed +dispatches and terminal cleanup, saving its scan cursor between invocations. +It cannot launch, suspend or resume workers; launching stays with the pinned +coordinator. A failed start response does not prove no worker exists. Capacity +remains reserved until shutdown is confirmed, or the full service lifetime has +elapsed for an unknown handle. The exact-attempt lease becomes `CLOSED` before +atomic release; `TERMINATING` alone is insufficient. + ### The idle timeout problem AgentCore terminates sessions after 15 minutes of inactivity. Since coding tasks may have long pauses between tool calls (builds, complex reasoning), the agent uses `add_async_task` to register background work. The SDK reports `HealthyBusy` via `/ping` while any async task is active, preventing idle termination. diff --git a/docs/src/content/docs/architecture/Orchestrator.md b/docs/src/content/docs/architecture/Orchestrator.md index ff67cd3c4..531312b71 100644 --- a/docs/src/content/docs/architecture/Orchestrator.md +++ b/docs/src/content/docs/architecture/Orchestrator.md @@ -303,17 +303,43 @@ When the session is unhealthy, the task transitions to `FAILED` with "Agent sess The P3 supervisor saves intent before control calls and rechecks the gate before and after them. Its durable state retains an absolute service lifetime, consecutive failures, recovery start time and next delay. Three failed cycles or 120 seconds of unconfirmed wake cannot become an indefinite wait. AWS RUNNING does not end recovery while the guest remains stuck on a decided/expired approval; fresh guest liveness is required. API approve/deny commit first, then attempt a bounded wake without changing the decision response. Automatic suspension defaults off via `microvm_approval_suspend_enabled`; disabling new sleep preserves wake and cleanup. See the [supervisor runbook](/sample-autonomous-cloud-coding-agents/architecture/645-p3-supervisor). Unanswered approvals have no deadline by default. A checkpointed MicroVM wait can -retire after an hour, or before that worker's lifetime ends. The coordinator -verifies immutable storage, fences the old attempt, confirms shutdown and then -releases capacity. A saved decision admits one replacement and invokes the -original published coordinator version. A scheduled manager retries lost signals -and performs terminal cleanup. See the [continuation protocol](/sample-autonomous-cloud-coding-agents/architecture/645-p3-continuation-protocol-20260917) -for ownership, capacity, expiry and failure behavior. +retire after an hour, or before that worker's lifetime ends. Retirement and +replacement follow the [retained approval protocol](#retained-microvm-approvals). `TERMINATED` is the normal terminal signal and remains observable for at least 10 minutes. `ResourceNotFoundException` maps to completion only as a late fallback after the control-plane record is eventually reaped; polling does not wait for `NotFound`. **`/ping` health endpoint (AgentCore only).** The agent's FastAPI server responds to AgentCore's `/ping` calls while the coding task runs in a separate thread. AgentCore sees `HealthyBusy` and keeps the session alive. +### Retained MicroVM approvals + +A pending approval has no deletion timer by default. Closing its task cancels +unanswered requests and retains the decision history for 90 days. An explicit +positive approval deadline still applies; waking or replacing a worker does not +restart it. The agent decides whether the approved action remains relevant. + +At the approval barrier, the worker saves the conversation, exact pending tool +inputs, workflow context, cumulative usage and Git/workspace archive. The +coordinator verifies the checksummed S3 object versions before fencing the old +attempt through its coordinator-owned `worker-lease#` record. Workers may +read and condition-check that lease but cannot modify it. Only confirmed shutdown +allows the task to become `PARKED` and release its concurrency reservation. + +An answer admits one replacement when capacity permits. Admission, the new lease +and capacity reservation are one DynamoDB transaction; a deterministic Durable +execution name deduplicates invocation. The replacement uses the original +published coordinator and exact image version, restores the files/conversation, +and consumes the recorded answer. Remaining cost and turns are the original +allowance minus accumulated usage. Repository-free tasks preserve scratch files +using a private directory and local Git baseline. + +The scheduled continuation manager retries unfinished retirement, missed +dispatches and terminal cleanup, saving its scan cursor between invocations. +It cannot launch, suspend or resume workers; launching stays with the pinned +coordinator. A failed start response does not prove no worker exists. Capacity +remains reserved until shutdown is confirmed, or the full service lifetime has +elapsed for an unknown handle. The exact-attempt lease becomes `CLOSED` before +atomic release; `TERMINATING` alone is insufficient. + ### The idle timeout problem AgentCore terminates sessions after 15 minutes of inactivity. Since coding tasks may have long pauses between tool calls (builds, complex reasoning), the agent uses `add_async_task` to register background work. The SDK reports `HealthyBusy` via `/ping` while any async task is active, preventing idle termination. From 7891b0bbf8235ba074fedfce97e6b900a8ec1e5d Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Fri, 18 Sep 2026 09:43:42 -0400 Subject: [PATCH 079/149] docs(microvm): replace investigation notebooks with focused guidance --- agent/README.md | 7 +- cdk/scripts/package-microvm-artifact.sh | 25 +- cdk/src/constructs/lambda-microvm-compute.ts | 9 +- .../strategies/lambda-microvm-strategy.ts | 38 +- .../constructs/lambda-microvm-compute.test.ts | 27 +- ...ADR-021-lambda-microvms-compute-backend.md | 18 +- docs/design/CEDAR_HITL_GATES.md | 8 +- docs/design/COMPUTE.md | 4 +- docs/design/ORCHESTRATOR.md | 2 +- docs/design/SECURITY.md | 8 +- docs/guides/USER_GUIDE.md | 5 +- .../docs/architecture/Cedar-hitl-gates.md | 8 +- docs/src/content/docs/architecture/Compute.md | 4 +- .../content/docs/architecture/Orchestrator.md | 2 +- .../src/content/docs/architecture/Security.md | 8 +- ...Adr-021-lambda-microvms-compute-backend.md | 18 +- .../docs/using/Approval-gates-cedar-hitl.md | 5 +- .../verification/645-capacity-reservations.md | 87 - docs/verification/645-coordinator-metadata.md | 77 - .../645-effective-iam-20260915.md | 109 - .../645-lambda-microvm-service-feedback.md | 495 +--- docs/verification/645-lifecycle-intent.md | 85 - .../645-microvm-image-rebuild-20260914.md | 134 - .../645-p1-lambda-microvm-runbook.md | 2198 ----------------- .../645-p2-clean-deployment-20260913.md | 351 --- .../verification/645-p2-live-task-20260914.md | 219 -- .../645-p2-payload-live-20260914.md | 199 -- .../645-p2-repository-config-20260914.md | 147 -- docs/verification/645-p2-smoke-runbook.md | 1905 -------------- .../645-p2-start-recovery-live-20260914.md | 209 -- .../verification/645-p3-agentcore-20260916.md | 110 - .../645-p3-agentcore-permissions-20260917.md | 108 - .../645-p3-approval-ux-20260917.md | 267 -- .../645-p3-callback-live-20260915.md | 226 -- docs/verification/645-p3-callback-timeout.md | 83 - .../645-p3-capacity-scan-20260916.md | 84 - .../645-p3-capacity-upgrade-20260916.md | 100 - .../645-p3-cloud-continuation-20260917.md | 129 - .../645-p3-command-races-20260916.md | 195 -- ...45-p3-connection-close-rollout-20260916.md | 196 -- .../645-p3-continuation-protocol-20260917.md | 77 - docs/verification/645-p3-credentials.md | 143 -- .../645-p3-diagnostics-rollout-20260916.md | 159 -- .../645-p3-durable-live-20260915.md | 295 --- .../645-p3-final-image-and-ecs-20260917.md | 249 -- docs/verification/645-p3-guest-barrier.md | 123 - docs/verification/645-p3-image-capability.md | 96 - .../645-p3-implementation-plan.md | 209 -- .../645-p3-lifecycle-diagnostics.md | 223 +- docs/verification/645-p3-lifecycle-hooks.md | 159 -- .../645-p3-listener-probe-20260916.md | 113 - .../645-p3-live-deployment-20260915.md | 253 -- .../645-p3-mcp-network-20260917.md | 160 -- docs/verification/645-p3-nested-stack.md | 404 +-- .../645-p3-normal-closure-20260918.md | 187 -- .../645-p3-normal-rollout-review-20260917.md | 137 - docs/verification/645-p3-pending-wake.md | 120 - .../645-p3-pid1-observer-20260916.md | 151 -- .../645-p3-process-observer-20260916.md | 187 -- .../645-p3-progress-race-20260918.md | 66 - docs/verification/645-p3-readiness-review.md | 187 -- .../645-p3-registration-20260916.md | 113 - .../645-p3-repository-path-20260917.md | 136 - .../645-p3-resume-refusal-investigation.md | 225 -- .../645-p3-session-recovery-20260917.md | 179 -- docs/verification/645-p3-supervisor.md | 233 -- .../645-p3-transport-control-20260916.md | 87 - .../645-p3-user-sleep-20260916.md | 148 -- .../645-p3-wake-feedback-20260916.md | 74 - .../645-p3-wake-transport-20260916.md | 265 -- .../645-p3-workspace-recovery-20260917.md | 134 - docs/verification/645-payload-bootstrap.md | 211 +- docs/verification/README.md | 91 + 73 files changed, 434 insertions(+), 13069 deletions(-) delete mode 100644 docs/verification/645-capacity-reservations.md delete mode 100644 docs/verification/645-coordinator-metadata.md delete mode 100644 docs/verification/645-effective-iam-20260915.md delete mode 100644 docs/verification/645-lifecycle-intent.md delete mode 100644 docs/verification/645-microvm-image-rebuild-20260914.md delete mode 100644 docs/verification/645-p1-lambda-microvm-runbook.md delete mode 100644 docs/verification/645-p2-clean-deployment-20260913.md delete mode 100644 docs/verification/645-p2-live-task-20260914.md delete mode 100644 docs/verification/645-p2-payload-live-20260914.md delete mode 100644 docs/verification/645-p2-repository-config-20260914.md delete mode 100644 docs/verification/645-p2-smoke-runbook.md delete mode 100644 docs/verification/645-p2-start-recovery-live-20260914.md delete mode 100644 docs/verification/645-p3-agentcore-20260916.md delete mode 100644 docs/verification/645-p3-agentcore-permissions-20260917.md delete mode 100644 docs/verification/645-p3-approval-ux-20260917.md delete mode 100644 docs/verification/645-p3-callback-live-20260915.md delete mode 100644 docs/verification/645-p3-callback-timeout.md delete mode 100644 docs/verification/645-p3-capacity-scan-20260916.md delete mode 100644 docs/verification/645-p3-capacity-upgrade-20260916.md delete mode 100644 docs/verification/645-p3-cloud-continuation-20260917.md delete mode 100644 docs/verification/645-p3-command-races-20260916.md delete mode 100644 docs/verification/645-p3-connection-close-rollout-20260916.md delete mode 100644 docs/verification/645-p3-continuation-protocol-20260917.md delete mode 100644 docs/verification/645-p3-credentials.md delete mode 100644 docs/verification/645-p3-diagnostics-rollout-20260916.md delete mode 100644 docs/verification/645-p3-durable-live-20260915.md delete mode 100644 docs/verification/645-p3-final-image-and-ecs-20260917.md delete mode 100644 docs/verification/645-p3-guest-barrier.md delete mode 100644 docs/verification/645-p3-image-capability.md delete mode 100644 docs/verification/645-p3-implementation-plan.md delete mode 100644 docs/verification/645-p3-lifecycle-hooks.md delete mode 100644 docs/verification/645-p3-listener-probe-20260916.md delete mode 100644 docs/verification/645-p3-live-deployment-20260915.md delete mode 100644 docs/verification/645-p3-mcp-network-20260917.md delete mode 100644 docs/verification/645-p3-normal-closure-20260918.md delete mode 100644 docs/verification/645-p3-normal-rollout-review-20260917.md delete mode 100644 docs/verification/645-p3-pending-wake.md delete mode 100644 docs/verification/645-p3-pid1-observer-20260916.md delete mode 100644 docs/verification/645-p3-process-observer-20260916.md delete mode 100644 docs/verification/645-p3-progress-race-20260918.md delete mode 100644 docs/verification/645-p3-readiness-review.md delete mode 100644 docs/verification/645-p3-registration-20260916.md delete mode 100644 docs/verification/645-p3-repository-path-20260917.md delete mode 100644 docs/verification/645-p3-resume-refusal-investigation.md delete mode 100644 docs/verification/645-p3-session-recovery-20260917.md delete mode 100644 docs/verification/645-p3-supervisor.md delete mode 100644 docs/verification/645-p3-transport-control-20260916.md delete mode 100644 docs/verification/645-p3-user-sleep-20260916.md delete mode 100644 docs/verification/645-p3-wake-feedback-20260916.md delete mode 100644 docs/verification/645-p3-wake-transport-20260916.md delete mode 100644 docs/verification/645-p3-workspace-recovery-20260917.md create mode 100644 docs/verification/README.md diff --git a/agent/README.md b/agent/README.md index 4a045ab9e..4d040f34b 100644 --- a/agent/README.md +++ b/agent/README.md @@ -145,8 +145,7 @@ sets `ABCA_MICROVM_CREDENTIAL_BROKER=1` **only in the Claude child**, points clears alternate credential sources in that child. Operators should not set this internal flag themselves. The endpoint serves the current task's scoped session; the parent's runtime credentials and other backends' attribution path remain -separate. See [P3 credential verification](../docs/verification/645-p3-credentials.md) -for implementation and live-verification limits. +separate. See [recorded credential-renewal acceptance](../docs/verification/README.md#recorded-acceptance). ### Examples @@ -253,7 +252,7 @@ Baked secrets are **reported, not enforced**: `warnings` lists the names (never `microvmId` is parsed defensively and **arrives empty in practice**: the service sends `""` here, unlike `/run` where it is populated (live-verified, ADR-021 P2-F8). So an empty id is expected-normal, not a degraded read — and this hook therefore **cannot** join the guest's record to the control-plane one. `/run`'s `hook accepted task_id=… microvm_id=…` line carries that correlation; `/terminate`'s value is the pipeline-state snapshot it reports. -**`POST /aws/lambda-microvms/runtime/v1/run`** — Authenticate and download a task, install `platform_config` (below), start the pipeline in a background thread, and return 200 inside the hook budget. Protocol v2 is deployed for MicroVM; [11 live transport/failure cases](../docs/verification/645-p2-payload-live-20260914.md) verified worker downloads/rejections, URL expiry/revocation and immediate launch replay. The [runbook](../docs/verification/645-payload-bootstrap.md) tracks the broader authorization and recovery matrix separately. +**`POST /aws/lambda-microvms/runtime/v1/run`** — Authenticate and download a task, install `platform_config` (below), start the pipeline in a background thread, and return 200 inside the hook budget. See the [payload contract and upgrade checks](../docs/verification/645-payload-bootstrap.md) and [recorded live acceptance](../docs/verification/README.md#recorded-acceptance). `runHookPayload` is a JSON **string** passed through by `RunMicrovm`, containing: @@ -307,7 +306,7 @@ Values are **non-secret configuration only**. Credentials are fetched at task st Rejections are structured so they are readable in the MicroVM log group: `400 MICROVM_RUN_PAYLOAD_INVALID` (unusable envelope — retrying the same body cannot help), `500 MICROVM_RUN_PAYLOAD_UNREADABLE` (manifest/payload read or stored bytes failed), `400 MICROVM_RUN_PLATFORM_CONFIG_INVALID` (key off the allowlist, non-object block, or non-string value — fix the producer), `400 MICROVM_RUN_PLATFORM_CONFIG_INCOMPLETE` (a required key missing or blank — fix the deployment wiring), `400 TASK_RECORD_INCOMPLETE` (same validator and vocabulary as `/invocations`). -`/suspend` and `/resume` are served by `microvm_http.py` and declared in managed images with 30-second service timeouts and a shared non-secret protocol marker. Pause requires an active original approval gate, matching coordinator intent, drained activity and an acknowledged checkpoint; wake renews credentials and atomically rechecks the original task/gate before allowing coding. Each handler has a 20-second total budget. `/validate` rejects a supplied incompatible image marker without contacting AWS. The coordinator checks the actual launched image version and persists support on that worker; missing support disables new suspension. See [image capability verification](../docs/verification/645-p3-image-capability.md) and [completed P3 verification](../docs/verification/645-p3-nested-stack.md). +`/suspend` and `/resume` are served by `microvm_http.py` and declared in managed images with 30-second service timeouts and a shared non-secret protocol marker. Pause requires an active original approval gate, matching coordinator intent, drained activity and an acknowledged checkpoint; wake renews credentials and atomically rechecks the original task/gate before allowing coding. Each handler has a 20-second total budget. `/validate` rejects a supplied incompatible image marker without contacting AWS. The coordinator checks the actual launched image version and persists support on that worker; missing support disables new suspension. See the [acceptance checklist](../docs/verification/README.md#live-acceptance-for-an-installation) and [nested deployment prerequisites](../docs/verification/645-p3-nested-stack.md). #### Conversation and workspace continuation diff --git a/cdk/scripts/package-microvm-artifact.sh b/cdk/scripts/package-microvm-artifact.sh index 70b989498..579f1cbeb 100755 --- a/cdk/scripts/package-microvm-artifact.sh +++ b/cdk/scripts/package-microvm-artifact.sh @@ -84,11 +84,10 @@ # contracts/ <- cross-language constants the agent reads at runtime # # --------------------------------------------------------------------------- -# P2 VERIFICATION STATUS +# VERIFICATION # --------------------------------------------------------------------------- -# The 2026-09-14 clean deployment and coding/iteration/cancellation runs passed -# without the earlier manual IAM workaround. The wider failure/security matrix -# remains open; see docs/verification/645-p2-live-task-20260914.md. +# Test the selected image/coordinator together before enabling automatic sleep. +# See docs/verification/README.md for acceptance criteria and remaining PR checks. # # ADR-021 sub-decision 3's hook-phasing table (corrected after the live P1 # verification run, then completed in P2) is now: @@ -265,16 +264,14 @@ echo " log group : ${LOG_GROUP}" print_p1_reminder() { cat <<'EOF' -REMINDER (ADR-021 P2): clean deployment and coding, iteration and cancellation - passed on 2026-09-14 with bootstrap bundle 1.7.0, without manual IAM changes. - Ready/validate hooks, heartbeat, logging, Memory writes and cleanup have live - evidence. Full P2 acceptance still needs the failure/recovery, effective IAM - and networking matrix in docs/verification/645-p3-implementation-plan.md. - /suspend and /resume are declared; supervisor integration is implemented. - P3 requires bootstrap bundle 1.8.0 and defaults new suspension off. - Live P3 acceptance remains open. - CDK retains warning ID abca:microvm-image-p1-smoke-unverified for compatibility; - its text describes the current verification gaps. +REMINDER (ADR-021): verify the deployed image and coordinator together before + enabling automatic suspension. Managed images declare all six hooks; + compatible coordinator, IAM and image configuration are required. + Nested deployments require bootstrap bundle 1.9.0 or later. Existing flat + deployments need the staged migration in docs/verification/645-p3-nested-stack.md. + Acceptance criteria and remaining PR checks: docs/verification/README.md. + Preserve a compatible published coordinator and explicit image pin for rollback. + CDK retains warning ID abca:microvm-image-p1-smoke-unverified for compatibility. EOF } diff --git a/cdk/src/constructs/lambda-microvm-compute.ts b/cdk/src/constructs/lambda-microvm-compute.ts index 2aee952c0..6d7303df7 100644 --- a/cdk/src/constructs/lambda-microvm-compute.ts +++ b/cdk/src/constructs/lambda-microvm-compute.ts @@ -379,7 +379,7 @@ export interface LambdaMicrovmComputeProps extends LambdaMicrovmImageInputs { * * Managed images enable all six served hooks. Automatic approval sleep requires * a compatible image and coordinator plus the deployment enable switch. P3 live - * acceptance is recorded in docs/verification/645-p3-implementation-plan.md; new + * acceptance is recorded in docs/verification/README.md; new * installations must verify their own configuration before enabling sleep. */ export class LambdaMicrovmCompute extends Construct { @@ -893,7 +893,7 @@ export class LambdaMicrovmCompute extends Construct { + '/run, /terminate, /suspend and /resume; managed images declare all six. ' + 'Nested deployments require bundle 1.9.0 and a reviewed migration from existing flat stacks. ' + 'Preserve a compatible coordinator and explicit image version for rollback. Follow ' - + 'docs/verification/645-p3-implementation-plan.md and docs/verification/645-p3-nested-stack.md.', + + 'docs/verification/README.md and docs/verification/645-p3-nested-stack.md.', ); } @@ -963,9 +963,8 @@ export class LambdaMicrovmCompute extends Construct { /** * Scope log writes to the MicroVM namespace. Only the build role can create log - * groups; runtime logs use the pre-created image group. See - * docs/verification/645-p3-callback-live-20260915.md for the verified policy and - * log inventory. Investigate a specific runtime denial before widening the grant. + * groups; runtime logs use the pre-created image group. Investigate a specific + * runtime denial before widening the grant. */ private grantMicrovmLogWrites( role: iam.IRole, diff --git a/cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts b/cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts index 1fb66ca57..cf07439a0 100644 --- a/cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts +++ b/cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts @@ -132,27 +132,9 @@ export const MICROVM_MAX_DURATION_SECONDS = sharedConstants.microvm_lifecycle.ma const RUN_HOOK_PAYLOAD_LIMIT_BYTES = 4_096; /** - * The ``GetMicrovm`` ``stateReason`` value that means "nothing to report". - * - * Live-observed, not guessed: an orchestrator-initiated ``TerminateMicrovm`` on the - * SUCCESS path leaves the MicroVM ``TERMINATED`` with exactly - * ``stateReason: "Success."`` — trailing period included. Recorded three times in - * ``docs/verification/645-p2-smoke-runbook.md``: **§5.1** ("Finalization called - * `TerminateMicrovm`", the verbatim CLI output), **§6.2** (the suspend/resume - * latency table) and **§2.9** ("Lifecycle — PASS", run 2). A healthy ``RUNNING`` - * MicroVM reports no reason at all (``None``, same §6.2 table). - * - * Normalized away in {@link LambdaMicrovmComputeStrategy.pollSession} so it never - * reaches the reconcile ``detail`` string, where it would append noise to every - * cleanly-finished task. - * - * A bare literal comparison is deliberate and the brittleness is bounded: this is - * a service-owned display string, so an exact match can only fail OPEN — a future - * ``"Success"`` without the period, or a different capitalisation, would leak one - * benign phrase into an operator-facing string. It cannot suppress a real reason, - * which is the direction that would matter. Left out of - * ``contracts/constants.json`` for the same reason: nothing in the agent reads it, - * so it is not a cross-language contract. + * The service reports "Success." after normal termination. Suppress only that + * exact benign display string; preserve other reasons for operator diagnosis. + * A future wording change may add harmless detail but cannot hide a failure. */ const MICROVM_BENIGN_STATE_REASON = 'Success.'; @@ -733,17 +715,9 @@ export class LambdaMicrovmComputeStrategy implements ComputeStrategy { * ``microvmState`` also reports the explicit observed state (or local UNKNOWN / * NOT_FOUND). P3 uses it to distinguish readiness from the coarse status. * - * ``stateReason`` is carried through on every mapped state as - * ``SessionStatus.reason``, VERBATIM and uninterpreted. It is the substrate's - * own account of WHY, and mapping it away is what made the dominant runtime - * failure unreadable: a ``/run`` hook 4xx self-terminates the VM within ~12 s - * (``docs/verification/645-p2-smoke-runbook.md`` §6.1) with - * ``stateReason = "Run lifecycle hook returned HTTP status 400. Please check - * your hook endpoint and application logs for more details."`` — and because - * ``TERMINATED → completed`` has no error slot, the orchestrator's reconcile - * detail read ``"substrate state completed"``, naming none of the three causes - * its remedy suggested. Reporting the reason keeps this method mechanical (no - * branch reads it) while giving the orchestrator something true to say. + * Service failure reasons are preserved in SessionStatus.reason so a failed + * hook is not reduced to "substrate completed". Suppress only the known benign + * success string; the orchestrator decides whether the task itself succeeded. */ async pollSession(handle: SessionHandle, options?: SessionControlOptions): Promise { if (handle.strategyType !== 'lambda-microvm') { diff --git a/cdk/test/constructs/lambda-microvm-compute.test.ts b/cdk/test/constructs/lambda-microvm-compute.test.ts index 7617ce303..10a9ccde8 100644 --- a/cdk/test/constructs/lambda-microvm-compute.test.ts +++ b/cdk/test/constructs/lambda-microvm-compute.test.ts @@ -674,29 +674,10 @@ describe('LambdaMicrovmCompute — image provisioned from a managed base image', }); test('NO source-key condition on any MicroVM-facing role trust (P2-F1/F3)', () => { - // The sharpest IAM assertion in this file, and the one most likely to be - // "fixed" back by a reviewer applying the standard service-principal - // confused-deputy pattern. It must not be. - // - // The Lambda MicroVMs service presents NO source condition key when it assumes - // these roles, so a trust policy carrying one is unassumable. Live 2026-08-06/07 - // (evidence inlined in ADR-021 §4; `docs/verification/645-p2-smoke-runbook.md` - // is the raw session log, additional detail rather than the sole proof), one - // root cause, two symptoms: - // both network connectors CREATE_FAILED deterministically on a freshly deleted - // stack (P2-F1), and RunMicrovm reported a MISLEADING caller-side - // `iam:PassRole` AccessDenied on the orchestrator (P2-F3) — with the grant - // present, `simulate-principal-policy` returning `allowed`, no permissions - // boundary, and an unconditioned PassRole ALSO denied. Removing the execution - // role's trust conditions made the next submission reach RUNNING in 6 s. - // - // What compensates is asserted elsewhere in this file and in - // `test/constructs/task-orchestrator.test.ts`: the EXECUTION role is passable by - // the orchestrator only, scoped to its EXACT ARN — and with NO - // `iam:PassedToService` condition either, because the same missing-context-key - // root cause blocks that path too (P2r2-F10), which is why the exact ARN is the - // whole of the scoping. Every resource these roles reach is account-scoped by - // ARN apart from two justified `Resource: '*'` statements. + // Recorded service calls rejected source-conditioned role trust; removing + // those conditions restored connector creation and worker launch. Keep exact + // resource grants and verify service support before adding conditions again. + // See ADR-021 §4 and docs/verification/645-lambda-microvm-service-feedback.md. const roles = Object.entries(template.findResources('AWS::IAM::Role')) .filter(([id]) => id.includes('LambdaMicrovmComputeBuildRole') || id.includes('LambdaMicrovmComputeExecutionRole') diff --git a/docs/decisions/ADR-021-lambda-microvms-compute-backend.md b/docs/decisions/ADR-021-lambda-microvms-compute-backend.md index 476a9a803..b4ab48f30 100644 --- a/docs/decisions/ADR-021-lambda-microvms-compute-backend.md +++ b/docs/decisions/ADR-021-lambda-microvms-compute-backend.md @@ -1,6 +1,6 @@ # ADR-021: AWS Lambda MicroVMs as a third ComputeStrategy backend -> **Implementation status (2026-09-18): P1 and P2 are merged; P3 implementation and normal deployment acceptance are complete.** P3 adds approval sleep/wake, retained requests, conversation/workspace recovery, replacement workers and nested infrastructure. See the [completed plan](../verification/645-p3-implementation-plan.md) and [normal acceptance record](../verification/645-p3-normal-closure-20260918.md). This ADR defines P1–P3; it does not define an official P4. +> **Implementation status (2026-09-18): P1 and P2 are merged; P3 is in draft review.** Approval sleep/wake, retained requests, conversation/workspace recovery and nested infrastructure have live acceptance evidence. Reusable migration and final integration checks remain open; see [verification status](../verification/README.md). This ADR defines P1–P3, not an official P4. **Status:** proposed **Date:** 2026-07-29 @@ -24,7 +24,7 @@ Lambda MicroVMs are managed Firecracker virtual machines, separate from ordinary See [COMPUTE.md](../design/COMPUTE.md) for the full comparison and costs. Suspending stops compute charges, but snapshot storage and save/restore charges remain. A shorter sleep delay does not guarantee lower total cost. -The [P1 probes](../verification/645-p1-lambda-microvm-runbook.md) established a 4,096-byte `runHookPayload` limit, an image-ARN requirement and accepted baseline values of 512, 1,024, 2,048, 4,096 and 8,192 MiB. Those observations override conflicting generated SDK descriptions for the tested account/Region. They do not measure guest-visible launch memory, vertical-scaling latency or sustained workload fit. The service guide’s capacity figures and live observations must remain distinguishable. +Recorded P1 probes established a 4,096-byte `runHookPayload` limit, an image-ARN requirement and accepted baseline values of 512, 1,024, 2,048, 4,096 and 8,192 MiB. Those observations override conflicting generated SDK descriptions for the tested account/Region. They do not measure guest-visible launch memory, vertical-scaling latency or sustained workload fit. The service guide’s capacity figures and live observations must remain distinguishable. ### Design tensions the strategy must resolve @@ -70,11 +70,11 @@ Unanswered approvals have no deadline by default (`approval_timeout_s=0`). Expli Automatic suspension also requires the deployment’s `microvm_approval_suspend_enabled` opt-in, which defaults false for new deployments. A live Parameter Store switch lets existing durable executions stop initiating new suspensions without changing their pinned Lambda environment. The verified normal deployment has this opt-in enabled. Turning it off does not abandon already-suspended workers. -The coordinator observes the exact pending gate and waits for the sleep delay. It skips sleep when a timed gate has too little time left before its wake margin. The guest holds a coding barrier, drains acknowledged progress and commits a checkpoint before accepting `/suspend`. Lifecycle HTTP responses explicitly close their connections before freeze to avoid reuse of a stale pooled connection; see the [transport experiment](../verification/645-p3-wake-transport-20260916.md). +The coordinator observes the exact pending gate and waits for the sleep delay. It skips sleep when a timed gate has too little time left before its wake margin. The guest holds a coding barrier, drains acknowledged progress and commits a checkpoint before accepting `/suspend`. Lifecycle HTTP responses explicitly close their connections before freeze to avoid reuse of a stale pooled connection; see the [transport evidence summary](../verification/README.md#recorded-acceptance). Approval and denial handlers commit the decision first. They then read the current handle consistently, persist wake intent and request resume best-effort. A wake failure records diagnostics and does not undo the accepted decision. The durable supervisor retries and observes both service state and guest consumption of the decision; RUNNING alone does not prove the tool was released. -A sleeping worker retains its concurrency reservation. For a longer wait, ABCA verifies a complete, version-pinned conversation/workspace checkpoint, fences the attempt, confirms shutdown and releases the reservation. A later decision can admit one replacement through the original published coordinator. It restores Git state, required workspace files, the actual SDK conversation, exact pending tool inputs and cumulative usage with fresh scoped credentials. See the [continuation protocol](../verification/645-p3-continuation-protocol-20260917.md). +A sleeping worker retains its concurrency reservation. For a longer wait, ABCA verifies a complete, version-pinned conversation/workspace checkpoint, fences the attempt, confirms shutdown and releases the reservation. A later decision can admit one replacement through the original published coordinator. It restores Git state, required workspace files, the actual SDK conversation, exact pending tool inputs and cumulative usage with fresh scoped credentials. See the [continuation protocol](../design/ORCHESTRATOR.md#retained-microvm-approvals). Normative requirements (EARS): @@ -104,7 +104,7 @@ All six hooks share the FastAPI listener on port 8080. AWS hook properties accep | `/suspend` | Drain acknowledged progress and commit the current safe checkpoint within the hook budget | | `/resume` | Renew credentials and reconcile the original gate before releasing coding | -Warm-up budgets come from `contracts/constants.json`; the total guest budget must remain below the image hook timeout. The [P2 runbook](../verification/645-p2-smoke-runbook.md) records the cold-binary failure and the enum/trust/logging fixes. Historical binary sizes and timings are observations of those images, not sizing guarantees for later builds. +Warm-up budgets come from `contracts/constants.json`; the total guest budget must remain below the image hook timeout. Cold-binary startup exceeded the hook budget in an earlier image; warm-up moved this work into image preparation. Historical sizes and timings are not sizing guarantees for later builds. **Authenticated v2 payload transport.** Every ECS and MicroVM task uses S3. The coordinator publishes a deployment manifest, conditionally creates the task payload and sends a short-lived signed URL for that one object. The serialized MicroVM reference must fit 4,096 bytes; payloads are bounded at 8 MiB and manifests at 16 KiB. @@ -127,7 +127,7 @@ Normative requirements (EARS): - Signed URLs shall not appear in ordinary logs, agent-readable task rows or repository subprocess environments. Downloads shall use the exact regional S3 HTTPS object without redirects/proxies and with bounded response sizes. - Producers, images and IAM shall be upgraded together; incompatible workers must be drained before switching transport. -The [payload contract](../verification/645-payload-bootstrap.md) and [live checks](../verification/645-p2-payload-live-20260914.md) record the implementation and validation. This transport does not establish complete hostile-worker isolation: other platform grants remain, and a stolen signed URL is usable until expiry or revocation. +The [payload contract](../verification/645-payload-bootstrap.md) and [live checks](../verification/README.md) record the implementation and validation. This transport does not establish complete hostile-worker isolation: other platform grants remain, and a stolen signed URL is usable until expiry or revocation. No ABCA endpoint consumer exists in P1–P3. The platform grants no `CreateMicrovmAuthToken` permission and mints no JWE tokens. `NO_INGRESS` can still return an endpoint URL; an unauthenticated 403 verifies the authentication boundary, not valid-token reachability. @@ -135,7 +135,7 @@ No ABCA endpoint consumer exists in P1–P3. The platform grants no `CreateMicro The backend adds build/runtime VPC connectors, build artifacts, launch payloads, logs, roles and a managed or external image. Its bootstrap policy is conditional on `ComputeTypes` including `lambda-microvm`. A VPC egress connector requires an operator role. Build egress permits ports 80/443 for package installation; runtime egress permits 443 through the platform VPC. -**Trust and PassRole limitation.** Recorded live checks rejected `aws:SourceAccount`/`aws:SourceArn` conditions on the MicroVM-facing roles and `iam:PassedToService` on the MicroVM PassRole paths. The working roles trust `lambda.amazonaws.com` without those conditions; build/execution roles also allow `sts:TagSession`. IAM simulation with caller-supplied condition values did not prove that the service supplied those values. See the [P2 evidence](../verification/645-p2-smoke-runbook.md), [clean deployment](../verification/645-p2-clean-deployment-20260913.md) and [effective IAM checks](../verification/645-effective-iam-20260915.md). Reintroduce a condition only after verifying service support. +**Trust and PassRole limitation.** Recorded live checks rejected `aws:SourceAccount`/`aws:SourceArn` conditions on the MicroVM-facing roles and `iam:PassedToService` on the MicroVM PassRole paths. The working roles trust `lambda.amazonaws.com` without those conditions; build/execution roles also allow `sts:TagSession`. IAM simulation with caller-supplied condition values did not prove that the service supplied those values. See the [IAM service questions](../verification/645-lambda-microvm-service-feedback.md#iam-setup-f02). Reintroduce a condition only after verifying service support. | Role/action | Scope and responsibility | |---|---| @@ -149,7 +149,7 @@ The backend adds build/runtime VPC connectors, build artifacts, launch payloads, `lambda:PassNetworkConnector` has no resource-level authorization support and therefore uses `Resource: *`. The operator role also has wildcard ENI permissions; some mutations can be scoped in IAM, so their current tested wildcard is not evidence that narrower permissions are impossible. DescribeAvailabilityZones needs a wildcard for fresh CDK repository lookups. These exceptions and namespace wildcards are documented in the construct’s cdk-nag suppressions. -**Nested infrastructure.** `microvm_nested_stack=true` puts MicroVM resources in a nested stack. The shared execution role stays in the parent to avoid a role-trust dependency cycle. Bootstrap 1.9.0 covers nested deployment roles. Existing flat installations need the reviewed overlap/drain migration, explicit image/coordinator pins and rollback checks in the [nested migration record](../verification/645-p3-nested-stack.md). Moving construct paths alone is not a safe migration. Preserve `microvm_resource_name_prefix` after a migrated deployment. +**Nested infrastructure.** `microvm_nested_stack=true` puts MicroVM resources in a nested stack. The shared execution role stays in the parent to avoid a role-trust dependency cycle. Bootstrap 1.9.0 covers nested deployment roles. Existing flat installations need the reviewed overlap/drain migration, explicit image/coordinator pins and rollback checks described in the [migration prerequisites](../verification/645-p3-nested-stack.md). Reusable migration commands are still being completed. Moving construct paths alone is not a safe migration. Preserve `microvm_resource_name_prefix` after a migrated deployment. #### Security bar vs existing backends ([#645](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/645) acceptance criterion) @@ -199,7 +199,7 @@ Changing the default backend, GPU support, native Slack approval buttons, approv ## Testing -The [completed plan](../verification/645-p3-implementation-plan.md) indexes exact suite results and dated evidence. Required coverage includes: +The [verification summary](../verification/README.md) distinguishes recorded live acceptance from open PR checks. Required coverage includes: - Strategy state mapping, explicit unsupported results, ARN/Region validation, NO_INGRESS, omitted idlePolicy and bounded uncertain-start recovery. - Hook readiness/warm-up, AWS-silent build hooks, authenticated payload installation, arbitrary terminate bodies and lifecycle connection closure. diff --git a/docs/design/CEDAR_HITL_GATES.md b/docs/design/CEDAR_HITL_GATES.md index 1bff5676b..f503a3d4c 100644 --- a/docs/design/CEDAR_HITL_GATES.md +++ b/docs/design/CEDAR_HITL_GATES.md @@ -17,10 +17,10 @@ > CLI response instructions are implemented for Slack/Linear notifications; > native channel decisions remain separate work. See the current > [user guide](../guides/USER_GUIDE.md#approval-gates-cedar-hitl) and -> [continuation protocol](../verification/645-p3-continuation-protocol-20260917.md). +> [continuation protocol](./ORCHESTRATOR.md#retained-microvm-approvals). > The normal deployment passed retained-request, ten-minute sleep, explicit-expiry > and sleep-off/rollback acceptance; see the -> [deployment record](../verification/645-p3-normal-closure-20260918.md). +> [deployment record](../verification/README.md). --- @@ -1039,7 +1039,7 @@ event. A later decision is rejected: the existing API returns `404 REQUEST_NOT_FOUND` for missing, foreign or already-decided approval rows, including a cancelled row. If approval committed first, cancellation preserves the recorded decision while cancelling the task. See the -[P3 approval verification record](../verification/645-p3-approval-ux-20260917.md) +[P3 approval verification record](../verification/README.md) for source versus deployment status. ### 7.2 `POST /v1/tasks/{task_id}/deny` @@ -1530,7 +1530,7 @@ path uses the CLI owner's authentication. Native Slack approval buttons and Line approval replies are not implemented by this notification change; the OAuth/button design below remains proposed. Email remains a log-only stub and GitHub does not receive approval messages. Deployment status is recorded in the -[P3 verification record](../verification/645-p3-approval-ux-20260917.md). +[P3 verification record](../verification/README.md). **TaskApprovalsTable Streams are not consumed by the fan-out Lambda**. The approval row is working state; the audit trail is in TaskEventsTable. Enabling Streams on TaskApprovalsTable would be redundant and add noise. Final design: TaskApprovalsTable DOES NOT have Streams enabled. (Retains the `stream` attribute commented out for future use if needed.) diff --git a/docs/design/COMPUTE.md b/docs/design/COMPUTE.md index c34ed2845..1f70a7a12 100644 --- a/docs/design/COMPUTE.md +++ b/docs/design/COMPUTE.md @@ -81,11 +81,11 @@ Lambda MicroVMs are an opt-in third backend, selected per repository with `compu For managed images, run `cdk/scripts/package-microvm-artifact.sh --stack-name ` after each agent change. It packages the Dockerfile's local inputs into a deterministic ZIP and uploads it under `microvm-images/agent-artifact-.zip`. The checksum is a fingerprint of the uploaded bytes: identical inputs reuse the same verified object, and changed inputs produce a new filename. Deploy with the printed `--context microvm_artifact_sha256=` alongside `microvm_base_image_arn` and `microvm_base_image_version`, and retain these inputs for later deployments. The changed S3 URI tells CloudFormation to update the existing image; overwriting the old fixed filename alone does not. Missing or malformed digests fail synthesis. An initial deployment without an image must create the bucket first. The script's explicit `--create-image` alternative retains the fixed base key and calls the image API directly. -Because a snapshot freezes its build-time environment, current deployment identifiers arrive through the v2 payload bootstrap. The coordinator publishes a non-secret deployment manifest and sends a single-object signed download URL for the task. The worker reads only its deployment's `bootstrap/*` with ambient credentials; other object reads and payload-bucket listing are explicitly denied. The downloaded task identity and configuration must match the authenticated manifest before configuration installation. The serialized reference fits the verified 4,096-byte hook limit; all payload sizes use S3. ECS shares this transport through `AGENT_PAYLOAD_REF`. [ADR-021 §3](../decisions/ADR-021-lambda-microvms-compute-backend.md#3-packaging-same-agent-image-source-new-build-path) defines the wire format and compatibility requirements; [live payload checks](../verification/645-p2-payload-live-20260914.md) record validation. +Because a snapshot freezes its build-time environment, current deployment identifiers arrive through the v2 payload bootstrap. The coordinator publishes a non-secret deployment manifest and sends a single-object signed download URL for the task. The worker reads only its deployment's `bootstrap/*` with ambient credentials; other object reads and payload-bucket listing are explicitly denied. The downloaded task identity and configuration must match the authenticated manifest before configuration installation. The serialized reference fits the verified 4,096-byte hook limit; all payload sizes use S3. ECS shares this transport through `AGENT_PAYLOAD_REF`. [ADR-021 §3](../decisions/ADR-021-lambda-microvms-compute-backend.md#3-packaging-same-agent-image-source-new-build-path) defines the wire format and compatibility requirements; [live payload checks](../verification/README.md) record validation. Networking separates image build from execution: the build-only connector permits TCP 80 and 443 because the Dockerfile uses `apt-get`, while running MicroVMs retain 443-only egress through the platform VPC. Every launch explicitly passes the Lambda-managed `NO_INGRESS` connector; omission would select the service's public-ingress default. Images declare and serve `/ready` and `/validate` at build time and `/run`, `/terminate`, `/suspend` and `/resume` at runtime. Automatic suspension is a separate deployment opt-in. Registry HTTP/SSE tools therefore need reachable HTTPS/443 endpoints; remote non-443 tools are unsupported under the default policies of AgentCore and ECS as well. Local `stdio` tools can run, with their outbound traffic subject to the same restriction. Asset resolution/loading does not probe connectivity; see [registry network support](./REGISTRY.md#2-asset-kinds-for-mvp). -P3 connects the guest checkpoint/credential hooks, actual image-version capability, durable supervisor and post-commit approval wake. The `microvm_approval_suspend_enabled` deployment context defaults false and controls both a static opt-in and a live Parameter Store switch. Existing durable executions reread the live switch before new suspension because their original Lambda environment is pinned. Recovery and the original service lifetime survive supervisor replay; the API preserves accepted decisions when optional wake fails. AgentCore/ECS retain explicit unsupported pause/wake results. Explicit approval deadlines use the original UTC/monotonic deadline. See the [supervisor runbook](../verification/645-p3-supervisor.md) for budgets and scoped permissions, and the [completion record](../verification/645-p3-implementation-plan.md) for deployment evidence. +P3 connects the guest checkpoint/credential hooks, actual image-version capability, durable supervisor and post-commit approval wake. The `microvm_approval_suspend_enabled` deployment context defaults false and controls both a static opt-in and a live Parameter Store switch. Existing durable executions reread the live switch before new suspension because their original Lambda environment is pinned. Recovery and the original service lifetime survive supervisor replay; the API preserves accepted decisions when optional wake fails. AgentCore/ECS retain explicit unsupported pause/wake results. Explicit approval deadlines use the original UTC/monotonic deadline. See the [lifecycle diagnostics](../verification/645-p3-lifecycle-diagnostics.md) for failure investigation, and the [acceptance status](../verification/README.md) for deployment evidence. Tasks can set `microvm_sleep_after_s` (CLI: `--microvm-sleep-after `). The default is 600 seconds of waiting for each approval; zero disables sleep. diff --git a/docs/design/ORCHESTRATOR.md b/docs/design/ORCHESTRATOR.md index e6a4fd27a..0ca3d0ad0 100644 --- a/docs/design/ORCHESTRATOR.md +++ b/docs/design/ORCHESTRATOR.md @@ -296,7 +296,7 @@ When the session is unhealthy, the task transitions to `FAILED` with "Agent sess - A terminal substrate report paired with a non-terminal task first checks for a complete, acknowledged approval checkpoint. Such a checkpoint can retire the old attempt and retain the task for a replacement. Without one, finalization strongly re-reads the task row before classifying a substrate failure. - Substrate state detects a dead VM; heartbeat staleness detects loss of the heartbeat writer inside a VM that still reports `RUNNING`. The independent heartbeat thread can continue during a pipeline hang, so a fresh timestamp is not proof of progress. -The P3 supervisor saves intent before control calls and rechecks the gate before and after them. Its durable state retains an absolute service lifetime, consecutive failures, recovery start time and next delay. Three failed cycles or 120 seconds of unconfirmed wake cannot become an indefinite wait. AWS RUNNING does not end recovery while the guest remains stuck on a decided/expired approval; fresh guest liveness is required. API approve/deny commit first, then attempt a bounded wake without changing the decision response. Automatic suspension defaults off via `microvm_approval_suspend_enabled`; disabling new sleep preserves wake and cleanup. See the [supervisor runbook](../verification/645-p3-supervisor.md). +The P3 supervisor saves intent before control calls and rechecks the gate before and after them. Its durable state retains an absolute service lifetime, consecutive failures, recovery start time and next delay. Three failed cycles or 120 seconds of unconfirmed wake cannot become an indefinite wait. AWS RUNNING does not end recovery while the guest remains stuck on a decided/expired approval; fresh guest liveness is required. API approve/deny commit first, then attempt a bounded wake without changing the decision response. Automatic suspension defaults off via `microvm_approval_suspend_enabled`; disabling new sleep preserves wake and cleanup. See the [lifecycle diagnostics](../verification/645-p3-lifecycle-diagnostics.md). Unanswered approvals have no deadline by default. A checkpointed MicroVM wait can retire after an hour, or before that worker's lifetime ends. Retirement and diff --git a/docs/design/SECURITY.md b/docs/design/SECURITY.md index 091e34fac..8d9467d70 100644 --- a/docs/design/SECURITY.md +++ b/docs/design/SECURITY.md @@ -44,11 +44,13 @@ Three authentication mechanisms protect the platform, matching its input channel - **DynamoDB**: item access on the four `task_id`-partitioned tables (task, events, approvals, nudges) is gated by a `dynamodb:LeadingKeys` condition equal to `${aws:PrincipalTag/task_id}` and requires that context key to be present. `Scan` is not granted. The main task table permits reads and only attribute-scoped `UpdateItem` writes: the agent can report status, heartbeat, results and approval details, but cannot replace/delete the row or edit coordinator-owned start receipts, capacity reservations, owner identity or compute handles. The reviewed write list is `cdk/src/constructs/agent-task-write-attributes.json`; Python tests exercise the current writers against it. Supporting tables retain task-scoped item writes. `LeadingKeys` binds the base-table partition key, not a GSI key such as `user_id`. - **S3**: trace writes and attachment reads are scoped to the `/${aws:PrincipalTag/user_id}/` object prefix. -The compute role retains only non-tenant access (Bedrock model invocation — already ARN-scoped; CloudWatch Logs; the GitHub PAT secret, read once before the SessionRole is assumed; AgentCore Memory) plus `sts:AssumeRole`/`sts:TagSession` on the SessionRole. Because the agent runs under credentials that are themselves an assumed role, its `AssumeRole` is *role chaining* — capped at one hour regardless of the role's max session duration — so the agent uses a **refreshable** credential provider that re-assumes before expiry (tasks can run up to the 8-hour `maxLifetime`). The design is backend-agnostic: the same SessionRole and agent code serve all three backends (AgentCore Runtime, ECS Fargate, Lambda MicroVMs). Existing scoped credentials are constrained to their tags, but the compute role chooses those tags when assuming the role. The trust policy does not independently bind a worker to one task; this is not complete isolation against a compromised worker with ambient credentials. No worker needs direct access to the shared capacity counter. ECS's legacy configuration without a session role retains unscoped task reads/reporting updates, with the same attribute restriction. The policy structure and Python writer compatibility are tested locally. A historical IAM simulator check covered the original tag conditions; it does not verify the new attribute restriction or deployed transaction authorization. The repository runbook `docs/verification/645-coordinator-metadata.md` contains the required AWS authorization checks. +The compute role retains only non-tenant access (Bedrock model invocation — already ARN-scoped; CloudWatch Logs; the GitHub PAT secret, read once before the SessionRole is assumed; AgentCore Memory) plus `sts:AssumeRole`/`sts:TagSession` on the SessionRole. Because the agent runs under credentials that are themselves an assumed role, its `AssumeRole` is *role chaining* — capped at one hour regardless of the role's max session duration — so the agent uses a **refreshable** credential provider that re-assumes before expiry (tasks can run up to the 8-hour `maxLifetime`). The design is backend-agnostic: the same SessionRole and agent code serve all three backends (AgentCore Runtime, ECS Fargate, Lambda MicroVMs). Existing scoped credentials are constrained to their tags, but the compute role chooses those tags when assuming the role. The trust policy does not independently bind a worker to one task; this is not complete isolation against a compromised worker with ambient credentials. No worker needs direct access to the shared capacity counter. ECS's legacy configuration without a session role retains unscoped task reads/reporting updates, with the same attribute restriction. The policy structure and Python writer compatibility are tested locally. A historical IAM simulator check covered the original tag conditions; it does not verify the new attribute restriction or deployed transaction authorization. The [acceptance checklist](../verification/README.md#live-acceptance-for-an-installation) includes effective-role authorization checks. -**Lambda MicroVMs compute-role delta** - On this backend the compute role additionally holds prefix-scoped `secretsmanager:GetSecretValue` on the per-workspace channel-OAuth secrets (`bgagent-linear-oauth-*`, `bgagent-jira-oauth-*`), read access only to its payload bucket's `bootstrap/*` manifests (other object reads and payload-bucket listing are explicitly denied), and `ec2:DescribeAvailabilityZones` (`Resource: *`, read-only — EC2 describe actions have no resource-level scoping) for a CDK repo's own synth gate. It is also **the only compute role in the platform whose trust policy carries no confused-deputy condition**: the Lambda MicroVMs service populates no `aws:SourceAccount` or `aws:SourceArn` when it assumes the role, so a trust policy carrying one is unassumable — verified live, not assumed (connector creation failed deterministically, and `RunMicrovm` surfaced the same root cause as a misleading caller-side `iam:PassRole` denial). The compensating controls are that each of the three MicroVM roles can be passed to `lambda.amazonaws.com` only by a named principal — the orchestrator, via an `iam:PassRole` scoped to the execution role's exact ARN, and the CloudFormation deployment role for the build and connector-operator roles — that every other resource they reach is account-scoped by ARN apart from two justified `Resource: *` read/create-time statements, and that none of them holds `iam:*`, cross-account trust, or any `sts:AssumeRole` beyond the execution role's scoped hop to the per-task SessionRole. Full evidence, the two-arm PassRole experiment, and the alternatives considered are in [ADR-021 §4](../decisions/ADR-021-lambda-microvms-compute-backend.md#4-infra-and-iam-conditional-resources-behind-bootstrap-computetypes). +**Lambda MicroVMs compute-role delta** — The compute role additionally reads the per-workspace channel-OAuth secrets (`bgagent-linear-oauth-*`, `bgagent-jira-oauth-*`) and its payload bucket's `bootstrap/*` manifests. Ambient task-object reads and payload-bucket listing are explicitly denied. It also has read-only `ec2:DescribeAvailabilityZones` for repository CDK synthesis; that API has no resource-level scope. -**ECS/MicroVM task bootstrap (v2)** — The coordinator publishes non-secret deployment manifests and supplies a short-lived signed URL for exactly one task's payload. The worker reads the manifest with its ambient role, whose explicit deny outside its own `bootstrap/*` prevents a foreign public bucket from authorizing fake configuration. It verifies downloaded task identity and exact manifest/config agreement before installing MicroVM settings. The coordinator privately persists the URL outside TaskTable for retry and deletes both task objects at finalization. Worker roles cannot read/list task objects using their ambient credentials. A signed URL is a bearer capability: redact it in errors and keep it out of task rows and repository subprocess environments. Existing session-tag trust and other platform grants remain separate limitations. This protocol is implemented locally; real IAM/network/expiry gates and the coordinated image/coordinator/policy rollout are documented in the repository's `docs/verification/645-payload-bootstrap.md`. +Recorded live calls rejected source-conditioned MicroVM role trust and service-conditioned PassRole grants. The integration therefore omits those conditions on the affected paths, while restricting which roles can be passed and which resources they may access. Reintroduce conditions only after verifying service support. [ADR-021 §4](../decisions/ADR-021-lambda-microvms-compute-backend.md#4-infra-and-iam-conditional-resources-behind-bootstrap-computetypes) records the role responsibilities and limits; [service feedback](../verification/645-lambda-microvm-service-feedback.md#iam-setup-f02) tracks the unresolved contract questions. + +**ECS/MicroVM task bootstrap (v2)** — The coordinator publishes non-secret deployment manifests and supplies a short-lived signed URL for exactly one task's payload. The worker reads the manifest with its ambient role, whose explicit deny outside its own `bootstrap/*` prevents a foreign public bucket from authorizing fake configuration. It verifies downloaded task identity and exact manifest/config agreement before installing MicroVM settings. The coordinator privately persists the URL outside TaskTable for retry and deletes both task objects at finalization. Worker roles cannot read/list task objects using their ambient credentials. A signed URL is a bearer capability: redact it in errors and keep it out of task rows and repository subprocess environments. Existing session-tag trust and other platform grants remain separate limitations. The [payload guide](../verification/645-payload-bootstrap.md) describes coordinated image/coordinator/policy upgrades; the [acceptance summary](../verification/README.md) distinguishes recorded AWS checks from remaining validation. > Out of scope for this control and tracked separately as GitHub issues: replacing the shared GitHub PAT (GitHub App / Token Vault), binding credentials to the MicroVM via attestation, and scoping AgentCore Memory (namespace isolation by `actorId`/`sessionId` remains its boundary). diff --git a/docs/guides/USER_GUIDE.md b/docs/guides/USER_GUIDE.md index cb55b6cad..991b692a8 100644 --- a/docs/guides/USER_GUIDE.md +++ b/docs/guides/USER_GUIDE.md @@ -949,8 +949,9 @@ remaining open does not mean its old computer must stay alive. Sleeping saves compute charges but adds snapshot save/restore charges and wake-up time; short pauses can cost more than staying awake. The API equivalent is `microvm_sleep_after_s` (zero means off); task -details return the saved setting. Automatic suspension remains disabled by -default pending the [P3 acceptance checks](../verification/645-p3-implementation-plan.md). +details return the saved setting. Automatic suspension is disabled by default for +new deployments. An operator enables it after [verifying the deployed image and +coordinator](../verification/README.md#live-acceptance-for-an-installation). ## Webhook integration diff --git a/docs/src/content/docs/architecture/Cedar-hitl-gates.md b/docs/src/content/docs/architecture/Cedar-hitl-gates.md index ce3ac49fe..2e709bcb4 100644 --- a/docs/src/content/docs/architecture/Cedar-hitl-gates.md +++ b/docs/src/content/docs/architecture/Cedar-hitl-gates.md @@ -21,10 +21,10 @@ title: Cedar hitl gates > CLI response instructions are implemented for Slack/Linear notifications; > native channel decisions remain separate work. See the current > [user guide](/sample-autonomous-cloud-coding-agents/using/overview#approval-gates-cedar-hitl) and -> [continuation protocol](/sample-autonomous-cloud-coding-agents/architecture/645-p3-continuation-protocol-20260917). +> [continuation protocol](/sample-autonomous-cloud-coding-agents/architecture/orchestrator#retained-microvm-approvals). > The normal deployment passed retained-request, ten-minute sleep, explicit-expiry > and sleep-off/rollback acceptance; see the -> [deployment record](/sample-autonomous-cloud-coding-agents/architecture/645-p3-normal-closure-20260918). +> [deployment record](/sample-autonomous-cloud-coding-agents/architecture/readme). --- @@ -1043,7 +1043,7 @@ event. A later decision is rejected: the existing API returns `404 REQUEST_NOT_FOUND` for missing, foreign or already-decided approval rows, including a cancelled row. If approval committed first, cancellation preserves the recorded decision while cancelling the task. See the -[P3 approval verification record](/sample-autonomous-cloud-coding-agents/architecture/645-p3-approval-ux-20260917) +[P3 approval verification record](/sample-autonomous-cloud-coding-agents/architecture/readme) for source versus deployment status. ### 7.2 `POST /v1/tasks/{task_id}/deny` @@ -1534,7 +1534,7 @@ path uses the CLI owner's authentication. Native Slack approval buttons and Line approval replies are not implemented by this notification change; the OAuth/button design below remains proposed. Email remains a log-only stub and GitHub does not receive approval messages. Deployment status is recorded in the -[P3 verification record](/sample-autonomous-cloud-coding-agents/architecture/645-p3-approval-ux-20260917). +[P3 verification record](/sample-autonomous-cloud-coding-agents/architecture/readme). **TaskApprovalsTable Streams are not consumed by the fan-out Lambda**. The approval row is working state; the audit trail is in TaskEventsTable. Enabling Streams on TaskApprovalsTable would be redundant and add noise. Final design: TaskApprovalsTable DOES NOT have Streams enabled. (Retains the `stream` attribute commented out for future use if needed.) diff --git a/docs/src/content/docs/architecture/Compute.md b/docs/src/content/docs/architecture/Compute.md index 4f72a8a22..7b6d30c24 100644 --- a/docs/src/content/docs/architecture/Compute.md +++ b/docs/src/content/docs/architecture/Compute.md @@ -85,11 +85,11 @@ Lambda MicroVMs are an opt-in third backend, selected per repository with `compu For managed images, run `cdk/scripts/package-microvm-artifact.sh --stack-name ` after each agent change. It packages the Dockerfile's local inputs into a deterministic ZIP and uploads it under `microvm-images/agent-artifact-.zip`. The checksum is a fingerprint of the uploaded bytes: identical inputs reuse the same verified object, and changed inputs produce a new filename. Deploy with the printed `--context microvm_artifact_sha256=` alongside `microvm_base_image_arn` and `microvm_base_image_version`, and retain these inputs for later deployments. The changed S3 URI tells CloudFormation to update the existing image; overwriting the old fixed filename alone does not. Missing or malformed digests fail synthesis. An initial deployment without an image must create the bucket first. The script's explicit `--create-image` alternative retains the fixed base key and calls the image API directly. -Because a snapshot freezes its build-time environment, current deployment identifiers arrive through the v2 payload bootstrap. The coordinator publishes a non-secret deployment manifest and sends a single-object signed download URL for the task. The worker reads only its deployment's `bootstrap/*` with ambient credentials; other object reads and payload-bucket listing are explicitly denied. The downloaded task identity and configuration must match the authenticated manifest before configuration installation. The serialized reference fits the verified 4,096-byte hook limit; all payload sizes use S3. ECS shares this transport through `AGENT_PAYLOAD_REF`. [ADR-021 §3](/sample-autonomous-cloud-coding-agents/architecture/adr-021-lambda-microvms-compute-backend#3-packaging-same-agent-image-source-new-build-path) defines the wire format and compatibility requirements; [live payload checks](/sample-autonomous-cloud-coding-agents/architecture/645-p2-payload-live-20260914) record validation. +Because a snapshot freezes its build-time environment, current deployment identifiers arrive through the v2 payload bootstrap. The coordinator publishes a non-secret deployment manifest and sends a single-object signed download URL for the task. The worker reads only its deployment's `bootstrap/*` with ambient credentials; other object reads and payload-bucket listing are explicitly denied. The downloaded task identity and configuration must match the authenticated manifest before configuration installation. The serialized reference fits the verified 4,096-byte hook limit; all payload sizes use S3. ECS shares this transport through `AGENT_PAYLOAD_REF`. [ADR-021 §3](/sample-autonomous-cloud-coding-agents/architecture/adr-021-lambda-microvms-compute-backend#3-packaging-same-agent-image-source-new-build-path) defines the wire format and compatibility requirements; [live payload checks](/sample-autonomous-cloud-coding-agents/architecture/readme) record validation. Networking separates image build from execution: the build-only connector permits TCP 80 and 443 because the Dockerfile uses `apt-get`, while running MicroVMs retain 443-only egress through the platform VPC. Every launch explicitly passes the Lambda-managed `NO_INGRESS` connector; omission would select the service's public-ingress default. Images declare and serve `/ready` and `/validate` at build time and `/run`, `/terminate`, `/suspend` and `/resume` at runtime. Automatic suspension is a separate deployment opt-in. Registry HTTP/SSE tools therefore need reachable HTTPS/443 endpoints; remote non-443 tools are unsupported under the default policies of AgentCore and ECS as well. Local `stdio` tools can run, with their outbound traffic subject to the same restriction. Asset resolution/loading does not probe connectivity; see [registry network support](/sample-autonomous-cloud-coding-agents/architecture/registry#2-asset-kinds-for-mvp). -P3 connects the guest checkpoint/credential hooks, actual image-version capability, durable supervisor and post-commit approval wake. The `microvm_approval_suspend_enabled` deployment context defaults false and controls both a static opt-in and a live Parameter Store switch. Existing durable executions reread the live switch before new suspension because their original Lambda environment is pinned. Recovery and the original service lifetime survive supervisor replay; the API preserves accepted decisions when optional wake fails. AgentCore/ECS retain explicit unsupported pause/wake results. Explicit approval deadlines use the original UTC/monotonic deadline. See the [supervisor runbook](/sample-autonomous-cloud-coding-agents/architecture/645-p3-supervisor) for budgets and scoped permissions, and the [completion record](/sample-autonomous-cloud-coding-agents/architecture/645-p3-implementation-plan) for deployment evidence. +P3 connects the guest checkpoint/credential hooks, actual image-version capability, durable supervisor and post-commit approval wake. The `microvm_approval_suspend_enabled` deployment context defaults false and controls both a static opt-in and a live Parameter Store switch. Existing durable executions reread the live switch before new suspension because their original Lambda environment is pinned. Recovery and the original service lifetime survive supervisor replay; the API preserves accepted decisions when optional wake fails. AgentCore/ECS retain explicit unsupported pause/wake results. Explicit approval deadlines use the original UTC/monotonic deadline. See the [lifecycle diagnostics](/sample-autonomous-cloud-coding-agents/architecture/645-p3-lifecycle-diagnostics) for failure investigation, and the [acceptance status](/sample-autonomous-cloud-coding-agents/architecture/readme) for deployment evidence. Tasks can set `microvm_sleep_after_s` (CLI: `--microvm-sleep-after `). The default is 600 seconds of waiting for each approval; zero disables sleep. diff --git a/docs/src/content/docs/architecture/Orchestrator.md b/docs/src/content/docs/architecture/Orchestrator.md index 531312b71..e5f6d60f1 100644 --- a/docs/src/content/docs/architecture/Orchestrator.md +++ b/docs/src/content/docs/architecture/Orchestrator.md @@ -300,7 +300,7 @@ When the session is unhealthy, the task transitions to `FAILED` with "Agent sess - A terminal substrate report paired with a non-terminal task first checks for a complete, acknowledged approval checkpoint. Such a checkpoint can retire the old attempt and retain the task for a replacement. Without one, finalization strongly re-reads the task row before classifying a substrate failure. - Substrate state detects a dead VM; heartbeat staleness detects loss of the heartbeat writer inside a VM that still reports `RUNNING`. The independent heartbeat thread can continue during a pipeline hang, so a fresh timestamp is not proof of progress. -The P3 supervisor saves intent before control calls and rechecks the gate before and after them. Its durable state retains an absolute service lifetime, consecutive failures, recovery start time and next delay. Three failed cycles or 120 seconds of unconfirmed wake cannot become an indefinite wait. AWS RUNNING does not end recovery while the guest remains stuck on a decided/expired approval; fresh guest liveness is required. API approve/deny commit first, then attempt a bounded wake without changing the decision response. Automatic suspension defaults off via `microvm_approval_suspend_enabled`; disabling new sleep preserves wake and cleanup. See the [supervisor runbook](/sample-autonomous-cloud-coding-agents/architecture/645-p3-supervisor). +The P3 supervisor saves intent before control calls and rechecks the gate before and after them. Its durable state retains an absolute service lifetime, consecutive failures, recovery start time and next delay. Three failed cycles or 120 seconds of unconfirmed wake cannot become an indefinite wait. AWS RUNNING does not end recovery while the guest remains stuck on a decided/expired approval; fresh guest liveness is required. API approve/deny commit first, then attempt a bounded wake without changing the decision response. Automatic suspension defaults off via `microvm_approval_suspend_enabled`; disabling new sleep preserves wake and cleanup. See the [lifecycle diagnostics](/sample-autonomous-cloud-coding-agents/architecture/645-p3-lifecycle-diagnostics). Unanswered approvals have no deadline by default. A checkpointed MicroVM wait can retire after an hour, or before that worker's lifetime ends. Retirement and diff --git a/docs/src/content/docs/architecture/Security.md b/docs/src/content/docs/architecture/Security.md index a392f8b8c..ee9cfb42d 100644 --- a/docs/src/content/docs/architecture/Security.md +++ b/docs/src/content/docs/architecture/Security.md @@ -48,11 +48,13 @@ Three authentication mechanisms protect the platform, matching its input channel - **DynamoDB**: item access on the four `task_id`-partitioned tables (task, events, approvals, nudges) is gated by a `dynamodb:LeadingKeys` condition equal to `${aws:PrincipalTag/task_id}` and requires that context key to be present. `Scan` is not granted. The main task table permits reads and only attribute-scoped `UpdateItem` writes: the agent can report status, heartbeat, results and approval details, but cannot replace/delete the row or edit coordinator-owned start receipts, capacity reservations, owner identity or compute handles. The reviewed write list is `cdk/src/constructs/agent-task-write-attributes.json`; Python tests exercise the current writers against it. Supporting tables retain task-scoped item writes. `LeadingKeys` binds the base-table partition key, not a GSI key such as `user_id`. - **S3**: trace writes and attachment reads are scoped to the `/${aws:PrincipalTag/user_id}/` object prefix. -The compute role retains only non-tenant access (Bedrock model invocation — already ARN-scoped; CloudWatch Logs; the GitHub PAT secret, read once before the SessionRole is assumed; AgentCore Memory) plus `sts:AssumeRole`/`sts:TagSession` on the SessionRole. Because the agent runs under credentials that are themselves an assumed role, its `AssumeRole` is *role chaining* — capped at one hour regardless of the role's max session duration — so the agent uses a **refreshable** credential provider that re-assumes before expiry (tasks can run up to the 8-hour `maxLifetime`). The design is backend-agnostic: the same SessionRole and agent code serve all three backends (AgentCore Runtime, ECS Fargate, Lambda MicroVMs). Existing scoped credentials are constrained to their tags, but the compute role chooses those tags when assuming the role. The trust policy does not independently bind a worker to one task; this is not complete isolation against a compromised worker with ambient credentials. No worker needs direct access to the shared capacity counter. ECS's legacy configuration without a session role retains unscoped task reads/reporting updates, with the same attribute restriction. The policy structure and Python writer compatibility are tested locally. A historical IAM simulator check covered the original tag conditions; it does not verify the new attribute restriction or deployed transaction authorization. The repository runbook `docs/verification/645-coordinator-metadata.md` contains the required AWS authorization checks. +The compute role retains only non-tenant access (Bedrock model invocation — already ARN-scoped; CloudWatch Logs; the GitHub PAT secret, read once before the SessionRole is assumed; AgentCore Memory) plus `sts:AssumeRole`/`sts:TagSession` on the SessionRole. Because the agent runs under credentials that are themselves an assumed role, its `AssumeRole` is *role chaining* — capped at one hour regardless of the role's max session duration — so the agent uses a **refreshable** credential provider that re-assumes before expiry (tasks can run up to the 8-hour `maxLifetime`). The design is backend-agnostic: the same SessionRole and agent code serve all three backends (AgentCore Runtime, ECS Fargate, Lambda MicroVMs). Existing scoped credentials are constrained to their tags, but the compute role chooses those tags when assuming the role. The trust policy does not independently bind a worker to one task; this is not complete isolation against a compromised worker with ambient credentials. No worker needs direct access to the shared capacity counter. ECS's legacy configuration without a session role retains unscoped task reads/reporting updates, with the same attribute restriction. The policy structure and Python writer compatibility are tested locally. A historical IAM simulator check covered the original tag conditions; it does not verify the new attribute restriction or deployed transaction authorization. The [acceptance checklist](/sample-autonomous-cloud-coding-agents/architecture/readme#live-acceptance-for-an-installation) includes effective-role authorization checks. -**Lambda MicroVMs compute-role delta** - On this backend the compute role additionally holds prefix-scoped `secretsmanager:GetSecretValue` on the per-workspace channel-OAuth secrets (`bgagent-linear-oauth-*`, `bgagent-jira-oauth-*`), read access only to its payload bucket's `bootstrap/*` manifests (other object reads and payload-bucket listing are explicitly denied), and `ec2:DescribeAvailabilityZones` (`Resource: *`, read-only — EC2 describe actions have no resource-level scoping) for a CDK repo's own synth gate. It is also **the only compute role in the platform whose trust policy carries no confused-deputy condition**: the Lambda MicroVMs service populates no `aws:SourceAccount` or `aws:SourceArn` when it assumes the role, so a trust policy carrying one is unassumable — verified live, not assumed (connector creation failed deterministically, and `RunMicrovm` surfaced the same root cause as a misleading caller-side `iam:PassRole` denial). The compensating controls are that each of the three MicroVM roles can be passed to `lambda.amazonaws.com` only by a named principal — the orchestrator, via an `iam:PassRole` scoped to the execution role's exact ARN, and the CloudFormation deployment role for the build and connector-operator roles — that every other resource they reach is account-scoped by ARN apart from two justified `Resource: *` read/create-time statements, and that none of them holds `iam:*`, cross-account trust, or any `sts:AssumeRole` beyond the execution role's scoped hop to the per-task SessionRole. Full evidence, the two-arm PassRole experiment, and the alternatives considered are in [ADR-021 §4](/sample-autonomous-cloud-coding-agents/architecture/adr-021-lambda-microvms-compute-backend#4-infra-and-iam-conditional-resources-behind-bootstrap-computetypes). +**Lambda MicroVMs compute-role delta** — The compute role additionally reads the per-workspace channel-OAuth secrets (`bgagent-linear-oauth-*`, `bgagent-jira-oauth-*`) and its payload bucket's `bootstrap/*` manifests. Ambient task-object reads and payload-bucket listing are explicitly denied. It also has read-only `ec2:DescribeAvailabilityZones` for repository CDK synthesis; that API has no resource-level scope. -**ECS/MicroVM task bootstrap (v2)** — The coordinator publishes non-secret deployment manifests and supplies a short-lived signed URL for exactly one task's payload. The worker reads the manifest with its ambient role, whose explicit deny outside its own `bootstrap/*` prevents a foreign public bucket from authorizing fake configuration. It verifies downloaded task identity and exact manifest/config agreement before installing MicroVM settings. The coordinator privately persists the URL outside TaskTable for retry and deletes both task objects at finalization. Worker roles cannot read/list task objects using their ambient credentials. A signed URL is a bearer capability: redact it in errors and keep it out of task rows and repository subprocess environments. Existing session-tag trust and other platform grants remain separate limitations. This protocol is implemented locally; real IAM/network/expiry gates and the coordinated image/coordinator/policy rollout are documented in the repository's `docs/verification/645-payload-bootstrap.md`. +Recorded live calls rejected source-conditioned MicroVM role trust and service-conditioned PassRole grants. The integration therefore omits those conditions on the affected paths, while restricting which roles can be passed and which resources they may access. Reintroduce conditions only after verifying service support. [ADR-021 §4](/sample-autonomous-cloud-coding-agents/architecture/adr-021-lambda-microvms-compute-backend#4-infra-and-iam-conditional-resources-behind-bootstrap-computetypes) records the role responsibilities and limits; [service feedback](/sample-autonomous-cloud-coding-agents/architecture/645-lambda-microvm-service-feedback#iam-setup-f02) tracks the unresolved contract questions. + +**ECS/MicroVM task bootstrap (v2)** — The coordinator publishes non-secret deployment manifests and supplies a short-lived signed URL for exactly one task's payload. The worker reads the manifest with its ambient role, whose explicit deny outside its own `bootstrap/*` prevents a foreign public bucket from authorizing fake configuration. It verifies downloaded task identity and exact manifest/config agreement before installing MicroVM settings. The coordinator privately persists the URL outside TaskTable for retry and deletes both task objects at finalization. Worker roles cannot read/list task objects using their ambient credentials. A signed URL is a bearer capability: redact it in errors and keep it out of task rows and repository subprocess environments. Existing session-tag trust and other platform grants remain separate limitations. The [payload guide](/sample-autonomous-cloud-coding-agents/architecture/645-payload-bootstrap) describes coordinated image/coordinator/policy upgrades; the [acceptance summary](/sample-autonomous-cloud-coding-agents/architecture/readme) distinguishes recorded AWS checks from remaining validation. > Out of scope for this control and tracked separately as GitHub issues: replacing the shared GitHub PAT (GitHub App / Token Vault), binding credentials to the MicroVM via attestation, and scoping AgentCore Memory (namespace isolation by `actorId`/`sessionId` remains its boundary). diff --git a/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md b/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md index 19ec3107c..ba7108892 100644 --- a/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md +++ b/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md @@ -4,7 +4,7 @@ title: Adr 021 lambda microvms compute backend # ADR-021: AWS Lambda MicroVMs as a third ComputeStrategy backend -> **Implementation status (2026-09-18): P1 and P2 are merged; P3 implementation and normal deployment acceptance are complete.** P3 adds approval sleep/wake, retained requests, conversation/workspace recovery, replacement workers and nested infrastructure. See the [completed plan](/sample-autonomous-cloud-coding-agents/architecture/645-p3-implementation-plan) and [normal acceptance record](/sample-autonomous-cloud-coding-agents/architecture/645-p3-normal-closure-20260918). This ADR defines P1–P3; it does not define an official P4. +> **Implementation status (2026-09-18): P1 and P2 are merged; P3 is in draft review.** Approval sleep/wake, retained requests, conversation/workspace recovery and nested infrastructure have live acceptance evidence. Reusable migration and final integration checks remain open; see [verification status](/sample-autonomous-cloud-coding-agents/architecture/readme). This ADR defines P1–P3, not an official P4. **Status:** proposed **Date:** 2026-07-29 @@ -28,7 +28,7 @@ Lambda MicroVMs are managed Firecracker virtual machines, separate from ordinary See [COMPUTE.md](/sample-autonomous-cloud-coding-agents/architecture/compute) for the full comparison and costs. Suspending stops compute charges, but snapshot storage and save/restore charges remain. A shorter sleep delay does not guarantee lower total cost. -The [P1 probes](/sample-autonomous-cloud-coding-agents/architecture/645-p1-lambda-microvm-runbook) established a 4,096-byte `runHookPayload` limit, an image-ARN requirement and accepted baseline values of 512, 1,024, 2,048, 4,096 and 8,192 MiB. Those observations override conflicting generated SDK descriptions for the tested account/Region. They do not measure guest-visible launch memory, vertical-scaling latency or sustained workload fit. The service guide’s capacity figures and live observations must remain distinguishable. +Recorded P1 probes established a 4,096-byte `runHookPayload` limit, an image-ARN requirement and accepted baseline values of 512, 1,024, 2,048, 4,096 and 8,192 MiB. Those observations override conflicting generated SDK descriptions for the tested account/Region. They do not measure guest-visible launch memory, vertical-scaling latency or sustained workload fit. The service guide’s capacity figures and live observations must remain distinguishable. ### Design tensions the strategy must resolve @@ -74,11 +74,11 @@ Unanswered approvals have no deadline by default (`approval_timeout_s=0`). Expli Automatic suspension also requires the deployment’s `microvm_approval_suspend_enabled` opt-in, which defaults false for new deployments. A live Parameter Store switch lets existing durable executions stop initiating new suspensions without changing their pinned Lambda environment. The verified normal deployment has this opt-in enabled. Turning it off does not abandon already-suspended workers. -The coordinator observes the exact pending gate and waits for the sleep delay. It skips sleep when a timed gate has too little time left before its wake margin. The guest holds a coding barrier, drains acknowledged progress and commits a checkpoint before accepting `/suspend`. Lifecycle HTTP responses explicitly close their connections before freeze to avoid reuse of a stale pooled connection; see the [transport experiment](/sample-autonomous-cloud-coding-agents/architecture/645-p3-wake-transport-20260916). +The coordinator observes the exact pending gate and waits for the sleep delay. It skips sleep when a timed gate has too little time left before its wake margin. The guest holds a coding barrier, drains acknowledged progress and commits a checkpoint before accepting `/suspend`. Lifecycle HTTP responses explicitly close their connections before freeze to avoid reuse of a stale pooled connection; see the [transport evidence summary](/sample-autonomous-cloud-coding-agents/architecture/readme#recorded-acceptance). Approval and denial handlers commit the decision first. They then read the current handle consistently, persist wake intent and request resume best-effort. A wake failure records diagnostics and does not undo the accepted decision. The durable supervisor retries and observes both service state and guest consumption of the decision; RUNNING alone does not prove the tool was released. -A sleeping worker retains its concurrency reservation. For a longer wait, ABCA verifies a complete, version-pinned conversation/workspace checkpoint, fences the attempt, confirms shutdown and releases the reservation. A later decision can admit one replacement through the original published coordinator. It restores Git state, required workspace files, the actual SDK conversation, exact pending tool inputs and cumulative usage with fresh scoped credentials. See the [continuation protocol](/sample-autonomous-cloud-coding-agents/architecture/645-p3-continuation-protocol-20260917). +A sleeping worker retains its concurrency reservation. For a longer wait, ABCA verifies a complete, version-pinned conversation/workspace checkpoint, fences the attempt, confirms shutdown and releases the reservation. A later decision can admit one replacement through the original published coordinator. It restores Git state, required workspace files, the actual SDK conversation, exact pending tool inputs and cumulative usage with fresh scoped credentials. See the [continuation protocol](/sample-autonomous-cloud-coding-agents/architecture/orchestrator#retained-microvm-approvals). Normative requirements (EARS): @@ -108,7 +108,7 @@ All six hooks share the FastAPI listener on port 8080. AWS hook properties accep | `/suspend` | Drain acknowledged progress and commit the current safe checkpoint within the hook budget | | `/resume` | Renew credentials and reconcile the original gate before releasing coding | -Warm-up budgets come from `contracts/constants.json`; the total guest budget must remain below the image hook timeout. The [P2 runbook](/sample-autonomous-cloud-coding-agents/architecture/645-p2-smoke-runbook) records the cold-binary failure and the enum/trust/logging fixes. Historical binary sizes and timings are observations of those images, not sizing guarantees for later builds. +Warm-up budgets come from `contracts/constants.json`; the total guest budget must remain below the image hook timeout. Cold-binary startup exceeded the hook budget in an earlier image; warm-up moved this work into image preparation. Historical sizes and timings are not sizing guarantees for later builds. **Authenticated v2 payload transport.** Every ECS and MicroVM task uses S3. The coordinator publishes a deployment manifest, conditionally creates the task payload and sends a short-lived signed URL for that one object. The serialized MicroVM reference must fit 4,096 bytes; payloads are bounded at 8 MiB and manifests at 16 KiB. @@ -131,7 +131,7 @@ Normative requirements (EARS): - Signed URLs shall not appear in ordinary logs, agent-readable task rows or repository subprocess environments. Downloads shall use the exact regional S3 HTTPS object without redirects/proxies and with bounded response sizes. - Producers, images and IAM shall be upgraded together; incompatible workers must be drained before switching transport. -The [payload contract](/sample-autonomous-cloud-coding-agents/architecture/645-payload-bootstrap) and [live checks](/sample-autonomous-cloud-coding-agents/architecture/645-p2-payload-live-20260914) record the implementation and validation. This transport does not establish complete hostile-worker isolation: other platform grants remain, and a stolen signed URL is usable until expiry or revocation. +The [payload contract](/sample-autonomous-cloud-coding-agents/architecture/645-payload-bootstrap) and [live checks](/sample-autonomous-cloud-coding-agents/architecture/readme) record the implementation and validation. This transport does not establish complete hostile-worker isolation: other platform grants remain, and a stolen signed URL is usable until expiry or revocation. No ABCA endpoint consumer exists in P1–P3. The platform grants no `CreateMicrovmAuthToken` permission and mints no JWE tokens. `NO_INGRESS` can still return an endpoint URL; an unauthenticated 403 verifies the authentication boundary, not valid-token reachability. @@ -139,7 +139,7 @@ No ABCA endpoint consumer exists in P1–P3. The platform grants no `CreateMicro The backend adds build/runtime VPC connectors, build artifacts, launch payloads, logs, roles and a managed or external image. Its bootstrap policy is conditional on `ComputeTypes` including `lambda-microvm`. A VPC egress connector requires an operator role. Build egress permits ports 80/443 for package installation; runtime egress permits 443 through the platform VPC. -**Trust and PassRole limitation.** Recorded live checks rejected `aws:SourceAccount`/`aws:SourceArn` conditions on the MicroVM-facing roles and `iam:PassedToService` on the MicroVM PassRole paths. The working roles trust `lambda.amazonaws.com` without those conditions; build/execution roles also allow `sts:TagSession`. IAM simulation with caller-supplied condition values did not prove that the service supplied those values. See the [P2 evidence](/sample-autonomous-cloud-coding-agents/architecture/645-p2-smoke-runbook), [clean deployment](/sample-autonomous-cloud-coding-agents/architecture/645-p2-clean-deployment-20260913) and [effective IAM checks](/sample-autonomous-cloud-coding-agents/architecture/645-effective-iam-20260915). Reintroduce a condition only after verifying service support. +**Trust and PassRole limitation.** Recorded live checks rejected `aws:SourceAccount`/`aws:SourceArn` conditions on the MicroVM-facing roles and `iam:PassedToService` on the MicroVM PassRole paths. The working roles trust `lambda.amazonaws.com` without those conditions; build/execution roles also allow `sts:TagSession`. IAM simulation with caller-supplied condition values did not prove that the service supplied those values. See the [IAM service questions](/sample-autonomous-cloud-coding-agents/architecture/645-lambda-microvm-service-feedback#iam-setup-f02). Reintroduce a condition only after verifying service support. | Role/action | Scope and responsibility | |---|---| @@ -153,7 +153,7 @@ The backend adds build/runtime VPC connectors, build artifacts, launch payloads, `lambda:PassNetworkConnector` has no resource-level authorization support and therefore uses `Resource: *`. The operator role also has wildcard ENI permissions; some mutations can be scoped in IAM, so their current tested wildcard is not evidence that narrower permissions are impossible. DescribeAvailabilityZones needs a wildcard for fresh CDK repository lookups. These exceptions and namespace wildcards are documented in the construct’s cdk-nag suppressions. -**Nested infrastructure.** `microvm_nested_stack=true` puts MicroVM resources in a nested stack. The shared execution role stays in the parent to avoid a role-trust dependency cycle. Bootstrap 1.9.0 covers nested deployment roles. Existing flat installations need the reviewed overlap/drain migration, explicit image/coordinator pins and rollback checks in the [nested migration record](/sample-autonomous-cloud-coding-agents/architecture/645-p3-nested-stack). Moving construct paths alone is not a safe migration. Preserve `microvm_resource_name_prefix` after a migrated deployment. +**Nested infrastructure.** `microvm_nested_stack=true` puts MicroVM resources in a nested stack. The shared execution role stays in the parent to avoid a role-trust dependency cycle. Bootstrap 1.9.0 covers nested deployment roles. Existing flat installations need the reviewed overlap/drain migration, explicit image/coordinator pins and rollback checks described in the [migration prerequisites](/sample-autonomous-cloud-coding-agents/architecture/645-p3-nested-stack). Reusable migration commands are still being completed. Moving construct paths alone is not a safe migration. Preserve `microvm_resource_name_prefix` after a migrated deployment. #### Security bar vs existing backends ([#645](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/645) acceptance criterion) @@ -203,7 +203,7 @@ Changing the default backend, GPU support, native Slack approval buttons, approv ## Testing -The [completed plan](/sample-autonomous-cloud-coding-agents/architecture/645-p3-implementation-plan) indexes exact suite results and dated evidence. Required coverage includes: +The [verification summary](/sample-autonomous-cloud-coding-agents/architecture/readme) distinguishes recorded live acceptance from open PR checks. Required coverage includes: - Strategy state mapping, explicit unsupported results, ARN/Region validation, NO_INGRESS, omitted idlePolicy and bounded uncertain-start recovery. - Hook readiness/warm-up, AWS-silent build hooks, authenticated payload installation, arbitrary terminate bodies and lifecycle connection closure. diff --git a/docs/src/content/docs/using/Approval-gates-cedar-hitl.md b/docs/src/content/docs/using/Approval-gates-cedar-hitl.md index a97543089..0d29c63de 100644 --- a/docs/src/content/docs/using/Approval-gates-cedar-hitl.md +++ b/docs/src/content/docs/using/Approval-gates-cedar-hitl.md @@ -122,5 +122,6 @@ remaining open does not mean its old computer must stay alive. Sleeping saves compute charges but adds snapshot save/restore charges and wake-up time; short pauses can cost more than staying awake. The API equivalent is `microvm_sleep_after_s` (zero means off); task -details return the saved setting. Automatic suspension remains disabled by -default pending the [P3 acceptance checks](/sample-autonomous-cloud-coding-agents/architecture/645-p3-implementation-plan). \ No newline at end of file +details return the saved setting. Automatic suspension is disabled by default for +new deployments. An operator enables it after [verifying the deployed image and +coordinator](/sample-autonomous-cloud-coding-agents/architecture/readme#live-acceptance-for-an-installation). \ No newline at end of file diff --git a/docs/verification/645-capacity-reservations.md b/docs/verification/645-capacity-reservations.md deleted file mode 100644 index 7824f0e85..000000000 --- a/docs/verification/645-capacity-reservations.md +++ /dev/null @@ -1,87 +0,0 @@ -# Capacity reservation verification for #645 - -The replay regression starts with two occupied seats, finalizes one task, then repeats that finalizer as if its checkpoint were lost. The old implementation reduced the counter to zero. The new implementation keeps the other task's reservation and a count of one. - -## Local transaction tests - -`cdk/test/handlers/shared/task-concurrency-local.test.ts` uses real DynamoDB Local transactions and conditional expressions. It replaces client construction only to select an explicit loopback endpoint and dummy credentials. Fault hooks can discard a response after the database commits. - -The suite covers repeated finalization, concurrent admission/cap enforcement, lost acquisition/release responses, competing cleaners, failure replay, cancellation before admission, approval occupancy, unadmitted/queued tasks, wrong owners, empty counters with a racing admission, periodic terminal cleanup, ambiguous legacy rows and stale repair after a revision change. - -Run an isolated in-memory instance: - -```sh -docker run --rm -d --name abca645-capacity-ddb \ - --memory 512m --cpus 1 -p 127.0.0.1::8000 \ - amazon/dynamodb-local@sha256:ff89bd48ff32cd8d9be5fee8873b65b8854dc408f1afe881be6eb00247bc0dab \ - -jar DynamoDBLocal.jar -inMemory -sharedDb -docker port abca645-capacity-ddb 8000/tcp -``` - -From `cdk/`, use the printed port: - -```sh -ABCA_DDB_LOCAL_ENDPOINT=http://127.0.0.1: mise run testf -- task-concurrency-local -``` - -The test rejects non-loopback endpoints. It creates uniquely named temporary tables, deletes them after the suite and closes its clients. Without the environment variable, this optional integration suite is skipped; ordinary helper/handler tests still run. Stop the temporary database after verification: - -```sh -docker stop abca645-capacity-ddb -``` - -## Upgrade and live verification - -1. Pause new submissions and allow old coordinator executions to finish, or terminate them through normal task cancellation/cleanup. Drain pending uploads/queue pickups as appropriate to prevent old code from starting work during the update. -2. Deploy all changed writers together: orchestrator, upload confirmation, stranded cleaner, counter reconciler and queue pickup. The counter reconciler now needs task-table `UpdateItem`; upload confirmation only needs counter reads. These changes use existing tables and roles. -3. Let older unmarked active tasks settle. The new helper does not guess their reservation ownership, and repair skips a user with ambiguous active rows. Inspect `CONCURRENCY_RESERVATION_UNKNOWN`; verify actual task/compute state before manual corrections. -4. Run reconciliation with admissions paused, check task reservations against counters, then reopen admission. For rollback, drain tasks using the new protocol before returning to old counter writers. -5. Verify one normal completion, start failure, cancellation, stranded task, approval wait, queued pickup and interrupted finalization against the deployed policies. Check that one task's cleanup preserves other tasks' seats. -6. Measure the two strongly consistent base-table scans at realistic retention volume. The function has a five-minute timeout; interrupted scans must not install partial counts. Monitor scan duration/read capacity and failures. - -`CONCURRENCY_EMPTY_COUNTER` means a held reservation was found without a positive counter. The fallback closes its marker without subtracting from later admissions. Concurrent repair can leave an overcount for the next sweep to correct. - -## Limits of this evidence - -The [September 16 AWS follow-up](./645-p3-capacity-scan-20260916.md) now verifies -600 users across real multi-page task/counter scans, an exact deployed reconciler -artifact with equivalent permissions on isolated tables, interrupted scans with -zero writes, and conservative handling of an older unmarked task. A separate -[normal-role check](./645-p3-user-sleep-20260916.md) repaired an overcount and -terminal reservation while preserving a real waiting worker. The bounded -volume and role checks do not complete the old-writer drain/upgrade/rollback -procedure or establish arbitrary production scale. - -The subsequent [isolated upgrade/rollback rehearsal](./645-p3-capacity-upgrade-20260916.md) -passed admission fences, legacy/current writer drains, lost-finalization replay, -drained rollback and re-upgrade using real AWS functions and restricted table -permissions. All temporary resources were removed. It exercised the table -protocol with fixture-managed task statuses; the normal deployment's complete -admission-route and durable-execution drain remains a separate operational gate. - -A subsequent read-only deployment audit checked upload confirmation, coordinator -version 8, counter reconciliation, queue pickup and stranded-task reconciliation. -All five referenced the same code assets as the `a81c565d` assembly, and each -downloaded S3 ZIP matched its deployed Lambda `CodeSha256`. At that observation, -none of the normal coordinator's retained versions 2–8 had a running durable -execution. This establishes artifact alignment and a point-in-time execution -inventory; it does not prove admissions were paused or the upgrade/rollback -sequence was rehearsed. Evidence is in -`/tmp/abca-645-p2-clean-20260913/p3-capacity-writer-inventory-20260916`. -The [PID 1 evidence record](./645-p3-pid1-observer-20260916.md) links its -permanent archive, which also contains this writer inventory. The subsequent -[feedback-only rollout](./645-p3-wake-feedback-20260916.md) advanced the normal -coordinator to version 9; it did not change the capacity protocol. - -The [September 17 normal-rollout review](./645-p3-normal-rollout-review-20260917.md) -extends the inventory through coordinator version 10 and identifies all nine -actual admission producers. The installation was idle at observation and already -used task-owned reservations in its clean-deployment baseline. A live admission -fence was not applied. Its asynchronous processors have no failure-retention -destination; setting their reserved concurrency to zero can discard input -instead of holding it for later. Follow the retained-input and image-rollback -steps in that review before a normal rollout rehearsal. - -Local tests prove the application requests and DynamoDB Local's transaction behavior. They do not establish deployed IAM, AWS scaling, successful rollout or MicroVM sleep/wake behavior. Terminal events may repeat or be lost independently of the atomic seat update. - -The reservation/start markers share the task row. Subsequent prerequisite work restricts agent updates to reporting/approval attributes and removes whole-row replacement/deletion plus direct worker access to the counter. Public-API omission alone was not protection. See [coordinator metadata verification](./645-coordinator-metadata.md) for the writer inventory, actual policy boundary, remaining status/tag trust limits and required AWS authorization checks. These local transaction tests do not prove that security boundary. diff --git a/docs/verification/645-coordinator-metadata.md b/docs/verification/645-coordinator-metadata.md deleted file mode 100644 index 33911eca0..000000000 --- a/docs/verification/645-coordinator-metadata.md +++ /dev/null @@ -1,77 +0,0 @@ -# Coordinator metadata write protection (#645) - -The coordinator is the platform code that assigns capacity and starts/stops a task's computer. The agent reports what happened inside that computer. Both use the task record, but the agent must not rewrite the coordinator's saved machine identity or capacity reservation. - -## Implemented boundary - -`AgentSessionRole` now treats the main task table separately from events, approvals and nudges: - -- Main task reads remain scoped to the session's `task_id`. These fields are not secrets. -- Main task writes allow only `UpdateItem`, with `dynamodb:Attributes` restricted to the reviewed list in `cdk/src/constructs/agent-task-write-attributes.json`. Both the attribute context and the scoped leading key must be present; `ForAllValues` alone accepts an absent context key. -- No main-table `PutItem`, `DeleteItem`, `BatchWriteItem` or PartiQL write permission is granted. A replacement could erase a protected field without mentioning its name, so attribute filtering alone would not make replacement safe. -- Supporting table access remains scoped to the tagged task. Approval transactions still use `PutItem` on the approval table and restricted `UpdateItem` on the main task table. -- The main table cannot also be supplied as a supporting table: the construct rejects that bypass at synthesis. -- AgentCore and configured ECS/MicroVM workers have no direct DynamoDB access; they assume the session role. ECS's legacy configuration without a session role uses the same attribute restriction, but its reads/updates are not task-scoped. None of the compute roles has the old shared-capacity-counter grant. - -This is an allowlist: a new coordinator field is protected without adding its name to a denylist. New agent reporting fields require a deliberate contract update. - -AWS documents [`dynamodb:Attributes`](https://docs.aws.amazon.com/amazondynamodb/latest/developerguide/specifying-conditions.html) as the top-level attributes referenced in a request, including the parent of a nested path. For example, `SET #receipt.#handle = :value` with aliases resolving to `microvm_start.handle` references `microvm_start` and is outside the allowed set. Read responses remain unrestricted within the tagged task; this change does not claim attribute confidentiality. - -## Writer inventory - -| Agent writer | Main-task fields | -| --- | --- | -| `write_running` | `status`, `status_created_at`, `started_at`, optional `logs_url` | -| `write_heartbeat` | `agent_heartbeat_at`; condition reads `status` | -| `write_terminal` | Status/completion time, cost/turns, verification results, PR/answer and trace/artifact references | -| `write_trace_uri_conditional` | `trace_s3_uri`; condition reads terminal `status` | -| `transact_write_approval_request` | `status`, `awaiting_approval_request_id`; also creates the supporting approval row | -| `transact_resume_from_approval` | `status`, refreshed heartbeat, removes approval request ID | -| `increment_approval_gate_count_in_ddb` | `approval_gate_count` | - -`progress_writer.py` writes only the supporting events table. `nudge_reader.py` updates only the supporting nudges table. Approval decisions/timeouts use the supporting approvals table. - -The unused `write_submitted` and `write_session_info` Python helpers were removed. Neither had a production caller; their comments incorrectly attributed task creation/session registration to the agent. The TypeScript coordinator already owns those operations. - -## Local checks and their limits - -The regression first failed against the old policy because the main-table grant included `PutItem`. Construct tests now check the restricted main-task statements, missing-context guards, duplicate-table rejection, unchanged session tags and absence of direct worker counter access. Stack tests check actual AgentCore and MicroVM wiring; ECS tests cover both configured and legacy direct access. - -Python contract tests run the real task writers against recording clients, including all terminal-result fields and both approval transactions. They compare the requested attributes against the same JSON list used by CDK and require review when a new writer is added. - -These local checks inspect generated policies and request compatibility; they do -not execute AWS authorization. DynamoDB Local also does not implement IAM. -The subsequent [37-case AWS matrix](./645-effective-iam-20260915.md) passed using -the unchanged deployed MicroVM role and real tagged sessions from a temporary -Lambda. Other backend ambient roles and migration/scale remain open. - -## AWS acceptance and rollout gate - -Use an isolated deployment and disposable task rows with known initial `microvm_start`, `concurrency_slot`, owner and compute metadata. Record the deployed commit, image version and effective role policies. Use task-scoped credentials for two different tasks and separately check the ambient compute roles. - -| Request | Required result | -| --- | --- | -| Own-task heartbeat, full terminal result and trace repair | Allowed; protected fields unchanged | -| Approval request and resume transactions | Allowed; both records consistent and heartbeat refreshed | -| Update of another task or a session with no task tag | Denied | -| Put/replacement, Delete, BatchWrite put/delete on own task | Denied; original row unchanged | -| SET/REMOVE/ADD/DELETE involving `concurrency_slot` or `microvm_start`, including aliases/nested paths | Denied | -| Change `user_id`, TTL, session ID, compute type or compute metadata | Denied | -| Transaction containing a forbidden main-task update/put/delete plus an otherwise valid approval write | Entire transaction denied; neither record changes | -| PartiQL update/delete/insert against the task table | Denied | -| Direct task-table or capacity-counter operations using any configured worker's ambient credentials | Denied | -| Coordinator start, cancellation, finalization and counter repair | Allowed; replay retains exactly-once reservation accounting | - -Drain old executions according to the [capacity rollout procedure](./645-capacity-reservations.md#upgrade-and-live-verification). Review effective policies for additional grants or resource policies that could bypass this allowlist. Deploy the code/policies together and publish the matching agent image. Existing agent writers already fit the list, but an external or old custom caller of the removed helpers must be migrated. Wait for IAM propagation and verify newly assumed and existing sessions before reopening admissions. No bootstrap policy change is required by this patch; it changes application-role permissions. - -Do not roll back to unrestricted worker writes while relying on protected reservation/start metadata. Drain first and explicitly review the security consequence of a rollback. - -## Remaining trust limits - -The agent can still report status and results. This patch does not authenticate whether its reported success/failure is truthful or constrain status values/transitions at IAM level. Supporting approval/event/nudge rows retain their prior permissions. - -The compute role chooses `{user_id, repo, task_id}` session tags. Current trust policies do not independently prove that the chosen task belongs to that worker. A compromised whole worker with ambient credentials can therefore try to assume a differently tagged session. The attribute restriction applies to that session too, but this is not complete tenant isolation. Trusted task/deployment identity and task-scoped payload reads remain separate prerequisites in the [P3 plan](./645-p3-implementation-plan.md). - -## Payload capability storage - -The subsequent [v2 payload bootstrap](./645-payload-bootstrap.md) stores signed download URLs only in coordinator-owned S3 launch records. They are not added to `microvm_start` or other TaskTable attributes: own-task reads still include internal metadata, so API omission and attribute write restrictions cannot provide confidentiality for a bearer URL. diff --git a/docs/verification/645-effective-iam-20260915.md b/docs/verification/645-effective-iam-20260915.md deleted file mode 100644 index befc530c7..000000000 --- a/docs/verification/645-effective-iam-20260915.md +++ /dev/null @@ -1,109 +0,0 @@ -# ADR-021: effective AWS metadata and payload permissions - -Date: 2026-09-15. These checks complement the -[metadata contract](./645-coordinator-metadata.md), -[payload bootstrap](./645-payload-bootstrap.md), and -[durable lifecycle checks](./645-p3-durable-live-20260915.md). - -## Scope - -Temporary Lambda functions used the unchanged deployed MicroVM execution role -in `backgroundagent-dev`, `us-west-2`. Its existing trust permits Lambda. Their -code was limited to fixed disposable records and objects. They assumed real -task-tagged STS sessions. No existing role or bucket policy was modified. - -This tests actual effective AWS authorization. It does not test the MicroVM -network connector, other backends' ambient roles, or ECS rollout. Compute roles -still choose their session tags; hostile-worker identity binding remains a limit. - -## Metadata: 37 checks passed - -Two disposable terminal task rows carried protected owner, worker, start-receipt -and reservation fields. No worker or capacity reservation was acquired for them. - -| Requests | Observed result | -|---|---| -| Each tagged session reads its own task | Allowed | -| Cross-task reads and reads without a task tag | `AccessDeniedException` | -| Own heartbeat and complete reporting attribute set | Allowed | -| Foreign/missing-task-tag updates | `AccessDeniedException` | -| Owner, TTL, session ID, compute type and compute metadata changes | `AccessDeniedException` | -| Aliased/nested start/reservation updates and removals | `AccessDeniedException` | -| ADD to protected TTL; DELETE from a protected set | `AccessDeniedException` | -| Put, Delete, BatchWrite put/delete and PartiQL insert/update/delete | `AccessDeniedException` | -| Approval Put plus reporting Update transaction | Allowed | -| Forbidden task Update/Put/Delete plus valid approval Put | Entire transaction denied; no approval created | -| Ambient worker task/counter reads and updates | `AccessDeniedException` | -| Scoped task session counter reads and updates | `AccessDeniedException` | - -The entire task record was unchanged afterward. False conditions protected -negative single-item writes where supported. Only disposable rows could be -affected if deletion unexpectedly succeeded. These are authorization checks of -request shapes, not a normal approval workflow on the synthetic terminal rows. - -Verifier `backgroundagent-dev-p3-iam-20260915` version 2 had ZIP SHA-256 -`IzStMi0xeUIUNRQGhvNwrhGy5gs2XFGClArQc4U79f4=`. Version 1 passed a 31-case subset. -All versions, both task rows and the owned approval were removed; cleanup was -verified at 22:20:28.279Z. Existing roles and shared logs were retained. - -## S3: 10 checks and public control passed - -Three harmless owned markers occupied a bootstrap manifest key, task payload key -and private launch key. They contained no credentials or real task instructions. - -| Request | Observed result | -|---|---| -| Ambient worker reads its own bootstrap prefix | Allowed | -| Ambient worker reads payload/private launch or lists payload bucket | `AccessDenied` | -| Ambient worker replaces/deletes manifest | `AccessDenied` | -| Scoped session SDK reads payload, launch or manifest | `AccessDenied` | -| Worker-signed SDK read of a public NOAA object | `AccessDenied` | - -From the same Lambda invocation, an anonymous one-byte request to -`https://noaa-goes16.s3.amazonaws.com/index.html` returned HTTP 206, while the -worker-signed request to the same object was denied. No public resources or -public-access settings were created or changed. - -The deployed worker policy explicitly denies `s3:GetObject*` outside its own -`bootstrap/*` prefix and denies `s3:List*` on the payload bucket. This public -control uses an existing NOAA object, not a crafted public manifest. Earlier -guest probes separately verified private foreign-manifest rejection. - -Version 1 incorrectly required the public error text to name an explicit deny. -AWS provides enhanced reasons mainly for -[same-account/organization requests](https://docs.aws.amazon.com/AmazonS3/latest/userguide/troubleshoot-403-errors.html). -Version 2 pairs same-function anonymous success with signed denial and retains -the deployed policy separately. - -Verifier `backgroundagent-dev-p3-s3-20260915` version 2 had ZIP SHA-256 -`ZN6YKfbf5XqqYiTQ/3BtdyOYSvBl3LuFNHsP3Tfem8g=`. All versions were deleted and -absence verified at 22:31:12.242Z. The three object bodies remained unchanged. - -## Signing credentials expire: passed - -A temporary signer role trusts only the operator's existing IAM principal and -can read exactly the harmless owned payload. Its STS session lasts 900 seconds; -the download URL has a nominal 3,600-second lifetime. Initial HTTP 200 returned -the expected body. - -The credentials expired at 22:41:22Z. At 22:41:27.332Z, the same URL returned -HTTP 400 `ExpiredToken`, with about 45 minutes of nominal URL lifetime remaining. -S3's response date was 22:41:27Z, request ID `30CHHS5HK249BHPF`. Neither the object -nor the role policy changed during the test. A signed link therefore cannot -outlive the temporary credentials that signed it. - -Cleanup was verified at 22:42:55.386Z: all three owned objects were deleted, -the payload prefix was empty, manifest HEAD returned 404, and the temporary -signer role and inline policy were absent. The private mode-0600 URL file was -deleted. No signing keys or URL appear in logs or this record. - -## Evidence and remaining gates - -Private code, policies, per-operation results, hashes and cleanup ledgers: - -- `/tmp/abca-645-p2-clean-20260913/p3-effective-iam-20260915` -- `/tmp/abca-645-p2-clean-20260913/p3-effective-s3-20260915` - -Migration/drain, reconciler scale, other ambient roles, runtime network negatives -and coordinated ECS rollout remain separate gates. No credential values appear -in the result documents. diff --git a/docs/verification/645-lambda-microvm-service-feedback.md b/docs/verification/645-lambda-microvm-service-feedback.md index ab5a77c5a..49324b181 100644 --- a/docs/verification/645-lambda-microvm-service-feedback.md +++ b/docs/verification/645-lambda-microvm-service-feedback.md @@ -1,421 +1,74 @@ -# Lambda MicroVM service-team feedback tracker - -Updated 2026-09-18. Working notes for the ADR-021 takeover. **Not submitted to the -service team.** Keep each item's evidence, question, service response and next -action here as verification continues. - -P3 application acceptance is complete and automatic approval suspension is -enabled in the [normal deployment](./645-p3-normal-closure-20260918.md). -The dated investigations below preserve earlier deployment settings; the -service questions remain open. Account IDs are redacted from this public copy. - -In plain language: a MicroVM is the worker's little computer. A lifecycle hook -is the doorbell AWS rings to tell it to start, pause or wake. An AWS request ID -is the receipt that lets the service team find a particular call. - -| ID | Priority | Topic | Evidence/status | -|---|---|---|---| -| F01 | High; service diagnosis open | Accepted wake ends in connection refusal | Six recorded failures; pre-freeze connection closure verified; image 6.0 correction deployed and nine Durable workflows passed | -| F02 | High | Supported IAM conditions and misleading permission errors | Reproduced in earlier P2 work; current service behavior needs confirmation | -| F03 | Medium | A service-side hook timeline and structured failure details | Diagnostic improvement request based on F01 | -| F04 | Medium | `PENDING` also means restoring an existing worker | Observed live; our timer bug is fixed | -| F05 | Medium | HTTP connection handling across suspend/resume | Exact refusal correlates with expired idle connection; explicit close header deployed; service dispatch details still needed | -| F06 | Medium | Conditional operator-role requirement for VPC connectors | Earlier deployment failure; application setup fixed | -| F07 | P2/P3 acceptance gap | Run token retention and recovery without a worker ID | Guest-log recovery verified; maximum retention, post-expiry behavior and recovery without identity logs unknown | -| F08 | High; service diagnosis open | Generic wake-hook failure with an observed listener | Historical image 5.0 failure; PID 1 owned its listener after restore, with no resume hook entry; connection-close correction now deployed | -| F09 | Blocks native image refactoring | Image move passes preview but execution rejects tag schema | Exact one-image move rolled back; original image, version, tags and resource identities preserved | - -## F01 — Wake request accepted, then the hook connection is refused - -**Impact:** the user approves an action, but the worker stops before continuing. -The coordinator releases capacity correctly; the requested coding workflow -failed in these recorded cases. The application correction below passed P3 -acceptance before automatic suspension was enabled. - -The [September 17 follow-up](./645-p3-final-image-and-ecs-20260917.md) adds two -repository sleep/wake cycles, late approval winning, and wake after actual STS -expiry on normal image 6.0. The long-sleep worker remained suspended after its -old credentials expired, then renewed with identical identity tags and retained -its original approval deadline. All three workflows finalized without repair. -These are additional application acceptance results, not service-side traces -or a measured failure rate. - -The [normal repository follow-up](./645-p3-repository-path-20260917.md) also -passes clone/setup, approval sleep/wake, build/lint and existing-PR resolution -on image 6.0, with unchanged GitHub content and verified private cleanup. - -**Observed:** five failures on normal images `3.0`, `4.0` and `5.0`, plus a sixth -on a private image retaining 5.0's application and connection behavior with -transport probes, in `us-west-2`, September 15–16. AWS accepted `ResumeMicrovm`, -then reported: - -> Resume lifecycle hook connection was refused. Please check your hook endpoint -> and application logs for more details. - -The guest had returned HTTP 200 from `/suspend`; retained logs contain no -subsequent `/resume` access entry. Single-issuer cases exclude overlapping API -and coordinator Resume calls as a necessary cause. - -**Best starting evidence for the service team:** - -- Account ``, region `us-west-2`, September 16, 21:14:27–21:15:30 UTC. -- Worker `microvm-bb1bfd4b-9ce3-3691-a60b-88f263081d43`, private image - `backgroundagent-dev-p3-wake-transport-20260916:1.0`, original server PID 1. -- Resume receipt `74943bd5-cfd1-4781-9f10-9946566089e8`, accepted - 21:15:26.786 UTC. At 21:15:26.948, the restored event loop expired the old - suspend connection's five-second idle timer. At 21:15:26.950, an independent - observer saw PID 1 owning the original port-8080 listener. AWS terminated the - worker at 21:15:27.471 with the exact refusal wording. -- No fresh HTTP connection or resume request appeared in the retained window. - The [transport investigation](./645-p3-wake-transport-20260916.md) - contains the full timeline and successful fresh-connection comparison. - -**Earlier unmodified image 5.0 evidence:** - -- Account ``, region `us-west-2`, September 16, 17:09:39–17:10:42 UTC. -- Worker `microvm-6ff103ab-a41d-348b-8a56-721c9050b623`, image - `backgroundagent-dev-abca-agent:5.0`, original server PID 1. -- Suspend receipt `98a53cab-7737-4675-944e-7a8202781149`; guest checkpoint - succeeded and `/suspend` returned HTTP 200 at 17:09:40.186 UTC. -- Resume receipt `d32f9929-e7cb-4603-a6c1-0bd2a801f5b5`, acknowledged - 17:10:39.404 UTC; service termination timestamp 17:10:40.415 UTC. -- The instrumented guest logged no resume hook entry or subsequent callback - stage. This case used a single supervisor wake before the original deadline, - with no injected failure. -- [Prepared investigation report](./645-p3-resume-refusal-investigation.md) - contains all five worker IDs, comparison runs and the complete example timelines. - -**Ask:** inspect the service's restore and hook-transport records for these -receipts. Was this a TCP refusal from the guest listener, a stale connection, -a process exit, or a networking/restore failure? At what point did the service -consider the guest network and listener ready, and what underlying error did it -map to this reason? - -**Limits:** the latest guest trace supports an old-connection race but cannot -show the service's actual dispatch error or socket choice. The Mac was the test -controller; the failing connection was inside AWS. Passing controls do not -establish a failure rate or discharge the earlier failures. - -**Application correction:** the [normal rollout](./645-p3-connection-close-rollout-20260916.md) -deployed explicit close headers in image 6.0, with coordinator 10 and both sleep -switches off. Approval, denial, timeout and cancellation while asleep passed on -that image through a private real Durable coordinator and normal decision APIs. -Two unchanged sleeps longer than 90 seconds and three candidate cases passed; -the candidate trace verifies actual connection closure before freeze. - -**Next:** obtain service-side dispatch details for the retained failures and -confirm supported connection handling across suspension. The application -evidence does not establish the exact service error in every historical case. -Service response: pending. F08 records an independently observed failed wake -with different wording; a shared cause remains unconfirmed. - -## F02 — IAM conditions and errors make correct setup difficult - -IAM roles are permission sets. A trust condition adds a rule about who may use -one. `PassRole` is permission to hand a role to a service. - -**Earlier evidence:** the [ADR infrastructure decision](../decisions/ADR-021-lambda-microvms-compute-backend.md) -records August 6–7 tests in which: - -- Trust policies using `aws:SourceAccount` / `aws:SourceArn` prevented the - MicroVM-facing roles from being assumed. Removing those conditions restored - the tested operations. -- A role-assumption problem surfaced as a caller-side `iam:PassRole` denial, - despite an existing grant and an `allowed` policy simulation. -- A separate clean comparison found exact-role `iam:PassRole` with - `iam:PassedToService: lambda.amazonaws.com` denied, while the same exact-role - grant without that condition succeeded. The earlier contaminated comparison - is explicitly corrected in the ADR. - -**Impact:** a normal attempt to tighten permissions breaks deployment or launch, -and the error sends the operator to the wrong policy. Our integration contains -the verified setup and exact-resource restrictions. - -**Ask:** publish the supported condition keys and values for build, execution, -connector-role assumption and both PassRole paths. Can the usual source/service -conditions be supported? Can errors distinguish missing caller permission from -failed target-role assumption? Is this behavior different in the current service? - -**Limits:** these are earlier reproduced integration findings, not a fresh -September 16 comparison or a demonstrated security exploit. Service response: -pending; recheck the recommended recipe before changing our working policies. - -## F03 — Expose the hook attempt that follows an accepted API request - -**Observed:** a successful Resume response provides an API receipt, but does not -establish that the guest received or completed its hook. In F01, the remaining -service explanation is a human-readable `stateReason`. A hook that is never -reached cannot write its own application diagnostic. - -**Our improvement:** [correlated lifecycle diagnostics](./645-p3-lifecycle-diagnostics.md) -now record hook entry/stage/result, PID, AWS receipts and coordinator state -changes. These passed isolated AWS verification and are -[deployed in coordinator 6 / image 5.0](./645-p3-diagnostics-rollout-20260916.md). -Two subsequent controlled missing-approval failures reached the guest, logged -the failed identity-read stage, returned HTTP 409 and surfaced specific failure -guidance through the deployed task API. This distinguishes an application -rejection from F01's missing hook-entry evidence. - -The [command-race follow-up](./645-p3-command-races-20260916.md) additionally -records a checkpoint transaction rejected after cancellation, cancellation during -observed `SUSPENDING` / restore `PENDING`, and three consecutive status-read -failures. The latter now has specific platform-error guidance deployed in -coordinator 7 and verified through the normal task API. These controlled failures -and passing race cases do not explain F01. - -F08 exposed another service message, `Resume lifecycle hook failed.`, without -an HTTP status or underlying connection error. The -[classifier correction deployed in coordinator 9](./645-p3-wake-feedback-20260916.md) -recognizes that observed wording and prevents misleading retry advice for newly -classified failures. Previously persisted stable error codes remain unchanged. -This does not establish why the service failed to complete the hook. - -The subsequent image 6.0 / coordinator 10 rollout also classifies -`Resume lifecycle hook timed out` as a nonretryable service failure. Six normal -task-handler checks cover raw/legacy/stable timeout forms, refusal, generic -failure and preservation of an older stable code. They use synthetic terminal -records and verify feedback, not a new service failure. - -**Ask:** provide a service-side lifecycle attempt timeline or equivalent -structured fields: originating API receipt, hook kind/attempt ID, start/end -times, connection versus HTTP failure, HTTP status, underlying error code and -whether a retry occurred. Link the attempt to worker/image identity and document -where customers can retrieve it after termination. - -**Impact:** this would show whether the doorbell failed or the worker received it -and failed while getting ready. It would also reduce dependence on parsing -human-readable failure strings. Service response: pending. - -## F04 — Document `PENDING` during restore - -**Observed:** a real worker went `SUSPENDED → PENDING → RUNNING` after Resume. -The installed SDK has no separate `RESUMING` state. The -[pending-wake record](./645-p3-pending-wake.md) includes timestamps and a verified -older-worker case with an API-issued wake. - -**Impact:** a client that treats every `PENDING` as first startup can use the -wrong timer. That was **our coordinator bug**; it is fixed and deployed in -coordinator version 5. It is separate from F01. - -**Ask:** publish a complete lifecycle transition table, including observable -restore states, and consider an explicit restoring state or transition -kind/start timestamp. Clarify which timestamps retain the original worker -lifetime and which describe the current transition. - -Service response: pending. Our next action is documentation/contract alignment, -not reopening the corrected timer bug. - -## F05 — Clarify hook HTTP connections across a freeze - -**Evidence:** some successful full-agent suspend/resume access logs used the -same peer port. An isolated Linux paused-process experiment reproduced a reset -of an old HTTP connection after a six-second pause, while every fresh connection -still succeeded. It did **not** reproduce an AWS connection refusal. -The interval from an observer's `SUSPENDED` sample is not the server's idle -connection age. The [timing correction](./645-p3-wake-transport-20260916.md#historical-timing-correction) -removes the earlier inference that short observed sleeps ruled out idle expiry. - -See the [transport control](./645-p3-transport-control-20260916.md) and -[process-observer record](./645-p3-process-observer-20260916.md). - -The [September 16 AWS comparison](./645-p3-diagnostics-rollout-20260916.md) -kept the original server as PID 1 and compared normal image 5.0 with a private -image differing only by `--timeout-keep-alive 0`. Quick approval wakes and -measured 10.205 / 8.697-second suspended holds passed on the respective images. -Both missing-approval controls reached the expected HTTP 409. Normal quick -cases reused the same client port; the longer normal wake and the connection-close -cases used fresh ports. No unexpected refusal or reset appeared. -Two original mistitled long-hold attempts are retained and excluded from that -acceptance; fresh corrected cases supplied the stated durations. - -The later [instrumented transport investigation](./645-p3-wake-transport-20260916.md) -captured a 59.451-second event-loop gap, closure of the old suspend connection -by `timeout_keep_alive_handler`, a live PID 1 listener, and the exact F01 refusal -without a new HTTP connection. Its passing control closed the old socket and -accepted a fresh resume connection about two milliseconds later. The candidate -uses an explicit response header, not the earlier timer flag. All three candidate -cases verified response-driven closure before freeze, no armed idle timer on the -suspend connection, and a fresh resume connection. - -That correction is now deployed in normal image 6.0. Four real Durable -approval/denial/timeout/cancellation workflows passed on the normal image; -the [rollout record](./645-p3-connection-close-rollout-20260916.md) includes -actual wake receipts, guest acknowledgments and final task/tool evidence. - -**Ask:** does the service reuse hook TCP connections across suspend/resume, -honor `Connection: close`, and retry a failed reused connection on a fresh socket? -How are reset, refused and timeout errors classified? What ordering is guaranteed -between guest unfreeze, network restoration and hook delivery? Which clock -semantics should guest timeout/keep-alive timers expect across suspension? - -**Limits:** the guest now records exact connection identity and timer closure, -but it cannot expose the service client's pool or failed dispatch. A paused -Linux process is not an AWS MicroVM restore, and the earlier timer-flag comparison -did not reproduce F01. The normal image now sends explicit close headers; these -passing cases do not reveal the service's earlier dispatch error. -Service response: pending; retain the service contract questions above. - -## F06 — Make the VPC connector role requirement obvious before deployment - -**Earlier evidence:** the generated CloudFormation/CDK property allowed omitting -`operatorRole`, but a `VPC_EGRESS` connector failed with: - -> NetworkConnectorOperatorRole is required for VPC_EGRESS connector type - -The [P1 live runbook](./645-p1-lambda-microvm-runbook.md) records the July 31 -failure and successful operator-role setup. Our construct now supplies that role. - -**Ask:** document or validate this conditional requirement in the schema/CDK -surface, with a complete example of the trust and ENI/tag/private-IP permissions. -Clarify whether a service-linked role is ever an alternative for this connector. - -**Limits:** an optional property can be correct for other connector types. This -is a request for clearer conditional validation and setup guidance, not a claim -that every connector requires the same role. Service response: pending. - -## F07 — Specify Run token retention and recovery after a lost reply - -A client token is an order number: repeating the same launch request with that -number should recover the original worker instead of ordering another one. - -**Evidence:** the [start/recovery probes](./645-p2-start-recovery-live-20260914.md) -verified simultaneous identical calls, changed-request rejection and replay of a -terminated worker through roughly five minutes. A replayed Run response could -still report `PENDING`; a separate Get correctly reported the worker's current -state. The installed SDK documents idempotency but gives no token-retention -duration. ABCA therefore stops automatic recovery without a saved handle after -its own conservative 120-second deadline. - -**Public documentation recheck (September 17):** the -[RunMicrovm API reference](https://docs.aws.amazon.com/lambda/latest/microvm-api/API_RunMicrovm.html) -is now reachable. It describes `clientToken` as “A unique, case-sensitive -identifier you provide to ensure the idempotency of the request,” with a -1–128-character limit. It still gives no retention duration or post-expiry -behavior. Its request schema has no per-worker tags field. This resolves the -earlier documentation-access problem, not the missing service contract. - -The same recheck of [ListMicrovms](https://docs.aws.amazon.com/lambda/latest/microvm-api/API_ListMicrovms.html) -and [GetMicrovm](https://docs.aws.amazon.com/lambda/latest/microvm-api/API_GetMicrovm.html) -found no returned client token, task identity, environment or per-worker tags. -List supports image/version filtering and returns the worker ID, image, -start time and state. Get adds the endpoint, execution role, connectors, -duration/idle settings and termination details. These fields can narrow an -operator's search, but shared image/role/time matches do not uniquely identify -the worker for a lost launch response. - -**Impact:** if AWS created a worker but its response was lost, an operator needs -to find that exact worker. An undocumented retention boundary prevents proving -that a late retry cannot create another one. ABCA must not invent a new token -or automatically submit a replacement task to hide the uncertainty. - -**Ask:** what is the guaranteed token-retention period, including after worker -termination or image-version changes? After expiry, is the same token rejected -or treated as a new launch? Which supported API, event or audit field maps the -original token/API receipt to the worker ID when the client never received it? -How long does that mapping remain available? - -**CloudTrail check:** at 17:36 UTC on September 16, event-history lookup for the -known image 5.0 launch at 17:08:54 UTC found no `RunMicrovm` event. The complete -17:08:45–17:09:05 window contained 74 events, including `GetMicrovm`, -`ListMicrovms` and `GetMicrovmImageVersion` from `lambda.amazonaws.com`. -Only event names, sources and window metadata were retained. This observation -does not establish the logging contract or exclude later delivery. Is Run a -management or data event, and what audit configuration is required to retain -the launch token-to-worker mapping? - -**Recovery follow-up:** the [September 16 durable check](./645-p3-registration-20260916.md) -discarded both accepted Run responses without saving their worker IDs. After the -task failed, the operator recovered the still-running worker from an exact -task/worker pair in its accepted `/run` guest log, checked its service identity -and terminated it. No Read result occurred and no replacement worker was -launched. This is a verified operator path when identity logs exist; it is not -a service token lookup or automatic recovery. - -**Limits:** five minutes of successful replay does not establish the maximum -retention period. The 120-second limit is ABCA policy, not an AWS guarantee. -Post-expiry behavior and recovery without unambiguous guest identity logs remain -unverified. Service response: pending. - -## F08 — Generic wake-hook failure while PID 1 owns its listener - -**Observed:** September 16, 18:55:13–18:56:16 UTC, account ``, -region `us-west-2`. A diagnostic image kept the original server as PID 1 and -added a separate observer child. Application code, connection keepalive, -8,192 MiB and lifecycle hooks matched normal image 5.0. - -- Worker: `microvm-da3668f9-04a6-393c-939a-33c059e4b2a0`. -- Image: `backgroundagent-dev-p3-pid1-observer-20260916:1.0`. -- Task: `01M2NRA1BNVMXGD5YDVYD4XH4T`. -- Suspend receipt: `802e1264-4569-49ef-aada-70be7c26e71b`; guest checkpoint - and suspend HTTP 200 completed at 18:55:14.013. -- Resume receipt: `11bc8d9b-ada5-440e-aef3-034d8ac04848`, accepted at - 18:56:13.121. -- At 18:56:13.350, observer PID 7 saw PID 1 running and owning its port-8080 - listening socket, inode `481`, which also existed before the freeze. -- AWS terminated the worker at 18:56:13.869 with - `Resume lifecycle hook failed. Please check your hook endpoint and application logs for more details.` - The retained log contains no resume hook entry, stage or access line. - -**Ask:** what exact connection/HTTP error maps to this generic reason? For this -receipt, was the hook attempted on a fresh or reused socket, what bytes/status -were received, and was another connection attempted before termination? -Does the service retain the hook-attempt timeline separately from the API -receipt? Compare it with the six exact connection refusals in F01. - -**Limits:** the observed process and listener existed 519 ms before termination. -This does not prove continuous health or event-loop responsiveness, and it does -not establish whether this failure shares F01's cause. The diagnostic process -can affect scheduling. The [full record](./645-p3-pid1-observer-20260916.md) -retains the passing comparison, excluded fixture assertion, sampled state, -timestamps and failure. Service response: pending. - -## F09 — Image refactor preview passes but execution rejects the tag schema - -**Observed:** September 17, 13:52–13:54 UTC, account ``, -region `us-west-2`. A freshly built isolated nested deployment used the production -MicroVM construct and image artifact. The test attempted to move only the image's -CloudFormation ownership from child to parent; no worker was running. - -- Image: `abca-645-refactor-probe-20260917:1.0`. -- Parent stack: `backgroundagent-dev-p3-refactor-20260917`, ID suffix - `434bf610-b29c-11f1-8640-0624a84210c3`. -- Child stack ID suffix: `44ff4b60-b29c-11f1-8386-06d8ca482b91`. -- Refactor: `b439bca8-a95f-4716-970d-dfad7c5b30d7`. -- Create request: `12b4130e-07ec-4540-aa80-81be74899243`. -- Execute request: `6f195c46-c1a5-4f40-912c-9fdc362eeac6`, - accepted at `2026-09-17T13:52:48.582Z`. -- Preview: `CREATE_COMPLETE` / `AVAILABLE`, one `MOVE`, with - `No configuration changes detected.` Both user tags would be preserved. -- Execution: `ROLLBACK_COMPLETE`, with: - `Stack Refactor does not support AWS::Lambda::MicrovmImage because the resource type defines an unsupported tag schema.` - -**Impact:** existing images cannot use this native stack-refactor path to move -into the requested nested layout. The successful preview does not expose the -failure before execution. The normal deployment was not modified. Independent -rollback checks confirmed the image's original ARN, only version 1.0, settings, -tags and all 21 resource identities. - -**Schema evidence:** `DescribeType` reports `FULLY_MUTABLE` and updatable tags. -`Tags` is an array of objects whose `required` list contains only `Key`; `Value` -is optional. `AWS::Lambda::NetworkConnector` has the same shape. By comparison, -S3 requires both `Key` and `Value`. The exact internal rejection condition and -connector behavior have not been established. - -**Ask:** can the MicroVM image provider support CloudFormation stack refactoring, -or document a supported image-preserving retain/import procedure? Is the -optional tag `Value` causing the schema rejection, and does the connector need -the same correction? Can the preview perform this validation before making the -operation executable? Please also confirm how refactoring preserves an explicit -resource tag that overlaps a source-stack tag. - -The [nested-stack record](./645-p3-nested-stack.md#image-ownership-refactor-execution-rejected-rollback-verified) -contains the preparation controls, exact receipts and rollback evidence. -Service response: pending. This feedback has not been submitted. - -## Updating this tracker - -For each new finding, add the actual trigger, UTC window, region, worker/image -versions, AWS receipt IDs, impact, smallest supported conclusion and a concrete -service question. Link raw evidence through the verification report. Retain -unsuccessful controls and distinguish application fixes from service findings. -Record any service answer and the verification needed before closing the item. +# Lambda MicroVM service-team feedback + +These questions come from the P1–P3 integration, most recently checked in +September 2026. They are not confirmed service defects in every Region/build. +Detailed worker timelines and receipts are archived privately for escalation. +No service-side trace confirmation has been obtained for the historical wakes. + +## Hook transport and diagnostics (F01, F03, F05, F08) + +**Observed:** some Resume API calls succeeded, then workers terminated with +“Resume lifecycle hook connection was refused” or the generic hook-failed reason. +In instrumented cases, the guest listener remained present and no new resume +handler entry was logged. Local paused-process controls reproduced failure when +reusing an old HTTP connection; a fresh connection succeeded. Longer-sleep +controls and explicit connection-close responses passed live checks. + +**Application correction:** lifecycle responses send `Connection: close` before +freeze. Approve/deny/expiry/cancellation and repeated wakes passed with that +correction. Passing controls do not prove the exact service error for each +historical worker. + +**Ask:** confirm the deployed hook client's pooling/retry behavior, whether stale +connections can survive suspend, and supported clock/timer semantics on restore. +Consider retiring pooled connections at suspend or retrying a failed dispatch on +a fresh connection. Distinguish refused connections, resets, incomplete responses +and timeouts in customer-visible errors. Expose hook-attempt timestamps, error +classification and correlation to the originating API receipt. A failed hook +that never reaches the guest cannot emit its own application log. + +## IAM setup (F02) + +**Observed:** tested source-conditioned role trust and `iam:PassedToService` +conditions rejected operations that worked without those conditions. A target-role +assumption problem also appeared as a caller-side PassRole denial. These are +historical integration findings, not a fresh comparison of every service build. + +**Ask:** document supported keys/values for build, execution and connector-role +assumption and both PassRole paths. Distinguish caller authorization failures +from target-role trust failures. Recheck service support before tightening the +working exact-role grants; IAM simulation alone does not establish which +condition values the service sends. + +## Restore state and connector schema (F04, F06) + +A worker was observed going `SUSPENDED → PENDING → RUNNING`. ABCA now distinguishes +restore from first startup using saved lifecycle intent. Publish the complete +transition/timestamp contract or expose a restoring state. + +A `VPC_EGRESS` connector required an operator role despite the generated property +being optional. Add conditional validation and a complete role/permission example; +this does not imply other connector types need the same role. + +## Lost Run response and idempotency (F07) + +ABCA preserves the exact Run request/token and bounds automatic replay. A lost +successful response can still leave a running worker with no saved handle. +Matching an image, role and start window is not a unique task identity. Accepted +guest identity logs allowed manual recovery in a tested case. + +**Ask:** specify token retention, behavior after expiry/termination/image changes, +and a supported token or request-ID lookup that returns the original worker. +Clarify required CloudTrail event configuration and retention. ABCA's 120-second +replay policy and observed successful replays are not service guarantees. + +## CloudFormation refactoring (F09) + +An image refactor preview succeeded, but execution rejected +`AWS::Lambda::MicrovmImage` because of an unsupported tag schema. Rollback +preserved the original image/resources. + +**Ask:** support image ownership refactoring or reject unsupported moves during +preview; publish a supported migration procedure. Until verified otherwise, use +the [overlap migration prerequisites](./645-p3-nested-stack.md), not a direct +flat-to-nested template switch. diff --git a/docs/verification/645-lifecycle-intent.md b/docs/verification/645-lifecycle-intent.md deleted file mode 100644 index 19d85652d..000000000 --- a/docs/verification/645-lifecycle-intent.md +++ /dev/null @@ -1,85 +0,0 @@ -# MicroVM lifecycle intent: local implementation and verification - -Status: foundation implemented locally on 2026-09-13; [production supervisor/API integration](./645-p3-supervisor.md) added on 2026-09-15. **Not deployed; new suspension defaults off.** This is the next foundation after `f3e684d4`. The [P3 checklist](./645-p3-implementation-plan.md) tracks the remaining integration and live gates. - -## In plain language - -The supervisor needs a saved instruction that says “this particular task's computer should sleep” or “it must wake up.” Saving it in the database lets a replacement supervisor continue after a restart. - -Each instruction has a unique revision stamp. A supervisor holding an older stamp cannot overwrite a newer instruction. Once “wake up” is saved for an approval request, that same request cannot become “sleep” again. The record stays until the task expires; deleting it would let an old supervisor mistake the empty space for permission to save its old sleep request. - -## What the code does - -- `cdk/src/handlers/shared/microvm-lifecycle.ts` reads the task and its current approval consistently, validates identity, and saves intent with database conditions. -- `cdk/src/handlers/shared/microvm-lifecycle-policy.ts` chooses an action from observations. It makes no AWS calls or human decisions. Its returned action is reconciled by the durable supervisor. -- `SessionStatus.microvmState` carries explicit MicroVM state alongside the existing coarse status. Known SDK states are preserved; absent/future states become local `UNKNOWN`, and a not-found response becomes local `NOT_FOUND`. These last two are observations defined by this application, not service states. Existing status/reason behavior is preserved. - -The internal `microvm_lifecycle` task attribute contains: - -| Field | Meaning | -|---|---| -| `version` | Record format, currently `1` | -| `generation` | Unique revision stamp; replaced when the gate/action changes | -| `microvm_id` | The computer this instruction belongs to | -| `request_id` | The approval request, or null when recovering a working task | -| `action` | `suspend` or `resume` | -| `requested_at_ms` | Original request time, milliseconds since the Unix epoch | -| `deadline_ms` | Original approval expiry, or null if no valid deadline is available for conservative wake | - -Repeated saves of the same action for the same VM/gate retain their generation and request time. This keeps retries from resetting the future recovery budget. Each database transaction gets its own AWS request token; reusing the persistent generation as that token would conflict when the transaction's conditions change on replay. - -This is coordinator data, with no public task-type or agent contract change. The existing worker attribute allowlist excludes it; the session-role regression now names it explicitly. Real AWS permission verification remains pending. - -## Conditions and races - -Every write checks the current task owner, status, session ID, VM ID, endpoint, approval-request identity and previous instruction generation. A first write requires the attribute to be absent. A suspend also condition-checks the **same approval row** in the transaction: still PENDING, same owner, original creation time and timeout. A concurrent decision, changed gate or cancellation rolls back the whole write. - -Resume does not require a readable approval row. If that row is missing, malformed or temporarily unreadable, waking an already-sleeping task allows the existing agent decision loop to recover or expire it. Approval-read failures are explicit observations, including a safe error class; integration must report and count them. - -A lost database reply triggers readback. Only the exact saved generation with still-eligible task/gate identity is returned as saved. A known committed write followed by cancellation or a decision returns stale. An unresolved outcome remains an error. Cancelled/closed tasks may retain their old approval pointer; reads preserve it for diagnosis without reading the approval again, and writes cannot revive those tasks. - -**Saving intent does not lock AWS.** There remains a gap between a database write and a compute command. The future caller must reread identity/intent/deadline before suspend, persist wake requests even when the VM still looks awake, and reconcile after both acknowledgements and uncertain outcomes. Guest hooks must also validate the live task/gate before allowing a freeze or resumed work. - -## Initial policy - -These are local policy choices to validate with live measurements: - -| Setting | Initial value | -|---|---| -| Wait before considering sleep | 30 seconds from gate creation | -| Wake before approval/session deadline | 60 seconds | -| Minimum useful sleep interval | 30 seconds | -| Poll during a transition/unconfirmed observation | At most 5 seconds | -| Database request/read-sequence budget | 5 seconds; write plus recovery can use two budgets | - -A caller supplies valid times, session expiry, its normal poll interval and an enable switch. New suspension also requires compatible-image evidence from the snapshot's saved worker handle. The [image capability implementation](./645-p3-image-capability.md) verifies the actual launched image ARN/version and persists the supported protocol in coordinator-owned metadata. Both policy and store reject unknown/legacy capability; the suspend transaction rejects image metadata changes after the read. Current deployment settings cannot authenticate an older worker. Disabling new suspends still permits wake and termination. Long poll intervals are shortened to the next grace/wake/session deadline. - -A PENDING approval alone is not a wake condition. An intended suspended VM can wait while there is sufficient time. APPROVED, DENIED, TIMED_OUT, STRANDED, deadline proximity, missing/invalid data or unintended suspension require wake/recovery. While the service reports SUSPENDING, save desired resume but return `requestReady: false`; issue ResumeMicrovm only after observing SUSPENDED. A wake acknowledgement followed by a delayed old suspend remains repairable because wake intent is retained. - -Terminal/cancelled/finalizing tasks cannot resume. Terminal VM observations go through existing task reconciliation. PENDING/UNKNOWN or a legacy coarse `running` result cannot authorize suspend or confirm wake. The supervisor now bounds persistent failures and unconfirmed recovery across serialized polls. - -## Verification and deployment gates - -The unit suite checks the policy and store contract. The opt-in DynamoDB Local suite checks actual transaction expressions, cross-table rollback, competing/stale writers, changed task/approval identities, later gates, lost committed replies, cancellation during recovery and a fresh module/client reading the saved intent. - -Recorded local results (2026-09-13): CDK lint/compilation passed; **158 suites / 3,738 tests** passed in the broad handler/session-role run, including **23 lifecycle** and **15 existing capacity** database tests. Five relevant suites passed **218 overlapping tests** and exited normally. The broad run exited successfully after a delay without an open-handle trace; the cause is unconfirmed. Documentation sync/build/link checks passed, and the temporary database container was removed. - -Run locally from `cdk/` with a loopback DynamoDB Local instance: - -```sh -ABCA_DDB_LOCAL_ENDPOINT=http://127.0.0.1: mise run testf -- test/handlers/shared/microvm-lifecycle-local.test.ts -``` - -The tests use dummy credentials, create uniquely named tables and delete them afterward. They do not prove AWS IAM or MicroVM behavior. - -Before enabling automatic sleep: - -1. Complete a clean P2 deployment/rerun, including the earlier bootstrap, metadata, capacity and managed-image gates. Follow the coordinated v2 drain/rollout procedure. -2. Implement guest lifecycle context, acknowledged progress durability, credential refresh preserving task identity, resume barriers and snapshot randomness handling. Reuse the original approval deadline. -3. **Implemented locally:** policy/store are connected to durable supervisor polling, preserving counters, next-poll delay, anomaly episodes and a bounded wake-recovery clock. Handle stale results by observing again. Never reset the recovery clock on repeated saves. -4. Connect approve/deny after the decision commits, with bounded best-effort wake and repair diagnostics. Preserve current decision responses on wake failure. -5. Add and verify the coordinator's required task/approval transaction permissions and scoped MicroVM lifecycle grants. Grant no lifecycle action or intent-write permission to workers. Check the total handler time budget, not only each individual call. -6. Deploy compatible hooks/image and coordinator together with automatic suspension initially disabled. Validate actual transition/conflict/timeout behavior and then the full P3 acceptance matrix in an isolated development deployment. -7. Exercise disable/rollback with sleeping tasks: stop new suspends, continue wake/expiry/termination, and drain before removing compatible code. Do not delete intent records to “reset” recovery. - -The existing approval API uses the PENDING/task/gate transaction conditions; it has no independent wall-clock expiry check. The agent owns the conditional TIMED_OUT write, and the first committed decision wins. This implementation preserves that behavior. Strict API deadline rejection would be a separate behavior change. diff --git a/docs/verification/645-microvm-image-rebuild-20260914.md b/docs/verification/645-microvm-image-rebuild-20260914.md deleted file mode 100644 index 9af17a9a7..000000000 --- a/docs/verification/645-microvm-image-rebuild-20260914.md +++ /dev/null @@ -1,134 +0,0 @@ -# Managed MicroVM image rebuild verification - -## Purpose and status - -The original managed image referenced `microvm-images/agent-artifact.zip`. -Uploading new bytes to that filename did not change the CloudFormation -template, so an ordinary deployment could leave the old agent image active. - -The fix is committed as `e1d5debe`. The full build passed, the normal -CloudFormation update built and activated image version `2.0`, and a repeat -deployment reported no changes. This work -does not complete the remaining P2 acceptance matrix or enable P3 suspension. - -## Deployment workflow - -1. Run the full root `mise run build` before deployment. -2. Use the intended AWS profile and Region, then run - `cdk/scripts/package-microvm-artifact.sh --stack-name `. - An initial deployment without an image must create the artifact bucket first. -3. The script packages the Dockerfile's local inputs and prints their ZIP's - SHA-256 digest. It uploads - `microvm-images/agent-artifact-.zip`, verifying the checksum with S3. - Repeating the upload reuses the object only after verifying its checksum. - It refuses to overwrite an existing object with a different checksum. -4. Synthesize and review the deployment using `compute_type=lambda-microvm`, - the existing `microvm_base_image_arn` and `microvm_base_image_version`, and - the printed `microvm_artifact_sha256`. Execute the reviewed change set. -5. Verify both CloudFormation completion and the actual image build/version. - Retain these context inputs for future deployments, including unrelated - infrastructure changes. After changing agent source, package again and use - the newly printed digest. - -The digest is a fingerprint of the ZIP bytes. Source file timestamps, checkout -paths and caches do not affect it. Changed runtime inputs change the filename, -which changes `CodeArtifact.Uri` and requests an update to the existing image. -The packaging step remains explicit; CDK does not upload this artifact itself. -Base-image changes can also request a rebuild while using the same artifact. - -The build role reads the selected immutable object and the legacy base object -only. The script's explicit `--create-image` mode keeps using the base object -and directly requests an out-of-band build. The new -`MicrovmArtifactBaseObjectKey` output prevents repeated managed packaging from -adding a second digest onto the previous filename. Older stack outputs work -for the first upgrade because their object key is the unsuffixed base. - -For an artifact rollback, retain the previous ZIP and deploy its digest after -reviewing the change set. This requests another build; it does not promise that -AWS retains or instantly reactivates a previous image version. Artifact objects -have no automatic expiration. Missing or malformed managed-image digests fail -synthesis with packaging instructions. - -## Local verification - -- The pre-fix regression reproduced identical image URIs for two artifact - revisions and acceptance of missing/invalid digests. -- Four targeted suites passed 229 tests: the real shell/Python packagers with - an isolated fake AWS CLI, construct image/IAM assertions, stack wiring and - the production CDK-nag path. -- Packaging cases cover byte-for-byte reuse despite changed timestamps/caches, - a changed runtime source, checksum mismatch, authorization failure, - symlink rejection, custom base keys and manual image creation. A guard - compares the packager's input manifest against every local Dockerfile COPY. -- Image assertions keep the logical ID and name stable while changing the URI. - Build-role reads remain exact object ARNs. -- Shell and Python syntax checks pass. -- Full root build: passed, exit 0 in 897.66 seconds. CDK: 223 suites / - 4,797 tests; CLI: 62 suites / 928 tests; Python: 1,823 tests. Compilation, - lint, formatting, types, drift checks, the 77-page docs build and links pass. - Two existing DynamoDB Local suites / 38 cases were skipped because that - service was not running; this change does not modify the capacity protocol. - The first full run caught an omitted hook list in the revised warning; the - final run includes the corrected warning and its passing regression. - -## Live verification - -Target: existing `backgroundagent-dev`, account ``, `us-west-2`, -profile `sphia-dev`, bootstrap policy bundle `1.7.0`. - -Before this update, CloudFormation was `UPDATE_COMPLETE`, the managed image -`backgroundagent-dev-abca-agent` had latest active version `1.0`, and all four -listed task MicroVMs were terminated. AWS's resource schema marks only `Name` -as create-only; `CodeArtifact.Uri` supports an in-place update. - -- [x] Upload and checksum-verify the immutable artifact in S3. -- [x] Review a change set preserving image identity and existing infrastructure. -- [x] Execute the normal update and verify a successful new active image version. -- [x] Verify repeating the same artifact digest requests no further image change. - -The uploaded ZIP contains **106 files / 467,322 bytes**. Every archived file -matches its source bytes; ZIP integrity checks pass. S3 reports the matching -SHA-256 checksum and AES256 encryption. The artifact digest is -`86219317d92fc501b58d21df02dcb298955576c28751703f0753cc655c1c0a21`. - -CDK prepared change set `abca-645-image-rebuild-20260914`. AWS reported ten -modifications and no additions/removals: the image URI and build-role policy, -six metadata-only changes, the AgentCore container reference and the existing -`awslabs/agent-plugins` blueprint timestamp refresh. The image has -`Replacement: False`. The blueprint's custom-resource physical ID is stable; -its update refreshes active status/time for that blueprint. No target-repository -verification overrides were restored. - -The incidental AgentCore update comes from its existing repository-root asset -fingerprint; `agent/` and `contracts/` source still match the earlier deployed -`29dcaa74`. A textual `GetTemplate` comparison also showed Unicode characters -as question marks, including a layer description, although the original -synthesized template contains Unicode. The actual AWS change set excludes -those apparent description changes and does not replace that layer. - -The build-role policy completed its update at **16:08:15 UTC**, before the image -update started at **16:08:17 UTC**. Version `2.0` used the hash-suffixed URI; -`/ready` and `/validate` returned HTTP 200 in its version-specific log streams. -Validation reported zero warnings. The version reached `SUCCESSFUL` / `ACTIVE`, -and the image's `latestActiveImageVersion` became `2.0` under its original ARN: -`arn:aws:lambda:us-west-2::microvm-image:backgroundagent-dev-abca-agent`. -CloudFormation reached `UPDATE_COMPLETE`, retaining **474 root resources**. -No manual IAM changes or image API updates were needed. - -Running the packager again against the upgraded stack outputs reused the same -checksum-verified object and printed the same digest. A normal CDK deployment -of the identical reviewed cloud assembly exited 0 with **`(no changes)`** and -zero deployment time. The subsequent version list contains `1.0` and `2.0`; -no `3.0` was created. Fresh synthesis can still refresh the unrelated blueprint -timestamps described above, while unchanged artifact input preserves the image URI. - -All four listed task MicroVMs remain terminated. This verification covers image -rebuild and activation; no new coding task was launched. The earlier coding, -iteration and cancellation evidence remains in the -[version 1.0 live task record](./645-p2-live-task-20260914.md). -P3 `/suspend` and `/resume` hooks remain undeclared. - -Private command output and AWS responses are retained under -`/tmp/abca-645-p2-clean-20260913/image-rebuild-*`. -The [P3 implementation plan](./645-p3-implementation-plan.md) tracks the -remaining failure/recovery, permission/network and sleep/wake work. diff --git a/docs/verification/645-p1-lambda-microvm-runbook.md b/docs/verification/645-p1-lambda-microvm-runbook.md deleted file mode 100644 index 12d1d56dd..000000000 --- a/docs/verification/645-p1-lambda-microvm-runbook.md +++ /dev/null @@ -1,2198 +0,0 @@ -# ADR-021 P1 Lambda MicroVM verification runbook - -Working verification document for issue #645 / PR #689 on branch -`feat/645-lambda-microvm-p1`. This is for a **new, CDK-bootstrapped sandbox -account in which ABCA has never been deployed**. It is not a production deploy -guide. - -`docs/scripts/sync-starlight.mjs` mirrors only selected guides, `docs/design/`, -`docs/decisions/`, `CONTRIBUTING.md`, and assets. It does **not** mirror -`docs/verification/`, so this file intentionally stays here. - -## ⚠️ This runbook predates the Stage D fixes - -The pass recorded under “Live execution results” below ran against the code as of -`0505f914`, and its findings (F1–F14) were then fixed. The **instructions** above -each step have been updated only where a fix renamed something an instruction -asserts on. The following instructions are known to still describe the pre-fix -world and must be adjusted by whoever runs this again: - -| Step | Stale instruction | Post-fix reality | -|---|---|---| -| 1.2 | ~~unescaped `ParameterValue=agentcore,lambda-microvm`; bootstrap without `--force`~~ | **CORRECTED IN PLACE** — the comma is now escaped (`agentcore\,lambda-microvm`) and both bootstrap invocations carry `--force`. F14/F10 are the evidence; nothing is left to adjust here | -| 2.2 | "a 443-only security group" and one `AWS::Lambda::NetworkConnector` | **two** security groups (443-only runtime, 443 + 80 build) and **two** connectors, plus a connector operator role; a seventh output `MicrovmBuildEgressConnectorArns` | -| 2.3 | `list-stack-resources --query "…|[0]"` | needs `--no-paginate` (F14) | -| 4.3 | `export IMAGE_VERSION=1` | the service returns `1.0`; there are two builds per version (one per chipset) | -| 5.1 / 5.9 | `--image-identifier "$IMAGE_NAME"` | an **ARN** is required (F3); use the `imageArn` the script prints | -| 5.9 | 16 384 / 16 385-byte probes | the real boundary is **4 096 / 4 097** (F6) | -| Phase 5 | "the hook-less P1 image" framing | the image now declares AND serves `/ready` + `/run`, so hook behaviour is a different experiment | - -## Important P1 and tooling limits - -- P1 provisions and can start the substrate, and — **as of the Stage D fixes** — - its image declares AND the agent serves `/ready` + `/run`, so the image is - creatable, launchable and payload-deliverable. What P1 has no guarantee of is - smoke parity (Memory grants, snapshot env parity, egress specifics from a - running MicroVM, heartbeats). *Pre-fix, this bullet read "not runnable end to - end" because the plan was to declare `/run` in P1 and serve it in P2; the live - run proved that is not a reachable service state (F1).* -- P1's orchestrator role intentionally has only `RunMicrovm`, `GetMicrovm`, - `TerminateMicrovm`, `PassNetworkConnector`, and the required `iam:PassRole`. - It does **not** have `SuspendMicrovm`, `ResumeMicrovm`, or - `CreateMicrovmAuthToken`. Therefore use the orchestrator role for the P1 IAM - checks when it can be assumed, but use the sandbox administrator identity for - the manual suspend/resume experiment. This is a deliberate P1/brief mismatch. -- The repository's AWS SDK model is - `@aws-sdk/client-lambda-microvms@3.1098.0`. It verifies operation names, - request keys, state enums, and `delete-microvm-image-version`. The local - `aws-cli/2.35.8` does **not** recognize `aws lambda-microvms`; consequently all - `aws lambda-microvms ...` commands below are **best-effort CLI spellings - derived from that SDK model and the repository packaging script**, not locally - CLI-validated. No minimum AWS CLI release containing this service could be - established. The executor must install a CLI build for which - `aws lambda-microvms help` succeeds. Do not proceed with image/lifecycle work - merely because `aws --version` is newer than 2.35.8. -- The packaging-script model drift is resolved: its direct service request now - uses SDK 3.1098.0's `ARM_64` architecture and `ENABLED|DISABLED` hook-state - shape (with port and timeout), rather than CloudFormation's `arm64` and hook - path strings. The CDK L1 remains intentionally unchanged because its generated - CloudFormation types accept string values and document no architecture/hook - allowed-value constraint. Step 4.2 still captures CLI help/input skeleton as - a live check because the local CLI cannot validate this service offline. -- Commands are run from the repository root. `mise` is primary. Commands marked - **raw fallback** are only for a machine without `mise`. -- **Run the whole thing under `set -o pipefail`.** Several steps pipe a command - through `tee`; without `pipefail` the pipeline reports `tee`'s exit status and a - failed command looks like a success. The 2026-07-31 pass recorded `EXIT=0` for a - `package-microvm-artifact.sh` run that had actually failed service validation - (F14). The script now also prints an explicit - `!! package-microvm-artifact.sh FAILED (exit N) !!` marker on any failure, so - the teed log carries the truth either way — but set the option anyway: - - ```bash - set -o pipefail - ``` - -## Variables and evidence directory - -**Purpose:** make every subsequent command target one account, Region, and stack. - -```bash -export AWS_REGION=us-east-1 -export AWS_DEFAULT_REGION="$AWS_REGION" -export CDK_DEFAULT_REGION="$AWS_REGION" -export STACK_NAME=backgroundagent-dev -export EXPECTED_BRANCH=feat/645-lambda-microvm-p1 -export EVIDENCE_DIR="/tmp/abca-645-p1-$(date -u +%Y%m%dT%H%M%SZ)" -mkdir -p "$EVIDENCE_DIR" -``` - -If using a profile, also `export AWS_PROFILE=`. Supported -Regions are `us-east-1`, `us-east-2`, `us-west-2`, `eu-west-1`, and -`ap-northeast-1`; this runbook defaults to `us-east-1`. - -**Expected:** the directory exists and all variables print non-empty. - -**Record:** variable values and evidence-directory path. - -**ADR-021 item:** regional availability enforcement and reproducibility. - ---- - -## Phase 0 — Preflight - -### 0.1 Verify identity, branch, and virgin account - -**Purpose:** prevent deploying to the wrong account/branch and fail fast if this -is not the assumed first ABCA deployment. - -```bash -aws sts get-caller-identity | tee "$EVIDENCE_DIR/caller-identity.json" -export ACCOUNT_ID="$(aws sts get-caller-identity --query Account --output text)" -test "$(git branch --show-current)" = "$EXPECTED_BRANCH" -git status --short --branch | tee "$EVIDENCE_DIR/git-status.txt" - -if aws cloudformation describe-stacks --stack-name "$STACK_NAME" \ - >"$EVIDENCE_DIR/unexpected-existing-stack.json" 2>"$EVIDENCE_DIR/stack-absence.txt"; then - echo "STOP: $STACK_NAME already exists; this runbook requires a virgin account." >&2 - exit 1 -fi -``` - -The expected CloudFormation error is `ValidationError: Stack ... does not -exist`. Any other error (especially `AccessDenied`) is not proof of absence; -stop and fix credentials. - -**Expected:** account is the intended sandbox, branch test passes, only known -local files are shown, and `backgroundagent-dev` is absent. `CDKToolkit` may and -normally will exist. - -**Record:** account ID, caller ARN, branch SHA (`git rev-parse HEAD`), status, -and the exact stack-absence response. - -**ADR-021 item:** clean first-deploy/substrate bootstrap path. - -### 0.2 Record tools and install dependencies - -**Purpose:** capture the exact client/service models and prepare CDK and CLI. - -```bash -aws --version 2>&1 | tee "$EVIDENCE_DIR/aws-version.txt" -node --version | tee "$EVIDENCE_DIR/node-version.txt" -npx cdk --version | tee "$EVIDENCE_DIR/cdk-version.txt" -mise --version | tee "$EVIDENCE_DIR/mise-version.txt" -python3 --version | tee "$EVIDENCE_DIR/python-version.txt" -zip -v | tee "$EVIDENCE_DIR/zip-version.txt" -rsync --version | tee "$EVIDENCE_DIR/rsync-version.txt" - -MISE_EXPERIMENTAL=1 mise run install -``` - -**Raw fallback if `mise` is absent:** - -```bash -yarn install --check-files -``` - -Then perform the mandatory service-client gate: - -```bash -aws lambda-microvms help >"$EVIDENCE_DIR/lambda-microvms-help.txt" -node -p "require('./node_modules/@aws-sdk/client-lambda-microvms/package.json').version" \ - | tee "$EVIDENCE_DIR/lambda-microvms-sdk-version.txt" -``` - -**Expected:** install succeeds; SDK version is `3.1098.0`; the help command -lists at least `run-microvm`, `get-microvm`, `suspend-microvm`, -`resume-microvm`, `terminate-microvm`, `create-microvm-auth-token`, and -`delete-microvm-image-version`. If help fails, stop and update/install the AWS -CLI distribution that exposes the preview/new service. - -**Record:** every version and whether service help was available. - -**ADR-021 item:** empirical IAM/API action-name verification. - ---- - -## Phase 1 — Least-privilege bootstrap - -### 1.1 Inspect the existing bootstrap - -**Purpose:** determine whether the standard bootstrap lacks ABCA's generated -template and custom compute policies. - -```bash -aws cloudformation describe-stacks --stack-name CDKToolkit \ - | tee "$EVIDENCE_DIR/cdktoolkit-before.json" -aws cloudformation get-template --stack-name CDKToolkit \ - --query TemplateBody --output text >"$EVIDENCE_DIR/cdktoolkit-template-before.txt" -python3 - "$EVIDENCE_DIR/cdktoolkit-template-before.txt" <<'PY' -import pathlib, sys -s = pathlib.Path(sys.argv[1]).read_text() -for needle in ("ComputeTypes", "IaCRoleABCAComputeLambdaMicrovms"): - print(needle, "present" if needle in s else "ABSENT") -PY -``` - -**Expected:** a standard bootstrap may report both markers absent. That is the -reason for the next step, not a failure. - -**Record:** template markers and current CDKToolkit parameters. - -**ADR-021 item:** conditional bootstrap policy exists only with the custom -template. - -### 1.2 Re-bootstrap, then set the CloudFormation parameter - -**Purpose:** replace the standard administrator bootstrap with the repository's -generated least-privilege template, then enable both AgentCore and Lambda -MicroVM deployment permissions. `cdk bootstrap` has no `--parameters`; CDK -context is not a substitute for this CloudFormation parameter. - -```bash -# `--force` is REQUIRED on an already-bootstrapped account: without it the CDK CLI -# refuses to replace the default template and exits 0 ("Bootstrap stack already -# exists, containing 'AWS CDK: Default Resources'. Not overwriting it…"), leaving -# AdministratorAccess attached while looking like a success (F10). Note also that -# BootstrapVariant stays 'AWS CDK: Default Resources' afterwards, so every future -# non-forced bootstrap refuses again. -MISE_EXPERIMENTAL=1 mise //cdk:bootstrap -- --force - -# The comma MUST be backslash-escaped. The CLI's shorthand parser otherwise splits -# on it and rejects the call (F14): -# "Invalid type for parameter Parameters[0].ParameterValue, -# value: ['agentcore', 'lambda-microvm'], type: , -# valid types: " -# The comment above `[tasks.bootstrap]` in cdk/mise.toml shows the same escaping. -aws cloudformation update-stack \ - --stack-name CDKToolkit \ - --use-previous-template \ - --capabilities CAPABILITY_NAMED_IAM \ - --parameters 'ParameterKey=ComputeTypes,ParameterValue=agentcore\,lambda-microvm' -aws cloudformation wait stack-update-complete --stack-name CDKToolkit -aws cloudformation describe-stacks --stack-name CDKToolkit \ - --query 'Stacks[0].Parameters' \ - | tee "$EVIDENCE_DIR/cdktoolkit-parameters-after.json" -``` - -**Raw fallback if `mise` is absent** (run from `cdk/`): - -```bash -npx tsx scripts/generate-bootstrap-artifacts.ts -npx tsx scripts/generate-bootstrap-template.ts -# --force for the same reason as above (F10). -npx cdk bootstrap --template bootstrap/bootstrap-template.yaml --force -``` - -Then run the same `aws cloudformation update-stack` parameter dance above, -including the escaped comma. - -**Expected:** `ComputeTypes` is exactly `agentcore,lambda-microvm`. The custom -template replaces default `AdministratorAccess` with generated ABCA policies. - -**Record:** update stack ID/events and final parameters. - -**ADR-021 item:** “where `ComputeTypes` includes `lambda-microvm`, attach -`IaCRole-ABCA-Compute-LambdaMicrovms`.” - -### 1.3 Verify policy creation and attachment - -**Purpose:** prove the MicroVM CloudFormation permissions are attached to the -actual execution role. - -```bash -export CFN_EXEC_ROLE="$(aws cloudformation describe-stack-resource \ - --stack-name CDKToolkit \ - --logical-resource-id CloudFormationExecutionRole \ - --query StackResourceDetail.PhysicalResourceId --output text)" - -aws iam list-policies --scope Local \ - --query "Policies[?contains(PolicyName, 'IaCRole-ABCA-Compute-LambdaMicrovms')].[PolicyName,Arn]" \ - --output table | tee "$EVIDENCE_DIR/microvm-bootstrap-policy.txt" -aws iam list-attached-role-policies --role-name "$CFN_EXEC_ROLE" \ - | tee "$EVIDENCE_DIR/cfn-exec-attached-policies.json" -``` - -**Expected:** one generated policy whose name contains -`IaCRole-ABCA-Compute-LambdaMicrovms` exists and its ARN is attached to -`$CFN_EXEC_ROLE`. - -**Record:** execution role name, policy ARN, and attachments. - -**ADR-021 item:** conditional bootstrap policy and verified IAM action names. - -**Optional negative deliberately skipped:** a scratch-qualifier bootstrap with -only `agentcore` would create another bootstrap stack, buckets, ECR repository, -roles, and policies merely to prove a template condition already covered by CDK -tests. It is not cheap enough for the core pass and complicates teardown. Run it -only if specifically requested, and destroy every scratch bootstrap resource. - ---- - -## Phase 2 — Substrate-only deploy (no image context) - -### 2.1 Synthesize and deploy the bootstrap state - -**Purpose:** verify the intended first-deploy state: connector, buckets, roles, -and logs exist while no image or orchestrator image configuration exists. - -```bash -MISE_EXPERIMENTAL=1 mise //cdk:synth -- \ - "$STACK_NAME" --context compute_type=lambda-microvm \ - 2>&1 | tee "$EVIDENCE_DIR/substrate-synth.txt" - -MISE_EXPERIMENTAL=1 mise //cdk:deploy -- \ - "$STACK_NAME" --require-approval never \ - --context compute_type=lambda-microvm \ - 2>&1 | tee "$EVIDENCE_DIR/substrate-deploy.txt" -``` - -**Raw fallback if `mise` is absent** (run from `cdk/`): - -```bash -npx cdk synth "$STACK_NAME" --context compute_type=lambda-microvm -npx cdk deploy "$STACK_NAME" --require-approval never --context compute_type=lambda-microvm -``` - -**Expected:** synth includes warning ID -`abca:microvm-image-not-provisioned`; deploy completes. This is intentionally -not `abca:microvm-image-p1-smoke-unverified` yet because no image is configured. -(The 2026-07-31 pass observed the pre-fix id `abca:microvm-image-p1-not-runnable`; -the warning was renamed when F1 was fixed.) - -**Record:** warning, deployment duration, stack ID/status, and failures/retries. - -**ADR-021 item:** conditional substrate and explicit no-image first-deploy -warning. - -### 2.2 Resolve exact outputs and resources - -**Purpose:** prove the script-facing substrate contract and capture physical IDs. - -```bash -aws cloudformation describe-stacks --stack-name "$STACK_NAME" \ - --query 'Stacks[0].Outputs' | tee "$EVIDENCE_DIR/stack-outputs-substrate.json" -aws cloudformation list-stack-resources --stack-name "$STACK_NAME" \ - | tee "$EVIDENCE_DIR/stack-resources-substrate.json" - -for key in ComputeSubstrate MicrovmArtifactBucketName MicrovmArtifactObjectKey \ - MicrovmBuildRoleArn MicrovmExecutionRoleArn MicrovmEgressConnectorArns \ - MicrovmLogGroupName; do - aws cloudformation describe-stacks --stack-name "$STACK_NAME" \ - --query "Stacks[0].Outputs[?OutputKey=='$key'].OutputValue | [0]" --output text -done | tee "$EVIDENCE_DIR/microvm-output-values.txt" -``` - -**Expected:** `ComputeSubstrate=lambda-microvm`; all six `Microvm...` outputs -are populated; artifact key is `microvm-images/agent-artifact.zip`. Stack -resources include two S3 buckets, build/execution roles, a 443-only security -group, `/aws/lambda-microvms/...` log group, and -`AWS::Lambda::NetworkConnector`; no `AWS::Lambda::MicrovmImage` exists. - -**Record:** outputs and physical IDs. - -**ADR-021 item:** construct resources, egress connector, build/execution roles, -artifact/payload buckets. - -### 2.3 Verify no orchestrator `MICROVM_*` environment - -**Purpose:** prove partial image configuration is not injected. - -```bash -export ORCHESTRATOR_FN="$(aws cloudformation list-stack-resources \ - --stack-name "$STACK_NAME" \ - --query "StackResourceSummaries[?ResourceType=='AWS::Lambda::Function' && contains(LogicalResourceId, 'TaskOrchestrator')].PhysicalResourceId | [0]" \ - --output text)" -aws lambda get-function-configuration --function-name "$ORCHESTRATOR_FN" \ - --query 'Environment.Variables' | tee "$EVIDENCE_DIR/orchestrator-env-no-image.json" -aws lambda get-function-configuration --function-name "$ORCHESTRATOR_FN" \ - --query 'Environment.Variables' --output json \ - | python3 -c 'import json,sys; d=json.load(sys.stdin); print([k for k in d if k.startswith("MICROVM_")])' -``` - -**Expected:** the final line is `[]`. - -**Record:** function name and environment-key list (do not publish environment -values if later deployments add sensitive configuration). - -**ADR-021 item:** reject tasks when deployed without an image; all-or-nothing -strategy configuration. - -### 2.4 Verify backend cost tags - -**Purpose:** verify deployed, taggable construct resources carry -`abca:compute-backend=lambda-microvm`. - -```bash -aws resourcegroupstaggingapi get-resources \ - --tag-filters Key=abca:compute-backend,Values=lambda-microvm \ - --query 'ResourceTagMappingList[].ResourceARN' --output text \ - | tee "$EVIDENCE_DIR/microvm-tagged-resource-arns.txt" - -aws cloudformation get-template --stack-name "$STACK_NAME" \ - --query TemplateBody --output json >"$EVIDENCE_DIR/deployed-template.json" -python3 - "$EVIDENCE_DIR/deployed-template.json" <<'PY' -import json, pathlib, sys -t = json.loads(pathlib.Path(sys.argv[1]).read_text()) -for logical_id, r in t["Resources"].items(): - if "LambdaMicrovmCompute" not in logical_id: - continue - tags = r.get("Properties", {}).get("Tags") - print(logical_id, r["Type"], tags if tags is not None else "NOT-TAGGABLE/NO-TAGS") -PY -``` - -**Expected:** every taggable construct resource (buckets, roles, security group, -log group, and connector where supported) shows the backend tag. Generated -policies/bucket policies are not independently taggable resources. Save any -service that omits tags as a defect rather than silently accepting it. - -**Record:** tagged ARNs and the per-logical-resource template report. - -**ADR-021 item:** backend-identifying cost-allocation tags. - ---- - -## Phase 3 — Region gate (synth only) - -### 3.1 Reject an unsupported Region - -**Purpose:** prove static fail-fast enforcement without deploying there. - -```bash -env AWS_REGION=eu-central-1 AWS_DEFAULT_REGION=eu-central-1 \ - CDK_DEFAULT_REGION=eu-central-1 CDK_DEFAULT_ACCOUNT="$ACCOUNT_ID" \ - MISE_EXPERIMENTAL=1 mise //cdk:synth -- \ - "$STACK_NAME" --context compute_type=lambda-microvm \ - >"$EVIDENCE_DIR/unsupported-region-synth.txt" 2>&1 && { - echo "ERROR: unsupported-region synth unexpectedly succeeded" >&2; exit 1; - } -``` - -**Raw fallback if `mise` is absent** (run from `cdk/`): - -```bash -env AWS_REGION=eu-central-1 AWS_DEFAULT_REGION=eu-central-1 \ - CDK_DEFAULT_REGION=eu-central-1 CDK_DEFAULT_ACCOUNT="$ACCOUNT_ID" \ - npx cdk synth "$STACK_NAME" --context compute_type=lambda-microvm -``` - -**Expected:** failure names `eu-central-1`, all five supported Regions, and -`--context microvm_region_override=true`. - -**Record:** complete stderr. - -**ADR-021 item:** static unsupported-Region synth failure. - -### 3.2 Exercise the escape hatch - -**Purpose:** prove newly launched Regions can bypass only the static list. - -```bash -env AWS_REGION=eu-central-1 AWS_DEFAULT_REGION=eu-central-1 \ - CDK_DEFAULT_REGION=eu-central-1 CDK_DEFAULT_ACCOUNT="$ACCOUNT_ID" \ - MISE_EXPERIMENTAL=1 mise //cdk:synth -- \ - "$STACK_NAME" --context compute_type=lambda-microvm \ - --context microvm_region_override=true \ - 2>&1 | tee "$EVIDENCE_DIR/unsupported-region-override-synth.txt" -``` - -**Expected:** synth succeeds with warning `abca:microvm-region-override`. - -**Record:** warning and exit status. - -**ADR-021 item:** Region-list escape hatch. - -**Gotcha:** `src/main.ts` reads `CDK_DEFAULT_REGION`, not merely `AWS_REGION`. -An unresolved/region-agnostic CDK token skips the static check by design. The -commands set both account and Region to force the real test. Synth makes no -MicroVM control-plane calls, but CDK context/asset bundling may still require -valid AWS credentials and the bootstrap version parameter. - ---- - -## Phase 4 — Package and build an image - -### 4.1 Select a managed base image - -**Purpose:** pin a real regional base-image ARN/version rather than guessing. - -```bash -aws lambda-microvms list-managed-microvm-images \ - | tee "$EVIDENCE_DIR/managed-images.json" -export BASE_IMAGE_ARN="$(aws lambda-microvms list-managed-microvm-images \ - --query 'items[0].imageArn' --output text)" -aws lambda-microvms list-managed-microvm-image-versions \ - --image-identifier "$BASE_IMAGE_ARN" \ - | tee "$EVIDENCE_DIR/managed-image-versions.json" -# NEWEST FIRST (measured 2026-07-31): items[0] is the latest version, items[-1] -# is the OLDEST. The original `items[-1]` here selected version 0 instead of 1. -export BASE_IMAGE_VERSION="$(aws lambda-microvms list-managed-microvm-image-versions \ - --image-identifier "$BASE_IMAGE_ARN" \ - --query 'items[0].imageVersion' --output text)" -test -n "$BASE_IMAGE_ARN" && test "$BASE_IMAGE_ARN" != None -test -n "$BASE_IMAGE_VERSION" && test "$BASE_IMAGE_VERSION" != None -``` - -**Expected:** the regional probe succeeds and returns at least one ARN/version. -Ordering is newest-to-oldest, so `items[0]` is correct; inspect the timestamps -and explicitly export the desired version if the installed CLI ever differs. - -**Record:** complete catalogs and selected pair. - -**ADR-021 item:** live regional availability probe and managed base-image API. - -### 4.2 Package, upload, and start the out-of-band build - -**Purpose:** exercise the actual script interface and avoid slow CloudFormation -iteration while still using CDK-created bucket, role, connector, and logs. - -Before running it, capture the installed CLI's authoritative request shape: - -```bash -aws lambda-microvms create-microvm-image help \ - >"$EVIDENCE_DIR/create-microvm-image-help.txt" -aws lambda-microvms create-microvm-image --generate-cli-skeleton input \ - >"$EVIDENCE_DIR/create-microvm-image-skeleton.json" -``` - -Confirm that `hooks.microvmHooks.run` is `ENABLED` and -`cpuConfigurations[].architecture` is `ARM_64`, matching SDK 3.1098.0. The CDK -L1 request is a separate CloudFormation surface and legitimately retains its -generated path/string shape. If the installed CLI skeleton differs from the SDK -model or rejects the script request, stop this phase, save the parser/service -error as a model-drift defect, and mark later image/runtime steps blocked. - -```bash -export IMAGE_NAME="${STACK_NAME}-abca-agent" -export BUILD_STARTED_AT="$(date -u +%Y-%m-%dT%H:%M:%SZ)" -cdk/scripts/package-microvm-artifact.sh \ - --stack-name "$STACK_NAME" \ - --create-image \ - --base-image-arn "$BASE_IMAGE_ARN" \ - --base-image-version "$BASE_IMAGE_VERSION" \ - --image-name "$IMAGE_NAME" \ - 2>&1 | tee "$EVIDENCE_DIR/package-and-create-image.txt" -# Explicit status check — `| tee` reports tee's status, so a bare `$?` here lies -# unless `set -o pipefail` is on (see "Important P1 and tooling limits"). -test "${PIPESTATUS[0]}" -eq 0 -``` - -The script requires `aws`, `zip`, `python3`, and `rsync`; reads outputs -`MicrovmArtifactBucketName`, `MicrovmArtifactObjectKey`, -`MicrovmBuildRoleArn`, `MicrovmBuildEgressConnectorArns`, -`MicrovmEgressConnectorArns`, and `MicrovmLogGroupName`; stages root -`Dockerfile`, `agent/`, and `contracts/`; uploads the zip; and calls -`create-microvm-image` with ARM64, 8,192 MiB, `/ready` **and** `/run` enabled on -port 8080, the **build-time** egress connector (443 + 80), and the backend tag. - -**Expected:** upload succeeds, create returns/starts image version `1.0` in this -virgin image name, and the output contains the conspicuous “P1 image is runnable -but NOT smoke-verified” reminder — printed BOTH before and after the create call, -so a failing create cannot swallow it. - -**Record:** artifact size printed by the script, S3 object size from the command -below, create response, and exact banner. - -```bash -export ARTIFACT_BUCKET="$(aws cloudformation describe-stacks --stack-name "$STACK_NAME" \ - --query "Stacks[0].Outputs[?OutputKey=='MicrovmArtifactBucketName'].OutputValue | [0]" --output text)" -export ARTIFACT_KEY="$(aws cloudformation describe-stacks --stack-name "$STACK_NAME" \ - --query "Stacks[0].Outputs[?OutputKey=='MicrovmArtifactObjectKey'].OutputValue | [0]" --output text)" -aws s3api head-object --bucket "$ARTIFACT_BUCKET" --key "$ARTIFACT_KEY" \ - | tee "$EVIDENCE_DIR/artifact-head.json" -``` - -**ADR-021 item:** zip+Dockerfile packaging plane, no secret build inputs, build -role/connector/log wiring, and the `abca:microvm-image-p1-smoke-unverified` -warning (renamed from `…-p1-not-runnable` when F1 was fixed: the image IS -creatable, launchable and payload-deliverable now that `/ready` + `/run` are -declared AND served — what is unverified is smoke parity). - -### 4.3 Poll build and image status to ACTIVE - -**Purpose:** capture real snapshot build duration, state, and component sizes. - -```bash -export IMAGE_VERSION=1 -while :; do - aws lambda-microvms list-microvm-image-builds \ - --image-identifier "$IMAGE_NAME" --image-version "$IMAGE_VERSION" \ - | tee "$EVIDENCE_DIR/image-builds-latest.json" - STATE="$(aws lambda-microvms list-microvm-image-builds \ - --image-identifier "$IMAGE_NAME" --image-version "$IMAGE_VERSION" \ - --query 'items[0].buildState' --output text)" - date -u '+%Y-%m-%dT%H:%M:%SZ buildState='"$STATE" - case "$STATE" in SUCCESSFUL) break;; FAILED) exit 1;; esac - sleep 30 -done - -export BUILD_ID="$(aws lambda-microvms list-microvm-image-builds \ - --image-identifier "$IMAGE_NAME" --image-version "$IMAGE_VERSION" \ - --query 'items[0].buildId' --output text)" -aws lambda-microvms get-microvm-image-build \ - --image-identifier "$IMAGE_NAME" --image-version "$IMAGE_VERSION" \ - --build-id "$BUILD_ID" | tee "$EVIDENCE_DIR/image-build-final.json" - -while :; do - aws lambda-microvms get-microvm-image-version \ - --image-identifier "$IMAGE_NAME" --image-version "$IMAGE_VERSION" \ - | tee "$EVIDENCE_DIR/image-version-latest.json" - STATUS="$(aws lambda-microvms get-microvm-image-version \ - --image-identifier "$IMAGE_NAME" --image-version "$IMAGE_VERSION" \ - --query status --output text)" - test "$STATUS" = ACTIVE && break - sleep 30 -done -export BUILD_FINISHED_AT="$(date -u +%Y-%m-%dT%H:%M:%SZ)" -``` - -**Expected:** build states may include `PENDING` and `IN_PROGRESS`, then -`SUCCESSFUL`; image-version `state` becomes `SUCCESSFUL` and `status` becomes -`ACTIVE`. On failure, save `stateReason` and tail the exact output log group: - -```bash -export MICROVM_LOG_GROUP="$(aws cloudformation describe-stacks --stack-name "$STACK_NAME" \ - --query "Stacks[0].Outputs[?OutputKey=='MicrovmLogGroupName'].OutputValue | [0]" --output text)" -aws logs tail "$MICROVM_LOG_GROUP" --since 2h -``` - -**Record:** start/finish timestamps, duration, all states/reasons, build ID, and -`snapshotBuild.memorySnapshotSizeInBytes`, `codeInstallSizeInBytes`, and -`diskSnapshotSizeInBytes`. Compare **code-install size** to AgentCore's 2 GB -container-image limit, while clearly noting that memory/disk snapshots are not -equivalent to an OCI image and must not be summed into a misleading comparison. -Also record the reported image resources/disk facts; the SDK exposes minimum -memory but no explicit disk-capacity field, so verify the 32 GB disk claim in -the service quota/console and report “not exposed” if that remains true. - -**ADR-021 item:** buildability, final image size, 2 GB-limit narrative, and disk -quota external fact. - -### 4.4 Redeploy against the built image and inspect IAM/env - -**Purpose:** hand the out-of-band image to the orchestrator and prove exact-image -IAM scoping. - -```bash -MISE_EXPERIMENTAL=1 mise //cdk:deploy -- \ - "$STACK_NAME" --require-approval never \ - --context compute_type=lambda-microvm \ - --context microvm_image_identifier="$IMAGE_NAME" \ - --context microvm_image_version="$IMAGE_VERSION" \ - 2>&1 | tee "$EVIDENCE_DIR/image-configured-deploy.txt" -``` - -**Expected:** synth/deploy emits `abca:microvm-image-p1-smoke-unverified` (the -2026-07-31 pass saw the pre-fix id `abca:microvm-image-p1-not-runnable`; the -warning was renamed when F1 was fixed). The orchestrator now has -`MICROVM_IMAGE_IDENTIFIER` — **a full image ARN**, not a bare name (F3) — -`MICROVM_IMAGE_VERSION`, `MICROVM_EXECUTION_ROLE_ARN`, -`MICROVM_EGRESS_CONNECTOR_ARNS`, `MICROVM_PAYLOAD_BUCKET`, and -`MICROVM_INGRESS_CONNECTOR_ARNS` carrying the Lambda-managed `NO_INGRESS` -connector (F7 — the pre-fix build had no ingress variable at all). - -```bash -aws lambda get-function-configuration --function-name "$ORCHESTRATOR_FN" \ - --query 'Environment.Variables' --output json \ - | python3 -c 'import json,sys; d=json.load(sys.stdin); print({k:d[k] for k in d if k.startswith("MICROVM_")})' \ - | tee "$EVIDENCE_DIR/orchestrator-microvm-env.txt" - -export ORCHESTRATOR_ROLE_ARN="$(aws lambda get-function-configuration \ - --function-name "$ORCHESTRATOR_FN" --query Role --output text)" -export ORCHESTRATOR_ROLE="${ORCHESTRATOR_ROLE_ARN##*/}" -aws iam list-role-policies --role-name "$ORCHESTRATOR_ROLE" \ - | tee "$EVIDENCE_DIR/orchestrator-inline-policy-names.json" -export ORCH_POLICY_NAME="$(aws iam list-role-policies --role-name "$ORCHESTRATOR_ROLE" \ - --query 'PolicyNames[0]' --output text)" -aws iam get-role-policy --role-name "$ORCHESTRATOR_ROLE" \ - --policy-name "$ORCH_POLICY_NAME" \ - | tee "$EVIDENCE_DIR/orchestrator-inline-policy.json" -``` - -The inline policy's physical name is CDK-generated and therefore cannot be -hard-coded; the actual name to fetch is `$ORCH_POLICY_NAME` returned by -`list-role-policies` (normally the role's `DefaultPolicy`). If more than one is -listed, fetch each and select the document containing `Sid=MicrovmLifecycle`. - -**Expected:** `MicrovmLifecycle` grants exactly `lambda:RunMicrovm`, -`lambda:GetMicrovm`, and `lambda:TerminateMicrovm` against exactly -`arn:...:microvm-image:$IMAGE_NAME` and its `:` suffix sibling; -`MicrovmPassNetworkConnector` has `lambda:PassNetworkConnector` on `*`; no -`SuspendMicrovm`, `ResumeMicrovm`, or `CreateMicrovmAuthToken` exists. - -**Record:** warning, environment-key/value map, role/policy names, statements, -and exact image ARN format observed. - -**ADR-021 item:** the `abca:microvm-image-p1-smoke-unverified` warning (formerly -`…-p1-not-runnable`), exact-ARN lifecycle IAM, no-JWE grant, and all-or-nothing -environment wiring — which now includes `MICROVM_INGRESS_CONNECTOR_ARNS`, always -injected, carrying the `NO_INGRESS` control. - ---- - -## Phase 5 — Manual lifecycle and empirical checklist - -### 5.0 Resolve launch inputs and IAM identity mode - -**Purpose:** use deployed values and distinguish true role-policy evidence from -admin-only lifecycle evidence. - -```bash -export EXECUTION_ROLE_ARN="$(aws cloudformation describe-stacks --stack-name "$STACK_NAME" \ - --query "Stacks[0].Outputs[?OutputKey=='MicrovmExecutionRoleArn'].OutputValue | [0]" --output text)" -export EGRESS_CONNECTORS="$(aws cloudformation describe-stacks --stack-name "$STACK_NAME" \ - --query "Stacks[0].Outputs[?OutputKey=='MicrovmEgressConnectorArns'].OutputValue | [0]" --output text)" - -aws sts assume-role --role-arn "$ORCHESTRATOR_ROLE_ARN" \ - --role-session-name abca-645-verification \ - >"$EVIDENCE_DIR/orchestrator-assume-role.json" \ - 2>"$EVIDENCE_DIR/orchestrator-assume-role-error.txt" || true -``` - -Lambda execution-role trust normally allows only `lambda.amazonaws.com`, so -operator assumption is expected to fail unless the sandbox has an explicit -test trust path. **Do not modify production-like trust just for this run.** If -assumption succeeds, open a subshell with those temporary credentials for steps -5.1, 5.2, 5.7, and 5.8(a/b), and mark evidence `ORCHESTRATOR_ROLE`. Otherwise -run as sandbox admin and mark scoping observations `ADMIN — advisory`; the -static inline-policy inspection in 4.4 remains authoritative. - -Example temporary-credential subshell setup: - -```bash -read AWS_ACCESS_KEY_ID AWS_SECRET_ACCESS_KEY AWS_SESSION_TOKEN < **Stale premise.** This step was written against the pre-fix world, where the -> plan was to declare `/run` in P1 and serve it in P2. F1 proved that is not a -> reachable service state, and the agent now serves `/ready` + `/run`. Two -> consequences for a re-run: (i) the image is no longer hook-less, so -> `--run-hook-payload` is ACCEPTED rather than rejected — the interesting -> observation becomes whether the payload reaches the pipeline, not what a -> hook-less VM does; and (ii) `--image-identifier` needs the image **ARN**, not -> `$IMAGE_NAME` (F3). What still holds unchanged, and is worth re-confirming, is -> everything below about the state enum, the 28,800-second bound, the -> omit-`idlePolicy` invariant, and the default public ingress. - -```bash -export RUN_STARTED_EPOCH="$(date +%s)" -aws lambda-microvms run-microvm \ - --image-identifier "$IMAGE_NAME" \ - --image-version "$IMAGE_VERSION" \ - --execution-role-arn "$EXECUTION_ROLE_ARN" \ - --egress-network-connectors "$EGRESS_CONNECTORS" \ - --run-hook-payload '{"verification":"issue-645-p1"}' \ - --maximum-duration-in-seconds 28800 \ - | tee "$EVIDENCE_DIR/run-microvm.json" -export MICROVM_ID="$(python3 -c 'import json,os; print(json.load(open(os.environ["EVIDENCE_DIR"] + "/run-microvm.json"))["microvmId"])')" -export MICROVM_ENDPOINT="$(python3 -c 'import json,os; print(json.load(open(os.environ["EVIDENCE_DIR"] + "/run-microvm.json"))["endpoint"])')" -``` - -Do **not** pass `--idle-policy`. Immediately poll: - -```bash -for delay in 0 2 5 10 20 30 60; do - sleep "$delay" - date -u '+%Y-%m-%dT%H:%M:%SZ' - aws lambda-microvms get-microvm --microvm-identifier "$MICROVM_ID" || true -done | tee "$EVIDENCE_DIR/hookless-state-timeline.txt" -``` - -**Expected:** `run-microvm` should return a `microvmId`, endpoint, image ARN, -version, and initial state if `/run` is asynchronous at control-plane return. -The eventual state is deliberately **not prescribed**: record whether it reaches -`RUNNING`, remains `PENDING`, becomes `TERMINATING/TERMINATED`, or disappears, -plus `stateReason` and elapsed time. If `run-microvm` itself rejects because the -hook returns 404/times out, record that exact exception and timing. This result -feeds the P2 hook/startup design. - -**Record:** full response, endpoint (not an auth token), all states/reasons, -time-to-first-state/time-to-terminal, and relevant log lines. - -**ADR-021 item:** real state enum, 28,800-second bound, and omit-`idlePolicy` -invariant. (The original "P1-not-runnable premise" this step was written to probe -no longer exists — see the Phase 5 row of the stale-instruction table.) - -### 5.2 Explicit `get-microvm` state mapping - -**Purpose:** validate the six SDK states used by strategy mapping. - -```bash -aws lambda-microvms get-microvm --microvm-identifier "$MICROVM_ID" \ - | tee "$EVIDENCE_DIR/get-microvm.json" -``` - -**Expected:** observed values come from `PENDING`, `RUNNING`, `SUSPENDING`, -`SUSPENDED`, `TERMINATING`, `TERMINATED`; a reaped ID returns -`ResourceNotFoundException`. - -**Record:** every distinct state actually observed and any unknown state. - -**ADR-021 item:** mechanical state mapping and future-enum safety premise. - -### 5.3 Manual suspend without `idlePolicy` - -**Purpose:** determine whether explicit suspend works independently of traffic -idle policy and whether a hook-less image survives long enough to suspend. - -Use the sandbox admin identity because P1's orchestrator correctly lacks this -permission: - -```bash -date -u '+%Y-%m-%dT%H:%M:%SZ suspend-request' -aws lambda-microvms suspend-microvm --microvm-identifier "$MICROVM_ID" \ - | tee "$EVIDENCE_DIR/suspend-microvm.json" -for i in 1 2 3 4 5 6; do - aws lambda-microvms get-microvm --microvm-identifier "$MICROVM_ID" || true - sleep 10 -done | tee "$EVIDENCE_DIR/suspend-state-timeline.txt" -``` - -**Expected:** if the VM reached a suspendible state, observe -`SUSPENDING → SUSPENDED`. A conflict/not-found caused by the failed `/run` hook -is a valid P1 result but means 5.4–5.6 cannot discharge TTL/resume empirically; -mark those **BLOCKED-BY-P1-HOOKLESS-IMAGE**, do not invent an answer. - -**Record:** identity, response/exception, states, and suspend latency. - -**ADR-021 item:** manual suspend without idle policy. - -### 5.4 Suspended TTL experiment - -**Purpose:** determine the default lifetime of a manually suspended VM when -`idlePolicy` (and therefore `suspendedDurationSeconds`) is omitted. - -Only run after observing `SUSPENDED`: - -```bash -for seconds in 0 900 3600 14400; do - sleep "$seconds" - printf '\ncheckpoint_after_sleep_seconds=%s at %s\n' "$seconds" "$(date -u +%Y-%m-%dT%H:%M:%SZ)" - aws lambda-microvms get-microvm --microvm-identifier "$MICROVM_ID" || true -done | tee "$EVIDENCE_DIR/suspended-ttl-checkpoints.txt" -``` - -The sleeps are incremental (0, then 15 min, then +1 h, then +4 h). A time-boxed -executor may truncate after 15 min or 1 h, but must say so. Never leave the VM -past the 28,800-second maximum; Phase 8 terminates it. - -**Expected:** unknown by design. Record whether it stays `SUSPENDED`, terminates, -or becomes NotFound and at what wall-clock age. `maximumDurationInSeconds=28800` -bounds the worst case even if no separate suspended TTL exists. - -**Record:** complete checkpoints, truncation, start age, and terminal time. - -**ADR-021 item:** P1 external fact — manual-suspend default TTL without -`idlePolicy`. - -### 5.5 Quota treatment while suspended - -**Purpose:** test the rationale that a suspended VM still holds account memory -quota. - -```bash -aws service-quotas list-services \ - --query "Services[?contains(ServiceName, 'Lambda')].[ServiceCode,ServiceName]" \ - --output table | tee "$EVIDENCE_DIR/lambda-service-codes.txt" -aws service-quotas list-service-quotas --service-code lambda \ - --query "Quotas[?contains(QuotaName, 'MicroVM') || contains(QuotaName, 'microVM') || contains(QuotaName, 'memory')]" \ - | tee "$EVIDENCE_DIR/microvm-service-quotas.json" -aws lambda-microvms list-microvms --image-identifier "$IMAGE_NAME" \ - | tee "$EVIDENCE_DIR/microvms-while-suspended.json" -``` - -Also open the Lambda MicroVM **account quota / memory utilization view** in the -AWS console before launch, while RUNNING, while SUSPENDED, and after termination; -capture values/timestamps. The SDK 3.1098 model has no “get account memory -usage” operation, and `service-quotas` normally reports limits rather than live -consumption. If the console has no utilization view and quota is too high to -safely saturate with a second 32 GiB VM, record **NOT OBSERVABLE SAFELY** rather -than launching VMs until failure. - -**Expected:** the suspended VM remains in `list-microvms`; the load-bearing -claim is discharged only if the account quota view continues counting its -32,768 MiB after `SUSPENDED`. - -**Record:** quota names/codes/limits, list response, and four utilization -snapshots. Distinguish “listed” from “proven to consume quota.” - -**ADR-021 item:** account-memory quota treatment of suspended VMs / concurrency -slot-held rationale. - -### 5.6 Resume and state transition - -**Purpose:** verify manual resume and preserved lifecycle identity. - -```bash -aws lambda-microvms resume-microvm --microvm-identifier "$MICROVM_ID" \ - | tee "$EVIDENCE_DIR/resume-microvm.json" -for i in 1 2 3 4 5 6; do - aws lambda-microvms get-microvm --microvm-identifier "$MICROVM_ID" || true - sleep 10 -done | tee "$EVIDENCE_DIR/resume-state-timeline.txt" -``` - -**Expected:** unknown for P1 because no `/resume` hook is declared and `/run` -may already have failed. Record whether resume succeeds and transitions to -`RUNNING`, conflicts, terminates, or disappears, and whether ID/endpoint remain -stable. - -**Record:** response, transitions, latency, ID/endpoint stability. - -**ADR-021 item:** empirical resume lifecycle input for P3 design. - -### 5.7 Terminate and observe reaping - -**Purpose:** validate active cleanup and the `NotFound → completed` strategy -mapping premise. - -```bash -date -u '+%Y-%m-%dT%H:%M:%SZ terminate-request' -aws lambda-microvms terminate-microvm --microvm-identifier "$MICROVM_ID" \ - | tee "$EVIDENCE_DIR/terminate-microvm.json" -for delay in 0 2 5 10 20 30 60 120; do - sleep "$delay" - date -u '+%Y-%m-%dT%H:%M:%SZ' - aws lambda-microvms get-microvm --microvm-identifier "$MICROVM_ID" || true -done | tee "$EVIDENCE_DIR/terminate-reap-timeline.txt" -``` - -**Expected:** `TERMINATING`/`TERMINATED` may be visible, followed by -`ResourceNotFoundException`. Exact reaping time is empirical. - -**Record:** transition and seconds from terminate request to NotFound, including -exception name/message/status code. - -**ADR-021 item:** explicit terminate path and `ResourceNotFoundException → -completed` mapping. - -### 5.8 IAM negative tests (only under assumed orchestrator role) - -**Purpose:** prove the deployed role cannot escape its exact image or mint JWE -tokens. - -Run only if step 5.0 successfully assumed the orchestrator role. Otherwise mark -**SKIPPED — LAMBDA ROLE TRUST DOES NOT ALLOW OPERATOR ASSUMPTION** and rely on -4.4's policy document. - -```bash -aws lambda-microvms run-microvm \ - --image-identifier "${IMAGE_NAME}-different" \ - --execution-role-arn "$EXECUTION_ROLE_ARN" \ - --egress-network-connectors "$EGRESS_CONNECTORS" \ - --maximum-duration-in-seconds 60 \ - 2>&1 | tee "$EVIDENCE_DIR/iam-negative-different-image.txt" - -aws lambda-microvms create-microvm-auth-token \ - --microvm-identifier "$MICROVM_ID" \ - --expiration-in-minutes 5 \ - --allowed-ports '[{"port":8080}]' \ - 2>&1 | tee "$EVIDENCE_DIR/iam-negative-auth-token.txt" -``` - -**Expected:** both return `AccessDeniedException`. If the different-name test -returns `ResourceNotFoundException`, it does not prove exact-ARN denial; create -or use a real second sandbox image only if already available, otherwise mark -that sub-check inconclusive. The `allowed-ports` CLI union syntax is -best-effort; the verified SDK request is -`{microvmIdentifier, expirationInMinutes, allowedPorts:[{port:8080}]}`. A local -argument-parser error is not an IAM result. - -**Record:** assumed caller ARN and full errors. - -**ADR-021 item:** exact-image ARN scoping and no -`CreateMicrovmAuthToken`/no-JWE posture. - -### 5.9 Direct service boundary: 16,384 vs 16,385 bytes - -**Purpose:** independently verify the documented `runHookPayload` service cap. -This is direct-service validation; the ABCA strategy already routes envelopes -larger than 16,384 bytes to S3. - -```bash -python3 - <<'PY' -from pathlib import Path -Path('/tmp/abca-payload-16384.txt').write_bytes(b'x' * 16384) -Path('/tmp/abca-payload-16385.txt').write_bytes(b'x' * 16385) -PY -wc -c /tmp/abca-payload-16384.txt /tmp/abca-payload-16385.txt - -aws lambda-microvms run-microvm \ - --image-identifier "$IMAGE_NAME" --image-version "$IMAGE_VERSION" \ - --execution-role-arn "$EXECUTION_ROLE_ARN" \ - --egress-network-connectors "$EGRESS_CONNECTORS" \ - --run-hook-payload file:///tmp/abca-payload-16384.txt \ - --maximum-duration-in-seconds 60 \ - | tee "$EVIDENCE_DIR/run-payload-16384.json" - -aws lambda-microvms run-microvm \ - --image-identifier "$IMAGE_NAME" --image-version "$IMAGE_VERSION" \ - --execution-role-arn "$EXECUTION_ROLE_ARN" \ - --egress-network-connectors "$EGRESS_CONNECTORS" \ - --run-hook-payload file:///tmp/abca-payload-16385.txt \ - --maximum-duration-in-seconds 60 \ - 2>&1 | tee "$EVIDENCE_DIR/run-payload-16385.txt" -``` - -Immediately terminate the MicroVM returned by the accepted request (if any). - -**Expected:** 16,384 bytes is accepted; 16,385 bytes is rejected, likely with -`ValidationException`. Record the actual exception rather than treating the -predicted name as normative. Confirm the installed CLI expands `file://` to file -contents; if it passes the literal URI, repeat with command substitution and -record the client behavior. - -**Record:** byte counts, both complete responses, exception name/message/status, -and cleanup ID. - -**ADR-021 item:** exact 16 KB `runHookPayload` boundary. - ---- - -## Phase 6 — CLI behavior - -### 6.1 Build and point the CLI at the stack - -**Purpose:** use the repository CLI's real resolution rules: operator commands -take `--region`/`--stack-name`; configured Region is a fallback; API commands -also need Cognito configuration/login. - -```bash -MISE_EXPERIMENTAL=1 mise //cli:build -bgagent() { node cli/lib/bin/bgagent.js "$@"; } -bgagent configure --stack-name "$STACK_NAME" --region "$AWS_REGION" -bgagent platform outputs --stack-name "$STACK_NAME" --region "$AWS_REGION" -``` - -**Expected:** config is written under `${BGAGENT_CONFIG_DIR:-$HOME/.bgagent}`; -stack outputs resolve. No Cognito login is needed for operator AWS commands. - -**Record:** CLI build result and redacted outputs. - -**ADR-021 item:** deploy/CLI substrate discovery contract. - -### 6.2 Onboard, probe, inspect, and clean up a dummy row - -**Purpose:** exercise the live managed-image probe, doctor check, and runtime -grouping without submitting a task. - -```bash -export DUMMY_REPO=verification-only/issue-645 -bgagent repo onboard "$DUMMY_REPO" \ - --compute-type lambda-microvm \ - --stack-name "$STACK_NAME" --region "$AWS_REGION" \ - --output json | tee "$EVIDENCE_DIR/cli-onboard.json" - -bgagent platform doctor --stack-name "$STACK_NAME" --region "$AWS_REGION" \ - --output json | tee "$EVIDENCE_DIR/cli-doctor.json" || true -bgagent runtime status --stack-name "$STACK_NAME" --region "$AWS_REGION" \ - --output json | tee "$EVIDENCE_DIR/cli-runtime-status.json" - -bgagent repo offboard "$DUMMY_REPO" \ - --stack-name "$STACK_NAME" --region "$AWS_REGION" \ - --output json | tee "$EVIDENCE_DIR/cli-offboard.json" -``` - -**Expected:** onboarding's `ListManagedMicrovmImages` probe passes and the row -has `compute_type=lambda-microvm`; doctor contains -`lambda_microvm_availability`; runtime status groups it under -`lambda_microvm_substrates`; offboard marks it removed. Doctor may still exit -non-zero in a virgin sandbox because its GitHub secret is an unpopulated -placeholder or no real repo/token/model access exists—record those independent -failures. - -**SKIP-IF-UNCONFIGURED:** if the sandbox principal lacks DDB or MicroVM catalog -read permission, record the IAM gap and skip the write. Onboarding itself does -not require a valid GitHub token, but a meaningful doctor pass and any task do. - -**Record:** probe result, row, doctor checks, grouping, and cleanup row status. - -**ADR-021 item:** onboarding region probe and doctor availability check. - ---- - -## Phase 7 — OPTIONAL negative task path - -### 7.1 Submit only in a fully configured sandbox - -**Purpose:** observe orchestrator classification and terminate-on-finalize, not -to claim P2 smoke parity. - -This is **OPTIONAL / usually DEFERRED-TO-P2-ENV**. A virgin account is missing a -Cognito user/login, populated GitHub token secret, and a genuinely accessible -onboarded repository unless the executor configures all three. Do not create -those merely for this P1 substrate pass. - -If they already exist: - -```bash -bgagent login -bgagent repo onboard --compute-type lambda-microvm \ - --stack-name "$STACK_NAME" --region "$AWS_REGION" -bgagent submit --repo --task "P1 negative: observe hook-less MicroVM failure" -# Use the returned task ID: -bgagent watch -``` - -**Expected:** the backend does not complete agent work. Capture the task's -failure classification/remedy, persisted `compute_metadata` (`microvmId` and -endpoint), orchestrator logs, and whether finalization calls -`TerminateMicrovm`. If no complete setup exists, write -`DEFERRED-TO-P2-ENV — missing Cognito user/login, GitHub token, and/or real repo -onboarding`. - -**Record:** task ID/evidence or exact deferral reason. - -**ADR-021 item:** defense-in-depth failure classification, handle persistence, -and terminate fire; this is not P2 clone→change→PR smoke parity. - ---- - -## Phase 8 — Teardown - -### 8.1 Terminate every MicroVM - -**Purpose:** stop compute billing before image/stack deletion. - -```bash -aws lambda-microvms list-microvms --image-identifier "$IMAGE_NAME" \ - | tee "$EVIDENCE_DIR/microvms-before-teardown.json" -for id in $(aws lambda-microvms list-microvms --image-identifier "$IMAGE_NAME" \ - --query 'items[].microvmId' --output text); do - aws lambda-microvms terminate-microvm --microvm-identifier "$id" || true -done -sleep 30 -aws lambda-microvms list-microvms --image-identifier "$IMAGE_NAME" \ - | tee "$EVIDENCE_DIR/microvms-after-terminate.json" -``` - -**Expected:** no nonterminal VM remains; wait/retry if necessary. - -**Record:** IDs and final states. - -**ADR-021 item:** explicit cleanup rather than relying on the eight-hour bound. - -### 8.2 Delete out-of-band image versions and image - -**Purpose:** stop snapshot storage charges. The verified operation/CLI command -name is `delete-microvm-image-version` with `imageIdentifier` and -`imageVersion`. - -```bash -aws lambda-microvms list-microvm-image-versions --image-identifier "$IMAGE_NAME" \ - | tee "$EVIDENCE_DIR/image-versions-before-delete.json" -for version in $(aws lambda-microvms list-microvm-image-versions \ - --image-identifier "$IMAGE_NAME" --query 'items[].imageVersion' --output text); do - aws lambda-microvms delete-microvm-image-version \ - --image-identifier "$IMAGE_NAME" --image-version "$version" -done -aws lambda-microvms delete-microvm-image --image-identifier "$IMAGE_NAME" -``` - -**Expected:** all versions enter deletion and the image is deleted. If the -service requires deleting the parent first/last or waiting between operations, -follow the returned conflict remedy and record actual order. - -**Record:** versions, responses, final NotFound/list absence. - -**ADR-021 item:** versioned image lifecycle cleanup. - -### 8.3 Destroy ABCA; retain bootstrap - -**Purpose:** remove the platform and recurring network costs while preserving -the account's reusable CDK bootstrap. - -```bash -MISE_EXPERIMENTAL=1 mise //cdk:destroy -- \ - "$STACK_NAME" --force \ - --context compute_type=lambda-microvm \ - --context microvm_image_identifier="$IMAGE_NAME" \ - --context microvm_image_version="$IMAGE_VERSION" -aws cloudformation wait stack-delete-complete --stack-name "$STACK_NAME" -aws cloudformation describe-stacks --stack-name CDKToolkit \ - --query 'Stacks[0].[StackStatus,Parameters]' \ - | tee "$EVIDENCE_DIR/bootstrap-left-in-place.json" -``` - -**Raw fallback if `mise` is absent** (run from `cdk/`): - -```bash -npx cdk destroy "$STACK_NAME" --force \ - --context compute_type=lambda-microvm \ - --context microvm_image_identifier="$IMAGE_NAME" \ - --context microvm_image_version="$IMAGE_VERSION" -``` - -**Expected:** `backgroundagent-dev` is absent; `CDKToolkit` remains -`CREATE_COMPLETE`/`UPDATE_COMPLETE` with the custom policies. VPC teardown can -lag while service-managed ENIs are reclaimed; wait and retry rather than -force-deleting resources past CloudFormation. - -**Record:** destroy duration/events, leftovers, and final bootstrap status. - -**ADR-021 item:** clean substrate/resource lifecycle. - -**Cost note:** this pass can accrue MicroVM running minutes, snapshot/image -storage, S3 artifact storage, NAT gateway hourly/data charges, VPC endpoint -hourly charges, CloudWatch logs, and brief Lambda/DynamoDB/API usage. Suspended -VMs should stop compute charges but may retain billed snapshot storage and -account memory quota; the experiment determines the latter. NAT gateways and -endpoints continue charging until stack deletion. - ---- - -## Live execution results — 2026-07-31, account , us-east-1 - -Executed against real AWS. Evidence directory: -`/tmp/abca-645-p1-20260731T184822Z`. Wall clock 18:48Z → 23:07Z (4 h 19 min). - -**Execution deviations from the runbook as written** (each is itself a result): - -1. `mise` is **not installed** on the executor; every **raw fallback** was used. -2. The account is **not virgin overall** — `serverless-api-powertools`, - `BuildingServerlessAPIs`, `aws-sam-cli-managed-default`, and `CDKToolkit` - pre-existed. `backgroundagent-dev` was absent, so the ABCA-specific - first-deploy premise held. -3. `docker` is absent; `finch` 1.x (`CDK_DOCKER=finch`) built the AgentCore - container asset. -4. Phase 2 could not deploy from unmodified sources. The stack was deployed from - a **hand-patched cloud assembly** (`/tmp/cdkout-p1*`, a build artifact — no - repository source file was modified). Two patches, both forced by live-service - rejections recorded below: a MicroVM connector **operator role**, and moving - subnets off `us-east-1a`. -5. A temporary **port-80 egress rule** (`sgr-07ed1fa48ef38467a`) was added to the - construct's security group to get any image to build at all (see 4.3). -6. Suspend-TTL observation was **truncated at ~1 h** (runbook allows this). - -### Phase 0 - -**0.1** — Account ``, caller -`arn:aws:sts:::assumed-role/AdminConsoleAccess/aamorosi-Isengard` -(administrator). Branch `feat/645-lambda-microvm-p1`, SHA -`0505f914fd7093cccd067b6346a24e1c40e50643`. Untracked: `docs/verification/`, -`opencode.json`. Stack absence returned exactly the expected error: - -``` -An error occurred (ValidationError) when calling the DescribeStacks operation: Stack with id backgroundagent-dev does not exist -``` - -**0.2** — `aws-cli/2.36.13 Python/3.14.6 Darwin/25.5.0 source/arm64`; -`node v24.16.0`; `cdk 2.1129.0`; `mise NOT INSTALLED`; `Python 3.9.6`; -`Zip 3.0`; **`openrsync` (protocol 29, "rsync 2.6.9 compatible")** — the -packaging script's `rsync -a --exclude` usage worked unmodified on macOS. -`@aws-sdk/client-lambda-microvms` = **3.1098.0**. - -`aws lambda-microvms help` **succeeded (exit 0)** and lists **24** commands -including all seven the runbook requires. The CLI command list is an **exact -match** to the SDK 3.1098.0 command list (24 vs 24), including -`create-microvm-shell-auth-token`. **No CLI/SDK operation-name drift.** - -`create-microvm-image --generate-cli-skeleton input` **confirms the packaging -script's shape and refutes the CDK L1 shape**: - -- `cpuConfigurations[].architecture` — help documents exactly one allowed value: - `ARM_64`. The L1's `'arm64'` is not a documented value. -- `hooks` — `{"port": integer, "microvmHooks": {"run": "DISABLED"|"ENABLED", - "runTimeoutInSeconds": integer, ...}}`. There is **no hook-path field at all**; - the L1's `run: '/run'` path string has no counterpart in the service model. -- `run-microvm` skeleton confirms `idlePolicy` - `{maxIdleDurationSeconds, suspendedDurationSeconds, autoResumeEnabled}` and - `runHookPayload` as a plain string. - -The CFN-vs-API question is therefore **half-adjudicated**: the API side is -settled, but the `AWS::Lambda::MicrovmImage` CFN path was never exercised, -because the construct only synthesizes it when `microvm_base_image_arn` + -`microvm_base_image_version` context is supplied, and the runbook's Phase 4 uses -the out-of-band script path. **The CFN value shapes remain untested** — and 4.2 -below shows the request would be rejected on hook semantics regardless of shape. - -### Phase 1 - -**1.1** — `CDKToolkit` `CREATE_COMPLETE`, created 2025-11-24, `BootstrapVariant` -= `AWS CDK: Default Resources`. Markers: `ComputeTypes` **ABSENT**, -`IaCRoleABCAComputeLambdaMicrovms` **ABSENT**, `AdministratorAccess` -**present** — exactly the standard-bootstrap starting state the step predicts. - -**1.2 — DEFECT (runbook + `mise //cdk:bootstrap`): the re-bootstrap is a silent -no-op on an already-bootstrapped account.** Verbatim: - -``` -Bootstrap stack already exists, containing 'AWS CDK: Default Resources'. Not overwriting it with a template containing 'ABCA: Least-Privilege Bootstrap' (use --force if you intend to overwrite) -✅ Environment aws:///us-east-1 bootstrapped (no changes). -``` - -Exit status **0**. A pass that trusts this would proceed to 1.3 with -`AdministratorAccess` still attached. `--force` was required. A **durable** -consequence: after the forced bootstrap, `BootstrapVariant` **remains** -`AWS CDK: Default Resources` (the CDK CLI re-sends the previous value rather than -the template default `ABCA: Least-Privilege Bootstrap`), so **every future -non-forced `mise //cdk:bootstrap` will refuse again**. - -**1.2 — DEFECT (runbook): the `ComputeTypes` parameter command as written is -rejected.** Verbatim: - -``` -An error occurred (ParamValidation): Parameter validation failed: -Invalid type for parameter Parameters[0].ParameterValue, value: ['agentcore', 'lambda-microvm'], type: , valid types: -``` - -The CLI shorthand parser splits on the comma. The escaped form -`ParameterValue=agentcore\,lambda-microvm` works — which is exactly what the -comment above `[tasks.bootstrap]` in `cdk/mise.toml` already shows; the runbook -dropped the escapes. Final state: `UPDATE_COMPLETE`, `ComputeTypes` = -`agentcore,lambda-microvm`. - -**1.3 — PASS.** `CFN_EXEC_ROLE` = -`cdk-hnb659fds-cfn-exec-role--us-east-1`. Exactly one local policy -matched: `cdk-hnb659fds-IaCRole-ABCA-Compute-LambdaMicrovms--us-east-1`, -and it is attached. Attachments are the five ABCA policies (Application, -Infrastructure, Observability, Compute-Agentcore, Compute-LambdaMicrovms) and -**no `AdministratorAccess`** — the template's replacement works. The -`LambdaMicrovms` statement grants 19 actions, all image/version/build/ -managed-catalog/network-connector, including `lambda:PassNetworkConnector`. -The optional scratch-qualifier negative was deliberately skipped as the runbook -directs. - -### Phase 2 - -**2.1 — synth PASS, deploy BLOCKED THREE TIMES.** Synth emitted -`abca:microvm-image-not-provisioned` and **not** -`abca:microvm-image-p1-not-runnable`, exactly as specified. Incidental synth -warnings worth noting: `Template size is approaching limit: 893273/1000000` and -`Number of resources: 463 is approaching allowed maximum of 500`. - -*Blocker A (environmental, not an ABCA defect).* The stack carries one Docker -image asset (`agent/Dockerfile`) and no container builder was installed. With -`finch`, the `gh-builder` stage failed twice on upstream flakiness: - -``` -pkg/mod/github.com/cli/go-gh/v2@v2.13.0/internal/yamlmap/yaml_map.go:8:2: unrecognized import path "gopkg.in/yaml.v3": reading https://gopkg.in/yaml.v3?go-get=1: 502 Proxy Error -``` - -`gopkg.in` alternated 200/502 from the host too. A direct `finch build` then -succeeded. **Data point for 4.3's size narrative:** the AgentCore container -image is **1.799 GB uncompressed / 629.7 MB compressed**. - -*Blocker B — **the P1 substrate cannot deploy from unmodified sources**.* -`AWS::Lambda::NetworkConnector` `CREATE_FAILED`, verbatim: - -``` -Resource handler returned message: "NetworkConnectorOperatorRole is required for VPC_EGRESS connector type (Service: Lambda, Status Code: 400, Request ID: 04726267-6c61-4ff5-bb1d-302122e9f955) (SDK Attempt Count: 1)" (RequestToken: 1a1652c3-2166-e865-c49d-6cdb5927bbfe, HandlerErrorCode: InvalidRequest) -``` - -This **directly refutes an explicit design assumption** stated in -`cdk/src/constructs/lambda-microvm-compute.ts` (~line 467): - -> `operatorRole` is left unset so Lambda manages the ENIs with its own -> service-linked role rather than a role we would have to trust. - -The generated L1 also marks `operatorRole` optional -(`readonly operatorRole?: string`) with no note that `VPC_EGRESS` requires it. -An independent probe stack (`abca645-connector-probe`) confirmed the minimal -working recipe: a role trusting `lambda.amazonaws.com` with -`AWSLambdaVPCAccessExecutionRole` plus `ec2:CreateNetworkInterface` / -`DeleteNetworkInterface` / `DescribeNetworkInterfaces` / `DescribeSubnets` / -`DescribeVpcs` / `DescribeSecurityGroups` / `CreateTags` / -`AssignPrivateIpAddresses` / `UnassignPrivateIpAddresses` / -`Describe|ModifyNetworkInterfaceAttribute` → connector `CREATE_COMPLETE`. - -*Blocker C (AgentCore, account-AZ-specific, blocks any ABCA deploy here).* - -``` -Resource handler returned message: "Agent runtime creation failed with status: CREATE_FAILED for runtime: backgroundagentdevRuntimeCC6E3A5A-yiKm9OEVPo. Reason: The following subnets are in unsupported availability zones in region us-east-1: subnet-02b0221802f3fee10 in us-east-1a (ID: use1-az6). Supported availability zones are: use1-az4, use1-az1, use1-az2" -``` - -This account maps `us-east-1a` → `use1-az6`. `AgentVpc` does not constrain AZ -selection, so CDK's default two-AZ pick lands on an AZ AgentCore rejects. -Patched `us-east-1a` → `us-east-1c` (`use1-az2`). - -*Two further teardown/iteration gotchas.* (i) Rollback itself failed once: -`Validation failed during DeleteMemory: Memory is in transitional state -CREATING. Cannot delete memory.` — `AWS::BedrockAgentCore::Memory` cannot be -deleted while creating, leaving `ROLLBACK_FAILED`; a plain `delete-stack` -cleared it. (ii) Post-synth template edits are **silently ignored** if the -template's S3 asset object already exists: the object key is the pre-edit content -hash recorded in `*.assets.json`, so `cdk-assets` skips the upload and CFN -re-uses the stale template. The stale object must be deleted. - -Successful deploy: `CREATE_COMPLETE`, 19:41:53Z → 19:55:35Z = **13 min 42 s**, -464 resources. - -**2.2 — PASS.** `ComputeSubstrate=lambda-microvm`; all six `Microvm…` outputs -populated; artifact key exactly `microvm-images/agent-artifact.zip`. The 13 -`LambdaMicrovmCompute` resources are: artifact + payload buckets (each with a -bucket policy and an auto-delete custom resource), build role + policy, execution -role + policy, `AWS::EC2::SecurityGroup sg-0e662dc0d6f6e9ade`, -`AWS::Logs::LogGroup /aws/lambda-microvms/backgroundagent-dev-abca-agent`, and -`AWS::Lambda::NetworkConnector nc-132ede11-cb63-4dfa-b75b-6a4713023c1a`. -**No `AWS::Lambda::MicrovmImage`** — correct for this state. The security group -has exactly one rule: egress `tcp/443 → 0.0.0.0/0`, *"Allow HTTPS egress (GitHub -API, AWS services)"*. (That single rule is what breaks the image build — 4.3.) - -**2.3 — PASS.** `MICROVM_*` keys = `[]` (14 env keys total). -**DEFECT (runbook): the `ORCHESTRATOR_FN` command is broken by pagination.** -With 464 resources, `list-stack-resources --query "…|[0]"` applies the query -**per page** and printed five lines (`None None None None `), which then -failed `get-function-configuration` on the multi-line value. Needs -`--no-paginate` (or local parsing). Resolved value: -`backgroundagent-dev-TaskOrchestratorOrchestratorFn-gM2sydgNVf1V`. - -**2.4 — PASS.** All **six** taggable construct resources carry -`abca:compute-backend=lambda-microvm`: security group, network connector, log -group, both buckets, and — verified via `iam list-role-tags` — both roles. -`resourcegroupstaggingapi` returned only **5** ARNs; **IAM roles are simply not -returned by that API**, which is an API coverage gap, not a missing tag. -Bucket policies and auto-delete custom resources are not independently taggable, -as the step anticipates. - -### Phase 3 - -**3.1 — PASS** (exit 1). Verbatim: - -``` -Error: AWS Lambda MicroVMs are not available in eu-central-1. The lambda-microvm compute backend is enabled (--context compute_type=lambda-microvm) but the stack Region is not one of: us-east-1, us-east-2, us-west-2, eu-west-1, ap-northeast-1. Either deploy the stack into a supported Region, drop the backend (--context compute_type=agentcore or ecs), or — if AWS has since launched Lambda MicroVMs in eu-central-1 — bypass this static check with --context microvm_region_override=true and add eu-central-1 to LAMBDA_MICROVM_SUPPORTED_REGIONS in cdk/src/handlers/shared/microvm-regions.ts. -``` - -Names the Region, all five supported Regions, and the override flag. - -**3.2 — PASS** (exit 0) with `abca:microvm-region-override` (and, correctly, the -`abca:microvm-image-not-provisioned` warning still present). - -### Phase 4 - -**4.1 — PASS, with a runbook selector bug.** Exactly **one** managed base image -exists in us-east-1: `arn:aws:lambda:us-east-1:aws:microvm-image:al2023-1`, with -versions `1` (created 2026-07-21) and `0` (2026-06-17). **Ordering is -newest-first**, so the runbook's `items[-1].imageVersion` returns **`0`** (the -older version); `items[0]` returns `1`. The runbook's own warning about ordering -is therefore *load-bearing here*. Selected `1`; the service echoes it back as -`baseImageVersion: "1.0"`. - -**4.2 — Artifact plane PASS; image creation FAILED TWICE on service -validation.** The script read all five outputs, staged `Dockerfile` + `agent/` + -`contracts/`, printed **`584K artifact`**, and uploaded successfully (S3 -`ContentLength` **597305**, SSE `AES256`). - -Then, on the exact request the script builds, verbatim: - -``` -An error occurred (ValidationException) when calling the CreateMicrovmImage operation: The ready (/ready) MicroVM image hook must be enabled when any MicroVM lifecycle hook (run, resume, suspend, or terminate) is enabled. The ready hook signals when the application has finished initializing so the snapshot is taken in a ready state. -``` - -This **refutes the P1 hook-phasing plan directly**. The construct comments state -`/ready` and `/validate` are *"omitted in P1 because the agent does not implement -them yet: configuring a `/validate` endpoint that 404s would fail every image -build"* — but the service **will not accept `run: ENABLED` without -`ready: ENABLED`**. "Declare `/run` in P1, serve it in P2" is not a reachable -state. - -Consequences of that failure: **the conspicuous "P1 image is NOT runnable end to -end" banner was never printed**, because the script's banner heredoc comes after -the `create-microvm-image` call. Also, the runbook's `2>&1 | tee` pipeline -reported `EXIT=0` while the script had failed — the tee status masks it. - -Retrying with `/ready` enabled surfaced the **second** rejection: - -``` -An error occurred (ValidationException) when calling the CreateMicrovmImage operation: The requested memory size of 32768 MiB is not supported by base MicroVM image arn:aws:lambda:us-east-1:aws:microvm-image:al2023-1. Supported memory sizes in MiB are: [512, 1024, 2048, 4096, 8192]. -``` - -The script's and construct's `32768` MiB (`DEFAULT_MINIMUM_MEMORY_MIB`, -documented as *"the service ceiling"*) is **not accepted**. The ceiling for this -base image is **8192 MiB (8 GiB)**, a quarter of what ADR-021 claims. - -**4.3 — Image ACTIVE only after two more corrections; the P1 hook shape is -UNBUILDABLE.** - -*Attempt 1* (hooks omitted, 8192 MiB): create succeeded, returning -`imageVersion: "1.0"` — **not `1`**, contradicting the runbook's -`IMAGE_VERSION=1` and the script's `--image-version 1` guidance. Both builds then -**FAILED** with `stateReason: "The container image build failed."` The exact -root cause, from `/aws/lambda-microvms/backgroundagent-dev-abca-agent`: - -``` -Could not connect to deb.debian.org:80 (146.75.38.132), connection timed out -E: Unable to locate package curl -E: Unable to locate package git -E: Unable to locate package build-essential -E: Package 'gnupg' has no installation candidate -ERROR: process "/bin/sh -c apt-get update && ... apt-get install -y --no-install-recommends curl git build-essential ca-certificates gnupg ..." did not complete successfully: exit code: 100 -``` - -**The construct's 443-only security group makes the agent image unbuildable.** -`agent/Dockerfile` runs `apt-get`, which uses HTTP on **port 80**; DNS resolution -worked, so only the port is the problem. Adding a temporary port-80 egress rule -(`sgr-07ed1fa48ef38467a`) fixed it immediately. - -*Attempt 2* (`update-microvm-image`, port 80 open) produced version **`2.0`**, -both builds `SUCCESSFUL`, `state=SUCCESSFUL`, `status=ACTIVE` in -20:11:18Z → 20:17:09Z = **5 min 51 s**. - -Further API-shape observations: - -- **Two builds per image version**, one per `chipsetGeneration` (`3` and `4`, - `chipset: GRAVITON`). The runbook's `items[0].buildState` inspects only one. -- `list-microvm-image-builds --image-identifier ` → - `ValidationException: Invalid ARN format: backgroundagent-dev-abca-agent`. - **An ARN is required.** `--image-version` accepts either `1` or `1.0`. -- `snapshotBuild` is returned by **`get-microvm-image-build`**, not by - `get-microvm-image-version` (which returned `snapshotBuild: null`). - -Sizes (`buildId 2033b7d8-1aa2-44a4-b174-3fc4bffebcea`, GRAVITON gen 4): - -| Field | Bytes | Human | -|---|---|---| -| `codeInstallSizeInBytes` | 2,334,748,672 | **2.17 GiB** | -| `memorySnapshotSizeInBytes` | 1,216,577,536 | 1.13 GiB | -| `diskSnapshotSizeInBytes` | 37,089,280 | 35.4 MiB | - -**Code-install size (2.17 GiB) exceeds AgentCore's 2 GB container-image limit**, -while the equivalent OCI image built locally was 1.799 GB (629.7 MB compressed). -So the same agent tree is *over* the AgentCore ceiling when measured as MicroVM -code-install and *under* it as an OCI image — the two are not interchangeable -measures, and the ADR narrative should say which one it means. Memory and disk -snapshots are deliberately **not** summed into that comparison. - -**Disk capacity: NOT EXPOSED.** No disk quota appears in `service-quotas` -(full list under 5.5) and no image/version field reports disk capacity. The -32 GB disk claim remains unverified; the 32 GB *memory* claim is refuted (8 GiB). - -*The decisive experiment.* A second image (`…-abca-agent-hooks`) was created with -the **exact P1 hook shape plus the service-mandated `/ready`**, and rebuilt after -port 80 was open. Both builds **FAILED**: - -``` -Ready hook check failed: the application returned a client error (HTTP 4xx) response -``` - -The agent **does** answer on port 8080 (an HTTP 4xx, not a connection failure), -but does not implement `/ready`. Combined with 4.2: **a P1 image that declares -`/run` cannot be built at all**, and an image that omits hooks cannot receive a -`runHookPayload` (5.1). P1 as specified is not merely "not runnable end to end" — -its image is **not creatable**. - -**4.4 — PASS, exactly as designed.** `UPDATE_COMPLETE`. Synth emitted -`abca:microvm-image-p1-not-runnable` with the full expected text. Orchestrator -env is exactly the five variables and **no ingress variable**: - -``` -MICROVM_EGRESS_CONNECTOR_ARNS = arn:aws:lambda:us-east-1::network-connector:nc-132ede11-cb63-4dfa-b75b-6a4713023c1a -MICROVM_EXECUTION_ROLE_ARN = arn:aws:iam:::role/backgroundagent-dev-LambdaMicrovmComputeExecutionRo-ZJu8Y1ybJt1N -MICROVM_IMAGE_IDENTIFIER = backgroundagent-dev-abca-agent -MICROVM_IMAGE_VERSION = 1.0 -MICROVM_PAYLOAD_BUCKET = backgroundagent-dev-lambdamicrovmcomputepayloadbuc-en08fimmvu6h -``` - -Inline policy `TaskOrchestratorOrchestratorFnServiceRoleDefaultPolicyDECF0D43`: - -- `Sid: MicrovmLifecycle` — exactly `lambda:RunMicrovm`, `lambda:GetMicrovm`, - `lambda:TerminateMicrovm` on exactly - `arn:aws:lambda:us-east-1::microvm-image:backgroundagent-dev-abca-agent` - and `…:backgroundagent-dev-abca-agent:*`. **Observed image ARN format matches** - the Service Authorization Reference pattern the construct derives. -- `Sid: MicrovmPassNetworkConnector` — `lambda:PassNetworkConnector` on `*`. -- `Sid: MicrovmPassExecutionRole` — `iam:PassRole` scoped to the execution role - with `iam:PassedToService = lambda.amazonaws.com`. -- **Zero** `SuspendMicrovm` / `ResumeMicrovm` / `CreateMicrovmAuthToken` actions - anywhere in the role (only attached managed policy is - `AWSLambdaBasicDurableExecutionRolePolicy`). - -**Critical mismatch:** `MICROVM_IMAGE_IDENTIFIER` is a **bare name**, and -`RunMicrovm` rejects bare names (5.1). The construct's comment asserts the -opposite — *"`imageIdentifier` may legitimately be a bare image NAME (that is -what `create-microvm-image --name` returns and what `run-microvm ---image-identifier` accepts)"*. `lambda-microvm-strategy.ts:236` passes that env -var straight through, so the P1 orchestrator would fail at `RunMicrovm`. - -### Phase 5 - -**5.0 — ADMIN — advisory.** AssumeRole denied, verbatim: - -``` -An error occurred (AccessDenied) when calling the AssumeRole operation: User: arn:aws:sts:::assumed-role/AdminConsoleAccess/aamorosi-Isengard is not authorized to perform: sts:AssumeRole on resource: arn:aws:iam:::role/backgroundagent-dev-TaskOrchestratorOrchestratorFnS-Bd7rBa2V6Jwf -``` - -Trust policy allows only `Service: lambda.amazonaws.com`. Trust was **not** -modified. All Phase 5 lifecycle evidence is therefore **admin-identity**; -4.4's static policy inspection remains the authoritative scoping evidence. - -**5.1 — Hook-less behaviour MEASURED: the VM runs and stays running.** -Three requests, in order: - -1. Bare name → `ValidationException: Malformed ARN - doesn't start with 'arn:'` -2. ARN + `--run-hook-payload` on the hook-less image → - `ValidationException: The run hook must be enabled in the MicroVM image to pass the run hook payload` -3. ARN, no payload → **success**: - -```json -{ "microvmId": "microvm-b44b69d9-f23b-30d1-97d1-a4ac558cfb5c", - "state": "PENDING", - "endpoint": ".lambda-microvm.us-east-1.on.aws", - "imageArn": "arn:aws:lambda:us-east-1::microvm-image:backgroundagent-dev-abca-agent", - "imageVersion": "2.0", - "maximumDurationInSeconds": 28800, - "ingressNetworkConnectors": ["arn:aws:lambda:us-east-1:aws:network-connector:aws-network-connector:HTTP_INGRESS"], - "egressNetworkConnectors": ["arn:aws:lambda:us-east-1::network-connector:nc-132ede11-cb63-4dfa-b75b-6a4713023c1a"] } -``` - -`maximumDurationInSeconds=28800` accepted; `idlePolicy` omitted as required. -Timeline: `RUNNING` at **+12 s**, and still `RUNNING` at +15/+21/+32/+53/+84/ -**+145 s**, `stateReason` always `None`. It did **not** terminate, stall in -`PENDING`, or disappear. A hook-less MicroVM is a healthy idle VM that bills. - -**Security finding — unrequested public ingress.** The service **auto-attached** -`arn:aws:lambda:us-east-1:aws:network-connector:aws-network-connector:HTTP_INGRESS` -even though `--ingress-network-connectors` was never passed, and returned a -public `*.lambda-microvm.us-east-1.on.aws` endpoint. ADR-021's "no ingress" -posture is not what the service defaults to; it is a *default-on* public HTTP -ingress that P1 neither requests nor suppresses. - -**5.2 — five of six states observed.** `PENDING`, `RUNNING`, `SUSPENDED`, -`TERMINATING`, `TERMINATED`. **`SUSPENDING` was never observable** — suspend -reached `SUSPENDED` in under 1 s. No unknown/unmapped state appeared. -`ResourceNotFoundException` for a reaped ID was **not** reached (5.7). - -**5.3 — PASS.** Admin `suspend-microvm` on a hook-less image with no -`idlePolicy`: **empty response body**, `SUSPENDED` at **+1 s**, stable across -+12/+23/+34/+44/+55 s. Explicit suspend works independently of any idle policy, -and a hook-less image is perfectly suspendible. - -**5.4 — Suspended TTL: survived the full observation window; TRUNCATED at ~1 h.** -Suspended at 20:21:12Z. `maximumDurationInSeconds` was 28800 and `startedAt` -stayed 20:18:05Z throughout. - -| Checkpoint | Wall clock | Suspended age | `get-microvm` state | `list-microvms` | Account quota view | -|---|---|---|---|---|---| -| start | 20:21:13Z | 1 s | `SUSPENDED` | present | `L-CD1C0CC4` = 1024 GB (limit only) | -| +15 min | 20:36:27Z | 915 s | `SUSPENDED` | present | 1024 GB (unchanged) | -| +45 min | 21:06:12Z | 2700 s | `SUSPENDED` | present | 1024 GB (unchanged) | -| ~+1 h (cap) | 21:21:29Z | 3617 s | `SUSPENDED` | present | 1024 GB (unchanged) | - -**No suspended TTL was observed within 1 h.** The runbook's 4-hour checkpoint was -**not** run (time-boxed, as the runbook permits). So the answer is bounded, not -final: *a manually suspended MicroVM with no `idlePolicy` survives at least -1 h 0 min 17 s (3617 s)*; whether a TTL exists between 1 h and the 8 h -`maximumDurationInSeconds` bound is **still open**. The VM was terminated in -Phase 8 rather than left to expire. - -**5.5 — Suspended VM stays listed; quota consumption NOT PROVABLE.** -`service-quotas list-service-quotas --service-code lambda` returned 24 MicroVM -quotas. The load-bearing ones: - -| Code | Name | Value | -|---|---|---| -| `L-CD1C0CC4` | Max allocated memory | **1024 Gigabytes** | -| `L-B430C318` | Max Execution Duration of a MicroVM (in Hours) | **8** | -| `L-942E56BE` | Number of MicroVM images | 100 | -| `L-F8BECE9C` | Versions per MicroVM Image | 50 | -| `L-72E0D058` | Number of concurrent MicroVM image builds | 10 | -| `L-535CA9B6` / `L-91B95582` | Rate / burst of `RunMicrovm` | 5 / 5 | -| `L-90045317` / `L-139F9A48` | Rate / burst of `SuspendMicrovm` | 2 / 2 | -| `L-118C44B3` / `L-25EEC0A4` | Rate / burst of `ResumeMicrovm` | 5 / 5 | -| `L-74787B8A` / `L-2CCA0501` | Rate / burst of `TerminateMicrovm` | 10 / 10 | -| `L-7712260B` / `L-D65D9F16` | Rate / burst of `CreateMicrovmAuthToken` | 50 / 50 | -| `L-772D8D8F` … `L-19741F6D` | Concurrent connections per 1/2/4/8/16 vCPU MicroVM | 8 / 16 / 32 / 64 / 128 | - -`L-B430C318 = 8 hours` independently confirms the 28,800-second bound. There is -**no disk quota at all**, corroborating "disk capacity not exposed". -`L-CD1C0CC4` is `QuotaAppliedAtLevel: ACCOUNT`, described as *"The maximum amount -of memory that can be allocated across all MicroVMs per account per region. -Customers can burst up to 4x this limit."* - -**Utilization is not observable.** `get-service-quota` returns **no -`UsageMetric`** for `L-CD1C0CC4`; `AWS/Usage` exposes only `CallCount` per API -name (`RunMicrovm`, `GetMicrovm`, `CreateMicrovmImage`, …) and **no memory -metric**; there is no MicroVM metric in `AWS/Lambda` and no MicroVM CloudWatch -namespace. Saturating the quota would require ~128 × 8 GiB VMs. - -**Verdict: LISTED, NOT PROVEN.** The suspended VM remained in `list-microvms` at -every checkpoint, but the claim *"a suspended VM still holds account memory -quota"* is **NOT OBSERVABLE SAFELY** in this account and remains undischarged. -Note also that the rationale was written for 32,768 MiB per VM against this -1024 GB quota; at the real 8192 MiB ceiling the arithmetic changes by 4×. - -**5.6 — Resume PASSES with no `/resume` hook declared.** Run on a second VM -(`microvm-d4992cba-92a5-328b-b534-d40ef8715ab3`) so the TTL observation was not -disturbed. `resume-microvm` returned an empty body; state `RUNNING` at **+1 s** -and stable through +55 s. **`microvmId` and `endpoint` were byte-identical before -and after** suspend/resume — lifecycle identity is preserved, so a stored -`SessionHandle` survives a suspend/resume cycle. - -**5.7 — Terminate is fast; `NotFound` did NOT arrive.** Terminate requested -20:26:09Z: `TERMINATING` at **+1 s**, `TERMINATED` at **+3 s**, then still -`TERMINATED` at +9/+20/+41/+72/+133/**+254 s**, and still `TERMINATED` at the -+15-min checkpoint (~10 min after termination) and in `list-microvms` at every -later checkpoint. **`ResourceNotFoundException` was never observed.** The -strategy's `NotFound → completed` mapping is not wrong, but it is **not the -near-term signal**: for at least ~10 minutes the observable terminal state is -`TERMINATED`, so `TERMINATED` must map to completed on its own. - -**5.8 — SKIPPED as a role-scoped test** (Lambda role trust does not allow -operator assumption — 5.0). Advisory admin-identity results: - -- Different image ARN → `ResourceNotFoundException: No active version found for - MicroVM image arn:aws:lambda:us-east-1::microvm-image:backgroundagent-dev-abca-agent-different`. - **Inconclusive** for exact-ARN denial, exactly as the runbook predicts. Note - the message is about *no active version*, not a missing image. -- `create-microvm-auth-token --expiration-in-minutes 5 --allowed-ports - '[{"port":8080}]'` → **SUCCEEDED**. The CLI union syntax is valid, and the - request shape matches the SDK. It returned a genuine JWE under key - `X-aws-proxy-auth` (header `{"kid":"9e81880a-…","alg":"dir","enc":"A256GCM"}`), - **minted against a `SUSPENDED` MicroVM**. So tokens are mintable at will by any - principal holding the action, including for suspended VMs; the no-JWE posture - rests entirely on the orchestrator role omitting the action, which 4.4 verified - statically. - -**5.9 — The 16 KB boundary is WRONG: the real cap is 4096.** Both runbook -payloads were rejected identically: - -``` -An error occurred (ValidationException) when calling the RunMicrovm operation: 1 validation error detected: Value at 'runHookPayload' failed to satisfy constraint: Member must have length less than or equal to 4096 -``` - -Re-measured at the real boundary: **4096 bytes passes length validation** (and -then fails only the hook-enabled check), **4097 bytes is rejected** with the -message above. The CLI **does** expand `file://` (a literal URI would have been -33 bytes and passed). Length validation runs **before** the hook-enabled check, -which is why the boundary was measurable on a hook-less image at all. - -`lambda-microvm-strategy.ts:84` sets `RUN_HOOK_PAYLOAD_LIMIT_BYTES = 16_384` and -inlines anything `<= 16384`; ADR-021 states `≤ 16 KB` in four places. Any -envelope between 4,097 and 16,384 bytes would be inlined by the strategy and -**rejected by the service**. No MicroVM was created by these calls, so no -cleanup was needed. - -### Phase 6 - -**6.1 — PASS.** `npx tsc -p cli/tsconfig.json` built cleanly (mise fallback). -`bgagent configure` wrote config (`BGAGENT_CONFIG_DIR=/tmp/abca-645-bgagent`); -`platform outputs` resolved all seven MicroVM outputs. No Cognito login was -needed for operator AWS commands, as documented. - -**6.2 — PASS.** `repo onboard verification-only/issue-645 --compute-type -lambda-microvm` succeeded (the live `ListManagedMicrovmImages` probe passed) with -`status: active`, `compute_type: lambda-microvm`. -`platform doctor` returned `passed: true` with **all six** checks passing, -including `lambda_microvm_availability` — *"Managed MicroVM images are available -in us-east-1"*. (`github_token` also passed: the CFN-generated secret holds a -32-character generated placeholder, which the check cannot distinguish from a -real PAT — worth noting, since the runbook expected doctor to fail here.) -`runtime status` grouped the row under `lambda_microvm_substrates` with -`used_by_repos: ["verification-only/issue-645"]`. -`repo offboard` set `status: removed` with a TTL. Nothing was skipped. - -### Phase 7 - -**7.1 — DEFERRED-TO-P2-ENV — missing Cognito user/login and real repo -onboarding.** The user pool `us-east-1_sYU3Rftw6` has **zero users**, so -`bgagent login` is impossible, and no genuinely accessible repository is -onboarded; the platform GitHub secret is a 32-character generated placeholder. -None of these were created, per the step's instruction. Independently, the task -path could not have produced meaningful classification evidence: `RunMicrovm` -rejects the bare-name identifier the orchestrator injects (4.4 / 5.1), so every -task would fail with a `ValidationException` at launch rather than at -"session start" as the P1 narrative predicts. - -### Phase 8 — teardown (executed) - -**8.1** — `list-microvms` before teardown showed -`microvm-b44b69d9-…` `SUSPENDED` and `microvm-d4992cba-…` `TERMINATED`. -`terminate-microvm` on the suspended VM (a `SUSPENDED` VM terminates directly, -no resume required) → both `TERMINATED`; no non-terminal VM remained. - -**8.2** — Both out-of-band images deleted (`list-microvm-images` now returns -**empty**). **Correction to the runbook's loop:** the *last remaining* version -cannot be deleted individually — -`ValidationException: This is the last version. Please delete the entire image` — -so the correct order is *delete every version except the last, then delete the -image*, which reaps the final version. A version delete in flight also puts the -image in `UPDATING` and makes concurrent calls fail with -`ConflictException: MicroVM Image is already in state: UPDATING` and -`ValidationException: Cannot delete MicroVM image in its current state: `; -both cleared on retry after ~40 s. Versions reaped: `1.0` + `2.0` on -`…-abca-agent` and `1.0` + `2.0` on `…-abca-agent-hooks` — note that **failed -builds still create versions that must be reaped**. `get-microvm-image` on the -first image now returns -`ResourceNotFoundException: MicroVMImage not found for MicroVMImageID: `. - -**8.3 — Stack deletion is INCOMPLETE: `DELETE_FAILED`, blocked by leaked -AgentCore ENIs. All billable resources are confirmed gone.** - -`cdk destroy` ran 21:24:34Z and deleted 460 of 464 resources. It then failed: - -``` -The following resource(s) failed to delete: [AgentVpcRuntimeSG96507CD0, AgentVpcPrivateSubnet1Subnet8051BB57, AgentVpcPrivateSubnet2SubnetC66971D0]. -resource sg-05f50b9950d41572e has a dependent object (Service: Ec2, Status Code: 400 …) -Resource handler returned message: "The subnet 'subnet-0befbffccbb83b718' has dependencies and cannot be deleted. (Service: Ec2, Status Code: 400 …)" -``` - -Cause: two ENIs of `InterfaceType: agentic_ai` (AgentCore-managed, requester -`AROA…[redacted]:[redacted]`) — `eni-080353b2356328ed7` and -`eni-04500acbfed377f4a` — remained `in-use` in the private subnets. The runbook's -own gotcha ("VPC teardown can lag while service-managed ENIs are reclaimed; wait -and retry rather than force-deleting resources past CloudFormation") was -followed: **three** `delete-stack` retries spread over ~1 h 40 min (21:48Z, -22:37Z, 23:05Z) all returned `DELETE_FAILED`, and the ENIs were still `in-use` -each time. A direct `delete-network-interface` was attempted once for diagnosis -and correctly refused (`InvalidParameterValue: Network interface … is currently -in use.`); nothing was force-deleted past CloudFormation. - -Final residual state of `backgroundagent-dev` — **4 resources, all zero-cost**: -`AWS::EC2::VPC AgentVpcA6796801` (`vpc-0a12c1a64cc960c6a`), two private subnets, -and one security group, plus the two AgentCore ENIs holding them. - -**Billing is stopped.** Verified after teardown: - -- MicroVMs: both `TERMINATED`; `list-microvm-images` empty (no snapshot storage). -- **NAT gateways: none in the ABCA VPC** (`nat-0c6cdbee97699f4e8` deleted - 21:24:43Z). The two `available` NAT gateways in the account belong to - pre-existing `vpc-01c9984d163d2965e` and were **not** created or touched by - this run. -- **VPC endpoints in the ABCA VPC: none.** -- ABCA S3 buckets: none (auto-delete custom resources emptied them). -- Elastic IPs: no unattached (billable) addresses. -- `/aws/lambda-microvms/backgroundagent-dev-abca-agent`: deleted. - -Retry command for whoever picks this up (should succeed once AgentCore releases -the ENIs): - -```bash -aws cloudformation delete-stack --stack-name backgroundagent-dev -aws cloudformation wait stack-delete-complete --stack-name backgroundagent-dev -``` - -**Verification-only resources created and removed:** the -`abca645-connector-probe` stack (deleted — `Stack with id -abca645-connector-probe does not exist`) and the temporary port-80 egress rule -`sgr-07ed1fa48ef38467a` (removed with its security group when the stack deleted -it). The local `finch` VM was stopped. - -**Bootstrap retained as instructed:** `CDKToolkit` `UPDATE_COMPLETE`, -`ComputeTypes = agentcore,lambda-microvm`, with all five ABCA policies attached -to `cdk-hnb659fds-cfn-exec-role--us-east-1` and **no -`AdministratorAccess`**. - ---- - -## Results table (fill this in) - -Use one row per material observation; add rows as needed. - -| Step | Expected | Observed | ADR item discharged | Feeds back to design? | -|---|---|---|---|---| -| Setup | Vars + evidence dir | `us-east-1`, `backgroundagent-dev`, evidence `/tmp/abca-645-p1-20260731T184822Z`. `mise` absent → all **raw fallbacks** used | Reproducibility | No | -| 0.1 | Virgin account, correct branch | Account ``, admin `AdminConsoleAccess/aamorosi-Isengard`, branch OK, SHA `0505f914`. `backgroundagent-dev` absent (`ValidationError … does not exist`). **Account not virgin overall** — 3 unrelated stacks + `CDKToolkit` pre-existed; ABCA itself never deployed, so the premise holds | Clean first-deploy path | No | -| 0.2 | CLI exposes Lambda MicroVMs; SDK 3.1098.0 | `aws lambda-microvms help` **exit 0**, 24 commands, all 7 required present. CLI command list is an **exact match** to SDK 3.1098.0 (24 vs 24). CLI 2.36.13, cdk 2.1129.0, node 24.16.0, python 3.9.6, **openrsync** (worked). Skeleton confirms `ARM_64`-only and `ENABLED\|DISABLED` hooks with a single `hooks.port` and **no hook-path field** | API/action-name verification | **Yes — CFN L1 shapes (`arm64`, `run:'/run'`) have no counterpart in the service model** | -| 1.2 | Custom template replaces admin bootstrap | **DEFECT: `cdk bootstrap --template …` is a silent no-op** on an already-bootstrapped account (`Not overwriting it with a template containing 'ABCA: Least-Privilege Bootstrap' (use --force …)`, **exit 0**). `--force` required. `BootstrapVariant` then *stays* `AWS CDK: Default Resources`, so every future non-forced bootstrap refuses again. **DEFECT: the runbook's `--parameters` shorthand is rejected** (`Invalid type for parameter Parameters[0].ParameterValue, value: ['agentcore', 'lambda-microvm'] … valid types: `); the escaped `agentcore\,lambda-microvm` works | Conditional bootstrap IAM | **Yes — `mise //cdk:bootstrap` + runbook/docs** | -| 1.3 | Custom policy attached for `agentcore,lambda-microvm` | `ComputeTypes=agentcore,lambda-microvm`; exactly one `IaCRole-ABCA-Compute-LambdaMicrovms` policy, attached to `cdk-hnb659fds-cfn-exec-role-…`; 5 ABCA policies, **no `AdministratorAccess`**; 19 MicroVM/connector actions incl. `PassNetworkConnector` | Conditional bootstrap IAM | No | -| 2.1 (synth) | No-image warning | `abca:microvm-image-not-provisioned` present, `…p1-not-runnable` correctly absent. Incidental: template 893273/1000000, 463/500 resources | First-deploy bootstrap state | No | -| 2.1 (deploy) | Substrate deploys | **BLOCKED — cannot deploy from unmodified sources.** `AWS::Lambda::NetworkConnector` CREATE_FAILED: `"NetworkConnectorOperatorRole is required for VPC_EGRESS connector type (… Status Code: 400 …)" HandlerErrorCode: InvalidRequest`. Refutes the construct's stated *"`operatorRole` is left unset so Lambda manages the ENIs with its own service-linked role"*. Deployed only after patching in an operator role (+ moving off `us-east-1a`): `CREATE_COMPLETE` in **13 min 42 s** | Conditional substrate — **only with a code fix** | **YES — construct must create + pass an operator role; L1 marks it optional** | -| 2.1 (deploy, 2nd) | — | **AgentCore blocker:** `The following subnets are in unsupported availability zones in region us-east-1: subnet-… in us-east-1a (ID: use1-az6). Supported availability zones are: use1-az4, use1-az1, use1-az2`. This account maps `us-east-1a`→`use1-az6`; `AgentVpc` does not constrain AZs | — | **YES — `AgentVpc` should pin AgentCore-supported AZs** | -| 2.1 (rollback) | — | Rollback itself failed: `Validation failed during DeleteMemory: Memory is in transitional state CREATING. Cannot delete memory.` → `ROLLBACK_FAILED`; plain `delete-stack` cleared it | — | Minor — yes (`AgentMemory` delete retry) | -| 2.2 | Outputs/resources present | `ComputeSubstrate=lambda-microvm`; all six `Microvm…` outputs populated; key `microvm-images/agent-artifact.zip`; 2 buckets, build+execution roles, **443-only SG** (`sg-0e662dc0d6f6e9ade`, one rule tcp/443), `/aws/lambda-microvms/…` log group, `AWS::Lambda::NetworkConnector`; **no `AWS::Lambda::MicrovmImage`** | Conditional substrate/config | No | -| 2.3 | No `MICROVM_*` env | `[]` ✓ (14 env keys). **DEFECT: the runbook's `ORCHESTRATOR_FN` query is broken by pagination** — 464 resources → the query ran per page and returned 5 values (`None None None None `), breaking the next call. Needs `--no-paginate` | Reject-without-image config | Yes — runbook | -| 2.4 | Construct resources have backend tag | All **6** taggable resources tagged `abca:compute-backend=lambda-microvm` (SG, connector, log group, 2 buckets, 2 roles). `resourcegroupstaggingapi` returned only 5 — **IAM roles are not returned by that API** (coverage gap, not a missing tag); confirmed via `iam list-role-tags` | Cost attribution | No | -| 3.1–3.2 | Unsupported failure; override warning | 3.1 exit 1 naming `eu-central-1`, all five Regions, and `--context microvm_region_override=true`. 3.2 exit 0 with `abca:microvm-region-override` (plus the not-provisioned warning) | Region gate/escape hatch | No | -| 4.1 | Live regional probe | Exactly **one** base image: `…:aws:microvm-image:al2023-1`, versions `1` and `0`. **Ordering is newest-first, so the runbook's `items[-1]` selects the OLDER version `0`**; `items[0]`=`1` is correct. Service echoes `baseImageVersion: "1.0"` | Regional availability probe | Yes — runbook selector | -| 4.2 (artifact) | Script uploads | Staged + zipped + uploaded fine on macOS/openrsync: script printed `584K artifact`, S3 `ContentLength 597305`, SSE `AES256` | Packaging plane | No | -| 4.2 (create) | Image created; P1 banner shown | **FAILED:** `The ready (/ready) MicroVM image hook must be enabled when any MicroVM lifecycle hook (run, resume, suspend, or terminate) is enabled.` → **the P1 banner was never printed** (it comes after the failing call), and the runbook's `2>&1 \| tee` reported `EXIT=0`, masking the failure | **NOT discharged** — packaging + operator warning | **YES — "declare `/run` in P1, serve it in P2" is not a reachable state** | -| 4.2 (memory) | 32,768 MiB accepted | **FAILED:** `The requested memory size of 32768 MiB is not supported by base MicroVM image …al2023-1. Supported memory sizes in MiB are: [512, 1024, 2048, 4096, 8192].` Real ceiling **8192 MiB (8 GiB)**, ¼ of the documented figure | Refutes sizing premise | **YES — `DEFAULT_MINIMUM_MEMORY_MIB` and ADR-021's "32 GB RAM"** | -| 4.3 (build 1) | Build successful | **FAILED** (`The container image build failed.`). Root cause in the log group: `Could not connect to deb.debian.org:80 (146.75.38.132), connection timed out` → `E: Unable to locate package curl/git/build-essential` → `exit code: 100`. **The construct's 443-only SG makes the agent image unbuildable** (`apt-get` needs port 80; DNS was fine) | **NOT discharged** without a fix | **YES — SG must allow 80, or the Dockerfile must not use HTTP apt** | -| 4.3 (build 2) | Image ACTIVE | After adding a temporary port-80 egress rule: version **`2.0`** `state=SUCCESSFUL`, `status=ACTIVE` in **5 min 51 s**, both builds `SUCCESSFUL` | Buildability (with fixes) | No | -| 4.3 (shapes) | `IMAGE_VERSION=1` | **Version is `1.0`, not `1`.** **Two builds per version** (`chipsetGeneration` 3 and 4, GRAVITON) — the runbook's `items[0]` checks only one. `list-microvm-image-builds --image-identifier ` → `ValidationException: Invalid ARN format: …` (**ARN required**). `snapshotBuild` lives on `get-microvm-image-build`, **not** on the version (which returns `null`) | State/shape mapping | Yes — runbook + script | -| 4.3 (sizes) | Size vs 2 GB narrative | `codeInstallSizeInBytes` **2,334,748,672 (2.17 GiB) — exceeds AgentCore's 2 GB container-image limit**; `memorySnapshotSizeInBytes` 1,216,577,536 (1.13 GiB); `diskSnapshotSizeInBytes` 37,089,280 (35.4 MiB). Same tree as an OCI image = 1.799 GB (629.7 MB compressed). Snapshots deliberately not summed | Sizing narrative | **Yes — state which measure the 2 GB comparison uses** | -| 4.3 (disk) | Verify 32 GB disk | **NOT EXPOSED** — no disk quota in `service-quotas`, no disk field on image/version. 32 GB disk unverified; 32 GB *memory* refuted | Disk external fact | Yes | -| 4.3 (P1 shape) | — | **Decisive:** the exact P1 hook shape + service-mandated `/ready` **FAILED both builds**: `Ready hook check failed: the application returned a client error (HTTP 4xx) response`. The agent *does* answer on 8080 but not `/ready`. **A P1 image declaring `/run` is not creatable at all** | Refutes P1 premise | **YES — P1/P2 hook phasing** | -| 4.4 | Env present; exact image IAM; no JWE grant | **PASS exactly as designed.** `abca:microvm-image-p1-not-runnable` emitted. Exactly 5 `MICROVM_*` vars, no ingress var. `MicrovmLifecycle` = exactly `RunMicrovm`/`GetMicrovm`/`TerminateMicrovm` on `…:microvm-image:backgroundagent-dev-abca-agent` + `:*`; `MicrovmPassNetworkConnector` on `*`; `MicrovmPassExecutionRole` with `iam:PassedToService=lambda.amazonaws.com`; **zero** Suspend/Resume/AuthToken actions | Least privilege | No | -| 4.4 (identifier) | Bare name accepted by RunMicrovm | **REFUTED.** `MICROVM_IMAGE_IDENTIFIER` is the bare name `backgroundagent-dev-abca-agent`; `run-microvm` with a bare name → `ValidationException: Malformed ARN - doesn't start with 'arn:'`. `lambda-microvm-strategy.ts:236` passes it straight through, so P1 would fail at launch | Refutes construct comment | **YES — inject the image ARN, not the name** | -| 5.0 | Assume-role identity or trust denial | **ADMIN — advisory.** `AccessDenied … not authorized to perform: sts:AssumeRole on resource: …TaskOrchestratorOrchestratorFnS-Bd7rBa2V6Jwf`; trust = `lambda.amazonaws.com` only. Trust not modified | Verification confidence | No | -| 5.1 | Hook-less behavior measured, not assumed | **Measured: it runs and keeps running.** `RUNNING` at **+12 s**, still `RUNNING` at +145 s, `stateReason` always `None` — no terminate, no stall, no disappearance. `maximumDurationInSeconds=28800` accepted, `idlePolicy` omitted. A payload on a hook-less image is rejected: `The run hook must be enabled in the MicroVM image to pass the run hook payload` | P1/P2 phase boundary | **Yes — P2 startup/hooks** | -| 5.1 (ingress) | No ingress | **Service auto-attached `…:aws:network-connector:aws-network-connector:HTTP_INGRESS`** with a public `*.lambda-microvm.us-east-1.on.aws` endpoint, though none was requested. "No ingress" is not the service default | Security posture | **YES — P1 must suppress or accept default public ingress** | -| 5.2 | Actual state enum values recorded | Observed 5 of 6: `PENDING`, `RUNNING`, `SUSPENDED`, `TERMINATING`, `TERMINATED`. **`SUSPENDING` never observable** (<1 s). No unknown state. `ResourceNotFoundException` not reached | State mapping | Yes (see 5.7) | -| 5.3 | Manual suspend without idle policy | **PASS.** Admin suspend on a hook-less image, no `idlePolicy`: **empty response body**, `SUSPENDED` at **+1 s**, stable | Explicit suspend external fact | **Yes — P3 lifecycle** | -| 5.4 | Suspended TTL/checkpoint result | **No TTL within 1 h.** `SUSPENDED` at start / +15 min / +45 min / +1 h (3617 s); `startedAt` and `maximumDurationInSeconds=28800` unchanged. **TRUNCATED at ~1 h**; the 4 h checkpoint was NOT run, so a TTL between 1 h and the 8 h bound is **still open** | Partially — bounded below only | **Yes — timeout policy** | -| 5.5 | Suspended quota consumption proven/inconclusive | **LISTED, NOT PROVEN — NOT OBSERVABLE SAFELY.** Suspended VM present in `list-microvms` at every checkpoint. `L-CD1C0CC4 Max allocated memory = 1024 GB` (ACCOUNT, "burst up to 4x"), **no `UsageMetric`**; `AWS/Usage` has only `CallCount`; no MicroVM memory metric anywhere. Proving it needs ~128 × 8 GiB VMs. `L-B430C318 = 8 hours` independently confirms the 28,800 s bound; **no disk quota exists** | **NOT discharged** | **Yes — concurrency policy; rationale was sized on 32 GiB/VM, real is 8 GiB** | -| 5.6 | Resume transitions/result | **PASS with no `/resume` hook declared.** `RUNNING` at **+1 s**, empty response body; **`microvmId` and `endpoint` byte-identical** across suspend→resume, so a stored `SessionHandle` survives | Resume external fact | **Yes — P3 hooks/reconciliation** | -| 5.7 | Terminate→NotFound timing | `TERMINATING` **+1 s** → `TERMINATED` **+3 s**, then `TERMINATED` at +254 s and still `TERMINATED` ~10 min later and at every later checkpoint. **`ResourceNotFoundException` never observed** | Partially — terminate path yes, `NotFound` mapping no | **YES — `TERMINATED` must map to completed; `NotFound` is not the near-term signal** | -| 5.8 | Different image/JWE denied under role | **SKIPPED — LAMBDA ROLE TRUST DOES NOT ALLOW OPERATOR ASSUMPTION.** Advisory (admin): different image ARN → `ResourceNotFoundException: No active version found for MicroVM image …-different` (**inconclusive**, as predicted). `create-microvm-auth-token … --allowed-ports '[{"port":8080}]'` **SUCCEEDED** as admin, returning a real JWE under `X-aws-proxy-auth` (`{"alg":"dir","enc":"A256GCM"}`) **against a SUSPENDED VM**; CLI union syntax valid. No-JWE posture rests solely on 4.4's role omission | Static only (4.4) | Yes — tokens are mintable for suspended VMs | -| 5.9 | 16,384 accepted; 16,385 rejected | **REFUTED — the cap is 4096, not 16,384.** Both 16,384 and 16,385 → `Value at 'runHookPayload' failed to satisfy constraint: Member must have length less than or equal to 4096`. Re-measured: **4096 passes, 4097 rejected**. CLI does expand `file://`. Length validation precedes the hook check. `RUN_HOOK_PAYLOAD_LIMIT_BYTES = 16_384` would inline 4,097–16,384-byte envelopes that the service rejects | **Refutes the documented boundary** | **YES — strategy threshold + ADR-021 (4 places)** | -| 6.1–6.2 | Onboard probe, doctor, grouping, cleanup | **PASS, nothing skipped.** `npx tsc` build clean; `platform outputs` resolved all 7 MicroVM outputs; onboard probe passed with `compute_type=lambda-microvm`, `status=active`; doctor `passed: true` with all 6 checks including `lambda_microvm_availability` ("Managed MicroVM images are available in us-east-1"); `runtime status` grouped under `lambda_microvm_substrates`; offboard `status=removed` + TTL. Note `github_token` **passed** on a 32-char generated placeholder | CLI regional enforcement | Minor — yes (`github_token` can't detect a placeholder) | -| 7.1 | Negative task evidence or explicit deferral | **DEFERRED-TO-P2-ENV — missing Cognito user/login, real GitHub token, and real repo onboarding.** User pool `us-east-1_sYU3Rftw6` has **zero users**; secret is a 32-char placeholder. Not created, per the step. Independently moot: `RunMicrovm` rejects the bare-name identifier, so tasks would fail at launch, not at session start | Not discharged (by design) | **Yes — P2 env** | -| 8.1–8.2 | VMs/images gone | **PASS.** Both MicroVMs `TERMINATED` (a `SUSPENDED` VM terminates directly, no resume needed). `list-microvm-images` **empty**. **Correction:** the last remaining version cannot be deleted alone (`This is the last version. Please delete the entire image`) — delete all but the last, then the image. Concurrent calls during a version delete give `ConflictException: MicroVM Image is already in state: UPDATING`. **Failed builds still create versions that must be reaped** (`1.0`+`2.0` on both images) | Versioned image lifecycle | Yes — runbook loop order | -| 8.3 | Stack gone; bootstrap retained | **PARTIAL — `DELETE_FAILED`.** 460/464 resources deleted; VPC + 2 private subnets + 1 SG remain, blocked by two leaked AgentCore ENIs (`InterfaceType: agentic_ai`) still `in-use` after 3 retries over ~1 h 40 min. **All billable resources confirmed gone** (no ABCA NAT gateway, no VPC endpoints, no buckets, no images/VMs, no unattached EIPs); residual 4 resources are zero-cost. Nothing force-deleted past CFN. `CDKToolkit` retained `UPDATE_COMPLETE`, `ComputeTypes=agentcore,lambda-microvm`, 5 ABCA policies, no `AdministratorAccess`. Verification-only extras (`abca645-connector-probe`, temp rule `sgr-07ed1fa48ef38467a`) removed | Partially — cleanup blocked by AgentCore | **YES — AgentCore ENI reclaim blocks clean `cdk destroy`** | - -## Findings summary - -Live run, 2026-07-31, account ``, `us-east-1`, branch -`feat/645-lambda-microvm-p1` @ `0505f914`. Evidence: -`/tmp/abca-645-p1-20260731T184822Z`. - -**Headline:** the *P1 substrate* is broadly correct — conditional bootstrap IAM, -outputs, tags, region gate, warnings, exact-ARN least privilege, and the CLI -surface all behave as designed. But **P1 cannot deploy, cannot build its image, -and cannot launch a MicroVM from unmodified sources**: five independent -live-service rejections had to be worked around to get any empirical result, and -three documented ADR-021 constants (32 GB memory, 16 KB payload, bare-name image -identifier) are **wrong**. - -### Items discharged (behaved exactly as designed) - -1. **0.2** — CLI/SDK operation names: `aws lambda-microvms` exposes 24 commands, - an exact match to SDK 3.1098.0. No action-name drift; the packaging script's - `ARM_64` / `ENABLED` shapes are confirmed correct against the live model. -2. **1.3** — Conditional bootstrap IAM: `IaCRole-ABCA-Compute-LambdaMicrovms` is - created and attached only with `ComputeTypes` including `lambda-microvm`, and - `AdministratorAccess` really is replaced. -3. **2.2** — Substrate contract: `ComputeSubstrate=lambda-microvm`, all six - `Microvm…` outputs, both buckets, both roles, 443-only SG, `/aws/lambda-microvms/` - log group, network connector, and **no** `AWS::Lambda::MicrovmImage`. -4. **2.3** — All-or-nothing config: zero `MICROVM_*` env vars without an image. -5. **2.4** — Cost tags: all six taggable construct resources carry - `abca:compute-backend=lambda-microvm`. -6. **3.1 / 3.2** — Region gate and escape hatch, verbatim as specified. -7. **4.1** — Live regional availability probe works. -8. **4.2 (artifact half)** — Packaging plane: zip+Dockerfile staging, no secret - build inputs, correct bucket/key, works on macOS with `openrsync`. -9. **4.4** — Least privilege: exactly `RunMicrovm`/`GetMicrovm`/`TerminateMicrovm` - on exactly the image ARN + `:*`; `PassNetworkConnector`; scoped `iam:PassRole`; - **zero** `SuspendMicrovm`/`ResumeMicrovm`/`CreateMicrovmAuthToken`. Both - no-image and image-configured warnings fire correctly. -10. **5.3** — Manual suspend works without any `idlePolicy` (`SUSPENDED` in ~1 s). -11. **5.6** — Manual resume works with no `/resume` hook declared; `microvmId` - **and** `endpoint` are preserved, so `SessionHandle` survives a cycle. -12. **5.7 (terminate half)** — Explicit terminate is near-instant - (`TERMINATING` +1 s → `TERMINATED` +3 s). -13. **6.1 / 6.2** — CLI: outputs discovery, live `ListManagedMicrovmImages` - onboarding probe, `lambda_microvm_availability` doctor check, - `lambda_microvm_substrates` grouping, and offboard all pass. -14. **8.1 / 8.2** — MicroVM and image cleanup paths work (with the version-order - correction below). - -### Items contradicting design assumptions — `feeds-back-to-design: YES` - -Ordered by severity. - -**F1. The P1 image is not creatable at all** (blocks the entire P1 premise). -`create-microvm-image` with the script's/construct's hook shape: - -``` -ValidationException: The ready (/ready) MicroVM image hook must be enabled when any MicroVM lifecycle hook (run, resume, suspend, or terminate) is enabled. The ready hook signals when the application has finished initializing so the snapshot is taken in a ready state. -``` - -And with `/ready` added as demanded, both builds fail: - -``` -Ready hook check failed: the application returned a client error (HTTP 4xx) response -``` - -ADR-021's hook-phasing plan ("declare `/run` in P1, serve it in P2; omit `/ready` -and `/validate` because the agent does not implement them") is **not a reachable -service state**. The only creatable P1 image is one with **no hooks at all**, and -such an image **cannot accept a `runHookPayload`** -(`The run hook must be enabled in the MicroVM image to pass the run hook -payload`) — so P1's payload-delivery path cannot function either. The agent does -answer on port 8080 (HTTP 4xx, not a connection refusal), so serving `/ready` -is the unblocking change. - -**F2. The substrate cannot deploy: the network connector requires an operator -role.** - -``` -"NetworkConnectorOperatorRole is required for VPC_EGRESS connector type (Service: Lambda, Status Code: 400, Request ID: 04726267-6c61-4ff5-bb1d-302122e9f955)" HandlerErrorCode: InvalidRequest -``` - -This refutes the explicit comment in `lambda-microvm-compute.ts` (~L467): -*"`operatorRole` is left unset so Lambda manages the ENIs with its own -service-linked role rather than a role we would have to trust."* The generated L1 -also mis-signals it as optional (`readonly operatorRole?: string`). Proven fix -(validated standalone): a role trusting `lambda.amazonaws.com` with -`AWSLambdaVPCAccessExecutionRole` + `ec2:CreateNetworkInterface` / -`DeleteNetworkInterface` / `DescribeNetworkInterfaces` / `DescribeSubnets` / -`DescribeVpcs` / `DescribeSecurityGroups` / `CreateTags` / -`AssignPrivateIpAddresses` / `UnassignPrivateIpAddresses` / -`Describe|ModifyNetworkInterfaceAttribute`. - -**F3. `RunMicrovm` requires an image ARN; the orchestrator injects a bare name.** - -``` -ValidationException: Malformed ARN - doesn't start with 'arn:' -``` - -`MICROVM_IMAGE_IDENTIFIER` is set to `backgroundagent-dev-abca-agent` and -`lambda-microvm-strategy.ts:236` passes it straight to `RunMicrovm`. The -construct's comment claims *"`run-microvm --image-identifier` accepts"* bare -names — it does not. Every P1 task would fail at launch. The same applies to -`list-microvm-image-builds` (`ValidationException: Invalid ARN format: …`), which -the packaging script's operator instructions also get wrong. Note the construct -already derives the correct ARN for IAM, so the fix is to inject that ARN. - -**F4. The 443-only security group makes the agent image unbuildable.** - -``` -Could not connect to deb.debian.org:80 (146.75.38.132), connection timed out -E: Unable to locate package curl / git / build-essential -… did not complete successfully: exit code: 100 -``` - -`agent/Dockerfile` runs `apt-get`, which uses **HTTP/80**; the construct's SG -allows only 443 (DNS resolution succeeded, so the port is the sole cause). Either -the SG must allow 80 for build-time egress, or the Dockerfile must use an -HTTPS apt transport/mirror. Opening port 80 made the build succeed immediately. - -**F5. Memory: 32,768 MiB is rejected; the real ceiling is 8,192 MiB.** - -``` -ValidationException: The requested memory size of 32768 MiB is not supported by base MicroVM image arn:aws:lambda:us-east-1:aws:microvm-image:al2023-1. Supported memory sizes in MiB are: [512, 1024, 2048, 4096, 8192]. -``` - -`DEFAULT_MINIMUM_MEMORY_MIB = 32768` is documented in the construct as *"the -service ceiling"*, and ADR-021 states **32 GB RAM** in at least three places -(comparison table, constraint paragraph, consequences). The real ceiling for the -only available base image (`al2023-1`) is **8 GiB — one quarter**. This -materially changes the "MicroVMs target the default-sized workload" positioning -*and* the 5.5 concurrency arithmetic (which was sized on 32 GiB/VM). - -**F6. `runHookPayload` cap is 4096 bytes, not 16,384.** - -``` -ValidationException: 1 validation error detected: Value at 'runHookPayload' failed to satisfy constraint: Member must have length less than or equal to 4096 -``` - -Boundary measured exactly: **4096 passes, 4097 fails**. Both of the runbook's -16 KB probes failed. `RUN_HOOK_PAYLOAD_LIMIT_BYTES = 16_384` -(`lambda-microvm-strategy.ts:84`) would inline every envelope from 4,097 to -16,384 bytes and the service would reject all of them; ADR-021 repeats `≤ 16 KB` -in four places, including the P1 requirement statement. - -**F7. `RunMicrovm` attaches a default public HTTP ingress connector.** -Without `--ingress-network-connectors`, the response contained: - -``` -"ingressNetworkConnectors": ["arn:aws:lambda:us-east-1:aws:network-connector:aws-network-connector:HTTP_INGRESS"] -``` - -plus a public `*.lambda-microvm.us-east-1.on.aws` endpoint. ADR-021's "no -orchestrator→agent HTTP path / no ingress in P1–P3" posture is **not the service -default**; P1 either has to suppress it explicitly or document that a public -ingress endpoint exists on every agent MicroVM. - -**F8. `NotFound` is not the near-term terminal signal.** After terminate, the VM -was `TERMINATED` at +3 s and **still `TERMINATED` ~10 min later** and at every -subsequent checkpoint; `ResourceNotFoundException` was **never observed**. The -strategy's `NotFound → completed` mapping is fine as a fallback, but `TERMINATED` -must map to completed in its own right or the orchestrator will poll a terminal -VM indefinitely. Relatedly, **`SUSPENDING` is not observable** (suspend reaches -`SUSPENDED` in <1 s), so any state machine that waits for `SUSPENDING` will hang. - -**F9. Hook-less MicroVMs run indefinitely and bill.** A P1-style hook-less image -reaches `RUNNING` in **12 s** and stays `RUNNING` (no `stateReason`, no -self-termination) up to its 8 h `maximumDurationInSeconds`. The P1 narrative -("a launch may return an ID and endpoint and then fail its hook or terminate") -is wrong in the safe direction for correctness and wrong in the *expensive* -direction for cost: nothing fails, so nothing cleans up. Active -`TerminateMicrovm` is mandatory, not belt-and-braces. - -**F10. `mise //cdk:bootstrap` silently no-ops on an already-bootstrapped -account** — `Not overwriting it with a template containing 'ABCA: Least-Privilege -Bootstrap' (use --force if you intend to overwrite)` with **exit 0**. Worse, after -a forced bootstrap `BootstrapVariant` remains `AWS CDK: Default Resources`, so -the refusal recurs forever. Any operator following ADR-002's documented flow on an -existing account keeps `AdministratorAccess`. - -**F11. `AgentVpc` picks AZs AgentCore rejects.** -`The following subnets are in unsupported availability zones in region us-east-1: -subnet-… in us-east-1a (ID: use1-az6). Supported availability zones are: -use1-az4, use1-az1, use1-az2`. AZ *names* are account-scoped, so this is a -latent first-deploy failure for any account whose `us-east-1a` maps to `use1-az6`. - -**F12. AgentCore leaks ENIs and blocks `cdk destroy`.** Two -`InterfaceType: agentic_ai` ENIs stayed `in-use` for >1 h 40 min after runtime -deletion, leaving the stack `DELETE_FAILED` with VPC/subnets/SG undeletable. -Also `AWS::BedrockAgentCore::Memory` cannot be deleted while `CREATING` -(`Validation failed during DeleteMemory: Memory is in transitional state -CREATING`), which turned one rollback into `ROLLBACK_FAILED`. - -**F13. `codeInstallSizeInBytes` = 2.17 GiB exceeds AgentCore's 2 GB image -limit**, while the same tree as an OCI image is 1.799 GB (629.7 MB compressed). -The ADR's "compare to AgentCore's 2 GB limit" narrative must say *which* measure -it means, because the two straddle the limit. - -**F14. Runbook/tooling defects found in execution** (lower severity, but they -would silently corrupt a future pass): - -- The `--parameters 'ParameterKey=ComputeTypes,ParameterValue=agentcore,lambda-microvm'` - form is rejected (`Invalid type for parameter … valid types: `); - needs `agentcore\,lambda-microvm`, as `cdk/mise.toml` already shows. -- `list-stack-resources --query "…|[0]"` is **paginated** at 464 resources and - returned five values, breaking `ORCHESTRATOR_FN`. Needs `--no-paginate`. -- `list-managed-microvm-image-versions` is **newest-first**, so `items[-1]` - selects the *older* version. -- Image version is **`1.0`**, not `1`, so `IMAGE_VERSION=1` is wrong. -- There are **two builds per version** (GRAVITON gen 3 and 4); `items[0]` checks - only one. -- `snapshotBuild` comes from `get-microvm-image-build`, not - `get-microvm-image-version` (which returns `null`). -- The script's failure is masked by `2>&1 | tee` (reported `EXIT=0`), and its - "P1 image is NOT runnable" banner never prints because it sits *after* the - `create-microvm-image` call that fails. -- Teardown: the **last** image version cannot be deleted individually - (`This is the last version. Please delete the entire image`). -- Post-synth cloud-assembly template edits are ignored if the template's S3 - asset object already exists (key = pre-edit content hash). - -### Items skipped, blocked, or inconclusive - -| Item | Verdict | Reason | -|---|---|---| -| 1.3 optional negative (scratch-qualifier bootstrap) | **SKIPPED** | The runbook directs skipping it; not cheap enough and complicates teardown. | -| CFN `AWS::Lambda::MicrovmImage` value shapes (`arm64`, `run:'/run'`) | **NOT TESTED** | The construct only synthesizes the L1 with `microvm_base_image_arn`/`_version` context, which the runbook's Phase 4 does not use. The API side is settled (0.2) and the request would be rejected on hook semantics anyway (F1), so the CFN-vs-API shape question stays **open**. | -| 5.4 suspended TTL beyond 1 h | **TRUNCATED / OPEN** | Time-boxed at ~1 h as the runbook permits. `SUSPENDED` held at start/+15/+45/+60 min (3617 s). The 4 h checkpoint was not run; a TTL between 1 h and the 8 h bound remains unknown. Bounded result: **survives ≥ 1 h with no `idlePolicy`**. | -| 5.5 suspended VM consumes account memory quota | **NOT OBSERVABLE SAFELY / UNDISCHARGED** | The VM stays in `list-microvms`, but that only proves *listed*. `L-CD1C0CC4` (1024 GB, ACCOUNT) exposes **no `UsageMetric`**; `AWS/Usage` has only `CallCount`; no MicroVM memory metric exists in any namespace; no console utilization view is reachable from a CLI-only session. Proving it would need ~128 × 8 GiB VMs. | -| 5.8 IAM negatives under the orchestrator role | **SKIPPED — LAMBDA ROLE TRUST DOES NOT ALLOW OPERATOR ASSUMPTION** | `AccessDenied … sts:AssumeRole`; trust is `lambda.amazonaws.com` only and was deliberately not modified. 4.4's static policy inspection is authoritative. | -| 5.8 exact-ARN denial sub-check | **INCONCLUSIVE** | As the runbook predicts, a different image name returns `ResourceNotFoundException: No active version found for MicroVM image …-different`, not `AccessDenied`. Advisory only (admin identity). | -| 5.8 no-JWE posture | **STATIC ONLY** | As admin, `create-microvm-auth-token` **succeeded**, returning a real JWE (`{"alg":"dir","enc":"A256GCM"}`, key `X-aws-proxy-auth`) — **against a `SUSPENDED` VM**. The posture depends entirely on the role omitting the action. | -| 7.1 negative task path | **DEFERRED-TO-P2-ENV** | Cognito pool `us-east-1_sYU3Rftw6` has **zero users**, no real repo onboarded, GitHub secret is a 32-char generated placeholder. Not created, per the step. Independently moot given F3. | -| 8.3 stack deletion | **DELETE_FAILED (billing stopped)** | Leaked AgentCore ENIs (F12). 4 zero-cost resources remain; all billable resources verified gone. Retry command recorded in 8.3. | -| Disk capacity / 32 GB disk claim | **NOT EXPOSED** | No disk quota in `service-quotas`, no disk field on image or version. Unverified. | - -### Elapsed and approximate cost - -**Elapsed:** 18:48Z → 23:07Z = **4 h 19 min** wall clock. Of that, ~1 h was the -suspend-TTL observation (run concurrently with the CLI phase, IAM checks, the -16 KB probes, and the second image build, per instructions — no idle waiting); -~1 h 15 min was consumed by the three blocked deploys plus rollbacks/redeploys; -~1 h 40 min was teardown retries. - -**Approximate cost: well under US$10, dominated by NAT gateways and VPC -endpoints, not by MicroVMs.** - -| Item | Quantity | Est. | -|---|---|---| -| NAT gateways (2 × $0.045/h) | ~2.3 h summed across 4 stack lifetimes | ~$0.21 + trivial data | -| Interface VPC endpoints (7 × $0.01/h × 2 AZ) | ~1.6 h | ~$0.22 | -| MicroVM runtime | VM1 ~3 min `RUNNING` + ~2 h 45 min `SUSPENDED`; VM2 ~3 min | < $0.50 (suspended compute is not billed; snapshot storage was < 3 h) | -| MicroVM image builds | 6 builds (2 versions × 2 chipsets × 2 images), ~6 min each | low single-digit $ at most | -| Snapshot/image storage | ~3.5 GiB × 2 images × < 3 h | negligible | -| AgentCore runtime | created 4×, never invoked | negligible | -| S3 / DynamoDB / Lambda / API GW / Cognito / Secrets / logs | brief, mostly idle | < $1 | -| ECR container asset (retained in bootstrap) | 630 MB stored | ~$0.06/month ongoing | - -The 8 h `maximumDurationInSeconds` worst case was never approached; both VMs were -explicitly terminated. - -### Deliberately left in place - -1. **`CDKToolkit`** — retained as the runbook instructs, now carrying - `ComputeTypes=agentcore,lambda-microvm` and the five ABCA least-privilege - policies **instead of `AdministratorAccess`**. ⚠️ This is a change to a - *shared* account: other CDK apps in `` now deploy through the - ABCA-scoped execution role. The original standard-bootstrap template is - captured at `$EVIDENCE_DIR/cdktoolkit-template-before.txt` if it needs - restoring. -2. **`backgroundagent-dev` in `DELETE_FAILED`** — VPC `vpc-0a12c1a64cc960c6a`, - two private subnets, one security group, and two leaked AgentCore ENIs. Zero - cost; retry `delete-stack` once AgentCore releases the ENIs. -3. **Bootstrap S3/ECR assets** — including the 630 MB agent container image, - normal bootstrap content. -4. **Service-vended log groups** — `/aws/bedrock-agentcore/runtimes/…` and - `/aws/lambda/backgroundagent-dev-…` created outside CloudFormation. - -Untouched and **not** created by this run: the pre-existing stacks -(`serverless-api-powertools`, `BuildingServerlessAPIs`, -`aws-sam-cli-managed-default`) and the two `available` NAT gateways in -`vpc-01c9984d163d2965e`. - -### Recommended follow-up before P1 merges - -F1, F2, F3, F4, F5, and F6 are each independently sufficient to make the -`lambda-microvm` backend non-functional. F1 (the `/ready` requirement) is the one -that changes the *shape* of the phase plan rather than a constant, so it should be -adjudicated first: either P1 grows a minimal `/ready` (and `/run`) responder, or -P1 ships a hook-less image and explicitly defers all payload delivery to P2. - -## Report back - -Attach or summarize the evidence paths, then provide the filled results table to -the orchestrator. Draft it as a comment for issue #645; **do not post it until -the orchestrator reviews it**. Highlight every `BLOCKED`, `INCONCLUSIVE`, -`SKIPPED`, and `DEFERRED-TO-P2-ENV` result and explicitly separate observed AWS -behavior from expectations inferred from the SDK model. diff --git a/docs/verification/645-p2-clean-deployment-20260913.md b/docs/verification/645-p2-clean-deployment-20260913.md deleted file mode 100644 index eb271e03f..000000000 --- a/docs/verification/645-p2-clean-deployment-20260913.md +++ /dev/null @@ -1,351 +0,0 @@ -# ADR-021 P2 clean deployment — 2026-09-13–14 - -Live verification of the P2 source fixes and deployment prerequisites from -[`645-p3-implementation-plan.md`](./645-p3-implementation-plan.md). - -**Status: infrastructure and managed image deployed.** CloudFormation reached -`UPDATE_COMPLETE`; image version `1.0` is `SUCCESSFUL` and `ACTIVE`. Build hooks -and authenticated API reads pass. The later -[live task verification](./645-p2-live-task-20260914.md) also passed normal-task, -PR-iteration and cancellation checks. Remaining P2 acceptance conditions are -listed there; the broad smoke warnings have not been discharged. - -The detailed chronology below records the infrastructure handoff at -2026-09-14 04:48 UTC. Repository/PAT setup and task execution occurred afterward -and are documented in the linked follow-up. - -## Target and isolation - -- Source: `fix/645-microvm-readiness`, deployed commit `29dcaa74`. - The initial runtime revision was `eb7e2071a3affa09ca0fd2f0064028545537d21f`; - the four deployment fixes below are included in the final source. -- AWS profile: `sphia-dev`; account: ``; Region: `us-west-2`. -- Application stack: `backgroundagent-dev`; bootstrap stack: `CDKToolkit`. -- Compute selection: `lambda-microvm`, with the existing AgentCore resources - still provisioned. Bootstrap compute policies: `agentcore,lambda-microvm`. - This provisions a backend; each repository separately chooses its backend. -- Managed base image: - `arn:aws:lambda:us-west-2:aws:microvm-image:al2023-1`, version `1`. -- Local evidence: `/tmp/abca-645-p2-clean-20260913`. - -The existing application and shared bootstrap in `us-east-1` are outside this -deployment. That Region already has five VPCs against a quota of five, and its -Bedrock invocation logging belongs to the existing development stack. No existing -VPC was deleted and no quota increase was requested. - -Before this run, `us-west-2` had no active CloudFormation stacks, one default VPC -(`vpc-053c23d0618a71e91`), and no Bedrock invocation logging configuration. -Using this Region permits the repository's normal stack name and bootstrap -qualifier without modifying another application's resources. - -## Local prerequisites - -Docker is available on native ARM64. The installed mise `2026.2.8` cannot parse -the root monorepo task configuration. This run uses the official macOS ARM64 -mise `2026.9.7` binary in the evidence directory's `tools/` folder, with its -SHA-256 checked against the GitHub release: -`3c3f377e7123a466274a20f01502ddd8c58f76028907f471c9bc42fbf83846e1`. -The global tool installation was not changed. - -The full root build runs with `JEST_MAX_WORKERS=2` and `MISE_JOBS=2`. -The final deployed source passed in 704.66 seconds; its output is retained as -`full-build-managed-image-nag.log`. - -**Final result: passed**, exit 0. CDK: 222 suites / 4,783 tests; -CLI: 62 suites / 928 tests; Python: 1,823 tests. Compilation, lint, formatting, -type checks, the 77-page documentation build and drift checks passed. Two CDK suites / 38 -DynamoDB Local integration cases were skipped because that local service was -not running; those cases passed during the preceding implementation session. - -Earlier builds are recorded below beside the fixes they validated. Target -synthesis was run after the final build and passed. Deployment cleanup must -not run concurrently with tests/synthesis using the shared CDK temporary files. - -## Acceptance record - -- [x] Full root build passes on deployed infrastructure source (`29dcaa74`). -- [x] Fresh least-privilege bootstrap has the required policies. -- [x] No-image application infrastructure deploys from source. -- [x] Current agent artifact is packaged and uploaded. -- [x] CloudFormation creates the managed image using the pinned base. -- [x] Image version is active; `/ready` and `/validate` return HTTP 200. -- [x] Coordinator image/network/role configuration and authenticated API reads are verified. -- [x] A normal task completes with live progress and heartbeat evidence (follow-up). -- [x] Runtime logs, successful/canceled-task payload cleanup and VM termination are verified (follow-up). -- [ ] Failure/recovery cleanup and automatic repository build/lint gates are verified. -- [ ] Relevant allowed/denied IAM and network cases are exercised. -- [x] Current resource state, temporary-login cleanup and remaining limits are recorded. - -Automatic P3 suspension remains disabled in this source revision. A successful -P2 deployment would not establish P3 lifecycle correctness. - -## Bootstrap and application synthesis - -`CDKToolkit` reached `CREATE_COMPLETE` with bootstrap version `32` and ABCA -policy bundle `1.6.0`. The original generated template was deployed through -CloudFormation so the custom `ComputeTypes=agentcore,lambda-microvm` parameter -could be supplied at creation. No generated template or live IAM policy was -patched. The reviewed change set contained 18 additions. - -At initial bootstrap 1.6.0 creation, the CloudFormation execution role had exactly -the five expected ABCA policies: -Infrastructure, Application, Observability, Compute-Agentcore and -Compute-LambdaMicrovms. Each deployed policy document equaled its source JSON. -There were no inline execution-role policies and no `AdministratorAccess`. -Bootstrap 1.7.0 adds the exact-role inline policy described below while -preserving those five managed policies. -Evidence: `bootstrap-policy-comparison.json`. - -The target-specific no-image synthesis passed for -`aws:///us-west-2`: 472 parent resources and a 693,882-byte parent -template. Registry resources occupy two nested stacks. Asset publication and -the application change-set preparation follow this reviewed cloud assembly. - -## First application attempt and source fix - -The agent container built and every code/image asset uploaded successfully. -Application change-set preparation then failed before resource creation: -CloudFormation's execution role lacked `iam:PassRole` on itself while validating -the registry nested stacks. Evidence: `substrate-prepare.log`. - -The source fix adds `PassExecutionRoleToCloudFormation` as a generated inline -policy, permitting only the exact execution role ARN and only the CloudFormation -service. Bootstrap bundle `1.7.0` is required. No live IAM patch was applied. - -The bootstrap hash also needed correction: its root-key JSON replacer omitted -nested Action, Effect, Resource and Condition changes. It now recursively sorts -keys, keeps those fields, and includes the new inline policy. - -Five new regressions failed on old source: the missing self-role policy and four -permission changes that left the old hash unchanged. After the fixes, the two -bootstrap suites pass all 58 tests, including object-key ordering and inline-policy -hash coverage. The generated template is 47,446 bytes on disk; the CDK CLI sends -46,307 characters, within both the budget and AWS's inline limit. - -The complete build passed again: 4,773 CDK tests, 928 CLI tests and 1,823 Python -tests; total runtime 362.54 seconds. The fix is committed as `16941828`. -Bootstrap 1.7.0 reached `UPDATE_COMPLETE`. Its actual inline policy equals the -resolved source policy, its five managed attachments are unchanged, and every -existing bootstrap parameter was preserved. AWS IAM simulation verified six -resource/service combinations, allowing only this role passed to CloudFormation. - -## Second application attempt: global names - -The application retry passed the self-PassRole check, then failed early -validation because `BackgroundAgent-Tasks-backgroundagent-dev` already exists. -CloudWatch dashboard names are account-global. The existing dashboard was last -modified on 2026-08-28 and was not changed by this run. - -An audit of the other global resources found a second collision before another -attempt: the screenshot CloudFront origin access control (OAC), -`backgroundagentdevGitHubScreOrigin1S3OriginAccessControl544B0DF8`, already exists -as `E7LV4IA65D59K`. The OAC controls how CloudFront authenticates requests to the -private screenshot bucket. It belongs to the existing deployment and must not -be deleted or repurposed. - -The source now gives the dashboard a Region suffix and gives each screenshot -OAC a bounded construct-path name plus Region. Four regressions failed the old -names; all 15 tests in the two construct suites now pass. Tests cover identical -stack names in two Regions, long names, distinct controls, and SigV4 signing. - -For existing deployments adopting this source, the dashboard and OAC resources -are replaced; the distribution references the new OAC. Dashboard bookmarks need -the new name. This run updates only the new Oregon deployment. - -The full build passed: 4,777 CDK tests, 928 CLI tests, 1,823 Python tests and -77 documentation pages; total runtime 358.95 seconds. The fix is committed as -`4ea7e88b`. Refreshed target synthesis passed with 472 parent resources. - -## Third application attempt: registry waiter - -The third change set passed validation and all 472 additions were reviewed -before execution. Actual creation failed in the registry nested stack: -CloudFormation generated the Step Functions name -`AgentRegistryProviderwaiterstatemachineE27177B2-SxOkWE4lO56U`, outside the -bootstrap policy's `backgroundagent-dev-*` namespace. The execution role therefore -denied `states:CreateStateMachine`. - -The registry construct now explicitly names this helper using the outer stack -name and a bounded construct-path hash. No bootstrap permission expansion is -needed. Four regressions failed on old source; the registry suite now passes -all 11 tests, including actual nested templates, long stack names, two distinct -registries, and comparison against the bootstrap ARN pattern. The full root -build passed in 715.27 seconds: 4,781 CDK tests, 928 CLI tests, 1,823 Python -tests and 77 documentation pages. The fix is committed as `edaa144c`. - -The target template now names the waiter -`backgroundagent-dev-AgentRegistryStack-AgentRegistry-Provider-77C0EB2A`. -IAM simulation using the actual deployed execution role allows its creation -and denies the old name. The registry nested template changes only that name; -the parent remains at 472 resources. Evidence: -`registry-waiter-iam-simulation.json` and `substrate-waiter-template-review.json`. - -Rollback initially could not delete the two MicroVM network connectors and -AgentCore memory while AWS was still creating them. A normal stack deletion -after they stabilized reached `DELETE_COMPLETE`, including the connectors, -memory and VPC. No forced deletion or VPC retention was used. This removed only -the new failed Oregon application; the Oregon bootstrap remains for the retry. - -Post-deletion inventory found one intentionally retained API Gateway logging -role and two regional logging settings. With no REST or HTTP APIs remaining, -the API Gateway setting was cleared and that exact failed-stack role deleted. -Bedrock invocation logging, still pointing to the deleted stack's log group and -role, was cleared too. Both settings now match the observed empty initial state. -Stack and VPC inventories now contain only the new bootstrap and the original -default VPC; existing resources in other Regions were not changed. - -## Fourth application attempt - -Change set `abca-645-p2-substrate-r4` passed AWS validation. Its 472 changes are -all additions, with the intended restricted CloudFormation execution role. -The proposal was executed for the new stack -`df112e20-afe6-11f1-963f-02fff900428b`. The stack and all 472 root resources -reached `CREATE_COMPLETE` at 2026-09-14 03:03 UTC. -Source fixes are `16941828`, `4ea7e88b` and `edaa144c` on top of the original -runtime revision recorded above. Bootstrap remains 1.7.0. - -The registry and its waiter both reached `CREATE_COMPLETE`. The registry ID is -`5v3pQEkpkvjdS4q6`; the waiter uses the explicit name above. This verifies the -previous failure is fixed during actual resource creation, not only simulation. -Evidence: `substrate-r4-registry-resources.json`. - -Both connectors are `ACTIVE`. `aws lambda-core` is the CLI namespace for network -connectors. Their actual security groups have no ingress rules; runtime egress -allows TCP 443, while build egress allows TCP 80 and 443. This verifies deployed -configuration; in-guest traffic tests remain outstanding. - -The DNS firewall is in the source's hardcoded observation mode. Live rules -allow baseline/additional domains at priorities 100/200 and use a catch-all -`ALERT` at priority 300. Unlisted domains are logged, not blocked; domain -allowlisting must not be claimed as an enforced isolation boundary here. -Evidence: `live-dns-firewall-rules.json`. - -## Agent artifact and API checks - -The official packaging script uploaded the artifact successfully. Downloading -that S3 object and comparing the 106 image inputs against the checkout verified -every byte. The ZIP is 1,046,220 bytes with SHA-256 -`e6f37030b0517374359927ac3c1480d85195fba37db8052ed339b9a34dcd94a1`. -Evidence: `uploaded-artifact-verification.json`. - -A temporary user in the new Cognito pool authenticated through the built CLI, -and an authenticated task list returned an empty result. Missing and invalid -tokens both returned HTTP 401. Invitation messages were suppressed. The CLI -used a separate private configuration, leaving the operator's default login -untouched. Authenticated listing passed again after the managed-image update. - -Platform doctor passes the API, Cognito, active-repo and model-catalog checks. -It fails only the GitHub token check: the new secret still contains its -placeholder. Repository choice and GitHub setup remain pending. Catalog -visibility alone does not prove runtime model invocation. -Evidence: `post-update-task-list.json` and `post-update-platform-doctor.json`. - -`bgagent runtime status` confirms that the only seeded repository, -`awslabs/agent-plugins`, still resolves to **AgentCore**, whose control plane is -`READY`. There are no repositories configured for MicroVM yet. Stack output -`ComputeSubstrate=lambda-microvm` describes provisioned infrastructure; it does -not change the seeded repository's selection. Onboard the chosen test repository -with `--compute-type lambda-microvm` before submitting a task. Evidence: -`post-update-runtime-status.json`. - -## Managed-image synthesis: overflow-policy check - -The first managed-image synth failed locally on one `AwsSolutions-IAM5` finding: -the orchestrator's Jira OAuth secret-prefix grant moved into `OverflowPolicy1` -when image lifecycle permissions enlarged the role's policy. Constructor-time -suppression metadata does not reach policies generated later during synthesis. -The source fix extends the existing overflow-policy Aspect only for that role -and that resource pattern. Both production-entry-point regression cases pass, -including an unrelated wildcard grant that still triggers an error. The fix is -committed as `29dcaa74`; the final full-build result is recorded above. - -The corrected managed-image synth passed with 474 parent resources and a -697,733-byte template. Compared with the failed synth, all IAM Role, Policy -and ManagedPolicy resource properties are identical: the fix changes the narrow -security-check exception metadata, not permissions. Evidence: -`managed-image-template-review.json` and `managed-image-synth-r2.log`. - -## Managed-image deployment and live checks - -Change set `abca-645-p2-managed-image-r1` contained 42 changes: four additions, -36 modifications and two removals. The removals were old immutable Lambda and -guardrail versions. No existing data bucket, table or VPC was replaced. -CloudFormation executed the update successfully; `UPDATE_COMPLETE` was observed -at 2026-09-14 04:39:46 UTC. The stack has 474 root resources. - -| Item | Observed result | -|---|---| -| Stack | `backgroundagent-dev`, ID suffix `df112e20-afe6-11f1-963f-02fff900428b`, `UPDATE_COMPLETE` | -| Image | `arn:aws:lambda:us-west-2::microvm-image:backgroundagent-dev-abca-agent` | -| Image record | `CREATED`, `latestActiveImageVersion=1.0` | -| Image version `1.0` | Build state `SUCCESSFUL`, status `ACTIVE` | -| Base | `al2023-1`; input version `1` is reported by AWS as `1.0` | -| Configuration | `ARM_64`, 8,192 MiB minimum memory, hook port 8080 | -| Build hooks | Ready enabled / 300 seconds; validate enabled / 60 seconds | -| Runtime hooks | Run enabled / 60 seconds; terminate enabled / 15 seconds; no suspend/resume hooks | -| CloudWatch logs | `/aws/lambda-microvms/backgroundagent-dev-abca-agent` | -| Active task VMs | None; no coding task has been submitted | - -The actual `/ready` log records warm-up of Claude CLI 2.1.191, git 2.47.3 and -Node 24.21.0, followed by HTTP 200. `/validate` reports Python 3.13.13, -13 supported platform configuration keys and zero validator warnings, followed -by HTTP 200. Some generic Uvicorn “Invalid HTTP request received” warnings -precede validation; their cause was not established. These build-hook results -do not exercise runtime AWS credentials, model invocation or task execution. - -The live coordinator alias points to version `2`, with Lambda state `Active` -and last update `Successful`. Its actual environment includes the managed image -ARN, the runtime execution role, runtime egress connector, AWS `NO_INGRESS` -connector and the dedicated payload bucket. No image-version override is set, -so launch resolves the latest active version. - -Evidence: `managed-image-change-set-r1.json`, `managed-image-r1-state.json`, -`managed-image-r1-resources.json`, `final-managed-image.json`, -`final-managed-image-version.json`, `final-orchestrator-configuration.json`, -`final-microvms.json` and `image-build-log-snapshot.json`. - -## Retained deployment, cleanup and next verification - -The successful application, managed image, artifact and bootstrap remain live. -The default Oregon VPC and existing deployment in other Regions were preserved. -The failed earlier application was fully removed as described above. -The final read at 2026-09-14 04:48:54 UTC confirms `UPDATE_COMPLETE`, no task -MicroVMs and zero objects in the payload bucket. Since no task ran, an empty -bucket does not prove the cleanup path works. Evidence: -`final-deployment-status.json`. - -After the final authenticated API check, the temporary Cognito user -`p2-verification-20260913@example.invalid` was deleted from the new pool. -`AdminGetUser` confirmed `UserNotFoundException`. Its three private request files -and cached CLI credentials were removed. The non-secret isolated CLI -configuration and deployment evidence remain. Evidence: -`verification-user-cleanup.json`. - -At infrastructure handoff, the planned verification sequence was the following. -The [live task record](./645-p2-live-task-20260914.md) records the completed -steps and the remaining limits: - -1. Select the test repository and populate the new deployment's GitHub token - using `bgagent github set-token --region us-west-2 --stack-name backgroundagent-dev`. - No GitHub token was copied and no repository, branch or PR was created by this run. -2. Onboard that repository with `bgagent repo onboard OWNER/REPO --compute-type - lambda-microvm --region us-west-2 --stack-name backgroundagent-dev`. Confirm - its effective backend with `bgagent runtime status --repo OWNER/REPO`. - Use this deployment's isolated CLI configuration and an authorized login. -3. Submit the normal clone/change/test/PR task from the P2 runbook and capture - progress, heartbeat, model invocation, Memory writes and logs while the VM - is running. The image build and empty API list are not substitutes. -4. Verify success, failure and cancellation cleanup, including S3 payload - deletion and service-reported VM termination. Exercise the pending IAM and - in-guest network positive/negative cases. The DNS observation-mode limit - remains applicable. -5. Record those results before discharging P2 warnings or enabling automatic - P3 suspension. - -The rebuild follow-up identified here is now resolved in the -[managed image update record](./645-microvm-image-rebuild-20260914.md): -overwriting the fixed S3 key did not change CloudFormation image properties. -Commit `e1d5debe` adds immutable hash-suffixed artifacts and requires their digest -in deployment context. A normal update built and activated image `2.0`; repeat -packaging reused the verified object and a same-assembly deployment made no -changes. Packaging and passing the printed digest remain explicit operator steps. diff --git a/docs/verification/645-p2-live-task-20260914.md b/docs/verification/645-p2-live-task-20260914.md deleted file mode 100644 index f1bbe4ae3..000000000 --- a/docs/verification/645-p2-live-task-20260914.md +++ /dev/null @@ -1,219 +0,0 @@ -# ADR-021 P2 live task verification — 2026-09-14 - -Follow-up to the [clean infrastructure and image deployment](./645-p2-clean-deployment-20260913.md). - -**Result:** a normal coding task and a PR-update task completed on managed -MicroVM image `1.0`. Both ran the repository's npm checks successfully, wrote -Memory events and terminated automatically. A separate cancellation test -removed its own VM, payload and capacity reservation while the other task -continued running. - -The reviewed result is [isadeks/vercel-abca-linear PR #584](https://github.com/isadeks/vercel-abca-linear/pull/584), -open and unmerged at commit `fd509fa63fa089df356574bdd654123c8621b12e`. -Only `README.md` changes. - -This verifies the observed success, PR-iteration and cancellation paths. -It does **not** discharge every P2 warning. The later -[configuration follow-up](./645-p2-repository-config-20260914.md) verifies real -automatic build/lint commands and a worker-reported failure cleanup; additional -failure/recovery cases and the wider IAM/network matrix remain outstanding. -The temporary CLI configuration was subsequently removed at user request; -repository mise tasks are still needed for the restored defaults. -Automatic P3 suspension remains disabled. - -## Environment and repository setup - -- AWS profile/account/Region: `sphia-dev` / `` / `us-west-2`. -- Stack: `backgroundagent-dev`; bootstrap bundle: `1.7.0`. -- Deployed source: `29dcaa74`; no runtime, image or IAM changes were needed for - these tests. The later local commits contain verification documentation. -- Image: `arn:aws:lambda:us-west-2::microvm-image:backgroundagent-dev-abca-agent`, - version `1.0`, `ACTIVE`. -- User-selected repository: `isadeks/vercel-abca-linear`. -- Local evidence: `/tmp/abca-645-p2-clean-20260913`. - -Oregon was chosen because the existing east deployment already occupied its -regional resources and `us-east-1` had five VPCs against a quota of five. This -run preserved that deployment and its resources. - -The existing effective PAT for the selected repository was checked without -printing its value. GitHub `/user` and `/repos/isadeks/vercel-abca-linear` both -returned HTTP 200 as `isadeks`, with repository push permission. GitHub reported -expiration **2026-09-27 12:52:45 UTC**. The verified value was copied into the -new west deployment's still-empty platform secret, using the built CLI helper. -Read-back equality passed; the east secret was not modified. - -The built CLI onboarded the selected repository with -`--compute-type lambda-microvm`. Runtime status confirmed that selection, and -platform doctor passed every check, including MicroVM service availability. -The previously seeded `awslabs/agent-plugins` repository retained its AgentCore -selection. - -Evidence: `selected-repository-pat-check.json`, `github-token-setup.json`, -`smoke-runtime-status.json` and `smoke-platform-doctor.json`. - -## Normal task and review correction - -| Run | Task ID | Workflow | Outcome | -|---|---|---|---| -| Initial README task | `01M2FYAHE8TDF8JZ9EB432F3SX` | `coding/new-task-v1@1.0.0` | `COMPLETED`, created PR #584 | -| README review correction | `01M2FYS5DVZH60J9TJWX3ZE9CA` | `coding/pr-iteration-v1@1.0.0` | `COMPLETED`, updated the same PR | -| Cancellation canary | `01M2FYV0SVT4YP3VKFDWTB3S7R` | `coding/new-task-v1@1.0.0` | `CANCELLED`, VM and payload cleaned up | - -The first task preserved the project title and documented `npm ci`, -`npm run lint` and `npm test`. It used a 40-turn / $5 limit. The actual VM -`microvm-86421c36-72e8-38cb-9cb3-b5f29222f497` reached `RUNNING` on image `1.0`. -The `/run` hook returned HTTP 200, scoped tenant credentials initialized, -repository cloning succeeded and heartbeats advanced throughout execution. - -Claude Opus 5 executed tools, changed only the README and pushed commit -`20c0717eec36b858b12a40a0b43cb13ea651feb1`. The task completed at -12:34:21 UTC with reported duration 202.3 seconds, 17 turns and model cost -$0.2655996. These duration/cost fields are the task's reported metrics, not -total AWS infrastructure cost. - -Review then caught two incorrect claims: `api/_lib/` contains only a design -README, so the booking API modules are planned rather than implemented; and -there is no repository CI workflow automatically running the npm checks. -The second task corrected those statements through the normal PR-iteration -workflow, with a 30-turn / $3 limit. - -Its VM was `microvm-d552daeb-dbd6-3300-8b2a-1ef7fae46429`. The task completed at -12:41:09 UTC with reported duration 143.2 seconds, nine turns and model cost -$0.2111514. The final diff correctly describes the current static site and -planned API. The PR description was shortened after review to state the final -change, actual validation and configuration limitation. Vercel checks passed -on the final head. - -For both runs, the explicit npm commands exited 0. The correction run's -CloudWatch tool result records: - -```text -npm ci exit=0 npm run lint exit=0 npm test exit=0 -``` - -Vitest ran one baseline placeholder test. That verifies the shipped test -command ran successfully; it does not establish application feature coverage. -`bgagent watch` observed the first task through completion and exited 0. - -Evidence: `smoke-task-{request,submit}.json`, -`smoke-review-task-{request,submit}.json`, `smoke-watch.ndjson`, -`smoke-observations.ndjson`, `smoke-review-validation-evidence.json`, -`smoke-pr-final-diff.patch` and the timestamped task/VM/log snapshots. - -## Automatic build/lint configuration in the original runs - -**Follow-up:** the [repository configuration record](./645-p2-repository-config-20260914.md) -records all four automatic pre/post npm checks passing under temporary overrides. -Those overrides and the CLI addition were subsequently removed at user request; -repository mise tasks are still needed. The findings below describe the original -two runs before the temporary configuration was applied. - -The default blueprint checks are `mise run build` and `mise run lint`. -This repository has no mise tasks. The agent's explicit npm checks passed, -but the platform's automatic checks still used those defaults. - -Both task records therefore show `build_passed=true` and `lint_passed=false`. -The build result reflects the existing inert/default-build behavior; it must -not be presented as a successful project build. The lint result reflects the -failing default mise command, not a failing `npm run lint`. - -Configure suitable `pipeline.buildCommand` / `pipeline.lintCommand` values on -the repository's canonical Blueprint, including dependency installation where -needed, then verify those automatic gates on another normal task. The runtime -onboarding command used here does not configure those fields. No placeholder -mise task, no-op build or unrelated repository change was added to hide the -mismatch. - -## Logging and Memory evidence - -Two separate logging paths were observed while the initial VM was running: - -1. Service/runtime logs in `/aws/lambda-microvms/backgroundagent-dev-abca-agent`, - under the VM's versioned log stream. -2. Direct server writes to - `/aws/vendedlogs/bedrock-agentcore/runtime/APPLICATION_LOGS/backgroundagent-dev`, - stream `server_debug/01M2FYAHE8TDF8JZ9EB432F3SX`. - -The second path contained configuration/thread/run-acceptance records. It -exercises actual application `CreateLogStream` / `PutLogEvents` calls under -the deployed runtime credentials, beyond checking logs after shutdown. - -AgentCore `ListEvents` independently returned: - -- Initial task: a `task_episode` and `repo_learnings` event, timestamp - 12:34:21 UTC, with schema version 3 and content hashes. -- Correction task: a `task_episode` event, timestamp 12:41:09 UTC, referencing - the same PR. - -These records verify Memory writes. The correction's hydration event reported -`has_memory_context=false`; this run does not establish successful retrieval -of prior Memory context. - -Evidence: `smoke-application-logs-running.json`, `smoke-memory-events.json`, -`smoke-review-memory-events.json` and `smoke-review-events.json`. - -## Termination, payload deletion and cancellation - -For the first successful task, S3 initially contained its `payload.json`, -private `launch.json` and the shared bootstrap manifest. After automatic -finalization, both task files were gone. The service reported `TERMINATED`, -and the `/terminate` hook returned HTTP 200 at 12:34:35 UTC with zero active -pipeline threads. - -The correction run also terminated automatically. Its hook returned HTTP 200 -at 12:41:31 UTC with zero active pipeline threads. An empty `microvm_id` in -these terminate records matches the service behavior already documented in -`server.py`; the versioned log stream and service state provide the VM identity. -The hook acknowledgement is not a P3 suspend durability barrier. - -The cancellation canary requested no coding or GitHub publication work and -used a one-turn / $0.05 limit. At 12:39:32 UTC: - -- Its VM `microvm-1a9f0024-2a4b-3b80-a172-b6b93c35af77` was `RUNNING`. -- The correction VM was also `RUNNING`. -- A strongly consistent read showed the user's `active_count=2`. - -Normal CLI cancellation returned `CANCELLED` at 12:39:33.958 UTC. By the -12:40:24 UTC verification, its VM was `TERMINATED`, its S3 task prefix was -empty, and its reservation was `released`. The count was **1**, with the -correction VM still running. That observed cancellation did not release the -other task's seat. - -After the correction completed, `active_count=0`, all three VMs were -`TERMINATED`, and only the shared bootstrap manifest remained in S3. -That manifest intentionally expires through the one-day lifecycle rule; -it is not a retained task payload. - -Evidence: `smoke-payload-objects-{running,after}.json`, -`smoke-cancel-{before,response,after}.json`, `smoke-final-counter.json`, -`smoke-final-payload-inventory.json` and `smoke-observations.ndjson`. - -## Handoff and remaining gates - -The temporary Cognito user `p2-smoke-20260914@example.invalid` was deleted after -verification; `AdminGetUser` confirmed `UserNotFoundException`. Its cached CLI -tokens were removed. The generated password was never saved or sent in an -invitation. Evidence: `smoke-verification-user-cleanup.json`. - -The intended deployment, repository/PAT configuration, open PR, task audit -records and non-secret local evidence remain. No application code, IAM policy, -agent image or resource in the existing east deployment was modified. - -- [x] Normal task and PR iteration complete with live heartbeat/model/tool evidence. -- [x] Real npm checks pass and the final README/PR is reviewed. -- [x] Application logging works while running; actual Memory events are stored. -- [x] Successful and canceled tasks terminate and delete their task payloads. -- [x] Cancellation preserves the other task's seat; final active count is zero. -- [x] Verify automatic npm build/lint checks under temporary overrides — see the [follow-up](./645-p2-repository-config-20260914.md). -- [ ] Make repository mise tasks available and verify default build/lint commands after withdrawal of the CLI configuration. -- [ ] Exercise failure cleanup, lost/replayed start responses and recovery cases. -- [ ] Exercise the full real-session IAM and in-guest/public-ingress network matrix. -- [ ] Verify Memory retrieval if claiming that behavior. -- [ ] Complete the separate P3 hooks, credential/durability barriers, supervisor - integration and live suspend/resume matrix. - -The clean deployment's DNS firewall remains in observation mode: unlisted -domains are logged rather than blocked. The code-only managed-image rebuild -trigger follow-up also remains open. None of these remaining conditions is -discharged by the successful tasks above. diff --git a/docs/verification/645-p2-payload-live-20260914.md b/docs/verification/645-p2-payload-live-20260914.md deleted file mode 100644 index 2b0e22ff9..000000000 --- a/docs/verification/645-p2-payload-live-20260914.md +++ /dev/null @@ -1,199 +0,0 @@ -# P2 live MicroVM payload verification - -## Result - -On 2026-09-14, **11 distinct cases passed against the deployed MicroVM image -`2.0`**. They verify delivery of synthetic task instructions, rejection of -invalid instructions/download references, S3 URL expiry and revocation, and -immediate identical-request replay. All **12 disposable workers terminated** -and all **29 synthetic file locations were confirmed absent** after cleanup. - -This is a subset of P2 acceptance. P3 sleep/wake remains unfinished. The -[implementation checklist](./645-p3-implementation-plan.md) tracks the remaining -work; the [bootstrap runbook](./645-payload-bootstrap.md) explains the protocol. - -## What the test does - -A worker needs two things before starting: a trusted settings file and a -temporary download ticket for its task instructions. The settings file is the -**manifest**; the ticket is a **presigned URL**. A bucket is an S3 storage -container, and an object is a file inside it. - -The [live runner](../../cdk/test/live/verify-microvm-payload.live.ts) uses the -real TypeScript producer to create both files and the saved launch reference. -It then starts a disposable worker through the real AWS `RunMicrovm` API. -The worker's `/run` hook is its startup check. - -Every case uses deliberately incomplete settings: -`{log_group_name: }`. When delivery succeeds, the startup -check reaches required-settings validation and returns HTTP **400** before -installing configuration or starting the coding pipeline. That expected -rejection is the positive control: it proves the files were fetched and -validated without making repository changes. - -The producer runs with the operator's AWS credentials. The worker uses the -deployment's unchanged execution role and runtime network connector, with -explicit `NO_INGRESS`. No shell connector, new image, IAM change, application -deployment, task row or capacity reservation is involved. - -## Deployment identity - -| Item | Value | -|---|---| -| Account / profile / Region | `` / `sphia-dev` / `us-west-2` | -| Stack | `backgroundagent-dev` | -| Deployed source | `e1d5debe`; the verifier was added on top of documentation commit `9bceb764` | -| Bootstrap policy bundle | `1.7.0` | -| Image | `backgroundagent-dev-abca-agent`, version `2.0` | -| Execution role | `backgroundagent-dev-LambdaMicrovmComputeExecutionRo-wYoHnFJEUj9S` | -| Runtime egress connector | `nc-e1601dbe-8821-45b8-bdee-3a59e7393031` | -| Ingress | AWS `NO_INGRESS` | -| Worker lifetime cap | 180 seconds per probe | -| Worker log group | `/aws/lambda-microvms/backgroundagent-dev-abca-agent` | - -The [image rebuild record](./645-microvm-image-rebuild-20260914.md) identifies -the deployed artifact and its full-build validation. - -## Observed cases - -All rows below passed the expected diagnostic and HTTP-status assertions. -HTTP 400 means the startup input was rejected; HTTP 500 here means the worker -could not read or authenticate the supplied files. These are expected test -outcomes. - -| Case | Observed startup result | HTTP | -|---|---|---| -| Valid transport | Manifest and signed task download passed; deliberately missing required settings rejected | 400 | -| Wrong task | Downloaded document did not belong to the referenced task | 400 | -| Wrong configuration | Same-account, other-workspace secret address did not match the authenticated manifest | 400 | -| Invalid JSON | Stored task bytes could not be parsed as JSON | 500 | -| Changed signature | S3 rejected the download with HTTP 403 | 500 | -| Expired URL | A real one-second URL expired; both operator control and worker received S3 HTTP 403 | 500 | -| Revoked URL | Production deletion helper removed both task files; signed download received S3 HTTP 404 | 500 | -| Foreign manifest | Same valid manifest bytes in the private artifact bucket could be read by the operator; worker received `AccessDenied` | 500 | -| Wrong download path | Reference did not sign this task's exact `payload.json` path; rejected before download | 400 | -| Corrupted manifest bytes | Appended whitespace left valid JSON but broke the SHA-256 fingerprint in its filename | 500 | -| Large task | **1,048,865 stored bytes** traveled through an **825-byte** hook reference and reached required-settings validation | 400 | - -Each case also exercised two concurrent producer preparations, a later -identical preparation, and changed instructions. The preparations returned -the same saved reference; changed instructions raised -`PAYLOAD_BOOTSTRAP_CONFLICT` without overwriting the original object. -The original signed URL returned HTTP 200 and the exact stored bytes before -fault injection. These observations exercise real S3 conditional writes with -operator credentials. - -Each passing case immediately repeated the exact `RunMicrovm` request with -the same client token. AWS returned the same worker ID. This proves immediate -identical-request replay only. - -The first foreign-manifest run correctly returned `AccessDenied`, but the -runner initially expected the generic exception name `ClientError`. That -assertion timed out and the batch exited 1 after cleanup. Boto3 exposes an -`AccessDenied` subclass here. The corrected expectation passed in a second -worker; this accounts for 12 workers across 11 distinct cases. - -## Permissions, logs and cleanup - -The live execution-role policy was read without modification. It contains: - -- `Allow s3:GetObject` on this deployment's payload-bucket `bootstrap/*`. -- `Deny s3:GetObject*` with `NotResource` set to that same prefix. -- `Deny s3:List*` on the payload bucket. - -The foreign-manifest case exercises the real worker's rejection of another -private bucket. The broader effective-permission matrix below remains open. - -The runner verifies rejection before configuration installation and pipeline -acceptance, and checks diagnostics for signed-URL credential/signature markers. -A final independent read of all 12 terminated workers' log streams found three -events per worker, no such markers, and no configuration-installed or -pipeline-accepted message. - -Cleanup uses the production task-file deletion helper, followed by operator -deletion of exact synthetic keys, including the run's unique manifests. -The revoked-URL case verifies both task files are absent **before** that -fallback. These direct probes do not exercise automatic coordinator -finalization. - -Fresh AWS reads confirmed all 12 owned worker IDs were `TERMINATED` and all -29 exact object locations returned not found. The payload bucket has -versioning disabled. The stack remained `UPDATE_COMPLETE`, with its last -update still `2026-09-14T16:07:52.421Z`; the latest active image remained `2.0`. -The test retains the intended development deployment. - -## Reproduce - -Use the repository's installed dependencies, Node 22, AWS CLI and credentials -for the intended development account. The output directory must not already -exist. The script makes no AWS calls unless `--execute` is present. - -```bash -cd cdk -mise exec -- npx tsx test/live/verify-microvm-payload.live.ts - -AWS_PROFILE=sphia-dev mise exec -- npx tsx test/live/verify-microvm-payload.live.ts \ - --execute \ - --account \ - --region us-west-2 \ - --stack backgroundagent-dev \ - --image-version 2.0 \ - --output /tmp/abca-microvm-payload-new-run -``` - -Optional `--cases` accepts comma-separated names shown by the dry run. -The script checks the actual account, stack, active image version and deployed -`NO_INGRESS` configuration before creating fixtures. Its output records worker -IDs, AWS request IDs, request fingerprints, statuses and cleanup coordinates; -signed URLs and request bodies stay in memory. - -The output directory is private. If the process is interrupted, use its -`context.json` exact `plannedCleanupObjects` and recorded worker IDs to finish -cleanup; do not delete whole buckets or prefixes. The 180-second service -lifetime bounds workers even if the process dies, but does not delete S3 files. -This is an operator-run live verifier, outside the normal Jest test pattern. - -Validation: ESLint, focused strict TypeScript checking and dry run passed. -The live batches used `--cases valid-transport`, then the seven original -negative cases, then -`--cases foreign-manifest,wrong-path,bad-manifest-digest,large-payload`. -The second batch's harness-only failure and corrected rerun are recorded above. - -## Evidence index - -Private evidence root: `/tmp/abca-645-p2-clean-20260913`. - -| Directory | Run ID | Result / cleanup | -|---|---|---| -| `payload-live-canary-20260914` | `p2-bootstrap-14632317-f022-4cb5-aad6-be45e6cb0144` | Control passed; 1 worker / 3 object locations | -| `payload-live-failures-20260914` | `p2-bootstrap-ff822345-11fc-47a7-8fb1-17bc46ed8b89` | Six assertions passed; foreign diagnostic misasserted; 7 workers / 16 locations cleaned | -| `payload-live-extended-20260914` | `p2-bootstrap-62ee9df3-eacf-4338-a43e-88999f66e3ee` | Four passed; 4 workers / 10 object locations | - -Each directory contains `context.json`, `results.json` and per-worker logs for -passing assertions. `results.json` links each case to its worker and Run/S3 -request IDs. The first foreign worker's actual rejection is separately retained -in `payload-live-foreign-investigation-logs.json` and its state file. -`payload-live-worker-s3-policy.json`, `payload-live-final-audit.json` and -`payload-live-final-log-audit.json` retain the policy and independent audits. - -For example, the passing foreign-manifest worker is -`microvm-9b27a667-8780-3439-91b0-55f8e00bf1a4`, Run request -`e0a364de-8178-4cf6-9567-004e4e6f24d3`. The large-payload worker is -`microvm-080080da-8637-3d8d-8bf7-eb65e25dface`, Run request -`fdb09b32-a2d7-4c2c-8a95-bab0fdd0be9a`. - -## Still required - -- Effective worker/session-role tests for listing, task/launch reads, - writes/deletes, public-bucket resource-policy grants, task-table restrictions - and transactions; equivalent ECS delivery and authorization cases. -- Expired **signer credentials**, coordinated upgrade/rollback, lost committed - S3 replies and coordinator restart under the deployed coordinator role. -- Lost Run replies, simultaneous or changed requests with the same token, - delayed replay/retention, recovery after process death, and unknown-ID cleanup. -- Coordinator terminal classification and automatic cleanup after rejected - startup hooks, plus injected cleanup failures and capacity migration/repair. -- Active public-ingress and port-denial attempts, remote MCP connectivity, and - end-to-end registry/tool use. The large case proves byte transport only. -- P3 guest suspend/resume hooks, safe credential refresh and durable state, - supervisor/approval integration, bounded recovery and live sleep/wake. diff --git a/docs/verification/645-p2-repository-config-20260914.md b/docs/verification/645-p2-repository-config-20260914.md deleted file mode 100644 index ea59f399e..000000000 --- a/docs/verification/645-p2-repository-config-20260914.md +++ /dev/null @@ -1,147 +0,0 @@ -# ADR-021 repository verification configuration and phase audit - -Date: 2026-09-14. Follows the [clean deployment](./645-p2-clean-deployment-20260913.md) -and [first live tasks](./645-p2-live-task-20260914.md). The user selected -`isadeks/vercel-abca-linear` and authorized completing its configuration. - -**Current status — withdrawn at user request:** commit `4e616679` reverts the -entire CLI addition `62cce37d`. At 13:29:11 UTC, the two live command overrides -were also removed from the west repository row using a conditional update. -A strongly consistent read confirmed their absence and retained -`lambda-microvm` routing. Future tasks use the existing `mise run build` / -`mise run lint` defaults. The remote repository still has no `mise.toml`, so -verification setup is open again. The results below preserve the earlier test -evidence; they do not describe the current configuration. - -## Historical configuration - -The selected repository has npm lint/test scripts but no mise tasks. All three -compute backends use the same worker defaults (`mise run build` and -`mise run lint`), and all accept per-repository command overrides. - -The temporary operator CLI exposed those existing settings and was used to set -`build_command = npm ci && npm test` and `lint_command = npm run lint` for -this repository in `us-west-2`, stack `backgroundagent-dev`. - -The configuration was applied to account ``. A separate -`repo show` confirmed both effective commands and `lambda-microvm` routing. -The east deployment was not changed. No infrastructure or image update is -required: the deployed coordinator already forwards these fields to the worker. -The build command installs the pinned dependencies before running tests, so the -following lint command has its tools available. - -The temporary CLI also preserved existing commands on re-onboarding and displayed -their effective values. Its original replacement row silently omitted both fields. -Two regressions failed before the preservation fix. Tests covered active and -removed repositories, backend changes, explicit overrides, empty-string resets, -fresh defaults, command parsing and text/JSON display. CLI compile/lint and all -**62 suites / 938 tests** passed on commit `62cce37d`. The full revert also removes -that preservation fix; the known omission remains a separate follow-up, not a -completed fix on the current branch. - -For a repository managed through CDK, put the same values in -`Blueprint.pipeline.buildCommand` / `lintCommand`; CLI changes alone do not edit -that source. This test repository was onboarded through the operator CLI. - -## Local mise alternative - -A local checkout of PR #584 contains commit `7bf72d3`, adding `mise.toml` and -updating README. The tasks share an `install` dependency (`npm ci`); `build` -runs `npm test`, and `lint` runs `npm run lint`. `mise run build ::: lint` -passed, including the repository's one baseline test. It does not compile site -assets or establish functional application coverage. - -Code Defender rejected the push because it considers the public repository -unapproved. The hook was not bypassed and the commit was not published through -another channel. PR #584 remains at its previously published README-only head. -The historical RepoTable configuration worked independently of that local file; -it is now removed. No publication was attempted as part of the CLI revert. - -`npm ci` reported nine existing dependency vulnerabilities (three moderate, -five high and one critical). Dependencies and the lockfile were not changed by -this configuration work; successful lint/tests are not a security audit. - -## Live verification - -An initial PR-review attempt, `01M2G0SN8CN8H2WJBK70S5GC60`, failed during hydration -because Bedrock Guardrails classified its PR context as -`CONTENT/PROMPT_ATTACK (MEDIUM)`. It had no session ID and launched no MicroVM. -No guardrail settings were changed. This is not runtime verification. The -platform's notification handler nevertheless posted a failed-task status comment -to PR #584 before agent execution; a read-only agent instruction does not disable -those platform notifications. The PR's source head remains -`fd509fa63fa089df356574bdd654123c8621b12e`. - -A separate default-branch inspection, `01M2G0VC4TY37SJ7GGE2JQ7YXF`, was submitted -at 13:14:06 UTC with a 12-turn / $2 limit. The request keeps files and GitHub -records unchanged. It launched -`microvm-f858bc9f-1235-3052-9792-2d9b263cd33b` using managed image `1.0`. -Live logs already confirm the automatic pre-agent build and lint commands -returned `OK` at 13:14:22 UTC. Automatic post-agent build and lint also returned -`OK` at 13:15:56 UTC. These are the platform's own four verification invocations, -not just commands the model chose to run. The task stored `build_passed=true` -and `lint_passed=true`. - -The overall task ended **FAILED** at 13:15:58 UTC: `coding/new-task-v1` requires a -commit/PR, and the inspection intentionally produced neither. Its error records -`agent_status=success, deliverable=lost`. This is evidence that the tested -configuration worked and a worker-reported failure was finalized, **not** another successful -coding/PR smoke test. The earlier successful coding/PR tests remain the evidence -for that path. Reported model cost was $0.14489115 and duration 95.1 seconds. - -Cleanup was verified independently: - -- Service state `TERMINATED`; no active MicroVMs remained in the region. -- `/terminate` returned 200 at 13:16:09 UTC with zero active pipeline threads. -- The task's payload prefix was empty. -- The task-owned reservation was `released` at 13:16:09.292 UTC, and the - temporary user's strongly read counter was zero. -- The temporary Cognito user was deleted after checking its recorded subject ID; - a follow-up read returned `UserNotFoundException` at 13:17:11 UTC. The isolated - cached credentials were removed. - -This failure-path observation does not cover a crashed worker, rejected run hook, -lost start reply, or failed cleanup API. Those need their own cases. - -## P2/P3 completion audit - -| Area | Current evidence | Remaining work | -|---|---|---| -| P1 start/poll/stop | Merged; exercised again by the clean deployment | Keep service-fact limits in the original runbook explicit | -| P2 managed build and normal work | Clean bootstrap/image deployment; coding, PR iteration, Memory writes, live logs, cancellation and a worker-reported failure cleanup observed | Complete failure/recovery and effective permission/network matrix | -| Repository command configuration | Temporary npm overrides passed all four pre/post checks; CLI addition and live overrides subsequently removed at user request | Repository mise tasks must be available before the restored default commands can pass | -| #817 review fixes | Error classification, deletion grants, trusted configuration, malformed-byte handling and comment fixes on takeover branch | Effective-role and ingress negatives; upstream review/merge | -| #700 task-scoped payloads | v2 signed references and deployment manifests implemented; real launches work | Cross-task/public-object denials, expiry, conditional-write and coordinated migration cases | -| #818 registry portability | Large-payload transport/local loader tests; shared HTTPS/443 limits documented | Real remote-tool DNS/TLS/auth/connectivity | -| #841 thread isolation | Regression and Python suite passed | Upstream merge and requested downstream #680 build confirmation | -| #810 logging diagnostics | Unused counter removed; structured failure events tested; normal live logs work | Do not equate normal log delivery with injected live writer-failure evidence | -| Start/capacity/metadata protocols | Local race/replay tests; normal live reservation release and cancellation work | AWS replay/token semantics, unknown-start recovery, effective session permissions, migration/drain and scan-scale checks | -| Managed image updates | First image build succeeded | Make code-only artifact changes trigger a managed-image rebuild and verify it | -| Optional nesting / #857 | Offline prototype only; clean deployment uses current root layout | Production split and full feature/migration validation if adopted | -| P3 sleep/wake | Strategy methods, persistent intent, policy helper and original-deadline handling implemented | Guest hooks, credential/durability barriers, bounded coordinator recovery, approval wiring, scoped grants, compatible-image control and live lifecycle matrix | - -All seven inspected upstream issues (#645, #817, #700, #818, #841, #857 and #810) -remain open. A local fix or one positive AWS run does not establish upstream -completion. ADR-021 defines P1–P3; it has no P4. - -The ordered [P3 implementation plan](./645-p3-implementation-plan.md) remains the -working checklist. Automatic suspension remains disabled. P3 is not complete. - -## Evidence - -Private evidence is retained under `/tmp/abca-645-p2-clean-20260913/`, including -CLI/docs build logs, redacted configuration output, task requests and timestamped -AWS/task/log observations. `config-command-execution-evidence.json` contains the -four commands and their successful completion lines; `config-final-*.json` and -`config-verification-user-cleanup.json` record cleanup. The documentation sync -and **77-page** site build also passed. - -Rollback evidence is in `cli-revert-config-{before,update,after}.json`. The -update required the old command values and timestamp to match before removing -only those two settings and refreshing `updated_at`. No test task was launched -for the revert, and the existing deployment/image and east environment were -unchanged. - -After the revert, CLI compile/lint and **62 suites / 928 tests** passed. The -entire `cli/` tree matches its pre-addition state at `396a31e0`, and the rebuilt -`repo onboard --help` no longer includes either new command flag. diff --git a/docs/verification/645-p2-smoke-runbook.md b/docs/verification/645-p2-smoke-runbook.md deleted file mode 100644 index d7d94e4e5..000000000 --- a/docs/verification/645-p2-smoke-runbook.md +++ /dev/null @@ -1,1905 +0,0 @@ -# ADR-021 P2 Stage D — live smoke runbook - -Working verification document for issue #645 on branch -`feat/645-lambda-microvm-p2` @ `3a4b61a97b22b7bcdd9832101f8d61a077fbf103`. -Companion to [`645-p1-lambda-microvm-runbook.md`](./645-p1-lambda-microvm-runbook.md), -whose command incantations this run reuses wholesale (escaped `ComputeTypes` -comma, `bootstrap --force`, newest-first version ordering, image-version `N.0` -spelling, two builds per version, delete-all-but-last then delete-image, -teardown-as-finally). - -`docs/scripts/sync-starlight.mjs` does not mirror `docs/verification/`, so this -file intentionally stays here and is not part of the Starlight site. - -**Live run:** 2026-08-06 22:38Z → 2026-08-07 01:26Z, account ``, -`us-east-1`. Evidence directory: `/tmp/abca-645-p2-20260806`. - -**Teardown: complete.** All MicroVMs terminated, image + version deleted, both -live IAM workarounds reverted, Cognito user deleted, secret died with the stack, -96/100 stack resources deleted (4 zero-cost residual, #702), **all billable -resources confirmed gone**, and every global/host configuration restored. See -Phase 8. - -## What Stage D was for - -P2 (`ab4808c`) wired everything short of the live run. Its commit message names -what it could not assert: - -> Remaining for P2 completion: the live smoke run (clone -> change -> PR with -> `bgagent watch`) and the deferred empirical items (suspend TTL >1h, -> SUSPENDED-vs-quota, `microvmImageHooks` API spelling, `NO_INGRESS` ARN). - -Five jobs, and their verdicts: - -| # | Job | Verdict | -|---|---|---| -| 1 | **THE SMOKE** — clone → change → PR through `bgagent watch` | **BLOCKED at `implement`, turn 0** — reproducible. Clone, branch, `/run`, `platform_config`, progress events and terminate all work; `claude --version` times out (P2-F5). **No PR was created.** | -| 2 | Adjudicate the CFN `AWS::Lambda::MicrovmImage` shape P1 left half-open | **DISCHARGED — the L1 is REFUTED on 5 values** (P2-F2) | -| 3 | `microvmImageHooks` spelling | **DISCHARGED both ways** — property name/nesting correct, hook *values* must be `ENABLED`/`DISABLED` (P2-F2); the API request built by the packaging script is correct and all four hooks were accepted and served | -| 4 | `NO_INGRESS` ARN name | **DISCHARGED** — the injected ARN is right and does suppress P1 F7's default public ingress | -| 5 | Extend suspend-TTL bound; re-probe SUSPENDED-vs-quota | **TTL extension SKIPPED** (time-boxed, see Skipped); quota **re-confirmed NOT OBSERVABLE** | - -**Headline:** the P2 substrate is much closer than P1 — the image builds with all -four hooks, the MicroVM launches, `/run` installs `platform_config`, the repo -clones, progress events stream to `bgagent watch`, and finalization terminates -the VM. But **four independent defects had to be worked around to get that far**, -and the run still ends one step short of a PR on a fifth. None of the four is -visible to `cdk synth`, `mise //cdk:test`, or any unit test: every one is a -live-service contract mismatch. - ---- - -## Variables - -```bash -set -o pipefail # P1's hard-won lesson: `| tee` masks failures -export AWS_PROFILE=aamorosi+workshops-AdminConsoleAccess -export AWS_REGION=us-east-1 -export AWS_DEFAULT_REGION="$AWS_REGION" -export CDK_DEFAULT_REGION="$AWS_REGION" -export CDK_DEFAULT_ACCOUNT="" -export STACK_NAME=backgroundagent-dev -export EXPECTED_BRANCH=feat/645-lambda-microvm-p2 -export SCRATCH_REPO=dreamorosi/batch-sync-triage -export SMOKE_USER="" -export CDK_DOCKER=finch # no docker on this box (P1 deviation 3) -export EVIDENCE_DIR=/tmp/abca-645-p2-20260806 -export BGAGENT_CONFIG_DIR=/tmp/abca-645-p2-bgagent -``` - -### ⚠️ zsh trap that bit this run - -The executor's shell is **zsh**, where `$VAR:latest` is parsed as the -`${VAR:l}` *lowercase modifier*, silently producing `…atest`. It cost one -mis-diagnosed container push. **Always brace: `${VAR}:latest`.** Likewise zsh has -no `PIPESTATUS` (it is `$pipestatus`, 1-indexed) and no `timeout(1)` — P1's -`test "${PIPESTATUS[0]}" -eq 0` silently evaluates to empty here. Use -`set -o pipefail` plus a plain `$?`. - ---- - -## Execution deviations (each is itself a result) - -1. **AZ pinning was required — P1 F11 is UNFIXED on this branch.** - `agent-vpc.ts` still does `maxAzs: 2` with no AZ constraint, and this account - still maps `us-east-1a → use1-az6`, which AgentCore rejects. Rather than - re-derive a finding P1 already recorded, the gitignored CDK context cache - (`cdk/cdk.context.json`, a build artifact — `git status` stayed clean) was - trimmed to lead with `us-east-1b` (`use1-az1`) + `us-east-1c` (`use1-az2`). - Original saved to `$EVIDENCE_DIR/cdk.context.json.ORIGINAL` and **restored at - teardown**. With the pin, AgentCore Memory and Runtime created cleanly, so - this is the whole of F11's remaining impact. -2. **The AgentCore container asset could not be pushed; an existing ECR manifest - was retagged instead.** `finch` (v1.17.2, no docker on this box) *built* the - image fine but every `finch push`/`finch pull` against ECR failed instantly - with `no basic auth credentials`. Root cause established: `finch push` shells - into the Lima VM as `limactl shell finch sudo -E nerdctl push`, and the VM's - `DOCKER_CONFIG` (`/home/aamorosi.guest/.finch-vm-config/config.json`) carries - `credsStore: finchhost`, a helper that cannot resolve this account's - Isengard `credential_process`. A host-side `finch login` succeeded and wrote - to `~/.finch/config.json`, but the VM never consulted it. Since the ECR image - is **only** consumed by the AgentCore runtime — which this run never invokes, - because the smoke runs on the MicroVM substrate built from the S3 zip by the - Lambda MicroVMs service — the required asset tag was added to an existing - manifest with `aws ecr put-image`. **This does not touch the MicroVM image - under test.** Consequence to be honest about: the deployed AgentCore runtime - carries a P1-era agent image. Nothing in this runbook depends on it. - *(All global config touched during that investigation — `~/.finch/config.json` - and the VM's docker config — was reverted; see Teardown.)* -3. **The connector operator role's trust policy had to be patched to deploy at - all** (P2-F1) — via a `/tmp` cloud-assembly patch, P1's technique. No - repository source file was modified. -4. **Two IAM workarounds were applied live to get past P2-F3** (execution-role - trust condition, plus a temporary unconditioned `iam:PassRole` that turned out - to be unnecessary). Both reverted; see Teardown. -5. **The CDK-managed `CfnMicrovmImage` path was attempted first, as briefed, and - failed** (P2-F2). The out-of-band `--create-image` script path was used - instead — the same path P1 used, and currently the only one that works. -6. **The suspend-TTL extension beyond P1's 1 h floor was skipped** (time-boxed). -7. **One self-inflicted error is recorded rather than hidden:** the GitHub PAT was - first written to Secrets Manager with a trailing newline (`gh auth token | - … --secret-string file:///dev/stdin`), which produced a real clone failure. - Corrected; see 3.2. - ---- - -## Phase 0 — Preflight - -### 0.1 Identity, region, branch - -`aws sts get-caller-identity` → -`arn:aws:sts:::assumed-role/AdminConsoleAccess/aamorosi-Isengard`, -account ``. Branch `feat/645-lambda-microvm-p2`, SHA -`3a4b61a97b22b7bcdd9832101f8d61a077fbf103`. Untracked: `docs/verification/`, -`opencode.json` — the same two P1 saw. - -> **Credential note.** The profile's default region is `eu-west-1`, so -> `AWS_REGION`/`AWS_DEFAULT_REGION` must be exported explicitly for every -> command. A bare `aws` call in this account goes to the wrong Region. -> Mid-run the credentials expired once; re-pinning `AWS_PROFILE` (which resolves -> through `credential_process` and auto-refreshes) fixed it. **No global AWS -> config was read or modified at any point.** - -`backgroundagent-dev` was **absent**, exactly as P1's Phase 8 hoped: - -``` -An error occurred (ValidationError) when calling the DescribeStacks operation: Stack with id backgroundagent-dev does not exist -``` - -**P1 F12's stack half is therefore RESOLVED**: P1 ended with `backgroundagent-dev` -in `DELETE_FAILED` (VPC + 2 subnets + 1 SG pinned by two leaked `agentic_ai` -ENIs) and a recorded retry command. AgentCore did eventually release them and the -stack is gone. The three unrelated pre-existing stacks -(`serverless-api-powertools`, `BuildingServerlessAPIs`, -`aws-sam-cli-managed-default`) plus `CDKToolkit` remain. - -### 0.2 Tooling - -| Tool | Version | Note | -|---|---|---| -| `aws-cli` | 2.36.13 | `aws lambda-microvms help` **exit 0**, **24 commands** — identical to P1 | -| `mise` | 2026.8.1 | **present this time** (P1 ran entirely on raw fallbacks) | -| node | v24.16.0 | | -| python3 | 3.9.6 | system python; the *guest* runs 3.13.13 | -| `zip` | 3.0 | | -| `rsync` | openrsync (protocol 29) | packaging script works unmodified | -| docker | **absent** | | -| `finch` | v1.17.2 | builds fine, **cannot push to ECR** (deviation 2) | - -### 0.3 Bootstrap — no re-bootstrap needed - -`CDKToolkit` `UPDATE_COMPLETE` (last updated 2026-07-31T18:54:48Z, i.e. P1's run), -and it already carries what P2 needs: - -- `ComputeTypes` = **`agentcore,lambda-microvm`** ✓ -- `BootstrapVariant` = **`ABCA: Least-Privilege Bootstrap`** -- bootstrap version SSM parameter = `32` -- `cdk-hnb659fds-cfn-exec-role--us-east-1` carries exactly the **five** - ABCA policies (Application, Infrastructure, Observability, Compute-Agentcore, - **Compute-LambdaMicrovms**) and **no `AdministratorAccess`**. - -So `bootstrap --force` was **not** re-run. - -> **P1 F10 is partly self-healing — correction to the P1 record.** P1 reported as -> a "durable consequence" that `BootstrapVariant` *stays* `AWS CDK: Default -> Resources` after a forced bootstrap, so every future non-forced bootstrap -> refuses forever. It now reads `ABCA: Least-Privilege Bootstrap`. The reason is -> the very next command in P1's own runbook: `update-stack -> --use-previous-template` with **only** `ParameterKey=ComputeTypes` supplied -> resets every unspecified parameter to its **template default** (that is -> CloudFormation's documented behaviour absent `UsePreviousValue=true`), and the -> ABCA template's default for `BootstrapVariant` is the ABCA string. F10's -> *first* half (the silent `exit 0` no-op without `--force`) stands; the -> "recurs forever" half does not. - -### 0.4 Managed base image - -Unchanged from P1: exactly **one** managed base image in `us-east-1`, -`arn:aws:lambda:us-east-1:aws:microvm-image:al2023-1`, versions `1` -(2026-07-21) and `0` (2026-06-17), **newest first**, so `items[0]` = `1`. -Selected version `1`; the service echoes `baseImageVersion: "1.0"`. - ---- - -## Phase 1 — Deploy - -### 1.1 Substrate synth — PASS - -`abca:microvm-image-not-provisioned` emitted with the full remedy text; -`abca:microvm-image-p1-smoke-unverified` correctly **absent**. Both incidental -warnings grew since P1 and one is close to the wall: - -| Metric | P1 | P2 | Limit | -|---|---|---|---| -| Template size | 893,273 | **977,650** (substrate) / **983,796** (with image) | 1,000,000 | -| Resource count | 463 | **485** / **486** | 500 | - -**Feeds back to design:** the template is at **98.4 %** of the 1 MB -CloudFormation limit with the image configured. That is ~16 KB of headroom — -roughly one more construct. `suppressTemplateIndentation` or a stack split is -now a near-term requirement, not a nicety. - -### 1.2 Substrate deploy — BLOCKED, then PASS after a trust-policy patch - -**Attempts 1–2 failed on the ECR push** (deviation 2), a purely local tooling -problem. **Attempts 3–4 failed identically on the network connectors** — this is -P2-F1: - -``` -Resource handler returned message: "The service is unable to assume the provided NetworkConnectorOperatorRole. Please verify the trust policy on the role. (Service: Lambda, Status Code: 400, Request ID: dbe1d2f4-dd0c-4319-8f5d-b4be4f076843) (SDK Attempt Count: 1)" (RequestToken: d6e21be9-9857-f452-bf12-3b93f89a4c75, HandlerErrorCode: InvalidRequest) -``` - -Both `LambdaMicrovmCompute/EgressConnector` and -`LambdaMicrovmCompute/BuildEgressConnector` `CREATE_FAILED`. Attempt 4 ran -against a **freshly deleted stack**, so this is **deterministic, not IAM -propagation lag** — an important distinction, because that error message is the -classic propagation symptom and a re-run is the obvious (wrong) first guess. - -**Attempt 5** deployed a `/tmp` cloud assembly with exactly one edit — the -`aws:SourceAccount` condition removed from -`LambdaMicrovmComputeConnectorOperatorRole`'s `AssumeRolePolicyDocument`: - -```json -{"Statement":[{"Action":"sts:AssumeRole","Effect":"Allow","Principal":{"Service":"lambda.amazonaws.com"}}],"Version":"2012-10-17"} -``` - -Both connectors went `CREATE_IN_PROGRESS → Resource creation Initiated` within a -second. **`CREATE_COMPLETE` 23:13:43Z → 23:27:03Z = 13 min 20 s**, 485 resources -(P1's comparable figure: 13 min 42 s / 464 resources). - -*Two P1 gotchas recurred verbatim and their remedies still work:* - -- `AWS::BedrockAgentCore::Memory` cannot be deleted while `CREATING` - (`Validation failed during DeleteMemory: Memory is in transitional state - CREATING. Cannot delete memory.`) → rollback ends `ROLLBACK_FAILED`; a plain - `aws cloudformation delete-stack` clears it (~3 min). **P1 F12's Memory half is - unfixed.** -- Post-synth template edits are silently ignored unless the template's S3 asset - object is deleted first (the object key is the *pre-edit* content hash). - -### 1.3 Substrate outputs — PASS - -`ComputeSubstrate = lambda-microvm`; all **seven** MicroVM outputs populated -(the six P1 had, plus `MicrovmBuildEgressConnectorArns`); artifact key exactly -`microvm-images/agent-artifact.zip`. Runtime and build egress connectors are -distinct ARNs, as designed. - -### 1.4 No-image orchestrator env — PASS, and a correction to P1 F14 - -`MICROVM_*` keys = **`[]`** (23 env keys total). All-or-nothing holds. - -> **P1 F14's `--no-paginate` remedy is WRONG and silently truncates.** P1 -> concluded that `list-stack-resources --query "…|[0]"` needs `--no-paginate`. -> With 485 resources, `--no-paginate` returns **only the first page** — it -> yielded 6 Lambda functions and *no* `TaskOrchestrator`, resolving -> `ORCHESTRATOR_FN` to the literal string `None` and producing a confident -> `Function not found: …:function:None`. That is worse than P1's original bug, -> because the original at least printed five values and broke loudly. **The -> correct approach is to let the CLI paginate (its default) and filter -> client-side**, which returns all 485: -> ```bash -> aws cloudformation list-stack-resources --stack-name "$STACK_NAME" --output json \ -> | python3 -c 'import json,sys; print([r["PhysicalResourceId"] for r in json.load(sys.stdin)["StackResourceSummaries"] if "TaskOrchestrator" in r["LogicalResourceId"] and r["ResourceType"]=="AWS::Lambda::Function"])' -> ``` - -**The new `agentPlatformConfig` env vars are present — 11 of 13.** All seven -`agentPlatformConfig` fields plus the four inherited ones landed on the -orchestrator: - -| `platform_config` key | Orchestrator env var | Present | -|---|---|---| -| `task_table_name` | `TASK_TABLE_NAME` | ✓ | -| `task_events_table_name` | `TASK_EVENTS_TABLE_NAME` | ✓ | -| `github_token_secret_arn` | `GITHUB_TOKEN_SECRET_ARN` | ✓ | -| `agent_session_role_arn` | `AGENT_SESSION_ROLE_ARN` | ✓ | -| `task_approvals_table_name` | `TASK_APPROVALS_TABLE_NAME` | ✓ | -| `nudges_table_name` | `NUDGES_TABLE_NAME` | ✓ | -| `log_group_name` | `LOG_GROUP_NAME` | ✓ | -| `artifacts_bucket_name` | `ARTIFACTS_BUCKET_NAME` | ✓ | -| `trace_artifacts_bucket_name` | `TRACE_ARTIFACTS_BUCKET_NAME` | ✓ | -| `aws_sdk_ua_app_id` | `AWS_SDK_UA_APP_ID` | ✓ (`uksb-wt64nei4u6#backgroundagent-dev`) | -| `anthropic_default_haiku_model` | `ANTHROPIC_DEFAULT_HAIKU_MODEL` | ✓ (`us.anthropic.claude-haiku-4-5-20251001-v1:0`) | -| `linear_oauth_secret_arn` | `LINEAR_OAUTH_SECRET_ARN` | — (per-workspace, CLI-created; correctly absent) | -| `jira_oauth_secret_arn` | `JIRA_OAUTH_SECRET_ARN` | — (same) | - -The two absentees are **not** in the contract's `required` list, so the -producer's optional-key handling is exercised and correct. - -*Incidental:* `ARTIFACTS_BUCKET_NAME` and `TRACE_ARTIFACTS_BUCKET_NAME` resolve -to the **same** bucket (`…-traceartifactsbucket8cbd5207-pwjbv0diorys`). Not a -MicroVM issue, but worth a glance — two distinct `platform_config` keys carrying -one bucket makes the trace/artifact separation notional. - -### 1.5 The CDK-managed `CfnMicrovmImage` path — **REFUTED** (P2-F2) - -This is the question P1 left open, and it is now closed. Synth produced the -warning correctly (`abca:microvm-image-p1-smoke-unverified`) and the exact L1 -under test: - -```json -"CpuConfigurations": [{ "Architecture": "arm64" }], -"Hooks": { - "Port": 8080, - "MicrovmHooks": { "Run": "/aws/lambda-microvms/runtime/v1/run", "RunTimeoutInSeconds": 60, - "Terminate": "/aws/lambda-microvms/runtime/v1/terminate", "TerminateTimeoutInSeconds": 15 }, - "MicrovmImageHooks": { "Ready": "/aws/lambda-microvms/runtime/v1/ready", "ReadyTimeoutInSeconds": 60, - "Validate": "/aws/lambda-microvms/runtime/v1/validate", "ValidateTimeoutInSeconds": 60 } -} -``` - -CloudFormation **rejected it at change-set early validation** — the stack was -never touched, so there was no rollback: - -``` -Early validation failed for change set cdk-deploy-change-set: -backgroundagent-dev/LambdaMicrovmCompute/Image (AWS::Lambda::MicrovmImage LambdaMicrovmComputeImage16B48539) - /aws/lambda-microvms/runtime/v1/run is not a valid enum value. Supported values: [DISABLED, ENABLED] (at - /Resources/LambdaMicrovmComputeImage16B48539/Properties/Hooks/MicrovmHooks/Run) - /aws/lambda-microvms/runtime/v1/terminate is not a valid enum value. Supported values: [DISABLED, ENABLED] (at - /Resources/LambdaMicrovmComputeImage16B48539/Properties/Hooks/MicrovmHooks/Terminate) - arm64 is not a valid enum value. Supported values: [ARM_64] (at - /Resources/LambdaMicrovmComputeImage16B48539/Properties/CpuConfigurations/0/Architecture) - /aws/lambda-microvms/runtime/v1/ready is not a valid enum value. Supported values: [DISABLED, ENABLED] (at - /Resources/LambdaMicrovmComputeImage16B48539/Properties/Hooks/MicrovmImageHooks/Ready) - /aws/lambda-microvms/runtime/v1/validate is not a valid enum value. Supported values: [DISABLED, ENABLED] (at - /Resources/LambdaMicrovmComputeImage16B48539/Properties/Hooks/MicrovmImageHooks/Validate) -``` - -**Five rejected values, and the CFN surface is identical to the API surface.** -This refutes the construct's explicit reasoning — *"The CDK L1 remains -intentionally unchanged because its generated CloudFormation types accept string -values and document no architecture/hook allowed-value constraint"*. The types -accept strings; the **service** enforces the enum, at change-set time. - -Two useful corollaries: - -- **The `microvmImageHooks` *spelling* is CORRECT.** The errors are scoped to - `…/Hooks/MicrovmImageHooks/Ready` and `…/Validate`, so CFN resolved the - property name and its children. Only the *values* are wrong. Combined with 2.1 - below (the API accepted the identical structure), the naming question is - discharged in both directions. -- **Hook paths are not configurable anywhere.** Neither CFN nor the API takes a - path; both take `ENABLED`/`DISABLED`. The service calls fixed well-known routes - — 2.2 proves they are exactly the `/aws/lambda-microvms/runtime/v1/*` strings - the agent serves. So `RUN_HOOK_PATH` and friends are correct *as route - constants for the agent* and simply must not be sent as property values. - -### 1.6 Wired deploy (out-of-band image) — PASS - -After the image existed (Phase 2), redeploying with -`--context microvm_image_identifier=` completed in **3 min 33 s** -(00:06:00Z → 00:09:33Z), `UPDATE_COMPLETE`, warning -`abca:microvm-image-p1-smoke-unverified` emitted. **All six `MICROVM_*` vars -present — all-or-nothing WITH the image confirmed:** - -``` -MICROVM_EGRESS_CONNECTOR_ARNS = arn:aws:lambda:us-east-1::network-connector:nc-d306f00f-1bd0-45ea-9457-0fcec0dab2a4 -MICROVM_EXECUTION_ROLE_ARN = arn:aws:iam:::role/backgroundagent-dev-LambdaMicrovmComputeExecutionRo-tF8Idpc9aT0R -MICROVM_IMAGE_IDENTIFIER = arn:aws:lambda:us-east-1::microvm-image:backgroundagent-dev-abca-agent -MICROVM_IMAGE_VERSION = 1.0 -MICROVM_INGRESS_CONNECTOR_ARNS = arn:aws:lambda:us-east-1:aws:network-connector:aws-network-connector:NO_INGRESS -MICROVM_PAYLOAD_BUCKET = backgroundagent-dev-lambdamicrovmcomputepayloadbuc-bctl5ej8aazr -``` - -**P1 F3 is FIXED:** `MICROVM_IMAGE_IDENTIFIER` is a full ARN, not a bare name. - ---- - -## Phase 2 — Image - -### 2.1 `create-microvm-image` — PASS with all four hooks - -`package-microvm-artifact.sh` (no `--create-image`) staged and uploaded first: -**904K artifact / 922,056 bytes in S3, SSE `AES256`** (P1: 584K / 597,305 — the -P2 `server.py` growth). Then with `--create-image`, exit **0**, and the service -echoed the request back: - -```json -"cpuConfigurations": [{ "architecture": "ARM_64" }], -"resources": [{ "minimumMemoryInMiB": 8192 }], -"hooks": { - "port": 8080, - "microvmHooks": { "run": "ENABLED", "runTimeoutInSeconds": 60, - "terminate": "ENABLED", "terminateTimeoutInSeconds": 15 }, - "microvmImageHooks": { "ready": "ENABLED", "readyTimeoutInSeconds": 60, - "validate": "ENABLED", "validateTimeoutInSeconds": 60 } -}, -"state": "CREATING", "imageVersion": "1.0", -"imageArn": "arn:aws:lambda:us-east-1::microvm-image:backgroundagent-dev-abca-agent" -``` - -**`microvmImageHooks` with `ready` + `validate`, and `microvmHooks` with `run` + -`terminate`, are all ACCEPTED.** P1 could not test this: it never got a -four-hook image created (F1), and its `/ready`-only attempt failed the build. - -The script's P2 reminder banner printed both before and after the call, and -`8192 MiB` was accepted — P1 F5's ceiling holds. - -### 2.2 Build — PASS in 4 min 35 s, `/ready` **and** `/validate` served - -Two builds per version again (`chipsetGeneration` 3 and 4), both `SUCCESSFUL`; -`state=SUCCESSFUL`, `status=ACTIVE`. **00:00:01Z → 00:04:36Z = 4 min 35 s** -(P1: 5 min 51 s). - -The decisive evidence, from `/aws/lambda-microvms/backgroundagent-dev-abca-agent` -— **this is the item P1 F1 blocked entirely**: - -``` -[server/build-hook] /ready hook: server is up, reporting ready for snapshot -INFO: 127.0.0.1:52856 - "POST /aws/lambda-microvms/runtime/v1/ready HTTP/1.1" 200 OK -[server/build-hook] /validate hook: ok (python=3.13.13, platform_config_keys=13, warnings=0) -INFO: 127.0.0.1:50860 - "POST /aws/lambda-microvms/runtime/v1/validate HTTP/1.1" 200 OK -``` - -(both lines twice, once per chipset). Note what this proves beyond "the hooks -work": the service POSTs to **exactly** the `/aws/lambda-microvms/runtime/v1/*` -paths, confirming the fixed-route model inferred in 1.5; `/validate` reports -`platform_config_keys=13`, so the cross-package contract loaded inside the -snapshot; and `warnings=0`, so the baked-secret scan found nothing. - -**P1 F4 is FIXED:** `apt-get` reached `deb.debian.org` over port 80 through the -dedicated build connector — the build log shows `Get:… http://deb.debian.org/…` -succeeding. No temporary security-group rule was needed this time. - -`agent/Dockerfile` also now installs Go tooling; the build log shows -`go: downloading …` completing, so build-time egress is sufficient. - -### 2.3 Sizes — P1 F13 reconfirmed and slightly worse - -| Field | Bytes | Human | vs P1 | -|---|---|---|---| -| `codeInstallSizeInBytes` | 2,342,203,392 | **2.18 GiB** | 2.17 GiB | -| `memorySnapshotSizeInBytes` | 1,223,421,952 | 1.14 GiB | 1.13 GiB | -| `diskSnapshotSizeInBytes` | 34,959,360 | 33.3 MiB | 35.4 MiB | - -`codeInstallSizeInBytes` still **exceeds AgentCore's 2 GB container-image -limit**, while the same tree as an OCI image is 1.803 GB / 631.1 MB compressed -(measured locally). P1 F13's "say which measure you mean" recommendation stands. -`snapshotBuild` is still only on `get-microvm-image-build`, not on the version. - ---- - -## Phase 3 — Platform user, secret, repo onboarding - -### 3.1 Cognito user — PASS, no first-login dance needed - -`bgagent admin invite-user --stack-name … --region …` -created the user with a **permanent** password (`UserStatus: CONFIRMED`, not -`FORCE_CHANGE_PASSWORD`), wrote credentials to -`$BGAGENT_CONFIG_DIR/invites/…txt` mode `0600`, and printed a `configure` -bundle. `bgagent configure --stack-name` + `bgagent login --username …` → -`Login successful.` **This is the answer to "find the exact bgagent commands": -`admin invite-user` is the whole first-login flow** — there is no -`RespondToAuthChallenge` step to drive, which is why P1's 7.1 blocker -("user pool has zero users") is a two-command fix, not an obstacle. - -### 3.2 GitHub PAT into the stack secret — PASS on the second attempt - -The token was piped from `gh auth token` straight into -`aws secretsmanager put-secret-value` and **never** written to disk, a log, or -this file. - -**Operator error worth recording, because it produced a convincing false -defect.** `gh auth token` emits a trailing newline and -`--secret-string file:///dev/stdin` stores it verbatim, so the secret was **41 -bytes**. My own verification used `.strip()` and reported "length 40", hiding it. -The task then failed inside the guest with: - -``` -RuntimeError: clone failed (non-transient): Post "https://api.github.com/graphql": net/http: invalid … -``` - -— i.e. an invalid HTTP header value, because the token carried `\n`. Rewriting -the secret stripped (40 bytes, no trailing whitespace) fixed the clone -immediately. - -*Verification lesson:* assert on the **raw** secret, never a stripped copy: - -```bash -aws secretsmanager get-secret-value --secret-id "$SECRET_ARN" --output json \ - | python3 -c 'import json,sys; s=json.load(sys.stdin)["SecretString"]; print(len(s), s!=s.strip())' -``` - -*Minor robustness observation (not a defect found by this run's design):* the -agent passes the secret value through to `gh`/`git` unstripped, so any -whitespace an operator introduces surfaces as a confusing `net/http` error rather -than "your token looks malformed". A `.strip()` at the token resolver would turn -a 20-minute misdiagnosis into a non-event. - -### 3.3 Repo onboarding — PASS, gate and probe both behaved - -`bgagent repo onboard dreamorosi/batch-sync-triage --compute-type lambda-microvm`: - -```json -{ "repo": "dreamorosi/batch-sync-triage", "status": "active", - "compute_type": "lambda-microvm", "onboarded_at": "2026-08-07T00:10:43.376Z" } -``` - -Both guards fired as designed and in the documented order: the **ComputeSubstrate -gate** passed because the stack output reads `lambda-microvm`, and the live -**`ListManagedMicrovmImages` availability probe** passed. The command also -printed the two ADR-021 advisory notes, including the smoke-unverified warning — -correct, and still accurate at the end of this run. - -`bgagent platform doctor` → `passed: true`, **all 7 checks**, including -`Managed MicroVM images are available in us-east-1` and — because 3.2 had already -run — `GitHubTokenSecretArn contains a token value`. (P1 noted this check cannot -distinguish a real PAT from the 32-char generated placeholder; that is still -true, it just happens to be a true positive here.) - ---- - -## Phase 4 — THE SMOKE - -Five submissions. Each failure moved the boundary forward, so all five are -recorded. - -| # | Task ID | Outcome | Finding | -|---|---|---|---| -| 1 | `01KZCRY70HRBP236GECR768JJX` | `FAILED` — `RunMicrovm … AccessDeniedException … iam:PassRole` | P2-F3 | -| 2 | `01KZCS6451HRPSXAG33Z4R5XRV` | identical, **with an unconditioned `iam:PassRole` attached** → so PassRole was never the real problem | P2-F3 | -| 3 | `01KZCSCD6PKDZNC8WRTFNRQG3H` | identical, 3 min after the IAM change → **not propagation lag** | P2-F3 | -| 4 | `01KZCSNM8MD4MB17ZTW1VDB6PY` | **`RUNNING`** after removing the execution-role trust condition, then `clone failed … net/http: invalid` | P2-F3 root cause proven; 3.2 token bug | -| 5 | `01KZCSVRHZXVHXQ4T29XYZKBAM` | **`RUNNING` → clone OK → branch OK → `implement` failed at turn 0** | **P2-F5** | -| 5r | `01KZCT8SWZ1DDZC7P0RYS982ES` | identical to #5 — **reproducible** | P2-F5 | - -### 4.1 P2-F3 — the `iam:PassRole` deny that was really a trust-policy deny - -``` -Session start failed: Error: MicroVM RunMicrovm failed: AccessDeniedException: User: arn:aws:sts:::assumed-role/backgroundagent-dev-TaskOrchestratorOrchestratorFnS-kyEY8iz3mrcm/backgroundagent-dev-TaskOrchestratorOrchestratorFn-huSs3tbuFbJs is not authorized to perform: iam:PassRole on resource: arn:aws:iam:::role/backgroundagent-dev-LambdaMicrovmComputeExecutionRo-tF8Idpc9aT0R because no identity-based policy allows the iam:PassRole action -``` - -The message is actively misleading, and the diagnosis is the useful part of this -section. The orchestrator's inline policy **does** carry the grant, exactly as -P1 4.4 recorded it: - -```json -{ "Sid": "MicrovmPassExecutionRole", "Effect": "Allow", "Action": "iam:PassRole", - "Resource": "arn:aws:iam:::role/backgroundagent-dev-LambdaMicrovmComputeExecutionRo-tF8Idpc9aT0R", - "Condition": { "StringEquals": { "iam:PassedToService": "lambda.amazonaws.com" } } } -``` - -Elimination sequence: - -1. Attached a **temporary unconditioned** `iam:PassRole` on the same resource → - **still denied** (submission 2). So the `iam:PassedToService` condition was - *not* the cause. -2. `aws iam get-role` → **no permissions boundary**, path `/`. -3. `aws iam simulate-principal-policy … --action-names iam:PassRole` → - **`allowed`**, matching my temporary statement. So IAM itself says yes. -4. Waited 3 minutes and resubmitted → **still denied** (submission 3). Not - propagation. -5. Read the **target** role's trust policy — and found the same - `aws:SourceAccount` condition that had just broken the network connectors: - -```json -{"Effect":"Allow","Principal":{"Service":"lambda.amazonaws.com"},"Action":"sts:AssumeRole", - "Condition":{"StringEquals":{"aws:SourceAccount":""}}} -{"Effect":"Allow","Principal":{"Service":"lambda.amazonaws.com"},"Action":"sts:TagSession", - "Condition":{"StringEquals":{"aws:SourceAccount":""}}} -``` - -6. Removed both conditions → **submission 4 reached `RUNNING` in 6 seconds.** - -**So `RunMicrovm` reports a role it cannot pass-and-assume as an `iam:PassRole` -identity-policy denial on the *caller*.** P2-F1 and P2-F3 are therefore **one -root cause with two symptoms**: the Lambda MicroVMs service does not present -`aws:SourceAccount` when assuming ABCA's MicroVM-facing roles, so every trust -policy carrying that confused-deputy condition is unassumable. - -### 4.2 THE SMOKE (submission 5) — `bgagent watch`, verbatim - -``` -Watching task 01KZCSVRHZXVHXQ4T29XYZKBAM... (Ctrl+C to stop) -[5:28:03 PM] ★ repo_setup_complete: branch=bgagent/01KZCSVRHZXVHXQ4T29XYZKBAM/add-a-codeowners-file-at-the-repository-root-conta build_before=False -[5:28:04 PM] ★ step:implement:start -[5:28:15 PM] ★ step:implement:failed -[5:28:15 PM] ★ agent_execution_complete: status=error turns=0 -Task 01KZCSVRHZXVHXQ4T29XYZKBAM failed. timeout: Build/tests didn't finish in time (timed out) — Workflow run_agent step failed: TimeoutExpired: Command '['claude', '--version']' timed out after 10 seconds -``` - -**Time-to-RUNNING: ~6 s.** Submitted 00:27:09Z, `started_at` -`2026-08-07T00:27:11`, VM `startedAt` 00:27:12Z. That is materially faster than -AgentCore or ECS cold start and is the backend's main selling point — worth -recording as the one positive performance result of the run. - -**`/run` accepted the envelope, and `platform_config` was installed** — the -required observation, from the guest log group: - -``` -[server/run-pre-config] /run hook received: microvm_id='microvm-8ad29e93-99c2-3ccc-b079-12f8bdab2936' bytes=2120 -[server/debug] /run hook installed platform_config env: ['AGENT_SESSION_ROLE_ARN', 'ANTHROPIC_DEFAULT_HAIKU_MODEL', 'ARTIFACTS_BUCKET_NAME', 'AWS_SDK_UA_APP_ID', 'GITHUB_TOKEN_SECRET_ARN', 'LOG_GROUP_NAME', 'NUDGES_TABLE_NAME', 'TASK_APPROVALS_TABLE_NAME', 'TASK_EVENTS_TABLE_NAME', 'TASK_TABLE_NAME', 'TRACE_ARTIFACTS_BUCKET_NAME'] -[server/debug] /run hook accepted task_id='01KZCSVRHZXVHXQ4T29XYZKBAM' microvm_id='microvm-8ad29e93-99c2-3ccc-b079-12f8bdab2936' -``` - -Everything in the P2 delivery design is confirmed here: **2,120 bytes**, so the -envelope went **inline** and stayed under the real 4,096-byte cap (P1 F6); the -installed set is **exactly the 11 available keys**, names only, no values; the -pre-install line is stdout-only (`run-pre-config`) and the post-install line is -the first to reach CloudWatch, exactly as the snapshot-credential-hygiene work -intended. - -**Progress events streamed** — `repo_setup_complete`, `step:implement:start`, -`step:implement:failed`, `agent_execution_complete` all arrived live in `watch`. - -**Clone → change → PR got exactly one step:** clone ✓, branch ✓, -`build_before=False` ✓ … then `implement` died at turn 0. **No commit, no push, -no PR.** - -**Heartbeats: NOT observed.** `agent_heartbeat_at` was `None` at every poll -across all six submissions. The task was `RUNNING` for only ~12 s and the agent -bumps the heartbeat every 45 s, so it never had a chance to fire. **The -dual-signal liveness path is therefore NOT discharged** — see Skipped. - -### 4.3 P2-F5 — `claude --version` times out in the guest - -``` -[00:28:04] AGENT claude-agent-sdk version: 0.2.110 -[00:28:15] ERROR step 'implement' handler raised: TimeoutExpired: Command '['claude', '--version']' timed out after 10 seconds -[00:28:15] WORKFLOW step 'implement' failed (on_failure=fail) — workflow FAILED -``` - -`METRICS_REPORT`: `"turns": 0, "duration_s": 12.7, "code_changed": null, -"pr_url": null, "memory_written": true`. - -The probe is `agent/src/runner.py:476`: - -```python -["claude", "--version"], capture_output=True, text=True, timeout=10 -``` - -Characterisation, to separate "broken binary" from "slow substrate": - -- **In the identical image, locally under finch: `claude --version` → `2.1.191 - (Claude Code)` in under 1 second.** So the binary, its symlink and `PATH` are - all fine. -- `/usr/bin/claude` → `../lib/node_modules/@anthropic-ai/claude-code/bin/claude.exe`, - a **236,305,136-byte (225 MiB) statically-linked ELF**. -- The MicroVM had been restored from a snapshot ~50 s earlier and had already - done a full `git clone` over the network. - -The consistent reading is **lazy snapshot hydration**: the first `exec` of a -225 MiB binary that was never touched before the snapshot was taken must fault -its pages in from lazily-restored storage, and that exceeds 10 s. Two -independent fixes suggest themselves, and the second is the more interesting -because P2 already built the mechanism and then deliberately declined to use it: - -1. Raise / make backend-aware the `timeout=10` (it is a *liveness probe for a - version string* — a tight bound buys nothing). -2. **Warm `claude` in the `/ready` build hook so it lands in the memory - snapshot.** P2's `/ready` deliberately does the minimum — `"server is up, - reporting ready for snapshot"` — and its own docstring explains that `/ready` - exists so "the snapshot is taken with a warm server". The snapshot is warm for - *uvicorn* and stone cold for the 225 MiB binary that does all the work. - -**`build_passed: true, lint_passed: false`** in the same report is incidental -noise from the scratch repo (`mise ERROR no tasks defined …`, `unknown command: -lint`) and unrelated to the substrate. - -### 4.4 P2-F4 — the execution role cannot write to the application log group - -Found while reading logs for F5, and independent of it. Every structured agent -log line to the log group that `platform_config` itself delivers is denied: - -``` -[server/debug/self] CloudWatch write failed: AccessDeniedException: … User: arn:aws:sts:::assumed-role/backgroundagent-dev-LambdaMicrovmComputeExecutionRo-tF8Idpc9aT0R/Lambda-microvmsExecutor-86cfecce-… is not authorized to perform: logs:CreateLogStream on resource: arn:aws:logs:us-east-1::log-group:/aws/vendedlogs/bedrock-agentcore/runtime/APPLICATION_LOGS/backgroundagent-dev:log-stream:server_debug/01KZCSVRHZXVHXQ4T29XYZKBAM because no identity-based policy allows the logs:CreateLogStream action -``` - -Same denial for the `metrics/` stream, so `METRICS_REPORT` never lands -either. P2 wired `log_group_name` into `platform_config` (making the agent -*attempt* the write) but the execution role's logs grant is scoped to -`/aws/lambda-microvms/*` only — the construct's own cdk-nag suppression says so: -*"the `MICROVM_LOG_GROUP_PREFIX/*` namespace"*. - -Not fatal — the agent degrades to stdout, which the MicroVM log group captures, -which is why this run could be debugged at all. But it is a genuine smoke-parity -hole: on `lambda-microvm`, the platform's canonical per-task observability -streams are empty, and anything reading them (rather than the guest's stdout) -sees nothing. Exactly the class of grant P2 added for Bedrock, Secrets Manager -and Memory — this one was missed. - ---- - -## Phase 5 — Post-smoke verification - -### 5.1 Finalization called `TerminateMicrovm` — PASS - -`get-microvm` on the smoke VM: - -``` -microvmId = microvm-8ad29e93-99c2-3ccc-b079-12f8bdab2936 -state = TERMINATED -stateReason= Success. -startedAt = 2026-08-06T17:27:12.144000-07:00 -``` - -and the in-guest breadcrumb fired, 200 OK: - -``` -[server/debug] /terminate hook: {"active_pipeline_threads": 0, "background_pipeline_failed": false, "event": "microvm_terminate", "microvm_id": "", "timestamp": "2026-08-07T00:28:42.926428+00:00"} -INFO: 127.0.0.1:37364 - "POST /aws/lambda-microvms/runtime/v1/terminate HTTP/1.1" 200 OK -``` - -So `/terminate` is served, returns 200, reports a clean pipeline state, and does -not write terminal task status. **Note `"microvm_id": ""`** — the hook body's -`microvmId` arrives empty, unlike `/run` where it is populated. Cosmetic, but it -defeats the hook's stated purpose of joining the guest record to the -control-plane one, and the `/run` line has to carry that correlation instead -(which it does). - -**P1 F8 reconfirmed:** the VM sat at `TERMINATED` for the whole observation -window; `ResourceNotFoundException` never arrived. `TERMINATED` must map to -completed in its own right. - -### 5.2 `NO_INGRESS` — DISCHARGED - -``` -ingressNetworkConnectors = ['arn:aws:lambda:us-east-1:aws:network-connector:aws-network-connector:NO_INGRESS'] -egressNetworkConnectors = ['arn:aws:lambda:us-east-1::network-connector:nc-d306f00f-1bd0-45ea-9457-0fcec0dab2a4'] -``` - -The ARN P2 injects is **correct and effective**: the service accepted it and -`HTTP_INGRESS` (P1 F7's unrequested default public ingress) is **gone**. P1's -security finding is mitigated. - -**One caveat worth carrying forward:** a `NO_INGRESS` VM **still returns a public -endpoint hostname** (`.lambda-microvm.us-east-1.on.aws`). -So the presence of an `endpoint` in a `RunMicrovm` response is *not* evidence of -reachability, and any doc or alarm that treats "endpoint exists" as "ingress is -open" will be wrong in both directions. - -### 5.3 Dual-signal liveness — NOT DISCHARGED - -`agent_heartbeat_at` stayed `None` throughout. The heartbeat interval is 45 s and -no task was `RUNNING` for more than ~13 s, because F5 kills every task at turn 0. -**This item is blocked behind P2-F5, not independently testable.** The -`heartbeatLivenessApplies` switch is `RUNNING`-scoped and never got a -long-enough `RUNNING` window to exercise. - ---- - -## Phase 6 — Lifecycle extras - -### 6.1 A run-hook 4xx **self-terminates** the VM — supersedes P1 F9 - -Launching a second MicroVM from the image with **no** `runHookPayload` (the image -has `run: ENABLED`) produced a result P1 could not see, because P1's only -creatable image was hook-less: - -``` -state = TERMINATED -stateReason = Run lifecycle hook returned HTTP status 400. Please check your hook endpoint and application logs for more details. -``` - -Terminal within ~12 s of launch, and `suspend-microvm` then correctly refused: -`The MicroVM … has been terminated and its state cannot be changed.` - -**P1 F9 said "hook-less MicroVMs run indefinitely and bill", and warned that -nothing self-cleans.** With `/run` ENABLED that is no longer true: **the service -reaps a VM whose run hook returns 4xx.** This materially improves the cost -posture and is a direct benefit of declaring hooks. Active `TerminateMicrovm` -remains correct for the *success* path, but the *failure* path now has a -service-side backstop. - -Incidental confirmation: `executionRoleArn` must be a **full ARN** — -a bare role name is rejected with -`Member must satisfy regular expression pattern: arn:aws[a-z\-]*:iam::[0-9]{12}:role/?…`. - -### 6.2 Suspend / resume on a hook-ENABLED image — PASS - -To get a VM that stays `RUNNING`, a **hand-built valid envelope** (713 bytes: -`agent_payload` with `task_id`/`repo_url`/`task_description`/`resolved_workflow` -plus the four required `platform_config` keys) was passed as `runHookPayload`. -`/run` returned 200 and the VM held `RUNNING`. This is itself a useful result: -**the documented envelope shape is reproducible by hand from the contract alone.** - -| Step | Latency | State | Notes | -|---|---|---|---| -| launch → `RUNNING` | ~11 s | `RUNNING`, `stateReason` `None` | | -| `suspend-microvm` | **~2 s** | `SUSPENDED` | `/suspend` hook **DISABLED** in the image | -| `resume-microvm` | **~3 s** | `RUNNING` | `/resume` hook **DISABLED** | -| `terminate-microvm` | **~3 s** | `TERMINATED`, `stateReason` `Success.` | `/terminate` hook fired, 200 OK | - -`microvmId` **and** `endpoint` -(`.lambda-microvm.us-east-1.on.aws`) were -**byte-identical** before and after the suspend/resume cycle, and `startedAt` / -`maximumDurationInSeconds` (28800) never moved. - -**This extends P1 5.3/5.6 to a hook-enabled image:** suspend and resume work -*without* `/suspend` and `/resume` being declared, so P3's interface widening is -not gated on the hooks — a stored `SessionHandle` survives a cycle here too. -`SUSPENDING` was again never observable. - -### 6.3 Quota — re-probed, still NOT OBSERVABLE - -`L-CD1C0CC4` "Max allocated MicroVM memory" = **1024 Gigabytes**, and critically -`UsageMetric` = **`null`**. `AWS/Usage` still exposes **only** `CallCount` per API -name (`GetMicrovm`, `CreateMicrovmImage`, `ListMicrovmImages`, …) and **no memory -metric in any namespace**. **P1 5.5's verdict is unchanged: the claim "a -suspended VM still holds account memory quota" remains NOT OBSERVABLE SAFELY and -undischarged.** Proving it would still need ~128 concurrent 8 GiB VMs. - ---- - -## Phase 7 — Scratch repo - -**Nothing to clean up, and no PR to leave open.** `dreamorosi/batch-sync-triage` -is byte-for-byte as it was found: - -- Branches: `main` + the five pre-existing `dependabot/*`. **No `bgagent/*` - branch was ever pushed** — the agent created the branch locally in the guest - and died at `implement` before any commit or push. -- PRs: the same five open dependabot PRs (#1–#5) from 2025-11-24. **No PR was - created by this run.** - -So the instruction to leave the CODEOWNERS PR open is moot: **P2-F5 prevented any -PR from existing.** That absence is the single most important line in this -document. - ---- - -## Phase 8 — Teardown (executed as a finally-block) - -### 8.1 MicroVMs — all TERMINATED - -Nine MicroVMs existed across the run (six from the six task submissions, plus -the two Phase 6 lifecycle VMs and one orchestrator retry). **All nine were -already `TERMINATED`** at teardown — every one either finalized by the -orchestrator's `TerminateMicrovm`, service-reaped after a run-hook 4xx (6.1), or -explicitly terminated in 6.2. No VM needed chasing, and none ever approached the -8 h bound. - -### 8.2 Image and versions — deleted - -Only one version existed (`1.0`, `ACTIVE`), so `delete-microvm-image` alone was -sufficient and reaped it. `list-microvm-images` → `{"items": []}`; -`get-microvm-image` → -`ResourceNotFoundException: MicroVMImage not found for MicroVMImageID: …`. -P1's ordering correction (delete all but the last version, then the image) was -therefore not exercised, but is not contradicted. - -*Incidental:* one `ListMicrovmImages` call returned `502 Bad Gateway (reached max -retries: 2)` while the image was `DELETING`; it succeeded 30 s later. Worth -retrying rather than treating as a failure. - -### 8.3 Live IAM workarounds — reverted - -- Temporary unconditioned `iam:PassRole` inline policy - (`abca645p2-verification-passrole`) **deleted** from the orchestrator role. -- The execution role's trust policy **restored** to its deployed form, i.e. with - both `aws:SourceAccount` conditions back (verified by re-reading it). The - defect is left exactly as the branch produces it. - -### 8.4 Cognito user — deleted - -`bgagent admin delete-user ` → -`✓ Deleted Cognito user`. `list-users` then returned **empty**. The local invite -file containing its password was `rm`'d. The GitHub-token secret was left to die -with the stack (it did). - -### 8.5 Stack — DELETE_FAILED, 96/100 deleted, identical to P1's residual - -Three delete attempts, and each failed differently — that progression is itself -the finding: - -| Attempt | Duration | Outcome | -|---|---|---| -| 1 | 00:55:35Z → 01:04:28Z (8 min 53 s) | `DELETE_FAILED` — `AWS::BedrockAgentCore::Runtime`: `"Request timed out while deleting AWS::BedrockAgentCore::Runtime"`, `HandlerErrorCode: NotStabilized` | -| 2 (the briefed retry) | 01:04:55Z → 01:06:00Z (1 min) | `DELETE_FAILED` — same resource, **different error**: `"Access denied for operation 'AWS::BedrockAgentCore::Runtime'."`, `HandlerErrorCode: AccessDenied` | -| 3 (informed) | 01:07:02Z → 01:24:47Z (17 min 45 s) | `DELETE_FAILED` — but the Runtime, Memory and both IAM roles **did** delete; only the VPC set remains | - -Attempt 3 was not a blind third retry: between attempts, -`bedrock-agentcore-control list-agent-runtimes` returned **empty**, proving the -runtime was already gone server-side and the failures were a handler -stabilization/authorization artifact rather than a real leftover. Acting on that -evidence took the residual from 7 resources to 4. - -**Final residual — 4 resources, all zero-cost, exactly P1's set:** - -``` -CREATE_COMPLETE AWS::EC2::VPC vpc-072fddf653ccdcfc4 -DELETE_FAILED AWS::EC2::Subnet subnet-0e6a6a0ed18100c8a -DELETE_FAILED AWS::EC2::Subnet subnet-02d91450e51cf72a0 -DELETE_FAILED AWS::EC2::SecurityGroup sg-00997a58c1f4c5775 -``` - -``` -resource sg-00997a58c1f4c5775 has a dependent object (Service: Ec2, Status Code: 400 …) -Resource handler returned message: "The subnet 'subnet-0e6a6a0ed18100c8a' has dependencies and cannot be deleted. (Service: Ec2, Status Code: 400 …)" HandlerErrorCode: InvalidRequest -``` - -Cause: the same two AgentCore-managed ENIs, still `in-use` — -`eni-04911e11e08d670f9` and `eni-04a1c5a27966ee08b`, both -`InterfaceType: agentic_ai`. **This is #702 / P1 F12 reproducing verbatim.** -Nothing was force-deleted past CloudFormation. P1's experience says these are -eventually released (its stack is gone now — see 0.1), so the retry command is: - -```bash -aws cloudformation delete-stack --stack-name backgroundagent-dev -aws cloudformation wait stack-delete-complete --stack-name backgroundagent-dev -``` - -### 8.6 Billing confirmed stopped - -| Check | Result | -|---|---| -| NAT gateways in the ABCA VPC | **none** (the 2 `available` ones belong to pre-existing `vpc-01c9984d163d2965e` and were not touched — same as P1) | -| VPC endpoints in the ABCA VPC | **none** | -| ABCA S3 buckets | **none** | -| Unattached (billable) EIPs | **none** | -| MicroVM images / non-terminated VMs | **none** | -| AgentCore runtimes / memories | **none** | -| `/aws/lambda-microvms/*` log groups | **none** | - -### 8.7 Environment restored — nothing global left modified - -- `cdk/cdk.context.json` **restored** to the original six-AZ list (gitignored - build artifact; `git status` shows only `docs/verification/` and the - pre-existing `opencode.json`). -- `~/.finch/config.json` **restored** to `credsStore: osxkeychain` with **no - stored credential**, and the Lima VM's `DOCKER_CONFIG` restored to - `{"credsStore":"finchhost"}`; the `/root/.docker/config.json` written during - the push investigation was removed. **The short-lived ECR token written during - that investigation is no longer on disk anywhere.** -- The ECR tag this run added (`34fbc1d4…`) was removed with - `batch-delete-image`, leaving the three pre-existing tags and the underlying - manifest exactly as found. -- The finch VM was stopped, as P1 did. -- **No AWS global configuration was read or modified at any point** - (`~/.aws/config`, `~/.aws/credentials` untouched); `~/.cdk.json` was never - created. All credential and Region selection was via environment variables in - the run's own shell. - -### 8.8 Deliberately retained - -1. **`CDKToolkit`** — `UPDATE_COMPLETE`, `ComputeTypes = agentcore,lambda-microvm`, - `BootstrapVariant = ABCA: Least-Privilege Bootstrap`, five ABCA policies, no - `AdministratorAccess`. Unchanged by this run. ⚠️ Still the shared-account - caveat P1 raised: other CDK apps in `` deploy through the - ABCA-scoped execution role. -2. **`backgroundagent-dev` in `DELETE_FAILED`** — the 4 zero-cost resources in 8.5. -3. **Bootstrap S3/ECR assets** — normal bootstrap content, including the three - pre-existing agent container images. -4. **Service-vended log groups** created outside CloudFormation - (`/aws/bedrock-agentcore/runtimes/…`, `/aws/lambda/backgroundagent-dev-…`). -5. **`dreamorosi/batch-sync-triage`** — untouched (Phase 7). - -### 8.9 Scratch repo — verified unchanged - -Branches: `main` + the five pre-existing `dependabot/*`. Open PRs: #1–#5, all -dependabot. **No `bgagent/*` branch, no smoke PR.** There was nothing to close, -nothing to delete, and — because of P2-F5 — nothing to leave open for the -operator to look at. - ---- - -## Findings summary - -Live run 2026-08-06/07, account ``, `us-east-1`, branch -`feat/645-lambda-microvm-p2` @ `3a4b61a9`. Evidence: -`/tmp/abca-645-p2-20260806`. - -**Verdict on Stage D's primary objective: the smoke did NOT pass. No pull request -was created.** The run got to `clone ✓ → branch ✓ → implement ✗ (turn 0)`, -reproducibly, and it took four separate live workarounds to get even that far. -Every one of the five defects below is invisible to synth, unit tests and -`cdk-nag`; all five are live-service contract mismatches, which is precisely what -Stage D exists to surface. - -### Items DISCHARGED (behaved as designed) - -1. **`microvmImageHooks` spelling — both directions.** The API accepted - `microvmImageHooks:{ready,validate}` + `microvmHooks:{run,terminate}` with - timeouts, and CloudFormation resolved the identical property path - (`…/Hooks/MicrovmImageHooks/Ready`). The open P2 item is closed. -2. **All four hooks are served and exercised.** `/ready` 200 and `/validate` 200 - (`python=3.13.13, platform_config_keys=13, warnings=0`) during the build, on - both chipsets; `/run` 200 with the envelope; `/terminate` 200 at teardown. - **P1 F1 — "a P1 image that declares `/run` is not creatable at all" — is - fully fixed.** -3. **`NO_INGRESS` ARN.** Correct, accepted, and it suppresses P1 F7's default - public `HTTP_INGRESS`. (Caveat: a public endpoint hostname is still returned.) -4. **`platform_config` delivery end to end.** 2,120-byte envelope inline under - the real 4,096-byte cap; exactly the 11 available keys installed; names-only - logging; pre-install logging correctly stdout-only. -5. **Image buildability.** 4 min 35 s, two builds per version, `ACTIVE`; 8192 MiB - accepted; `apt-get`/Go egress over port 80 through the dedicated build - connector. **P1 F4 (443-only SG) and F5 (32 GiB memory) are fixed.** -6. **`MICROVM_IMAGE_IDENTIFIER` is a full ARN — P1 F3 fixed**, and all six - `MICROVM_*` vars appear together (all-or-nothing, both directions). -7. **`agentPlatformConfig` wiring.** 11 of 13 keys on the orchestrator; the two - absent ones are optional per-workspace secrets, so the optional-key path is - also verified. -8. **CLI end to end.** `admin invite-user` (permanent password — no first-login - challenge to drive), `configure --stack-name`, `login`, `repo onboard - --compute-type lambda-microvm` (ComputeSubstrate gate **and** live - `ListManagedMicrovmImages` probe both pass), `platform doctor` 7/7, - `submit`, `watch` streaming progress events. -9. **Finalization terminates the VM** (`TERMINATED`, `stateReason: Success.`). -10. **Suspend/resume on a hook-enabled image** with `/suspend` + `/resume` - undeclared: ~2 s / ~3 s, `microvmId` **and** `endpoint` preserved. -11. **Time-to-RUNNING ≈ 6 s** — the backend's headline advantage, measured. -12. **P1 F12's stack half resolved**: the leaked AgentCore ENIs were eventually - released and P1's `DELETE_FAILED` stack is gone. - -### Items CONTRADICTING design assumptions — `feeds-back-to-design: YES` - -**P2-F1 + P2-F3 (ONE root cause, two symptoms, both blocking). The -`aws:SourceAccount` confused-deputy condition makes ABCA's MicroVM-facing roles -unassumable by the Lambda MicroVMs service.** - -*Symptom A — the substrate cannot deploy.* Both `AWS::Lambda::NetworkConnector` -resources `CREATE_FAILED`, deterministically, on a freshly deleted stack: - -``` -"The service is unable to assume the provided NetworkConnectorOperatorRole. Please verify the trust policy on the role. (Service: Lambda, Status Code: 400, Request ID: dbe1d2f4-dd0c-4319-8f5d-b4be4f076843)" HandlerErrorCode: InvalidRequest -``` - -Removing the condition from `ConnectorOperatorRole`'s trust → both connectors -create within a second. Note this is a **regression introduced by the P1 F2 -fix**: P1's validated probe role trusted `lambda.amazonaws.com` with **no -conditions**, and the construct's comment says trust "mirrors the build/execution -roles (`lambda.amazonaws.com` + `aws:SourceAccount`)" — which is exactly what -breaks it. - -*Symptom B — no task can start.* `RunMicrovm` fails with a misleading -`iam:PassRole` denial **on the caller**, even though the orchestrator's grant is -present and `simulate-principal-policy` returns `allowed` and there is no -permissions boundary. Proven by elimination (unconditioned `iam:PassRole` still -denied; 3-minute wait still denied); removing the **execution role's** trust -conditions made the very next submission reach `RUNNING` in 6 s. - -Both the build role and the execution role carry the same pattern, so the fix is -one decision applied consistently: **drop `aws:SourceAccount` from every -MicroVM-facing role's trust policy, or find the condition key the service does -present** (`aws:SourceArn`/`aws:SourceAccount` are simply not populated on this -path). Note the *identity-side* `iam:PassedToService: lambda.amazonaws.com` -condition on the orchestrator was never shown to be wrong — it was exonerated by -step 1 and can stay. - -**P2-F2. The `CfnMicrovmImage` L1 is rejected by CloudFormation on five values — -the CDK-managed image path does not work at all.** Verbatim early-validation -output is in 1.5. `arm64` must be `ARM_64`; all four hooks must be -`ENABLED`/`DISABLED`, not paths. This **refutes the construct's stated reasoning** -that the L1 could keep CloudFormation's "path/string shape" because the generated -types "document no architecture/hook allowed-value constraint" — the service -enforces the enum at change-set time. Consequences: (a) the documented -"CDK-managed (recommended)" bootstrap path in -`cdk/scripts/package-microvm-artifact.sh` is currently non-functional and the -"out-of-band alternative" is the *only* working path; (b) hook paths are not -configurable on either surface, so the `*_HOOK_PATH` constants are agent route -constants only and must never be sent as property values. - -**P2-F4. The MicroVM execution role cannot write to the application log group -that `platform_config` tells the agent to use.** `logs:CreateLogStream` denied on -`/aws/vendedlogs/bedrock-agentcore/runtime/APPLICATION_LOGS/backgroundagent-dev` -for both the `server_debug/` and `metrics/` streams, because -the role's logs grant is scoped to `/aws/lambda-microvms/*`. P2 delivered -`log_group_name` (so the agent *attempts* the write) without the matching grant — -the same omission class the P2 Bedrock/Secrets/Memory grants were added to fix. -Non-fatal (stdout fallback lands in the MicroVM log group) but the canonical -per-task observability streams are empty on this backend, including -`METRICS_REPORT`. - -**P2-F5. `claude --version` times out after 10 s inside the MicroVM, failing -every task at turn 0. This is the blocker that prevented the PR.** Reproduced on -two consecutive submissions: - -``` -TimeoutExpired: Command '['claude', '--version']' timed out after 10 seconds -``` - -`agent/src/runner.py:476` uses `timeout=10`. The binary is fine: in the identical -image locally it answers `2.1.191 (Claude Code)` in **under 1 s**. It is a -**225 MiB (236,305,136-byte) statically-linked ELF** whose pages were never -touched before the snapshot was taken, exec'd on a guest restored ~50 s earlier — -consistent with lazy snapshot hydration. Two fixes: raise/make backend-aware the -timeout (a version-string probe gains nothing from a tight bound), and — more -interestingly — **warm `claude` in the `/ready` build hook**, which exists -precisely so "the snapshot is taken with a warm server" but currently warms only -uvicorn while leaving the 225 MiB binary that does all the work cold. - -**P2-F6. P1 F9 is superseded, in the safe direction.** With `run: ENABLED`, a -run-hook 4xx makes the **service** terminate the VM -(`stateReason: "Run lifecycle hook returned HTTP status 400."`) within ~12 s. P1's -"hook-less MicroVMs run indefinitely and bill; nothing self-cleans" no longer -describes this image. Worth correcting in ADR-021, because it changes the -cost-risk argument for the failure path. - -**P2-F7. Template size is at 98.4 % of the CloudFormation 1 MB limit** -(983,796/1,000,000) and 486/500 resources with the image configured. ~16 KB of -headroom — roughly one more construct before deploys start failing for reasons -unrelated to MicroVMs. - -**P2-F8. Runbook/tooling corrections (lower severity, but each would corrupt a -future pass).** - -- **P1 F14's `--no-paginate` remedy is wrong and silently truncating.** At 485 - resources it returns only the first page, resolved `ORCHESTRATOR_FN` to the - literal `None`, and produced `Function not found: …:function:None`. Let the CLI - paginate and filter client-side. -- **P1 F10's "recurs forever" half is wrong**: `BootstrapVariant` now reads - `ABCA: Least-Privilege Bootstrap`, because P1's own - `update-stack --use-previous-template` with only `ComputeTypes` supplied resets - unspecified parameters to template defaults. The silent-`exit 0` half stands. -- **`--execution-role-arn` requires a full ARN** (regex in 6.1); a bare role name - is rejected. -- **`zsh`**: `"$VAR:latest"` is the `${VAR:l}` lowercase modifier (cost one - mis-diagnosed push); no `PIPESTATUS` (it is `$pipestatus`, 1-indexed), so P1's - `test "${PIPESTATUS[0]}" -eq 0` silently evaluates empty; no `timeout(1)`. -- **Verify secrets on the raw value, never a `.strip()`ed copy** — a trailing - newline from `gh auth token` produced a convincing false clone defect (3.2). -- **`finch` cannot push to ECR on this box**: `finch push` runs - `limactl shell finch sudo -E nerdctl push` and the VM's `DOCKER_CONFIG` uses - the `finchhost` creds helper, which cannot resolve an Isengard - `credential_process`; a host-side `finch login` does not help. Both `push` and - `pull` fail with `no basic auth credentials`. -- The `/terminate` hook body's **`microvmId` arrives empty** (`"microvm_id": ""`), - defeating the hook's stated guest↔control-plane correlation purpose. -- **`ARTIFACTS_BUCKET_NAME` and `TRACE_ARTIFACTS_BUCKET_NAME` resolve to the same - bucket**, making that separation notional. - -### Items SKIPPED, BLOCKED, or INCONCLUSIVE - -| Item | Verdict | Reason | -|---|---|---| -| **clone → change → PR (the point of Stage D)** | **FAILED — no PR** | P2-F5, reproduced twice. Stopped at two retries as briefed. | -| Dual-signal liveness / `agent_heartbeat_at` fresh during RUN | **BLOCKED behind P2-F5** | Heartbeat interval is 45 s; no task stayed `RUNNING` beyond ~13 s. `agent_heartbeat_at` was `None` on all six submissions. Not independently testable until F5 is fixed. | -| Suspend TTL beyond P1's ≥1 h floor | **SKIPPED (time-boxed)** | Would have added 2 h+ of wall clock for a marginal bound after ~2.5 h already spent on five blocking defects. P1's result stands unchanged: **survives ≥ 1 h 0 min 17 s with no `idlePolicy`**; a TTL between 1 h and the 8 h `maximumDurationInSeconds` bound remains **OPEN**. | -| SUSPENDED VM consumes account memory quota | **STILL NOT OBSERVABLE — undischarged** | Re-probed: `L-CD1C0CC4` = 1024 GB with `UsageMetric: null`; `AWS/Usage` has only `CallCount`; no memory metric in any namespace. Identical to P1 5.5. | -| CDK-managed `CfnMicrovmImage` deploy | **REFUTED (P2-F2)** | Fell back to the out-of-band script path, as the brief directed. | -| AgentCore container asset push | **WORKED AROUND** | finch/ECR auth (deviation 2). ECR manifest retagged; irrelevant to the MicroVM image under test. | -| P1 F11 (AgentCore-unsupported AZ) | **STILL UNFIXED** | `agent-vpc.ts` unchanged; worked around via the gitignored AZ context cache, restored at teardown. | -| P1 F12 Memory-delete half | **STILL UNFIXED** | `AWS::BedrockAgentCore::Memory` still undeletable while `CREATING`; still turns a rollback into `ROLLBACK_FAILED`; plain `delete-stack` still clears it. | -| Orchestrator-role IAM negatives | **NOT ATTEMPTED** | P1 5.0 established the Lambda trust does not allow operator assumption; trust was not modified for this purpose. | - -### Elapsed and approximate cost - -**Elapsed:** 22:38Z → 01:26Z ≈ **2 h 48 min**. Roughly: ~35 min on the finch/ECR -push dead end and the ECR-retag workaround; ~40 min on the five deploy attempts -plus two `ROLLBACK_FAILED`/`delete-stack` cycles; ~10 min on image create+build; -~35 min on the six submissions and the P2-F3 elimination sequence; ~10 min on -lifecycle extras; ~30 min on teardown (three delete attempts); the remainder on -evidence capture and this document. - -**Approximate cost: well under US$5**, dominated as in P1 by NAT/VPC-endpoint -hours rather than MicroVMs. - -| Item | Quantity | Est. | -|---|---|---| -| NAT gateway (1 × $0.045/h) | ~1.6 h across 3 stack lifetimes | ~$0.08 | -| Interface VPC endpoints (7 × $0.01/h × 2 AZ) | ~1.6 h | ~$0.22 | -| MicroVM runtime | 9 VMs, all short-lived (~12 s to ~3 min each); longest suspended window ~1 min | < $0.10 | -| MicroVM image builds | 2 builds (1 version × 2 chipsets), ~4.5 min each | low single-digit cents | -| Snapshot/image storage | ~3.6 GiB × ~1 h | negligible | -| AgentCore runtime + Memory | created 3×, **never invoked** | negligible | -| Bedrock | **zero model tokens** — every task died before turn 1 | $0 | -| S3 / DynamoDB / Lambda / API GW / Cognito / Secrets / logs | brief, mostly idle | < $1 | - -The 8 h `maximumDurationInSeconds` was never approached; every VM was either -explicitly terminated or service-reaped. - -### Recommended follow-up before P2 is called complete - -Ordered by what unblocks what: - -1. **P2-F1/F3** (`aws:SourceAccount` on MicroVM-facing role trust) — nothing - deploys or runs without this. One decision, three roles. -2. **P2-F5** (`claude --version` timeout) — nothing *completes* without this. It - is the only thing between this run and a PR, and the `/ready`-warming option - is worth considering on its merits rather than just raising the timeout. -3. **P2-F2** (L1 enum values) — the documented recommended path is dead until - fixed; five one-word changes plus the tests that assert the old strings. -4. **P2-F4** (application-log-group grant) — cheap, and it is what makes the next - failure debuggable through the platform rather than through guest stdout. -5. **P2-F7** (template size) — unrelated to MicroVMs but it will bite soon. -6. Re-run Stage D after 1–2. The dual-signal-liveness item and the suspend-TTL - extension both need a task that stays `RUNNING` for minutes, which only F5's - fix provides. - ---- - -## Stage D-redux (run 2) - -Narrow re-run on branch `feat/645-lambda-microvm-p2` @ -`b927d1d6a58e2b040c2ed4ce4e9f1dc9be9fc981` — the commit that fixed P2-F1..F5 -against run 1's evidence. **Purpose: convert "fixed-against-evidence" into -"re-exercised live", and get the pull request.** - -**Live run:** 2026-08-07 02:58Z → 04:55Z (≈1 h 57 min), account ``, -`us-east-1`. Evidence directory: `/tmp/abca-645-p2r2-20260806`. - -**Teardown: complete.** Live IAM workaround reverted, Cognito user deleted, image -+ version deleted, ECR retag removed, AZ context cache restored, 481/485 stack -resources deleted (4 zero-cost residual, #702), **all billable resources confirmed -gone**, and no global/host configuration touched. Both smoke PRs left open as -briefed. See 2.12. - -> **Provenance note — HEAD moved mid-run, from outside this run.** Preflight -> confirmed `HEAD = b927d1d` with a clean tree at 02:58Z. At **03:04Z** an -> unrelated user commit landed on the branch — `045722c fix(deps): refresh -> js-yaml lock entry to 4.3.1`, **`yarn.lock` only, 3 insertions / 3 deletions**. -> No `git` write command was issued by this run, and `b927d1d` remains an -> ancestor of `HEAD`. **It cannot have affected any finding:** `mise run install` -> / `yarn install` was never re-run, so `cdk/node_modules` still reflected -> `b927d1d`'s lock for every synth and deploy; the MicroVM artifact is built from -> `agent/` + `contracts/` + `agent/Dockerfile` (Python/uv), which the commit does -> not touch. Every verdict below is therefore against `b927d1d`'s tree. - -### 2.0 Headline - -> **THE SMOKE PASSED. `https://github.com/dreamorosi/batch-sync-triage/pull/6`** -> — clone → change → commit → push → PR, `COMPLETED`, 12 turns, $0.279, 153 s. -> Run 1's single most important line ("no PR was created") is retired. - -But the PR required **one live IAM workaround**, and establishing *why* produced -the run's most consequential result: **P2-F3 is NOT fixed, and run 1's -exoneration of its identity-side condition was a false negative.** Two of the -five P2 fixes are fully discharged, two are discharged, one is refuted, and one -brand-new blocking defect was found on the path run 1 never reached. - -| Fix | Run-1 verdict | Run-2 live verdict | -|---|---|---| -| **P2-F1** (connector trust) | blocking | ✅ **DISCHARGED** — substrate deployed **first try**, zero workarounds, 0 `CREATE_FAILED` in 485 resources | -| **P2-F2** (`ARM_64`/`ENABLED` enums) | REFUTED at change-set validation | ✅ **DISCHARGED** — early validation **passed**, resource reached `CREATE_IN_PROGRESS` | -| **P2-F3** (`RunMicrovm` PassRole) | blocking; "trust was the sole cause" | ❌ **NOT FIXED (P2r2-F10)** — isolated to the *identity-side* `iam:PassedToService` condition run 1 explicitly exonerated | -| **P2-F4** (application-log grant) | blocking observability | ✅ **DISCHARGED** — `server_debug/`, `metrics/` **and** `trajectory/` streams exist, with content | -| **P2-F5** (`claude` warm-up) | **the blocker** — no PR | ✅ **DISCHARGED** — cold `claude` measured at **17–38 s**, warm **0.1 s**, `/ready` 200, no 503 | -| *(new)* **P2r2-F9** | not reachable in run 1 | ❌ CDK-managed image path blocked: bootstrap `IAMPassRole` denies the build role to CloudFormation | -| *(new)* **P2r2-F11** | mis-attributed in run 1 | ⚠️ `agent_heartbeat_at` is never projected into the API response | -| Dual-signal liveness | BLOCKED behind F5 | ✅ **DISCHARGED** — 45 s cadence observed live over a 181 s `RUNNING` window | - -### 2.1 Deltas from run 1's setup - -```bash -export SMOKE_USER="" -export EVIDENCE_DIR=/tmp/abca-645-p2r2-20260806 -export BGAGENT_CONFIG_DIR=/tmp/abca-645-p2r2-bgagent -``` - -Everything else — the zsh traps, `set -o pipefail`, explicit `AWS_REGION`, -newest-first version ordering — carried over unchanged and all of it still -applies. Two run-1 notes paid for themselves immediately: the **raw-secret -assertion** (2.6) and **client-side pagination** for `TaskOrchestrator` lookup. - -`bgagent` was invoked as `node cli/lib/bin/bgagent.js` after -`mise //cli:compile` (there is no linked binary in this tree). - -### 2.2 Preflight — run 1's residual had to be cleared first, and it did not clear itself - -`backgroundagent-dev` was still `DELETE_FAILED` with run 1's exact 4-resource -residual, and **the two `agentic_ai` ENIs were still `in-use` 1 h 32 min later** -(`eni-04911e11e08d670f9`, `eni-04a1c5a27966ee08b`, both requester `amazon-aws`, -`InstanceOwnerId: amazon-aws`), while -`bedrock-agentcore-control list-agent-runtimes` and `list-memories` both returned -**empty**. So the ENIs outlive the resources that created them by a wide margin. - -Run 1's documented retry command was executed verbatim and **failed again after -17 min 17 s** (02:58:41Z → 03:15:58Z): - -``` -The following resource(s) failed to delete: [AgentVpcRuntimeSG96507CD0, AgentVpcPrivateSubnet1Subnet8051BB57, AgentVpcPrivateSubnet2SubnetC66971D0]. -``` - -**Correction to run 1's §8.5 advice.** Run 1 concluded from P1's experience that -"these are eventually released, so the retry command is `delete-stack`". That is -true on a multi-day horizon and **useless on a same-session horizon** — a -17-minute retry that fails identically is not a remedy. The remedy that works is -`--retain-resources`, which cleared the stack record in **33 seconds**: - -```bash -aws cloudformation delete-stack --stack-name backgroundagent-dev \ - --retain-resources AgentVpcA6796801 AgentVpcPrivateSubnet1Subnet8051BB57 \ - AgentVpcPrivateSubnet2SubnetC66971D0 AgentVpcRuntimeSG96507CD0 -``` - -Note the VPC (`CREATE_COMPLETE`, never attempted) must be listed alongside the -three `DELETE_FAILED` children or the delete fails on it. Cost: an orphaned -zero-cost VPC + 2 subnets + 1 SG, now outside CloudFormation's knowledge (2.12). - -*Unchanged from run 1:* `CDKToolkit` `UPDATE_COMPLETE`, `ComputeTypes = -agentcore,lambda-microvm`, `BootstrapVariant = ABCA: Least-Privilege -Bootstrap`, bootstrap SSM version `32`, five ABCA policies, no -`AdministratorAccess` — so **no re-bootstrap was run**. That decision turns out -to matter; see P2r2-F9. - -**Managed base image** — still exactly one in `us-east-1`, -`arn:aws:lambda:us-east-1:aws:microvm-image:al2023-1`, versions `1` (newest -first) and `0`. Selected `1`; the service echoed `baseImageVersion: "1.0"`. - -**P1 F11 is still UNFIXED** (`agent-vpc.ts` is still `maxAzs: 2` with no AZ -constraint, `us-east-1a` is still `use1-az6`), so the gitignored -`cdk/cdk.context.json` AZ cache was trimmed to lead with `us-east-1b`/`us-east-1c` -again, saved to `$EVIDENCE_DIR/cdk.context.json.ORIGINAL`, and **restored at -teardown**. `git status` stayed clean throughout. - -### 2.3 The AgentCore container asset — retag workaround reused, and the tag had moved - -`finch` still cannot push to ECR (run 1 deviation 2 — the Lima VM's -`credsStore: finchhost` cannot resolve an Isengard `credential_process`). The -required asset tag was **`7a71005f…`, not run 1's `34fbc1d4…`**, because -`b927d1d` grew `agent/src/server.py`; run 1's tag had been correctly removed at -its teardown. The same `aws ecr put-image` retag was applied to the existing -manifest `sha256:cdf5436a…`: - -``` -{"imageDigest": "sha256:cdf5436ab8d9d17bcd5b555ad22c52b4e6d6622f1fc0373cdc97a73c3eb8e6a4", - "imageTag": "7a71005fe6f520a2741cdd1fb2a47ffa919b13e63476226ec39bc66eb4c150c5"} -``` - -Same caveat, restated because it is easy to lose: **the deployed AgentCore -runtime therefore carries a stale agent image, and nothing in this run depends on -it** — the smoke runs on the MicroVM substrate built from the S3 zip. The finch VM -was never started this run, and `~/.finch/config.json` was never touched -(verified still `credsStore: osxkeychain` with no stored credential). - -### 2.4 Substrate deploy — P2-F1 DISCHARGED, first try, zero workarounds - -Synth: `abca:microvm-image-not-provisioned` emitted, `…-p1-smoke-unverified` -correctly absent. Deploy `03:19:25Z → 03:33:59Z = 14 min 34 s`, **`EXIT=0` on the -first attempt**, 485 resources. - -**This is the whole of P2-F1's verification and it is unambiguous.** Run 1 needed -five attempts and a `/tmp` cloud-assembly trust-policy patch to get here. Run 2 -needed none: - -``` -2026-08-07T03:21:47Z LambdaMicrovmComputeEgressConnector9C36AAC2 CREATE_IN_PROGRESS -2026-08-07T03:21:48Z LambdaMicrovmComputeBuildEgressConnector3B762F80 CREATE_IN_PROGRESS -2026-08-07T03:26:05Z LambdaMicrovmComputeEgressConnector9C36AAC2 CREATE_COMPLETE -2026-08-07T03:26:06Z LambdaMicrovmComputeBuildEgressConnector3B762F80 CREATE_COMPLETE -``` - -and a scan of every stack event for `*FAILED` returned **`NONE`**. The live trust -policies on both the execution and build roles read exactly as source intends — -`lambda.amazonaws.com`, `sts:AssumeRole` + `sts:TagSession`, **no conditions**. - -All seven MicroVM outputs populated; `ComputeSubstrate = lambda-microvm`; -artifact key exactly `microvm-images/agent-artifact.zip`. - -**P2-F7 reconfirmed and marginally worse:** 979,867 B / 485 resources (substrate), -**985,886 B / 486** with the image — **98.6 %** of the 1 MB limit, ~14 KB of -headroom. Down from run 1's ~16 KB. - -### 2.5 The CDK-managed image path — P2-F2 DISCHARGED, then blocked by a NEW defect (P2r2-F9) - -The synthesized L1 now carries exactly the five values CloudFormation rejected in -run 1: - -```json -"CpuConfigurations": [{ "Architecture": "ARM_64" }], -"Hooks": { - "Port": 8080, - "MicrovmHooks": { "Run": "ENABLED", "RunTimeoutInSeconds": 60, - "Terminate": "ENABLED", "TerminateTimeoutInSeconds": 15 }, - "MicrovmImageHooks": { "Ready": "ENABLED", "ReadyTimeoutInSeconds": 300, - "Validate": "ENABLED", "ValidateTimeoutInSeconds": 60 } -} -``` - -**P2-F2 is DISCHARGED.** Change-set early validation **passed** — zero -`not a valid enum value` errors, no `Early validation failed` — and the resource -progressed to `CREATE_IN_PROGRESS`. Run 1's §1.5 refutation is fully answered and -the enum fix is correct on the CloudFormation surface. - -It then failed on something run 1 could never have seen, because run 1 never got -past early validation: - -``` -LambdaMicrovmComputeImage16B48539 CREATE_FAILED -Resource handler returned message: "User: arn:aws:sts:::assumed-role/cdk-hnb659fds-cfn-exec-role--us-east-1/AWSCloudFormation is not authorized to perform: iam:PassRole on resource: arn:aws:iam:::role/backgroundagent-dev-LambdaMicrovmComputeBuildRoleF0-9FxjQbiJC3px because no identity-based policy allows the iam:PassRole action (Service: LambdaMicrovms, Status Code: 403, Request ID: f8c45ab5-49c4-42fd-ac61-a6a3ac00dc26) (SDK Attempt Count: 1)" HandlerErrorCode: AccessDenied -``` - -`UPDATE_ROLLBACK_COMPLETE`; the substrate survived intact (all seven outputs still -populated), so this cost one deploy cycle and nothing else. - -**Diagnosis — and it is NOT a stale bootstrap.** The live -`…IaCRole-ABCA-Infrastructure…` policy is **byte-identical to -`cdk/bootstrap/policies/infrastructure.json` on this branch**, so -`bootstrap --force` would change nothing. Its `IAMPassRole` statement is: - -```json -{ "Sid": "IAMPassRole", "Effect": "Allow", "Action": "iam:PassRole", - "Resource": "arn:aws:iam::*:role/backgroundagent-dev-*", - "Condition": { "StringEquals": { "iam:PassedToService": [ - "lambda.amazonaws.com", "ecs-tasks.amazonaws.com", "ecs.amazonaws.com", - "apigateway.amazonaws.com", "logs.amazonaws.com", "bedrock.amazonaws.com", - "bedrock-agentcore.amazonaws.com", "events.amazonaws.com", - "vpc-flow-logs.amazonaws.com" ] } } } -``` - -The resource pattern matches the build role. `simulate-principal-policy` on the -deploy role returns **`allowed`** with -`iam:PassedToService=lambda.amazonaws.com` and **`implicitDeny`** with no context -— so the resource is right and the *condition* is the only remaining variable. - -**The control that makes this airtight:** the out-of-band `--create-image` path -(2.6) passed **the same build role** to the same service successfully, using -operator credentials that carry no such condition. So the build role's trust is -fine and the service can assume it; the denial is genuinely caller-side. - -This directly contradicts the construct's own accounting, which asserts -`buildRole` is passed "via the bootstrap `infrastructure` policy's `IAMPassRole`" -— live, it is not. - -*Incidental:* **CloudTrail has no `lambda-microvms` events at all** for this run -(`lookup-events` across the window returns nothing for `CreateMicrovmImage` / -`RunMicrovm`). The service appears not to emit management events yet, which is why -the actual `iam:PassedToService` value cannot simply be read out of a log — every -determination in this section had to be made by elimination. - -### 2.6 Image + build — P2-F5 DISCHARGED, with numbers - -Artifact: **935,874 bytes**, SSE `AES256` (run 1: 922,056). `--create-image` exit -**0**; the service echoed `ARM_64`, `minimumMemoryInMiB: 8192`, all four hooks -`ENABLED`, and `readyTimeoutInSeconds: 300`. - -Build `03:42:53Z → 03:47:23Z = 4 min 30 s`, both chipsets -(`GRAVITON` generation `3` and `4`) `SUCCESSFUL`, `state=SUCCESSFUL`, -`status=ACTIVE`. **The warm-up cost the build ~22 s** versus run 1's 4 min 35 s -hook-less-warm-up baseline — an outstanding trade for what it buys. - -**The decisive evidence, verbatim** (per build stream): - -``` -[server/build-hook] /ready hook: server is up, warming the snapshot before it is taken -[server/build-hook] /ready hook: warmed 'claude' in 17.2s (version='2.1.191 (Claude Code)') -[server/build-hook] /ready hook: warmed 'claude' in 37.8s (version='2.1.191 (Claude Code)') -[server/build-hook] /ready hook: warmed 'git' in 0.0s (version='git version 2.47.3') -[server/build-hook] /ready hook: warmed 'claude' in 0.1s (version='2.1.191 (Claude Code)') -[server/build-hook] /ready hook: warmed 'node' in 3.0s (version='v24.19.0') -[server/build-hook] /ready hook: reporting ready for snapshot -INFO: 127.0.0.1:33946 - "POST /aws/lambda-microvms/runtime/v1/ready HTTP/1.1" 200 OK -[server/build-hook] /validate hook: ok (python=3.13.13, platform_config_keys=13, warnings=0) -INFO: 127.0.0.1:53154 - "POST /aws/lambda-microvms/runtime/v1/validate HTTP/1.1" 200 OK -``` - -**Four independent things are proven here, and three of them are new:** - -1. **The lazy-hydration diagnosis was correct, and now it is measured.** A cold - `exec` of the 225 MiB `claude` binary takes **17.1–37.8 s**. Run 1 inferred - this from a 10 s timeout; run 2 has the number. **`timeout=10` could never - have passed** — it was 2–4× short. -2. **The warm-up works.** The same binary in the same guest answers in **0.1 s** - once its pages are faulted in. That is the mechanism doing exactly what the - `/ready` docstring claims. -3. **The service issues `/ready` three times per build, and the first two run - concurrently.** Both concurrent calls warm `claude` simultaneously and contend - (17.2 s and 37.8 s in the same stream). Worst-case single call ≈ 37.8 + 0.0 + - 6.4 ≈ **44 s**. So the 120 s required budget carries ~3.2× margin and the move - from `readyTimeoutInSeconds` 60 → 300 was not merely defensive — a 60 s hook - budget would have had ~16 s of slack against a contended cold start. -4. **No 503, ever.** The "required warm-up failed" path never fired, and - `/validate` still reports `platform_config_keys=13, warnings=0`. - -*Not re-captured:* the 2.3 size table (`codeInstallSizeInBytes` etc.) — the image -was deleted at teardown before those fields were read. Run 1's figures stand; -this run adds nothing to or against P1 F13. - -**Wired deploy** with `--context microvm_image_identifier=`: -`03:48:34Z → 03:53:37Z = 5 min 3 s`, `UPDATE_COMPLETE`. - -**`MICROVM_*` is FIVE vars this run, not run 1's six**, and that is **correct by -design**: `MICROVM_IMAGE_VERSION` is emitted only when the optional -`microvm_image_version` context is supplied (`agent.ts:245` → -`task-orchestrator.ts:467`), and the construct documents its absent state as -"let the service pick". Run 1 recorded six because it supplied the version. -Not a regression — but worth stating, because "all-or-nothing" is asserted of the -`MICROVM_*` block and the version is the one member that legitimately opts out. - -`platform_config` wiring reconfirmed at **11 of 13** keys, with -`LINEAR_OAUTH_SECRET_ARN` / `JIRA_OAUTH_SECRET_ARN` correctly absent. - -**P2-F4's grant is present on the live execution role** — the second statement is -new versus run 1: - -```json -{ "Action": ["logs:CreateLogStream","logs:PutLogEvents"], - "Resource": "arn:aws:logs:us-east-1::log-group:/aws/vendedlogs/bedrock-agentcore/runtime/APPLICATION_LOGS/backgroundagent-dev:*", - "Effect": "Allow" } -``` - -### 2.7 Platform user, secret, onboarding — all PASS - -`admin invite-user` → `CONFIRMED` with a permanent password; `configure ---stack-name`; `login` → `Login successful.` Exactly run 1's two-command flow. - -**The PAT was written correctly this time** — `gh auth token | tr -d '\n\r'` into -`--secret-string file:///dev/stdin`, then asserted on the **raw** value per run -1's §3.2 lesson: - -``` -len 40 differs_from_stripped False prefix gho_ -``` - -Run 1's self-inflicted 41-byte defect did not recur. `repo onboard ---compute-type lambda-microvm` → `status: active`, both the ComputeSubstrate gate -and the live `ListManagedMicrovmImages` probe passing; `platform doctor` → **7/7**. - -### 2.8 THE SMOKE — five submissions, and the last two are an experiment - -| # | Task ID | Orchestrator `iam:PassRole` grant in force | Settle | Outcome | -|---|---|---|---|---| -| 1 | `01KZD5SZ55N707GMWY14P09JGR` | **source only** (exact ARN + `iam:PassedToService`) | ~2.5 min | `FAILED` — PassRole denial | -| 2 | `01KZD61FHYH3QT774PFETDNVKY` | + unconditioned, exact ARN | 20 s | `FAILED` — identical | -| 3 | `01KZD6D95P5VXJTJ06BFJXF8J6` | + unconditioned, `backgroundagent-dev-*` | 3.5 min | ✅ **`RUNNING` in 5 s → `COMPLETED` → PR #6** | -| 4 | `01KZD71DJSTGZK2A91MZJ8GFHJ` | **source only** (workaround removed) | **5 min** | `FAILED` — identical → **control** | -| 5 | `01KZD7D7HD3CT2APBT3DH12XFE` | + unconditioned, **exact ARN** (source's resource) | **5 min** | ✅ **`RUNNING` in 9 s → `COMPLETED` → PR #7** | - -**Submissions 4 and 5 are the isolation experiment, and they are the most -important result in this document.** Same exact-ARN resource as source, same -5-minute settle, one variable — the `iam:PassedToService: lambda.amazonaws.com` -condition. With it: denied. Without it: `RUNNING` in 9 s. See P2r2-F10. - -#### 2.8.1 THE SMOKE (submission 3), `bgagent watch`, verbatim highlights - -``` -[9:07:17 PM] ★ repo_setup_complete: branch=bgagent/01KZD6D95P5VXJTJ06BFJXF8J6/add-a-codeowners-file-at-the-repository-root-conta build_before=False -[9:07:18 PM] ★ step:implement:start -[9:08:14 PM] ★ pre_approvals_loaded {"scopes":[],"count":0} -[9:08:31 PM] Turn #1 (claude-opus-4-8, 0 tool calls) - Text: This is a simple, well-defined task. Let me create the CODEOWNERS file. -[9:08:37 PM] ▶ Write: {'file_path': '/workspace/01KZD6D95P5VXJTJ06BFJXF8J6/CODEOWNERS', 'content': '* @dreamorosi\n'} -[9:09:27 PM] ▶ Bash: git add CODEOWNERS && git commit -m "chore(github): add CODEOWNERS file" && git push -u origin bgagent/… -[9:09:43 PM] ▶ Bash: gh pr create --repo dreamorosi/batch-sync-triage --head bgagent/… --base main --title "chore(github): add CODEO… -[9:09:45 PM] ◀ Bash: https://github.com/dreamorosi/batch-sync-triage/pull/6 -[9:09:50 PM] Cost: $0.2792 (1947 in / 3554 out tokens) -[9:09:50 PM] ★ step:implement:succeeded -[9:09:50 PM] ★ agent_execution_complete: status=success turns=22 -[9:09:50 PM] ★ pr_created: https://github.com/dreamorosi/batch-sync-triage/pull/6 -Task 01KZD6D95P5VXJTJ06BFJXF8J6 completed. -``` - -Final record: `status COMPLETED`, `duration_s 153.4`, `cost_usd 0.279`, -`turns_completed 12`, `build_passed true`, `lint_passed false`, -`session_id microvm-082ebde7-…`. **Time-to-`RUNNING` 5 s**, confirming run 1's -headline performance result on a task that actually finishes. - -*Minor inconsistency worth a glance:* `watch` prints -`agent_execution_complete: turns=22` while the persisted record and -`METRICS_REPORT` both say `turns: 12`. Two different notions of "turn" (SDK -messages vs. counted iterations) surfaced under one word in the same stream. - -*Incidental, both tasks:* `lint_passed: false` is scratch-repo noise -(`tsc: not found`, `biome` schema mismatch, `mise ERROR no tasks defined`), and -the agent correctly reasoned about it as pre-existing before proceeding. - -#### 2.8.2 P2-F4 — DISCHARGED with content, not just stream existence - -The brief's test was whether `server_debug/` now *exists*. It does, and -so does more: - -``` -metrics/01KZD6D95P5VXJTJ06BFJXF8J6 -server_debug/01KZD6D95P5VXJTJ06BFJXF8J6 -server_debug/server -trajectory/01KZD6D95P5VXJTJ06BFJXF8J6 -``` - -All three per-task streams were `AccessDenied` in run 1. `server_debug` carries -the `/run` breadcrumbs — including the **11-key** `platform_config` install line, -names only — and `metrics/` carries the `METRICS_REPORT` that run 1 lost -entirely: - -```json -{"event": "METRICS_REPORT", "status": "success", "agent_status": "success", - "pr_url": "https://github.com/dreamorosi/batch-sync-triage/pull/6", - "build_passed": true, "lint_passed": false, "cost_usd": 0.27916625, - "turns": 12, "duration_s": 153.4, "task_id": "01KZD6D95P5VXJTJ06BFJXF8J6", - "disk_before": "167.5 KB", "disk_after": "261.0 MB", …} -``` - -**P2-F4 is fully discharged**, and the platform's canonical per-task -observability is no longer empty on this backend. - -#### 2.8.3 Dual-signal liveness — DISCHARGED, and run 1's verdict was partly an artifact - -Polled **against DynamoDB** during submission 5's `RUNNING` window: - -| Wall clock | Status | Running for | `agent_heartbeat_at` | Freshness | -|---|---|---|---|---| -| 04:25:11Z | `RUNNING` | 71 s | `2026-08-07T04:24:47Z` | 24 s | -| 04:25:38Z | `RUNNING` | 99 s | `2026-08-07T04:25:32Z` | **6 s** | -| 04:26:05Z | `RUNNING` | 126 s | `2026-08-07T04:25:32Z` | 34 s | -| 04:26:33Z | `RUNNING` | 153 s | `2026-08-07T04:26:17Z` | 16 s | -| 04:27:00Z | `COMPLETED` | 181 s | `2026-08-07T04:26:17Z` | 44 s | - -Successive values are `04:24:47 → 04:25:32 → 04:26:17`: **exactly the 45 s -`_HEARTBEAT_INTERVAL_SECONDS` cadence**, freshness never worse than 34 s while -`RUNNING`. **The dual-signal liveness path is DISCHARGED on `lambda-microvm`** — -`heartbeatLivenessApplies` has a real, fresh signal to read. Progress events -streamed live in `watch` concurrently (2.8.1). - -**But this only became visible by reading DynamoDB directly.** `bgagent status` -reported `agent_heartbeat_at = None` on every poll of *both* completed tasks — -including submission 3, whose stored value was `2026-08-07T04:09:39Z`, 12 s before -`completed_at`. Cause: `toTaskDetail` (`cdk/src/handlers/shared/types.ts:786`) -never maps the field, though `TaskRecord` declares it at line 95 and -`orchestrator.ts` consumes it for liveness. See P2r2-F11 — and note this makes -run 1's "heartbeats NOT observed" partly a measurement artifact rather than a -pure consequence of P2-F5. - -### 2.9 Lifecycle — PASS - -Both smoke MicroVMs finalized cleanly: `state TERMINATED`, `stateReason -Success.`, `maximumDurationInSeconds 28800` never approached. `NO_INGRESS` -reconfirmed on both, and **run 1's caveat holds** — a `NO_INGRESS` VM still -returns a public endpoint hostname -(`76cfed33-…lambda-microvm.us-east-1.on.aws`), so "an endpoint exists" remains no -evidence of reachability. - -`list-microvms` showed **11** VMs, all `TERMINATED` — run 1's 9 plus this run's 2, -so the list is cumulative across runs and none of run 1's ever resurfaced. - -### 2.10 Not attempted this run - -Deliberately out of the narrow scope: suspend/resume and suspend-TTL (run 1 §6.2 -covered a hook-enabled image), the SUSPENDED-vs-quota probe (still -`UsageMetric: null` territory), and run 1's §6.1 run-hook-4xx self-termination -(P2-F6 — already recorded, and `b927d1d` corrected the ADR for it). - -### 2.11 Scratch repo — TWO PRs left open - -``` -#7 docs(contributors): add CONTRIBUTORS.md | bgagent/01KZD7D7HD3CT2APBT3DH12XFE/… -#6 chore(github): add CODEOWNERS file | bgagent/01KZD6D95P5VXJTJ06BFJXF8J6/… -#1–#5 pre-existing dependabot PRs -``` - -**#6 is the briefed smoke and is left open as instructed.** **#7 is a by-product -of the 2.8 isolation experiment** (submission 5 needed a task that would actually -run, and a second distinct file avoided colliding with #6) and is left open -alongside it rather than closed, so the evidence for the experiment survives. -Two `bgagent/*` branches were pushed; `main` is untouched. - -### 2.12 Teardown (executed as a finally-block) - -| Step | Result | -|---|---| -| **Live IAM workaround** | `abca645p2r2-verification-passrole` **deleted** from the orchestrator role; `list-role-policies` shows only the CDK-managed default policy. **No trust policy was modified at any point this run** (verified: execution role still `sts:AssumeRole` + `sts:TagSession`, conditionless, exactly as source produces). | -| **MicroVMs** | All 11 `TERMINATED` before teardown began; none needed chasing. | -| **Image + version** | Single version `1.0`; `delete-microvm-image` → `DELETING`, reaped it. | -| **Cognito user** | `admin delete-user` → `✓ Deleted`; `list-users` → `[]`; invite file `rm`'d. | -| **Secret** | Left to die with the stack. | -| **`cdk/cdk.context.json`** | **Restored** to the original six-AZ list; `git status` shows only `docs/verification/` + pre-existing `opencode.json`. | -| **ECR retag** | `7a71005f…` removed with `batch-delete-image`; pre-existing tags and the underlying manifest untouched. | -| **finch** | Never started this run; `~/.finch/config.json` never touched (still `credsStore: osxkeychain`, no stored credential). | -| **Global/host config** | **`~/.aws/*` never read or modified.** All credential and Region selection via environment variables in the run's own shell. `~/.cdk.json` never created. | - -**Stack — two delete attempts, ending at run 1's exact residual:** - -| Attempt | Window | Outcome | -|---|---|---| -| 1 | 04:29:02Z → 04:37:24Z (8 min 22 s) | `DELETE_FAILED` — `AWS::BedrockAgentCore::Runtime`: `"Request timed out while deleting AWS::BedrockAgentCore::Runtime"`, `HandlerErrorCode: NotStabilized`. **19** resources left. | -| 2 (informed) | 04:37:50Z → 04:55:17Z (17 min 27 s) | `DELETE_FAILED` — but the Runtime, Memory, all IAM roles, all DynamoDB tables, the S3 bucket and the secret **did** delete. **4** resources left. | - -Attempt 2 was not a blind retry: `bedrock-agentcore-control list-agent-runtimes` -returned **empty** first, proving the runtime was already gone server-side and the -failure was a handler stabilization artifact. **Run 1's attempt-3 technique -reproduced exactly and is confirmed as the right procedure** — it took the -residual from 19 to 4. - -**Final residual — 4 resources, all zero-cost, identical in shape to run 1 and P1:** - -``` -CREATE_COMPLETE AWS::EC2::VPC vpc-07dad8897791f477b -DELETE_FAILED AWS::EC2::Subnet subnet-09a1ff1f7568d05ef -DELETE_FAILED AWS::EC2::Subnet subnet-068397b48a25bf13f -DELETE_FAILED AWS::EC2::SecurityGroup sg-05e3f48665c47b358 -``` - -Pinned by two fresh `agentic_ai` ENIs (`eni-0942f1b0b3b6553ea`, -`eni-07876963c2818da63`, both `in-use`). **#702 / P1 F12 reproducing for the third -consecutive run.** Nothing was force-deleted past CloudFormation. - -**Billing confirmed stopped:** - -| Check | Result | -|---|---| -| NAT gateways in either orphaned ABCA VPC | **none** | -| VPC endpoints | **none** | -| ABCA S3 buckets | **none** | -| Unattached (billable) EIPs | **none** | -| MicroVM images / non-`TERMINATED` VMs | **none** | -| AgentCore runtimes / memories | **none** | -| `/aws/lambda-microvms/*` log groups | **none** | -| ABCA DynamoDB tables / GitHub-token secret | **none** | - -**Deliberately retained:** `CDKToolkit` (unchanged — still the shared-account -caveat P1 raised); bootstrap S3/ECR assets; service-vended log groups created -outside CloudFormation; the two scratch-repo PRs (2.11); and **two** orphaned -zero-cost VPC sets — run 2's four resources above (still inside the -`DELETE_FAILED` stack) plus **run 1's**, which `--retain-resources` moved outside -CloudFormation entirely (`vpc-072fddf653ccdcfc4`, `subnet-0e6a6a0ed18100c8a`, -`subnet-02d91450e51cf72a0`, `sg-00997a58c1f4c5775`). - -A best-effort hand cleanup of run 1's set was attempted and **refused** — -`DependencyViolation` on both subnets, the SG and the VPC, because -`eni-04911e11e08d670f9` and `eni-04a1c5a27966ee08b` were **still `in-use` -3 h 30 min after run 1's teardown**. Nothing was forced. Retry for both sets: - -```bash -aws cloudformation delete-stack --stack-name backgroundagent-dev # run 2's set -# run 1's set is no longer CFN-managed: -aws ec2 delete-subnet --subnet-id subnet-0e6a6a0ed18100c8a -aws ec2 delete-subnet --subnet-id subnet-02d91450e51cf72a0 -aws ec2 delete-security-group --group-id sg-00997a58c1f4c5775 -aws ec2 delete-vpc --vpc-id vpc-072fddf653ccdcfc4 -``` - -### 2.13 Findings summary - -Live run 2026-08-07 02:58Z → 04:55Z, account ``, `us-east-1`, branch -`feat/645-lambda-microvm-p2` @ `b927d1d6`. Evidence: -`/tmp/abca-645-p2r2-20260806`. - -**Verdict on the primary objective: THE SMOKE PASSED. -`https://github.com/dreamorosi/batch-sync-triage/pull/6`.** Clone → change → -commit → push → PR, `COMPLETED` in 153 s for $0.28, with progress events and a -live heartbeat. That retires run 1's headline. **It required one live IAM -workaround, and pinning down why is the run's most valuable output.** - -#### Fixes CONVERTED to "re-exercised live" - -1. **P2-F1 — DISCHARGED.** Substrate deployed **first try**, `EXIT=0`, - 14 min 34 s, **zero workarounds**, both `AWS::Lambda::NetworkConnector` - resources `CREATE_COMPLETE`, and **no `*FAILED` event anywhere** in 485 - resources. Run 1 needed five attempts and a cloud-assembly patch. -2. **P2-F2 — DISCHARGED.** Change-set **early validation passed** with `ARM_64` - and four `ENABLED` hooks; the resource reached `CREATE_IN_PROGRESS`. Run 1's - five-value refutation is answered on the CloudFormation surface. (The path is - still blocked downstream — P2r2-F9 — but *not* on the enums.) -3. **P2-F4 — DISCHARGED with content.** `server_debug/`, - `metrics/` **and** `trajectory/` all exist; the - `METRICS_REPORT` run 1 lost entirely now lands. -4. **P2-F5 — DISCHARGED, and now quantified.** Cold `claude` exec measured at - **17.1–37.8 s** (so `timeout=10` was 2–4× short and could never have passed); - **0.1 s once warm**; `/ready` 200, no 503; `/validate` still - `platform_config_keys=13, warnings=0`; whole build 4 min 30 s, i.e. the - warm-up cost ~22 s. **New empirical detail:** the service calls `/ready` - **three times per build, the first two concurrently**, so two cold `claude` - execs contend — worst single call ≈ 44 s. The 120 s required budget and the - `readyTimeoutInSeconds` 60 → 300 move are both correctly sized; 60 s would - have left ~16 s of slack. -5. **Dual-signal liveness — DISCHARGED** (run 1: BLOCKED). `agent_heartbeat_at` - advanced `04:24:47 → 04:25:32 → 04:26:17` — exact 45 s cadence — across a - 181 s `RUNNING` window, freshness ≤ 34 s. - -#### Items CONTRADICTING design assumptions — `feeds-back-to-design: YES` - -**P2r2-F10 (BLOCKING). P2-F3 is NOT fixed. The orchestrator's *identity-side* -`iam:PassedToService: lambda.amazonaws.com` condition — the one run 1 explicitly -exonerated and `b927d1d` deliberately kept — is a second, independent blocker.** - -Isolated by a clean two-arm experiment, same exact-ARN resource as source, same -5-minute settle, one variable: - -| Orchestrator grant | Result | -|---|---| -| exact ARN **+ `iam:PassedToService: lambda.amazonaws.com`** (source as written) | **DENIED** — submissions 1 *and* 4 | -| exact ARN, **no condition** | **`RUNNING` in 9 s** — submission 5 | - -``` -Session start failed: Error: MicroVM RunMicrovm failed: AccessDeniedException: User: arn:aws:sts:::assumed-role/backgroundagent-dev-TaskOrchestratorOrchestratorFnS-a7sP6rFzoIkU/backgroundagent-dev-TaskOrchestratorOrchestratorFn-p41lJqwFmxNG is not authorized to perform: iam:PassRole on resource: arn:aws:iam:::role/backgroundagent-dev-LambdaMicrovmComputeExecutionRo-pZQWXvKsITBa because no identity-based policy allows the iam:PassRole action -``` - -The trust half of run 1's fix was **necessary but not sufficient**. Run 1's -exoneration was a **false negative with an identifiable cause**: its temporary -unconditioned `iam:PassRole` was attached at §4.1 step 1 and, per its own §8.3, -**was still attached through submissions 4 and 5** — the ones that reached -`RUNNING`. So run 1 never tested the conditioned grant against a *working* trust, -and attributed the whole effect to the trust change. - -Two source statements must therefore change, and the comment in -`task-orchestrator.ts` asserting the condition "was EXONERATED live … so it -stays" must be reversed: - -- `cdk/src/constructs/task-orchestrator.ts`, sid `MicrovmPassExecutionRole` -- `cdk/bootstrap/policies/infrastructure.json`, sid `IAMPassRole` (P2r2-F9) - -**P2r2-F9 (BLOCKING, same root cause). The CDK-managed image path is dead one -step later than run 1 thought: CloudFormation cannot pass the build role.** -Verbatim `CREATE_FAILED` in 2.5. **This is not a stale bootstrap** — the live -policy is byte-identical to this branch's -`cdk/bootstrap/policies/infrastructure.json`. `simulate-principal-policy` returns -`allowed` for `iam:PassedToService=lambda.amazonaws.com` and `implicitDeny` -without it, so the resource pattern is right and the condition is the variable; -and the out-of-band `--create-image` call **succeeded passing the same build -role**, proving the trust is fine and the denial is caller-side. - -**P2r2-F9 and P2r2-F10 are ONE root cause with TWO symptoms** — precisely the -shape of run 1's P2-F1/F3, one layer in: **the Lambda MicroVMs service does not -present `iam:PassedToService: lambda.amazonaws.com` on either `PassRole` path** -(CloudFormation → build role at `CreateMicrovmImage`; orchestrator → execution -role at `RunMicrovm`). Consequence for the docs: the "CDK-managed (recommended)" -bootstrap path in `package-microvm-artifact.sh` is **still** non-functional and -`--create-image` is **still** the only working path — for a new reason. - -**P2r2-F11. `agent_heartbeat_at` is written and consumed correctly but is -invisible through the API.** `toTaskDetail` -(`cdk/src/handlers/shared/types.ts:786`) does not map it, though `TaskRecord` -declares it (line 95) and `orchestrator.ts` reads it for liveness; `cli/src/types.ts` -has no such field at all. So `bgagent status` / `watch` report `None` even when -DynamoDB holds a 6-second-old value. Cheap to fix, and worth fixing because **it -already caused a wrong conclusion**: run 1 recorded "heartbeats NOT observed" and -attributed it wholly to P2-F5. - -**P2r2-F12. Run 1's §8.5 stack-delete retry advice does not work on a -same-session horizon, and the ENI leak is worse than #702 records.** Run 1's -documented `delete-stack` retry failed **identically after 17 min 17 s**, with the -`agentic_ai` ENIs still `in-use` 1 h 32 min after run 1's teardown and *no* -AgentCore runtimes or memories in existence. At the end of *this* run those same -two ENIs were **still `in-use` 3 h 30 min on**, and a hand `delete-subnet` / -`delete-security-group` / `delete-vpc` sweep was refused with -`DependencyViolation` on all four. So the leak is not "slow" — it is **unbounded -relative to a developer's session**, and it now compounds: each run strands -another VPC set. The working escape hatch is `--retain-resources` (33 s), which -must list the `CREATE_COMPLETE` VPC alongside the three `DELETE_FAILED` children: - -```bash -aws cloudformation delete-stack --stack-name backgroundagent-dev \ - --retain-resources AgentVpcA6796801 AgentVpcPrivateSubnet1Subnet8051BB57 \ - AgentVpcPrivateSubnet2SubnetC66971D0 AgentVpcRuntimeSG96507CD0 -``` - -#702 should carry both the unbounded-hold evidence and this escape hatch, rather -than have each run rediscover them. Run 1's attempt-3 technique (verify -`list-agent-runtimes` is empty, *then* retry) is separately **confirmed correct** -— it took this run's residual from 19 resources to 4. - -**P2r2-F13. `MICROVM_IMAGE_VERSION` is legitimately optional** — five -`MICROVM_*` vars this run vs run 1's six, because the optional -`microvm_image_version` context was not supplied. Correct by design -(`task-orchestrator.ts:467`), but it qualifies the "all-or-nothing `MICROVM_*`" -claim, which should name the version as the one member that may be absent. - -**P2r2-F14. `turns` is reported inconsistently in one stream.** `watch` prints -`agent_execution_complete: turns=22`; the persisted record and `METRICS_REPORT` -both say `12`. - -**P2-F7 reconfirmed, worse.** 985,886 B / 486 resources = **98.6 %** of the 1 MB -limit, ~14 KB of headroom (run 1: ~16 KB). - -**CloudTrail blind spot.** No `lambda-microvms` management events at all, so -`PassRole` context values cannot be read from logs — every determination in -2.5/2.8 had to be by elimination. Worth knowing before the next person tries. - -**Still unfixed from earlier runs:** P1 F11 (AZ constraint — worked around again -via the gitignored AZ cache, restored at teardown); the finch→ECR push failure; -P1 F8 (`TERMINATED` is terminal, `ResourceNotFoundException` never arrives). - -#### Recommended follow-up - -1. **P2r2-F10 + P2r2-F9 together** — drop or correct `iam:PassedToService` on - both the orchestrator statement and the bootstrap `IAMPassRole`. Nothing runs - without the first; the recommended image path stays dead without the second. - Determining the value the service *does* present needs either AWS - confirmation or a bounded candidate sweep — `microvms.lambda.amazonaws.com`, - `lambda-microvms.amazonaws.com` and `microvms.amazonaws.com` are all - `implicitDeny` against the current policy, so any of them would work as the - allow-list entry if it is the right one. Reverse the "EXONERATED … so it - stays" comment while you are there. -2. **P2r2-F11** — one line in `toTaskDetail` plus the CLI type; it is what makes - the liveness signal observable to the people who need it. -3. **P2r2-F12** — put the `--retain-resources` escape hatch in #702. -4. **P2-F7** — `suppressTemplateIndentation` or a stack split; ~14 KB left. -5. **A third Stage D is NOT needed for P2-F1/F2/F4/F5 or dual-signal liveness** — - all five are now live-verified. The next run's scope is P2r2-F9/F10 plus the - still-deferred empirical items (suspend TTL > 1 h, SUSPENDED-vs-quota). - -### 2.14 Elapsed and cost - -**Elapsed:** 02:58Z → 04:55Z ≈ **1 h 57 min**, of which ~18 min was the failed -delete retry, ~15 min the substrate deploy, ~6 min the refuted CDK-managed image -attempt, ~10 min image create+build, ~5 min the wired deploy, ~30 min the five -submissions and the isolation experiment (mostly IAM settle waits), ~26 min the -two teardown delete attempts, and the remainder evidence capture. - -**Approximate cost: well under US$5**, dominated as always by NAT/VPC-endpoint -hours. Unlike run 1 this run actually spent Bedrock tokens: **$0.478 total across -two completed tasks** ($0.279 + $0.199). Two MicroVMs ran ~2.5 min each; two -image builds ~4.5 min each; one NAT gateway and 7×2 interface endpoints for -~1.6 h across a single stack lifetime. - -The 8 h `maximumDurationInSeconds` was never approached; every VM was explicitly -terminated by the orchestrator's finalization. diff --git a/docs/verification/645-p2-start-recovery-live-20260914.md b/docs/verification/645-p2-start-recovery-live-20260914.md deleted file mode 100644 index 6c4a93b7b..000000000 --- a/docs/verification/645-p2-start-recovery-live-20260914.md +++ /dev/null @@ -1,209 +0,0 @@ -# P2 live MicroVM start and recovery verification - -## Result - -On 2026-09-14, the AWS service replay checks and **13 application start/recovery -cases passed** against the existing `backgroundagent-dev` deployment in -`us-west-2`, account ``, profile `sphia-dev`, image `2.0`. -The deployed source remains `e1d5debe`, with bootstrap policy bundle `1.7.0`. - -This extends the [payload verification](./645-p2-payload-live-20260914.md). -It does not complete the full P2 matrix or implement P3 sleep/wake. -The [implementation checklist](./645-p3-implementation-plan.md) tracks those gates. - -## What “recovery” means here - -The coordinator starts a task's worker. Before asking AWS, it saves a -**receipt** in DynamoDB, the task database. The receipt contains a stable -request number, called the **client token**, and a fingerprint of the request. -Once AWS returns the worker ID, that ID is saved too. - -If a reply gets lost, retrying with the same receipt should find the same -worker. It should not change the instructions or buy another computer. -If the application no longer has enough information to retry safely, it must -stop and report uncertainty. - -These checks use two separate runners: - -- [Service replay](../../cdk/test/live/verify-microvm-replay.live.ts) calls - `RunMicrovm` directly and measures AWS's actual token behavior. -- [Application recovery](../../cdk/test/live/verify-microvm-start.live.ts) runs - the production strategy, S3 producer and DynamoDB receipt code in fresh local - child processes against real AWS. Its - [fault injector](../../cdk/test/live/microvm-start-child.ts) discards successful - replies or exits the child after an acknowledged operation. The next child - receives only the original task/settings, and recovers through AWS storage. - -The fault injector retains worker IDs separately for the test's cleanup audit. -The recovering application does not read that observer trace. This tests a lost -reply/process boundary, but it does not prove recovery when neither the -application nor an operator has any record of the worker ID. - -## AWS service observations - -The test uses an invalid startup reference, so the real guest rejects it before -S3 reads, configuration installation or pipeline startup. - -| Request | Observed result | -|---|---| -| Two simultaneous identical starts | Both returned the same worker ID | -| Immediate identical replay | Same ID and original response | -| Same token, duration changed from 180 to 181 seconds | `ValidationException`: “The provided clientToken was used with different request parameters.” | -| Replay after `GetMicrovm` reported termination | Same ID; no replacement worker | -| Replay at elapsed 30.194, 144.855 and 305.207 seconds | Same ID and request fingerprint | - -The worker was `microvm-c87a143e-1132-36ee-987f-ea778e23f955`. -The last replay's AWS request ID was -`1cc4e9be-b3c9-492d-95e4-e92a30cef006`. - -The replay response kept saying **`PENDING`**, even after termination. -An independent `GetMicrovm` read confirmed the original start time -`17:25:58.707Z`, termination time `17:26:04.155Z`, and state `TERMINATED`. -The replay returns the original response; callers must poll current state -separately. These observations through roughly five minutes do not establish -AWS's maximum token-retention period or its behavior after that period expires. -The application's 120-second cutoff remains a separate conservative limit. - -## Application observations - -Each task uses synthetic configuration with a deliberately malformed -`agent_session_role_arn`. The producer accepts the nonempty identifier, but the -guest rejects its shape before installing configuration, fetching secrets or -starting a pipeline. The control verifies this barrier through real worker logs. - -The operator runs the control-plane code. The guest uses the unchanged deployed -MicroVM execution role and runtime egress connector, with explicit `NO_INGRESS`. -The test changes two outgoing Run fields consistently: it caps worker lifetime -at **180 seconds** instead of the production eight hours, and sets the existing -MicroVM log group. Receipt hashing, payload preparation and application retry -logic remain unchanged. No deployed Lambda is killed or modified. - -| Case | Verified behavior | -|---|---| -| Control | Real S3 files, start receipt and worker-ID registration completed | -| Successful Run reply discarded | Production automatic retry recovered the same ID; `autoRetried` was true | -| Process exits after Run success, before receiving its result | Fresh process reused the saved token and exact signed request; same worker ID | -| Process exits after task-file write | Fresh process reused the immutable task bytes and completed the launch | -| Process exits after private launch-record write | Fresh process recovered the saved launch reference and completed the launch | -| Task-file write reply discarded | Producer read back the committed object and continued | -| Launch-record write reply discarded | Producer recovered the committed reference and continued | -| Worker-ID database write reply discarded | Strategy read back the committed handle and succeeded | -| Process exits after worker-ID write | Fresh process returned the saved handle without another Run request | -| Worker-ID write deliberately rejected before submission | Strategy stopped the known worker and reported `MICROVM_START_RECEIPT_SAVE_FAILED` | -| Changed instructions after interrupted start | `MICROVM_START_INPUT_CHANGED`; no Run call and no overwrite; original request could still recover | -| Cancellation with a saved handle | `MICROVM_START_TASK_CLOSED`; no new Run request; worker stopped | -| Actual receipt deadline passes after interrupted start | `MICROVM_START_OUTCOME_UNKNOWN`; no further Run request | - -Each case that reached AWS used exactly one worker ID. Repeated Run requests -had identical fingerprints, including the signed URL. Successful registration -saved matching `microvm_start.handle`, `session_id` and task-stable client token. -No capacity reservation or production channel/repository operation was created. - -The denied-write case injects an exception locally; it is not a test of an -actual IAM denial. Cancellation is observed through a fresh strategy invocation, -not a deployed approval/cancel handler racing a durable checkpoint. - -## Cleanup, logs and test corrections - -Independent reads confirmed **17 owned workers terminated**, all **36 planned -S3 object locations absent**, and all **16 planned task IDs absent**, including -setup failures and cases skipped when an earlier assertion stopped a batch. -Those counts include three service-probe workers and fourteen application-probe -workers across the control, fault batch and corrected cancellation rerun. -They are cleanup coordinates, not counts of distinct passing test cases. - -The final log audit found no signed-URL credential/signature markers and no -configuration-installed or pipeline-accepted messages. Four workers stopped -before emitting guest logs; the other thirteen had three events each. The test -does not use an empty log stream alone to prove a successful rejection. - -The runner calls the production task-file deletion helper, verifies both -files are absent, deletes only its own synthetic task rows, and removes its -unique manifest. This is operator cleanup; it does not replace the remaining -deployed-coordinator finalization checks. - -The stack remains `UPDATE_COMPLETE`, with last update -`2026-09-14T16:07:52.421Z`. No image, role policy or network configuration changed. - -Several harness expectations were corrected during calibration: - -- `ListMicrovms` permits at most 50 results per page. -- Parameter mismatch uses `ValidationException`; the state reason says - `HTTP status 400`. -- Child processes must reuse the parent's resolved `tsx` loader, since `npx` - can supply it from its cache. -- The task status is `TaskStatus.CANCELLED`. The misspelled fixture `CANCELED` - correctly failed as an unknown state. The corrected cancellation test passed. - -All affected fixtures were cleaned, and the corrected cases were rerun. -The production behavior required no fix in this batch. Comment cleanup in -`agent/src/server.py` and the MicroVM strategy removes stale claims about P1 -envelope compatibility, logged payloads and absence of automatic termination. -The helper's direct `None` no-op is not a legacy v2 startup path; the eight-hour -service lifetime is a backstop, not prompt cleanup. - -## Reproduce and validate - -Run from `cdk/` with the installed repository dependencies, Node 22 and AWS CLI. -Both scripts make no AWS calls without `--execute`. Output directories must -not already exist. - -```bash -mise exec -- npx tsx test/live/verify-microvm-replay.live.ts -mise exec -- npx tsx test/live/verify-microvm-start.live.ts - -AWS_PROFILE=sphia-dev mise exec -- npx tsx test/live/verify-microvm-replay.live.ts \ - --execute --account --region us-west-2 \ - --stack backgroundagent-dev --image-version 2.0 \ - --output /tmp/abca-p2-replay-new-run - -AWS_PROFILE=sphia-dev mise exec -- npx tsx test/live/verify-microvm-start.live.ts \ - --execute --account --region us-west-2 \ - --stack backgroundagent-dev --image-version 2.0 \ - --output /tmp/abca-p2-start-new-run -``` - -The application runner supports a comma-separated `--cases` selection. -Do not run independent worker-creating probes concurrently: each checks its -before/after inventory for unaccounted workers and reports them without deleting -them. If interrupted, use the private context, task plans and observer traces -for exact cleanup. The 180-second worker cap does not delete database/S3 files. -Context and traces contain identifiers and fingerprints, not signed URLs or -credentials. Both runners are outside the normal Jest test pattern. - -Validation: three existing CDK suites passed **134 tests**, and the selected -Python platform-config tests passed **5 tests**. Focused strict TypeScript, -ESLint and both dry runs passed. Documentation sync and changed-file link -checks accompany this record. No full application redeployment was needed. - -## Evidence and remaining gates - -Private evidence root: `/tmp/abca-645-p2-clean-20260913`. - -- `start-service-replay-v4-20260914`: successful service probe; earlier - `start-service-replay*` records retain calibration and cleanup. -- `start-recovery-control-v2-20260914`: successful application control. -- `start-recovery-faults-20260914`: ten passing fault cases and the misspelled - cancellation fixture; cleanup confirmed. -- `start-recovery-final-20260914`: corrected cancellation and real expiry passed. -- `start-recovery-final-audit.json`, `start-recovery-final-log-audit.json`, - `start-recovery-stack-status.json`: independent final audits. - -Per-run `results.json` links cases to task/worker IDs and AWS Run request IDs; -child traces record committed effects, injected faults and request fingerprints. -The five-minute service probe and the 120-second application cutoff are measured -separately. - -Still required: deployed durable-Lambda interruption/checkpoint recovery, -registration/cancellation races, automatic finalization after rejected hooks, -injected cleanup failures, and an operator recovery procedure for a genuinely -unknown worker ID. A CloudTrail Event History lookup for `RunMicrovm` during -this test interval returned no events; it did not establish such a procedure. -`ListMicrovms`/`GetMicrovm` expose no task token in the installed SDK shapes. - -The wider effective-role/session/transaction matrix, expired signer credentials, -ECS, public-bucket policy grants, network negatives, capacity migration and P3 -sleep/wake also remain open. Read-only trust inspection confirmed the session -role accepts the exact worker roles and the MicroVM execution role trusts -`lambda.amazonaws.com`; an isolated Lambda using that unchanged role is a -possible follow-up for actual IAM requests. No such function was created. diff --git a/docs/verification/645-p3-agentcore-20260916.md b/docs/verification/645-p3-agentcore-20260916.md deleted file mode 100644 index 6b816b64c..000000000 --- a/docs/verification/645-p3-agentcore-20260916.md +++ /dev/null @@ -1,110 +0,0 @@ -# ADR-021 P3: AgentCore compatibility verification - -Verified on 2026-09-16 in `backgroundagent-dev`, account ``, -region `us-west-2`. These results cover the shared AgentCore runtime and -approval/cancellation paths. The [MicroVM wake refusal](./645-p3-resume-refusal-investigation.md) -and other unfinished [P3 gates](./645-p3-implementation-plan.md) remain open. - -## Reviewed runtime update - -The normal AgentCore container was still older than the deployed MicroVM code. -The update uses the ARM64 container built from source `a81c565d`: - -- ECR asset tag: `3b50ba93fcff4e8129ee6140a8c65d876b45e47744c92775bc80bd078fd5d486`. -- Image digest: `sha256:8490592823b913558407b7b03d9c4f346c72fde2f15503849a50539fb7a654df`. -- Runtime: `backgroundagentdevRuntimeCC6E3A5A-Eq3mE6Gg2d`. -- Runtime version and `DEFAULT` endpoint advanced from `4` to `5`, both `READY`. - -The actual CloudFormation change set modified only -`Runtime99E3DDFA.AgentRuntimeArtifact.ContainerConfiguration.ContainerUri`, -with no replacement. All 475 resource identities remained unchanged. -Coordinator alias `live` still points to version `8`; MicroVM image `5.0` -remains `ACTIVE`/`SUCCESSFUL`; both normal suspension gates remain off. -The old AgentCore version `4` is still readable and references its original -container. This establishes retention, not a live rollback exercise. - -The reviewed template came from the exact deployed S3 bytes, SHA-256 -`4957871b24243f900e16fa9a04b58db63d85b1da5fcf7265d5f403fe5ecd3e3c`. -The new compact template is 706,153 bytes, SHA-256 -`9741f2d4ea6f43e8dde6deff0cb791780a54efdb933e7046c9851b9e6a2e3f9d`. -An initial preparation failed locally because stack metadata was missing. -A subsequent attempt uploaded the image but failed template validation: -pretty-printing expanded the JSON to 1,281,334 bytes. Restoring compact JSON -fixed that size error without changing resource properties. Neither failed -attempt updated the stack. - -## Bounded live cases - -A private durable Lambda ran the production coordinator with a fixed AgentCore -blueprint and two owned tasks. Its role restricted table operations to those -task/user keys and runtime calls to the normal AgentCore runtime. The tasks had -no repository or notification destination, a six-turn/$1 budget, and a -180-second approval window. Both requested `microvm_sleep_after_s=30`. -Only the private fixture's suspension gates were enabled. - -The fixture refused every MicroVM SDK call. It recorded each generated AgentCore -session ID before the actual invocation, then retained the AWS invocation -receipt. The approval API and cancellation API were the normal deployed -handlers. Their Lambda invocations supplied a synthetic trusted user context; -this did not exercise API Gateway authentication. - -| Case | Task | Session | Result | -|---|---|---|---| -| Approve after the sleep delay | `01M2NQFXGBXSAPV0PKFMXYTVX3` | `11991ca4-f2f5-4a83-89d0-cc161a76f92b` | `COMPLETED`; one successful `Read` | -| Cancel at the approval gate | `01M2NQFXGBQVTBQPDWKTTTJD44` | `3c0398a3-ce01-4444-b370-b7c111b009f2` | `CANCELLED`; no tool result | - -The approval case stayed at the gate for an observed 41.273 seconds before the -API returned HTTP 202. It never acquired MicroVM lifecycle/start state or called -the MicroVM service. The approved action read only `/etc/os-release`; task events -show exactly one call and one successful result. Its stored trace had zero -dropped events and SHA-256 -`453919a3accb63bd93dbec08685c121c011da8573254379ff6308bec224958d6`. -Invocation receipt: `24ca6184-fe2c-47aa-a521-552cc29d3cb2`. -Approval receipt: `0eb437b1-029f-4269-a5ce-6782caf0182e`. - -Cancellation returned HTTP 200, kept the task terminal, and stopped its session. -The blocked `Read` attempt produced no result. Its approval record remained -`PENDING`; cancellation did not approve the tool. No final trace was available -for this interrupted task. Invocation receipt: -`cf1dee27-8ead-4c46-8fb3-3c3cc7165028`. -Cancellation receipt: `5c9ca126-f456-42ba-b7c6-f850fa072045`. - -Both durable executions finished `SUCCEEDED`, both task reservations were -released, and both counters reached zero without independent reservation repair. -The successful session was explicitly stopped by fixture cleanup -(`53369953-9911-4aae-989c-da2cffed0c4a`). -Both sessions subsequently returned `ResourceNotFoundException` to another -owned-session stop request, verifying absence. - -## Excluded attempt and scope limits - -An initial fixture patched the root workspace's AgentCore SDK copy, while the -production strategy imported a separate copy under `cdk/node_modules`. -Its task completed, but the missing pre-invocation audit invalidated the intended -fixture. That attempt, task `01M2NQ6087WM33JR0EZT7WRDNJ`, is excluded from the -acceptance results. Its session was explicitly stopped and its zero counter -removed. The corrected bundle resolved the exact production SDK and used fresh -task identities. Both versions and all raw evidence were retained for audit -before infrastructure cleanup. - -Cleanup completed at **18:30:28.935 UTC**. The private coordinator's two -versions, role, switch, log group and zero counters were removed after ownership -checks. All three durable executions, including the excluded attempt, had -finished. The archive includes 252 coordinator log events and 17 relevant -runtime log events. Read-only checks confirmed the temporary resources were -absent; the normal runtime and its shared logs were retained. - -This verifies AgentCore's shared approval/cancellation behavior and exclusion -from MicroVM sleep. It does not verify the full other-backend IAM/network matrix, -remote MCP connectivity, ECS, capacity migration, or a cloned-repository P3 run. -The normal stack contains no ECS cluster or task definition. - -Evidence directories: -`/tmp/abca-645-p2-clean-20260913/p3-agentcore-compatibility-20260916` and -`/tmp/abca-645-p2-clean-20260913/p3-agentcore-fixtures-20260916`. - -The durable private archive is -`/Users/sphias/.local/share/abca-verification/645-p3-20260916/agentcore-compatibility-evidence.tar.gz`. -It contains 77 files, including source `9cd74c7e`, and is 38,952,488 bytes with -SHA-256 `0409a18b6ebd73a538f49ec4e41f1a3a9fa20ab8e0edc55179e5eb179fe0653b`. -Every file was verified against its manifest; archive permissions are `0600`. diff --git a/docs/verification/645-p3-agentcore-permissions-20260917.md b/docs/verification/645-p3-agentcore-permissions-20260917.md deleted file mode 100644 index 5aed0957e..000000000 --- a/docs/verification/645-p3-agentcore-permissions-20260917.md +++ /dev/null @@ -1,108 +0,0 @@ -# ADR-021 P3: actual AgentCore permission checks - -Verified September 17, 2026, in account ``, `us-west-2`. -The normal AgentCore runtime's `DEFAULT` endpoint remained **READY**, version -**5**. This extends the earlier -[AgentCore approval/cancellation checks](./645-p3-agentcore-20260916.md) -with real permission requests from that runtime. - -## Test and results - -Task `01M2PDAMMMMH8GMSS0YJP2J9R6` ran one exact Python diagnostic through -the agent's Bash tool. It had no repository or notification destination, -a four-turn/$1 limit, and an eight-minute watcher limit. A private Lambda -ran the production Durable handler. Its code SHA-256 was -`T4bq5saVKeGxQd0yUHN3TRCPKTrqUNH+TJHVT2YyVtc=`. - -The diagnostic removed inherited static credential overrides in its child -process before loading the production `aws_session` providers. This made -the ambient checks use AgentCore's container credential provider. Actual STS -identity requests verified the normal runtime role and the task session role. -No credential values were printed or saved. - -The independent audit passed at **00:55:43.375 UTC**: - -| Requests | Count | Result | -|---|---:|---| -| Ambient and scoped STS identities | 2 | Expected roles | -| Ambient task/counter reads | 2 | Denied | -| Ambient artifact read/list | 2 | Denied | -| Scoped own task / other owned task / counter reads | 3 | Allowed / denied / denied | -| Scoped reporting update with deliberately false condition | 1 | Authorized shape; condition rejected | -| Scoped writes to owner, compute handle, start receipt, reservation and lifecycle fields | 5 | Denied | -| Scoped own artifact write/read | 2 | Allowed / denied | -| Scoped write under another owned task's artifact prefix | 1 | Denied | -| Scoped MicroVM bootstrap-object read | 1 | Denied | - -All **19** checks retained actual AWS request IDs. Artifact access is -deliberately write-only in the deployed session role; the read denial is -expected. Every DynamoDB update used a false condition, so even unexpected -authorization could not change an existing task row. The other owned task's -complete record was unchanged. - -The audit verified exactly one Bash call matching the prepared command, one -successful tool result, zero approval records and zero dropped trace events. -The task completed, and its Durable execution succeeded. No MicroVM SDK call, -start receipt or lifecycle state appeared. - -Permission-result SHA-256: -`07237297d088b12743b5237a8655bfd5fca0cc072f84bddbdec3d15fa1913fdc`. -Trace SHA-256: -`9dc90a5df6077b2d27ef777d88400c505cb6c5e75dd2d634a12cb9599a50b377`. - -These checks establish the observed grants for the issued credentials. -They do not prove isolation against a compromised worker choosing different -tags when assuming the session role. - -## Watcher timing error - -The original watcher failed after the diagnostic had completed. It read -the task while its slot was still held, then read the newer Durable status -`SUCCEEDED`, and incorrectly asserted against the older task snapshot. - -| Evidence | UTC time | -|---|---| -| Actual reservation release | 00:53:28.261 | -| Durable `finalize` step succeeded | 00:53:28.301 | -| Durable execution succeeded | 00:53:28.318 | -| Watcher's stale-snapshot assertion failed | 00:53:28.391 | - -The fallback invoked the idempotent release helper after release had already -completed. It also explicitly stopped the owned AgentCore session. The -independent audit records this watcher failure and verifies the original -permission results separately; it does not relabel the watcher as passing. - -The original script is retained as `run-executed.cjs`. The corrected watcher -rereads the task after observing Durable completion. That corrected script -was syntax-checked but was not used to rerun the already completed AWS checks. - -## Cleanup and retained evidence - -Session `adbb6743-2a03-4161-9f87-f206aa738088` was stopped at -**00:53:40.358**, request ID `9077a23c-f13b-47ca-883f-26dac1c066c9`, -and subsequently returned `ResourceNotFoundException`. - -Private infrastructure cleanup was verified at **00:58:36.593**. The -coordinator and all versions, role, switch, log group and zero counter were -removed. Both harmless probe objects were archived and removed; the negative -write created no object. All 32 private function log events were retained. -Normal task/trace records follow their existing retention policy. - -The initial immediate absence check briefly still saw the deleted Lambda. -A subsequent bounded, read-only check verified absence. The original cleanup -script and failure, the follow-up verification, and the corrected future -cleanup script are retained. - -The normal runtime endpoint's full before/after snapshots and ambient-role -policies were identical. Normal MicroVM automatic sleep stayed off. -Evidence directory: -`/tmp/abca-645-p2-clean-20260913/p3-agentcore-permissions-20260917`. - -The permanent private archive is -`/Users/sphias/.local/share/abca-verification/645-p3-20260916/agentcore-permissions-evidence.tar.gz`. -It contains **45 files**, **14,912,232 bytes**, with mode `0600`; every member's -hash was checked against the manifest. Archive SHA-256: -`34123cd0d4599598a0489d9716d34ab86ccb97c0c5944b53072dbf33f21aa2fc`. - -Runtime ingress, remote MCP connectivity and the remaining deployment/service -gates remain in the [implementation plan](./645-p3-implementation-plan.md). diff --git a/docs/verification/645-p3-approval-ux-20260917.md b/docs/verification/645-p3-approval-ux-20260917.md deleted file mode 100644 index f5e68fdcb..000000000 --- a/docs/verification/645-p3-approval-ux-20260917.md +++ /dev/null @@ -1,267 +0,0 @@ -# P3 cancellation and approval notifications - -This records the September 16–17 implementation while the user was away. -The cancellation/API/notification update is deployed on `backgroundagent-dev` in -`us-west-2`. The subsequent managed-image update deploys the guest hook change in -image 7.0. Normal automatic sleep remains disabled. - -## Cancellation behavior - -REST/CLI and Slack cancellation use one shared state operation. Cancelling a task -with a pending approval writes all three changes in one DynamoDB transaction: - -1. The task becomes `CANCELLED`. -2. Its unanswered request becomes `CANCELLED`, with the owner's cancellation reason. -3. An `approval_cancelled` audit event is saved. - -The write checks ownership, the observed task status and the exact request ID. -If a decision or new request wins the race, cancellation reloads state and retries -up to three times within its five-second budget. A previously recorded approval or -denial remains recorded. Unexpected database failures do not fall back to cancelling -only half the state. - -The existing approve/deny API deliberately reports `404 REQUEST_NOT_FOUND` for -any failed approval-row condition, including already-decided/cancelled requests. -It cannot distinguish missing rows from foreign ownership using the current -transaction response. A task-only condition failure returns 409. The cancellation -reason is recorded in the approval row and closure event. - -The pending endpoint checks the owning tasks with consistent batch reads. Requests -whose tasks are terminal, missing or waiting on another request are omitted. This -also hides orphaned requests created by older cancellation code. It does not judge -whether the requested action is still relevant. - -A subsequent review found that a full first page of legacy cancelled requests -could hide a current request on the next page. The source now follows the index -cursor until it finds 100 live requests or exhausts the results, sharing a -five-second read budget. Read failures return an error instead of an incomplete -empty list. The 21 focused tests pass. The follow-up is deployed and its real -101-request check returned the single current request from page two after -examining 100 cancelled requests on page one. All 202 temporary task/approval -records were removed and consistent reads confirmed their absence. -The deployed pagination proof and cleanup are archived at -`/Users/sphias/.local/share/abca-verification/645-p3-20260916/pending-pagination`. - -The guest hook recognizes cancellation immediately, including when cancellation -wins a timeout race. It denies that tool call without restoring the task to -`RUNNING` or inventing a human denial for the agent's decision cache. - -## Notification and response path - -Slack and Linear receive pending requests, saved decisions and closure reasons. -The message identifies the task/request, tool, reason, action preview and current -decision deadline. It supplies these commands with the actual IDs: - -```text -bgagent approve TASK_ID REQUEST_ID --scope this_call -bgagent deny TASK_ID REQUEST_ID -``` - -The user runs them while signed in as the task owner. Slack messages stay in the -original thread; Linear posts on the originating issue. Native Slack buttons and -Linear comment-to-decision parsing remain unimplemented. - -The router now unwraps approval milestones from `agent_milestone`, as it already -did for PR creation. The Slack renderer no longer drops those events, and Linear -has explicit approval subscriptions. Approval handling returns before Linear's -terminal-reply path, so asking a question cannot accidentally mark a task finished. - -The dispatcher reads saved approval state rather than trusting delayed event -previews. It suppresses old pending notifications after closure and records -successful delivery separately for each request, event type and channel. External -post success followed by a lost receipt write can still duplicate a notification. -Known credential shapes are redacted before truncation. Slack uses plain text; -Linear keeps previews inside a literal code block. Repository text cannot supply -the generated approval command. - -CLI watch shows the action, reason, exact response commands and cancellation -feedback. A saved decision is described as recorded; it does not claim the sleeping -worker has successfully resumed. - -The final review also corrects three closure-feedback cases. Timeout messages -use the saved polling failure or explain that the deadline elapsed, instead of -repeating the original policy reason. A delayed approval/denial message reports -an already-ended task's status instead of promising that it will continue. -The current stranded-task reconciler emits `approval_stranded` without a -request ID and leaves the approval row `PENDING`; the notifier handles that -legacy shape only when the consistently read failed task, saved stranded cause, -exact pending request and ownership agree. This supplies feedback without changing -the approval row or adding a relevance check. Missing identity is logged. - -The review also found a pre-existing stream retry defect. Failed deliveries -returned the opaque event ID; AWS requires the DynamoDB record's sequence number. -The handler now returns the sequence, retaining it as a string, and rejects the -whole batch if a failed record has no sequence. AWS retries from the lowest -failed sequence onward, so successful later records can repeat and still need -delivery receipts. See the -[AWS partial-batch contract](https://docs.aws.amazon.com/lambda/latest/dg/services-ddb-batchfailurereporting.html). - -## Verification - -- Focused cancellation/store/handler checks: 149 CDK tests passed. -- Full Python quality: 1,957 passed, 11 skipped; lint/format/type checks passed. -- Full CLI build: 943 tests passed; compilation and lint passed. -- Full CDK suite: 5,126 tests passed, 56 skipped; the snapshot passed. -- Documentation sync and the 77-page site build passed. -- Additional construct IAM checks: 17 passed. Type synchronization, final - compilation and lint passed. -- After the pagination follow-up: 21 pending-handler tests, compilation and lint - passed. The packaging instruction cleanup passed seven existing regression tests - and Bash syntax validation. -- After the final closure/retry corrections: 241 notification and routing tests, - compilation, lint and type synchronization passed. - -The production TypeScript helpers also ran against three isolated real DynamoDB -tables in ``, `us-west-2`. Seven scenario groups passed: atomic pending -cancellation, late approval refusal, preservation of an existing approval, a new -gate racing an old task snapshot, cancellation without a gate, 12 simultaneous -approve/cancel transactions, and per-channel delivery receipts/closure feedback. -Approval won first in all 12 simultaneous races; cancellation preserved that -decision and cancelled the task. The separate cancel-first case rejected approval. - -All temporary tables were deleted and `DescribeTable` returned -`ResourceNotFoundException` for each. No Slack or Linear messages were sent during -verification. Raw evidence: `/tmp/abca-645-p3-approval-live-20260916`; permanent -archive: `/Users/sphias/.local/share/abca-verification/645-p3-20260916/approval-ux`. - -## Bounded deployment - -CloudFormation completed `p3-approval-ux-20260917-r2`, started at -`2026-09-17T03:47:12.105Z`. The reviewed template SHA-256 is -`2d75f6c84ea625ce2ff12cf189795476b22ee8e342ca3e7daddde59313b7a6e3`. -It changes five function packages (cancel, pending, Slack interactions, fan-out -and upload confirmation) and four additive table-access policies. CloudFormation -also lists eight existing API references because they depend on those function -ARNs. Their template definitions are unchanged. - -All 475 resource identities, compute resources, image 6.0, coordinator alias 10 and -parent outputs are preserved. The upload-confirmation package includes the earlier -correction explaining that startup was not confirmed, instead of promising an -automatic retry that does not exist. - -The initial review caught a synthesis-region mismatch before deployment. The final -assembly explicitly uses `AWS_REGION=us-west-2`, `AWS_DEFAULT_REGION=us-west-2` and -checks the manifest environment. The first change set was deleted without execution -so the final packages include the lint cleanup. Deployment artifacts and postchecks -are under `/tmp/abca-645-p3-approval-rollout`. - -Post-deployment verification at `2026-09-17T12:06:26Z` confirmed all five function -code hashes against the actual S3 ZIP bytes and exercised the deployed handlers -with two owned, API-origin task fixtures. Pending requests on cancelled tasks were -hidden; cancelling the live fixture atomically closed its task and request and -saved the audit event; a later approval returned `404 REQUEST_NOT_FOUND`; the -pending list was then empty. These calls used the handler's authenticated-event -shape through Lambda invocation, not an additional API Gateway login test. -Neither fixture launched a worker or sent a channel message. - -All owned task, approval, event and rate-limit records were removed. Eight -independent consistent reads confirmed their absence. Alias 10 and the live sleep -switch set to `false` were verified again. The two earlier verification attempts -used an incomplete request body and an incorrect expected HTTP status; their -fixtures were also cleaned up, and their logs are retained separately. - -The reviewed template, deployment receipts, code hashes, postchecks and cleanup -proof are archived with SHA-256 hashes at -`/Users/sphias/.local/share/abca-verification/645-p3-20260916/approval-rollout`. - -## Guest image deployment - -The subsequent `p3-cancel-image-20260917-r2` update deployed image **7.0**. -The artifact has 111 files and is 486,955 bytes, SHA-256 -`b4bc0c628f4c7976e18850ff47f92e78ecdccec683bc38974e0da82fc8600049`. -Compared with image 6.0, only `hooks.py` changes executable behavior; three other -files contain comment corrections and 107 files, including the Dockerfile, are -identical. - -The executed template SHA-256 is -`aa27f9e38107ad28b3bfeac7773797daadfae954b550cd78d21fd71f8fcc4c11`. -It changes the image artifact and its exact build read permission. The five -additional CloudFormation notices refer to unchanged consumers of the image ARN. -An expanded preview removed automatic notices for the unchanged existing nested -stacks; the initial preview was deleted without execution. - -Postchecks confirm `ACTIVE` / `SUCCESSFUL`, 8,192 MiB, unchanged image configuration, -retention of image 6.0, all 475 physical resource identities, coordinator alias 10 -and both sleep switches off. - -Two real Durable workflows then passed on image 7.0 using a private coordinator, -the production handler and normal decision APIs: - -| Case | Task | Result | -| --- | --- | --- | -| Approve after 60 seconds asleep | `01M2QMKXPQMHF2HZWS49R2G2S0` | Original PID 1 resumed with HTTP 200, one Read succeeded, task completed | -| Cancel while asleep | `01M2QMKXPZVE13FED8FG4EAYT8` | Task and approval became `CANCELLED`, one closure event, worker terminated without Resume or a successful Read | - -The approved wake has AWS request ID `233838bc-a1a9-4527-9a16-39e91bfe8b7e`. -Both executions finalized without watcher repair, released their reservations and -deleted their launch payloads. The cancellation case verifies platform closure and -termination; the guest's immediate cancelled-poll branch is covered by the Python -tests. Normal automatic sleep stayed disabled. - -The private function, role, switch, log group and two empty capacity counters were -deleted. All 147 function log events were archived. The function remained visible -beyond the original 45-second deletion-verification window; independent reads at -`2026-09-17T12:33:47Z` confirmed all private resources absent. No deletion was repeated. -Normal task/audit records retain their existing retention policy. - -The 67-file image and acceptance archive, with a SHA-256 manifest, is -`/Users/sphias/.local/share/abca-verification/645-p3-20260916/image7-cancellation`. - -## Pagination follow-up deployment - -`p3-pending-pages-20260917` changed only the pending function package; the two other -change-set notices were unchanged API references. Template SHA-256: -`128bfd2135643fd43b0763dbc52c85e0f5c41664bdf03b4e46e2a9693b535680`. -The deployed ZIP hash is -`W++CETdMQwJ2hrSQIcyCQ+6yTAKkjzByAzVo6NBlwQg=`. -The live 101-request regression passed at `2026-09-17T12:36:47Z`, invocation request -ID `4a488ff1-37ab-49ac-80d9-a142c6dda9c4`; cleanup was verified five seconds later. -Evidence is under `/tmp/abca-645-p3-pending-pages-20260917`. - -## Final closure-feedback deployment - -The reviewed `p3-approval-closure-feedback-20260917-r2` change set -`08d5af7d-b938-4d75-9343-5240f63642c6` changed only the fan-out function's code. -CloudFormation reached `UPDATE_COMPLETE`; its update started at -`2026-09-17T13:16:58.640Z`. The current template SHA-256 is -`1be45cbc55d63fe8370a638984f226366a57e6b29317b9b064c8a67b0baf1dc7`. -The live code hash matches the reviewed ZIP: -`fYNQWLpyeKEz9Bz0qlDdxKYitLX/pYoLrnrcqPzkQAE=`. -All 475 physical identities, parent outputs, image 7.0, coordinator alias 10 and -the disabled sleep setting are preserved. - -Three deployed checks passed at `2026-09-17T13:18:04Z`: - -1. The actual legacy stranded-event shape rendered from saved task/approval - state, then exited at the intentionally absent chat destination. -2. A malformed approval event returned its numeric stream sequence as the - retry cursor, preserving the value as a string. -3. The same failed event without a sequence rejected the invocation so a batch - cannot be silently acknowledged. - -The approval row remained unchanged. Both owned fixture rows were deleted; -consistent reads confirmed their absence. No chat destinations, worker sessions -or external messages were involved. These checks verify the deployed handler -contract, not a live external delivery or the event-source mapping's retry loop. - -The initial pretty-printed template exceeded CloudFormation's size limit and -was rejected before preview creation. Compact JSON preserved the same content -within the limit. The first valid preview was deleted without execution so the -final package also contains the stream retry correction. - -The 38-file evidence archive and SHA-256 manifest are at -`/Users/sphias/.local/share/abca-verification/645-p3-20260916/closure-feedback`. - -## Remaining boundaries - -This change does not make requests indefinite. The current default is still -300 seconds, with a 3,600-second submission maximum and worker-lifetime clipping. -Keeping unanswered requests available beyond a worker's lifetime needs durable -continuation of the agent's workspace/session and the waiting tool call. Simply -removing the deadline would leave an apparently actionable request pointing at -a worker that can no longer resume. - -The separate per-user notification throttle proposed in the Cedar design is also -not implemented. Native chat responses, indefinite waiting, nested resource -migration and normal automatic-sleep activation remain on the -[P3 checklist](./645-p3-implementation-plan.md). diff --git a/docs/verification/645-p3-callback-live-20260915.md b/docs/verification/645-p3-callback-live-20260915.md deleted file mode 100644 index eebc3ebf7..000000000 --- a/docs/verification/645-p3-callback-live-20260915.md +++ /dev/null @@ -1,226 +0,0 @@ -# ADR-021 P3: live verification of the approval callback fix - -Verified 2026-09-15–16 UTC. This records the deployment and live retest of -[the explicit callback timeout](./645-p3-callback-timeout.md), source `58ed7bbf`. -All nine core callback cases passed, including real credential expiry during -suspension. The separate intermittent wake failure remains unresolved; this is -not a P3 completion record. - -## Deployment - -The `backgroundagent-dev` stack in `us-west-2` reached `UPDATE_COMPLETE`. -Change set `p3-callback-image-20260915`, executed at 23:39:52.809Z, modified -exactly two resources without replacement: the managed MicroVM image and its -build role's permission to read the new artifact. The image URI and permission -were changed together to the same immutable object: - -```text -0585a80606aaa66e9ce4dbff35a093451488f72d5a732c6cce6036b88fbcd4ed -``` - -Image `backgroundagent-dev-abca-agent`, version `4.0`, became `ACTIVE` with -build state `SUCCESSFUL`. Ready and validate returned HTTP 200. All six hooks -remain on port 8080 with the same protocol and 30-second service budget. -Bootstrap remains `1.8.0`; the root stack still has 475 resources. - -The main coordinator's code, published version `3`, and alias are unchanged. -Both production suspension settings remain off. The ECS and AgentCore runtime -images have not received this source update. - -The deployment used the exact previously uploaded S3 template, with reviewed -artifact substitutions. It did not use the Unicode-damaged `GetTemplate` -response or an unrelated fresh synthesis. - -## Isolated live checks - -A temporary Lambda uses the compiled production durable coordinator, its -production lifecycle policy and real approval/cancel APIs. Its wrapper allows -only fixed, owned, repository-free tasks and pins image `4.0`. Tasks may read -`/etc/os-release`; the credential-expiry task first runs a foreground -`sleep 180`. No repository publication or notification is part of these checks. - -The fixture has its own IAM role, sleep switch and log group. A separate -temporary Lambda isolates the late-deadline supervisor outage from the long -credential-expiry execution. The long execution stays on fixture version `1`; -new boundary cases use published version `2`. - -Acceptance requires final task, approval and tool evidence; stable durable -supervisor clocks; one admission/start/finalization; a terminated worker; empty -payload prefix; and a released slot with counter zero. An empty independent -cleanup record proves the watcher did not repair the result. - -| Case | Current result | -|---|---| -| Ordinary task | Passed on image `4.0` | -| Approval after sleep, first attempt | Approval/read passed; cleanup acceptance invalidated by watcher timing | -| Fresh approval after sleep | Passed | -| Denial after sleep | Passed; one authoritative denial, no successful read | -| Cancellation while asleep | Passed; no tool result, task stayed canceled | -| Original approval deadline | Passed; timed out at the original deadline | -| Short approval window | Passed; 30-second timeout without suspension | -| Approval during grace period | Passed; quick approval without suspension | -| Wake after original deadline | Passed after an isolated supervisor outage | -| Credentials expire while frozen | Passed; fresh keys after expiry and timeout at the original deadline | -| Newly written files and two approval gates | Passed | -| Missing approval record during freeze | First attempt intercepted by connection refusal; fresh attempt returned expected HTTP 409 | - -### First approval attempt: watcher timing error - -Task `01M2KQ4D59154CQ89ZW1VWNXD0`, worker -`microvm-2532572a-6fe0-3a44-b2a7-e6edf3a4a687`, returned HTTP 200 from -`/suspend` at 23:48:19.210Z and `/resume` at 23:48:24.662Z. Approval became -`APPROVED` without changing its original clock, and exactly one Read succeeded -at 23:48:24.909Z. - -At 23:48:29.945Z the task was `COMPLETED`, the durable execution `SUCCEEDED`, -and its slot released, but AWS still reported `TERMINATING`. The private -watcher immediately required `TERMINATED` and entered independent cleanup. -That makes this attempt unsuitable as proof of untouched coordinator cleanup. - -The watcher now observes `TERMINATING` with read-only polling for up to -60 seconds before deciding whether termination failed. It sends no cleanup -request during that wait. Fresh task `01M2KQV00H6CHJKCC2D184DZBR` passed complete -acceptance, including untouched cleanup; the original evidence is retained. - -### Real credential expiry and the original approval deadline - -Task `01M2KQ4D590P0GSK2VGWXB2987`, worker -`microvm-39599381-c0f2-3298-b9e2-6fcb561daa6a`, stayed on the same image `4.0` -worker throughout this test. It completed one foreground `sleep 180`, then -requested exactly one Read of `/etc/os-release`. - -| UTC time | Evidence | -|---|---| -| Sep 15, 23:51:09 | Initial task-scoped STS credentials issued, expiring Sep 16 at 00:51:09 | -| Sep 15, 23:54:24 | Approval created with a 3,600-second window; original deadline 00:54:24 | -| Sep 15, 23:54:55.242 | Guest `/suspend` returned HTTP 200 | -| Sep 16, 00:51:24.348 | Independent read-only observer confirmed `SUSPENDED`, after actual credential expiry | -| Sep 16, 00:53:25 | Fresh credentials issued for the same role and user/task tags, expiring 01:53:25 | -| Sep 16, 00:53:25.372 | Guest `/resume` returned HTTP 200 | -| Sep 16, 00:54:24.136 | Read returned `User timed_out`; the original approval became `TIMED_OUT` | -| Sep 16, 00:54:29 | A further Bedrock response, persisted task events and S3 trace upload succeeded | -| Sep 16, 00:54:31.065 | Coordinator released the task's capacity reservation | - -CloudTrail request `d87e98e4-676f-4640-9ad8-881f7cb79e22` identifies the initial -STS issuance; `bedf0aaa-f2ad-4989-ace5-b421558a188c` identifies renewal after -expiry. The evidence stores timestamps, role/session identity and tags, without -credential values. - -All 744 durable invocations/poll callbacks preserved one first-observed time and -one session deadline. The task finished `COMPLETED`, its durable execution -`SUCCEEDED`, and the worker `TERMINATED` with reason `Success.` Its payload prefix -was empty, counter zero and independent cleanup record empty. A task completing -after correctly reporting a denied/timed-out tool is expected; the Read itself -did not succeed. - -Unlike the previous long test, the approval callback remained alive until the -original deadline. This passes both real credential renewal and callback -semantics; the earlier failed evidence remains in the historical record. - -### Repeated wakes, file persistence and missing approval - -Six further approval cases passed, three each on images `3.0` and `4.0`. -They used the same coordinator bundle, with the image version pinned separately. -Each retained its worker identity and supervisor clocks, ran one approved Read -and completed cleanup without watcher repair. These successful repeats do not -resolve the intermittent refusal. - -Task `01M2KRWNPPRV7CGJQSRFYY3375`, worker -`microvm-09ebf679-e78c-398d-8023-f1a176d5fb51`, wrote two new marker files in an -owned temporary directory. Its first gate was created at 00:21:33Z on Sep 16, -and its second at 00:22:12Z. Both independently suspended and resumed with HTTP -200, retaining distinct approval IDs and their original 600-second windows. -The two successful Reads returned the original marker bytes after their -respective freezes. The full recorded Bash and Read inputs match the intended -commands and paths. - -The pre-freeze SHA-256 values were: - -```text -b81be383132e7a49275fcf91cb22b0f4b015aa7b26387a3f300f7bc6ed395781 -ba01a509d1b49275a6b7c4a936172e507e51debea893b9b4a34e7e02e532e4af -``` - -All 21 durable invocations retained one pair of supervisor clocks. This proves -mutable temporary-filesystem persistence across two approval generations; it -does not replace a full P3 run against a cloned repository. - -The first missing-approval case, task `01M2KRWNPP4BQCJ80SVG6473Z8`, reproduced -the [connection refusal](./645-p3-resume-refusal-investigation.md) on image `4.0`. -The task failed and cleanup passed, but the expected guest response was absent. -Its acceptance record remains failed. - -One fresh attempt, task `01M2KSTSR1V31NTPQ0CKCJJJMN`, worker -`microvm-a093863f-edf0-39f4-9d79-6893ff08042c`, reached the intended failure: -after conditional deletion of its own pending approval, `/resume` returned HTTP -409 at 00:30:46.418Z. The service reported that HTTP status explicitly. No Read -result was produced; the task was `FAILED`, the worker terminated, its slot -released and payload removed without independent cleanup. This is a passing -negative test: the missing gate never allowed the tool to run. - -## Database checks and comment review - -The optional database suites skipped by the full build were run separately -against the documented pinned DynamoDB Local image, using a loopback endpoint -and dummy credentials. All 56 CDK lifecycle/capacity cases and all 47 Python -checkpoint cases passed. The temporary in-memory container was removed and -its absence verified. - -The subsequent source-comment edits passed TypeScript compilation and focused -ESLint. Python type checking also passed, covering the probe's final output -redirection change. - -Source comments now distinguish asynchronous `TERMINATING` from finished -shutdown, describe bounded recovery for unknown/suspending states, and specify -which roles receive lifecycle permissions. - -### Runtime logging inventory - -At 00:13:54.991Z on 2026-09-16, worker -`microvm-82e58ec5-ee2a-399b-aa8b-38a8469ac4a7` on image `3.0` was `RUNNING` -both before and after enumerating `/aws/lambda-microvms/` log groups. The only -group was `/aws/lambda-microvms/backgroundagent-dev-abca-agent`, with 90-day -retention. CloudFormation confirms that -`LambdaMicrovmComputeMicrovmLogGroup46EC26A7` owns that group. Image `4.0` -runtime and build records also arrived there. - -The deployed execution role has one inline policy and no attached managed -policies. Its logging grants allow `CreateLogStream` and `PutLogEvents`, with -no `CreateLogGroup` permission. No permission changes were needed for these -runs. This replaces the old comment's pending during-run inventory check; -it does not guarantee that future service versions will never need another -group. - -## Final cleanup - -Before infrastructure removal, all 18 main-function durable executions were -terminal; the separate late-deadline execution had already finished. All 19 owned -tasks had terminated workers, released reservations, zero counters and empty -task-payload prefixes. Task statuses were 16 `COMPLETED`, one `CANCELLED` and two -`FAILED`. Those counts include the retained failed/mis-measured attempts and must -not be read as 19 passing acceptance cases. - -The main function's six published versions, the separate late-deadline function, -their two log groups, the shared verification-only role and its policies, and -the private suspension switch were removed. Read-only checks at -2026-09-16T00:59:28.099Z confirmed both functions, the role, parameter and log -groups were absent. The in-memory DynamoDB Local container was also removed. - -Full guest streams, 4,028 main-function log events, durable histories, task events, -redacted credential issuance and resource checks are retained in the private -verification archive. Task/event/trace audit records and shared immutable image -artifacts retain their normal lifecycle; they were not deleted as test payloads. -Production suspension settings remain off. - -## Remaining scope - -The three original image `3.0` resume-hook connection refusals and the new image -`4.0` refusal remain unresolved. The -[investigation record](./645-p3-resume-refusal-investigation.md) contains all -four timelines and the next diagnostic steps. The callback fix changes how long -Claude waits for Python; it does not resolve the separate connection failure. -Successful later wakes do not discharge those failures. - -The [implementation plan](./645-p3-implementation-plan.md) retains the wider -P2/P3 acceptance gates. This completed callback retest and cleanup do not close -the separate wake-failure, full-repository, other-backend or migration gates. diff --git a/docs/verification/645-p3-callback-timeout.md b/docs/verification/645-p3-callback-timeout.md deleted file mode 100644 index ab0e97ee1..000000000 --- a/docs/verification/645-p3-callback-timeout.md +++ /dev/null @@ -1,83 +0,0 @@ -# ADR-021 P3: preserve the SDK approval callback - -Date: 2026-09-15. This follows the -[real long-sleep failure](./645-p3-durable-live-20260915.md#expired-credentials-passed-long-approval-semantics-failed). -The change below is deployed in MicroVM image `4.0`; the -[live retest record](./645-p3-callback-live-20260915.md) tracks AWS acceptance. - -## Problem - -The approval timer and the coding program's callback timer are separate. -The first controls how long a person may approve a tool. The second controls how -long Claude waits for our Python hook to answer. A suspended VM must preserve -that unanswered callback until its approval can be reconciled. - -The live hour-long case renewed expired AWS credentials successfully, but -Claude abandoned its pending Read callback when the VM woke. It produced a -generic denial while the approval record remained `PENDING`. The task then -completed before the original approval deadline. This is a failed lifecycle -case even though the tool did not run and cleanup succeeded. - -`build_hook_matchers` supplied no explicit `HookMatcher.timeout`. A local probe -with the actual pinned SDK `0.2.110` and Claude `2.1.191` reproduced the same -generic denial and Python `CancelledError` when a one-second callback budget -expired during a three-second wait. A ten-second budget allowed that wait and -one successful read. The SDK's documented 60-second default is not reliable for -this binary: an unset timeout allowed a 65-second wait. The longer default probe -then cancelled at 600.034 seconds and returned the same generic denial. The -actual default for this pinned binary is ten minutes. - -## Local change - -PreToolUse matchers now receive an explicit timeout: - -| Worker | Callback budget | -|---|---| -| ECS / AgentCore | Maximum approval window, 3,600 seconds, plus 120 seconds | -| Registered MicroVM | Maximum VM lifetime, 28,800 seconds, plus 120 seconds | - -The larger MicroVM budget covers a supervisor that recovers after the approval -deadline. The existing approval loop still uses its original deadline and denies -an expired gate. PostToolUse and Stop callback settings are unchanged. - -The existing eight-hour `RunMicrovm` limit now comes from -`contracts/constants.json`, shared with the Python callback calculation. The -service duration is unchanged. The drift checker rejects invalid durations and -a new hardcoded strategy copy. An outdated exception comment was also corrected: -Python cancellation propagates through `finally`; it is not caught by -`except Exception`. - -## Reproduction and validation - -The opt-in [probe](../../agent/scripts/verify_approval_hook_timeout.py) runs the -actual pinned coding program against a loopback fake Bedrock stream, synthetic -AWS keys and one disposable marker file. It makes no paid model call or AWS -request. It takes production matcher settings and replaces only the callback -with a controlled wait. - -```bash -agent/.venv/bin/python agent/scripts/verify_approval_hook_timeout.py --timeout 1 --delay 3 -agent/.venv/bin/python agent/scripts/verify_approval_hook_timeout.py --configured microvm --delay 650 -agent/.venv/bin/python agent/scripts/verify_approval_hook_timeout.py --configured standard --delay 650 -``` - -The full root build passed in 364.12 seconds: 4,993 CDK tests passed with 56 -optional DynamoDB Local tests skipped; 1,942 Python tests passed with 11 skipped; -all 928 CLI tests passed. Compilation, lint, drift checks, synthesis and docs -build also passed. The focused CDK run passed all 201 assertions, though its -partial coverage could not satisfy the global full-suite threshold; the later -full build passed that threshold. - -Both production-configured comparisons passed a 650-second wait: MicroVM -callback budget 28,920 seconds, standard budget 3,720 seconds. Each produced -one successful marker read and one post-tool callback. The unset default -cancelled at 600.034 seconds; the explicit one-second failure control cancelled -without reading the marker. A final short configured probe also verified that -diagnostics go to stderr and stdout remains valid JSON. - -The local probe does not simulate a real VM snapshot. The subsequent -[image `4.0` AWS retest](./645-p3-callback-live-20260915.md#real-credential-expiry-and-the-original-approval-deadline) -passed a real freeze beyond credential expiry, renewed the same task's keys and -preserved the callback until its original approval deadline. Approval, denial, -cancellation, short/grace windows and wake after a supervisor outage also passed. -The separate service-reported resume-hook connection refusal remains unresolved. diff --git a/docs/verification/645-p3-capacity-scan-20260916.md b/docs/verification/645-p3-capacity-scan-20260916.md deleted file mode 100644 index 824f420fa..000000000 --- a/docs/verification/645-p3-capacity-scan-20260916.md +++ /dev/null @@ -1,84 +0,0 @@ -# ADR-021 P3: AWS capacity scan and interrupted-read checks - -Verified September 16, 2026, in account ``, `us-west-2`. -All checks passed and all temporary resources were removed. This is bounded -pagination and reconciliation evidence; it does not establish an arbitrary -production retention volume or complete the coordinated writer upgrade/rollback. - -## Deployment and data - -The temporary Lambda used the exact deployed ConcurrencyReconciler artifact, -SHA-256 `ZbDOQoSBUD3t4e0efIyopTKifQwKkQBBF8IA2wBLMRw=`, with the same Node.js 24, -ARM64, 256 MB memory and 300-second timeout. Its DynamoDB action sets matched the -normal reconciler's policy, with resource ARNs substituted to two owned tables. -It had no normal-table access, event schedule or coding workers. - -Each table contained 600 records, with 5,000 bytes of padding per record to -exercise real AWS pagination. Task records covered active held reservations, -one terminal held reservation and an older active task without a reservation -marker. Three counters were deliberately too high or too low. - -| Strongly consistent scan | Rows | Observed pages | Consumed read capacity units | -|---|---|---|---| -| Tasks | 600 | 4 | 779 | -| Counters | 600 | 3 | 761 | - -These page and capacity measurements come from separate observer scans of the -same tables. The deployed handler uses projections and does not itself request -consumed-capacity telemetry. Its successful results required accounting for -all 600 users. - -A subsequent read-only measurement of the normal development deployment found -86 task rows in one page (58 read capacity units) and 36 counter rows in one page -(4 units). The 600-row fixture therefore exceeded this deployment's current -row count and scan volume. It does not establish capacity for a future production -retention policy or workload. - -## Results - -The first deployed invocation: - -- Repaired counters from 3, 0 and 7 to their actual one held reservation. -- Released the terminal task's reservation, changing its counter from 1 to 0. -- Preserved the other 598 active held reservations. -- Left the ambiguous older task's counter at 2 and logged - `CONCURRENCY_RESERVATION_UNKNOWN`. -- Logged `scanned=600`, `corrected=3`, `errors=0`. - -Lambda reported **1,475.42 ms** duration and **112 MB** maximum memory. -The invocation receipt is `2c55ca9e-9d47-4b06-947c-56d4f665e455`. -These numbers describe this fixture, not a throughput or maximum-size guarantee. - -Two interrupted-read cases ran the production handler locally against the -real AWS tables. The SDK boundary discarded the next read by throwing an -explicit fixture error before the second page of either the counter scan or -the task scan. Actual earlier AWS pages and their receipts were retained. -In both cases the handler failed with **zero update/transaction requests**, and -all 600 counters remained unchanged. A deliberate overcount was left in place -before these checks so that a premature repair would have been visible. -These were injected interruptions, not observed AWS outages. - -After the older unmarked task became terminal, a second deployed invocation -repaired its counter from 2 to 0. It also repaired the deliberate overcount -from 9 to 1. It logged `scanned=600`, `corrected=2`, `errors=0`, with -**343.39 ms** duration and **113 MB** maximum memory. -Receipt: `5bba4ca9-014e-47de-abbf-e0d4be98c9ab`. - -The earlier [normal-role live repair](./645-p3-user-sleep-20260916.md) separately -verified correction and terminal release while a real MicroVM waited for -approval. Together, these results add actual deployed writer permissions, -multiple scan pages, partial-read safety and conservative legacy handling to -the existing [transaction evidence and upgrade procedure](./645-capacity-reservations.md). - -## Cleanup and limits - -The private `backgroundagent-dev-p3-capacity-20260916` function, role, log group -and both 600-row tables were deleted after ownership-tag checks. Seventeen -function log events and final task/counter snapshots were saved first. -Read-only absence checks completed at **17:35:25.915 UTC**. The normal -reconciler's artifact and environment remained unchanged. - -Evidence is retained in -`/tmp/abca-645-p2-clean-20260913/p3-capacity-scan-20260916`. -The full old-writer drain/upgrade/rollback procedure and production retention -volume remain separate gates. No workload-based memory change is proposed. diff --git a/docs/verification/645-p3-capacity-upgrade-20260916.md b/docs/verification/645-p3-capacity-upgrade-20260916.md deleted file mode 100644 index 80a71d8f2..000000000 --- a/docs/verification/645-p3-capacity-upgrade-20260916.md +++ /dev/null @@ -1,100 +0,0 @@ -# Capacity protocol upgrade and rollback rehearsal — 2026-09-16 - -The isolated AWS rehearsal passed: old counter writers can lose another task's -seat when cleanup repeats; the current reservation protocol preserves it. -Pausing admissions, draining tasks, switching protocols, rolling back after -another drain, and upgrading again all worked in the bounded fixture. - -This is a table-protocol rehearsal. It does not claim that the normal -deployment's admission routes were paused or that its durable executions were -drained. - -## Exact scope - -The fixture ran in account ``, `us-west-2`, from **19:39:32 to -19:39:57 UTC**. It used two private DynamoDB tables, two restricted Lambda roles -and five Lambda functions under -`backgroundagent-dev-p3-capacity-upgrade-20260916`: - -- Separate old admission and release entry points. -- Separate current admission and release entry points. -- The exact deployed concurrency-reconciler ZIP, with only environment table - names redirected to the private tables. - -The old `admissionControl` and `decrementConcurrency` function bodies were -extracted unchanged from -`0f4545c77f01c5d905d9e32b8b375d9910e18324`. The current -`task-concurrency.ts` module was bundled unchanged from source `35d5515b`. -The wrapper admitted only nine fixed task IDs with one fixed owner. Its role -allowed `GetItem`/`UpdateItem` only on the two private tables. The reconciler -used the previously reviewed equivalent table permissions, redirected to those -tables. - -The writer ZIP SHA-256 was -`485093d998cc9c11576f08306b5fbb8e3696c0defabe45650ccd19fad4210261`. -The reconciler's Lambda code hash was -`ZbDOQoSBUD3t4e0efIyopTKifQwKkQBBF8IA2wBLMRw=`, matching the normal deployment. -Source-function hashes, bundle inputs, deployed configurations and role policies -are retained with the evidence. - -## Observed checks - -All **12 checks** passed, using 27 completed Lambda invocations and seven actual -invocation rejections while reserved concurrency was zero. - -| Check | Evidence | -|---|---| -| Enforced admission pause | AWS rejected invocation of the paused old/current admission functions | -| Old replay defect | Two active old tasks gave count 2; cleaning one twice produced 0 while the other remained active | -| Ambiguous legacy task | The current reconciler left the unmarked active task and its counter untouched; the fixture's drain check prevented progression | -| Approval occupancy | A current task awaiting approval retained its reservation; early release returned false | -| Capacity limit | A third admission at limit 2 was rejected without a reservation | -| Lost finalization reply | The wrapper deliberately threw after the real release committed; a repeated release preserved the other task's count of 1 | -| Cancellation | A terminal cancelled task released its own slot | -| Completion | A completed task released its own slot | -| Queue boundary | `QUEUED` could not acquire; changing to `SUBMITTED` permitted acquisition | -| Rollback with active work | Admissions remained paused and the fixture's drain check detected an active held reservation | -| Drained rollback | After all current reservations were released, a fresh old-protocol task admitted and finished with count 0 | -| Re-upgrade | After draining the old protocol again, a fresh current-protocol task admitted and released successfully | - -The fixture's runner explicitly changed task statuses to represent work, -approval, failure and cancellation. It did not start a coding worker, run the -production queue-pickup handler, or reproduce a durable execution's entire -finalization path. The deliberately lost reply occurred after the release helper -returned, before its caller received a successful Lambda result. It establishes -safe repeated release after a committed transaction, not an AWS service outage. - -The drain check is part of this operator rehearsal, **not a newly implemented -production deployment interlock**. Separate admission/release functions let the -fixture stop admissions while allowing cleanup. A normal rollout must identify -and stop every actual admission route without disabling cleanup for old work. - -## Cleanup and remaining gate - -At the final strong reads, all nine owned task records were terminal, no held -reservation remained and the single counter was zero. All five functions were -paused before cleanup. Their 120 log events, final table rows and invocation -receipts were saved. - -All five functions, two roles, two tables and five log groups were deleted. -Ownership checks and **14 absence checks** passed at -**19:42:37.730 UTC**. No normal task record, counter or deployment setting changed. - -The [capacity runbook](./645-capacity-reservations.md) still requires a deployment -specific inventory and drain of all old admission/cleanup writers, including -pending uploads, queue pickups and retained durable executions. The -[600-user scan test](./645-p3-capacity-scan-20260916.md) supplies separate bounded -volume evidence. Together these checks narrow the remaining rollout work; they -do not establish arbitrary retention scale or an already-executed normal-fleet -migration. - -Raw evidence: -`/tmp/abca-645-p2-clean-20260913/p3-capacity-upgrade-20260916`. - -Permanent private archive: -`/Users/sphias/.local/share/abca-verification/645-p3-20260916/capacity-upgrade-evidence.tar.gz` -(56 files, 9,265,702 bytes, mode `0600`, SHA-256 -`e1499cdde2ad8573996365b09553f4a62fc7ec790a932c0054370b5ded03784e`). -Every file was checked against its hash manifest. It includes source -`2b12ce88`, both exact function ZIPs, the legacy function bodies, all receipts, -logs and cleanup evidence, plus the subsequent ECS sizing-comment check. diff --git a/docs/verification/645-p3-cloud-continuation-20260917.md b/docs/verification/645-p3-cloud-continuation-20260917.md deleted file mode 100644 index df70be4d7..000000000 --- a/docs/verification/645-p3-cloud-continuation-20260917.md +++ /dev/null @@ -1,129 +0,0 @@ -# P3 cloud continuation acceptance — September 17, 2026 - -The final private MicroVM matrix passes approval, denial, cancellation and explicit -expiry after planned worker retirement, plus approval and denial on the original -sleeping worker. These checks exercise the real guest, model, SDK, lifecycle hooks, -AWS storage, Durable coordinator and API decision handlers. - -This record covers the isolated acceptance deployment. Normal nested migration, -activation and the signed-in CLI path have their own completion gates. - -## Exact build and fixture boundaries - -- Account/Region: `` / `us-west-2`. -- Stack: `backgroundagent-dev-p3-recovery-20260917`. -- Image: `abca-645-p3-recovery-20260917`, version `3.0`, 8,192 MiB. -- Guest ZIP SHA-256: - `a4ab66250ab20aceeefab05d5d13eea8387f7fa4d6a91376bc9265a06cb68013`. -- The ZIP contains 116 files and is 518,981 bytes. -- Coordinator version `4` ran the retirement matrix and wake-deny; version `5` - refined the private command-specific gate for wake-approve. Both use the same - image and production lifecycle implementation. - -The private wrapper selects dedicated tables/buckets and a repository-free -workflow. It sets the real AWS worker lifetime to 600 seconds so production -retirement is reached in approximately five minutes. It does not mock the clock, -AWS responses, lifecycle transitions, checkpoint transfers or model responses. -The configured sleep delay is 30 seconds; production's default remains 600. - -The wrapper invokes the production scheduled manager every 20 seconds to keep the -test short. Production schedules it every five minutes. Approval and denial call -the production handlers with the fixture owner's identity; authenticated HTTP/CLI -acceptance is a separate normal-deployment check. - -## Passing cases - -| Case | Task ID | Verified outcome | -|---|---|---| -| Approve after retirement | `01M2RHVXTC3VZT63008V2D2BVX` | One replacement; the approved command reads the original marker and its exact SHA-256 from restored disk | -| Deny after retirement | `01M2RHVXTGPRGR01ATR0CEGE0N` | One replacement; the denied command does not execute and the agent acknowledges the saved denial | -| Cancel while parked | `01M2RHVXTGPSSCBRM42GD46ANG` | Request/task cancelled; no replacement or terminal delivery | -| Explicit expiry while parked | `01M2RHVXTGB9H7DFHATTF2VEVJ` | The 480-second deadline closes the request; one replacement acknowledges the timeout without executing the command | -| Approve on the original worker | `01M2RN4WWRK0BHJ0VCWDAAT125` | Same worker wakes, reads the saved marker/checksum and finishes | -| Deny on the original worker | `01M2RH8HN8GPGYW08R4HBXWCB6` | Same worker wakes, respects denial and finishes | - -For every retirement case the original worker was observed `SUSPENDED`, then -`TERMINATED` with `stateReason: "Success."`. The coordinator fenced the old worker, -parked the task and released its capacity reservation before the human decision. -An unanswered request had no DynamoDB TTL. - -The harness verifies each immutable manifest's object version, byte count and -SHA-256, saved workspace location and positive accumulated model cost. Replacement -tasks keep their original image version and final costs include both worker runs. -Successful cases finish with released reservations, zero user capacity and no -outstanding continuation launch. Duplicate decision submissions return 404. - -The agent's normal reasoning decides whether the saved action is still relevant. -The platform checks exact task/request ownership and approved tool inputs; this -does not add an action-relevance rechecking system. - -## Findings corrected during integration - -The real retirement tests found a workflow boundary missing from the earlier -process-level tests: hydration overwrote the prepared continuation prompt with -the original task prompt. The resumed model therefore never received the saved -human decision. `workflow/runner.py` now fills the user prompt only when it is -empty, matching the existing treatment of a prepared system prompt. Five tests -exercise the real hydration and agent-run handlers; four failed before the fix. - -The approved continuation instructions also explicitly preserve every saved tool -input field, including descriptions. The exact-input permission check remains -unchanged. Restored denial feedback reports its original timestamp instead of -incorrectly describing a decision many minutes old as a recent 60-second denial. - -Earlier integration corrected retained-request hook checkpoint validation and -AWS's 64-character Durable execution-name limit. Dispatch errors now include the -operation, coordinator version, name length and bounded validation detail. - -One first wake-approve fixture paused at an incidental `ls` command before the -intended marker command. That run was cancelled and excluded from marker -acceptance. The reproducible fixture now gates only the exact intended command, -validates the approval's input before answering, and correlates tool results with -their trace/turn. It fails if parallel calls make that correlation ambiguous. - -## Cleanup and reproducibility - -All 15 task records from final and diagnostic attempts were terminal, all 15 -worker leases were closed, reservations were released and continuation launch -records were absent before deletion. The stack and its 40 tracked resources were -removed. Cleanup also removed automatically created Lambda log groups and checked -absence of the exact owned functions, buckets, images and network connectors. - -Private scripts and raw evidence are outside the Git worktree: - -```text -/Users/sphias/.local/share/abca-verification/645-p3-integration/ -``` - -The final matrix is in `cloud-replacement/results/`; `cleanup.json`, -`cleanup-resources.json` and `terminal-task-and-lease-audit.json` preserve cleanup -proof. Failed and invalidated earlier cases remain archived separately. - -`run.py` provides local, SDK, real S3, DynamoDB and coordinator checks; `cloud.py` -creates a fresh isolated stack and image for the real worker matrix. It uses the -selected worktree's production implementations and records the artifact digest. -The scripts deliberately do not belong in the PR. - -Full agent verification after these corrections passes **2,159 tests**, with -13 explicitly opt-in cases skipped and **86.34%** coverage. The real SDK suite was -run separately. The full infrastructure suite passed 5,223 tests before the -managed runtime-pin addition; 245 focused infrastructure tests and compilation/ -lint subsequently passed for that addition and dispatch diagnostics. - -## Corrected-image follow-up — September 18 - -The later normal-workflow progress/capture race is documented in the -[dedicated record](./645-p3-progress-race-20260918.md). After its correction, a -fresh portable cloud run built image `abca-645-p3-it-09180207b210de:1.0` from -artifact `924a1b51fe6b9aa62f61191a6bde9b10df01d65873181a9385489191492cadbe`. -It passed both same-worker wake/approval and approval after confirmed retirement -and replacement, with real marker/checksum assertions. - -- approve: `01M2S4A6MB2CZJQRP0ZYMT7EYZ`. -- wake-approve: `01M2S4A6MB7WH4A781WFFMQCPK`. - -The owned stack and all 40 tracked resources were removed; cleanup finished at -`2026-09-18T02:30:30.707Z` with no leaked resources. Source hashes, -image receipt, full timelines and cleanup checks are in the private harness -`runs/20260918-progress-race-cloud`. The full agent suite with this fix passes -2,162 tests, with 13 opt-in skips and 86.37% coverage. diff --git a/docs/verification/645-p3-command-races-20260916.md b/docs/verification/645-p3-command-races-20260916.md deleted file mode 100644 index 031f31c31..000000000 --- a/docs/verification/645-p3-command-races-20260916.md +++ /dev/null @@ -1,195 +0,0 @@ -# ADR-021 P3: live cancellation, approval and polling-failure races - -Six required AWS cases passed on September 16. They cover cancellation around -sleep/wake commands, directly observed `SUSPENDING` and restore `PENDING` states, -approval during an accepted suspension, and three consecutive failed status -reads. Nine workers were used: three harness-invalidated attempts were retained, -excluded and replaced. - -**P3 remains incomplete.** These checks do not explain the four original -[resume connection refusals](./645-p3-resume-refusal-investigation.md). -Normal automatic suspension remains disabled. - -## Test boundary - -- Account ``, profile `sphia-dev`, region `us-west-2`. -- Normal worker image `backgroundagent-dev-abca-agent:5.0`, artifact SHA-256 - `9b3150e9e5991cc9fcc8a4adbb2cfbd8f97399f0c16baa9c2b5fe5b101e7b035`. -- Temporary durable coordinator `backgroundagent-dev-p3-races-20260916`, - with an exact task whitelist, dedicated role and private suspension switch. -- Compiled production handler and supervisor, plus a private wrapper that - invokes the normal approve/cancel handlers at specific SDK command boundaries - or deliberately hides selected successful Get responses as `TimeoutError`. -- No guest changes. Observed lifecycle hooks ran as PID 1. Each task allowed - one harmless `Read` of `/etc/os-release`, six turns and a $1 budget; worker - and durable execution lifetimes were capped at 1,800 seconds. -- Original approval windows were 600 seconds, except the restore-cancellation - case's 150 seconds. Its normal pre-deadline wake occurred while approval was - still pending. -- Decision/GetTask handlers received fixed synthetic owner events through - direct Lambda invocation. This checks their deployed code and permissions, - not API Gateway authentication. - -The wrapper records the real service receipt and the decision invocation before -returning control to the supervisor. Thus a test proves which command boundary -was crossed. A separate AWS state read records the state actually observed; -an accepted Suspend request alone does not prove `SUSPENDING`. - -## Required results - -| Case | Task | Observed state at intervention | Result | -|---|---|---|---| -| Cancel after pre-command read, before Suspend submission | `01M2NDHS2AC0569CXMJGA5ZR6Z` | `RUNNING` | Cancel committed; subsequent real Suspend rejected; task stayed canceled | -| Cancel after Suspend acceptance, before supervisor post-command read | `01M2NDHS2BYXTGNKVTW1MMHSRJ` | `RUNNING` | Cancel committed; checkpoint refused the changed task; cleanup passed | -| Cancel after Resume acceptance | `01M2NDHS2B73V9S034FBFFE7RM` | Restore `PENDING` | Task stayed canceled; no tool result | -| Approve after Suspend acceptance | `01M2NE0YFTXKRAVH2XPW702K2N` | `RUNNING`; later observed sleep/restore | Same gate resumed; approved Read succeeded once | -| Cancel during observed suspension | `01M2NE0YG0PT357AS5K2MVD33G` | `SUSPENDING` | Task stayed canceled; no tool result | -| Three failed status reads while asleep | `01M2NENZNP5BE95TCSE96SYCJ3` | `SUSPENDED` for all three hidden responses | Failure count reached three; task failed and cleanup passed | - -All six reached terminal worker state, released their saved capacity reservation, -left a zero user counter and removed their launch/payload objects **before any -independent watcher cleanup**. Durable executions succeeded by completing their -expected task finalization; that does not mean the intentionally failed task -became successful. - -Strong final reads and logs retain the original approval creation time/window, -coordinator first-observation time and non-increasing worker lifetime deadline. -The approved case has one Read result containing `PRETTY_NAME` and a complete -trace with zero dropped records. Each negative case has one gated Read attempt -and no tool result. Those terminated tasks do not all have a final trajectory; -the retained task events and lifecycle/control evidence establish this narrower -claim. - -## Specific race evidence - -**Cancellation before submission.** Worker -`microvm-53ec3d4d-d6cc-3542-a910-9d00cc7c4ae0` entered the wrapper's Suspend -boundary at 15:36:35.245 UTC. Cancel returned 200 at 15:36:36.932, before the real -Suspend returned `ValidationException` at 15:36:38.788, receipt -`edccbf8a-436e-4e06-9fec-f3e287db9afa`. The supervisor's post-command read preserved -cancellation despite that service error. - -**Cancellation during checkpoint.** Worker -`microvm-dd64aab4-23f6-3714-91d7-25ea3d9d20b4` received Suspend acceptance at -15:37:50.256, receipt `f7d03f96-a405-4c39-a81d-68c97f324361`. Cancellation committed -before its checkpoint transaction failed with `TransactionCanceledException`, -receipt `7B2F9PRFP9JSUS95BPI0LNMPEVVV4KQNSO5AEMVJF66Q9ASUAAJG`. -The hook returned HTTP 503 at 15:37:50.431 and marked suspension ineligible. -No resume hook ran. This is an observed application checkpoint rejection after -cancellation, not an unexplained wake-connection failure. - -**Cancellation during restore.** Worker -`microvm-ece908cf-7fd0-385e-aff9-bd3878500bb8` received Resume acceptance at -15:41:51.965, receipt `528dbbed-3b34-43f4-9a9f-08332f78fd0e`. A fresh Get reported -`PENDING` at 15:41:52.245, receipt `3612d0cb-3f28-4676-b171-94849583ffaf`. -Cancel returned 200 at 15:41:52.809, before the wrapper returned at 15:41:52.814. -The 150-second approval remained pending; no approved tool ran. - -**Cancellation during suspension.** Worker -`microvm-1f1f49e0-217e-3c65-a44b-45ffe9fe3fd9` received Suspend acceptance at -15:56:03.561, receipt `4e767b18-89d1-4557-ab76-4437b51ab69f`. Get reported -`SUSPENDING` at 15:56:03.744, receipt `e5ca583d-7ab4-4c4b-aed1-ec114ebe5e81`. -Cancel returned 200 at 15:56:05.344, before the supervisor regained control. - -**Repeated polling failure.** Worker -`microvm-ad53c767-8fae-3f61-8b94-eb6ee0202e57` remained genuinely suspended while -the wrapper deliberately hid three real Get responses. Across consecutive -durable cycles, logs recorded `stage=substrate-read`, `error_type=TimeoutError` -and failure counts 1, 2 and 3. The final reason was -`MicroVM supervisor: substrate-read-failed`. The coordinator sent Terminate; -a fresh worker read confirmed `TERMINATED` at 15:54:25.536. There was no watcher -repair. - -## Rejected attempts and harness corrections - -These attempts are not included in the six-case acceptance count: - -1. Approval task `01M2NDHS2B6087T1925SRRPDSX` completed its Read, but the watcher - demanded `TERMINATED` while AWS still reported `TERMINATING`. Its fallback - cleanup ran. A fresh approval task supplied clean acceptance. -2. Poll task `01M2NDHS2BZYNEMHAQ7JDY41ZP` never received its intended faults: - esbuild renamed `GetMicrovmCommand` to `GetMicrovmCommand2`, defeating a - `constructor.name` comparison. It was explicitly canceled. The injector now - checks SDK command classes with `instanceof`. -3. Poll task `01M2NEBTY1CVS4B56ZJDVKS8CE` received all three intended faults and - the coordinator sent Terminate, but the watcher compared a `SUSPENDED` - sample taken **before** it read the completed durable execution. Fallback - cleanup invalidated that acceptance attempt. - -The final watcher observes a fresh terminal worker state for at most 30 seconds -after durable finalization. Task, worker and execution reads are separate calls, -so their values are not one atomic snapshot. This observation window sends no -repair commands. Original sources, logs, failures and replacement identities -remain in the private evidence. - -## Feedback correction and validation - -The polling test exposed an application feedback gap: the persisted supervisor -reason was classified as `unknown` / `user`, with the title “Unexpected error.” -Commit `d150991715c27dafecbe1eb4ab7ce44e98e02bff` gives those exact known reasons a -compute/platform classification and the title “The MicroVM status could not be -checked.” Guidance names `microvm_supervisor_request_failed`, the task/worker -IDs, stage, error type and AWS receipt, and requires checking termination and -saved progress before replacement. - -The same commit corrects two stale cancel-handler comments: the coordinator -saves the runtime ARN, and sleeping-worker AWS memory-quota consumption remains -unverified. The cancel handler's commentless TypeScript output is unchanged. - -Validation passed: 134 focused classifier tests and the full repository build, -including 5,025 CDK, 928 CLI and 1,947 Python tests; Python coverage was 86.43%. -The optional DynamoDB Local suites were skipped (56 CDK and 11 Python cases); -earlier transaction evidence remains separate. - -## Normal deployment of the feedback fix - -At 16:13:15.851 UTC, `backgroundagent-dev` was verified `UPDATE_COMPLETE`, -running coordinator alias `live → 7`, code SHA-256 -`/JcZuu12fMVSOoO4NwxsLGlLIJ5w9XYjAyx7cuBRA+I=`. Version 6 remains available, -and its environment matches version 7 exactly. Image 5.0, Cedar layer version 2 -and both disabled suspension settings were preserved. - -The reviewed change set contained 29 Lambda `Code` updates plus the retained -old/new coordinator version and alias changes: 32 changes, no replacements. -All 29 deployed code hashes matched the actual reviewed S3 ZIP contents -(409,965,183 bytes hashed). - -A direct invocation of the normal GetTask handler for the real failed poll -task returned the new title, compute/service classification, nonretryable flag -and specific supervisor-log/termination guidance. Existing saved task errors -therefore receive the corrected explanation without rewriting their records. - -One preview was rejected before execution. In this environment, the SDK's -`GetTemplate` result replaced every non-ASCII character with `?`, including `§` -and dash characters in descriptions. Reusing that result would have changed -unrelated descriptions/dashboard text and replaced the Cedar layer. The -replacement preview used the exact previously deployed S3 template, verified -against SHA-256 -`e9e58901d4fef62a9b9a07b24068b61cc2cf51849847c261df1f6ab7597b8698`. -AWS then reported only the intended 32 changes. The rejected preview and the -character-for-character loss comparison remain evidence; the underlying cause -within the retrieval path was not isolated. - -Deployment proof is under -`/tmp/abca-645-p2-clean-20260913/p3-race-feedback-rollout-20260916`. - -## Cleanup, evidence and remaining scope - -Cleanup was verified at 16:04:09.508 UTC. All nine workers were terminated. -The private function and its four versions, role/policies, suspension parameter -and log group were deleted; 1,236 function log records were archived first. -Nine zero counters were removed with revision checks. Task/event/trace audit -records remain. Immediate function-deletion observation lagged; subsequent -read-only checks confirmed all temporary resources absent. - -Private evidence is under -`/tmp/abca-645-p2-clean-20260913/p3-command-races-20260916`, including four exact -function bundles, the task whitelist/policies, run histories, hook/API logs, -strict audit, excluded attempts and cleanup proof. - -Still open: late approval versus timeout races, credential-refresh failure -injection, remaining durable registration/unknown-worker recovery, other-backend -permissions/networking, a final cloned-repository workflow, capacity migration -and compatible rollout checks. The [implementation plan](./645-p3-implementation-plan.md) -tracks their order. No original connection refusal occurred in this matrix; -that does not close its separate investigation. diff --git a/docs/verification/645-p3-connection-close-rollout-20260916.md b/docs/verification/645-p3-connection-close-rollout-20260916.md deleted file mode 100644 index 254c917c7..000000000 --- a/docs/verification/645-p3-connection-close-rollout-20260916.md +++ /dev/null @@ -1,196 +0,0 @@ -# ADR-021 P3: connection-close rollout and normal-image acceptance - -Verified September 16, 2026, in `us-west-2`, account ``. -The normal `backgroundagent-dev` stack now runs MicroVM image **6.0** and -coordinator **10**. Approval, denial, timeout and cancellation while asleep -passed on that image using a private AWS Durable coordinator and normal decision -handlers. Both normal automatic-suspension switches remain **off**. - -## What changed and why - -An HTTP connection is the channel AWS uses to deliver a lifecycle message. -Keeping it open normally saves the work of opening another one. A frozen worker, -however, can wake with both an old connection and an expired timer that wants -to close it. - -The [instrumented investigation](./645-p3-wake-transport-20260916.md) reproduced -the exact refusal while the original server still owned its listening socket. -Its restored event loop closed the old suspend connection; no fresh connection -or resume request appeared. Two unchanged cases sleeping longer than 90 seconds -passed. Three candidate cases directly verified that an explicit -`Connection: close` response closed the suspend connection before freezing and -that resume arrived on a new connection. - -Source `ccab2eecbf8aa28d4b9782e003f790f14078c7a5` adds that header to both -suspend and resume JSON responses, including errors. It leaves the original -server command, five-second idle timeout, checkpointing and approval barriers -intact. It also recognizes `Resume lifecycle hook timed out` as -`MICROVM_RESUME_HOOK_FAILED`, with service/admin guidance rather than automatic -retry advice. - -These guest observations support a specific connection-lifetime correction. -They do not reveal the service client's failed dispatch, establish a failure -rate, or prove that every historical wake failure had the same cause. The -service-side questions remain in [F01, F03, F05 and F08](./645-lambda-microvm-service-feedback.md). -No report was sent to the service team. - -## Reviewed normal deployment - -CloudFormation reached `UPDATE_COMPLETE` at **22:22:56.959 UTC**. - -| Item | Verified result | -|---|---| -| Source | `ccab2eecbf8aa28d4b9782e003f790f14078c7a5` | -| Root resource count | 475 | -| Coordinator live alias | Version 10; previous version 9 retained | -| Coordinator code SHA-256, base64 | `hrLwpI9HJ15OctduM1tv4VelPBXIRQ2qeGOqdq+zdqw=` | -| Normal image | `backgroundagent-dev-abca-agent:6.0`, `ACTIVE` / `SUCCESSFUL` | -| Previous image | Version 5.0 retained | -| Worker memory | 8,192 MiB | -| AgentCore | Version 5 retained; this rollout did not change its container | -| Bootstrap | 1.8.0 | -| Normal sleep gates | Environment `false`; SSM `false`, parameter version 1 | - -The normal image artifact is 486,771 bytes, SHA-256 -`191ae368a2a4d28bb8910caabef6a2b99b5521a474cff60ef6542456fb679490`. -Compared with the exact image 5.0 ZIP, 107 entries are identical and four differ: -the lifecycle HTTP module, a runner docstring with an unchanged executable -syntax tree, and the JSON/Markdown copies of the already-established adjustable -sleep contract. The Dockerfile is identical and the image contains no transport -probes. All image properties other than the source artifact match version 5.0. - -The reviewed change set, `p3-connection-close-20260916`, contained 16 resource -changes: 11 classifier-consumer function code updates, the coordinator version -and alias transition with retention of the old version, the image artifact, and -the build role's exact artifact digest. No modified resource required -replacement. The executed template is 706,153 bytes, SHA-256 -`856bb6aaba07bf2eb1c8d5c0ad874217c6fa7c1f19b84a03b92b0dec14624f51`. -It was prepared from the exact previous S3 template bytes, avoiding the lossy -`GetTemplate` baseline discovered in an earlier rollout. - -After deployment, all 11 function code hashes matched the actual reviewed ZIPs -downloaded from S3. The image version, unchanged image configuration, retained -coordinator version and disabled sleep settings were checked separately. - -## Four real Durable workflows - -The temporary function `backgroundagent-dev-p3-close-acceptance-20260916:1` -ran the production Durable handler with five-second polling. Its wrapper bound -launches and permissions to four fresh fixture identities, the normal image -6.0, and a 1,800-second maximum worker lifetime. Only its private sleep switch -was enabled. Each case had a six-minute watcher, a $1 task budget and a six-turn -limit. No transport failure was injected. - -The normal approval, denial, cancellation and read handlers were invoked through -Lambda with the fixture identities. This tests the deployed handlers, not an -additional API Gateway authentication/ingress path. These were repository-free -tasks requesting one guarded Read of `/etc/os-release`; no notifications or -repository publication were requested. - -| Case | Task | Worker | Result | -|---|---|---|---| -| Approve after sleep | `01M2P4WX07MY6AP3VAJ1X5BQXX` | `microvm-2e038148-adce-37eb-8e83-629ef9c16eda` | Approved; exactly one successful Read; task `COMPLETED` | -| Deny after sleep | `01M2P4WX086HH030697Z4XK8W8` | `microvm-4149fb1d-42ba-3fdc-b509-a548bba70627` | Denied; Read blocked; task `COMPLETED` | -| Timeout wins | `01M2P4WX08SKVPCA55E4AX0W7T` | `microvm-92ff298a-dc28-3e59-9994-1375f79a43a7` | Original approval `TIMED_OUT`; late approve returned HTTP 404; Read blocked | -| Cancel while asleep | `01M2P4WX08WQZH86E72YQGSDNA` | `microvm-b8414dd8-6e85-35c1-b32c-a496a4c553c2` | Task `CANCELLED`; worker terminated without a Resume request | - -All four AWS Durable executions ended `SUCCEEDED`. Each worker terminated, -each reservation was released, each counter reached zero, and each launch -payload prefix was empty before fixture cleanup. No watcher repair was needed. -The three non-cancelled tasks retained complete traces with one Read intent -each and zero dropped records. The denied Read returned `AUTHORITATIVE DENY`; -it did not execute. The cancelled task was not required to produce a completed -agent trace after termination. - -### Actual wake delivery - -The normal sleep switch prevents new automatic suspension. It intentionally -does not prevent a decision handler from waking an already-suspended worker. -Approve and deny therefore used the normal API's immediate wake path; timeout -used the private coordinator's production supervisor. - -| Case | Resume accepted, UTC | AWS Resume request ID | Original PID 1 resume hook | -|---|---|---|---| -| Approve | 22:35:14.410 | `04923c77-f85c-4274-85ee-ebbdc92b2935` | HTTP 200 at 22:35:14.832 | -| Deny | 22:37:32.764 | `5110864c-a2b1-4141-99ef-5d1d7e374e14` | HTTP 200 at 22:37:33.259 | -| Timeout | 22:40:07.397 | `154a247f-2c0e-4e0a-9221-31f8da62c76e` | HTTP 200 at 22:40:07.760 | - -Approve was delivered after more than 60 seconds of observed suspension. The -timeout case woke before its original 150-second approval deadline, then let -that deadline expire; wake did not grant a new window. Its later approval -received HTTP 404 `REQUEST_NOT_FOUND`. - -The cancel case was observed `SUSPENDED` at 22:42:45.314. The normal cancellation -handler returned HTTP 200 at 22:42:56.987, Lambda invocation receipt -`34e57a5e-eba9-459f-aee5-cb7a0c5d7503`. The worker terminated, the approval -remained pending, and no Resume command or successful Read appeared. - -An accepted API call alone was never the pass condition. The audit required -the guest's hook results where applicable, original decision/deadline outcomes, -tool evidence, final Durable/task state and resource cleanup. - -## Error feedback and local checks - -Six synthetic terminal records exercised the normal deployed GetTask handler: -raw, legacy and stable-code timeout forms; raw refusal; raw generic wake -failure; and preservation of an older stable classification code. All passed, -and all six synthetic rows were deleted. Recognized new wake failures receive -nonretryable service/admin guidance explaining that a refused-connection message -alone does not establish a closed listener. Existing stable classifications -keep their precedence. - -The source passed agent lint, formatting and type checks, **1,955 agent tests**, -TypeScript compilation, ESLint, and **5,053 CDK tests** across 229 suites. -The 11 agent and 56 CDK opt-in DynamoDB Local cases were skipped in these runs; -this change did not alter their transaction conditions. Eight new route cases -cover both lifecycle endpoints with HTTP 200/400/409/503 despite an incoming -keep-alive request. The classifier suite covers timeout classification and -resulting retry guidance. - -These four acceptance executions used real AWS Durable execution. Earlier -transport comparisons used a local step adapter after two separate AWS -`Stopped.ByService` errors during startup. Passing this round does not identify -the cause of those earlier service stops. - -## Cleanup and evidence - -The four-case audit passed with zero errors. Scoped cleanup completed at -**22:49:26.544 UTC**. It removed the private coordinator and versions, its role, -private SSM parameter, private function log group and four zero-valued fixture -counters. Ownership tags, terminal executions and task/counter versions were -checked before deletion; absence was verified afterward. All 331 private -function log events were archived. Normal image versions, shared logs and -ordinary task records retain their existing retention policy. - -A fresh normal-state check at **22:49:27.184 UTC** confirmed coordinator 10, -image 6.0 `ACTIVE` / `SUCCESSFUL`, 8,192 MiB, stack `UPDATE_COMPLETE`, and the -normal SSM switch still `false` at version 1. - -Private working evidence is under: - -- `/tmp/abca-645-p2-clean-20260913/p3-connection-close-rollout-20260916` -- `/tmp/abca-645-p2-clean-20260913/p3-close-acceptance-20260916` - -The permanent archive is -`~/.local/share/abca-verification/645-p3-20260916/connection-close-rollout-evidence.tar.gz`. -It contains 122 files, occupies 827,593,551 bytes, and has SHA-256 -`92aaa3c29a6ff5dbc1cf7acf81d7fe354b7aed4fdbcf774359e793ad466ce237`. -Every member's size and hash were verified against the manifest; permissions -are `0600`. It includes the exact rollout templates/artifacts, deployed function -ZIPs, source comparison, fixture code, API receipts, guest logs, Durable history, -task/tool evidence and cleanup proofs. The decisive earlier refusal and -instrumented comparison have their separate archive in the transport report. - -## Remaining completion gates - -The application correction and these four normal-image workflows are complete. -The broader [P3 plan](./645-p3-implementation-plan.md#implementation-progress) -still requires the wider final-image workspace/credential/decision matrix, -other-backend permission and network checks, and the normal deployment's -coordinated capacity upgrade/drain and rollback validation. Service token -retention and recovery without guest identity logs also remain open. - -Automatic suspension stays disabled while those gates are open. The user sleep -setting retains its 600-second default and zero-to-stay-awake option. Service -trace requests remain separate from the verified application change; finite -passing runs cannot certify every future wake. diff --git a/docs/verification/645-p3-continuation-protocol-20260917.md b/docs/verification/645-p3-continuation-protocol-20260917.md deleted file mode 100644 index 9ee9d71d1..000000000 --- a/docs/verification/645-p3-continuation-protocol-20260917.md +++ /dev/null @@ -1,77 +0,0 @@ -# Retained approval continuation — September 17, 2026 - -The local implementation and isolated persistence checks are complete. Full -replacement-worker execution and the normal nested deployment remain open. -This record does not claim P3 deployment acceptance. - -An unanswered request now has no automatic deadline by default -(`approval_timeout_s: 0`). A positive, explicitly selected timeout remains -available. Pending requests have no DynamoDB deletion timer; closing their task -cancels unanswered requests and retains the decision history for 90 days. -The agent decides how to continue after an answer. There is no new system that -tries to determine whether the proposed action is still relevant. - -## Worker and task lifetimes - -At an approval boundary the worker saves the conversation, pending proposal, -workflow context, exact accumulated usage, and full Git/workspace archive. -Objects are versioned and checksummed. New tool execution stays blocked while -the checkpoint is ready. - -The coordinator verifies all saved versions before fencing the old worker. -Its authority is a separate, coordinator-owned `worker-lease#` record: -the worker can read and condition-check that record, but cannot change it. -After confirmed shutdown, the task becomes `PARKED` and returns its capacity -reservation while the approval stays available. - -An answer triggers admission of one replacement attempt when capacity permits. -Admission, its new lease and capacity reservation form one DynamoDB transaction. -A deterministic Durable execution name deduplicates invocations. The replacement -uses the original published coordinator version and exact source image version, -then restores the saved files/conversation and consumes the recorded answer. -Its budget is the original allowance minus the accumulated cost and turns. - -Repository-free MicroVM tasks use a private directory per task, with a local Git -baseline for the same archive format. Recovery preserves those scratch files -without creating a remote or installing a GitHub credential helper. A closed -worker cannot proceed to the workflow's artifact or PR delivery steps. - -The scheduled continuation manager retries unfinished retirement, missed -dispatches and terminal cleanup. Its persistent scan cursor advances across -invocations. It does not have permission to launch, suspend or resume workers; -launching remains with the pinned coordinator. - -A failed start request does not prove that AWS failed to create a worker. -Terminal saved tasks keep their capacity until the manager confirms shutdown, -or until the full service lifetime has elapsed for an unknown handle. The -coordinator records `CLOSED` in the exact-attempt lease before the atomic release. -`TERMINATING` alone is insufficient. - -## Verification - -- TypeScript compilation and ESLint passed. -- Focused coordinator/cleanup checks: **107 passed, 15 skipped**, 11 suites. -- Python recovery entry point and runner checks: **69 passed**. These include - actual Git/archive restoration through `restore_for_task`, registration races, - approve/deny/timeout handling, identity rejection and cumulative budget limits. -- Private real-SDK approve/deny recovery: **2 passed** using the pinned SDK/CLI. -- Full agent quality after repository-free recovery: **2,139 passed, 13 skipped**, - with 86.21% coverage. CLI compilation and **943 tests** passed. -- Private AWS coordinator suite: **12 passed**, including six competing - admissions, deliberately lost successful fence/park/admission replies, a - day-old unanswered request, repository-free admission, opt-in expiry, capacity - ownership and cleanup. - All three temporary tables and the temporary versioned bucket were deleted; - absence was verified. This suite launched no MicroVM worker. - -The standalone integration runner, README, source hashes and raw evidence are -outside Git at -`/Users/sphias/.local/share/abca-verification/645-p3-integration`. -The coordinator run is `runs/20260917T191209Z-6beab7bb`; the SDK run after the -retained-request default is `runs/20260917T184225Z-f4ac16e3`. - -The requested -[microvms-agentd reference](https://github.com/laithalsaadoon/microvms-agentd/tree/78304e361fbbe62e3a6b255b43c6f6c372b47510) -informed bounded transfers, disk-reserve checks, artifact assertions and cleanup -receipts. No reference executable was run and no source was copied. ABCA keeps -its task-scoped credential broker and complete Git/workspace preservation. diff --git a/docs/verification/645-p3-credentials.md b/docs/verification/645-p3-credentials.md deleted file mode 100644 index 81566cfc0..000000000 --- a/docs/verification/645-p3-credentials.md +++ /dev/null @@ -1,143 +0,0 @@ -# #645 P3: scoped credentials across sleep - -**Follow-up (2026-09-15):** [per-worker image capability](./645-p3-image-capability.md) declares the served hooks and verifies the actual launched image version. The [supervisor milestone](./645-p3-supervisor.md) implements durable recovery and approval wake; the full repository build passes. The record below preserves this milestone's original scope. Live sleep/wake acceptance remains open. - -**Live follow-up (2026-09-16):** the -[image `4.0` long-sleep test](./645-p3-callback-live-20260915.md#real-credential-expiry-and-the-original-approval-deadline) -kept the worker frozen beyond real STS expiry, renewed the same role and task/user -tags, and completed a further model response, database writes and S3 upload. -The approval retained its original deadline. Gateway signing and the wider -lifecycle matrix remain separate gates; this supersedes the untested long-expiry -item below without completing P3. - -Date: 2026-09-14. Local implementation and actual pinned Claude process probes. -No AWS deployment or suspension was performed for this milestone. Deployed -image **2.0** was unchanged at that milestone; P3 was not complete. - -## What changed - -Think of AWS credentials as a visitor badge with an expiry time. A sleeping -worker may wake up holding an expired badge. Every part of the worker needs a -working replacement before continuing, with the same task permissions. - -`microvm_credentials.py` now serves the task's scoped credentials through an -authenticated HTTP endpoint bound only to `127.0.0.1`, on a runtime-selected port. -The Claude child receives that endpoint as its AWS container credential provider. -Its alternate AWS key, profile, SSO/process, web-identity, metadata and bearer-token -sources are cleared; controlled empty configuration files suppress local profiles. -The parent environment is preserved. - -The image's managed `awsCredentialExport` command stays in place. In this internal -MicroVM mode, `bedrock_creds_helper.py` emits `{"Credentials": {}}` without reading -attribution files or resolving AWS credentials. That leaves resolution to the -scoped container provider and keeps keys out of Claude's stale export cache. -AgentCore/ECS/local attribution retains its existing helper behavior. - -With pinned Claude `2.1.191`, this deliberately empty export logs -`awsCredentialExport did not return valid AWS STS output structure`. The live -long-sleep case showed that message both before and after suspension, followed -by successful model requests through the scoped provider. It is expected for -this provider-selection design; returning cached keys merely to silence the -message would restore the stale-key path the implementation avoids. - -`aws_session.py` exports a coherent key/expiry pair under botocore's refresh lock. -The production resume callback forces recorded ambient providers to refresh first, -then force the original tenant credential object to renew with the original STS -tags. Existing clients keep their references. Changing the configured identity -after session construction is rejected. Mandatory renewal propagates failure -even if an older cached key remains valid. - -Normal deployments already launch a fresh AgentCore runtime session or ECS/MicroVM -worker per task. Manually reusing one Python process for a different identity after -credential construction now fails explicitly. MicroVM STS calls use short network -timeouts; the other backends retain their default timeout/retry behavior. - -The broker participates in the guest activity drain. Paused, failed or closed -controllers reject credential requests before consulting AWS. SDK success, startup -failure, query failure, receive failure and cancellation close the client and broker. -The earlier runner did not disconnect its SDK client. - -This is provider selection, not isolation from arbitrary code running as the same -OS user. The bearer token is created from OS entropy at runtime; it and the real -credential responses must never be logged or included in verification evidence. -The broker requires no ingress connector or additional AWS IAM permission. - -## Actual CLI evidence - -The opt-in verifier runs **claude-agent-sdk 0.2.110 / Claude 2.1.191** against a -loopback fake Bedrock server. It returns valid AWS event-stream frames containing -synthetic model text. All keys are synthetic. There are no paid model calls. - -Each case requires a successful initial model query before a 12-second credential -expiry, waits past that actual wall-clock expiry, and checks both the next request's -signing key and the SDK result. The verifier rejects unreviewed version pins. -`CLAUDE_CODE_MAX_RETRIES=0` bounds this experiment's error result; production model -retry settings are unchanged. - -| Case | First query after expiry | Result | -|---|---|---| -| Existing `awsCredentialExport` | Still signs with `SYNTHETIC_INITIAL` | Reproduces stale-key reuse | -| `credential_process` renewal | Signs with `SYNTHETIC_RENEWED` | Renewal itself works | -| Working ambient-only control | Signs with `SYNTHETIC_AMBIENT` | Calibrates fallback endpoint | -| Failed process renewal plus working ambient provider | Signs with `SYNTHETIC_AMBIENT` | Demonstrates unsafe fallback | -| Production scoped broker plus production helper | Signs with `SYNTHETIC_RENEWED` | Waits for replacement before signing | -| Failed production broker renewal | No second model request; SDK reports credential error | Fails closed | - -The fake server accepts synthetic signatures. The export case deliberately proves -which key Claude sends; its successful fake response does not mean AWS would accept -an expired key. Non-streaming startup model-availability checks are recorded -separately from the two main queries. - -Run from `agent/`: - -```bash -.venv/bin/python scripts/verify_microvm_credentials.py --mode export -.venv/bin/python scripts/verify_microvm_credentials.py --mode process -.venv/bin/python scripts/verify_microvm_credentials.py --mode ambient --ambient-fallback -.venv/bin/python scripts/verify_microvm_credentials.py --mode process-failure --ambient-fallback -.venv/bin/python scripts/verify_microvm_credentials.py --mode broker -.venv/bin/python scripts/verify_microvm_credentials.py --mode broker-failure -``` - -Each successful verification prints `"verified": true`; failed expectations exit -nonzero. All temporary files, CLI processes and loopback servers are cleaned up. -Reports for these six runs were retained locally as -`/tmp/abca-645-p2-clean-20260913/p3-credentials--20260914.json`. - -## Regression coverage and remaining work - -The agent quality gate passes **1,881 tests** (24 added), with **86.14%** total -branch coverage, plus Ruff lint/format and type checking. The configured Bandit -high-severity gate and Vulture dead-code gate pass. - -The full monorepo build also passes: **4,797 CDK tests** (38 existing skips across -two suites), **928 CLI tests**, **11 Forge tests**, CDK compile/lint/synth, -documentation build/link checks and contract drift checks. The final agent quality -run includes the two additional backend-timeout tests added during that build. -Logs are retained as `p3-credentials-build-20260914.log` and -`p3-credentials-agent-quality-20260914.log` in the same private evidence directory. - -Focused tests use real botocore refreshable credentials and retained S3/DynamoDB -clients. They verify ambient-before-tenant ordering, exact tag preservation, -signing with replaced keys, mandatory failure, identity mismatch, unknown/static -providers, expired replacements and UTC expiry formatting. Loopback tests verify -authentication, child-only environment changes, no secret error details, suspend -drain and continued denial after failed resume. Runner tests cover cleanup on -normal completion and failure/cancellation for MicroVM and other backends. - -Before automatic suspension can ship: - -1. **Implemented locally in the [HTTP hook milestone](./645-p3-lifecycle-hooks.md):** - bounded `/resume` invokes refresh before atomic task/gate reconciliation; - `/suspend` requires an acknowledged checkpoint. -2. Verify actual MicroVM runtime credential-provider type and renewal after sleep. - Static or unknown providers currently reject resume; rereading an unchanged - environment is not proof of renewal. Botocore's forced-refresh private API is - isolated in one adapter and requires review when the SDK changes. -3. Verify the built image's managed-settings path, Gateway signing, detached - subprocess behavior and long-expiry sleep in AWS. The local CLI probe uses an - explicit settings file containing the production helper command; it does not - install `/etc/claude-code/managed-settings.json` on the developer machine. -4. **Implemented locally in the later image/supervisor milestones:** image capability, - durable supervisor recovery and approval-triggered wake. Complete their live - verification and the [P3 plan](./645-p3-implementation-plan.md), including its P2 gates. diff --git a/docs/verification/645-p3-diagnostics-rollout-20260916.md b/docs/verification/645-p3-diagnostics-rollout-20260916.md deleted file mode 100644 index 94270ad6e..000000000 --- a/docs/verification/645-p3-diagnostics-rollout-20260916.md +++ /dev/null @@ -1,159 +0,0 @@ -# ADR-021 P3: diagnostics rollout and connection comparison - -Updated 2026-09-16 UTC. The normal development stack now runs the -[lifecycle diagnostics and wake feedback](./645-p3-lifecycle-diagnostics.md). -Automatic suspension remains disabled. The four earlier -[wake refusals](./645-p3-resume-refusal-investigation.md) remain open. - -## Normal deployment - -| Item | Verified value | -|---|---| -| Account / region | `` / `us-west-2` | -| Stack | `backgroundagent-dev`, `UPDATE_COMPLETE` | -| Source | `dff860c63a7e8c211d493d3630c38907497c157a` | -| Coordinator `live` version | `6`; version `5` remains available | -| Coordinator code SHA-256 | `4EJoQ2SGpuPv/vlT7Za3peYIrL3DBKznzK6IyA+9EiU=` | -| Managed worker image | `backgroundagent-dev-abca-agent:5.0`, `SUCCESSFUL` / `ACTIVE` | -| Worker artifact SHA-256 | `9b3150e9e5991cc9fcc8a4adbb2cfbd8f97399f0c16baa9c2b5fe5b101e7b035` | -| Bootstrap / guardrail version | `1.8.0` / `3` | -| Root resources | 475 | -| Static suspension setting / live SSM switch | Both `false` | - -The deployment was verified at **14:54:31 UTC**. Coordinator versions 5 and 6 -have identical environments. An older durable execution can continue to use its -original retained coordinator version. - -The new worker artifact has 111 files and is 486,428 bytes. Relative to image -4.0, it adds `microvm_diagnostics.py` and changes only the three lifecycle -modules. The Dockerfile, startup command, dependencies and hook settings are -unchanged. - -The reviewed CloudFormation change set had 34 changes: - -- 29 Lambda code updates for consumers of the shared diagnostics/classifier. -- The image artifact URI and its build role's exact artifact permission. -- A new coordinator version, alias update and retained previous version. - -No existing resource required replacement. Function environments, database, -network, guardrail and suspension settings were unchanged. Each of the 29 -deployed functions was `Active` / `Successful`, and its `CodeSha256` matched -the reviewed S3 ZIP bytes. - -The assembly preserves the exact previous S3 deployment template and substitutes -the reviewed code/image changes. This avoids the previously recorded unrelated -asset and guardrail-version churn during full synthesis. The full synthesis, -restricted assembly, AWS change set and per-function hash audit are retained in -private evidence. - -## Controlled connection comparison - -An HTTP connection is a communication line. Uvicorn normally keeps that line -open briefly so the next request can reuse it. The private comparison image adds -`--timeout-keep-alive 0`, which closes it after the response. This tests one -possible explanation for the earlier failures; it is not a production fix. - -The private image is `backgroundagent-dev-p3-no-keepalive-20260916:1.0`. -Its artifact SHA-256 is -`0fb02e8cf6935d62abd4622b22791a0213fbf1cb24c31f209fcef67dffd98922`. -All 111 filenames and every file except that one Dockerfile command match the -normal image artifact. The original server command remains PID 1 in both. - -Six fixed, owned tasks compare an immediate wake, a wake after at least eight -seconds observed suspended, and a missing-approval rejection on each image. -Each task requests one read of `/etc/os-release`, with a $1 / six-turn limit, -600-second original approval window and 1,800-second worker ceiling. -Two fresh replacement tasks were needed after the watcher failed to apply the -longer hold to the original pair; eight workers actually ran. The original -mistitled results and watcher are retained. - -The private durable coordinator uses the compiled production implementation. -Version 1 pins normal image 5.0; version 2 changes only its image identifier and -version to the comparison image. Both have code SHA-256 -`v3IUVmZa7U1gAj4l8ytBhfJlL19hbimNOYisgh7rB/Q=`. -Versions 3 and 4 expand the fixed task whitelist for the two replacements, -with code SHA-256 `E+I/xFbrwlnu1U7tOkfyK0spJRW7WtMl4SjrDhkgg+I=`; -they pin the comparison and normal images respectively. The production -supervisor implementation is identical between these private versions. -One-second verification polling and fixed task/configuration wrappers are -explicit test differences from normal task routing. - -The private approval handler commits the real production decision but omits its -immediate wake call identically for both images. The coordinator therefore issues -the sole Resume. Its role is restricted to the eight task/rate keys and its own -logs; it has no MicroVM API permission. Missing-approval cases conditionally -delete only their own still-pending approval after checking task ownership and -the original gate. They do not call the approval API. - -## Results and limits - -The strict lifecycle audit passed at **15:12:51 UTC**. The independent tool and -deployed API feedback audit also passed. - -| Required case | Measured hold after observing suspended | Suspend / resume client ports | Resume HTTP | AWS Resume request ID | -|---|---:|---|---|---| -| Normal, quick | 1.527 s | 54904 / 54904 | 200 | `0dfd1993-5fba-4758-8d58-16ee6aa52e20` | -| Close connections, quick | 0.554 s | 60656 / 60664 | 200 | `b3581707-7432-46af-80d5-8c1f3597d2b0` | -| Normal, longer pause | 10.205 s | 58984 / 56118 | 200 | `c7e545d9-0342-4805-97e1-3c7eb4c4475f` | -| Close connections, longer pause | 8.697 s | 58984 / 56118 | 200 | `32df71a5-f150-40e0-81a6-69c1466f8a54` | -| Normal, missing approval | 0.236 s | 37364 / 37364 | 409 | `d102840b-2d38-49a7-a448-01d1808dc505` | -| Close connections, missing approval | 0.227 s | 58984 / 56118 | 409 | `424ce2c0-0dbb-4a4b-b5ec-f7948041a75c` | - -These holds measure from the watcher's first suspended observation to the -decision/deletion receipt. They are not exact service freeze durations. -Canonical lifecycle access logs supplied the client ports. The normal quick -cases reused the same port; the longer normal case and all close-connection -cases used different ports. - -Every worker retained server PID 1 and had exactly one coordinator-issued -Suspend and Resume acknowledgment with an AWS request ID. Successful workflows -produced 32 guest diagnostic records. Each missing-approval case produced 30: -credential refresh succeeded, `resume-identity-read` failed with -`LifecycleUnavailable`, and the hook returned HTTP 409 with -`MICROVM_LIFECYCLE_UNAVAILABLE`. AWS independently reported the HTTP 409; -neither rejection was a connection refusal. - -Both failed tasks returned `MICROVM_RESUME_HOOK_FAILED` through the normal -deployed GetTask handler, with the specific wake-failure title, diagnostic steps -and `retryable: false`. This used direct Lambda invocation with the fixture's -identity, not API Gateway authentication. - -All eight tasks released their capacity reservation, returned their counter to -zero, terminated their worker and emptied their launch prefix before independent -cleanup. Six tasks completed, each with exactly one successful Read result in -task events and one Read call in a complete retained trace with zero dropped -records. The two rejected tasks have one gated attempt and no tool-result event. -They did not publish a final trajectory before service termination; their event -records and failed guest barrier are the available evidence. - -The original `default-long` and `closed-long` attempts actually waited only -0.361 and 0.474 seconds. The audit rejected their intended hold requirement. -The watcher was corrected before the missing-approval tests and two fresh -long-pause tasks were run. Both original attempts remain documented successful -quick wakes and **do not count as long-pause acceptance**. - -No unexpected connection refusal or reset was reproduced. These controls show -that both connection settings can work; they do not establish a failure rate, -identify the cause of the four original failures, or justify changing production -connection handling. Service-side restore/transport evidence or a fresh failure -with independent process/listener evidence remains necessary. - -## Cleanup and retained evidence - -Cleanup completed at **15:15:37 UTC** after confirming all eight owned workers -were terminated, reservations released, counters zero, payload prefixes empty -and private durable executions finished. It archived all three private log -groups before removing both private functions and every version, the comparison -image, three roles, private SSM switch, artifact and log groups. The eight -synthetic counters were removed only while their count and reservation version -still matched. Explicit absence checks passed. - -Task/approval history, task events and trace objects retain their normal -retention. Private evidence also contains the exact source/artifacts, original -failed audit, corrected watcher, all task histories, six complete traces, guest -logs, service receipts and cleanup proof. - -The normal image, artifact and log group were excluded from cleanup. A final -read confirmed `UPDATE_COMPLETE`, coordinator `live:6`, image `5.0` active and -the suspension switch still false. These results complete the diagnostic -rollout and bounded comparison, not the remaining P3 acceptance matrix. diff --git a/docs/verification/645-p3-durable-live-20260915.md b/docs/verification/645-p3-durable-live-20260915.md deleted file mode 100644 index 793779419..000000000 --- a/docs/verification/645-p3-durable-live-20260915.md +++ /dev/null @@ -1,295 +0,0 @@ -# ADR-021 P3: real AWS durable supervision - -Date: 2026-09-15. This continues the -[deployment and isolated guest checks](./645-p3-live-deployment-20260915.md). -Production automatic suspension remains disabled. The full -[acceptance matrix](./645-p3-implementation-plan.md#acceptance-matrix-and-completion-gates) -is not complete. - -## Scope and isolation - -Temporary Lambda `backgroundagent-dev-p3-durable-20260915` runs the compiled -production durable handler, supervisor, hydration, strategy, admission and -finalization code from `42c0bde9`, using development image `3.0` in `us-west-2`. -Its wrapper permits fixed synthetic task IDs and owners, selects a repository-free -MicroVM blueprint, uses five-second polling, and optionally crashes immediately -after a saved worker-start receipt. Each task allows six turns and a one-dollar -model budget. Its requested tool action is `Read("/etc/os-release")`; the long -credential-expiry case first runs a foreground `sleep 180` to age its credentials. - -The fixture has its own role and suspension parameter. DynamoDB permissions name -the fixture task/owner keys; payload writes name their object paths. It has no -public endpoint or event trigger. Optional global Memory is disabled. The real -deployed approve, deny and cancel handlers receive synthetic owned API events. -No repository, PR, issue or notification destination is configured. - -This proves real AWS durable execution and worker lifecycle behavior under those -fixture settings. It does not prove API Gateway authorization, normal repository -configuration routing, Memory, publication, or the entire effective-policy matrix. -The temporary role is not the production coordinator role. - -AWS sends a durable-execution envelope to Lambda. The production SDK unwraps it; -the fixture checks its task allowlist inside `loadTask`, before a database read. -It must not treat the outer envelope as the ordinary `{task_id}` input. - -| Fixture version | Purpose | ZIP SHA-256, base64 | -|---|---|---| -| 1 | Unused preparation version; no executions | `yQgkDHEjsqt+xS8whUn4g6dHqYxcF5xirSHPo78aybI=` | -| 2 | Preserve the production SDK envelope | `eh9wh2+/wnoURXkFGa46rlH9FGGKRC8UF9xetYnZGIE=` | -| 3 | Add command timing and AWS request IDs | `WBoCDV9SXZ8dP4/9FgVkD6bCyqJhAdmDtdq/komlzik=` | -| 4 | Add fresh approval retry ID; correct diagnostic worker-ID field | `fPxIeB6wRlyv2xaPDUdqWU6HBDw7s1iy2e+B4GO6cLc=` | -| 5 | Add credential-expiry fixture; 90-minute execution bound | `LYsZfWmwZuGUiu87kr+AnGd4sNbpS8bwqxlcwkquKZQ=` | -| 6 | Add fixed approval-fault and late-deadline identities | `NTxEsRcA2QK7TYa5tQGJcfkZdcMj6O7hXKIsZTsbfj4=` | - -Versions 3–4 change only private fixture diagnostics/allowlisting, not production -lifecycle behavior. Diagnostics omit credentials and launch bodies. - -## Acceptance evidence - -A case passes only after the coordinator itself has produced the expected -terminal task, terminated its worker, deleted its task payload prefix, and -released its saved capacity reservation with the owner's counter at zero. -Independent fixture cleanup runs afterward or after failure; it cannot make a -failed coordinator check pass. - -| Case | Task ID | Result | -|---|---|---| -| Ordinary, no approval | `01M2KENV5CKDZVM1HAT70AKRV9` | Passed on version 2 | -| Automatic sleep, approve | `01M2KENV5EMWVJ7FT312YWXJAD` | Failed during wake; diagnosis open | -| Automatic sleep, deny | `01M2KENV5EMWVJ7FT312YWXJAE` | Passed on version 3 | -| Cancel automatically suspended worker | `01M2KENV5EMWVJ7FT312YWXJAF` | Passed on version 3 | -| No decision; original deadline | `01M2KENV5EMWVJ7FT312YWXJAG` | Passed on version 4 | -| Crash after saved start, then approve | `01M2KENV5EMWVJ7FT312YWXJAH` | Start recovery passed; later wake failed | -| Cancel during start-crash recovery | `01M2KENV5EMWVJ7FT312YWXJAJ` | Passed on version 4 | -| Worker dies while suspended | `01M2KENV5EMWVJ7FT312YWXJAK` | Expected failure and cleanup passed on version 4 | -| Disable suspension during grace | `01M2KENV5EMWVJ7FT312YWXJAM` | Passed on version 4 | -| Fresh approval retry | `01M2KGTJFZDTT7WJVR70B2GMEN` | Passed on version 4 | -| Credentials expire while frozen | `01M2KHNBGNGV55MZY0M4F90C1A` | Credential renewal passed; approval callback was lost | -| Immediate state read fails after approval | `01M2KKXYPDQXFA1RB2M1H9D7MA` | Passed on version 6 | -| Immediate Resume submission fails after approval | `01M2KKXYPD2DQTZS4MMYQ6PCNH` | Approval survived; supervisor wake failed | -| State read and audit writes fail after approval | `01M2KKXYPDEWYWBE76XMZRG708` | Passed on version 6 | -| Supervisor unavailable beyond approval deadline | `01M2KM4RC576WFTA9YW6HT65H0` | Passed in isolated late-deadline function | - -### Ordinary durable execution - -Worker `microvm-2e68dad5-0553-3c05-94ff-94b2a9451481` ran one successful Read. -Execution `dd5da645-4cb6-35a6-a146-b075db6db4c6` ran from -21:27:36.828Z to 21:28:12.055Z. Its history contains seven distinct -`InvocationCompleted` request IDs and seven poll callbacks. Load, admission, -preflight, hydration, start and finalization each started once. - -All seven saved supervisor states retain `firstObservedAtMs=1789507660909` -and `sessionDeadlineMs=1789536460730`, with verified service lifetime. Replay -did not reset either clock. The slot was acquired at 21:27:40.081Z and released at -21:28:11.836Z. Worker termination, payload absence and counter zero were verified -before private cleanup. - -### Approval wake failure: unresolved - -Worker `microvm-c1d17307-e488-3149-b2c9-5d7f25f6278e` created its approval at -21:31:10Z with a 600-second timeout. The guest wrote its suspend checkpoint at -21:31:41.230Z and returned HTTP 200 from `/suspend`. AWS was observed -`SUSPENDED` at 21:31:44.124Z. Approval committed at 21:31:44.844Z and the deployed -API returned 202 at 21:31:46.234Z. - -AWS terminated the worker at 21:31:46.958Z with: - -> Resume lifecycle hook connection was refused. Please check your hook endpoint -> and application logs for more details. - -The guest stream contains no subsequent `/resume` access log or tool result. -The supervisor recorded `MICROVM_SUBSTRATE_TERMINATED`, left the task `FAILED`, -and released its slot. The durable execution itself reported `SUCCEEDED` -because the handler completed failure reconciliation; that does **not** make -the coding task or this acceptance case successful. - -The start-crash case below encountered the same refusal. Its telemetry proves -that overlapping API/supervisor Resume requests are not required for this failure. -The root cause remains unknown. No product workaround has been applied. A fresh -approval retry subsequently passed without a product change; that does not erase -the failed wakes. The inline Resume-failure case below reproduced the refusal -with only the supervisor submitting an actual Resume request. - -### Denial after automatic sleep - -Worker `microvm-bafa9cab-58e0-38da-b36f-b75ee699df59` received one automatic -Suspend request at 21:49:15.533Z. AWS was observed `SUSPENDED` at -21:49:18.625Z; denial committed at 21:49:19.417Z. The supervisor also sent Resume -at 21:49:20.758Z while the decision API was finishing its wake request. -The worker was observed `RUNNING` at 21:49:26.113Z. - -TaskEvents contain one Read attempt and one authoritative-denial result, with no -successful tool execution. Original creation time 21:48:45Z and the 600-second -timeout are unchanged. At 21:49:36.664Z the task was `COMPLETED`, the durable -execution `SUCCEEDED`, the VM `TERMINATED`, and the slot released. Payload absence -and counter zero were verified before private cleanup. - -### Cancellation after automatic sleep - -Worker `microvm-263ecb51-d404-3862-96e2-4e5d2b3698a1` was observed -`SUSPENDED` at 21:51:12.152Z. The repaired deployed cancel handler returned 200 -at 21:51:13.907Z. At 21:51:19.174Z the task remained `CANCELLED`, its execution -had succeeded, its worker was terminated, and its slot released. Counter zero -and payload absence were verified before private cleanup. - -The original approval remains `PENDING`, with unchanged creation time and timeout. -There is one gated tool attempt and no tool result. API and coordinator each -record a cancellation event; the saved capacity reservation changes to released -once, at 21:51:17.273Z. - -### Original deadline and credential renewal - -Worker `microvm-be8e4e0a-85b1-3cb2-96f6-f3a749d56113` created its 180-second -approval at 21:52:22Z. It was observed suspended at 21:52:53.939Z and running at -21:54:25.043Z, about 57 seconds before the original deadline. There was one -supervisor Resume request at 21:54:22.525Z and no decision API call. - -The guest recorded timeout at 21:55:22.126Z and one Read timeout result at -21:55:22.169Z. Completion and cleanup passed by 21:55:34.891Z. All 42 durable -invocations retained the same supervisor clocks and original approval deadline. - -CloudTrail records initial tagged `AssumeRole` at 21:52:09Z, expiring at -22:52:09Z, and renewal during resume at 21:54:23Z, expiring at 22:54:23Z. -Role, user tag and task tag were unchanged. Only timestamps, tags and request IDs -were retained. This proves renewal before expiry; the longer case tests expired keys. - -### Real process crash and cancellation race - -The first crash fixture saved worker -`microvm-9e6bdb3f-bc10-3921-9c50-287bbe20b574`, then called `process.exit(42)`. -AWS recorded the failed invocation at 21:56:05.731Z, started another at -21:56:06.763Z, and completed the saved start step at 21:56:08.983Z. The same worker -continued. Later its `/suspend` returned 200 at 21:56:48.531Z; approval returned -202 at 21:56:55.364Z. AWS terminated it at 21:56:56.279Z with the same hook -connection refusal. Telemetry shows one Suspend and no supervisor Resume before -termination. Overall acceptance failed; start recovery and failure cleanup worked. - -The cancellation race used worker -`microvm-6cba9cf9-ebf2-3d95-9c8a-c3d628454c58`. Its invocation exited at -21:58:32.834Z. The watcher saw `HYDRATING` and the saved worker at 21:58:33.199Z; -cancel returned 200 at 21:58:34.875Z. The replayed start step returned `null` at -21:58:36.182Z and `finalize-before-session` finished at 21:58:36.424Z. -Cancellation, termination, payload absence and counter zero passed. There were -two real invocations, one admission and no tool result. - -### Fresh approval, worker death and live rollback - -Fresh approval worker `microvm-94aa4204-435e-36f5-9ebf-ee6686f18c20` was observed -suspended at 22:00:37.831Z. Approval returned 202 at 22:00:38.750Z and one Read -succeeded at 22:00:39.437Z. Completion and cleanup passed by 22:00:49.713Z. -Eleven invocations retained the original 22:00:07Z/600-second approval clock. - -The worker-death fixture terminated -`microvm-0b21daf9-f196-3e6f-a318-1991663a10b2` at 22:02:21.578Z after automatic -suspension. The coordinator marked the task `FAILED`, released its reservation, -deleted the payload and completed its durable execution. Acceptance passed by -22:02:32.693Z; no tool result was produced. - -The rollback fixture disabled its own live switch at 22:04:16.566Z, during grace -for an approval created at 22:04:11Z. Worker -`microvm-909268eb-2ea0-321f-853d-782322215fdc` stayed running for over 60 seconds -afterward. Approval and cleanup passed by 22:05:32.733Z. The existing durable -execution honored the switch despite its immutable static flag still being true. - -### Approval survives immediate wake and audit faults - -Temporary `backgroundagent-dev-p3-approve-fault-20260915:1` wrapped the compiled -production approval handler with fixed task/owner checks and injected SDK -failures. It used the unchanged deployed approval role and configuration. -Its ZIP SHA-256 was `WG4JXZvqDmQlpwHss4W/tcIzGm6EY0esThypYcxRvwY=`. - -The state-read case injected `GetMicrovm` `TimeoutError` at 22:51:55.332Z. -The handler persisted a `microvm_resume_orphan` warning event and returned 202. -The durable supervisor woke worker -`microvm-8f4f231a-e997-3aee-9d8c-2b8e5193722e`; one Read succeeded, and completion -and cleanup passed by 22:52:06.537Z. - -The Resume-submission case injected `ResumeMicrovm` `TimeoutError` at -22:54:42.026Z. The approval stayed committed and the API returned 202. The -supervisor submitted the real wake request, but AWS terminated worker -`microvm-dca019ee-7d19-389e-b90b-bde129f77329` at 22:54:45.753Z with the same -resume-hook connection refusal. The task became `FAILED`, the durable execution -`SUCCEEDED`, and the reservation was released. Overall acceptance failed; the -runner then performed independent cleanup. This case cannot count as a -successful wake-recovery test. - -The audit-failure case injected both the state-read failure and all TaskEvents -writes from the decision handler. Logs confirm failures of both -`approval_decision_recorded` and the orphan warning write. The API still returned -202 at 23:02:05.851Z. Worker -`microvm-0f74d6a8-057e-36b5-9c8b-e3d8ab2fd079` performed one successful Read at -23:02:06.757Z. Thirteen durable invocations retained the original clocks; -completion and coordinator cleanup passed without independent repair. - -All fault-handler versions and its own log group were removed after log -collection; function absence was verified at 23:03:09Z. Its production role -was left unchanged. - -### Supervisor outage beyond the original deadline - -`backgroundagent-dev-p3-late-20260915:1` used the exact version-6 bundle and the -temporary fixture role, with a separate function concurrency limit. The main -supervisor and concurrent long-expiry test were not paused. - -Worker `microvm-f1594855-b72c-3952-8670-57700e870ba5` created its approval at -23:02:48Z with a 180-second deadline, 23:05:48Z. After observing it suspended, -the operator set only the isolated function's reserved concurrency to zero at -23:03:21.847Z. The worker remained suspended. The limit was removed at -23:06:07.845Z, after the original deadline. - -The supervisor submitted Resume at 23:06:27.476Z, request ID -`e070a904-9720-4e79-8937-7098293048f0`. The guest returned `/resume` HTTP 200 at -23:06:28.010Z and the Read timeout at 23:06:28.146Z. It did not execute the Read -or create a new approval window. Twenty-one recorded invocations and thirteen -poll callbacks retained the original supervisor clocks. Coordinator cleanup -passed at 23:06:40.234Z with no independent repair. - -### Expired credentials passed; long approval semantics failed - -The shared approval maximum is 3,600 seconds. This fixture first runs harmless -foreground `sleep 180`, then requests the gated Read. It changes no production -limits and never asks the agent to read credentials. - -CloudTrail records original task credentials at 22:07:17Z, expiring at -23:07:17Z. Approval was created at 22:10:32Z with a 3,600-second timeout; -suspension was observed at 22:11:08.410Z. An independent check at 23:08:55.158Z -still found the VM suspended after those credentials expired. - -The worker woke at 23:09:33Z. CloudTrail records a new `AssumeRole` at -23:09:33Z, 136 seconds after the original expiration, with the same role and -task/user tags. New expiration is 00:09:33Z on September 16; request ID is -`f9b97857-4c4f-4f7f-af94-21cc760aa147`. Subsequent model completion, task writes -and trace upload worked. This proves actual scoped credential renewal after -expiry. It does not cover Gateway signing or other backends. - -The approval behavior failed independently. The guest logged a CLI-generated -generic denial at 23:09:33.528Z, before `/resume` returned 200 at 23:09:33.908Z. -The model reported that the read was not permitted and finished. The original -approval remained `PENDING`, with no Read-result TaskEvent, although its deadline -was still 23:10:32Z. The runner's terminal/cleanup checks alone passed; the -independent semantic audit correctly rejected overall acceptance. - -All 733 durable invocations retained the same supervisor clocks. Coordinator -cleanup finished by 23:09:45.465Z, with counter zero, worker terminated and no task -payload. The [callback timeout follow-up](./645-p3-callback-timeout.md) investigates -the lost wait and records the local fix and remaining deployment check. - -## Evidence and outstanding work - -Private evidence is under -`/tmp/abca-645-p2-clean-20260913/p3-durable-fixture-20260915`. -Each case has observations, durable history and TaskEvents. Full execution-data -history is retained privately when needed to inspect persisted clocks. -`deployment-ledger.json` records the exact temporary resources and versions. - -Eleven complete durable cases passed. Three cases failed during approval wake, -and the long-sleep case passed credential renewal but failed approval semantics. -The intermittent refusal and the callback-timeout fix's AWS validation remain -open, along with the remaining lifecycle faults, networking and migration. -See the [effective IAM checks](./645-effective-iam-20260915.md). - -All temporary durable/fault/late-deadline Lambda functions and versions, owned -log groups, the durable fixture role and its suspension parameter were removed. -Main cleanup completed at 23:16:39Z after verifying all 14 main executions were -terminal and all 23 workers listed for the development image were terminated. -Completed fixture task records and traces remain audit data. diff --git a/docs/verification/645-p3-final-image-and-ecs-20260917.md b/docs/verification/645-p3-final-image-and-ecs-20260917.md deleted file mode 100644 index cd5675c5b..000000000 --- a/docs/verification/645-p3-final-image-and-ecs-20260917.md +++ /dev/null @@ -1,249 +0,0 @@ -# ADR-021 P3: final-image persistence, credential expiry and ECS compatibility - -Verification ran September 16–17, 2026, in `us-west-2`, account -``. The normal deployment remains MicroVM image **6.0**, -coordinator **10**, 8,192 MiB, with both automatic-suspension gates **off**. -The [connection-close rollout](./645-p3-connection-close-rollout-20260916.md) -records the preceding four core lifecycle cases. - -## Repository state across two sleeps - -Task `01M2P84R90WESJH1GPTYKZ20PM` used normal image 6.0, its original server as -PID 1, and the production Durable handler in a private coordinator. It cloned -public repository `isadeks/vercel-abca-linear`, main commit -`f5be1e23a964b661f1bf3d98ead55679d65978c4`, into a temporary directory. - -The actual tool sequence was: - -1. Clone, verify the commit, run `npm ci`, lint and tests, then write a marker. -2. Request approval to read the marker, sleep, approve through the normal API, - and read it after waking. -3. Verify the first marker's hash and write a second marker. -4. Request another approval, sleep again, approve and read the second marker. -5. Verify both hashes and the unchanged repository commit, check the tracked - files are unchanged, and run lint and tests again. - -Both lint runs and both Vitest runs passed; the repository has one test. -The marker hashes were: - -| Marker | SHA-256 prefix (16 hex characters) | -|---|---| -| First | `8437311a6e403b72` | -| Second | `cb1e409f22270af5` | - -Worker `microvm-3a114559-4c86-34a0-af36-0a04e6232a5f` recorded two suspend and -two resume HTTP 200 results from PID 1. Independent normal approval-handler -logs supplied both actual `ResumeMicrovm` request IDs. Each gate kept its -original creation time and 600-second timeout. Finalization passed at -**23:25:41.137 UTC**: task `COMPLETED`, Durable execution `SUCCEEDED`, worker -`TERMINATED`, reservation released, counter zero and payload absent. -The watcher performed no repair. - -This verifies repository files and running application state across multiple -sleeps. The task used artifact delivery and explicitly cloned into a temporary -directory. It does **not** exercise the platform's normal repository-bound -clone/PR-delivery workflow. No commit, push, PR or external comment was created. - -## Late approval wins its decision race - -Task `01M2P84R94PY36ZV5D43RVN7MT`, worker -`microvm-f191abfa-de75-3a89-bb5f-d1b4b9865ce1`, used image 6.0 and a 150-second -approval window. The original deadline was **23:28:44 UTC**. - -The private coordinator deliberately withheld its scheduled resume until five -seconds after that deadline. Those omissions were recorded as injected missing -commands, never as successful AWS requests. The normal approval request began -at **23:28:46.349** and returned HTTP 202 at **23:28:47.593**. Its actual resume -request ID was `c3eed6c5-5029-494a-a45b-162de6dfce63`. - -The original PID 1 finished the resume hook with HTTP 200, and the permitted -Read completed. The approval remained `APPROVED` with its original clock. -Task completion, successful Durable execution, worker termination, reservation -release and payload cleanup passed at **23:29:01.093**, without watcher repair. - -This is the existing decision contract: the first committed decision wins. -The agent writes `TIMED_OUT`; the approval API does not independently reject a -request solely because the clock deadline passed. The preceding image 6.0 -acceptance separately verified timeout winning and a later approval returning -HTTP 404. - -The independent two-case audit passed at **23:29:55.019**. Its private -coordinator, role, switch, log group and two zero counters were removed, with -absence verified at **23:31:04.765**. All 237 function log events were retained. - -## Real credential expiry - -The separate task `01M2P7E6SF68775QCBYXS1Z0YP` uses normal image 6.0 and worker -`microvm-ddc38544-f478-3578-a78d-79e1ac7a32fa`. Its initial credentials expire at -**00:07:25 UTC on September 17**, according to the actual STS issuance recorded -in CloudTrail. Its original approval deadline is **00:11:10**. - -An independent `GetMicrovm` observed the worker still `SUSPENDED` at -**00:07:40.565**, after that actual expiry. Request ID: -`1761febc-81d7-43d5-97b2-841c8f553f94`. - -Both the lifecycle audit and the independent credential audit passed. - -| Event | UTC time | -|---|---| -| Initial STS session issued | September 16, 23:07:25 | -| Worker observed suspended | September 16, 23:11:41.565 | -| Initial credentials expired | September 17, 00:07:25 | -| Independent observation still `SUSPENDED` | 00:07:40.565 | -| Actual Resume accepted | 00:10:11.100 | -| New STS session issued | 00:10:12 | -| Original PID 1 finished resume with HTTP 200 | 00:10:12.242 | -| Original approval deadline reached | 00:11:10 | -| Late approval returned HTTP 404 | 00:11:12.097 | -| Finalization verified without watcher repair | 00:11:22.157 | - -The new STS session expires at **01:10:12**. CloudTrail request -`e1320fee-c291-44da-8f32-dff210efcd38` records the same session role, session name, -task tag and user tag as the initial request -`85c3d6af-9bf7-4bfc-8555-ab523f28eb64`. CloudTrail timestamps have second -precision. The original PID 1 completed its refresh/reconciliation hook in -132 ms; the actual Resume request ID was -`1a817784-6702-43ed-828e-532a18ec06c4`. - -The gate became `TIMED_OUT`, and the unapproved Read did not succeed. The task -then completed its refusal response, the Durable execution succeeded, and the -worker terminated. Its reservation was released, counter was zero and payload -was absent. The retained trace had no dropped events. This distinguishes -completing the task's refusal response from permitting the timed-out tool. - -The renewal record became visible in CloudTrail about three minutes after -issuance. The independent audit required every pre-observation session to -have expired, a successful subsequent issuance with unchanged identity tags, -and the successful resume/lifecycle evidence. No credential values were saved. - -## ECS approval configuration fix - -ECS runs the same agent in a container. Its task definition is the recipe that -tells the container its image, memory, permissions and configuration. - -The production ECS recipe omitted `TASK_APPROVALS_TABLE_NAME`. The agent's -approval code therefore raised `ApprovalTablesUnavailable`, even though the -shared session role already had approval-table permissions. - -Local commit `c14d2ba0` fixes this in the source: - -- Pass the real approval table from `AgentStack` into `EcsAgentCluster`. -- Give both build and planning containers the same approval-table name. -- Prevent deployment-time build settings from erasing that name. -- Preserve session-role ownership of production data permissions. The - optional path without a session role gets only approval Get/Put/Update. -- Correct nearby comments about task sizing, shared environment and signed - payload delivery. - -Compilation, ESLint and 186 targeted construct/stack tests passed. The full -CDK suite passed **5,057 tests in 229 suites**, with one snapshot; the existing -56 optional DynamoDB tests were skipped. No Python runtime change was needed. - -## ECS live approval and cancellation - -A separate 23-resource stack used the production `EcsAgentCluster`, -`AgentSessionRole` and `EcsPayloadBucket` constructs in the existing private -subnets. It created its own roles, buckets, cluster, security group, task -definitions, switch and logs. The normal deployment and its roles were not -modified. - -The unchanged Dockerfile image was built for ARM64 and published to an -immutable tag in a private verification repository. Its OCI index digest was -`sha256:e892253aaa2565d2064da2fbb13c09f85061cbb5095355b8993fa249fa10c2c0`. -Both workers reported that digest. Production build sizing remained -4 vCPU / 16 GiB / 50 GiB; planning remained 2 vCPU / 8 GiB. - -The production Durable handler ran in private coordinator version 2. Approval -used the normal API. Cancellation used the production cancellation handler in -a private function with its own restricted role, because the normal deployment -has no ECS substrate or ECS cancellation grant. A `timeout` wrapper bounded -each test container to 900 seconds; it ran the normal batch bootstrap command -inside that limit. Coordinator polling was shortened to five seconds. - -| Case | Task | Result | -|---|---|---| -| Approve while waiting | `01M2PA7M84P99ZDV97920NRSEX` | `COMPLETED`; one successful Read | -| Cancel while waiting | `01M2PA7M891EW8VW3FMW9APFXV` | `CANCELLED`; no successful Read | - -Each task requested a five-second MicroVM sleep delay but remained ECS -`RUNNING` throughout an approval wait longer than 20 seconds. Neither acquired -MicroVM lifecycle/start metadata. Both retained their original approval clocks, -finished their actual Durable executions successfully, reached ECS `STOPPED`, -released their reservations and deleted their payloads without watcher repair. -The independent audit passed at **00:01:21.042 UTC**. - -The first deployment attempt used an incorrect managed-policy name in the -scratch cancellation role and rolled back before any task ran. A subsequent -watcher incorrectly treated the JSON-text approval preview as an object; that -task was stopped and repaired, retained as excluded evidence, and replaced with -fresh IDs. The final audit also corrected its event-name lookup to the actual -`agent_tool_result` schema. These verification errors are not counted as product -failures or successful unattended acceptance. - -## Actual ECS permissions and network ports - -A separate planning container used the real ECS task role and the production -scoped-session provider. It passed **19 checks**, exited zero, and reached -`STOPPED`: - -| Check | Result | -|---|---| -| Ambient task/counter reads | Denied | -| Ambient own bootstrap marker read | Allowed | -| Ambient payload read and payload-bucket listing | Denied | -| Scoped own task read | Allowed | -| Scoped other-task and counter reads | Denied | -| Scoped reporting update shape | Authorized; deliberately false condition prevented mutation | -| Scoped owner, compute handle, start receipt, reservation and lifecycle updates | Denied | -| Scoped bootstrap read | Denied | -| Scoped own artifact write / other-task artifact write | Allowed / denied | -| TCP 443 / TCP 80 to the same address | Connected in 8 ms / timed out after 5 seconds | - -Both task records were unchanged afterward. The operator could reach both ports -on that same external address before and after the guest probe. The deployed -security group has no ingress and permits only outbound TCP 443; public IP -assignment was disabled. The network probe verifies this egress restriction, -not remote-MCP behavior or a live public-ingress negative test. - -The session role limits an existing tagged session to its task. The compute role -chooses the tags when assuming it; these checks do not establish protection -against a compromised worker minting a different session identity. - -## Cleanup, evidence and remaining scope - -The long-expiry coordinator, role, private switch, log group and zero counter -were removed with absence verified at **00:15:38.640 UTC**. All 1,524 private -function log events were retained. Together with the earlier two-case cleanup, -all three verification deployments in this record have been removed. - -ECS teardown was verified at **00:07:31.420 UTC**. The private stack, both -functions and all versions, five roles, three buckets, registry and private -logs were removed. All four ECS workers were stopped, and the three owned -zero counters were deleted. The automatically created Container Insights -performance log was archived and removed separately from CloudFormation. - -The image is an OCI index referencing an ARM64 manifest and a build-attestation -manifest. Cleanup first stopped on an incorrect one-record assumption, then -verified and removed the exact three-digest set. The exact image is retained -locally as a Docker archive. Both inactive task definitions were submitted for -deletion and reported `DELETE_IN_PROGRESS`; these are service records, not -running containers. - -Private evidence directories: - -- `/tmp/abca-645-p2-clean-20260913/p3-image6-matrix-20260916` -- `/tmp/abca-645-p2-clean-20260913/p3-image6-expiry-20260916` -- `/tmp/abca-645-p2-clean-20260913/p3-ecs-compatibility-20260916` - -Permanent private archive: -`/Users/sphias/.local/share/abca-verification/645-p3-20260916/final-image-and-ecs-evidence.tar.gz`. -It contains 188 files, including the exact Docker image, and is 701,261,332 bytes -with mode `0600`. SHA-256: -`576a9d5171fc6b97562310d5f12809e601a88d73e30ac6a3c500d8d9d9bb093f`. -Every archived file hash was checked against the retained manifest. - -The normal automatic-suspension switches stay off. Service-side wake traces, -Run-token retention and recovery without guest identity logs, the remaining -backend/network/MCP checks, normal repository-bound delivery, and the applicable -deployment/drain/rollback procedure remain separately tracked in the -[implementation plan](./645-p3-implementation-plan.md). diff --git a/docs/verification/645-p3-guest-barrier.md b/docs/verification/645-p3-guest-barrier.md deleted file mode 100644 index 30ad56699..000000000 --- a/docs/verification/645-p3-guest-barrier.md +++ /dev/null @@ -1,123 +0,0 @@ -# #645 P3 guest pause controller - -**Follow-up (2026-09-15):** [per-worker image capability](./645-p3-image-capability.md) declares the served hooks and verifies the actual launched image version. The [supervisor milestone](./645-p3-supervisor.md) implements durable recovery and approval wake; the full repository build passes. The record below preserves this milestone's original scope. Live sleep/wake acceptance remains open. - -Date: 2026-09-14. Local implementation; at this milestone, deployed image `2.0` -had only ready, validate, run and terminate hooks. Automatic sleeping was disabled. - -## What is implemented - -`agent/src/microvm_lifecycle.py` owns the local pause boundary. It does not call -AWS, authorize a human decision or extend an approval's deadline. - -- The MicroVM pipeline registers task and VM identity, then removes the context - on completion or failure. Hook closures retain that closed controller and - recheck it before allowing a tool, so registry removal cannot release late - callbacks. AgentCore/ECS do not register a MicroVM context. -- SDK pre/post/failure hooks track parallel tool calls. Suspension requires - exactly the tool currently parked for approval. Missing/duplicate identities - and known background work conservatively disable suspension. -- The approval hook registers the same `_ApprovalDeadline` created before its - database transaction. Before leaving the approval wait, it must pass the - barrier and remove the safe point. Approval/cancellation still use the existing - conditional task-state transition. -- Approval reads and heartbeat writes already in progress drain before the - checkpoint callback. New approval reads wait; heartbeat ticks skip while - paused. -- All progress writer instances report their actual write acknowledgments to - the same context. Missing tables, open circuit breakers and uncertain/failed - writes prevent suspension. A later event cannot repair a previously dropped - event, so this latch is separate from the transient failure counter. -- Suspend/resume callbacks run within a caller-supplied budget. Locks protect - local state only. Generation checks prevent a callback from committing a - transition after teardown or a superseding operation. -- A failed pre-suspend checkpoint leaves the unfrozen waiter able to proceed - and disables further suspension. An acknowledged suspend keeps the gate - closed until successful resume. Failed/timed-out/cancelled resume closes the - barrier; a late callback cannot release coding. -- `/run` and successful controller resume reseed Python's application PRNG with - `os.urandom(32)`. This does not make `random` suitable for secrets. - -Concurrent lifecycle requests receive a controlled rejection without holding a -lock across network work. The later [HTTP hook milestone](./645-p3-lifecycle-hooks.md) -adds cached successful acknowledgments and clears an old wake result for each new gate. - -## Why the image is not eligible for sleep yet - -The later [HTTP hook milestone](./645-p3-lifecycle-hooks.md) supplies production -callbacks for atomic checkpoint writes, credential renewal and gate reconciliation. -At this milestone, image capability, supervisor integration and live service -verification remained open; the follow-up above records later implementation. - -The worker has several independent credential consumers: - -| Consumer | Current owner | Required wake work | -|---|---|---| -| Task/approval/events/nudges and tenant S3 | `aws_session` tenant session; existing clients retain its credential object | Refresh without losing tags or leaving existing clients attached to stale credentials | -| Memory, Logs, trajectory and platform secrets | Ambient `platform_client` factory, including cached clients | Refresh the actual provider used by those clients | -| Claude Bedrock requests | Separate Claude subprocess and `awsCredentialExport` cache | Prove a blocking refresh path before its first request after sleep | -| Gateway request signing | Fresh botocore session per signing operation | Verify runtime-provider renewal and signing after wake | - -Inspection of the installed SDK `0.2.110`'s bundled Claude `2.1.191` found that -its helper cache uses wall-clock expiry but returns the previous value while -refreshing in the background. It also falls back to a one-hour cache lifetime -when the helper expiration is missing or six minutes or less away. Therefore: - -- Refreshing Python does not clear Claude's cache. -- Returning a very short helper expiration does not force a safe refresh. -- Current documentation for newer Claude default-chain caching is not evidence - that the pinned helper path has those semantics. - -**Follow-up implemented locally:** the [credential verification](./645-p3-credentials.md) -now exercises that actual CLI. `credential_process` renewed successfully but -fell through to ambient credentials on failure. The chosen MicroVM path uses a -single authenticated, scoped container provider; renewal waits before signing, -and failure sends no model request. Python refresh updates retained credential -objects with the original identity. The HTTP resume callback now invokes this -operation; live runtime provider/snapshot verification remains open. -No minified CLI internals were patched. - -Known background tool flags are tracked, but arbitrary shell/MCP subprocesses -may also detach work. Their safe-point behavior still requires a conservative -policy or process-level proof. Tool completion tracking alone must not be -advertised as proof that every process in the VM is idle. - -## Verification - -The agent quality gate passes: Ruff lint/format, type checking and **1,857 Python -tests** (34 added for this milestone; total branch coverage **85.02%**). The -configured Bandit high-severity gate and Vulture dead-code gate pass. A pre-existing -Vulture false positive for urllib's required `newurl` callback argument was -documented in the existing narrow allowlist; redirect behavior is unchanged. -New regressions exercise: - -- Parallel tools, original gate/deadline identity and an approval arriving during - the pause boundary. -- Cancellation winning the existing conditional resume transition. -- In-flight progress drain, failed/missing acknowledgments shared across - writers, and heartbeat/read suppression while paused. -- Credential callback failure, cancellation, timeout and late completion; - teardown during refresh; competing lifecycle requests. -- Frozen monotonic time with elapsed UTC approval deadline, fresh OS entropy and - pipeline registration/cleanup. - -These are local synchronization and existing-hook integration tests. -They do not prove actual AWS freeze/resume, credential renewal, hook HTTP -budgets, durable supervisor recovery or complete P3 acceptance. - -## Remaining integration order - -1. **Local credential implementation complete:** retained-client refresh is now - connected to resume; verify the deployed runtime provider and snapshot behavior. -2. **Local HTTP integration complete:** acknowledged production checkpoints, - shared budgets, duplicate handling and teardown are covered in the - [hook verification](./645-p3-lifecycle-hooks.md). -3. Bind lifecycle capability to the image/version that launched each worker; - declare compatible hooks, initially leaving automatic suspension disabled. -4. Wire persistent intent/policy into durable supervisor polling with bounded - failure/wake recovery. Wake after committed approve/deny decisions. -5. Deploy through the normal CDK/bootstrap path and complete the P3 live matrix, - including delayed suspend, expiry, cancellation, refresh failure and rollback. - -The [implementation plan](./645-p3-implementation-plan.md) remains the complete -task list, including unfinished P2 live gates. diff --git a/docs/verification/645-p3-image-capability.md b/docs/verification/645-p3-image-capability.md deleted file mode 100644 index d1216e4be..000000000 --- a/docs/verification/645-p3-image-capability.md +++ /dev/null @@ -1,96 +0,0 @@ -# ADR-021 P3: per-worker image capability - -Date: 2026-09-15. Local implementation on `fix/645-microvm-readiness`, following -the [guest hook milestone](./645-p3-lifecycle-hooks.md). At this milestone, the -last verified AWS image was `2.0` and automatic suspension was disabled. - -**Deployment follow-up:** the [P3 live record](./645-p3-live-deployment-20260915.md) -verifies active image `3.0`, protocol `1`, and initial isolated guest checks. -Automatic suspension remains disabled; full live acceptance is still open. - -## What this adds - -An image is the saved starting computer. Its version identifies which saved copy a -worker actually started from. Updating the image later does not update an existing -worker. The supervisor therefore needs the worker's own version, rather than the -deployment's newest image, before deciding whether it can safely sleep. - -Managed images and the out-of-band packaging helper now declare all six served -hooks. Suspend/resume use the shared 30-second service timeout, leaving headroom -above the guest's 20-second handler budget. The image also stores the non-secret -`ABCA_MICROVM_LIFECYCLE_PROTOCOL=1` marker. `/validate` rejects a supplied -incompatible marker without contacting AWS. A missing marker remains acceptable -for ordinary legacy/AgentCore startup, but cannot establish sleep support. - -After `RunMicrovm`, the coordinator: - -1. Saves the known worker ID/endpoint and actual returned `imageArn`/`imageVersion` - before any optional image query. -2. Calls `GetMicrovmImageVersion` for that exact ARN/version, with a three-second - request budget. A foreign or missing image identity cannot borrow deployment - configuration. -3. Checks the returned identity, shared marker and port, all six enabled hooks, - and suspend/resume budgets. -4. Conditionally adds `lifecycleProtocol` to both the saved start handle and - `compute_metadata`, only while both records still identify the same launch. - This preserves task status, including cancellation. - -Failed lookup, unsupported hooks/marker or a rejected capability write keeps the -saved worker available for ordinary tasks and cleanup, with new suspension -disabled. Error logs contain the error type, not returned image environment data -or exception text. Saved handles are replayed without reinterpreting them through -the current deployment configuration. - -Both the policy and the lifecycle intent store require persisted capability for -new suspend requests. The transaction also compares image identity and protocol -with the original read, so a stale decision cannot admit sleep. Resume remains -available for legacy/unknown images that need recovery. - -## AWS contract and permission scope - -The pinned SDK exposes actual `imageArn`/`imageVersion` in `RunMicrovmResponse`. -`GetMicrovmImageVersionOutput` exposes that version's hooks and environment. -`UpdateMicrovmImageVersionRequest` accepts identity plus `status`, not replacement -hooks, environment or code. Changing a version to `INACTIVE` stops new launches; -it does not erase an existing worker's support. - -The [official AWS Lambda service reference](https://servicereference.us-east-1.amazonaws.com/v1/lambda/lambda.json) -classifies `GetMicrovmImageVersion` as a read action on `microvmImage`. -The coordinator adds this action to its existing configured-image ARN scope. -It still has no Suspend/Resume grant or automatic suspension caller at this -milestone. Bootstrap policy files did not change. - -## Verification - -- Capability tests cover exact versus requested/current identity, missing and - conflicting markers/hooks/budgets, failed lookups/writes and saved-handle replay. -- DynamoDB Local passes 39 lifecycle tests, including 16 new cases for conditional - capability enrichment, concurrent cancellation, each changed identity field, - lost committed replies, stale suspend admission and legacy wake recovery. -- Construct tests compare the helper's actual hook/environment JSON with the - synthesized image and reject overrides of the reserved protocol marker. -- Contract mutation tests reject unsafe values and literal redeclarations. -- Guest validation tests cover absent, supported, empty and incompatible markers, - with AWS calls forbidden during image validation. - -Local evidence is under `/tmp/abca-645-p2-clean-20260913/p3-image-*20260915.log`. -The focused image suite passed **444 tests**, the orchestrator composition suite -passed **31**, and agent quality passed **1,940** with 11 opt-in database cases -skipped and **86.28%** branch-inclusive coverage. - -The full root build completed compile, lint, synth, docs, contract checks, -928 CLI tests, 11 Forge tests and agent quality. Its CDK run passed **4,847** -tests and failed three old assertions in two suites: the prior action list and -two expectations that omitted image metadata. Those assertions were updated; -the affected stack grant and all six start-session composition tests then passed. -The full build command itself exited nonzero; the corrections were verified by -focused reruns. There was no production-code failure in that run. - -## Remaining completion gates - -Durable supervisor recovery and approval-triggered wake are now [connected locally](./645-p3-supervisor.md). Next deploy and -test the matching coordinator/image together. Verify effective image-read and -lifecycle permissions, actual service hooks, expired credentials, approval and -cancellation races, failures and cleanup in AWS. The -[implementation plan](./645-p3-implementation-plan.md) retains these gates and the -remaining P2 checks. Declared hooks and local tests alone do not complete P3. diff --git a/docs/verification/645-p3-implementation-plan.md b/docs/verification/645-p3-implementation-plan.md deleted file mode 100644 index 5ef06713a..000000000 --- a/docs/verification/645-p3-implementation-plan.md +++ /dev/null @@ -1,209 +0,0 @@ -# ADR-021 P3 implementation and completion plan - -Updated September 18, 2026 UTC. The [original review](./645-p3-readiness-review.md) -explains P1 and P2 and introduces the service. This is the current checklist; -dated verification records describe what was true at each earlier milestone. - -**Current status: P3 complete.** Implementation, cloud continuation, normal -activation, live off-switch/rollback checks, retirement of the old flat -infrastructure, final smoke and cleanup are verified. See the -[normal acceptance record](./645-p3-normal-closure-20260918.md). - -## The behavior being delivered - -When the agent needs permission, it saves a question for the person. After ten -minutes, its computer can sleep while the question stays available. A person can -approve or deny later. If the wait is long, ABCA saves the conversation and files, -confirms that the old computer has stopped, and releases its capacity. The answer -can then start a replacement computer with the saved work. - -- Unanswered approvals have no deadline by default: `approval_timeout_s=0`. -- An explicit decision timeout remains available, from 30 to 3,600 seconds. - A positive policy-rule deadline can also apply. -- The default sleep delay is 600 seconds. `--microvm-sleep-after off` keeps that - task awake; a deployment switch can disable new suspensions globally. -- An expired explicit deadline denies the proposed tool call. Waking or replacing - a worker does not restart the decision clock. -- Cancellation closes unanswered requests atomically. Already-recorded decisions - remain part of task history. -- The platform checks authenticated ownership, task/request identity, exact - approved tool inputs and cancellation. The agent decides whether the action - still makes sense. A separate relevance/staleness rechecking system is excluded - by the user's scope decision. - -“Durable” means saved outside the worker. A checkpoint is the saved conversation, -files and task context. A capacity reservation is the worker's place in the -deployment's concurrency limit. - -## Implementation progress - -- [x] Finish P2 follow-ups for payload provenance, task-scoped permissions, - bounded uncertain-start recovery, atomic capacity ownership, worker cleanup, - error classification, thread isolation and logging. See the - [clean deployment](./645-p2-clean-deployment-20260913.md), - [payload checks](./645-p2-payload-live-20260914.md), - [registration recovery](./645-p3-registration-20260916.md) and - [effective permissions](./645-effective-iam-20260915.md). -- [x] Implement all six image/runtime hooks, guest activity barriers, original - deadlines, credential renewal, exact image capability and Durable supervision. - [Lifecycle hooks](./645-p3-lifecycle-hooks.md) and the - [supervisor](./645-p3-supervisor.md) document the contract. -- [x] Reproduce and correct stale pooled hook connections. Lifecycle responses - explicitly close their HTTP connections before freeze. The - [transport comparison](./645-p3-wake-transport-20260916.md) includes the - greater-than-90-second control and closure-before-freeze evidence. -- [x] Verify approval, denial, cancellation, deadline races, multiple sleeps, - repository work, networking and actual AWS credential renewal after expiry. - See [final-image/ECS checks](./645-p3-final-image-and-ecs-20260917.md), - [repository acceptance](./645-p3-repository-path-20260917.md), - [MCP/network checks](./645-p3-mcp-network-20260917.md) and - [AgentCore permissions](./645-p3-agentcore-permissions-20260917.md). -- [x] Implement retained approvals, versioned conversation/workspace storage, - budget accounting, attempt ownership, confirmed retirement, replacement - admission and scheduled cleanup for repository and repository-free tasks. - See the [continuation protocol](./645-p3-continuation-protocol-20260917.md). -- [x] Pass the real cloud continuation matrix: approve, deny and expiry after - retirement; cancellation without replacement; approval and denial on the - original sleeping worker. Verify real output/checksums and resource cleanup. - See [cloud acceptance](./645-p3-cloud-continuation-20260917.md). -- [x] Verify normal signed-in CLI pending/decision feedback, a request answered - after more than two hours, one replacement, exact Read execution and final - capacity release. The [progress-race record](./645-p3-progress-race-20260918.md) - distinguishes recovered cloud evidence from the interrupted local watcher. -- [x] Fix the normal-workflow progress/checkpoint race. Progress can finish - during upload; capture drains acknowledgments before publication. General - activity and actual suspension retain their barriers. Three regressions - cover successful writes, in-flight writes and failed writes. -- [x] Implement the requested nested MicroVM stack, install bootstrap 1.9.0, - verify fresh deployment and cleanup, and rehearse overlapping old/new - resources with a live permission probe and consumer rollback. -- [x] Add the normal nested stack, build its image, preserve the shared - execution-role identity, and switch compatible consumers while retaining old - resources for drain. The [nested record](./645-p3-nested-stack.md) also retains - the failed native-refactor attempt and verified automatic rollback. -- [x] Complete corrected-image normal default/expiry/off-switch acceptance, - rollback to its compatible sleep-off coordinator and restore activation. -- [x] Verify the old worker/execution drain, remove the 16 obsolete flat - resources, and check final resource identities, outputs and permissions. -- [x] Remove owned acceptance data/identity and temporary infrastructure; finish - documentation sync/check/build and the final evidence manifest. - -The full agent suite after the Linear vault follow-up passes **2,170 tests**, with -13 explicitly opt-in skips and **86.39%** coverage. Lint, formatting and types pass. -The CDK suite passed **5,223 tests**, 56 skips and its snapshot; the later -managed-image pin/dispatch checks passed **245 tests**. CLI compilation, lint and -**943 tests** passed. These are suite results, not additive independent counts. - -The subsequent real Linear submission also passed with AgentCore Identity vault: -the guest obtained its token from the existing vault grant, opened a one-file -test PR, posted completion feedback to Linear, terminated and released capacity. -Its fallback secret contained metadata only. This check caught and fixed missing -vault fields in the shared guest configuration and a metadata-only fallback -crash. The test PR was closed unmerged and the owned Linear fixture removed. -The private reproducible harness and raw receipts are maintained outside Git. - -## Unanswered approvals implementation order - -The prerequisites are implemented in this order: - -1. Persist exact task/request/tool identity and original decision deadlines. -2. Save and restore the actual SDK conversation, including the pending action. -3. Preserve Git commits, index, worktree, required ignored files and workflow - context in bounded, checksum-verified, version-pinned storage. -4. Keep tool execution behind an ownership barrier while capturing or restoring. -5. Confirm retirement before releasing the old reservation. Admit one replacement - through a conditional coordinator transaction. -6. Deliver the saved decision to the restored agent, preserve all approved input - fields, and account for cost and turns across both worker runs. -7. Keep pending rows free of retention TTL; close them on task cancellation or - completion. Clean terminal checkpoints, launch records and worker leases. -8. Deploy compatible producers, APIs, coordinator and image before activating - automatic suspension; verify the normal submission/response path. - -The implementation does not retain an old worker indefinitely. It also does not -claim recovery when a complete checkpoint could not be verified. Unsafe parallel -or detached work, failed storage and uncertain writes remain explicit failures -or prevent suspension. - -## Approval UX and scope - -The supported response path is the signed-in CLI. `bgagent pending` shows the -task, proposed tool, reason, creation time, deadline or “no automatic expiry,” -and exact approve/deny commands. `bgagent watch` displays the request and recorded -decision. Restarting the CLI does not remove the saved request. - -Slack and Linear notification renderers carry the same response instructions. -Their deployed packages, event routing, retry receipts and closure messages -have been verified; see [approval UX](./645-p3-approval-ux-20260917.md). -No external Slack/Linear messages were sent during this acceptance run. -Native Slack decision buttons, approval-by-Linear-reply and notification -throttling remain separate product follow-ups. - -Repository mise/build-command changes are excluded at the user's request. -The earlier CLI installer/configuration addition was withdrawn. The private -verification harness's `--mise` option only selects the local test runner tool. - -## Acceptance matrix and completion gates - -| Area | Evidence and required outcome | -|---|---| -| Ordinary coding and other substrates | Clean P2 repository flow, ECS approval/cancellation and AgentCore permission checks pass | -| Same-worker sleep/wake | Approve, deny, original-deadline expiry and cancellation pass; files and identity survive | -| Credential expiry | Real expired AWS session is renewed with unchanged task/user/repo tags; no ambient fallback | -| Hook transport | Old pooled-connection race reproduced; explicit close verified before freeze | -| Retained decision | Pending request survives retirement without TTL; a later answer remains actionable | -| Replacement | One admitted worker restores exact work/action context and cumulative usage | -| Cancellation and races | Task cannot become RUNNING after cancellation; competing decisions and lost replies preserve ownership | -| Failure feedback | Bounded recovery and stable task/API guidance; hook stage, identity and timing diagnostics | -| Storage and permissions | Corrupt/interrupted transfers rejected; wrong-task reads and forbidden payload access denied | -| Capacity | Task-owned reservations, exact-attempt leases and confirmed shutdown before release | -| Nested migration | Rehearsed overlap, one merged payload-deny exception list, explicit image/coordinator pins and drain before deletion | -| Normal activation | Corrected image, actual 600-second default, timed request, off switch, compatible rollback and restored activation | -| Cleanup | No owned test worker remains alive; temporary resources, test identity and private credentials removed | - -Real AWS requests establish service behavior and effective permissions. Unit -tests cover deterministic races and clocks; the real SDK and private AWS suites -cover process loss, disk loss and transaction ownership. An accepted resume API -response alone is never counted as a successful wake. - -## Reproducible tests and handoff - -The independent harness and raw evidence live outside Git: - -`~/.local/share/abca-verification/645-p3-integration/README.md` - -Its local, real-SDK, S3, DynamoDB, coordinator and full-cloud suites record source -hashes and cleanup results. `all` runs the first five; `cloud` is an explicit -option because it builds images and uses real MicroVM/model capacity. Fresh -cloud runs use isolated tagged resources and the existing deployment's VPC. -The normal migration/CLI fixtures are installation-specific and are labelled -separately. - -The [microvms-agentd reference](https://github.com/laithalsaadoon/microvms-agentd/tree/78304e361fbbe62e3a6b255b43c6f6c372b47510) -informed reproducible commands, output-file assertions, disk-reserve checks and -cleanup receipts. No source was copied and no third-party executable was run. -ABCA keeps its task-scoped credential broker and complete Git/workspace recovery. - -## Service and later-phase follow-ups - -The [service feedback tracker](./645-lambda-microvm-service-feedback.md) preserves -requests for internal hook-dispatch traces, precise transport-error wording, -fresh-connection/retry behavior and the reproduced CloudFormation refactor tag -schema limitation. Historical worker traces remain unavailable. The application -fixes and migration do not claim that those service questions were answered. - -Unknown Run outcomes remain bounded: reuse the saved start token only within -the supported recovery window; do not blindly create a second worker. If no -unique worker can be recovered, preserve the diagnostic and the full service -lifetime bound. This is a documented operator limitation, not proof of an -unbounded AWS idempotency guarantee. - -The legacy/current capacity migration passed its isolated upgrade/rollback -rehearsal and a 600-user scan check. This installation already used task-owned -reservations, so its normal rollout preserves that protocol. Arbitrary -production scale is not established by these checks. Linear-vault integration -(#857) has its own validation, separate from this P3 acceptance record. - -ADR-021 defines P1, P2 and P3. It does not define an official P4. Native channel -approval controls, broader liveness detection (#491), operator shell access and -additional deployment combinations can be scoped as follow-up work. diff --git a/docs/verification/645-p3-lifecycle-diagnostics.md b/docs/verification/645-p3-lifecycle-diagnostics.md index d753f5315..acdec8a7f 100644 --- a/docs/verification/645-p3-lifecycle-diagnostics.md +++ b/docs/verification/645-p3-lifecycle-diagnostics.md @@ -1,54 +1,30 @@ -# ADR-021 P3: lifecycle diagnostics +# MicroVM lifecycle diagnostics -Updated 2026-09-16. The logging changes passed local checks and three isolated -AWS workflows. The [normal-stack rollout](./645-p3-diagnostics-rollout-20260916.md) -first deployed coordinator version 6 and image 5.0. The latest -[connection-close rollout](./645-p3-connection-close-rollout-20260916.md) runs -coordinator 10 and image 6.0, adds specific timeout feedback, and passed four -real Durable workflows. Automatic suspension remains disabled. The six -[recorded wake refusals](./645-p3-resume-refusal-investigation.md) are retained -for service-side diagnosis. +An approval being saved, AWS accepting Resume, and the guest resuming work are +three different events. Diagnose each independently; an accepted API response +is not a completed wake. -## What the records tell us +## Correlate coordinator and guest logs -Think of waking a worker as delivering a message, then waiting for it to finish -getting ready. These are separate checkpoints: +Search coordinator/approval Lambda logs and the MicroVM image log group using +`task_id` and `microvm_id`. `request_id` identifies the approval gate; +`aws_request_id` identifies an AWS call; `hook_id` identifies one guest HTTP hook. -1. The approval API saves the decision and a wake intent. That saved decision can - succeed even if the immediate wake request fails. -2. AWS accepts `ResumeMicrovm`. Its request ID is a receipt for that API call; - it does not prove that the guest received `/resume` or resumed execution. -3. The guest receives `/resume`, refreshes credentials, checks the original task - and approval, and returns its HTTP response. -4. The coordinator observes the resulting worker/task state. It retains the - original wake and approval deadlines across retries. - -## Cloud logs - -Search the deployed coordinator and approval API Lambda log groups by `task_id` -and `microvm_id`. The approval gate is `request_id`; an AWS service receipt is -`aws_request_id`. These identifiers have different meanings. - -| Record/message | Useful fields | +| Coordinator record | Meaning | |---|---| -| `MicroVM observed after approval decision` | Observed state, saved intent generation/time, Get request ID | -| `MicroVM wake request started after approval decision` | Task, gate, worker, image, saved generation/time | -| `MicroVM wake requested after approval decision` | Same correlation plus Resume request ID and elapsed milliseconds | -| `MicroVM lifecycle request started/acknowledged/failed` | Suspend or Resume operation, worker/image, elapsed time; AWS request ID when supplied | -| `MicroVM supervisor observation changed` | Task/worker/approval state, intent generation, recovery kind/original start, failure counters, outcome/reason, original deadlines | -| `MicroVM reached a terminal state with a substrate reason` | AWS state reason, worker/image, Get request ID | -| `Lambda MicroVM termination requested` | Worker, cleanup reason, Terminate request ID and elapsed time | +| `MicroVM observed after approval decision` | State and saved lifecycle intent observed after the decision. | +| `MicroVM wake request started/requested after approval decision` | Wake dispatch and acknowledgment, including receipt and elapsed time when available. | +| `MicroVM lifecycle request started/acknowledged/failed` | Suspend/Resume operation and result. | +| `MicroVM supervisor observation changed` | Task/worker/gate state, recovery timing, failure counters and outcome. | +| `MicroVM reached a terminal state with a substrate reason` | Service reason and worker/image identity. | +| `Lambda MicroVM termination requested` | Cleanup request and receipt. | -The supervisor saves its last diagnostic signature in durable state, so an -unchanged poll after replay does not repeat the same record. A state change, -recovery change, failure-count change or terminal outcome does. Timestamps and -elapsed time are not part of that signature. Old saved states without a signature -remain usable and log their first new observation. +Unchanged supervisor observations are deduplicated across durable replay. +Recovery timeouts retain their original start; investigating or retrying must +not reset them. -## Guest logs - -Select `/aws/lambda-microvms/`. A typical CloudWatch Logs Insights -query is: +For guest logs, select the deployed image's `/aws/lambda-microvms/` +log group and run a CloudWatch Logs Insights query such as: ```text fields @timestamp, event, action, stage, callback_stage, code, http_status, @@ -59,127 +35,40 @@ fields @timestamp, event, action, stage, callback_stage, code, http_status, | limit 500 ``` -Each HTTP invocation generates a fresh `hook_id`. All its callback-thread -records share that ID. The registered task, worker and gate identify the work; -the request body cannot override those diagnostic identities. - -| Event | Meaning | +| Guest event | Meaning | |---|---| -| `microvm_hook_started` | The HTTP handler was entered, before reading its body | -| `microvm_hook_stage` | About to perform `callback_stage`; emitted before potentially blocking work | -| `microvm_hook_stage_finished` | That piece of work returned; this alone is not a hook acknowledgment | -| `microvm_hook_stage_failed` | That piece of work failed; includes exception type and safe AWS code/request ID when available | -| `microvm_hook_finished` | Handler result: HTTP status and stable diagnostic code; cancellation has no HTTP status | - -Stages include body reading, local identity checks, the controller, draining -active writes/reads, checkpoint reads/transaction, credential refresh, and resume -identity reads/transaction. `stage` retains the last entered operation, including -when a callback is still blocked at timeout. Nested stage records also identify -their enclosing operation in `callback_stage`. - -Records include the server PID, local phase/generation, active tool/activity -counts and whether earlier progress failure disabled suspension. These are local -observations; they do not prove the listener remained healthy during a freeze. -`microvm_hook_finished` records the response the handler selected; the service's -state and HTTP access logs establish what AWS subsequently observed. - -`late: true` means a callback logged after the handler had already finished. -It cannot turn a timed-out wake into success. The existing controller still owns -the barrier that prevents tools from continuing after an uncertain wake. -Logging failure is best effort and cannot change the hook result. - -The new diagnostics omit bodies, tool arguments, approval contents, credentials, -SDK response bodies, raw exception messages and tracebacks. AWS error codes and -request IDs are bounded identifiers. Existing terminal `state_reason` retention -continues independently for service diagnosis. - -## Reading a failed wake - -1. Find the saved intent and the actual Resume acknowledgment or failure. Check - its generation and original request time; do not restart the timer while - investigating. -2. Find the corresponding guest hook start. If absent, account for log delivery - delay and retention. Absence alone cannot distinguish a dead process, missing - listener, service transport failure or missing logs. -3. If the hook started, inspect the last stage and final code. For example, - `credential-refresh` plus `AccessDenied` points to credential renewal, while - `resume-identity-read` plus `MICROVM_LIFECYCLE_UNAVAILABLE` points to task/gate - reconciliation. A timeout names the operation that was still outstanding. -4. Check the coordinator's outcome, task progress, worker cleanup and capacity - release. A saved approval and successful cleanup do not mean coding resumed. -5. Retain the task/worker IDs, exact image and coordinator versions, UTC window, - AWS request IDs, service state reason and relevant logs. For connection - refusal before hook entry, independent listener/process or service-side - evidence is still required. - -Known Resume generic failures, connection-refused, timeout and HTTP 4xx/5xx -reasons produce the stable -`MICROVM_RESUME_HOOK_FAILED` task error. It asks an admin to inspect this evidence -and saved progress before starting a replacement. It does not promise that -retrying repairs the fault. Persisted classification codes take priority over -diagnostic words; unrecognized AWS wording keeps the generic terminal code. - -The service's connection-refused wording alone does not establish that the -listener was closed. The [instrumented transport failure](./645-p3-wake-transport-20260916.md#exact-refusal-with-connection-and-listener-evidence) -recorded that wording while the original listener was still observed, after an -expired idle timer closed the old HTTP connection. Preserve the raw reason and -receipts; distinguish the guest observations from the service's unavailable -dispatch details. - -## Verification and deployment - -Local regressions cover hook correlation, sensitive-text exclusion, safe AWS -identifiers, logging-sink failure, callback timeout/late completion, durable -observation deduplication without moving deadlines, and wake error feedback. -The root `mise run build` exited **0**: **5,019 CDK tests** and **928 CLI tests** -passed, along with lint, types, contracts, synthesis and documentation checks. -After the final concurrent-output/cancellation coverage and guide edits, agent -quality and documentation checks passed again: **1,947 Python tests**, **86.43%** -coverage. **56 CDK** and **11 Python** optional DynamoDB Local cases were skipped; -the database transaction conditions were unchanged. - -The private image `backgroundagent-dev-p3-diagnostics-20260916:1.0` was derived -from the exact image `4.0` source artifact. It replaced three lifecycle Python -modules and added `microvm_diagnostics.py`; the Dockerfile, original server -command, dependencies and hook configuration were unchanged. Its artifact SHA-256 -was `15b8867dfe7f45270246695d9c87d2f1ef42d0b338fd331a88fbe5bfc26d7733`. -Private coordinator and approval Lambda versions 1 and 2 used the checked -production code with fixed verification task identities. - -The strict log audit passed at **2026-09-16 14:21:03 UTC**: - -| Case | Actual wake path | Resume AWS request ID | -|---|---|---| -| `approve-a` | Coordinator won the race; API observed restore `PENDING` and retained its successful approval response | `c6ba7eea-5f18-4ccf-b032-b0aeb0cce15a` | -| `approve-b` | Verification wrapper omitted immediate API wake; coordinator issued the sole Resume | `29ebc13a-cf70-4ed9-84b2-a0f31533406c` | -| `approve-inline` | API issued the sole Resume during a verification-only 8-second coordinator polling delay | `4b7d0ba8-3106-4126-bbaa-7ad89df9fc72` | - -The first case was originally intended to exercise the API's Resume call. Its -strict assertion failed, and the new logs showed the actual successful fallback. -That evidence was retained; the fresh third case established the missing API -path. The timing delay affected only the private test wrapper, after suspend -intent. It changed neither production supervisor logic nor original deadlines. - -Every worker produced **32 guest diagnostic records**, one successful suspend -and resume hook pair, and the expected checkpoint/refresh/reconciliation stage -records. The original server remained **PID 1** across each wake. The coordinator -recorded **9, 8 and 5** meaningful observations respectively. AWS receipts were -present for suspend, the actual resume issuer, and cleanup. No guest failure or -late-callback record appeared in these successful workflows. - -All three tasks completed with their original approval deadlines, released -capacity, zero counters, terminated workers and empty launch-payload prefixes. -Complete retained traces each contain exactly one approved `Read` of -`/etc/os-release`. These short successful wakes establish working diagnostics; -failure/redaction paths were verified by local fault injection. The original -connection-refused defect was not reproduced. - -Raw logs, traces, durable histories, artifacts, source digests, original failed -assertions and corrected audit scripts are retained in private verification -evidence. Cleanup verified removal of both private functions (all versions), -the diagnostic image, three roles, the private SSM switch, three log groups, -the artifact object and three owned zero-valued counters. Task/approval history -and trace objects retain their normal retention. -The subsequent [normal-stack rollout](./645-p3-diagnostics-rollout-20260916.md) -deployed coordinator **6** and image **5.0**, including this instrumentation. -Automatic suspension remains disabled until the remaining P3 gates pass. +| `microvm_hook_started` | Handler entry, before body reading. | +| `microvm_hook_stage` | The next potentially blocking operation. | +| `microvm_hook_stage_finished` / `microvm_hook_stage_failed` | Operation completion or safe error metadata. | +| `microvm_hook_finished` | Selected handler status/code; not proof of service receipt. | + +`stage` records the last entered operation, such as credential refresh or approval +identity reconciliation. `late: true` means a callback finished after the handler +had already returned; it cannot turn a timed-out wake into success. The coding +barrier remains responsible for preventing tools after an uncertain wake. + +## Diagnose a failed wake + +1. Locate the saved decision/intent and actual Resume acknowledgment or failure. + Retain the original generation, deadlines and API request ID. +2. Find guest hook entry. If absent, account for log delivery and retention before + inferring anything about the listener or process. +3. If entered, inspect the final stage/code. Credential-refresh `AccessDenied` + points to renewal permissions; identity-read failures point to task/gate + reconciliation. A timeout identifies the outstanding operation. +4. Check subsequent guest progress, coordinator outcome, worker termination and + capacity release. A saved approval or successful cleanup does not establish + that the approved tool ran. +5. Retain UTC timestamps, task/worker identifiers, exact image/coordinator + versions, service state reason, AWS receipts and relevant sanitized logs. + +`MICROVM_RESUME_HOOK_FAILED` identifies recognized service resume-hook failures. +Preserve the raw service reason for diagnosis. The wording “connection was +refused” alone does not prove a closed listener: historical guest observations +support a stale pooled-connection race, while service-side dispatch traces remain +unavailable. Lifecycle responses explicitly close connections before freeze. +See the [service questions](./645-lambda-microvm-service-feedback.md). + +Diagnostics omit hook bodies, tool arguments, approval contents, credentials, +raw exception messages and SDK response bodies. Do not attach signed payload +URLs or conversation/workspace checkpoints when escalating an incident. diff --git a/docs/verification/645-p3-lifecycle-hooks.md b/docs/verification/645-p3-lifecycle-hooks.md deleted file mode 100644 index 636c412ae..000000000 --- a/docs/verification/645-p3-lifecycle-hooks.md +++ /dev/null @@ -1,159 +0,0 @@ -# #645 P3: worker suspend/resume hooks - -**Follow-up (2026-09-15):** [per-worker image capability](./645-p3-image-capability.md) declares the served hooks and verifies the actual launched image version. The [supervisor milestone](./645-p3-supervisor.md) implements durable recovery and approval wake; the full repository build passes. The record below preserves this milestone's original scope. Live sleep/wake acceptance remains open. - -Date: 2026-09-14. Local implementation and DynamoDB Local verification. -Nothing in this milestone was deployed. Managed image **2.0** and automatic -suspension were unchanged. At that point, P3 still required image capability, -supervisor integration and the live acceptance matrix in the [plan](./645-p3-implementation-plan.md). - -## Behavior - -Think of a checkpoint as a signed-off bookmark: “this worker stopped here, waiting -for this answer.” It is saved outside the worker so the supervisor can inspect it. -It does not prove that AWS put the worker to sleep. - -`server.py` serves POST `/suspend` and `/resume` under -`/aws/lambda-microvms/runtime/v1`. They use the sole context registered by `/run`. -An absent/empty body, `{}`, or empty `microvmId` uses that context; a supplied -nonempty ID must match. Actual AWS suspend/resume body behavior still needs live -verification. This tolerance comes from observed empty-ID terminate requests, -not a claim about a verified sleep/wake service payload. - -The handler shares a **20-second total limit** across reading the body, draining -activity and AWS work. Bodies over 4,096 bytes are rejected. The shared contract -reserves a **30-second service timeout** for the later image hook declaration. -Invalid bodies return 400/413, unavailable or conflicting local state returns -409, and failed/uncertain work returns 503. Error responses contain a code and, -for unexpected failures, the exception type; they never echo AWS exception text. - -## Before sleep - -1. The controller requires the original approval park and exactly its blocked - tool. Parallel or unaccounted background work prevents admission. -2. It blocks new activity and waits for already-running approval reads, - credential requests, progress writes and heartbeat calls to drain. - Any previously lost progress acknowledgment keeps suspension disabled. -3. The checkpoint callback strongly reads the task and approval. It verifies - task/user/repository/VM/gate identity, the original creation time and timeout, - task status `AWAITING_APPROVAL`, a matching coordinator `suspend` intent and - approval status `PENDING`. The original deadline must still have time left. -4. One DynamoDB transaction checks both records again and writes a TaskEvents - `agent_milestone` with `milestone: microvm_suspend_checkpoint`. Metadata records - the VM, request, intent generation and original deadline. A change between the - reads and transaction aborts the whole save. -5. Only an acknowledged transaction can produce HTTP 200. The controller checks - the deadline and acknowledgment latch again before returning success. - -The strict `_ProgressWriter.write_microvm_checkpoint` path raises on missing -storage, disabled progress or failed/uncertain writes. Ordinary progress remains -best effort. Existing task-scoped IAM permissions already permit the event Put -and task/approval ConditionChecks; this milestone adds no grants or task metadata -writes. Real deployed permission validation remains a P2/P3 acceptance gate. - -A lost transaction reply can leave a bookmark in DynamoDB while the HTTP hook -fails. That bookmark alone never authorizes the supervisor to assume suspension. -There is no transaction token reused across separate HTTP attempts: DynamoDB's -cached success must not bypass a later changed approval or cancellation. - -## After wake - -The coding barrier stays closed while the callback forces renewal of the retained -ambient providers, then the same tenant credential object with the original -task/user/repository tags. Only then does it read AWS state. The -[credential milestone](./645-p3-credentials.md) covers the Claude subprocess -provider and actual pinned-CLI renewal/failure probes. - -The callback requires the current task and original gate plus a matching -coordinator `resume` intent. A transaction containing only two ConditionChecks -rechecks task and approval together. It changes neither record. A concurrent -valid approval/denial is allowed; cancellation or a changed identity, intent or -original deadline fails reconciliation. - -Successful resume releases the **same** approval loop with the **same** deadline -object. Waking grants no extra time. If the window expired, that loop applies its -existing conditional timeout and late-decision rules, including honoring a timely -decision already saved. Rejecting every expired wake would strand those decisions. - -## Retries and teardown - -- A completed duplicate suspend acknowledges the same parked checkpoint while it - remains safe; it does not write a second event. A completed duplicate resume - returns its cached result without rerunning credential refresh on active coding. -- Concurrent lifecycle requests receive 409 while the first owns the transition. - Each new approval gate clears the old wake acknowledgment. One gate can sleep - only once after a successful wake. -- After admission, failed suspend leaves the unfrozen approval waiter able to - continue and disables another suspend. Failed/timed-out resume closes the - barrier permanently. A rejected body or mismatched VM does not begin a transition. -- A thread can finish a network call after its handler times out. The controller's - generation check prevents that late completion from releasing work. `/terminate` - closes the controller before processing its body and retains its best-effort - 200 cleanup response. Its optional body read now has a one-second limit inside - the existing 15-second service timeout; a stalled stream cannot hold teardown - indefinitely. A regression reproduced the unbounded wait before this fix. -- There is no timer that opens an acknowledged suspend barrier. The supervisor - must repair or terminate an ambiguous lifecycle transition within a bound. - -Build hooks remain AWS-silent. `/validate` checks all six served routes. Direct -FastAPI route registration keeps this check valid with the installed FastAPI -version, whose included routers are lazy objects without a top-level `path`. -Image declaration remains a separate rollout gate. - -## Verification - -The focused Python tests exercise real HTTP dispatch and controller state, -duplicate/concurrent requests, slow bodies, timed-out threads, failed refresh, -termination during refresh, identity/deadline mismatch and expired wake. - -Opt-in DynamoDB Local cases execute the actual condition expressions and -transactions. They inject cancellation, approval, deadline changes and replacement -intent after strong reads. Additional cases connect HTTP → controller → production -callbacks → local database for approval, expiry, cancellation and changed intent. -They assert that wake leaves task and approval records untouched and duplicate -suspend creates only one checkpoint. - -Run from `agent/`, with your own DynamoDB Local container listening on a loopback -port: - -```bash -ABCA_DDB_LOCAL_ENDPOINT=http://127.0.0.1:8000 \ - .venv/bin/pytest tests/test_microvm_checkpoint.py tests/test_microvm_http.py \ - tests/test_microvm_lifecycle.py tests/test_server.py tests/test_hooks.py \ - tests/test_progress_writer.py -q --no-cov -``` - -The fixture accepts only `http://127.0.0.1`, supplies synthetic credentials and -creates/deletes uniquely named temporary tables. Without the opt-in endpoint, -local database cases skip. This verifies DynamoDB expression semantics, not AWS -IAM enforcement, service hook ordering, snapshot clocks or actual frozen threads. - -The CLI already renders arbitrary `agent_milestone` metadata through its generic -milestone path; the new checkpoint needs no CLI configuration changes. - -The initial full agent quality run passed **1,946 tests**, including all **11** -DynamoDB Local cases. After the additional teardown-timeout regression, the final -run passed **1,936 tests** with those 11 opt-in cases skipped because the database -container had been removed. There are **66 new tests** in total; total coverage -including branches is **86.28%**. Ruff lint/format, type checking, Vulture and the -configured Bandit high-severity gate pass. Local tables were verified empty and -the dedicated container was stopped. Evidence is retained privately as -`p3-http-agent-quality-20260914.log` and `p3-http-final-agent-quality-20260914.log`. - -The full monorepo build passed in about 13 minutes: **4,801 CDK tests** (38 -existing optional skips), **928 CLI tests**, **11 Forge tests**, infrastructure -compile/lint/synthesis, documentation build/links and contract drift checks. -The final targeted constants suite passed **37 tests** after making its negative -budget cases independent of the configured service timeout. Its lint check passed. -Evidence: `p3-http-build-20260914.log`, `p3-http-constants-final-20260914.log` and -`p3-http-final-eslint-20260914.log`. The final Python run above includes the -teardown regression added during the build. - -## Next integration - -Bind capability to the actual image/version used by each worker and declare the -matching hooks, initially with automatic sleep disabled. Wire persistent intent -and policy into supervisor polling with bounded retries/recovery, and request -wake after an approval/denial transaction commits. Then deploy and verify real -hook bodies, runtime credential renewal, managed Claude settings, Gateway signing, -long sleep, delayed transitions, cancellation, expiry and rollback. diff --git a/docs/verification/645-p3-listener-probe-20260916.md b/docs/verification/645-p3-listener-probe-20260916.md deleted file mode 100644 index 474f2176d..000000000 --- a/docs/verification/645-p3-listener-probe-20260916.md +++ /dev/null @@ -1,113 +0,0 @@ -# ADR-021 P3: isolate the resume connection failure - -Verified 2026-09-16 UTC. This diagnostic experiment follows the four -[resume-hook connection refusals](./645-p3-resume-refusal-investigation.md). -The original refusal did not recur in this bounded sample. The deliberate -closed-listener control produced the expected refusal. All temporary resources -were removed; the production image and both suspension switches are unchanged. - -## Question and scope - -Can a small HTTP listener reproduce the refusal without the coding agent? -The [probe](../../agent/scripts/microvm_lifecycle_listener_probe.py) uses Python's -standard library to serve the six lifecycle hooks. It makes no AWS SDK calls, -runs no coding task, and receives no repository or tenant configuration. - -The owned image uses the same managed base `al2023-1`, base version `1.0`, -ARM64 architecture, 8,192 MiB memory and hook budgets as production image `4.0`. -Its Dockerfile uses the same pinned Python base: - -```text -python:3.13-slim@sha256:dc1546eefcbe8caaa1f004f16ab76b204b5e1dbd58ff81b899f21cd40541232f -``` - -It has its own build/runtime roles and log group. The build role can read only -the diagnostic artifact; both roles can write only diagnostic logs. Each worker -uses the existing restricted runtime connector, explicit `NO_INGRESS` and a -300-second maximum lifetime. A standalone operator runner controls only workers -belonging to this uniquely named image. - -This is not the production ASGI server, approval controller or durable -coordinator. Responses explicitly close each accepted HTTP connection so the next -hook must contact the listener again. Success here would not prove the full -application or its keep-alive behavior correct. - -## Evidence collected - -The probe writes structured stdout records for request entry, acknowledgment, -listener health, process signals and clock gaps across suspension. `/run` creates -a random disposable marker; each `/resume` checks its original hash and records -the retained suspend/resume counts. - -The failure control deliberately stops and closes the listening socket during -`/suspend`, then sends HTTP 200 on the already accepted connection. The process -remains alive, and its observer can report that the listener is closed. That -case must not be counted as an unexpected platform failure. - -Locally, three normal cycles retained the same marker. The failure control -acknowledged suspension and a new connection then failed with `ECONNREFUSED`. -The two subprocesses were terminated after verification. Ruff, formatting and -type checks passed. - -## Fixed live matrix - -| Cases | Time from observing `SUSPENDED` to requesting Resume | Cycles per worker | -|---|---|---| -| `delay-0-a`, `delay-0-b` | No additional delay | 3 | -| `delay-2-a`, `delay-2-b` | 2 seconds | 3 | -| `delay-10-a`, `delay-10-b` | 10 seconds | 3 | -| `closed-listener` | 2 seconds; deliberate refusal control | 1 | - -The six normal cases target 18 wakes. Failed workers are not replaced or retried -until a case passes. Request IDs, actual state observations, guest records and -termination evidence are retained for each case. A launch whose response is lost -is recovered by enumerating this owned image's workers for cleanup. - -Before these cases, the private runner incorrectly sent runtime image version -`1`; AWS rejected all seven requests without creating a worker. Their records -are retained separately. The corrected request uses `1.0`, matching the image's -published version. This differs from `CreateMicrovmImage.baseImageVersion`, -which requires the major version string `1` despite returning `1.0` in reads. - -## Interpretation and cleanup - -The completed matrix produced four fully passing normal cases, 13 cycles with -all runner checks, and 14 successful resume acknowledgments in guest logs. -Three cases were interrupted by 15-second AWS request timeouts: - -| Case | Fully checked cycles | Result | -|---|---|---| -| `delay-0-a` | 3 | Passed | -| `delay-0-b` | 0 | Read-side timeout; one successful guest resume is recorded | -| `delay-2-a` | 3 | Passed | -| `delay-2-b` | 3 | Passed | -| `delay-10-a` | 3 | Passed | -| `delay-10-b` | 1 | Second Suspend request timed out; guest logs prove it nevertheless acknowledged suspension | -| `closed-listener` | 0 | Timeout before the runner requested suspension | - -These interruptions remain incomplete tests. An API timeout is not proof that -the operation never occurred. The runner terminated the owned workers without -issuing replacement runs for these cases. - -One fresh control, `closed-listener-r2`, used a bounded 45-second request timeout. -Worker `microvm-2167fc9d-21bf-35f3-a3fc-f340813f03a8` deliberately closed its -listener at 03:16:07.093Z and acknowledged `/suspend` on the existing connection. -Resume request `96147502-07f3-4512-9397-0cf1063cca67` was acknowledged at -03:16:10.143Z. At 03:16:10.420Z the observer still ran as PID 1, reporting a -closed listener. AWS recorded termination at 03:16:10.811Z with exactly the -original connection-refused reason. No `/resume` request entered the handler. -This passes the intentional failure control; it is not a fifth unexplained -refusal. - -The normal cases show non-reproduction in a small, different application. -They do not clear the four original failures or identify their cause. The next -useful evidence is listener/process health in the full agent image. - -The experiment also observed `PENDING` during real restoration. That exposed a -separate [supervisor timer bug](./645-p3-pending-wake.md), reproduced and corrected -locally. - -All eight actual workers were verified `TERMINATED`. The owned image, its build -and runtime roles, exact S3 artifact and log group were removed. Read-only checks -at 03:21:47.813Z verified their absence. The archive retains all per-case records, -208 log events, image settings, request IDs and cleanup evidence. diff --git a/docs/verification/645-p3-live-deployment-20260915.md b/docs/verification/645-p3-live-deployment-20260915.md deleted file mode 100644 index 36bd0e33a..000000000 --- a/docs/verification/645-p3-live-deployment-20260915.md +++ /dev/null @@ -1,253 +0,0 @@ -# ADR-021 P3 development deployment and guest checks - -Date: 2026-09-15. The development stack is deployed with the P3 supervisor and -six-hook image. Automatic suspension remains disabled. Six isolated guest -cases pass; suspended cancellation exposed a stop-path bug that was repaired, -deployed and verified with a stronger probe below. -The full [P3 acceptance matrix](./645-p3-implementation-plan.md#acceptance-matrix-and-completion-gates) -is not complete. - -## Deployed configuration - -| Item | Verified value | -|---|---| -| Stack | `backgroundagent-dev`, account ``, `us-west-2` | -| Source | `9a5f4606` on `fix/645-microvm-readiness` | -| Cancellation follow-up | `aff5e637`; cancellation Lambda code updated separately | -| Bootstrap | `1.8.0` | -| Stack status | `UPDATE_COMPLETE` | -| Root resources | 475 | -| MicroVM image | `backgroundagent-dev-abca-agent`, version `3.0`, `ACTIVE` / `SUCCESSFUL` | -| Guest protocol | `ABCA_MICROVM_LIFECYCLE_PROTOCOL=1` | -| Coordinator `live` alias | Lambda version `3` | -| Static suspension flag | `false` | -| Live SSM parameter | `/backgroundagent-dev/microvm-approval-suspend-enabled`, String `false`, version 1 | - -The Lambda version identifies the supervisor's saved code. The MicroVM image -version identifies the worker's saved starting computer. They are separate -version sequences. - -The image has ready, validate, run, suspend, resume and terminate hooks. Suspend -and resume each have a 30-second hook timeout. The verified image artifact is: - -```text -SHA256: 6c695b782f108064c9d5ad461e67f2071d15d1cae71e7aef47afb6dfaba7f3c3 -110 files; 483188 bytes -``` - -The coordinator's deployed code SHA256 is -`r9JIjsu84GQOfFYAepGVHKZrdps4q/bd2PeOEMCS8PY=`. - -## Rollout review and retention repair - -The initial application change set exposed a missing prerequisite: it removed -the previous coordinator and guardrail versions without retaining them. A -durable execution keeps its original Lambda version and environment, including -its guardrail version. Deleting either dependency can break later recovery. - -The fix in `9a5f4606` retains both kinds of version across updates. It passed -192 focused infrastructure tests and the full root build in 657.59 seconds: -5,030 CDK tests, one snapshot, 1,951 Python tests, 928 CLI tests and the other -configured build checks. The 56 DynamoDB Local lifecycle/capacity cases ran. - -Before replacing the existing unretained versions, a separate update adopted -retention for them. Its resource properties were identical to the previous -deployment. The only direct changes were `DeletionPolicy` and -`UpdateReplacePolicy` on those two versions; CloudFormation also listed -dependent references for reevaluation. Both live policies were verified as -`Retain` after that update. - -The final application change set contained 164 modifications, three additions -and two removals. Both removals had `PolicyAction: Retain`. There were no -table, bucket, user-pool or secret changes, no required resource replacements, -and the coordinator alias and MicroVM image retained their identities. -The stack update started at 19:11:06 UTC and was observed complete at 19:19:02 UTC. - -The combined coordinator policy comparison found no removed permissions. Added -permissions were image-scoped GetMicrovmImageVersion/Suspend/Resume, approval -GetItem/ConditionCheckItem, and exact-parameter GetParameter. Template/policy -comparison is distinct from exercising every permission against the service. - -The IAM simulator evaluated the deployed coordinator, API and worker roles -against the configured image and an unrelated image. The coordinator's tested -lifecycle actions were allowed only on the configured image; approve/deny had -Get/Resume, cancellation had Terminate, and the execution role had none of those -actions. The coordinator could read its exact live-switch parameter and could -not read the unrelated test parameter. No unrelated-image action was allowed. -The tenant-session simulation also had no allowed lifecycle action, but reported -missing DynamoDB/tag context; it is not the broader cross-task IAM acceptance -test. Actual approve/deny calls separately verified the positive Resume path. - -### Preserve the original deployment template - -In this environment, CloudFormation `GetTemplate` returned question marks in -place of Unicode characters through both the CLI and JavaScript SDK. Reusing -that response produced unrelated changes, including a Cedar layer replacement. -That change set was discarded without execution. - -The original template downloaded from the deployment's S3 asset matched the -saved assembly and preserved those characters. The retention adoption used that -artifact. Minifying the JSON also kept it within the 1,000,000-byte S3-template -limit; pretty-printing it exceeded that limit. - -## Isolated guest acceptance - -The private probe creates one owned task with no repository or notification -destination. It uses the production launch strategy, image-capability checks and -local production supervisor. For long waits it saves and rechecks a valid -suspend intent before manually requesting Suspend. Decisions invoke the -deployed approve/deny/cancel handlers with the fixture's owner identity. - -This covers actual guest hooks and decision-handler IAM. It does not exercise -API Gateway authentication, admission/capacity reservation or the deployed -durable coordinator entrypoint. The global suspension settings remain false. - -Each task asks for exactly one `Read` of `/etc/os-release`. Tool-result events -distinguish a successful read from a denied attempt. Cleanup confirms worker -termination before deleting launch payloads, then verifies the task's S3 prefix -is empty. The strengthened cancellation check requires API-driven termination -before this independent cleanup runs. - -| Case | Observed result | -|---|---| -| Ordinary, no gate | `COMPLETED`; one Read call and successful result; 17 watcher cycles, no scheduling gaps or recovery failures; service lifetime verified; VM terminated and payload deletion requested | -| Long gate + approve | Observed `SUSPENDED`; deployed approve returned 202; same task completed; one successful Read; one approval gate; original approval clock unchanged; VM terminated and payload deletion requested | -| Long gate + deny | Frozen worker woke; deployed deny returned 202; one Read attempt returned an authoritative denial, with no retry; task completed and VM terminated | -| Original-deadline timeout | Worker was RUNNING about 54 seconds before expiry; original request became TIMED_OUT; tool error arrived 140 ms after the original deadline; no retry; task completed and VM terminated | -| Decision during grace | Enabled pure policy returned `wait` / `suspend-grace`; deployed approval committed within seven seconds of gate creation; no Suspend request; one successful Read; task completed | -| Suspended cancellation | After repairing the first run's stop omission, the API logged TerminateMicrovm and AWS reported TERMINATED before fixture cleanup; task stayed CANCELLED; one gated Read attempt and no tool result | - -Ordinary task: `01M2K9H3J4V23EST0SWNE228WR`, worker -`microvm-a724cc8d-f31e-3e2d-8866-3b1ef47beaab`. - -Approve task: `01M2K9QKBSP5NQ3Q8BNNB058MH`, worker -`microvm-2f927077-c8fb-3ae5-a177-e92818409da2`. Its request -`01M2K9SMP624BEKKN4ME2K3P6R` was created at 19:48:10 UTC with a 300-second timeout. -Suspend was requested at 19:48:44.782; AWS reported SUSPENDED at 19:48:50.056. -Approval committed at 19:48:50.782 and returned 202 at 19:48:52.035. Completion -was observed at 19:48:57.119. The request's creation time and timeout were -unchanged. The worker completed between polls, so AWS RUNNING after wake was not -separately sampled. - -Deny task: `01M2KA3MFW5RAX7TT9GRHPN2WQ`, worker -`microvm-59ae9e19-4722-3143-97a9-725af4381a64`. The request was created at -19:54:23 UTC with a 300-second timeout. AWS reported SUSPENDED at 19:55:03.289; -the denial committed at 19:55:03.941 and returned 202 at 19:55:05.267. -RUNNING was observed at 19:55:10.465 and completion at 19:55:15.549. -The single Read result had `is_error=true` and an authoritative human denial. -The approval row retained its original creation time and timeout. - -Timeout task: `01M2KA8QAVWAZ61VQH12KAZS0X`, worker -`microvm-b669614a-3613-3e1f-9747-91c0c83d2f42`. Request -`01M2KAACH57GRDTMNK3M1Z5W6F` was created at 19:57:19 UTC with a 180-second timeout. -The worker was observed SUSPENDED at 19:57:55.132, gained a resume intent at -19:59:20.009 and was RUNNING at 19:59:25.281 while the approval remained pending. -Its one Read attempt returned `User timed_out` at 20:00:19.140, 140 ms after the -original 20:00:19 deadline. Completion was observed at 20:00:23.076. The original -approval row became TIMED_OUT without changing its creation time or timeout. - -Short-gate task: `01M2KAPN555FJJ0C5779GQT1RA`, worker -`microvm-9d71d590-524d-33a3-85c4-ab64e27f74f3`. Its request was created at -20:04:43 UTC with a 300-second timeout. At 20:04:48.970 the policy still required -the grace period. Approval committed at 20:04:49.706, the one Read succeeded at -20:04:52.118, and completion was observed at 20:04:55.835. The original clock -was unchanged. No Suspend request was issued. - -Read-only S3 checks confirmed zero objects under all eight completed task -prefixes, including the initial ordinary attempt below. Shared bootstrap -manifests retain their normal lifecycle policy. Task/trace evidence is retained. - -An earlier ordinary probe's guest also completed, but the local Mac entered -Maintenance Sleep for 900 seconds immediately after the first poll. The watcher -then exceeded its own deadline. `pmset` timestamps confirmed the cause. That run -does not establish supervisor timing; subsequent probes use command-scoped -`caffeinate -is` and record scheduling gaps. - -### Cancellation defect found by the audit - -The first suspended-cancellation task was `01M2KB0SX2M4JW413E8BWN4ZY2`, worker -`microvm-b7cd04f8-ee57-3341-b5ec-9f607570e310`. AWS reported SUSPENDED at -20:10:54.053; the deployed cancel handler returned 200 at 20:10:55.759. -The task became CANCELLED and the pending Read produced no tool result. - -However, the handler logged no TerminateMicrovm call. Source review confirmed -that its `status === RUNNING` guard excluded AWAITING_APPROVAL, the state used -by a sleeping approval worker. The fixture's independent cleanup subsequently -terminated the VM, masking the API's omission in the original exit-code check. -This run does **not** pass API-termination acceptance. - -The repair attempts to stop a saved session during HYDRATING, RUNNING, -AWAITING_APPROVAL or FINALIZING across all three compute substrates. It still -commits cancellation first, preserves a successful response if stopping fails, -and makes no stop call when the conditional cancellation loses a terminal race. -Pre-session states continue to skip stopping compute. - -Ten new regression assertions failed before the fix; all 34 cancellation tests -passed afterward. The related approval/supervisor/cancellation run passed 149 -tests in five suites, and compilation/lint passed. The strengthened live probe -polls for TERMINATED/not-found with a 60-second deadline after calling the API, -before any independent cleanup. - -The repair is committed as `aff5e637`. Fresh full synthesis also changed -unrelated runtime asset references and guardrail version IDs. The Bedrock alpha -construct derives the inner version ID from an unresolved UpdatedAt token; -identical guardrail settings did not produce identical version IDs in these -syntheses. Investigating that churn remains a follow-up. - -The repair's cloud assembly therefore uses the exact previously deployed -template with only the cancellation Lambda's Code/Metadata replaced by the -new CDK bundle. It publishes two file assets and no Docker assets. AWS's reviewed -change set contained that one direct modification and two unchanged API ARN -references for reevaluation. There were no image, guardrail, coordinator, IAM or -suspension-setting changes. The update was executed at 20:28:39 UTC and reached -UPDATE_COMPLETE. The live function checksum matches the published ZIP: -`L1sQUO2flAB3sO2nA+7xcuR058eJH2haBUIvhgyvhPg=`. - -The strict retry task was `01M2KC7632R97JN3YZ17RSN9GT`, worker -`microvm-0641d6d3-afd4-3d73-88a2-06a0d304c86e`. AWS reported SUSPENDED at -20:31:46.938. The deployed API logged TerminateMicrovm at 20:31:48.532 and -returned 200 at 20:31:48.576. AWS reported TERMINATED at 20:31:51.047, before -fixture cleanup. The task stayed CANCELLED, the one gated Read had no tool -result, and its payload prefix was empty. The original approval clock stayed -unchanged; cancellation leaves the approval row PENDING under a terminal task. -This passes API-driven suspended termination, but does not establish slot -accounting or cancellation races during suspend/resume transitions. - -## Evidence and remaining gates - -Private evidence directory: `/tmp/abca-645-p2-clean-20260913/`. - -- `p3-retention-root-build-20260915.log`: final full validation. -- `p3-retention-adoption-verified-20260915.json`: existing version protection. -- `p3-supervisor-reviewed-changeset-r2-20260915.json`: executed application review. -- `p3-live-configuration-verified-20260915.json`: live image, alias and flags. -- `p3-live-iam-simulation-summary-20260915.json`: per-resource lifecycle evaluations. -- `p3-live-ssm-iam-simulation-20260915.json`: exact switch versus unrelated setting. -- `p3-guest-ordinary-r2-20260915/acceptance-review.json`: ordinary task and tool evidence. -- `p3-guest-approve-20260915/acceptance-review.json`: freeze, decision, tool and cleanup evidence. -- `p3-guest-deny-20260915/acceptance-review.json`: denial prevented execution. -- `p3-guest-timeout-20260915/acceptance-review.json`: original-deadline comparison. -- `p3-guest-short-20260915/acceptance-review.json`: early decision and successful read. -- `p3-guest-cancel-suspended-20260915/acceptance-review.json`: original cancellation defect. -- `p3-guest-cancel-suspended-r2-20260915/acceptance-review.json`: API termination before cleanup. -- `p3-cancel-regression-before-20260915.log`: ten failing assertions before repair. -- `p3-cancel-regression-after-20260915.log`: all 34 cancellation tests pass after repair. -- `p3-cancel-related-tests-20260915.log`: 149 related tests pass. -- `p3-cancel-reviewed-changeset-20260915.json`: narrow CloudFormation repair. -- `p3-cancel-deployment-verified-20260915.json`: deployed code checksum. -- `p3-guest-payload-cleanup-verified-20260915.json`: empty task-owned S3 prefixes. - -The subsequent [AWS durable record](./645-p3-durable-live-20260915.md) adds -automatic suspension, timeout, rollback, crash/cancellation recovery and cleanup -evidence, plus an unresolved intermittent approval-wake failure. -The [effective IAM record](./645-effective-iam-20260915.md) adds actual metadata -and S3 permission checks plus real signer-credential expiry. The long durable -case also proves scoped credential renewal after expiry, but exposes an -[approval callback timeout](./645-p3-callback-timeout.md). Image `4.0` now carries -the correction; the [fresh live record](./645-p3-callback-live-20260915.md) -verifies nine core callback cases, including real expired-key renewal with the -original approval deadline preserved. A fourth -[connection refusal](./645-p3-resume-refusal-investigation.md) occurred on that -image. Remaining lifecycle faults/deadline races, network and -capacity/migration checks remain tracked in the -[implementation plan](./645-p3-implementation-plan.md). diff --git a/docs/verification/645-p3-mcp-network-20260917.md b/docs/verification/645-p3-mcp-network-20260917.md deleted file mode 100644 index d293b32de..000000000 --- a/docs/verification/645-p3-mcp-network-20260917.md +++ /dev/null @@ -1,160 +0,0 @@ -# P3 remote MCP and runtime networking — September 17, 2026 - -The independent audit passed on normal MicroVM image **6.0**, with the original -PID 1 server and **8192 MiB**. A real remote documentation tool worked before -and after suspension. Runtime HTTPS connections worked in both phases, port 80 -timed out, and unauthenticated requests to the worker endpoint returned 403. - -The original watcher did not pass: it missed its after-wake ingress observation. -That observation was collected independently while the worker was running. -The original tool results and Durable history establish the results below. - -## Scope - -Account ``, Region `us-west-2`, profile `sphia-dev`. -Task `01M2PENMJXNHESFSP95ECR8MMD`, worker -`microvm-1685f819-8853-377e-a764-4e409211e19b`. - -A private coordinator ran the production Durable handler and normal repository -pipeline against existing PR 584 in `isadeks/vercel-abca-linear`. Its version 2 -code hash was `GZhMPmybPf3//lClZWOkomI9BzA2I63mFEgftlFk8Lc=`. -The fixture capped the worker at 900 seconds and the agent at ten turns/$2. - -The wrapper supplied one explicit resolved MCP asset to the normal payload -transport and project loader: - -```json -{ - "kind": "mcp_server", - "namespace": "verification", - "name": "aws_knowledge", - "version": "1.0.0", - "runtime": { - "transport": "http", - "url": "https://knowledge-mcp.global.api.aws" - } -} -``` - -This tests the actual Claude CLI/SDK remote-tool path and project configuration. -It does not claim that a registry lookup occurred in this fixture. The normal -registry was unchanged. The official public AWS Knowledge MCP endpoint required -no credentials. - -Private copies of the deployed worker/session policies redirected TaskEvents to -a private table without a notification consumer. A supported prompt override -and locally verified Cedar rules allowed only the exact network command, the -read-only documentation tool and tool discovery. The watcher approved only the -exact README Read. The normal repository, role policies and PR were unchanged. - -## Actual requests before and after sleep - -The agent made five sequential substantive calls: network diagnostic, remote -documentation read, approved README Read, a new remote documentation read, and -the same network diagnostic. It also used `ToolSearch` once to load the deferred -MCP tool definition. - -Both MCP calls invoked -`mcp__verification__aws_knowledge__aws___read_documentation` with the same -GetMicrovm documentation URL and `max_length: 2500`. - -| Check | Before sleep | After wake | -|---|---|---| -| Remote MCP tool | `SUCCESS`, 01:14:02.596 UTC | `SUCCESS`, 01:15:49.755 UTC | -| TCP 443 to `docs.aws.amazon.com`, IP `52.85.31.75` | Connected, TLS 1.3 | Connected, TLS 1.3 | -| TCP 80 to that same IP | Timeout after 4.02 seconds | Timeout after 4.02 seconds | -| Unauthenticated `GET /ping` on the worker endpoint | HTTP 403 | HTTP 403 | - -The laptop control connected to both ports on that same IP. Actual EC2 rules -for runtime security group `sg-07217b11d74ddc22f` allowed only IPv4 TCP 443 -outbound, with no inbound rules. Cloud Control confirmed the active runtime -connector used that group. Its rules were identical after the test. - -Both endpoint checks were bracketed by `GetMicrovm` returning `RUNNING`. -The exact response was **“Request missing authentication.”** Current -[networking documentation](https://docs.aws.amazon.com/lambda/latest/dg/microvms-networking.html) -requires an `X-aws-proxy-auth` token for endpoint requests. Thus these checks -prove unauthenticated denial. They do not establish how `NO_INGRESS` handles a -valid token; no token was minted. The returned endpoint URL alone does not prove -guest reachability. - -The output scanner redacted part of the public documentation response. Both -explicit `SUCCESS` results and exact tool inputs remained available. This -redaction does not indicate a transport failure. - -## Sleep, wake and completion - -Approval `01M2PEV9TFC74GEAQHVR5VGR71` retained its original creation time -**01:14:11** and 300-second timeout. The watcher observed `SUSPENDED` at -**01:14:42.966** and called the normal approval handler after more than 60 seconds. -The actual `ResumeMicrovm` request ID was -`60c6af18-966f-4442-a308-f6db63465b86`. - -PID 1 acknowledged `/suspend` with HTTP 200 in **84 ms**, and `/resume` with -HTTP 200 in **149 ms**. The same worker resumed and performed the remaining -calls. This establishes successful remote-client use across the freeze; it does -not prove that the underlying TCP socket was reused. - -| Completion evidence | UTC time | -|---|---| -| Original task reservation released | 01:16:40.183 | -| Durable `finalize` step succeeded | 01:16:40.383 | -| Durable execution succeeded | 01:16:40.399 | -| Worker observed `TERMINATED` | 01:16:41.999 | -| Watcher assertion failed | 01:16:44.459 | - -The task completed with passing build/lint, existing-PR resolution, zero counter -and absent launch payload. PR head, branch, state, update time, six comments and -zero reviews remained unchanged. - -The independent audit passed at **01:18:52.171**. It verified exact tool order -and inputs, real tool results, original approval timing, request IDs and the -completion timeline. Trace SHA-256: -`439f4a120ee5bb409e70837ed14e28f9728bb9597b6fc7e2c253f750d4a201f1`; -zero trace events were dropped. - -## Fixture and observer failures retained - -The first task, `01M2PEC4NZHE3G1PNZ27GCRAVT`, used private coordinator version 1. -Its first network diagnostic passed, but the fixture had not permitted -`ToolSearch`, which CLI 2.1.191 requires for deferred MCP discovery. The watcher -rejected that unexpected approval, stopped the execution, terminated the worker -and released its slot. No MCP call or suspension occurred. This attempt is -excluded from wake acceptance. - -Version 2 permitted read-only discovery. Its watcher later searched for the -approval using `awaiting_approval_request_id`, which production correctly clears -after approval. It therefore skipped the after-wake ingress observation and -eventually asserted `1 !== 2`. An independent observer had already captured the -second 403 at **01:16:35.621–01:16:36.197**, while the worker remained running. - -The fallback called idempotent payload-deletion and reservation-release helpers -after product finalization. No execution stop or worker termination request was -needed. The independent audit verifies the earlier completion instead of -relabeling this watcher as passing. The original scripts and failures are -retained; the corrected future watcher was syntax-checked, not rerun. - -## Cleanup and archive - -All private infrastructure was removed by **01:21:57.979**: coordinator and -both versions, three roles, switch, log group, event table and two zero counters. -Both workers were terminated and their launch payload prefixes empty. -Normal task/trace records retain their existing retention policy. - -The first cleanup exhausted its 45-second polling window after AWS accepted -`DeleteFunction`. A later read returned `ResourceNotFoundException`; the -follow-up verified completed deletions and removed the remaining table/roles. -The original cleanup failure and subsequent verification are retained. -All 142 private function log events and 58 private TaskEvents were archived. - -Normal runtime role trust/policies and firewall rules were unchanged. Normal -image 6.0 remained active, and normal automatic suspension remained disabled. - -Permanent private archive: -`/Users/sphias/.local/share/abca-verification/645-p3-20260916/mcp-network-evidence.tar.gz`. -It contains **85 files**, **29,507,839 bytes**, mode `0600`, with every member -verified against its hash manifest. Archive SHA-256: -`68b806d250e6f7c829a021d8a1b7b9fb6d53c3819f18fed535b87eb7c23e2a58`. - -The [implementation plan](./645-p3-implementation-plan.md) retains the remaining -deployment and service-contract gates. diff --git a/docs/verification/645-p3-nested-stack.md b/docs/verification/645-p3-nested-stack.md index e95a8f6e5..3d0992442 100644 --- a/docs/verification/645-p3-nested-stack.md +++ b/docs/verification/645-p3-nested-stack.md @@ -1,345 +1,67 @@ -# ADR-021 nested MicroVM stack +# Nested MicroVM infrastructure -The user requested this split after the P3 approval UX review. Implementation, -fresh deployment and an overlapping-resource migration rehearsal are complete. -The normal `backgroundagent-dev` deployment now runs the nested image. Its old -flat resources have been removed after the verified drain and rollback/restore -acceptance. The parent has 471 resources and the MicroVM child has 18; -all 459 original resources outside the removal set preserved their identities. -See the [normal acceptance record](./645-p3-normal-closure-20260918.md). -This infrastructure split does not nest virtual machines. - -An isolated image-ownership refactor was also exercised on September 17. -CloudFormation accepted the preview but rejected execution because -`AWS::Lambda::MicrovmImage` has an unsupported tag schema. Its automatic rollback -preserved the original image and every resource identity. Native image refactoring -is therefore not an available migration path with the provider tested here. +This is a CloudFormation infrastructure split, not a virtual machine running +inside another virtual machine. Fresh nested deployment and a deployment-specific +migration were verified; reusable migration commands remain unfinished. ## Resource ownership -`AgentStack` creates a `LambdaMicrovmStack` child named `Microvm`. The child owns -the managed image when configured, two network connectors and security groups, -artifact and payload buckets, the log group, and build/operator roles. - -The execution role stays at the original parent path -`LambdaMicrovmCompute/ExecutionRole`, beside `AgentSessionRole`. This preserves -its logical ID and avoids a parent/child cycle through session-role trust. -Existing parent `Microvm*` outputs and orchestrator/API consumers refer to child -outputs; packaging and CLI discovery keep their existing output names. - -Image, connector and log names derive from the concrete parent deployment name. -Never sanitize the generated child stack name: it is an unresolved token during -synthesis. Child IAM roles have explicit names `-MicrovmBuildRole` -and `-MicrovmConnectorRole`; names exceeding IAM's 64-character limit -are rejected at synthesis. - -## Configuration and bootstrap - -For `compute_type=lambda-microvm`, nesting defaults to enabled. -`microvm_nested_stack=false` preserves the existing flat resource paths for -deployments that have not migrated. The setting accepts booleans or the strings -`true` and `false`. - -`microvm_managed_image_version` optionally pins new workers to a verified version -of the managed image, such as `7.0`. It changes the coordinator's runtime -selection without removing, replacing or rebuilding the image resource. Omit it -to keep the latest-active behavior. When pinned, building a newer image does not -switch new tasks to it until the operator updates the pin. Record the image ARN -and version alongside the coordinator version for rollback. - -Nested deployments require bootstrap bundle **1.9.0**. It adds only the two exact -child build/operator names to the backend-specific, unconditioned `iam:PassRole` -statement. Legacy role prefixes remain for flat deployments; the runtime -execution role is not added to this statement. Generic nested-stack execution -permissions already existed in bundle 1.7.0. - -Bundle 1.9.0 is now installed in account ``, `us-west-2`. -CloudFormation completed the reviewed one-policy update on September 17. -The installed policy exactly matches source and the bundle hash is -`dc6301b65558c8fb51200e7b712a754e1b85adda7a512b14ec4f647dc954f541`. -Effective-role simulation without a service condition allows both exact names -and denies the runtime role and a similarly named extra role. Existing parameters -and all other bootstrap resources are unchanged. Evidence is archived under -`/Users/sphias/.local/share/abca-verification/645-p3-20260916/bootstrap19`. -Installing this prerequisite does not move the normal stack's resources. - -Nesting does not change the 8,192 MiB memory baseline, hook configuration, -runtime HTTPS-only egress, separate build HTTP/HTTPS egress, task-scoped -permissions or sleep gates. Linear vault configuration is covered separately in -the [setup guide](../guides/LINEAR_SETUP_GUIDE.md#using-the-vault-with-lambda-microvms). - -## Existing deployment migration - -Do not apply the default nested template directly to an existing flat stack. -CloudFormation sees the old resources removed and child resources added. -The existing image, connector and log names may collide, and the old bucket -auto-delete resources can erase artifacts or pending task payloads. - -Keep `microvm_nested_stack=false` in the existing deployment configuration while -preparing a concrete migration: - -1. Record the actual templates, physical IDs, image versions, grants and - retained coordinator versions; preserve the build artifacts. -2. Rehearse the chosen resource-transfer or replacement procedure in an isolated - deployment, including its rollback. Inspect each change set for deletions, - replacements, custom-resource effects and named-resource conflicts. -3. Either pause admission while preserving accepted inputs, or keep old and new - resources available during a compatible consumer switch. The latter requires - verified overlapping permissions and explicit image/coordinator pins. Drain - every old worker and retained execution before removing its resources; an idle - inventory alone is not an admission fence. -4. Deploy bootstrap 1.9.0 and apply only the reviewed migration. Confirm outputs, - permission boundaries, managed image build and normal task lifecycle. -5. Keep both sleep gates off until the normal activation checks are complete; - remove temporary migration resources only after verification. - -Read-only `GetTemplateSummary` on the normal stack returned identifiers -`ImageArn`/`Name` for `AWS::Lambda::MicrovmImage`, `Arn`/`Name` for -`AWS::Lambda::NetworkConnector`, and `BucketName` for S3. That is useful migration -metadata, not proof that a nested import/refactor will succeed. - -[CloudFormation stack refactoring](https://docs.aws.amazon.com/AWSCloudFormation/latest/UserGuide/stack-refactoring.html) -supports moves between nested stacks. Live `DescribeType` reports both MicroVM -resource types as `FULLY_MUTABLE`, with `Name` their only create-only property. -However, refactoring cannot simultaneously add/delete resources, change their -configuration, or add/change parameters, conditions or mappings. The new explicit -IAM role names and child parameters therefore need staged migration templates; -the final application template is not itself a resource-transfer plan. - -Live resource-provider inspection also reports `AWS::IAM::Policy` as -`NON_PROVISIONABLE`. CDK emits these separate inline-policy resources alongside -the roles. They must not be assumed eligible for a refactor just because their -roles are `FULLY_MUTABLE`. A full migration needs a separately reviewed procedure -for inline policies and S3 auto-delete custom resources, preserving permissions -and bucket contents throughout. Existing generated role names also differ from -the new explicit names; moving ownership must not silently rename those roles. - -The failed refactor led to the explicit replacement procedure below. Import was -not assumed to work. The replacement preserves artifacts, payload access and -compatible images during overlap, using non-conflicting resource names. - -## Overlapping replacement rehearsal and normal cutover - -The private `backgroundagent-dev-p3-migration-20260917` stack first deployed flat -resources, then added a separately named nested image and its infrastructure. -A Lambda using the exact shared MicroVM execution role verified that both old -and new bootstrap markers were readable, and that `tasks/forbidden.txt` was denied -in both payload buckets. Consumer outputs switched to the new image, rolled back -to the old image, and switched forward again while both resource sets existed. -The execution-role ARN and saved marker contents stayed unchanged. - -One permission detail matters: the payload policy has an explicit -`Deny`/`NotResource`. During overlap, its single exception list must contain both -bootstrap prefixes. Adding a second deny would make the two policies deny each -other's allowed bucket. The rehearsal tested the effective permissions. - -After the old resources were removed, the rehearsal had three parent resources -and 18 child resources. The whole private stack was then deleted. Its cleanup -manifest tracks 43 old/new resource identities, verifies absence and reports no -leaks; automatically created provider log groups were removed too. - -The normal cutover follows the same staged procedure: - -- Add the child without changing the original 475 physical resource identities. -- Build `backgroundagent-dev-p3-abca-agent` under the new child. -- Switch consumers with overlapping permissions and a compatible pinned image. - Deploy retained-request producers after the compatible coordinator alias. -- Verify normal CLI decisions, worker recovery and capacity release. -- Drain old executions/workers before deleting the 16 obsolete flat resources. - Preserve the old log group intentionally for historical diagnosis. - -The normal nested image is now version 2.0, with artifact SHA-256 -`924a1b51fe6b9aa62f61191a6bde9b10df01d65873181a9385489191492cadbe`. -Compatible coordinator 13 pins that image with automatic sleep disabled; -coordinator 14 pins the same image with sleep enabled. Final off-switch acceptance -and old-resource removal passed, as recorded in the -[normal acceptance record](./645-p3-normal-closure-20260918.md). -Future updates must retain `microvm_resource_name_prefix=backgroundagent-dev-p3`; -that migration prefix now identifies the installed service resources. - -No global admission pause or zero-concurrency setting was used. The -[earlier rollout review](./645-p3-normal-rollout-review-20260917.md) explains why -that shortcut could lose asynchronous inputs on this installation. - -Exact templates, change sets, probes and cleanup results are outside Git under -`~/.local/share/abca-verification/645-p3-integration/nested-overlap-rehearsal` -and `normal-nested-rollout`. The following sections preserve earlier milestones. - -## Verification - -The dedicated construct tests exercise bootstrap, imported-image and -managed-image modes, actual parent consumers, session-role trust, stable names, -network separation, backend tags and invalid nesting inputs. Stack tests inspect -both templates and preserve a flat-layout compatibility case. Bootstrap tests -cover the exact new role names, exclusions, generated artifacts and policy hash. - -Production `mise //cdk:synth` with managed-image inputs in `us-west-2` succeeded, -including Lambda asset bundling and the production aspects. The corrected output is -under `/tmp/abca-645-nested-20260916/managed-us-west-2`; its manifest explicitly -reports `aws:///us-west-2`. This was synthesis, not deployment. - -The earlier `managed` assembly actually targeted the profile's `us-east-1` default, -despite the previous version of this record saying `us-west-2`. Setting only -`CDK_DEFAULT_REGION` did not override the CDK CLI's environment selection. The -corrected run explicitly sets `AWS_REGION` and `AWS_DEFAULT_REGION` to `us-west-2` -and checks the generated manifest. Neither synthesis run deployed a nested stack; -the separate live deployment below subsequently exercised `us-west-2`. -AWS `ValidateTemplate` also accepted the MicroVM child, reporting three parameters -and `CAPABILITY_NAMED_IAM`. That API checks template structure; it does not build -the image or validate a migration. - -| Template | Resources | File bytes | Parameters | Outputs | -| --- | ---: | ---: | ---: | ---: | -| Parent | 460 | 680,945 | 1 | 49 | -| MicroVM child | 19 | 43,895 | 3 | 8 | -| Existing registry child | 20 | 52,259 | 0 | 2 | -| Existing registry API child | 36 | 64,560 | 3 | 3 | - -Every template fits the 500-resource, 1 MiB and 200-parameter/output limits. -The hierarchy totals 535 resources, below the 2,500-resource nested-operation -limit even if every resource changed. Unit synthesis additionally covers -AgentCore, ECS and all three MicroVM image modes, each with the tool gateway -disabled and enabled. At that point it preserved the separate #857 vault guard; -the subsequent vault integration replaces that guard with parity and budget checks. - -The flat and nested templates retain the execution role logical ID -`LambdaMicrovmComputeExecutionRoleAA0C4A0D`, the same trust document and the same -parent execution-role output reference. The child stack carries the backend tag -so its stack-scoped S3 cleanup helpers inherit it as well. - -The first broad test attempt exhausted local disk while staging repeated Docker -source contexts. Generated test directories were cleared. The rerun uses -`CDK_CONTEXT_JSON='{"aws:cdk:disable-asset-staging":true}'` for structural unit -tests; the production synthesis above used actual staging and bundling. - -Final local checks: - -- CDK suite after the approval work: **232 passed, 2 skipped**; - **5,126 tests passed, 56 skipped**; - the snapshot passed. -- The final child-stack tag assertions also passed in a focused **14-test** run. -- TypeScript compilation, ESLint and whitespace checks passed. -- Documentation sync and the **77-page** site build passed. -- Bootstrap golden-baseline, artifact-sync, hash and policy coverage tests passed - as part of the CDK suite. - -Templates, test/build logs, measurements, role-identity comparison and a SHA-256 -manifest are archived at -`/Users/sphias/.local/share/abca-verification/645-p3-20260916/nested-us-west-2`. -The older `nested-stack` archive remains historical evidence of the original -structural checks, with its region correction recorded here. - -## Fresh nested deployment and cleanup - -The isolated stack `backgroundagent-dev-p3-nested-20260917` uses the production -`LambdaMicrovmStack` construct and normal CloudFormation execution role. It -imports the existing VPC and uses separate image/connector names. Its owner tag -is `abca:verification=645-p3-nested-20260917`. No normal resources were moved. - -The bootstrap phase created three parent resources and 17 child resources. -The second reviewed change set added only the managed image to the child. -Image `abca-645-nested-probe-20260917:1.0` reached `ACTIVE` / `SUCCESSFUL`; -the stack reached `UPDATE_COMPLETE`. - -Verification at `2026-09-17T12:58:18.525Z` confirmed: - -- All six lifecycle hooks, the base image, CPU, 8,192 MiB memory and environment - match normal image 7.0. -- The build uses artifact SHA-256 - `b4bc0c628f4c7976e18850ff47f92e78ecdccec683bc38974e0da82fc8600049` - from its own tagged artifact bucket. -- The build and connector roles use the exact bootstrap 1.9.0 names. - The parent execution-role identity is unchanged between phases. -- Both connectors are active in the expected subnets. Runtime egress allows - port 443; build egress allows 80 and 443; neither security group has ingress. - -No worker was launched. After archiving 1,471 image-log events, the owned stack -was deleted. Cleanup verification at `2026-09-17T13:07:47.380Z` confirms absence -of both buckets, connectors, security groups, image/version, roles, provider -function and log groups. The implicitly created provider log group was archived -and removed separately. - -At that milestone, the normal deployment retained all 475 physical resource identities, image 7.0, -coordinator alias 10, its disabled live sleep switch and the shared VPC. -Both phases' exact templates, 36 evidence files and a SHA-256 manifest are archived at -`/Users/sphias/.local/share/abca-verification/645-p3-20260916/nested-live`. - -This established fresh nested deployment, image build and deletion. The later -overlapping migration and normal task checks are recorded above. - -## Image ownership refactor: execution rejected, rollback verified - -The separate stack `backgroundagent-dev-p3-refactor-20260917` used the same -production construct, bootstrap 1.9.0 and artifact as the successful fresh build. -Its image `abca-645-refactor-probe-20260917:1.0` reached `ACTIVE` / `SUCCESSFUL`. -Baseline verification at `2026-09-17T13:47:15.644Z` checked all six hooks, roles, -network rules and the three parent / 18 child resources; 1,450 image-log events -were retained. No worker was launched. - -The planned transfer moved only child resource `ComputeImageC9058F98` to parent -resource `MovedMicrovmImage`. Image settings resolved to the same values through -existing child outputs. Every other resource remained in its original stack. -Moving the image back was conditional on successful execution and verification. - -Two preview findings required staging: - -- Changing the parent's nested-stack `TemplateURL` produced - `Found an action type that is not permitted during refactor operations: Modify`. - Keeping that property unchanged and providing the child's revised template as - its own `StackDefinition` produced an accepted preview. -- The preview removed source-stack tags and applied destination-stack tags. It - would have dropped `abca:compute-backend`, despite that tag also appearing on - the image resource. A separately reviewed tag-only update aligned this - MicroVM-only test parent's tags. Its two root changes had `Tags` scope, - identical before/after resource properties and no replacements. Image 1.0 - remained the only version. This staging choice must not be copied blindly to - the mixed-backend normal parent stack. - -Final refactor `b439bca8-a95f-4716-970d-dfad7c5b30d7` reached `CREATE_COMPLETE` / -`AVAILABLE`. Its only action was the expected image `MOVE`, described as -`No configuration changes detected.` Both user tags were preserved by the -proposed remove/reapply operations. Execution was accepted at -`2026-09-17T13:52:48.582Z`, request ID -`6f195c46-c1a5-4f40-912c-9fdc362eeac6`, then automatically rolled back: - -> Stack Refactor does not support AWS::Lambda::MicrovmImage because the resource type defines an unsupported tag schema. - -The refactor reached `ROLLBACK_COMPLETE`; both stacks reached -`UPDATE_ROLLBACK_COMPLETE`. Verification at `2026-09-17T13:54:29.943Z` proved: - -- The image retained its ARN, settings, creation time, `ACTIVE` / `SUCCESSFUL` - state and only version **1.0**. -- The original three parent and 18 child physical identities and all parent - outputs were preserved. -- The image retained both user tags and its original child-stack ownership tags. - -No successful ownership transfer occurred, so the planned reverse move was not -attempted. This is a reproduced provider limitation, not a passing migration -rehearsal. It is tracked as -[service feedback F09](./645-lambda-microvm-service-feedback.md#f09--image-refactor-preview-passes-but-execution-rejects-the-tag-schema). - -The live provider schema declares `FULLY_MUTABLE`, updatable tags and -`ImageArn` as its identifier. Its tag object requires `Key` but makes `Value` -optional. The connector has the same requirement; S3's tag object requires both. -This comparison supplies a service-team diagnostic question, not proof of the -internal validator's cause or of connector-refactor failure. - -Evidence includes all rejected previews, the accepted action list, execution -receipts, schema snapshots and rollback checks. Cleanup verification at -`2026-09-17T13:57:26.518Z` passed 14 independent absence checks for the owned image, -roles, buckets, connectors, security groups, provider function and log groups. -The three uploaded verification template versions were removed from their exact -toolkit-bucket prefix, with no remaining versions or delete markers. Normal CDK -assets were retained. - -At that milestone, the normal deployment still had all 475 original resource identities, image 7.0, -coordinator alias 10, its disabled live sleep switch and the shared VPC. The -rehearsal resources are fully deleted. Evidence and a SHA-256 manifest are -archived at -`/Users/sphias/.local/share/abca-verification/645-p3-20260916/nested-refactor`. - -One preparation guard also caught a CDK asset-key assumption: a published -template's key matched the hash of compact JSON, while its stored bytes used -formatted JSON. Their parsed contents were identical. The rehearsal therefore -used its own prefix and hashes of the actual uploaded bytes, without replacing -the existing CDK object. +`AgentStack` creates the `Microvm` child (`LambdaMicrovmStack`). It owns the +managed image when configured, build/runtime network connectors and security +groups, artifact/payload buckets, logs, and build/operator roles. + +The execution role remains at `LambdaMicrovmCompute/ExecutionRole` in the parent. +This preserves its logical ID and avoids a dependency cycle through SessionRole +trust. Parent `Microvm*` outputs retain the names used by packaging and consumers. +Names derive from the concrete parent deployment name, not the child stack token. + +## Configuration + +| Setting | Behavior | +|---|---| +| `microvm_nested_stack` | Defaults to `true` for MicroVM compute. Set explicitly to `false` for an existing flat deployment until migration. | +| `microvm_resource_name_prefix` | Supplies distinct names for overlapping nested resources; preserve it after migration. It does not retain old resources or permissions by itself. | +| `microvm_managed_image_version` | Pins new tasks to an explicitly verified image version. Without a pin, selection follows the latest active version. | +| `microvm_approval_suspend_enabled` | Defaults to `false`; enable only after testing the deployed image/coordinator. Disabling new sleep preserves wake and cleanup. | + +Nested deployment requires bootstrap bundle **1.9.0** or later. See +[deployment roles](../design/DEPLOYMENT_ROLES.md) and the +[artifact packaging instructions](../../cdk/scripts/README.md). +Build a new image before switching its runtime pin; building alone must not +silently change the image used by a pinned coordinator. Retain a compatible +published coordinator and exact image version for rollback. + +## Existing flat deployments + +Do not deploy the nested template directly over a flat deployment. CloudFormation +sees removed parent resources and newly created child resources; names can collide +and bucket auto-delete handlers can erase artifacts or pending payloads. + +The tested provider rejected native `AWS::Lambda::MicrovmImage` stack refactoring +with an unsupported tag-schema error even though the preview succeeded. Do not +assume resource import/refactoring is supported from a successful preview. + +The required overlap migration has these stages. This checklist is not yet a +runnable migration command: + +1. Preserve the exact deployed template/cloud assembly, image and published + coordinator versions. Capture current resource identities and configuration. + Keep `microvm_nested_stack=false` on ordinary updates before migration. +2. Create the child with distinct names while retaining the old resources and + their permissions. Build and verify the new image before switching consumers. + Reject intermediate templates that exceed CloudFormation's 500-resource limit. +3. Deploy compatible approval handlers/coordinator before producers of retained + requests. Preserve permissions for old workers, including the old payload + access that P3's new bootstrap deny would otherwise block. +4. Switch consumers with an explicit image pin. Verify normal tasks, retained + approvals, sleep/wake and rollback while both resource sets are available. +5. Retire old resources only after old workers, Durable executions and uncertain + starts are accounted for and old tasks/leases are drained. An idle inventory + snapshot alone is not an admission fence. Preserve checkpoint data and any + resources still needed by published coordinator versions. + +Setting a concurrency counter to zero is not a reliable pause for asynchronous +producers. A rollback must restore compatible code, image selection and IAM +without deleting pending requests or saved work. The reusable tool must enforce +these prerequisites; migration template and permission helpers alone do not +constitute that tool. Track remaining acceptance in [verification status](./README.md#open-pr-checks). diff --git a/docs/verification/645-p3-normal-closure-20260918.md b/docs/verification/645-p3-normal-closure-20260918.md deleted file mode 100644 index c5082f452..000000000 --- a/docs/verification/645-p3-normal-closure-20260918.md +++ /dev/null @@ -1,187 +0,0 @@ -# P3 normal deployment acceptance — September 18, 2026 - -The normal `backgroundagent-dev` deployment in account ``, -`us-west-2`, now uses the nested MicroVM infrastructure. P3 implementation and -deployment acceptance are complete: all four corrected-image normal cases, the -final test after removing old resources, and owned test-data cleanup passed. - -## Deployed configuration - -| Setting | Verified value | -|---|---| -| Managed image | `backgroundagent-dev-p3-abca-agent:2.0`, `ACTIVE` / `SUCCESSFUL` | -| Guest artifact SHA-256 | `924a1b51fe6b9aa62f61191a6bde9b10df01d65873181a9385489191492cadbe` | -| Artifact size | 519,287 bytes, 116 files | -| Memory baseline | 8,192 MiB | -| Lifecycle hooks | All six image/runtime hooks enabled | -| Active coordinator | Published version **14**, alias `live` | -| Compatible sleep-off rollback | Published version **13**, also pinned to image **2.0** | -| Live sleep switch | `/backgroundagent-dev/microvm-approval-suspend-enabled=true` | -| Default sleep delay | 600 seconds; per-task `0` disables sleep | -| Default approval deadline | `0`, no automatic decision expiry | -| Optional finite deadline | 30–3,600 seconds; positive policy-rule deadlines can also apply | -| Bootstrap bundle | 1.9.0 | - -The [progress/checkpoint correction](./645-p3-progress-race-20260918.md) is in -image 2.0. Reverting to image 1.0 would restore that known race; the tested -rollback therefore uses coordinator 13 with the corrected image. - -Subsequent updates to this installation must preserve the recorded CDK context: - -```json -{ - "microvm_nested_stack": true, - "microvm_resource_name_prefix": "backgroundagent-dev-p3", - "microvm_managed_image_version": "2.0", - "microvm_approval_suspend_enabled": true -} -``` - -The prefix remains part of this installation's configuration after migration. -Removing it would change the image, connector and log names again. The private -`normal-nested-rollout/progress-race-image/active-deployment-context.json` -contains the full context, including the exact base image and artifact hash. -Review the next generated change set; the archived migration templates preserve -unrelated deployed assets and are not substitutes for future source builds. - -## Actual signed-in CLI acceptance - -These cases used an owned Cognito identity, the normal API, normal Durable -coordinator, and `isadeks/vercel-abca-linear`. The fixture required approval for -one `Read` of that task's `README.md`, then used the existing clarification -workflow to finish without editing the repository. No PR, issue comment, email, -Slack message or Linear message was sent. - -The observer checked `bgagent pending`, `bgagent watch`, the actual CLI decision -receipt, stored approval identity, tool result, worker identity, terminal task -state and released capacity. A successful resume API response alone was -insufficient to pass. - -| Case | Task | Observed result | -|---|---|---| -| Omitted sleep/deadline overrides | `01M2S55A38R7V4FM532FF298KC` | Request remained pending without TTL; worker suspended after about 602 seconds; real CLI approval woke the same worker and exactly one Read succeeded | -| Explicit 300-second deadline, 30-second sleep delay | `01M2S55B567E0CZ613410VB2CW` | Same worker slept, woke before the original deadline and received `TIMED_OUT`; Read was denied; task completed and released capacity | -| New task while sleep disabled | `01M2S5W2NRDEVJVTM5VTQAHHHX` | Coordinator 13, static/live sleep flags false; worker stayed running for more than 75 seconds despite a 30-second delay, then accepted approval | -| Existing coordinator 14 task while live switch disabled | `01M2S5TMJK9JXPFF6SV7SHA5AK` | Version 14 execution was verified in its Lambda log stream; pending worker stayed running for 195 seconds despite a 120-second delay, then accepted approval | - -The default request was created at `02:24:19Z`; its worker first appeared -suspended at `02:34:21.491Z`. While it slept, the reviewed two-change rollback -switched the alias from 14 to 13 and the live SSM flag to false. The new-task -sleep-off case passed before the default request was approved at about -`02:37:49Z`. That same sleeping worker returned to `RUNNING` at `02:37:56Z`, -completed and terminated. Disabling new sleeps did not prevent an existing -sleeping worker from waking. - -All four cases ended `COMPLETED`; all workers terminated and all task-owned -capacity reservations were released. The scheduled manager released the last -two terminal reservations at `02:42:48Z`. The reviewed two-change restore -returned the alias to 14 and the SSM flag to true at `02:44:41Z`. - -An earlier normal request also survived more than two hours and retirement of -its original worker. Its later CLI approval admitted one replacement and -executed the exact Read. That evidence, including an interrupted local observer, -is described separately in the -[progress-race record](./645-p3-progress-race-20260918.md). - -## Final nested migration - -The [overlap rehearsal](./645-p3-nested-stack.md) established compatible -old/new permissions and rollback before normal migration. Native CloudFormation -image refactoring was rejected by the provider and was not used. - -Before final removal, a consistent inventory at `02:43:56Z` found: - -- No active tasks, held reservations, open worker leases or nonzero user counters. -- No live workers on the old image. -- No running Durable executions in retained coordinator versions 2–14. - -Change set `p3-normal-retire-flat-v2-20260918` removed the 16 obsolete flat -resource declarations and pruned old image/payload references from five IAM -policies. There were no replacements or unrelated modifications. The exact -final parent template hash is: - -`d497bcefaf81e217ce3b77b98cc36bc89bfcd7796c8714d2b4d63a2432848c04` - -The parent reached `UPDATE_COMPLETE` with **471 resources** in a 702,583-byte -template; its MicroVM child has **18 resources**. All **459** original resources outside the removal set -preserved their physical IDs, including the shared MicroVM execution role. -Parent output names continue to refer to the new child resources. - -Nine direct absence checks confirmed deletion of the old image, two buckets, -two connectors, two security groups and two roles. Their inline policies and -auto-delete resources completed removal. The old MicroVM log group was -intentionally retained with its 90-day retention for diagnosis; it is no longer -owned by the parent template. Seven old build-artifact versions were archived -before their bucket was removed. - -The resulting IAM simulation allows the new bootstrap-object read and explicitly -denies both the new forbidden payload path and the old bootstrap path. The final -normal smoke test exercises the actual worker and approval APIs after these -permissions were narrowed. - -## Final smoke and cleanup - -Final task `01M2S6S0YQ296KA52HM7HM0RJK` passed after the old resources and their -permissions were removed. Worker -`microvm-59f7debe-131d-3a8b-bb20-977d780f6df6` suspended at `02:52:52Z`, -received the actual CLI approval, resumed at `02:53:00Z`, completed at -`02:53:05Z` and terminated at `02:53:32Z`. Its exact Read succeeded once. -The scheduled manager released its reservation at `02:57:47.823Z`; -the observer recorded the complete pass at `02:57:53.216Z`. - -Normal test cleanup completed at `03:00:49.739Z`. All 11 synthetic-user tasks, -their closed leases, 341 approval/event/nudge rows, task object versions and -Memory episodes were removed. The zero counter, Cognito identity and private -CLI credentials were deleted and their absence verified. Repository settings -match their exact original values. - -Temporary cloud suites and the overlap rehearsal have already been deleted: -the corrected-image portable run verified removal of 40 resources, and the -overlap rehearsal verified removal of 43 recorded resources. Both report no -leaks. - -Normal fixture cleanup preserves real incurred-cost accounting and diagnostic -CloudWatch logs. It removes only the recorded synthetic user's task data, -approval/event rows, closed leases, task artifacts, conversation/workspace -objects, short-term Memory episodes, exact task memory namespaces, test identity -and private CLI credentials. Shared repository memory and shared bootstrap -configuration are outside that deletion boundary. - -An initial cleanup check observed a deleted Memory event briefly remaining in -`ListEvents`. The cleanup verifier now waits for confirmed absence with a bounded -deadline. The initial attempt and successful retry are both archived; no task -rows were deleted before that first visibility check stopped. - -## Evidence and reproduction - -The harness stays outside Git and the PR: - -`~/.local/share/abca-verification/645-p3-integration/README.md` - -The portable suites cover local restoration, the real pinned SDK with a -deterministic model, real S3, DynamoDB, the production coordinator, and fresh -cloud MicroVMs. Each run records source hashes, assertions and cleanup results. -The normal deployment scripts are explicitly installation-specific records, -not generic smoke commands. - -Key evidence directories: - -- `normal-acceptance-v2`: CLI cases, rollback/restore, final smoke and cleanup. -- `normal-nested-rollout/progress-race-image`: reviewed phase templates, final - resource identities, IAM simulation and old-resource absence. -- `runs/20260918-progress-race-cloud`: a fresh corrected image passing both - replacement approval and same-worker wake, followed by complete deletion. -- `cloud-replacement/results`: the full approve/deny/cancel/expiry and same-worker - approval/denial matrix. - -Full agent validation after the latest correction passed 2,162 tests with 13 -opt-in skips and 86.37% coverage, plus lint, formatting and types. CDK validation -passed 5,223 tests with 56 skips; the subsequent pin/dispatch checks passed 245. -CLI compilation, lint and 943 tests passed. -All 116 files in the deployed guest archive were compared with the final source -and matched exactly. - -Native Slack/Linear decision controls are separate product follow-ups; this -acceptance did not send external channel messages. Historical service-side -transport traces remain unavailable and are tracked in the -[service feedback record](./645-lambda-microvm-service-feedback.md). diff --git a/docs/verification/645-p3-normal-rollout-review-20260917.md b/docs/verification/645-p3-normal-rollout-review-20260917.md deleted file mode 100644 index a0130a9c9..000000000 --- a/docs/verification/645-p3-normal-rollout-review-20260917.md +++ /dev/null @@ -1,137 +0,0 @@ -# P3 normal deployment rollout review — September 17, 2026 - -Historical review. The later [overlapping nested cutover](./645-p3-nested-stack.md#overlapping-replacement-rehearsal-and-normal-cutover) -keeps both resource sets available, uses explicit compatible image/coordinator -pins, and drains old work before removal. That procedure supersedes the proposed -global pause below. The [completion plan](./645-p3-implementation-plan.md) tracks -current deployment status. - -This was a read-only review of `backgroundagent-dev` in account ``, -Region `us-west-2`. No admission settings, aliases, roles or task records were -changed. It identifies the remaining operational work; it is not a completed -normal deployment drain or rollback rehearsal. - -## Observed installation - -At **01:27:41.617 UTC**: - -- Normal coordinator versions **2–10** had no running Durable executions. -- All **112 task rows** were terminal; none held a capacity reservation. -- All **36 counter rows** had `active_count: 0`. -- Nine deployed functions referenced the normal coordinator's `live` alias. -- None of those nine had a configured dead-letter queue or asynchronous - on-failure destination. -- The normal coordinator alias remained on version **10**. Automatic MicroVM - suspension remained disabled. - -The initial clean-deployment source, `29dcaa74`, already contained -`acquireTaskSlot`, `reservation_version` and task-owned release markers. -This installation does not require conversion from the older counter protocol. -The [isolated legacy/current rehearsal](./645-p3-capacity-upgrade-20260916.md) -remains evidence for installations that do need that migration. -The [600-user scan check](./645-p3-capacity-scan-20260916.md) exceeds this -installation's observed table volume, without establishing arbitrary scale. - -An idle snapshot is not a fence: new work could arrive immediately afterward. - -## Actual admission producers - -Source callers and deployed `ORCHESTRATOR_FUNCTION_ARN` configuration identify -these nine entry points: - -| Producer | Work it can start | -|---|---| -| TaskApi CreateTask | Direct submissions | -| TaskApi WebhookCreateTask | Webhook submissions | -| TaskApi ConfirmUploads | Tasks whose uploads become ready | -| AdmissionQueuePickup | Previously queued tasks | -| Slack CommandProcessor | Slack submissions | -| Linear WebhookProcessor | Linear submissions and follow-ups | -| Jira WebhookProcessor | Jira submissions and follow-ups | -| OrchestrationReconciler | Released dependent/child tasks | -| StrandedOrchestrationReconciler | Recovery that can release child tasks | - -Stopping only the public create-task route does not stop the other eight. -Stopping the coordinator itself would also prevent existing Durable executions -from continuing through cleanup, so that is not a suitable drain mechanism. - -## Two unsafe shortcuts - -Slack, Linear and Jira use asynchronous Lambda processor invocation. -[AWS documents](https://docs.aws.amazon.com/lambda/latest/dg/invocation-async-retain-records.html) -that setting reserved concurrency to zero sends new asynchronous events -directly to a configured dead-letter queue or failure destination **without -retries**. The observed functions have neither configured. The zero-concurrency -method used on private fixture functions must not be copied blindly to these -normal processors. - -Denying the producers' coordinator-invocation permission is also insufficient. -`createTaskCore` and upload confirmation persist `SUBMITTED` before invoking -the coordinator. A denied or uncertain invocation can leave a submitted task -without a running execution. The stranded-task reconciler eventually marks -stuck tasks failed; it does not start them. - -The upload-confirmation log and user warning previously promised automatic -pickup. The local source now explains that startup could not be confirmed, -advises checking status before retrying, and describes failed cleanup of stuck -tasks. The existing 13 upload-confirmation tests, compilation and lint pass. -This text correction has not been deployed. - -## Image rollback is separate from coordinator rollback - -Normal coordinator versions 9 and 10 both name the image -`backgroundagent-dev-abca-agent` without `MICROVM_IMAGE_VERSION`. -The managed-image construct deliberately selects the latest active version: -the latest-active attribute can be empty during the initial image build. - -Consequently, moving the coordinator alias from 10 to 9 does **not** restore an -older MicroVM image. Once a worker starts, its saved handle records the actual -returned image version, but that does not select the version for future starts. - -Before rehearsing rollback, make image selection explicit in the reviewed -rollout procedure. The new `microvm_managed_image_version` context option pins -runtime selection while keeping the managed image under CloudFormation. Use a -version verified through `GetMicrovmImageVersion`, and preserve that coordinator -version together with its image ARN/version. This option is implemented locally; -normal deployment and rollback verification remain open. - -The external-image path also supports a version pin, -but switching an existing managed image resource to that path is an -infrastructure migration and must be reviewed for deletion/replacement. -Do not remove a managed image merely to obtain a pin. - -## Remaining execution sequence - -1. Prepare an admission pause that retains asynchronous input and has a tested - replay procedure. Inventory the receivers and scheduled/stream triggers - feeding all nine producers. Verify retention and restoration before relying - on a pause. Keep coordinator continuations, approvals, cancellation and - task finalization available. -2. Choose and record compatible coordinator **and image** rollback targets. - Preserve the current exact template, code hashes, image identity, role - policies, live switch and original producer/trigger settings. -3. Exercise the pause, then wait for already-running producer invocations and - previously accepted dispatches to settle. Repeatedly inventory all retained - coordinator versions, task states, compute handles and reservations. - Account explicitly for queued tasks, pending uploads and dependent tasks. -4. With admission fenced and the drain verified, run reconciliation and compare - held reservations with counters. This installation uses the current protocol; - do not introduce old writers merely to simulate a legacy migration. -5. Exercise the reviewed compatible rollout/rollback/restore sequence. Verify - actual returned worker image versions, completion and task-owned release. - Restore input delivery, replay retained input through its normal deduplication - path, and verify that no accepted work was lost. -6. Include the upload-feedback correction in the controlled deployment. Keep - automatic suspension disabled until the remaining applicable rollout and - service-contract gates in the [P3 plan](./645-p3-implementation-plan.md) close. - -Raw read-only evidence is in -`/tmp/abca-645-p2-clean-20260913/p3-normal-drain-20260917`. -It contains exact function names/hashes, asynchronous settings, version -inventories, table scans and the AWS documentation used for this review. - -Permanent private archive: -`/Users/sphias/.local/share/abca-verification/645-p3-20260916/normal-rollout-review-evidence.tar.gz`. -It contains nine files, 34,608 bytes, mode `0600`; every member was checked -against its hash manifest. Archive SHA-256: -`4a1e4dfd93b78c85463fdad590fe8cf93547cb01733d190080e5b8153ec65b54`. diff --git a/docs/verification/645-p3-pending-wake.md b/docs/verification/645-p3-pending-wake.md deleted file mode 100644 index caff724ba..000000000 --- a/docs/verification/645-p3-pending-wake.md +++ /dev/null @@ -1,120 +0,0 @@ -# ADR-021 P3: distinguish a pending wake from initial startup - -Date: 2026-09-16. The pending-wake correction and startup follow-up are deployed -in coordinator version 5. The exact direct pending-wake branch passed in AWS. - -## Trigger - -The [minimal listener experiment](./645-p3-listener-probe-20260916.md) observed -this real AWS sequence for worker -`microvm-47df351d-efe7-33f9-bf86-1892b66dc203`: - -| UTC time | Observation | -|---|---| -| 03:09:26.768 | `SUSPENDED` | -| 03:09:27.054 | `PENDING`, after Resume was requested | -| 03:09:27.684 | `RUNNING` | - -The service has no separate `RESUMING` enum, but `PENDING` can appear during -restoration. It does not always mean a worker's first startup. - -## Bug and correction - -When a worker is stably asleep, the supervisor has no active recovery timer. -The approval API can save a wake instruction and request Resume between -supervisor polls. If the next poll sees `PENDING`, the old supervisor starts a -startup timer using the worker's original first-observed time. - -For a worker older than five minutes, that timer is already expired. The -supervisor can fail an otherwise healthy wake immediately. - -A regression reproduced this with the production supervisor function: after -six minutes asleep, a committed approval and wake instruction followed by -`PENDING` returned `failure` with recovery kind `starting`. This is a local -reproduction of the coordinator bug, informed by a real AWS state sequence; -the minimal experiment itself does not run the durable coordinator. - -The supervisor now handles `PENDING` and unknown observations as follows: - -- A wake instruction matching the same worker and approval uses its saved - request time for bounded wake recovery. -- In deployed version 4, a worker still in `HYDRATING` may use the original - initial-startup timer. The follow-up below also covers the coordinator's - transition to `RUNNING` before AWS startup completes. -- An unexpected pending state after coding began receives one bounded - uncertainty window. - -Existing recovery timers retain their start time across replay. The original -worker lifetime and first-observed time do not change. An already-expired wake -instruction still fails, and a future instruction timestamp cannot extend the -window. The version-4 correction added no durable-state field, IAM permission -or image hook. - -## Follow-up: task RUNNING does not establish worker readiness - -The [full-agent observer experiment](./645-p3-process-observer-20260916.md) -found another startup case. Task `01M2N176C2WQ21PJ92FMTSZA9A` was already -`RUNNING` while AWS still reported its new worker as `PENDING`. Its actual first -supervisor result at 12:09:28.207 UTC used `unconfirmed`, a two-minute allowance, -instead of the intended five-minute startup allowance. - -The coordinator deliberately transitions the task before observing worker -readiness. The follow-up therefore records an optional internal -`startupConfirmed` flag, initially false and set true after AWS reports -`RUNNING`, `SUSPENDING` or `SUSPENDED`. Initial `PENDING` observations can then -use the original startup clock regardless of task status. Failed initial reads -retain that phase across replay. An older saved state without the flag does not -receive a new startup phase. Wake instructions and existing recovery clocks keep -their previous behavior. - -Two regressions failed before this follow-up and passed after it; current and -legacy confirmed-worker cases also retain bounded uncertainty recovery. -All 154 focused tests passed. The full build passed 4,995 CDK tests but hit eight -disk-space failures in one stack suite. After removing obsolete generated -assemblies, that entire suite passed all 138 tests, covering all eight failures. -Python, CLI, lint, types, contracts, documentation and synthesis passed. The -successful west-region synthesis was repeated after the same disk-space issue. - -Four real first-start `PENDING` polls in the private coordinator also retained -the original startup clock. The follow-up was deployed from `27521f85` and -verified at 12:39:13 UTC: live coordinator version `5`, code SHA-256 -`Y8SVypD7E959O6rHOY64D2m96Schdw9gb9urhQl3YWg=`. -Version 4 is retained. The environment, agent image 4.0, guardrail version 3 -and both disabled suspension switches are unchanged. This changes no IAM -permission or agent hook. - -## Validation and remaining scope - -The regression failed before the correction. All 150 tests in the supervisor, -MicroVM orchestrator and lifecycle suites passed afterward. Coverage includes -replay, stale wake instructions, pending/unknown observations, initial startup, -and completion after the guest consumes the decision and reports liveness. - -The full repository build passed in 488.56 seconds: 4,999 CDK tests passed with -56 optional DynamoDB Local tests skipped; 1,942 Python tests passed with 11 -skipped; all 928 CLI tests passed. Compilation, lint, type checks, contract/drift -checks, synthesis and the 77-page documentation build also passed. - -The first narrow deployment completed at 11:48:07 UTC in `sphia-dev`, `us-west-2`. -It selected coordinator version 4, with code SHA-256 -`1aITI4Dx3HdUMuAn9DnVLphZTpX6s4jyn+Odoz/N1Rs=`. Version 3 remains retained. -The reviewed change set modified only coordinator code/version/alias resources. -Resolved environment values, guardrail version 3 and agent image 4.0 are unchanged. -Both the environment and SSM automatic-suspension switches remain `false`. - -The exact published coordinator ZIP and deployment evidence are preserved in -the private persistent archive -`/Users/sphias/.local/share/abca-verification/645-p3-20260916/listener-pending-evidence.tar.gz` -(SHA-256 `221dffad6be148b296e95d89b501f6489db9038416d482cb0753c78465d59994`). - -The [full-agent observer's final timing control](./645-p3-process-observer-20260916.md) -passed on task `01M2N507M71T8F18ETT32Q51M2`. At 13:11:01.153Z, the production -supervisor observed real `PENDING` with no previous recovery on a worker older -than 300 seconds, and initialized wake recovery from the saved API request time. -The same worker completed, preserving its original lifetime and approval clocks, -then terminated with normal capacity/payload cleanup. Private scheduling controls -made that interleaving observable; stored states and clocks were not fabricated. - -This correction does not explain or resolve the four separate -[resume-hook connection refusals](./645-p3-resume-refusal-investigation.md). -Production automatic suspension remains off. diff --git a/docs/verification/645-p3-pid1-observer-20260916.md b/docs/verification/645-p3-pid1-observer-20260916.md deleted file mode 100644 index 2a948deb9..000000000 --- a/docs/verification/645-p3-pid1-observer-20260916.md +++ /dev/null @@ -1,151 +0,0 @@ -# ADR-021 P3: independent observation with the original PID 1 server - -This bounded diagnostic addresses a gap in the -[resume-refusal investigation](./645-p3-resume-refusal-investigation.md). -The earlier independent observer became the server's parent. Here the launcher -starts an observer child and then executes the original server command in its -own process. The server therefore remains PID 1, as in the five recorded -failures. This is private diagnostic instrumentation; the normal image is -unchanged. - -## Image and measurement - -The diagnostic derives from the exact normal image 5.0 artifact, SHA-256 -`9b3150e9e5991cc9fcc8a4adbb2cfbd8f97399f0c16baa9c2b5fe5b101e7b035`. -Only the Dockerfile command changes and `agent/scripts/pid1_observer.py` is -added. All other 110 artifact entries remain byte-identical. - -Private image: -`arn:aws:lambda:us-west-2::microvm-image:backgroundagent-dev-p3-pid1-observer-20260916`, -version `1.0`. Artifact SHA-256: -`c8f90180a2358678fa850e58a2dab8f5dc6632a00d0814e6b999d889c0cfe077`. -The image retains ARM64, 8,192 MiB, the original application, connection -keepalive settings, six hooks and existing connectors. - -The observer samples `/proc` every 100 ms. It reports the server's state, -thread count, RSS, pending signals, port-8080 listener inodes and whether -the server owns those sockets. It emits state changes, five-second summaries -and gaps in wall or monotonic time. It opens no sockets and records no command -arguments, environment, request bodies or credentials. The guest did not expose -the sampled cgroup `memory.events` file. - -A local ARM64 container control kept the server as PID 1 and the observer as -PID 7. The observer detected an open listener, then its intentional closure -while PID 1 remained alive. The control used no network access and was removed -afterward. The actual AWS worker also showed observer PID 7 independently -reading PID 1's listener. - -## Bounded cases - -The private coordinator uses the production durable handler from `a81c565d`, -with fixed owned task IDs and a repository-free `Read /etc/os-release` approval -gate. Each task allows six turns/$1, has a 150-second approval deadline, and -requests sleep after 30 seconds. The normal policy wakes it 60 seconds before -the original deadline. The observer then waits for the approval row to become -`TIMED_OUT` before submitting a late approval through the normal API. - -Only the private switch is enabled during a run. The four attempts are -sequential, stop on a failing attempt, and have six-minute observer and -1,800-second worker limits. No wake delivery, credentials, guest callback or -service response is deliberately changed. The wrapper records actual -Suspend/Resume request receipts. - -| Attempt | Task | Scope | -|---|---|---| -| 1 | `01M2NRA1BJNSBH4R5WR8VV3SHW` | Valid wake observation; excluded from automatic-finalization acceptance | -| 2 | `01M2NRA1BMK7ZBGDRCB3PBS0WH` | Passed original timeout, late rejection and automatic cleanup | -| 3 | `01M2NRA1BNVMXGD5YDVYD4XH4T` | Unexpected generic resume-hook failure; captured independent listener evidence | -| 4 | `01M2NRA1BN7RKF8E95B75S4VNT` | Not started; the run stopped after attempt 3 | - -Attempt 1 woke successfully. Its fixture expected HTTP 409 after the timeout, -but the deployed approval contract deliberately returns HTTP 404 -`REQUEST_NOT_FOUND` for missing, wrong-owner **and already-decided** approval -rows. The same approval-row condition failed after timeout, so the late -approval was correctly rejected. The fixture's assertion nevertheless triggered -fallback cleanup, including stopping its durable execution. The task finished -`COMPLETED` and the worker was terminated, but this attempt cannot establish -automatic finalization. The raw assertion and cleanup evidence are retained. - -The remaining attempts use the correct HTTP 404 contract. This did not change -the production API. A stale comment claiming an already-decided approval would -return 409 was corrected in `approve-task.ts`. - -The complete timeout case must retain the original deadline, reject the late -approval, return `User timed_out` as an error tool result, finish the task, -terminate the worker, release its reservation and empty its payload prefix -without observer repair. An error tool result here reports the denial; it is -not evidence that the file was read. - -## Failed wake with an observed listener - -Attempt 3 used worker `microvm-da3668f9-04a6-393c-939a-33c059e4b2a0`. -Its single coordinator wake was accepted, then AWS terminated the worker with: - -> Resume lifecycle hook failed. Please check your hook endpoint and application -> logs for more details. - -This is **different wording** from the five connection refusals. It is retained -as a separate failed-wake observation, without assuming a shared cause. - -| UTC, September 16 | Evidence | -|---|---| -| 18:55:13.690 | Coordinator starts Suspend | -| 18:55:13.865 | Suspend accepted; receipt `802e1264-4569-49ef-aada-70be7c26e71b` | -| 18:55:14.013 | Guest checkpoint succeeds; suspend hook finishes HTTP 200 in 91 ms | -| 18:56:13.043 | Coordinator starts Resume | -| 18:56:13.121 | Resume accepted; receipt `11bc8d9b-ada5-440e-aef3-034d8ac04848` | -| 18:56:13.350 | Independent observer runs after a 59.366-second monotonic gap | -| 18:56:13.869 | AWS records termination with the generic resume-hook failure | -| 18:56:15.893 | Observer sees task `FAILED` and durable execution `SUCCEEDED` | - -The post-restoration sample saw PID 1 in `R (running)` state, 29 threads, -156,784 kB RSS, no pending signals and ownership of the port-8080 `LISTEN` -socket, inode `481`. That inode was also present before suspension. -The sample preceded AWS's termination timestamp by approximately 519 ms. -There was no resume hook entry, stage or HTTP access line in the retained -worker log. - -This establishes that the server process and its listening socket existed at -that sample. It does not prove event-loop responsiveness, the state of a reused -HTTP connection, continuous listener health until termination, or the service's -underlying transport error. The observer samples every 100 ms but emits unchanged -state only every five seconds. - -The old classifier persisted this generic wording as -`MICROVM_SUBSTRATE_TERMINATED`, suggesting a retry. The correction recognizes -the observed message as `MICROVM_RESUME_HOOK_FAILED` and supplies administrator -guidance with `retryable=false`. Classifier and production-finalizer regression -tests passed. This corrects future feedback; it does not repair wake transport -or rewrite the historical task record. The subsequent -[normal deployment and API checks](./645-p3-wake-feedback-20260916.md) passed -with coordinator version 9 and automatic suspension still disabled. - -## Limits and evidence - -Successful wakes do not close the five earlier connection refusals. An -independent child still adds work and can affect scheduling. A listener observed -healthy around a successful wake does not establish its state during another -worker's failure. F08 supplies one failed wake with independent process/listener -observations, but its underlying connection error is still unknown. Service-side -connection evidence and event-loop/connection observations remain needed. - -Evidence directory: -`/tmp/abca-645-p2-clean-20260913/p3-pid1-observer-20260916`. -The audit records one accepted timeout case, one excluded assertion and one -unexpected wake failure, with no audit errors. The accepted trace has zero -dropped events and SHA-256 -`b8d63225e8a09b70a938bad8b36932ed6e08a5f29d928f545baab994395c8896`. -All three workers are terminated, reservations released and payload prefixes -empty. The private coordinator, its role/switch/log group and three zero counters -were removed at **19:02:01.271 UTC**, after saving 1,124 coordinator log events. -The private image, build role, artifact object and image log group were removed -at **19:05:57.158 UTC**, after saving 2,038 image log events. Ownership and -read-only absence checks passed. The normal deployment remained unchanged. - -The permanent private archive is -`/Users/sphias/.local/share/abca-verification/645-p3-20260916/pid1-observer-capacity-writers-evidence.tar.gz` -(74 files, 114,478,688 bytes, mode `0600`, SHA-256 -`a2e69354cab8fe7250efbbf8998621a023e26ae705b870e80cff0e1d4268a69f`). -Every archived file was checked against its hash manifest. It includes source -commit `b1837ec5`, diagnostic artifacts, logs, cleanup evidence, and the separate -read-only capacity-writer inventory. diff --git a/docs/verification/645-p3-process-observer-20260916.md b/docs/verification/645-p3-process-observer-20260916.md deleted file mode 100644 index 1fe48622c..000000000 --- a/docs/verification/645-p3-process-observer-20260916.md +++ /dev/null @@ -1,187 +0,0 @@ -# ADR-021 P3: full-agent process observer - -Date: 2026-09-16. Six task cases passed, including the exact API-first pending-wake -check. One test-role configuration failure is preserved separately. Temporary -resource cleanup and private evidence archiving are complete. - -## Question and scope - -The [four intermittent wake failures](./645-p3-resume-refusal-investigation.md) -have a successful `/suspend` response followed by an AWS report that the -`/resume` connection was refused. The -[minimal listener control](./645-p3-listener-probe-20260916.md) reproduced that -message by deliberately closing its listener, but did not establish what happened -inside the full agent. - -This experiment runs the full image-4 agent under an independent parent observer. -The parent records child exit status, received signals, port-8080 listener -inodes/ownership, selected process-state fields and cgroup memory counters. -It samples `/proc` without opening connections. It records no environment, -command arguments, task payloads or credentials. - -The observer changes the process tree: the server becomes a child of the -diagnostic parent. That can affect signal handling and timing. Results narrow -the investigation; they do not prove an unmodified image is reliable. - -## Artifact and isolation - -The source archive is the exact production image-4 artifact, -SHA-256 `0585a80606aaa66e9ce4dbff35a093451488f72d5a732c6cce6036b88fbcd4ed`. -The diagnostic archive changes only the Dockerfile startup wrapper and adds -`agent/scripts/microvm_process_observer.py`; 109 other entries are byte-identical. -The child still runs the original `opentelemetry-instrument uvicorn server:app` -command with its original arguments. - -Diagnostic artifact SHA-256: -`fd0875173bc9f0cc7ec7b321923e355d6d5022e4eae6ea7b958f5e7e0f92bd57`. - -The first diagnostic build failed before any task launch: the added ZIP entry -defaulted to mode `0600`, so the non-root agent could not read the root-owned -observer script. Both validation processes reported `Permission denied`; -the image eventually failed its five-minute readiness timeout. The corrected -archive explicitly uses regular-file mode `100644`. Failed image version `1.0` -and its logs remain part of the evidence; the corrected build is version `2.0`. - -The temporary image is `backgroundagent-dev-p3-observer-20260916`, in account -``, `us-west-2`. It uses the same managed base, architecture, memory, -six hooks and restricted network connectors. It has its own build role, artifact -and log group, with ownership tag `abca:verification=645-p3-observer-20260916`. -It reuses the existing worker execution role. No public ingress is added. - -The initial private durable fixture used production coordinator source -`c3b5c617`; private versions `3` through `6` use the startup-confirmation fix -in `27521f85`. All use fixed owned task IDs, one-second polls and a 1,800-second -worker lifetime cap. The fixture's SSM suspension switch is separate from production. -The fixtures ask only for one approved `Read` of `/etc/os-release`. - -Two ordinary approval cases are followed by a case that remains suspended for -six minutes before approval. The latter checks the -[pending-wake timer correction](./645-p3-pending-wake.md). Four further fixed task -IDs cover direct wake, a corrected test-role policy, and API-first timing below. -Their results are recorded separately from the fallback cases. - -## Observed results and direct-wake control - -The first three cases completed with the original worker and deadlines, -one successful `Read`, released capacity, zero user counters, empty launch -prefixes and observed worker termination. Independent parent observations -confirmed the child owned its listening socket; no child exit was logged. - -| Case | Worker | Durable polls | Owned-listener observations | -|---|---|---|---| -| Short approval A | `microvm-7897f87c-c612-3b0f-9958-3f49f9c1f0f2` | 59 | 41 | -| Short approval B | `microvm-af89965a-5607-3550-980d-900bd36c4a87` | 62 | 41 | -| Six-minute suspension | `microvm-e487702f-48d8-3a7a-93e0-93ba4ea573c9` | 363 | 24 | - -These are **fallback-wake** results. The production approval handler's intended -image-scoped permissions reject `GetMicrovm` on the separate diagnostic image. -It committed the approval and wake intent, returned 202, and the private -coordinator performed Resume. The short cases reused the same client port across -`/suspend` and `/resume`; this is a transport difference from the minimal probe. -Neither result establishes a cause for the four original failures. - -The direct-wake control uses a private copy of the production approval handler -with a dedicated role restricted to that task and the diagnostic image. Image -version `3.0` adds an explicit five-second delay before entering `/resume`. -This delay exists only in the diagnostic archive, whose SHA-256 is -`295fd2cee914ec56ee0ebef47d44df9fb9fc9f530a3baefc7da64826f791bd85`. -The production image and permissions are unchanged. - -The private coordinator's version `3` includes the startup-confirmation -follow-up. Four real initial `PENDING` polls already confirmed the original -startup clock was retained despite the task being `RUNNING`. - -The fourth task, `01M2N176C2X2M85VMAA415P4JX`, failed before approval: the private -API role lacked access to its user's synthetic rate-limit counter. It returned -500, and the watcher stopped the execution, cancelled the task, terminated -`microvm-625dca00-bc29-36c6-9a63-d3f348accfcb`, deleted launch files and released -capacity. This is a test-role configuration failure, not another resume refusal. - -The corrected policy grants `UpdateItem` on the exact owned -`RATE##APPROVE` partition. A preflight against a nonexistent approval -returned the expected 404 and verified counter value 1, without creating a task -or approval. A fresh fifth task, `01M2N3H3JVAFCK8JXGN44H6KVA`, uses private -coordinator version `4` and private approval-handler version `2`, with the same -diagnostic image `3.0`. - -The fifth case completed successfully on -`microvm-d996d527-64b5-3cb1-9f88-0a143fac9f05`. The private API acknowledged its -single Resume at 12:46:29.760Z, request -`eb6ea613-7cb6-4b62-8e84-d53828a32a31`. The coordinator first observed `SUSPENDED` -with the resume intent and initialized recovery at 12:46:24.145Z; five subsequent -real `PENDING` observations preserved that recovery clock. The task completed, -the worker terminated, and normal coordinator cleanup passed. - -This verifies direct API wake with an already-started recovery timer. The -stricter check for first observing `PENDING` with **no prior recovery** failed, -so that original failed audit remains in the evidence. Private API logs use -Lambda's text prefix; the audit was corrected to decode that prefix before -counting the actual Resume acknowledgment. - -The sixth task, `01M2N4DVR9VPEARNP0DK85PZZY`, used private coordinator version `5` -and approval-handler version `3`. It delayed completion of lifecycle snapshot -reads for at most three seconds after a matching resume intent existed with no -prior recovery. This still allowed the coordinator to win the intent write -before the delay began. The API correctly deferred with `wake-intent/intent-stale`; -the coordinator resumed the worker and completed normal cleanup. Worker -`microvm-a9d311ac-6f2b-3801-a579-71388cf3a7be` terminated at 13:01:31.203Z. -The audit recorded 382 supervisor polls and 32 owned-listener observations. -This is a successful stale-intent fallback, with the failed exact-branch audit -preserved separately. - -A seventh task, `01M2N507M71T8F18ETT32Q51M2`, uses private coordinator version `6` -and approval-handler version `4`. Its wrapper delays only the first lifecycle -snapshot read per supervisor poll when the real approval is already `APPROVED` -or a matching resume intent exists, with no prior recovery. This places the -delay before the coordinator can write a competing intent. It polls actual -`GetMicrovm` for at most three seconds until `PENDING`, then returns a fresh -database snapshot to the production supervisor. Stored states and clocks are -unchanged. The diagnostic guest still delays `/resume` entry for five seconds. -The seventh case passed the exact branch at 13:11:01.153Z: actual `PENDING`, -no prior recovery, and worker age 418,860 ms, beyond the 300-second startup -allowance. Recovery began at `1789564260293`, exactly the saved API wake-request -time. The sole private API Resume acknowledgment was at 13:11:00.823Z, request -`635ac95d-7e3e-49cf-a5b9-25566f45c3eb`. The guest's real delay lasted just over -five seconds. - -Worker `microvm-75c3deba-6819-35ec-8f33-a3b98f1f05ad` completed the single Read -and terminated at 13:11:13.063Z. The audit counted 381 supervisor polls and -32 owned-listener observations. The initial observation time and original -service deadline remained unchanged through durable replay; capacity was -released, the user counter reached zero, and launch files were absent without -watcher repair. This closes the deployed timer correction's live branch check. -It does not resolve the four original connection refusals. - -## Local checks and evidence - -Ruff lint/format and Ty passed. Real subprocess checks preserved exit code 42 and -forwarded SIGTERM to the child, reporting its termination and exiting 143. -Live Linux validation confirmed child/listener visibility. The guest does not -expose `/sys/fs/cgroup/memory.events`, so no OOM-counter evidence is claimed. - -Cleanup was verified at 13:12:44 UTC. All seven workers were terminated; -six tasks completed and the test-role failure was cancelled. All reservations -were released and launch prefixes were empty. The private coordinator and -approval function, every published version, all three diagnostic image versions, -three roles, three log groups, the private suspension parameter and three -diagnostic artifact objects were removed, with absence checks. - -The seven fake users' zero-valued capacity counters had no TTL. Their snapshots -were archived, then each row was deleted conditionally on its unchanged -reservation version and zero count; subsequent reads confirmed absence. -Task history, rate-limit rows and traces retain their normal retention policies. - -Private inputs, ownership ledgers, request IDs and results originated under -`/tmp/abca-645-p2-clean-20260913/p3-process-observer-20260916`. The persistent -archive is -`/Users/sphias/.local/share/abca-verification/645-p3-20260916/observer-startup-evidence.tar.gz`: -172,444,846 bytes, 222 entries, SHA-256 -`d5103c683c16fdc8f63504c2cded5ccec5dc435ed2bf8a8d8d5b37bbdc8be7ec`. -It includes the failed cases, 5,144 image-log events, 8,775 coordinator-log events, -39 private-API log events, durable histories and the exact deployed coordinator -ZIP. Extracting that ZIP reproduced its published code hash. The archive and its -directory are owner-only because raw histories may contain signed launch references. - -The final production check confirmed `UPDATE_COMPLETE`, coordinator version `5`, -active agent image `4.0`, guardrail version `3` and both suspension switches off. -The original resume refusals and the wider P3 acceptance gates remain open. diff --git a/docs/verification/645-p3-progress-race-20260918.md b/docs/verification/645-p3-progress-race-20260918.md deleted file mode 100644 index 3e2136d80..000000000 --- a/docs/verification/645-p3-progress-race-20260918.md +++ /dev/null @@ -1,66 +0,0 @@ -# P3 progress during checkpoint capture — September 18, 2026 - -The normal repository workflow exposed a progress-write race while saving a -continuation. This is separate from the earlier pooled HTTP connection problem. - -Task `01M2S3QH4RA02SMV7GKP0PSKR1` captured its conversation and workspace on -`backgroundagent-dev-p3-abca-agent:1.0`. An SDK progress event arrived during -the upload, while the lifecycle phase was `checkpointing`. The progress writer -treated the closed activity barrier as a lost write and permanently set -`progress_failed`. Capture nevertheless completed. Thirty seconds after the -approval request, the suspend hook correctly rejected this unsafe state with -HTTP 409; AWS terminated worker -`microvm-bfb43f74-bc32-37e5-860f-c0359bb6d94d`. - -The saved request and checkpoint survived. The coordinator confirmed termination, -parked the task and released capacity. A subsequent signed-in CLI approval -launched a replacement, which read the intended README and reached `COMPLETED`. -That recovery does not make the rejected suspend acceptable: the race is fixed -before final activation. - -## Correction and regression - -Progress events can write during continuation upload because these writes do not -modify the saved workspace. This exception applies only to the progress writer; -general activity and new tools remain paused. Capture drains progress a second -time after upload and refuses publication if a write failed or remains uncertain. -Actual suspension and credential renewal still close the activity barrier. - -The progress rejection log now includes task ID, event type and lifecycle phase. -It omits message contents and tool arguments. - -Two new regression tests failed before the correction. The corrected focused -suite passes 150 tests, including: - -- A progress event arriving during capture is acknowledged and permits suspend. -- A write started during upload must finish before capture is published. -- A real failed write prevents checkpoint publication and suspension. -- General activity remains blocked during upload; progress remains blocked during - actual suspension. - -The corrected guest artifact is 519,287 bytes with 116 files: - -`924a1b51fe6b9aa62f61191a6bde9b10df01d65873181a9385489191492cadbe` - -The normal deployment was returned to compatible coordinator 11 and the disabled -live sleep switch while rebuilding. Corrected image 2.0 is now active with -coordinator 14. Normal ten-minute sleep, explicit expiry, new and existing -sleep-off tasks, and compatible coordinator-13 rollback/restore all passed. -The [normal acceptance record](./645-p3-normal-closure-20260918.md) contains that -deployment evidence separately from the local regression results. - -## Independent evidence recovered after a watcher interruption - -The local watcher disconnected and later exceeded its deadline. The cloud task -continued without it. Full guest logs for retained request -`01M2RX085T9T5865XX1TBEVTYE` record a successful suspend acknowledgment -**601.45 seconds** after its creation. Its worker retired successfully after -about an hour, leaving the request pending without a TTL. - -The real CLI recorded approval at `2026-09-18T02:02:07.439Z`, more than two hours -after the request. A replacement worker consumed that decision and completed -the README task. This is cloud evidence recovered after the interruption, not -a passing result from the original local watcher. - -Private logs, task snapshots, exact approval receipts and regression results are -under `~/.local/share/abca-verification/645-p3-integration/normal-acceptance`. diff --git a/docs/verification/645-p3-readiness-review.md b/docs/verification/645-p3-readiness-review.md deleted file mode 100644 index 014b87b19..000000000 --- a/docs/verification/645-p3-readiness-review.md +++ /dev/null @@ -1,187 +0,0 @@ -# ADR-021 takeover review: P3 readiness - -Reviewed 2026-09-13 against `main` commit `5e10038c7e28179b302ac4de78b709795aeba3ce`. - -This review covers the existing MicroVM implementation, related P2 follow-ups, comment accuracy and nested-stack feasibility. It includes local experiments, not an AWS deployment or a new live smoke run. The associated [implementation plan](./645-p3-implementation-plan.md) turns the findings into ordered work. - -**Current status (2026-09-17):** subsequent -[normal-image acceptance](./645-p3-connection-close-rollout-20260916.md) and the -[final-image/ECS follow-up](./645-p3-final-image-and-ecs-20260917.md) verify seven -real Durable workflows on image 6.0: approval, denial, timeout, cancellation, -repository files across two sleeps, late approval winning, and wake after real -credential expiry. The ECS follow-up fixes missing approval-table configuration -and passes approval/cancellation plus 19 permission/port checks. Private -verification infrastructure was removed. Normal automatic sleep remains off -until the separately listed delivery, permission/network and deployment gates -are complete. Earlier paragraphs below preserve their dated findings. - -The later [normal repository check](./645-p3-repository-path-20260917.md) adds -an eighth successful image 6.0 Durable workflow: normal clone/setup, -approval sleep/wake, build/lint and existing-PR resolution. It uses isolated -notification storage and changes no GitHub content. Its private infrastructure -was removed. The review also corrects stale source comments that treated SDK -auto-approval as a tool restriction and repository-optional as repository-free. - -The [remote MCP/network check](./645-p3-mcp-network-20260917.md) adds a ninth -successful image 6.0 Durable workflow. Actual remote tool calls and HTTPS -connections worked before/after sleep; port 80 timed out in both phases. -Unauthenticated endpoint requests returned 403 with the worker running. -The independent audit retains the fixture/observer failures and verifies -product finalization before fallback cleanup. All private resources were removed. -The [AgentCore permission check](./645-p3-agentcore-permissions-20260917.md) -also passes 19 actual ambient/scoped requests. Normal rollout and service-contract -gates remain open. - -**Live update (2026-09-14):** subsequent work completed a [clean deployment](./645-p2-clean-deployment-20260913.md) and [real coding, PR iteration and cancellation tests](./645-p2-live-task-20260914.md), including Memory writes and runtime logging. A normal [image rebuild](./645-microvm-image-rebuild-20260914.md) activated version `2.0`; [11 live payload cases](./645-p2-payload-live-20260914.md) then verified transport/rejection, URL expiry/revocation and immediate Run replay. Those records supersede the corresponding gaps in the historical notes below. Full P2 acceptance and integrated P3 sleep/wake remain open; the original review findings are retained as a dated baseline. - -**Implementation update (2026-09-13):** subsequent local batches fix thread isolation, deletion/error/byte/contract bugs, approval heartbeat, stable MicroVM start recovery, atomic capacity reservations and coordinator metadata permissions. The latest batch implements v2 authenticated deployment manifests and single-object payload links for both ECS and MicroVM (#817/#700), with no old unsigned fallback. See the [bootstrap runbook](./645-payload-bootstrap.md) and [implementation progress](./645-p3-implementation-plan.md#implementation-progress). A further local batch removes unused logging counters in favor of structured stdout failures and verifies large registry assets through v2 delivery and the local loader. At that batch's completion, effective AWS policies, expiry/networking, stdout ingestion, remote-tool connectivity, clean deployment and P3 sleep/wake were pending; the live update above records later evidence. Findings below preserve the original reviewed baseline, rather than describing all of them as current defects. - -Further local work adds durable MicroVM start receipts, stable request tokens and bounded handle recovery. AWS token-retention/conflict behavior and unknown-ID cleanup remain live verification gates. The shared finalizer's repeated-decrement risk was subsequently reproduced and fixed locally with task-owned reservation transactions, unified counter writers and revision-guarded reconciliation, including approval waits. DynamoDB Local exercises the real transaction conditions; deployed IAM, scan scale and upgrade/drain behavior still need AWS verification. - -The first local P3 foundation adds the supervisor's pause/wake command methods and fixes approval timing. The timer now keeps both an elapsed-time stopwatch and the original clock deadline, using whichever runs out first. A sleeping VM therefore does not get a fresh approval window. At that foundation milestone, guest hooks and supervisor repair remained unfinished; subsequent milestones below supersede those gaps. The installed AWS SDK does not expose a `RESUMING` state; receiving a wake acknowledgement alone does not prove that the VM is awake. - -A second foundation adds saved sleep/wake instructions with revision stamps, so an old supervisor cannot overwrite a newer wake request. A policy helper now handles the observed state and current approval deadline. Real DynamoDB Local transaction tests cover competing writers and restart/readback cases; this is a local database test, not an AWS deployment. See the [lifecycle runbook](./645-lifecycle-intent.md). At that foundation milestone, guest hooks, supervisor integration and live verification remained open. - -**Further P3 work (2026-09-14):** the [guest pause controller](./645-p3-guest-barrier.md) -now keeps coding behind a controlled door while approval work is paused. The -[credential implementation](./645-p3-credentials.md) renews the existing AWS key -objects and gives Claude one source of task-specific keys. Actual pinned Claude -tests with fake AWS responses prove renewal before the next request and safe -failure without borrowing the parent's keys. The subsequent -[HTTP hook milestone](./645-p3-lifecycle-hooks.md) connects pause to an atomic -checkpoint and wake to credential renewal plus task/gate reconciliation. -The [image capability milestone](./645-p3-image-capability.md) (2026-09-15) declares the six hooks and checks/persists support for the actual launched image version. The [supervisor milestone](./645-p3-supervisor.md) adds durable recovery, post-commit approval wake, bounded cleanup and default-off suspension settings. The [development deployment](./645-p3-live-deployment-20260915.md) verifies image `3.0` and six isolated guest cases, including API-driven suspended cancellation after repairing its stop-path defect. Full AWS durable-workflow and sleep/wake acceptance remains open. - -The reservation review found a separate trust boundary: the old agent role could write/replace/delete its task row, including coordinator metadata. A subsequent local fix restricts main-task writes to reporting/approval attributes, removes replacement/deletion permissions and removes unused worker counter grants. It also removes unused Python submission/session-info helpers and corrects overstated tenant-isolation comments. [Metadata verification](./645-coordinator-metadata.md) records the tests and pending AWS gate. Status reports still come from the agent, and the compute role chooses session tags; this is not complete hostile-worker isolation. - -The [AWS durable follow-up](./645-p3-durable-live-20260915.md) records eleven -passing cases, including real process-crash/cancellation recovery, automatic -deadline wake, supervisor outage and coordinator-owned cleanup. Three approval -wakes failed with a service-reported connection-refused error. A long frozen -worker renewed expired task credentials but lost its approval callback; the -[callback fix](./645-p3-callback-timeout.md) is deployed in image `4.0`. -[Fresh AWS acceptance](./645-p3-callback-live-20260915.md) passed all nine core -callback cases, including real expired-key renewal and the original approval -deadline. A fourth [connection refusal](./645-p3-resume-refusal-investigation.md) -occurred on image `4.0`; it remains a separate P3 blocker. -The [effective permissions record](./645-effective-iam-20260915.md) -adds 37 metadata checks, 10 S3 checks and real signer-credential expiry. -Production automatic suspension remains disabled. - -## Start here: the pieces in plain language - -A **MicroVM** is a small, isolated computer rented from AWS. **Firecracker** is the technology that keeps these small computers separate. A **backend** is the kind of rented computer ABCA chooses to run a coding task. - -A **snapshot** is a saved picture of the computer's memory and disk. An **image** is the prepared starting snapshot: installed tools, server and warm files, without a particular user's task. Suspending saves the current task's computer so it can continue later. It is like closing a laptop halfway through homework. Suspending stops compute charges, but snapshot storage and read/write charges remain. The eight-hour session limit includes time asleep. - -The **orchestrator** is ABCA's supervisor. It starts computers, checks tasks, and cleans up. A **hook** is a small HTTP handler AWS calls at a lifecycle event, such as “prepare the image” or “about to stop.” These lifecycle calls do not require opening the agent to the public internet. - -**CloudFormation** is AWS's deployment system. A **stack** is a group of resources it creates together. **CDK** is code that produces the deployment recipe, called a **template**. A **construct** is a code-organizing box; it does not give its resources a separate CloudFormation quota. A **nested stack** does: it is a smaller deployment managed through the parent deployment. Nesting infrastructure does **not** mean running a VM inside another VM. This review evaluates nested CloudFormation stacks, not nested hardware virtualization. - -An **IAM role** is a permission badge. A **trust policy** says who may wear that badge. **PassRole** lets a caller hand a specific badge to an AWS service. A **session role** is the more tightly limited badge for one task. Its **tags** carry task/user/repository identity so access can be restricted to that task's data. - -## What is finished? - -| Phase | Purpose | Current state | -|---|---|---| -| P1 | Build the computer, start it, deliver a task, check it and stop it | Merged in [#689](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/689). Strategy, infrastructure, bootstrap permissions, packaging, types, `/ready` and `/run` exist. | -| P2 | Make a real coding task work with configuration, permissions, logs and progress | Merged in [#733](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/733). `/validate`, `/terminate`, warm-up, runtime grants and heartbeat support exist. The September 14 clean rerun passed coding, iteration and cancellation without manual IAM workarounds, followed by image 2.0 payload/start checks. The broader deployed recovery, effective IAM and network matrix remains open. | -| P3 | Sleep during a human approval wait, wake correctly, and keep deadlines/credentials safe | Implemented and deployed. Nine image 6.0 Durable workflows verify the core lifecycle, repository-file persistence, normal clone/PR resolution, real expired-credential renewal and remote MCP/network behavior. Normal rollout and service-contract gates remain open; automatic sleep stays off. | -| P4 | — | ADR-021 defines no P4. Verification runbooks have their own numbered phases; those are not extra ADR milestones. | - -The old unchecked checklist and the word “proposed” do not erase the merged work. Conversely, merged code is not proof that the final deployment path works unattended. - -## Can MicroVM infrastructure be nested? - -**Yes. A local prototype works with a deliberate permission boundary. A one-line wrap is undeployable.** - -The naive change puts the existing `LambdaMicrovmCompute` construct inside a `NestedStack`. Two things go wrong: - -1. The parent `AgentSessionRole` trusts the child's MicroVM execution role. The child also needs the parent session role's ARN (AWS resource address). Each needs the other created first. This is a **circular dependency**. `app.synth()` wrote templates, but `Template.fromStack()` rejected the result as undeployable with `AgentSessionRole → MicrovmNested → AgentSessionRole` in the cycle. Merely producing a template is insufficient verification. -2. The construct sanitizes `Stack.of(this).stackName` for image/connector names. A nested stack's generated name is an unresolved CDK **token**, a placeholder filled in later. String sanitization destroyed that placeholder, producing names such as `--Token-TOKEN-8353---abca-agent`. Names also varied with construct creation order. - -The successful prototype kept the MicroVM **execution role in the parent**, beside the shared session role, and derived resource names from the stable parent name. The child contained image/build/network/bucket/log resources. This removed the cycle; it is not yet a production refactor. - -```mermaid -flowchart TB - Parent["Parent: shared platform + session role + MicroVM execution role"] - Child["Child: MicroVM image, build role, connectors, buckets and logs"] - Parent -->|"VPC and subnet inputs"| Child - Child -->|"resource addresses returned as outputs"| Parent -``` - -The arrows show configuration passing, not a circular creation dependency: parent resources that depend on child outputs must be separate from the parent resources the child needs to create itself. - -### Fresh measurements - -Offline synth, bundling disabled, `backgroundagent-dev`, example account `123456789012`, `us-east-1`, `suppressTemplateIndentation: true`. “Image” means a managed image with both `microvm_base_image_arn` and `microvm_base_image_version`. The vault guard was disabled **only in the scratch experiment** to count that combination. - -| Configuration | Current root resources | Prototype root resources | MicroVM child resources | -|---|---:|---:|---:| -| Default AgentCore | 454 | 454 | — | -| MicroVM, no image yet | 471 | 457 | 17 | -| MicroVM + managed image | 472 | 457 | 18 | -| Above + tool gateway | 479 | 464 | 18 | -| Above + Linear identity vault | 489 | 474 | 18 | - -The fullest root template measured 536,098 bytes; the prototype root measured 515,859 bytes, with a 31,884-byte MicroVM child. Existing registry children had 19 and 35 resources; the optional consent-page child had 12. These are counts per stack, not totals for the whole deployment. The full prototype saves **15 root resources** and about **20 KB**, leaving 26 root resource slots under the 500-resource limit. - -The historical 505-resource refusal and 98.6%-full byte claim are stale for this source revision. The guard still exists in shipped code, and these unbundled experiments do not justify silently removing it. A production refactor must also measure real bundled templates and all supported feature combinations. See [probe results](./645-nesting-probe-results.json). - -### Conditions for a production split - -- Add an explicit execution-role injection point and stable deployment-name input. Do not put `nestedStackParent` lookups throughout production code simply because the scratch probe used one. -- Preserve parent `Microvm*` output keys consumed by the packaging helper and CLI; return child identifiers through outputs. -- Recheck bootstrap IAM. The current MicroVM PassRole patterns match `backgroundagent-dev-LambdaMicrovmComputeBuild*` and `...Connector*`. Moving resources changes generated physical role names, which may no longer match. Use narrowly scoped stable names/patterns, regenerate the bundle and bump its version if permissions change. -- Check every cross-boundary grant, tags, solution user-agent aspect and all nested templates. Parent-only resource assertions no longer cover the child. -- Plan migration. Moving a resource to another stack changes its identity to CloudFormation and can cause replacement. Artifact/payload buckets currently use destructive removal settings; named images/connectors can also collide with their old copies. Review the actual change set and choose a fresh experimental deployment or a supported resource-preserving migration. Never assume moving CDK code moves live resources safely. - -Nesting is useful preparation, especially for the vault combination, but it does not implement P3 and is not a fundamental prerequisite for an isolated MicroVM pause/resume prototype. - -## Behavior findings that need follow-up - -“Confirmed” below means visible in the baseline source or reproduced locally during the initial review. It does not mean reproduced on AWS during this review. - -| Finding | Evidence and consequence | Treatment | -|---|---|---| -| Finalize deletes without permission | `orchestrate-task.ts` calls `deleteMicrovmPayload`; `task-orchestrator.ts` grants only PutObject. Tests explicitly assert deletion permission is absent. Prompt data remains until lifecycle deletion when the call is denied. | Confirmed; [#817](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/817). Add narrowly scoped coordinator delete permission and reverse the wrong assertions. | -| Raw AWS reason text changes error classification | Executing the real classifier with a substrate-completed message plus `MicroVM host unavailable.` produces unsupported-region/non-retryable classification. Hook HTTP 400 is already classified correctly; that old example is stale. | Reproduced locally; #817. Use structured category/code for decisions and preserve service text as diagnostics. | -| ARN validation is consistency, not deployment identity | `_reject_foreign_arns` accepts another workspace's secret in the same account. Its anchor comes from the same supplied block. It checks partition/account agreement, not trusted provenance. | Local validator proof; #817. No forged `/run` reachability or credential theft demonstrated. Trusted deployment binding and negative ingress tests are needed. | -| Payload reader is broader than one task | Worker execution roles can read/list the payload bucket, so prompt data for other tasks may be accessible to untrusted task code. | Confirmed; [#700](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/700), shared with ECS. Task-scoped transport design is separate from deleting completed payloads. | -| A long approval wait leaves a stale heartbeat on resume | `write_heartbeat` only writes while status is RUNNING. `transact_resume_from_approval` restores RUNNING without refreshing the timestamp. A poll before the next 45-second tick can see an age over 240 seconds and fail a healthy task. | Source-proven race window, no live reproduction. Add atomic refresh plus an adversarial ordering test before P3. | -| Heartbeat is not a general progress watchdog | It runs on an independent thread. A stuck coding thread can coexist with fresh heartbeats. | Confirmed by call structure. Corrected comments; broader progress detection belongs with [#491](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/491). | -| Start retries have an uncertain-outcome window | `startSessionWithRetry` calls start again on transient errors. No application-stable MicroVM client token links those calls. Lost success response does not prove the first VM was never created. | Source risk, not a demonstrated live leak. Add response-loss fault injection and attempt-scoped idempotency/reconciliation design. Corrected the false guarantee. | -| Snapshot refresh must preserve task identity | `reset_session_cache()` clears tenant tags as well as the session; existing clients and the Claude credential helper can hold separate state. | Confirmed. It is a test reset, not a ready-made `/resume` implementation. | -| Existing progress writes do not provide a suspend barrier | `ProgressWriter._put_event` writes synchronously but catches/drops failures and can disable itself. There is no acknowledged queue-flush contract. | Confirmed. P3 must add a narrow durability barrier with a failure path. | -| Repeated MicroVM poll errors never escalate | The MicroVM branch logs and continues; ECS already has counters. | Confirmed P3 gap. Persist retry counters in durable poll state and bound recovery. | -| Logging failure count is invisible | `_debug_cw_failures` is incremented but never read/exported; `_DEBUG_CW_FAILURE_EMIT_EVERY` is unused. | Confirmed; [#810](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/810). Comments corrected; telemetry behavior remains work. | -| Server tests can leak background work | Test setup clears `_active_threads` without ensuring the threads finished; late work can use the next test's mocks. | Confirmed structure; [#841](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/841). Fix before trusting expanded hook tests. | - -The security findings are readiness work, not cosmetic cleanup. This review does not silently implement them or claim that their absence makes every existing deployment exploitable. - -### P2 issue map - -- **#817 is the primary tracker.** [#813](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/813), [#814](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/814), [#815](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/815) and [#816](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/816) closed because they were consolidated, not because all fixes landed. Its five tracks are classifier correctness, payload deletion IAM, contract/provenance checks, stale docs, and truncated-S3-body route coverage. `_PayloadFetchError` already distinguishes unreadable payloads, but the real bad-byte path needs a route-level regression. -- **[#818](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/818): registry tool networking and payload size.** Runtime networking is HTTPS/443-only on all three shipped backends, not uniquely MicroVM. Correct the premise, define supported tool ports, and test oversized resolved registry assets through the 4,096-byte/S3 path. -- **[#701](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/701): bootstrap refresh.** Checking the source version is insufficient; verify the deployed bundle and effective roles. [#867](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/867) fixed bootstrap template byte size, a different problem. -- **[#857](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/857): vault + MicroVM guard.** It is on main. [#854](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/854) reclaimed room; this review's newer measurements supersede old counts, not deployment verification. -- **[#811](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/811): ECS Haiku model environment parity.** Separate backend fix; not a reason to block MicroVM pause/resume. -- **[#702](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/702): teardown leaks.** Account for AgentCore ENIs (network attachments) and Memory deletion state when cleaning the test deployment. Separate platform operations work. -- **[#736](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/736): future IAM conditions.** Revisit when the service supports suitable context keys. Do not restore the previously broken `iam:PassedToService` condition just to make policies look tighter. - -## Cleanup in the original review commit - -Changes are comments, Python docstrings, documentation, one cdk-nag explanation string and the vault guard’s error wording. The error no longer claims a current 505-resource overflow; the guard still rejects exactly the same combination. The cdk-nag change affects template metadata, not IAM permissions. - -- Describe P2's successful workaround-assisted smoke and the remaining clean verification precisely; retain the stable warning ID. -- Remove the false no-delete-by-design explanation and point to the missing permission. -- Correct heartbeat, retry-idempotency, progress-durability and suspended-quota claims. -- Correct unexported logging-counter claims and the region/credential distinction. An offline probe on boto3/botocore 1.43.78 changed `AWS_DEFAULT_REGION`: new clients used the new region while the resolved credentials object remained cached. -- Explain that missing `platform_config` is accepted only when the effective environment already has required identifiers; describe the limits of ARN consistency validation. -- Fix stale ECS sizing/context instructions, root/nested deployment wording, old template counts, and PRNG terminology. Reseeding Python `random` does not make it suitable for secrets. - -## Review coverage and limits - -Reviewed the MicroVM construct, strategy, shared strategy interface, orchestrator start/poll/finalize paths, approval handlers and task API grant seams; the session role and bootstrap policy; agent hook dispatch/config installation, heartbeat, approval transactions/timers, credentials and progress writer; relevant tests/contracts; ADR, compute/orchestrator/deployment docs, packaging and recorded P1/P2 runbooks. - -Local checks and results are recorded in the implementation plan's review-validation section. Source review and mocked tests cannot establish AWS hook ordering, snapshot clock behavior, credential refresh after a long freeze, actual network reachability, or a safe migration of deployed resources. Those remain explicit live gates rather than claims of completion. diff --git a/docs/verification/645-p3-registration-20260916.md b/docs/verification/645-p3-registration-20260916.md deleted file mode 100644 index 62db3bf76..000000000 --- a/docs/verification/645-p3-registration-20260916.md +++ /dev/null @@ -1,113 +0,0 @@ -# ADR-021 P3: durable registration and unknown-worker recovery - -Verified September 16, 2026, in account ``, `us-west-2`. -All three cases passed on normal image 5.0 with sleep disabled. They used the -production durable coordinator from source `a81c565d` with explicit SDK-boundary -faults in a private wrapper. Normal coordinator version 8 was unchanged. - -## What was verified - -| Case | Task | Worker | Result | -|---|---|---|---| -| Reply lost after registration commits | `01M2NMZS79FGKJVA3EP10AZB2W` | `microvm-ca2ceb94-be3f-3bec-a02a-9d4bb4396ee0` | Recovered saved identity; one Run, one approved Read, normal cleanup | -| Cancellation during registration | `01M2NMZS7F6JPKPBF9YVDJBY3J` | `microvm-162d5f5c-ed2c-3a3e-b066-af2685ea199d` | Cancellation committed first; identity retained for automatic termination; zero tools | -| All launch replies lost | `01M2NMZS7FARVEH9VE1ES3DGZB` | `microvm-3e8f25f0-0d9d-34fe-84e4-b27bcfa5e4f9` | Task reported uncertain start; operator recovered and stopped the live worker from an exact guest-log identity match | - -Each repository-free fixture permitted at most one attempted Read of -`/etc/os-release`, with a six-turn/$1 model budget and 1,800-second worker and -durable execution ceilings. No publication or notification was requested. - -### Lost registration reply - -The actual DynamoDB handle-registration write committed at 17:48:07.652 UTC, -request `I5PMH0TGHU49H2U0GSBL82FGQFVV4KQNSO5AEMVJF66Q9ASUAAJG`. -The wrapper then threw a timeout instead of delivering the successful reply. -The production code read the saved identity and continued with that worker. - -Only one Run call occurred, receipt `47b3b36e-1da7-423f-b170-ca9c0ec737d0`. -The task completed after one approved Read. Its complete trace contains exactly -that tool call and no dropped events. The coordinator terminated the worker, -released its reservation and removed payloads without observer repair. - -### Cancellation before identity registration - -The normal cancellation API returned HTTP 200 at 17:48:37.385 UTC, receipt -`aabf2ae3-feaa-4bd5-9070-3bf5e89e1649`. The subsequent identity write committed -at 17:48:37.406 with the task already `CANCELLED`. - -Saving the known worker ID after cancellation is intentional: it gives cleanup -a computer to stop. It does not restore the task to running. The coordinator -observed cancellation and terminated the same worker. No tool call or result -occurred. Reservation release and payload deletion completed automatically. - -### Worker ID absent from coordinator state - -The first Run response was held until the owned guest reached its approval -wait, then discarded. This ensures a live worker exists for the intended -recovery check. The wrapper discarded the retry's response too; neither -returned worker ID nor endpoint was written to the task, wrapper steps or -coordinator logs before operator recovery. - -| Time UTC | Evidence | -|---|---| -| 17:48:42.835 | First Run request uses the task ID as its stable token | -| 17:48:44.867 | Guest logs the accepted `/run` with both task ID and worker ID | -| 17:48:50.039 | Owned task observed waiting for approval, without a saved worker ID | -| 17:48:50.045 | First accepted reply discarded; receipt `60a86e07-2eb9-4fdb-ad66-92f04a4a6529` | -| 17:48:50.258 | Retry reply discarded; receipt `b3d0ab9c-f3e3-4210-b65a-a6c36a0e9433` | -| 17:48:51.604 | Observer reads failed task and completed durable execution | -| 17:48:55.814 | Operator recovers the exact worker from the guest log and verifies it is `RUNNING` | -| 17:48:55.983 | Operator requests termination; receipt `a8f66ec1-d23c-4ae3-b085-ca2bb5a08b94` | -| 17:48:57.662 | Worker observed `TERMINATED` | - -Both Run attempts used the same token. The task and normal task API reported -`MICROVM_START_OUTCOME_UNKNOWN`. A snapshot taken before operator recovery -contains no session ID, saved start handle or compute worker ID. The pending -Read never produced a result. - -The coordinator released capacity and removed payloads. The operator explicitly -terminated the recovered computer; this case does **not** claim automatic -cleanup of an unknown ID. The final inventory contained exactly the three new -workers and no active worker. This check exhausted the normal two start attempts -within seconds; it does not retest the separate 120-second receipt cutoff. - -## Operator procedure when the start result is uncertain - -1. Read the task consistently and inspect its saved start receipt and compute - metadata. If an ID is already saved, use that identity and normal cleanup. - Preserve the original task ID, client token, timestamp and error evidence. -2. If the ID is absent, inspect the configured MicroVM guest log group in the - original region and launch window. In this deployment it is - `/aws/lambda-microvms/backgroundagent-dev-abca-agent`. Require an explicit - accepted `/run` entry containing **both** the exact task ID and worker ID. - A nearby timestamp or membership in the same shared image is insufficient. -3. Require one unambiguous identity. Read `GetMicrovm` and check its image - ARN/version, execution role, start time and current state against the original - launch. Confirm the task is terminal or has been canceled before stopping - a worker that may still be active. -4. Terminate that exact worker, verify its terminal state, and verify task - reservation/counter and payload cleanup. Retain the operation receipt and - identity evidence privately. Do not start another worker to mask the unknown - result. - -This procedure requires retained guest identity logs. A bootstrap failure before -that log entry, missing logs or multiple candidates requires further -investigation; do not guess which shared worker to terminate. The -[service feedback F07](./645-lambda-microvm-service-feedback.md) still asks for a -supported token/receipt-to-worker mapping and guaranteed token-retention behavior. - -## Evidence and cleanup - -The private function was `backgroundagent-dev-p3-registration-20260916`, -version 2, code SHA-256 -`khdHcavtr07wQQDqzWowtMNo/q5XZak+2JodEMhmtXk=`. -Unused version 1 and its original artifacts were retained in the evidence; no -task used it. Version 2's wrapper bundle SHA-256 is -`cc058531418d339cb295817b153353ab0c8e68b7a9ee231def58c533f671d5e6`. - -Evidence under `/tmp/abca-645-p2-clean-20260913/p3-registration-20260916` -includes durable histories, all task events/approvals, complete diagnostic -windows, the pre-recovery task snapshot, exact guest identity entry, inventory, -trace and guarded cleanup ledger. All worker instances are terminated; the -cleanup ledger records removal and absence checks for the private function -versions, role, parameter, log group and zero counters. diff --git a/docs/verification/645-p3-repository-path-20260917.md b/docs/verification/645-p3-repository-path-20260917.md deleted file mode 100644 index d71b84897..000000000 --- a/docs/verification/645-p3-repository-path-20260917.md +++ /dev/null @@ -1,136 +0,0 @@ -# ADR-021 P3: normal repository workflow across sleep and wake - -Verified September 17, 2026, in account ``, `us-west-2`. This -extends the [final-image matrix](./645-p3-final-image-and-ecs-20260917.md) -from an explicit temporary clone to the platform's normal repository setup -and existing-PR resolution path. - -## Test boundary - -The test used normal image **6.0**, its original server as PID 1, and **8,192 -MiB**. The packaged `coding/pr-review-v1` workflow cloned -`isadeks/vercel-abca-linear` and checked out existing -[PR #584](https://github.com/isadeks/vercel-abca-linear/pull/584). - -A private coordinator ran the production Durable handler. Its temporary -configuration limited the worker to 900 seconds, the task to five turns/$1, -and the approval to 300 seconds. Automatic sleep was enabled only for this -fixture, with a 30-second delay. The shared deployment's switches stayed off. - -The test deliberately excluded publication: - -- A private event table had no DynamoDB stream or notification consumer. -- Private copies of the deployed worker/session roles substituted the session - role and event-table ARNs. The session role trusted only the private worker - role. Normal deployed roles were unchanged. -- The supported `system_prompt_overrides` setting restricted the agent to one - README read and a final response. Every tool call required approval; the - watcher approved only that exact Read. -- Normal post-hooks used the workflow's `resolve` strategy, which looks up the - existing PR without pushing. Private build/lint settings selected - `npm ci --no-audit --no-fund && npm test` and `npm run lint`; repository - configuration was unchanged. - -The normal approval handler supplied the decision using a synthetic trusted -user context. This verifies that handler and its AWS wake call, not API Gateway -authentication. Its approval/wake audit events used the normal event table; -the agent/coordinator progress and terminal events used the private table. - -This isolation matters: the GitHub notification handler posts for a task with -a repository and PR number even when its source is `api`. Its entry point -currently ignores per-task notification overrides. Separately, the PR-review -system prompt tells the agent to post reviews and comments through Bash. -Neither the `read_only` flag nor `ensure_pr(strategy: resolve)` alone disables -those publication paths. - -## Successful run - -Task: `01M2PCJYY9Z0DK7Z0F4R4QFX0Y`. -Worker: `microvm-4e50cefb-66c7-34d5-8e28-32f1766c0c68`. -Private coordinator version: **3**, code SHA-256: -`DipaZN3h7+L0uJgyWSD9KIhhjX4XvVIshvxVItsvwnM=`. - -| Event | UTC time | -|---|---| -| Task created | 00:36:00.258 | -| Worker observed running | 00:36:10.401 | -| Original approval created | 00:37:07 | -| Worker observed suspended | 00:37:38.708 | -| Normal approval API returned HTTP 202 | 00:37:51.476 | -| Worker observed running after wake | 00:37:52.616 | -| Task observed completed | 00:38:13.514 | -| Finalization verified without repair | 00:38:17.812 | - -Approval `01M2PCQE38M9NB9ZJ3NXYP6TNP` retained its original 300-second -deadline, **00:42:07**. The actual normal-handler `ResumeMicrovm` request ID -was `96e85ab7-0936-499d-91f5-399bca675488`. PID 1 completed suspend and resume -with HTTP 200, in 91 ms and 141 ms respectively. - -The independent audit passed at **00:39:09.507**. It verified: - -- The normal clone/setup path and exactly one successful Read of - `/workspace/01M2PCJYY9Z0DK7Z0F4R4QFX0Y/README.md`. -- The expected `# vercel-abca-linear` heading, passing build/test and lint - results, and resolve-only post-hooks returning the existing PR URL. -- Task `COMPLETED`, Durable execution `SUCCEEDED`, worker `TERMINATED`, - reservation released, counter zero and task payload absent. -- No watcher repair and no dropped trace events. Trace SHA-256: - `537bb8894d8da3c34308456fa8c1753c1a1a0e9a5ae8bbfaa0eae27bceb8d66f`. -- Identical GitHub snapshots before and after: head - `fd509fa63fa089df356574bdd654123c8621b12e`, branch, base, state, update - timestamp, all six comments and zero reviews. - -The reference README was 1,744 bytes with SHA-256 -`dfa6bc9de7dc213fac31c35f6b4f22717f2df1371525eeb2fa195b6dbcc7649c`. -The reference hash identifies the expected remote file; the runtime audit -checks the recorded Read and heading, not a separate guest-side file hash. - -## Excluded attempts and diagnostics - -The initial infrastructure setup encountered IAM propagation: the newly -created worker role was not yet accepted as a trust-policy principal. -The exact error was retained, and a bounded retry for that specific error -completed setup. No worker existed during this failure. - -Private coordinator version 1 was never invoked. Version 2 ran task -`01M2PC1VQ0N3X3KTY0HDC37HX1`, whose verbose test request was rejected during -PR context screening as `CONTENT/PROMPT_ATTACK (MEDIUM)`. It created no -worker and released its reservation before the watcher performed redundant -failure cleanup. Its failed Durable execution remains excluded. - -A read-only hydration comparison with the same PR and the concise request -“Read the README and report its first heading” passed the unchanged filter. -The fresh version 3 task used that wording and passed its own normal screening. -This suggests the extra test instructions contributed to the rejection; it -does not identify an exact offending span or establish deterministic classifier -behavior. No guardrail setting was disabled or weakened. - -## Evidence and remaining scope - -Raw scripts, exact bundles for all three versions, role snapshots, events, -logs, traces and GitHub comparisons are retained under -`/tmp/abca-645-p2-clean-20260913/p3-repository-path-20260917`. - -Cleanup completed at **00:40:40.868 UTC**. The private coordinator and all three -versions, three roles, switch, log group and event table were removed, along -with both owned zero counters. The archive retains 91 function log events and -24 private task events. Subsequent comparisons confirmed the normal worker -and session-role policies and repository configuration were unchanged. - -The permanent private archive is -`/Users/sphias/.local/share/abca-verification/645-p3-20260916/repository-path-evidence.tar.gz`. -It contains 60 files, is 43,934,346 bytes, and has mode `0600`. Every member -was verified against its manifest. SHA-256: -`75850ffb172f3ed5c15198dc0da1ee8c194a87a0edeb89b317ddb0b025c20a11`. - -The accompanying source review corrected comment-only claims in -`agent/src/models.py` and the default/PR-review workflow YAML files: -SDK `allowed_tools` controls auto-approval, and `requires_repo: false` makes -the repository optional. Neither setting is the stronger restriction that -the old comments described. No executable Python or workflow values changed. - -This verifies normal cloning and resolve-only delivery after wake. It does not -verify creating/pushing a new PR or publishing review comments on image 6.0. -Earlier P2 publication results retain their recorded scope. Remaining -permission/network, service-contract and deployment gates are tracked in the -[implementation plan](./645-p3-implementation-plan.md). diff --git a/docs/verification/645-p3-resume-refusal-investigation.md b/docs/verification/645-p3-resume-refusal-investigation.md deleted file mode 100644 index a3ef7cab9..000000000 --- a/docs/verification/645-p3-resume-refusal-investigation.md +++ /dev/null @@ -1,225 +0,0 @@ -# ADR-021 P3: intermittent resume-hook connection refusal - -**Latest September 16 follow-up:** the -[HTTP connection investigation](./645-p3-wake-transport-20260916.md#exact-refusal-with-connection-and-listener-evidence) -captured a sixth exact refusal. The old connection's five-second idle timer -expired at restoration; PID 1 still owned its listener, and no fresh HTTP -connection or resume request appeared. Three explicit `Connection: close` -candidate cases passed on a private image, proving closure before freeze and -fresh resume connections. The subsequent -[normal rollout](./645-p3-connection-close-rollout-20260916.md) deployed image 6.0 -and coordinator 10; approval, denial, timeout and cancellation while asleep -passed on the normal image using a private real Durable coordinator. -Automatic suspension remains disabled while the remaining P3 gates are open. - -An earlier -[diagnostic retaining the original PID 1 server](./645-p3-pid1-observer-20260916.md) -captured a separate wake failure with the generic reason -`Resume lifecycle hook failed.` An independent child observed PID 1 owning its -listening socket after restoration, approximately 519 ms before AWS terminated -the worker. No resume hook entry appeared. This adds process/listener evidence -for that new failure; it does not establish the cause of the five exact -connection refusals below. The service question is tracked separately as -[F08](./645-lambda-microvm-service-feedback.md#f08--generic-wake-hook-failure-while-pid-1-owns-its-listener). - -Updated 2026-09-16 UTC. This is an investigation record and a prepared report; -it has not been submitted to AWS or published as an issue. -The [service-team feedback tracker](./645-lambda-microvm-service-feedback.md) -keeps this blocker alongside related service questions and earlier P2 findings. - -## Observed problem - -An automatically suspended worker sometimes terminates immediately after AWS -acknowledges `ResumeMicrovm`. Its final `stateReason` is: - -> Resume lifecycle hook connection was refused. Please check your hook endpoint -> and application logs for more details. - -The guest previously returned HTTP 200 from `/suspend`. Its retained application -stream has no subsequent `/resume` access log. Application logs alone cannot -distinguish a listener, process, network-restoration or hook-transport failure. -The latest case adds an independent listener sample and exact idle-timer closure; -the first five cases below did not capture those observations. - -The coordinator detects termination, marks the task failed and releases its -capacity reservation. Successful failure cleanup does not make the requested -approval workflow successful. - -The [lifecycle diagnostics guide](./645-p3-lifecycle-diagnostics.md) documents -the new logging and wake-failure feedback. Three isolated AWS workflows verified -the instrumentation with the original server running as PID 1, including actual -API wake and coordinator recovery. None reproduced refusal. The -[normal-stack rollout](./645-p3-diagnostics-rollout-20260916.md) first installed -these diagnostics in image 5.0. The fifth failure on that image includes them, -as recorded below. The current image and coordinator are identified above. - -## Recorded failures - -The first five failures below are on image `backgroundagent-dev-abca-agent`. -All times are UTC, in `us-west-2`. The sixth, on an instrumented private image -derived from 5.0, has its [own detailed timeline](./645-p3-wake-transport-20260916.md#exact-refusal-with-connection-and-listener-evidence). -The earlier cases are detailed in the -[durable verification record](./645-p3-durable-live-20260915.md). - -| Case | Image | Worker | Termination | -|---|---|---|---| -| Approval | `3.0` | `microvm-c1d17307-e488-3149-b2c9-5d7f25f6278e` | Sep 15, 21:31:46.958 | -| Approval after coordinator process-crash recovery | `3.0` | `microvm-9e6bdb3f-bc10-3921-9c50-287bbe20b574` | Sep 15, 21:56:56.279 | -| Supervisor repair after an injected inline Resume failure | `3.0` | `microvm-dca019ee-7d19-389e-b90b-bde129f77329` | Sep 15, 22:54:45.753 | -| Supervisor wake after an owned approval record was deleted | `4.0` | `microvm-2da070a7-43cf-3bb3-9c5d-69bc106f2cb1` | Sep 16, 00:25:07.524 | -| Supervisor wake before the original approval deadline | `5.0` | `microvm-6ff103ab-a41d-348b-8a56-721c9050b623` | Sep 16, 17:10:40.415 | - -### Image 5.0 diagnostic timeline - -Task `01M2NHJ8YSRTRY3XPT06SHT9QD` used the normal image and original server -as PID 1, with a 150-second approval window and a custom 30-second sleep delay. -The private coordinator ran source `a81c565d`; this case did not inject a -command or guest failure. The intended check was an approval timeout winning -before a subsequent approval API call. The unexpected refusal prevented that -check from reaching its decision race. - -| Time on Sep 16 | Evidence | -|---|---| -| 17:09:09 | Read approval created; original deadline 17:11:39 | -| 17:09:39.879 | Supervisor sends Suspend | -| 17:09:39.940 | AWS accepts; request `98a53cab-7737-4675-944e-7a8202781149` | -| 17:09:39.971 | Guest logs suspend hook entry, PID 1 | -| 17:09:40.185 | Checkpoint transaction and controller finish; hook acknowledges HTTP 200 after 214 ms | -| 17:09:40.186 | HTTP access log records `/suspend` 200 | -| 17:09:41.524 | Observer reads `SUSPENDED` | -| 17:10:39.344 | Supervisor sends its sole Resume, about 60 seconds before the original deadline | -| 17:10:39.404 | AWS accepts; request `d32f9929-e7cb-4603-a6c1-0bd2a801f5b5` | -| 17:10:40.415 | Service records termination with the connection-refused reason | -| 17:10:41.716 | Coordinator writes `FAILED` with `MICROVM_RESUME_HOOK_FAILED` | -| 17:10:41.752 | Coordinator releases the task's reservation | - -The fully paginated guest log window contains no resume hook entry, callback -stage, HTTP access record or later application output. Thus the newly -instrumented credential-refresh and identity-reconciliation callbacks did not -leave evidence of starting. This does not establish that the process or -listener survived restoration. The failed harness subsequently called its -idempotent cleanup helpers; the task's recorded failure and reservation release -predate that fallback, but this case is excluded from successful lifecycle -acceptance. - -Four preceding settings/decision cases completed: the default sleep request -occurred after 600.385 seconds, an explicit off setting never requested sleep, -a custom delay requested sleep after 30.395 seconds, and an approval committed -after the deadline won its conditional decision race. These passing controls -do not explain or discharge this fifth failure. - -### Image 4.0 request timeline - -The image 4.0 case used source `58ed7bbf`'s image, including the explicit approval -callback timeout. Task ID: `01M2KRWNPP4BQCJ80SVG6473Z8`. - -| Time on Sep 16 | Evidence | -|---|---| -| 00:24:31 | Read approval created with its original 600-second window | -| 00:25:01.476 | Supervisor sends Suspend | -| 00:25:01.534 | Suspend acknowledged; request `e2199854-5ad0-46c9-bc59-248ac68ead82` | -| 00:25:01.657 | Guest `/suspend` returns HTTP 200 | -| 00:25:04.601 | Independent observer sees `SUSPENDED` | -| 00:25:04.767 | Fixture conditionally deletes only its own still-pending approval | -| 00:25:06.696 | Supervisor sends the sole Resume request | -| 00:25:06.766 | Resume acknowledged; request `0acb3fec-8b18-4bed-b457-0271a15ffdd6` | -| 00:25:07.524 | AWS records termination with the connection-refused reason | -| 00:25:11.935 | Coordinator releases the task's reservation | - -No decision API was invoked in this case. No Read result was produced. The -coordinator left an empty payload prefix and counter zero without independent -cleanup. The intended missing-approval check still failed acceptance because -the guest's expected HTTP 409/503 response was not observed. - -A single fresh control, task `01M2KSTSR1V31NTPQ0CKCJJJMN`, worker -`microvm-a093863f-edf0-39f4-9d79-6893ff08042c`, did reach `/resume`, which returned -HTTP 409 at 00:30:46.418Z after the same owned-approval deletion. AWS reported -the HTTP status explicitly. No tool ran and cleanup passed. Thus the missing -record can produce the expected application error, distinct from the refusal. - -## What the comparison establishes - -Six fresh approval tasks used identical production coordinator code, five-second -polling, a 600-second gate and one read-only action: three on image `3.0`, then -three on `4.0`. All six completed automatic suspension, HTTP 200 resume, exactly -one approved Read and untouched coordinator cleanup. - -The image `4.0` refusal above happened afterward. Therefore: - -- The failure is intermittent in the observed runs. -- The explicit callback-timeout fix does not eliminate it. -- Overlapping API/supervisor Resume requests are not required. Both the - process-crash case and the supervisor-only cases exclude that explanation. -- The observed failure is separate from the old ten-minute Claude callback - cancellation: the image 4.0 case failed less than a minute after its gate - was created; the image 5.0 case failed before its original deadline. -- These runs do not establish a failure rate or identify the responsible - component. - -The [guest lifecycle HTTP handler](../../agent/src/microvm_http.py) and -[pause controller](../../agent/src/microvm_lifecycle.py) were reviewed. -Suspend drains tracked work and checkpoints; it does not intentionally close -the HTTP listener. The server's shutdown path is separate. That source review -does not exclude a process crash or a lower-level restore problem. - -## Earlier experiments and remaining investigation - -The [minimal listener experiment](./645-p3-listener-probe-20260916.md) completed -four full normal cases and recorded 14 guest resume acknowledgments without an -unexpected refusal. Request timeouts interrupted three cases. A fresh deliberate -closed-listener control produced the exact refusal reason while an independent -observer still ran in the guest. This calibrates the diagnostics; it does not -establish why the full agent loses its listener or connection. - -That experiment also exposed a separate [pending-wake timer bug](./645-p3-pending-wake.md) -in the supervisor. Its correction does not account for these failures, -whose worker termination reason was the service-reported connection refusal. - -The [full-agent observer experiment](./645-p3-process-observer-20260916.md) -adds the independent parent and listener sampling described below. Its completed -fallback and direct API wakes retained a healthy child-owned listener, without -reproducing the refusal. Short full-agent cases reused the same client port for -`/suspend` and `/resume`, unlike the minimal listener's closed connections. -That experiment established a transport difference. None of the first five -recorded failures has independent process/listener evidence at restore; the -sixth now does. The guest does not expose the cgroup OOM counters sampled by -the observer. - -The subsequent [local transport control](./645-p3-transport-control-20260916.md) -reproduced a reset on an old connection after a six-second process pause, while -all eight fresh-connection checks succeeded and the servers remained alive. -This was not a reproduction of the AWS refusal. It supplied a specific comparison -for the later instrumented cloud investigation. - -The earlier [AWS connection comparison](./645-p3-diagnostics-rollout-20260916.md) -tested normal image 5.0 against a private image with -`--timeout-keep-alive 0`, keeping the original server as PID 1. Both quick and longer wakes -passed, and both missing-approval controls produced the expected guest HTTP 409 -with a failed identity-read stage. The longer normal wake used a fresh client -port; quick normal wakes reused one. No refusal was reproduced. These results -alone did not identify F01's cause. The later instrumented refusal and explicit -response-header comparison supply the additional evidence for the deployed fix. - -1. Use the recorded worker IDs, region, timestamps and Resume request IDs to - inspect service-side lifecycle diagnostics. Determine the actual connection - error and whether the request reached the guest, including any transport retry. -2. Correlate guest process/kernel health and the port-8080 listener at restoration. - Application access logs do not provide that missing evidence. - The latest failure now supplies these observations with the original PID 1. - Compare its expired connection against the successful fresh-connection wake. -3. Retain the verified private-candidate socket closure and normal-image - acceptance evidence. Do not hide a failed wake by silently - launching another worker: the approved action and workspace may already have - changed. -4. Approval, denial, the original timeout winning over a late approval, and - cancellation while asleep now pass on image 6.0. Complete the wider final-image - checks, including multiple-gate workspace persistence, the late-decision - winner and actual expired-credential renewal. Earlier-image results remain - evidence for their recorded scope. -5. Leave production automatic suspension off until this gate and the - [remaining acceptance plan](./645-p3-implementation-plan.md) are satisfied. - -The private evidence archive contains full guest streams, redacted control -request telemetry, durable history and task events. It contains no saved AWS -credential values in the report. Temporary verification resources are tracked -separately in the [callback live record](./645-p3-callback-live-20260915.md). diff --git a/docs/verification/645-p3-session-recovery-20260917.md b/docs/verification/645-p3-session-recovery-20260917.md deleted file mode 100644 index ab71972a7..000000000 --- a/docs/verification/645-p3-session-recovery-20260917.md +++ /dev/null @@ -1,179 +0,0 @@ -# ADR-021 pending approval session recovery - -September 17 diagnostics verified that the pinned agent SDK can resume a saved -conversation in a new process after its original process dies while waiting in -a tool hook. It does **not** resume that pending hook. The implemented SDK -conversation store now retains the exact proposed action alongside the session; -approve and deny passed after deleting the original configuration. Nine live -S3 checks also passed, including task isolation and reads pinned to an immutable -object version. These are prerequisites for retaining unanswered approvals -beyond a worker's lifetime; production continuation remains unimplemented. - -## Initial recovery diagnostic - -The probe used Python `claude-agent-sdk` **0.2.110** and its bundled Claude Code -**2.1.191**, matching the reviewed application versions. The model endpoint was -a loopback HTTP server returning deterministic responses. AWS credentials were -synthetic, configuration directories were isolated, and the only available tool -was `Read` against an owned temporary marker file. No model service, repository -or notification channel was contacted. - -1. The first process received a model response proposing `Read`, with tool ID - `toolu_original`. Its `PreToolUse` callback remained blocked. -2. The probe waited until the assistant tool call appeared in the saved session - file, then copied the session directory and workspace. -3. It killed the original process group, including the bundled CLI. No - `PostToolUse` callback occurred in that process. -4. It restored the workspace at the same path and started a separate process - with the copied configuration and `ClaudeAgentOptions.resume` set to the - original session ID. -5. A new user turn told the agent that a decision had arrived. The simulated model - proposed a new `Read`, ID `toolu_restored`. That call passed through a new - permission hook, read the restored marker once, and completed in the same - conversation session. - -## What recovery preserved - -The restored model request contained the original user prompt and the new -continuation prompt. The original pending tool call was absent; its assistant -turn was represented as `No response requested.` The new tool call had a new ID. -The session ID remained `bb94901c-a91f-4854-8e63-edbc7a476f91`. - -Therefore, passing `resume=` is not sufficient to consume the saved -approval or recover its exact proposed action. The application must retain that -request identity, action and decision separately and make them available to the -agent after restoration. Newly proposed tool calls still pass through the normal -authorization hooks. The recorded approval must not become blanket permission -for a different action. - -This requirement does not add a separate relevance or staleness checker. The -agent assesses relevance through its ordinary reasoning, as requested. - -## Supported conversation checkpoint implementation - -The initial configuration-copy diagnostic established the SDK behavior. The new -`agent/src/continuation_session.py` uses the SDK's supported `SessionStore` -contract instead. It never reads or copies CLI configuration or authentication -files. - -- `CheckpointSessionStore.append()` preserves opaque SDK journal entries, updates - existing UUID entries in place and retains entries without UUIDs. It rejects - mixed sessions and subagent transcripts; detached/subagent work is outside the - current runner's supported waiting state. -- With `session_store_flush="eager"`, the SDK still mirrors asynchronously. - `checkpoint_pending()` waits for the exact assistant tool ID, name and full - input to appear. A missing batch, mismatch or existing tool result prevents - acknowledgement. A rejected batch poisons the buffer, so later data cannot - conceal the possible gap. -- The versioned envelope contains task, attempt, request, user and repository - identity; SDK project/session identity; full pending action and its approval-row - compatible hash; and transcript entries. It is bounded to 16 MiB and 50,000 - entries. Extra top-level configuration or environment fields are rejected. -- `S3ContinuationCheckpoints.save()` conditionally creates a SHA-256-addressed - object with encryption and checksum. It returns a receipt only after checking - the exact bytes and a real S3 version. A lost write reply or repeated save can - recover through read-back; failure to prove persistence never reports success. -- `load()` reads the receipt's exact version, validates its checksum, size, - envelope and identity, then supplies the journal to a new store for SDK - materialization. Later changes to the current object cannot change that - version's saved data. - -Storage uses `continuations////.json`. -The default AWS client requires an active task-scoped session before using the -attributed factory. This explicit check matters because `tenant_client()` alone -can fall back to ambient credentials. The future deployment needs a private, -versioned bucket and `PutObject`, `GetObject`, `GetObjectVersion` permission only -under the task's prefix. Existing artifact-write permissions are insufficient. -This component does not configure retention or delete checkpoints. - -Transcript and action contents remain private task data and may themselves -contain sensitive text. Excluding authentication files does not mean the -conversation is safe to log or publish. - -## Real SDK approve and deny checks - -`agent/tests/test_continuation_sdk_probe.py` uses the same SDK **0.2.110** and CLI -**2.1.191** with synthetic AWS credentials and a deterministic loopback model: - -1. The original `Read` hook waits for the new store to acknowledge the exact - proposed action and writes that checkpoint. -2. The test kills the entire original process group, verifies that its tool did - not execute, and deletes the original configuration directory. -3. The initial store test copied an owned marker as its workspace fixture. The - [workspace follow-up](./645-p3-workspace-recovery-20260917.md) now deletes the - original workspace and restores its Git history, tracked marker and untracked - file using the new archive component. A fresh process starts at the same - path with `resume`, the restored store and eager mirroring. -4. The continuation prompt carries the full saved action and human decision. - The model proposes `toolu_restored`; the fresh permission hook allows one - `Read` for approval and denies it for denial. Both sessions complete with - their original session ID. -5. A synthetic authentication-file sentinel is absent from the saved checkpoint. - No original CLI configuration is used during restoration. - -The initial conversation-component full agent quality run enabled both tests: - -```bash -ABCA_TEST_SDK_CONTINUATION=1 MISE_EXPERIMENTAL=1 mise run //agent:quality -``` - -Lint, formatting and type checks passed; **2,004 tests passed, 11 skipped**, with -86.57% total coverage. This includes **45 checkpoint unit tests** and both real SDK -cases. The skipped cases are the existing opt-in DynamoDB Local checks. One -dependency warning concerns Starlette's deprecated `httpx` test-client support. -Unit failure injection covers dropped mirror data, corrupt or cross-task -checkpoints, cancelled waits, ambiguous write replies and unverifiable storage. -The agent Bandit high-severity check and documentation build passed. The full -repository silent-success scan reported 92 findings in 50 files unchanged from -the pre-change commit `95fd6746`; none were in this component. The corresponding -change-only scan against that commit passed. - -## Live S3 verification and cleanup - -The isolated AWS probe used the development account, region `us-west-2`, private -versioned bucket `abca-645-checkpoint--20260917` and role -`abca-645-checkpoint-probe-20260917`. The role allowed only -`PutObject`, `GetObject` and `GetObjectVersion` under -`continuations/${aws:PrincipalTag/task_id}/*`, with task/user/repository session -tags. It used a real checkpoint produced by the SDK test above. - -| Check | Result | -| --- | --- | -| Save, read and repeat the same save | Same verified version receipt | -| Object encryption and SHA-256 checksum | AES256 and matching checksum | -| Overwrite current object, then load saved receipt | Original version and bytes restored | -| Another task reads current object | Access denied | -| Another task reads pinned version | Access denied | -| Another task writes this task's prefix | Access denied | -| Owning task lists the bucket | Access denied | -| Owning task deletes current object | Access denied | -| Owning task deletes pinned version | Access denied | - -All nine checks passed. Cleanup verified ownership tags, removed both exact -object versions, the bucket, inline policy and role, and confirmed bucket `404` -and role `NoSuchEntity`. No normal deployment resource or setting changed. - -## Limits and next steps - -These tests prove local SDK conversation recovery after abrupt process loss, -explicit transfer of the action/decision, and the S3 storage contract under real -task-scoped permissions. The deterministic model demonstrates transport and hook -behavior; it does not prove how a real model will interpret the continuation -prompt. - -Production `runner.py` does not yet install this store or resume from it. The -workspace follow-up implements local file/Git preservation; its durable storage -and integration remain required. Other remaining work includes recovery on a -replacement cloud worker; enforcing the stable workspace path; lifecycle barrier and -conditional publication of the acknowledged receipt; task-attempt fencing and -reservation transfer; exactly-once decision consumption; cancellation while -parked; lost launch replies; and request retention/deadline changes. A conversation -receipt alone never permits worker release. Those requirements remain in the -[unanswered-approval implementation order](./645-p3-implementation-plan.md#unanswered-approvals-implementation-order). - -The initial executable probe, model requests, hook audits and synthetic session -files are archived with a SHA-256 manifest under -`/Users/sphias/.local/share/abca-verification/645-p3-20260916/session-recovery`. -The new implementation snapshot, SDK fixtures, test logs and live S3 policy, -verification and cleanup receipts are archived separately under -`/Users/sphias/.local/share/abca-verification/645-p3-20260916/continuation-checkpoint`. diff --git a/docs/verification/645-p3-supervisor.md b/docs/verification/645-p3-supervisor.md deleted file mode 100644 index 507aaf83b..000000000 --- a/docs/verification/645-p3-supervisor.md +++ /dev/null @@ -1,233 +0,0 @@ -# ADR-021 P3: supervisor and approval wake - -Status (2026-09-16): production integration and full repository validation pass. -The development deployment now runs coordinator version `5`, including the -[pending-wake correction](./645-p3-pending-wake.md), and agent image `4.0`, -including the [approval callback correction](./645-p3-callback-timeout.md). -Automatic suspension remains off. The [P3 plan](./645-p3-implementation-plan.md) -retains the remaining live acceptance gates and P2 checks. - -## What this does - -The supervisor is the part of ABCA that watches the rented computer. It now -remembers both its instructions and its deadlines when the supervising Lambda -restarts. “AWS accepted my wake request” and “the agent continued its work” are -different observations. - -For a long approval wait, the supervisor checks the live suspension setting, -saves an instruction to sleep, rereads the setting, checks -the same approval again, requests suspension, and checks again afterward. The -guest's checkpoint hook must complete before AWS freezes it. New sleep requires -verified support from the worker's actual image version. - -An approval can arrive while suspension is still happening. Both the decision -API and supervisor save “wake” immediately. They request Resume only after AWS -reports SUSPENDED. That instruction remains even after an acknowledgment or a -RUNNING observation, so a delayed suspension cannot silently strand the worker. -When the task leaves its gate, the supervisor can re-scope wake to the new -no-gate identity before ending recovery on the following observation. - -AWS RUNNING alone cannot hide an agent stuck on an already decided or expired -gate. Recovery remains bounded until the agent moves forward. An intentional -early wake may remain AWAITING_APPROVAL while its original decision is pending. -An ordinary RUNNING worker still gets the shared heartbeat checks. - -## Bounds and state - -These are application choices requiring live timing verification, except the -configured eight-hour service maximum. - -| Bound | Value | -|---|---| -| Entire supervisor cycle | 45 seconds | -| Individual control request, including Get/Terminate | At most 10 seconds, shortened by caller deadline | -| Lifecycle database read sequence/write | At most 5 seconds, shortened by caller deadline | -| Live suspension setting read | At most 3 seconds, shortened by caller deadline; failure disables new sleep | -| Transition poll | At most 5 seconds | -| Suspend/wake/unknown recovery | 120 seconds; repeated polls do not reset it | -| Startup/HYDRATING recovery | 300 seconds | -| Consecutive failed cycles or Resume requests | 3; permanent control denials can fail earlier | -| Session deadline | Earlier saved deadline or AWS start time + service duration, capped at 28,800 seconds | -| API work after a committed decision | At most 8 seconds, retaining 1 second of remaining Lambda time for the response | -| Each API audit attempt | At most 2 seconds inside that shared budget | -| Final cleanup | At most 25 seconds, with at most two termination attempts | - -The durable poll state contains only JSON values: VM identity, original -observation/deadline, failure counters, recovery kind/start time, anomaly state -and next delay. Before verified service timing is available, a fixed deadline -from the first durable observation bounds supervision and new sleep is disabled. -Faster polls do not consume the older 1,020-attempt limit used by other backends. - -AWS can report `PENDING` while restoring a suspended worker. A matching saved -wake instruction starts wake recovery at that instruction's original request -time. The startup follow-up also preserves the startup clock until an actual AWS -observation confirms startup, because the coordinator can mark a task `RUNNING` -earlier. Another unexpected pending observation gets bounded uncertainty -recovery. Existing recovery clocks are preserved. Both corrections are deployed; -the exact API-issued pending-wake branch and initial-startup clock have live -evidence in the linked verification record. - -Control calls and database recovery reads share the caller's AbortSignal, a -cancellation notice. An expired operation cannot start another request with a -fresh independent budget. - -## Decisions, failures and cleanup - -Approve/deny keep the existing authorization and conditional transaction. -Optional wake runs only after commit. A failed read, Resume or audit cannot undo -the decision or turn the accepted response into HTTP 500. The guest still owns -approval expiry; this change adds no independent API expiry rule. - -Before a supervisor failure becomes a task outcome, finalization strongly reads -the task and preserves a committed completion/cancellation. Conditional failure -writes also check the original worker ID. A replacement worker is not followed. -Normal capacity release remains task-owned and happens once. - -Infrastructure exhaustion during AWAITING_APPROVAL/HYDRATING uses FAILED, since -those statuses do not allow TIMED_OUT. RUNNING/FINALIZING session expiry uses -TIMED_OUT. The approval row's decision is not rewritten. This also fixes the old -approval-wait finalizer's forbidden AWAITING_APPROVAL → TIMED_OUT transition. - -Termination is attempted even if database finalization fails. Its result -distinguishes requested, not-found and unconfirmed. ConflictException is -unconfirmed: a conflicting lifecycle operation does not prove termination. -An acknowledgment is not reported as observed teardown. - -Diagnostics: - -- `microvm_suspend_anomaly`: once per unexpected suspension episode; a new - episode can be reported after recovery. -- `microvm_supervisor_request_failed` and command-failure events: safe error - identifiers and the stage that failed. -- `microvm_resume_orphan`: an inline wake failed or became ineligible; includes - task/gate, VM when known, stage, reason and validated AWS request ID when - available. A recoverable race is not proof of a permanent orphan. -- `microvm_cleanup_unconfirmed`: bounded cleanup could not establish even a - successful termination request; retain the original handle for recovery. - -Audit writes are best-effort within the existing budget. Structured logs remain -the fallback when that budget is exhausted or the event store fails. No SDK -message, credential or signed payload URL is copied into these diagnostics. - -## Deployment configuration and permissions - -`microvm_approval_suspend_enabled` accepts true/false and defaults false. -It sets both `MICROVM_APPROVAL_SUSPEND_ENABLED` on a configured coordinator and -the String parameter `//microvm-approval-suspend-enabled`. -The stable parameter name is passed as `MICROVM_APPROVAL_SUSPEND_PARAMETER_NAME`. -Both must allow sleep. Compatible per-worker image evidence remains an -independent admission check. - -Task submission accepts `microvm_sleep_after_s`: an integer from 0 to 3600, -default 600 seconds. Zero disables new sleep for that task. The CLI exposes -`--microvm-sleep-after `. Creation stores the resolved default, -and task details return it. Legacy task rows without the setting use 600; -malformed stored values prohibit new sleep and permit wake/cleanup. -Fresh observations recheck the preference before Suspend. Supervisor logs -include the validated delay and policy reason, without echoing malformed input. -The worker role cannot modify this coordinator-owned setting. - -The delay is measured from each original approval creation time. The -60-second pre-deadline wake margin and 30-second minimum available sleep -window remain safety bounds, not financial break-even promises. Default -five-minute approvals stay awake with a ten-minute sleep delay. The choice -does not change approval deadlines, the eight-hour worker lifetime, or the -global enablement requirements. - -Durable executions retain their original Lambda version and environment. -An environment-only redeploy therefore cannot disable an existing execution. -The stack retains published coordinator versions and their immutable guardrail -versions across updates. Otherwise, a rollout could delete the code or guardrail -that an older execution still references. The live alias advances to the new -coordinator version. Retained versions require operator cleanup only after no -execution can resume or retry them; changing the live alias is not that proof. -For the first upgrade from an unretained deployment, drain existing executions -before removing the old versions, or retain those existing resources in a -separate update before replacing them. - -The supervisor rereads Parameter Store without caching before saving new -suspend intent and again before the pre-command gate read. Missing, invalid or -unavailable values disable new sleep without counting as compute failure. -Disable after intent commit records compensating wake and skips Suspend. -Waking, timeout and cleanup never depend on reading this setting. - -| Role | Added permissions | -|---|---| -| Coordinator | SuspendMicrovm/ResumeMicrovm on the configured image ARN and its version suffix | -| Coordinator | GetItem/ConditionCheckItem on the existing approvals table | -| Coordinator | GetParameter on this deployment's exact suspension parameter | -| Approve/deny API | GetMicrovm/ResumeMicrovm on that same image scope | -| CloudFormation execution role | Parameter lifecycle/tagging only under `backgroundagent-*/microvm-approval-suspend-enabled`, in the conditional MicroVM bootstrap policy | - -Cancel retains its Terminate-only grant. Worker roles gain no lifecycle actions -or approval-control authority. No token minting or ingress permissions are added. -The table supplied in `microvmConfig` is the source for both the coordinator's -approval-table environment variable and its read/condition-check grant. - -Refresh bootstrap to **1.8.0** before deploying the parameter. The old -`/cdk-bootstrap/*` SSM grant does not cover application settings. -Deploy the matching coordinator, API roles and six-hook image with the flag -false first. Verify the actual image version/capability and effective role -permissions. Then enable it for controlled development acceptance cases. - -For operational rollback, set the live parameter to `false` and verify its -stored value. Also deploy context `false` so the declared configuration agrees. -A direct Parameter Store update is immediate configuration drift until the -declaration is reconciled. CloudFormation need not rewrite a parameter whose -declared value has not changed, so always verify the live value. A suspension -already dispatched can still finish; existing sleepers retain normal wake and -deadline recovery. Do not remove Resume/Terminate permissions while workers can -still be suspended. A function version originally deployed with static false -stays opted out even if the live parameter is later enabled. - -## Validation record - -The initial combined run passed 673 tests across 15 suites, including 41 real -DynamoDB Local cases. Two new database cases exercise approval during Suspend -and approval just after the intent transaction. They use the real supervisor, -store and conditional expressions, with mocked compute only. - -The configured full-stack fixture initially passed 17 MicroVM checks. After the -live switch was added, 340 tests across 13 suites passed, covering uncached reads, -disable after intent commit, real handler/config composition, exact IAM and the -bootstrap policy/golden/size checks. The final root run includes all stack fixtures. - -Full root `mise run build` passed, exit 0 in 640.39 seconds: - -- CDK: 230 suites, 5,028 tests and one snapshot. -- Agent: 1,951 Python tests. -- CLI: 62 suites and 928 tests; Forge: 11 tests. -- Compilation, lint, formatting, types/drift checks, bundled synthesis, - documentation build and links all pass. -- DynamoDB Local was enabled: 41 lifecycle and 15 capacity cases ran. - -The full run caught two old start-recovery assertions omitting the newly bounded -Terminate request's AbortSignal; they now assert it. A subsequent comment lint -failure was corrected before the final passing run. Final evidence: -`p3-supervisor-root-build-r4-20260915.log`. - -The deployment change-set review then exposed missing retention for published -coordinator and guardrail versions. The fix passed 192 focused infrastructure -tests and another complete root build in 657.59 seconds: 5,030 CDK tests, -1,951 Python tests, 928 CLI tests and all other configured checks. DynamoDB Local -remained enabled. Evidence: `p3-retention-root-build-20260915.log`. - -Local evidence does not establish AWS timing, IAM effectiveness, actual frozen -credential renewal or deployed durable replay. - -Evidence directory: `/tmp/abca-645-p2-clean-20260913/`. - -## Relevant code - -- `cdk/src/handlers/shared/microvm-supervisor.ts`: durable reconciliation and - bounded cleanup diagnostics. -- `cdk/src/handlers/shared/microvm-approval-wake.ts`: bounded post-commit wake. -- `cdk/src/handlers/shared/microvm-lifecycle.ts`: strong snapshots and conditional intent. -- `cdk/src/handlers/shared/agent-heartbeat.ts`: shared liveness thresholds. -- `cdk/src/handlers/shared/microvm-control.ts`: safe diagnostic identifiers. -- `cdk/src/handlers/shared/microvm-suspend-config.ts`: uncached bounded live switch. -- `cdk/src/handlers/orchestrate-task.ts` and `shared/orchestrator.ts`: poll/finalize ownership. -- `cdk/src/handlers/approve-task.ts` and `deny-task.ts`: accepted decisions and - best-effort wake/event callbacks. -- `cdk/src/constructs/task-orchestrator.ts`, `task-api.ts` and `stacks/agent.ts`: - flag, table and scoped IAM wiring. diff --git a/docs/verification/645-p3-transport-control-20260916.md b/docs/verification/645-p3-transport-control-20260916.md deleted file mode 100644 index 7906b3e4e..000000000 --- a/docs/verification/645-p3-transport-control-20260916.md +++ /dev/null @@ -1,87 +0,0 @@ -# ADR-021 P3: paused-server HTTP transport control - -Date: 2026-09-16. Local diagnostic completed; the original AWS resume refusal -was **not reproduced** by this local experiment. It changed no application code -or AWS deployment. Subsequent AWS transport instrumentation is recorded in the -[wake transport investigation](./645-p3-wake-transport-20260916.md). - -## Question - -The full-agent hooks sometimes reuse one HTTP connection across suspension. -Could an old connection fail after a pause even though the server still accepts -new connections? This is a narrower question than the recorded -[AWS resume refusals](./645-p3-resume-refusal-investigation.md). - -The earlier full-agent observer measured a 363.501-second wall-clock gap and -363.507-second monotonic-clock gap across one long suspension, with the child -still owning its listener afterward. Another long case measured 367.621 and -367.620 seconds respectively. Thus elapsed-time timers advanced during those -observed freezes. - -## Isolated experiment - -Eight cases ran in a disposable ARM64 Linux container, with no external network -or AWS credentials. It used Python 3.13.13, Uvicorn 0.50.0, FastAPI 0.139.0, -the asyncio loop and the h11 HTTP implementation. The cached image digest was -`sha256:37a88f276acd700dbd0d6d2a69eff2536180bde11620b6084220414d8d2b63f2`. - -A minimal FastAPI application acknowledged `/suspend` and `/resume`. The -controller stopped the server process with `SIGSTOP` for one or six seconds, -then continued it with `SIGCONT`. These signals pause and restart a process. -They do not perform an AWS MicroVM snapshot or restore its networking. - -Each duration was tested with the wake request queued before continuation or -sent just afterward. The comparison response included `Connection: close`, -which tells an HTTP client to open a new connection for its next request. -Every case also made a separate fresh connection and checked that the server -was still alive. - -| Response behavior | Pause | Request relative to continuation | Wake result | Fresh connection | -|---|---:|---|---|---| -| Default keep-alive | 1 s | After | 200 | 200 | -| Default keep-alive | 1 s | Queued before | 200 | 200 | -| Default keep-alive | 6 s | After | Connection reset, errno 104 | 200 | -| Default keep-alive | 6 s | Queued before | 200 | 200 | -| Connection close | 1 s | After | 200 | 200 | -| Connection close | 1 s | Queued before | 200 | 200 | -| Connection close | 6 s | After | 200 | 200 | -| Connection close | 6 s | Queued before | 200 | 200 | - -All eight server processes remained alive until explicit test cleanup. The -container exited successfully and its absence was verified. - -## Interpretation and next reproduction - -One reused connection reset after exceeding the server's five-second keep-alive -timeout. That is different from a new connection being refused: the listener -remained reachable in every case. This does not establish how AWS classifies -its underlying transport errors. An earlier interpretation measured from the -observer's `SUSPENDED` sample and used an interval shorter than five seconds to -discount idle expiration. That uses the wrong starting point: the server arms -its idle timer after finishing the preceding HTTP response, before that later -state sample. The [historical timing correction](./645-p3-wake-transport-20260916.md#historical-timing-correction) -retains the original timestamps and explains why they neither establish nor -exclude this cause. - -The next bounded AWS comparison should: - -1. Preserve the original server command and its position as the first process - in the guest. The previous parent observer changed that position. -2. Compare the deployed image with a diagnostic variant that changes only - lifecycle response connection handling. Preserve exact artifact hashes. -3. Use fresh owned task IDs for ordinary approval and supervisor-only wake. - Record AWS request IDs, actual worker states, hook entry/response times, - connection identity and available process/listener observations. -4. Stop and preserve evidence on a refusal. Determine whether the process died, - the listening socket disappeared, or a healthy listener was unreachable. -5. Apply a correction only when supported by the failing path, then demonstrate - the same trigger passing and recheck normal approval, denial, deadlines and - credential renewal. A few successful comparison runs alone do not prove a fix. -6. Remove only new owned resources. The deployed image and its existing roles - are comparison inputs and must not be deleted by fixture cleanup. - -This AWS comparison is planned, not executed by this local experiment. -Production automatic suspension remains off. - -Scripts, raw results and the clock comparison are retained privately under -`/Users/sphias/.local/share/abca-verification/645-p3-20260916/transport-control`. diff --git a/docs/verification/645-p3-user-sleep-20260916.md b/docs/verification/645-p3-user-sleep-20260916.md deleted file mode 100644 index 3f8cb7488..000000000 --- a/docs/verification/645-p3-user-sleep-20260916.md +++ /dev/null @@ -1,148 +0,0 @@ -# ADR-021 P3: adjustable approval sleep and live verification - -Verified September 16, 2026, in account ``, `us-west-2`. -Source `a81c565d` is deployed with automatic suspension still disabled. -The settings and five live cases pass. P3 acceptance remains incomplete because -the sixth case reproduced the [resume connection refusal](./645-p3-resume-refusal-investigation.md). - -## User behavior - -`microvm_sleep_after_s` controls how long a MicroVM waits at an approval gate -before requesting sleep. The API accepts whole seconds from 0 through 3600, -persists 600 when omitted, and returns the value in task details. Zero means -stay awake. Existing records without the field use the same 600-second default. - -The CLI exposes `bgagent submit --microvm-sleep-after `. -For example, `--microvm-sleep-after 120` waits two minutes, and -`--microvm-sleep-after off` keeps the worker awake. This setting does not extend -the approval deadline. A five-minute approval window ends before a ten-minute -sleep delay, so that gate stays awake. - -The deployment flag and live Parameter Store switch still control whether -automatic sleep is allowed at all. Approval, denial, timeout and cancellation -continue to trigger the appropriate wake or cleanup behavior. - -## Validation and rollout - -The full build passed 5,050 CDK, 942 CLI and 1,947 Python tests: **7,939 total**. -A separate pinned DynamoDB Local run passed all **56** lifecycle/capacity -transaction tests. Type parity, compilation, lint, documentation generation, -the 77-page documentation build and link checks passed. - -CloudFormation reached `UPDATE_COMPLETE` at **17:21:50.192 UTC**: - -- Normal coordinator alias `live` points to version **8**, code SHA-256 - `lYgbdJRz0bLxPXT3XsW0VGaTqLCM8hB1vVGk9JuTRB0=`. -- Version **7** was retained; CloudFormation recorded `DELETE_SKIPPED`, and - its previous code checksum remains readable. -- All **475** resources retain their physical identity except the expected - coordinator version rotation. Image **5.0** remains `ACTIVE` / `SUCCESSFUL`. -- The coordinator environment is unchanged. Its suspension flag is false; - the normal Parameter Store switch remains false at version 1. -- The coordinator and CreateTask code checksums match the exact reviewed S3 - artifact bytes. - -The exact reviewed template contains 31 Lambda code changes plus the retained -version/alias rotation: 34 changes. `DescribeChangeSet` reported 34 entries with -`IncludePropertyValues=true`, but 87 without it. The additional 53 entries were -unchanged resources with dynamic ARN dependencies. Both their template -properties and final physical identities were checked. The reviewed template -SHA-256 is `4957871b24243f900e16fa9a04b58db63d85b1da5fcf7265d5f403fe5ecd3e3c`. -The exact S3 template bytes, rather than lossy `GetTemplate` text, supplied the -deployment baseline. - -Seven checks exercised the newly deployed normal Lambda API handlers: - -| Request | Verified result | -|---|---| -| Omitted setting | Stored and returned 600 | -| Setting 0 | Stored and returned 0 | -| Setting 120 | Stored and returned 120 | -| -1, 3601, string `"600"`, fraction 0.5 | Each rejected with HTTP 400 and the valid range | - -The three accepted tasks used metadata-only pending attachments. They stayed -`PENDING_UPLOADS`, launched no worker, acquired no reservation, and were canceled -through the normal API. No attachment was uploaded or signed upload instruction -retained. Direct handler invocation with a synthetic trusted identity verifies -deployed validation and persistence; it does not test API Gateway authentication. - -## Six live worker cases - -The private durable coordinator used the production handler, source `a81c565d`, -normal image 5.0 and the original server as PID 1. Each task was repository-free, -limited to one attempted Read of `/etc/os-release`, six model turns, a $1 model -budget and a 1,800-second worker/durable ceiling. Normal suspension stayed off. - -| Case | Task | Result | -|---|---|---| -| Default ten minutes | `01M2NHJ8YCQAM7VPMQFES4N2HJ` | Sleep requested after 600.385 seconds; approval completed | -| Sleep off | `01M2NHJ8YEVH062H60ZX45XKCE` | No Suspend request; approval completed | -| Custom thirty seconds | `01M2NHJ8YF5709VYHPXNMAMC0N` | Sleep requested after 30.395 seconds; approval completed | -| Late approval wins | `01M2NHJ8YQZQ0T1QFTA5B0J8ZB` | Approval committed after the original deadline; the approved Read completed | -| Timeout wins | `01M2NHJ8YSRTRY3XPT06SHT9QD` | **Failed acceptance:** pre-deadline wake ended in connection refusal before the intended timeout race | -| Credential renewal denied | `01M2NHJ8YS7WTCVPWQTABNM5D3` | Expected failure: resume HTTP 503, precise credential-refresh diagnostic, no Read result | - -The four successful coding cases each produced exactly one Read result and a -complete trace with no dropped events. All five passing cases preserved the -original deadline and lifetime observations, terminated their worker, released -capacity and removed payloads without observer repair. - -The late-approval fixture intentionally suppressed coordinator wake delivery -until after the deadline. Those wrapper responses are explicitly marked -synthetic acknowledgments, not AWS receipts. The normal approval API committed -the decision after expiry while the worker was still suspended. This verifies -the existing conditional-write rule: a committed approval can win before the -worker's timeout write. - -For the credential fault, an exact task-tag condition temporarily denied STS -role renewal for this fixture alone. The guest reached `/resume` and logged: - -- `callback_stage=credential-refresh` -- `error_type=ClientError`, `aws_error_code=AccessDenied` -- AWS request ID `149c138b-54e6-4dc8-ba95-1138136d4a14` -- Resume HTTP 503 after 441 ms; no subsequent identity transaction or Read result - -The coordinator surfaced `MICROVM_RESUME_HOOK_FAILED` and cleaned up. The -temporary deny was removed and its absence verified. This application rejection -is distinct from the unexpected case, whose retained guest stream contains no -resume hook entry at all. - -## Capacity and cleanup - -While the default worker waited, a bounded check exercised the normal deployed -ConcurrencyReconciler. An owned counter was deliberately increased from 1 to 3, -representing the real waiting worker, an added terminal task with a held -reservation, and one extra overcount. The reconciler repaired 3 to 2, then released -the terminal reservation to leave 1. The waiting worker and its reservation -were preserved. The auxiliary task was removed without independent counter -repair. Lambda receipt: `8710433a-4624-488d-b123-5cfd0c2c3736`. - -All six worker instances are terminated. Their reservations are released and -payload prefixes empty. The private function and all versions, role, Parameter -Store switch, log group and six zero counters were removed; 2,213 coordinator -log events were saved before deletion. An immediate post-delete function read -still returned a value, so the first absence assertion failed. A subsequent -read-only check confirmed that the function, role, parameter and log group were -absent. Both observations are retained. - -The failed timeout fixture called its fallback cleanup helpers after the -coordinator had already recorded failure and released its reservation. It is -excluded from successful lifecycle acceptance. Cleanup does not turn this -failure into a pass. - -Evidence is retained under -`/tmp/abca-645-p2-clean-20260913/p3-user-sleep-20260916` and -`/tmp/abca-645-p2-clean-20260913/p3-user-sleep-rollout-20260916`. -The [implementation plan](./645-p3-implementation-plan.md) tracks the remaining -acceptance gates. The [service feedback](./645-lambda-microvm-service-feedback.md) -contains the fifth failure's worker ID, timestamps and API receipts; it has not -been submitted externally. - -The durable private archive is -`~/.local/share/abca-verification/645-p3-20260916/sleep-registration-capacity-evidence.tar.gz` -(208 files, 462,751,059 bytes, mode 0600). Its SHA-256 is -`359dd94c366bd9b9122e4e6ff46e7b717c1ea00830cdf0d63e848e66fcc7c7ae`. -Every archived file was checked against its manifest. It also includes the -[capacity scan](./645-p3-capacity-scan-20260916.md) and -[registration/recovery](./645-p3-registration-20260916.md) evidence, cleanup -ledgers, private wrapper bundles and exact source snapshot `0ddcb4ea`. diff --git a/docs/verification/645-p3-wake-feedback-20260916.md b/docs/verification/645-p3-wake-feedback-20260916.md deleted file mode 100644 index ad937c421..000000000 --- a/docs/verification/645-p3-wake-feedback-20260916.md +++ /dev/null @@ -1,74 +0,0 @@ -# Generic wake-failure feedback deployment — 2026-09-16 - -The [independent PID 1 diagnostic](./645-p3-pid1-observer-20260916.md) captured -AWS's generic `Resume lifecycle hook failed.` message. The previous classifier -did not recognize that wording and suggested retrying a generic compute failure. -Source commit `35d5515b` now classifies it as `MICROVM_RESUME_HOOK_FAILED`, with -service/admin guidance, the relevant log and request-ID locations, and -`retryable: false`. This fixes the explanation; it does not fix the failed wake. - -## Reviewed deployment - -The `backgroundagent-dev` stack in account ``, `us-west-2`, reached -`UPDATE_COMPLETE`. Change set `p3-wake-feedback-20260916` contained exactly 14 -resource changes: 11 Lambda code-object updates, a new coordinator version, -retention of the previous version, and the live alias update. No continuing -resource was replaced; all 475 continuing/new resource entries were accounted -for. No worker image, environment, policy or AgentCore container property changed. -Execution request ID: `170357ab-2b5c-48b0-9a5a-adecd8a15d11`. - -The live coordinator is version **9**, with code SHA-256 -`zoYSUwO5qeLX0pR78DU8OKmFu6qeU9kes31YRj5up4o=`. -Version **8** remains readable with its original code hash. All 11 deployed ZIPs -were downloaded and their hashes matched Lambda's `CodeSha256`. - -Normal MicroVM image **5.0** remains `ACTIVE`/`SUCCESSFUL`, with **8192 MiB** and -the original server process/connection settings. AgentCore remains version -**5**, `READY`. Both automatic-suspension switches remain **false**. - -The deployment used the exact prior S3 template bytes, SHA-256 -`9741f2d4ea6f43e8dde6deff0cb791780a54efdb933e7046c9851b9e6a2e3f9d`. -The reviewed compact target was 706,153 bytes, SHA-256 -`50331a6fe3bcbe5218bfde4cd7c0185f40987c7bc0f3dd95de24de22ffb82d15`. -An explicit `us-west-2` synthesis supplied the new code assets. The default -build's separately generated `us-east-1` assembly was not deployed. - -## Validation and limits - -The full CDK build passed compilation, lint, synthesis and **5,052 tests in 229 -suites**, plus one snapshot. The 56 optional DynamoDB Local tests were skipped -in this build; the previous real-database run passed and this change does not -alter that protocol. The focused classifier/orchestrator run passed 156 tests. - -Four owned synthetic terminal records verified the normal deployed GetTask -handler: - -| Stored error | Observed response | -|---|---| -| Actual generic AWS wake-failure wording | Service error, specific wake explanation, not retryable | -| Legacy reconciliation message without a stable code | Same specific wake guidance | -| New `MICROVM_RESUME_HOOK_FAILED` code | Same specific wake guidance | -| Previously persisted `MICROVM_SUBSTRATE_TERMINATED` code | Existing generic classification preserved | - -Each temporary row was conditionally deleted and its absence verified. No -worker or durable execution was started. Direct handler invocation supplied a -trusted test identity; it does not test API Gateway authentication. Local -orchestrator regression tests cover the newly persisted failure code; these -four live reads do not constitute another end-to-end wake/finalization run. - -The actual F08 task's historical error is unchanged. Its already-persisted -generic code continues to take precedence over diagnostic wording. The five -connection-refusal failures and the distinct F08 generic failure remain open -in the [service feedback tracker](./645-lambda-microvm-service-feedback.md). -Neither retained versions nor this code-only deployment constitute a completed -capacity-protocol rollback exercise. - -Raw deployment, package and API evidence: -`/tmp/abca-645-p2-clean-20260913/p3-wake-feedback-20260916`. - -The permanent private archive is -`/Users/sphias/.local/share/abca-verification/645-p3-20260916/generic-wake-feedback-evidence.tar.gz` -(60 files, 820,897,318 bytes, mode `0600`, SHA-256 -`ab24324c8b6b12c3fb8efce4fd6f75a9d6f0ecbd1dae25d4fbfbe260939e79e9`). -All file hashes were verified against its manifest. It retains the exact 11 -deployed Lambda ZIPs and source/docs commit `0090a713`. diff --git a/docs/verification/645-p3-wake-transport-20260916.md b/docs/verification/645-p3-wake-transport-20260916.md deleted file mode 100644 index aeaaaf0d9..000000000 --- a/docs/verification/645-p3-wake-transport-20260916.md +++ /dev/null @@ -1,265 +0,0 @@ -# ADR-021 P3: wake HTTP connection and event-loop investigation - -Comparison completed on September 16, 2026, in `us-west-2`. -A fresh failure now correlates the exact refusal wording with an expired -connection timer and a live listener. Three `Connection: close` cases passed -with actual closure before freeze and fresh resume connections. The normal -deployment was unchanged during this comparison. This record separates -actual AWS wake evidence from local calibration and attempts that failed before -any sleep. Nothing here has been submitted to the service team. - -The subsequent [normal rollout](./645-p3-connection-close-rollout-20260916.md) -deployed image 6.0 and coordinator 10. Four real Durable workflows passed on that -image; the normal automatic-suspension switches remain off. - -## Question and instrumentation - -An HTTP keep-alive connection lets a client reuse an existing connection for its -next request. Uvicorn normally closes an unused connection after five seconds. -If its timer advances while a MicroVM is frozen, closing that connection can -race with the next request when the worker wakes. - -The private image was derived from the exact deployed image 5.0 artifact, -SHA-256 `9b3150e9e5991cc9fcc8a4adbb2cfbd8f97399f0c16baa9c2b5fe5b101e7b035`. -It retains the original server command, server as PID 1, default keep-alive -behavior, 8,192 MiB memory and all six hooks. Its additions are: - -- HTTP protocol metadata: connection identity, request arrival, response - completion, idle-timer deadline, and the function that closes a socket. -- A timer on the server's event loop, showing when that loop resumes execution. -- A separate process sampling the server and its listening socket. This observer - opens no network connections. - -The probes do not record request bodies, headers or credentials. The artifact -SHA-256 is `d894f553c1f43983e30591f38a727ce6b9ae7db58dd3f8246feb11415e39f63b`; -109 other ZIP entries are byte-identical to the baseline. The private image is -`backgroundagent-dev-p3-wake-transport-20260916`, version `1.0`. - -## Local calibration - -A disposable ARM64 Linux container used Uvicorn 0.50.0, Python 3.13.13, -FastAPI 0.139.0 and the asyncio/h11 server. It had no external network. -The controller waited for logged response completion and an armed idle timer -before pausing the server process. - -| HTTP response behavior | Process pause | Next request | Separate fresh connection | -|---|---:|---|---| -| Default keep-alive | 0.3 s | 200 | 200 | -| Default keep-alive | 6 s | Old connection failed with `BrokenPipeError` | 200 | -| Explicit `Connection: close` | 0.3 s | 200 | 200 | -| Explicit `Connection: close` | 6 s | 200 | 200 | - -The failing old connection was closed by `timeout_keep_alive_handler`; the server -stayed alive. `Connection: close` tells the client to use a new connection. It -is different from setting the server's idle timer to zero. - -Two preliminary local attempts are retained and excluded. The first started -the heartbeat during module import, before Uvicorn created its event loop. -The second paused immediately after the client read the response body, before -the server finished its connection bookkeeping. It consequently armed the timer -after continuation. Reading a response body is not proof that the server has -already armed its idle timer. - -Pausing a local process does not reproduce AWS snapshot or network restoration. - -## Attempts excluded before wake - -| Case | Task | Reason | Cleanup | -|---|---|---|---| -| 1 | `01M2NZ3AB465YF8ZN6ED3ERAB4` | Diagnostic IAM policy retained old fixture IDs; initial task read denied | Cancelled; no worker or reservation created | -| 2 | `01M2NZ3AB92EGG6GCSG2XT9DVZ` | AWS durable execution stopped with `Stopped.ByService` | Owned worker cancelled and terminated; reservation released | -| 3 | `01M2NZ3AB9PVXP9QHAB79DMSTB` | Second `Stopped.ByService`, during startup | Owned worker cancelled and terminated; reservation released | - -The copied IAM allowlists were corrected and checked against the fresh task -IDs. The two durable service errors contain correlation IDs -`b56d7524-7d7c-49c3-afab-13f343b3ab85` at 20:56:03.507Z and -`75b494cd-f102-4f1d-9a14-02c86533c808` at 20:58:46.239Z. Neither attempted a -Suspend or Resume. They are separate from the MicroVM wake failure. - -Subsequent transport tests use the production coordinator body and supervisor -with a local, single-pass step/poll adapter. Workers and lifecycle calls remain -in AWS. This tests real wake behavior; it does not test durable replay or resolve -the separate durable service stops. Each worker has a 1,800-second maximum -lifetime, with a five-minute local coordinator bound and six-minute watcher. - -## First actual AWS wake - -Task `01M2NZ3ABA61SYKQSRJ9S8ZAZ4` used worker -`microvm-a0df3925-9773-325a-9e64-7ce295f548db`, a 150-second approval window -and a 30-second sleep delay. - -| UTC time | Observation | -|---|---| -| 21:03:48.732 | Suspend response finishes on connection 3; five-second deadline armed | -| 21:03:50.039 | Watcher observes `SUSPENDED` | -| 21:04:46.524 | Sole Resume request starts | -| 21:04:46.779 | AWS accepts, receipt `da6f3495-c60b-44c6-a6ab-542533e43f20` | -| 21:04:46.979 | Server event loop resumes after a 58.306-second gap | -| 21:04:46.979 | Expired idle timer closes connection 3 | -| 21:04:46.981 | Server accepts fresh connection 4 | -| 21:04:46.983 | Resume request arrives on connection 4 | -| 21:04:47.428 | Resume response completes with HTTP 200 | -| 21:05:46.693 | Original approval timeout observed; subsequent approve API returns 404 | -| 21:05:57.960 | Task and cleanup checks pass without watcher repair | - -The independent observer saw the original server still owning its listener. -The original approval became `TIMED_OUT`; the Read did not execute. The task -completed, its worker terminated, its reservation was released, its counter was -zero and its payload absent. The complete trace contained no dropped records. - -This directly shows an idle timer expiring after an AWS freeze. It also shows -a successful fresh connection afterward. Timer expiry alone therefore does not -establish the cause of the intermittent failure. - -## Exact refusal with connection and listener evidence - -A second unchanged-image round completed one more successful wake, then -reproduced the exact refusal. Its third planned case was not started. -The failing task was `01M2P0QTA9M6D2J6B5RM42KDNH`, worker -`microvm-bb1bfd4b-9ce3-3691-a60b-88f263081d43`, on private image `1.0`. -No command or guest failure was injected. - -| UTC time | Observation | -|---|---| -| 21:14:27.502 | Suspend accepted, receipt `26d37f89-141a-4938-ba07-904c3edaf6b9` | -| 21:14:27.565 | Suspend HTTP 200 finishes on connection 3, peer port 55068; idle deadline 157.88734039 on the monotonic clock | -| 21:15:26.540 | Sole Resume request starts | -| 21:15:26.786 | Resume accepted, receipt `74943bd5-cfd1-4781-9f10-9946566089e8` | -| 21:15:26.947 | Event loop resumes after a 59.451-second gap, at monotonic time 212.277492979 | -| 21:15:26.948 | `timeout_keep_alive_handler` closes connection 3 | -| 21:15:26.949 | Connection 3 reports `connection_lost` | -| 21:15:26.950 | Independent observer sees PID 1 running and owning the original port-8080 listener, inode 13544 | -| 21:15:27.471 | AWS terminates the worker with `Resume lifecycle hook connection was refused` | - -The retained window has no fresh HTTP connection, resume request bytes, resume -hook entry or resume access log. The observer is a sample, not continuous proof -of health; its process/socket reads are not atomic. The 271 raw guest events and -immediate service response are retained. - -The coordinator recorded `FAILED` and released the reservation before the -runner's idempotent fallback cleanup. This is a failed wake, not a passing -approval test. Together with the successful fresh-connection control, it -supports an old-connection race. The service's actual dispatch error and -connection choice remain unavailable from customer-visible telemetry. - -## Explicit connection-close candidate - -The candidate adds `Connection: close` to suspend and resume JSON responses, -including error responses. Status codes, response bodies, checkpointing and -approval barriers remain the same. This makes the HTTP response end the -connection before a freeze instead of leaving closure to an idle timer. - -The private image's version `2.0` differs from diagnostic `1.0` only in -`agent/src/microvm_http.py`; 112 other archive entries are byte-identical. -Candidate artifact SHA-256: -`6eeddb08a4314d939e87a4bcc3a517319e107d1bcc3a5eda71d814346f60fbeb`. -The original server remains PID 1 with 8,192 MiB. - -Agent quality checks passed: lint, formatting, type checking, and 1,955 tests -with 11 opt-in DynamoDB Local tests skipped. Eight new route cases verify that -both hooks return the close header for HTTP 200, 400, 409 and 503, even when the -client requests keep-alive. - -The live comparison first ran two unchanged-image cases with a 210-second -approval window and a 30-second sleep delay, checking an actual suspended hold -longer than 90 seconds. These cases used a seven-minute local coordinator bound -and eight-minute watcher. It then ran three candidate-image cases with the -original 150-second approval window and approximately 60-second freeze. -Each phase was configured to stop at its first failure. - -| Case | Image | Observed suspended hold | Result | -|---|---|---:|---| -| `over-90s-1` | Unchanged `1.0` | 117.829 s | Passed | -| `over-90s-2` | Unchanged `1.0` | 117.935 s | Passed | -| `close-1` | Candidate `2.0` | About 58 s | Passed | -| `close-2` | Candidate `2.0` | About 59 s | Passed | -| `close-3` | Candidate `2.0` | About 58 s | Passed | - -The candidate audit additionally checked the actual HTTP transport: - -| Case | Suspend connection closed | First observed `SUSPENDED` | Fresh resume connection | -|---|---|---|---| -| `close-1` | 21:42:17.331 | 21:42:18.562 | 21:43:16.699 | -| `close-2` | 21:46:26.263 | 21:46:27.040 | 21:47:26.058 | -| `close-3` | 21:50:09.743 | 21:50:10.231 | 21:51:08.939 | - -In every candidate, connection 3 closed from Uvicorn's response-send path before -the freeze; no idle deadline was armed and no idle-expiry callback ran on that -connection. Resume arrived on new connection 4. The internal `keep_alive` -boolean remained true even though h11 closed the transport, so that boolean -alone would be an incorrect verification criterion. - -All five comparison tasks preserved their original approval deadline, applied -the timeout, rejected a subsequent approval with HTTP 404 `REQUEST_NOT_FOUND`, -and completed without executing the unapproved Read. Complete traces contained -one Read intent each and no dropped records. Every worker terminated, every -reservation was released, counters reached zero, and payload prefixes were -empty without watcher repair. All four phase audits had zero errors. - -A longer-sleep pass or a finite series of successful resumes alone cannot -establish the service-side cause or eliminate an intermittent failure. The -candidate's direct transport evidence establishes removal of the retained -suspend connection in these tests. It does not reveal the service's failed -dispatch error or prove that every historical wake failure had this cause. - -## Error feedback correction - -Review also found that the timeout wording `Resume lifecycle hook timed out` -fell into the generic retryable substrate category. It now maps to -`MICROVM_RESUME_HOOK_FAILED`, preserving the raw reason and requiring the same -service/admin diagnosis as the other recognized wake failures. The regression -checks current, legacy and raw error forms and the resulting retry guidance. -All 136 focused classifier tests passed, followed by the complete CDK suite: -5,053 passed and 56 opt-in DynamoDB Local tests skipped. TypeScript compilation -and ESLint passed. The remedy explicitly avoids treating the service's refused -wording as proof of a closed listener. - -## Historical timing correction - -The first four refusal records had short intervals measured from an observed -`SUSPENDED` state. Their archived HTTP access timestamps are earlier: - -| Earlier case | Suspend HTTP 200 access log | AWS termination | Access-to-termination interval | Observed `SUSPENDED`-to-termination interval | -|---|---|---|---:|---:| -| Approval, Sep 15 | 21:31:41.279 | 21:31:46.958 | 5.679 s | 2.834 s | -| Start-crash recovery, Sep 15 | 21:56:48.531 | 21:56:56.279 | 7.748 s | 3.006 s | -| Inline Resume failure recovery, Sep 15 | 22:54:39.936 | 22:54:45.753 | 5.817 s | 4.759 s | -| Missing approval, Sep 16 | 00:25:01.657 | 00:25:07.524 | 5.867 s | 2.923 s | - -An access log precedes final response completion and timer arming. AWS's -termination timestamp is not the exact time its failed hook request was sent. -The old records do not contain those precise events. These intervals cannot -prove idle expiry caused a refusal, but a short interval measured from -`SUSPENDED` does not rule it out. - -## Evidence and remaining work - -Raw scripts, exact artifacts, command receipts, paginated guest logs, original -failure timing extracts and audit results are retained privately under: - -- `/tmp/abca-645-p2-clean-20260913/p3-wake-transport-20260916` -- `/tmp/abca-645-p2-clean-20260913/p3-wake-transport-round2-20260916` -- `/tmp/abca-645-p2-clean-20260913/p3-wake-pool-window-20260916` -- `/tmp/abca-645-p2-clean-20260913/p3-wake-connection-close-20260916` - -Cleanup finished at 21:58:39.471Z. All ten created workers were terminated. -The private image and its versions, function and versions, two roles, parameter, -two log groups, two exact artifact objects and ten zero-valued fixture counters -were removed and their absence checked. Ordinary task records remain under -their existing retention policy. The archive retains 6,636 guest log records -and 73 coordinator log records. - -An initial cleanup preflight rejected `/tmp` versus `/private/tmp` spellings of -the same directory before any resource deletion. It was corrected by comparing -resolved paths; that failed preflight and the successful cleanup are both retained. - -The evidence is preserved in the private permanent archive -`~/.local/share/abca-verification/645-p3-20260916/wake-transport-evidence.tar.gz`. -It contains 255 files, occupies 17,981,013 bytes, and has SHA-256 -`c949e28f68b0884d84c3b9d0ced6d166ed307d4f2fc8f876b89a1f1474e9b0f8`. -Every member's size and hash were checked against its manifest; archive -permissions are `0600`. - -At comparison cleanup, the normal deployment remained coordinator 9 / image -5.0, 8,192 MiB, with automatic suspension disabled. The subsequent normal rollout -and its separate acceptance archive are recorded in the linked rollout report. diff --git a/docs/verification/645-p3-workspace-recovery-20260917.md b/docs/verification/645-p3-workspace-recovery-20260917.md deleted file mode 100644 index 5e9f12670..000000000 --- a/docs/verification/645-p3-workspace-recovery-20260917.md +++ /dev/null @@ -1,134 +0,0 @@ -# ADR-021 workspace recovery prerequisite - -`agent/src/continuation_workspace.py` adds offline capture and restoration of a -coding workspace. Together with the -[conversation checkpoint](./645-p3-session-recovery-20260917.md), it lets a new -local agent process recover its conversation and files. This is a prerequisite; -the production pipeline does not yet use either component for worker replacement. - -In plain language, a Git commit saves a named version of the code. The **index** -holds changes selected for the next commit. The **working tree** contains the -files currently on disk, including changes that have not been selected or -committed. Recovery must preserve all three, rather than merely clone the last -version uploaded to GitHub. - -## What the archive preserves - -- A Git bundle containing reachable history, local commits, refs and tags, - together with the current branch or detached HEAD. -- A binary staged patch and checksum of the original index entries, preserving - the difference between staged and unstaged changes. -- The repository-local `.git/info/exclude` rules, so its ignored files remain - ignored after restoration. -- Every regular working file, including untracked and ignored files, binary - contents, executable permissions, directories and leaf symlinks. -- Task, attempt, approval request, owner and repository identity; the original - absolute workspace path; and checksums for the archive and each data member. - -Ignored files are included because a Git ignore rule does not prove a file is -disposable. The default limit is **1 GiB including the archive**, with **100,000 -working-tree entries** and a **60-second limit per Git command**. An oversized -workspace fails capture and must keep its worker available; it is not silently -trimmed to fit. - -The original `.git` administration directory is not copied. Restoration rebuilds -it from the bundle and staged patch, sets a credential-free GitHub origin URL and -the `gh` credential helper, and verifies the restored index. Original Git hooks, -config, credential headers and global configuration are not restored. Platform -Git identity and commit-attribution hooks must be installed by the future -pipeline integration. This archive does not replace the separate saved -`RepoSetup`/workflow context needed by that integration. -The local ignore file is an explicit data-only exception; symlinks to an external -ignore file are rejected. Global ignore/filter configuration still belongs to -the fresh worker's trusted setup. - -The module does not read the home directory, copy authentication files from it, -or serialize environment variables. Repository contents can still contain -sensitive data, so the complete archive is private task data. A symlink is -preserved as a leaf, including an external target address, without reading the -target's contents. Archive extraction cannot write through a symlink. -External tools and home-directory caches are not captured. The future worker -setup must recreate required tool installations before using links to them. - -## Capture and restore boundaries - -The caller must hold the lifecycle barrier to stop agent/tool writes. Capture -also compares file metadata and Git state before and after writing. It opens -regular files through directory descriptors without following substituted -symlinks. A detected change prevents publication. - -Capture writes to an owned temporary directory, validates the archive, then -publishes its completed file without replacing an existing destination. Failures -remove the temporary capture data. The original workspace is not modified. - -Restore requires the expected archive checksum and identity and the original -stable workspace path, normally `/workspace/`. It validates member -checksums, paths, types, parent relationships, manifest and Git refs before -publishing a new workspace. It rebuilds the repository offline, with hooks and -external diff helpers disabled. A failed restoration removes only its newly -created staging data; an existing workspace is never cleared. - -The following currently fail with explicit feedback instead of losing state: - -- Linked Git worktrees, shallow/sparse repositories, alternate object stores, - replacement refs, submodules, merge conflicts and active Git operations. -- Intent-to-add, skip-worktree and assume-unchanged index flags. -- Hard-linked regular files, special files, mounted directories and special - permission bits; nested `.git` entries and case-colliding paths. -- Changed or corrupt archives, another task/request's identity, a different - workspace path, unlisted/duplicate members, path traversal and symlink parents. - -`WorkspaceCheckpointError.code` identifies stages such as `size_limit`, -`entry_limit`, `unsupported_git`, `workspace_changed`, `checksum_mismatch`, -`git_timeout` and `destination_exists`. Error messages omit file contents and -raw Git stderr. The future lifecycle integration must expose these failures -without claiming that the worker can be released. - -## Verification - -The unit tests use real offline Git repositories. They preserve two commits, -local branches/tags, staged and unstaged binary/text edits, a staged rename, -an unstaged deletion, ignored and untracked files, executable permissions and -symlinks. Failure injection covers limits, concurrent changes, existing -destinations, unsupported states and hostile archive members. - -The opt-in SDK test now uses this archive instead of copying the working -directory as a fixture: - -1. Start the pinned SDK/CLI and block its original `Read` permission hook. -2. Acknowledge the exact conversation action and capture the workspace. -3. Kill the original process group and delete its configuration and workspace. -4. Restore the files from the archive and the conversation from `SessionStore`. -5. Verify approval executes one new `Read`, denial executes none, the original - session ID is retained and the untracked marker survives. - -The model is deterministic and runs on loopback. No AWS resources, paid model -service or notification channel are used by these tests. - -The final source passed `ABCA_TEST_SDK_CONTINUATION=1 mise run //agent:quality`: -lint, formatting, types and **2,054 tests**, with **11 skipped** and **86.66%** -total coverage. This includes **50 workspace tests** and both composed SDK cases. -The skipped tests are the existing opt-in DynamoDB Local cases; the one warning -is Starlette's existing `httpx` test-client deprecation. - -The agent Bandit high-severity scan, staged secrets scan and silent-success scan -for changes since `f9ee0b49` passed. The full repository silent-success scan's -previously recorded findings are not claimed fixed by this patch. - -The source snapshot, final test logs, actual SDK model requests, hook audits, -workspace/conversation archives and checksums are retained under -`/Users/sphias/.local/share/abca-verification/645-p3-20260916/workspace-checkpoint`. - -## Remaining integration - -The archive is local only. It must still be uploaded under scoped permissions, -read back and pinned to an immutable object version. The existing conversation -S3 adapter accepts its JSON envelope and **does not upload this tar archive**. -Both receipts must be conditionally published together for the owning attempt -before any worker release. - -Production work still includes saved workflow/baseline context, fresh credential -setup, restoring without the destructive fresh-clone path, worker-attempt -fencing, capacity transfer and replacement cloud-worker acceptance. Approval -deadlines and normal automatic sleep are unchanged. See the -[P3 implementation order](./645-p3-implementation-plan.md#unanswered-approvals-implementation-order). diff --git a/docs/verification/645-payload-bootstrap.md b/docs/verification/645-payload-bootstrap.md index 48adb73b6..57add082a 100644 --- a/docs/verification/645-payload-bootstrap.md +++ b/docs/verification/645-payload-bootstrap.md @@ -1,142 +1,79 @@ -# #645: trusted task delivery for ECS and MicroVM +# Trusted task delivery for ECS and MicroVM -Implementation date: 2026-09-13. Tracks the prerequisites in #817 and #700. **Deployed for MicroVM; 11 live transport/failure cases passed on 2026-09-14.** The [live evidence](./645-p2-payload-live-20260914.md) covers real worker reads/rejections, URL expiry/revocation, operator-side conditional preparation and immediate Run replay. The broader authorization/recovery matrix and ECS rollout below remain open. This does not implement P3 sleep/wake. +The coordinator delivers each task through an IAM-authenticated deployment +manifest and a signed URL for one task object. This is application startup +configuration, distinct from the CDK infrastructure bootstrap. -## What changed, in plain language +## Contract and permissions -Think of S3 as a locked filing cabinet. A **bucket** is one cabinet and an **object** is one file. The **coordinator** is the supervisor that starts tasks. A **worker** is the ECS container or MicroVM that runs the coding agent. +Version 2 is defined in `contracts/constants.json` under `payload_bootstrap`. +The [ADR wire contract](../decisions/ADR-021-lambda-microvms-compute-backend.md#3-packaging-same-agent-image-source-new-build-path) +and [security design](../design/SECURITY.md) describe its trust boundary. -Previously, each worker's permission badge could read other tasks' instruction files. Also, checking that two AWS addresses named the same account did not prove either address belonged to this deployment. - -The new protocol gives the worker: - -1. A small **manifest**: a list of the deployment's non-secret settings. The worker reads this with its own AWS permission badge. Only this deployment's bootstrap directory is allowed. -2. A **presigned URL**: a temporary download ticket, created by the coordinator, for exactly one task's instruction file. AWS checks its signature. Possessing the ticket permits the read, so the complete URL must be treated like a password. - -The downloaded task must name the expected task ID and contain exactly the settings in the authenticated manifest. Changing a secret's address to another workspace in the same AWS account now fails this comparison. - -This bootstrap means “load the information needed to start a task.” It is distinct from the infrastructure bootstrap that installs deployment permissions. - -## Storage and wire contract - -Both backends use `contracts/constants.json` → `payload_bootstrap`, version 2. All task payloads use this path, including small tasks. - -| Location | Content / owner | -|---|---| -| `bootstrap/.json` | `{version:2, backend, platform_config}`. Coordinator writes; worker reads. The filename contains a SHA-256 digest: a fingerprint of the exact JSON bytes. | -| `/payload.json` | `{version:2, task_id, agent_payload, platform_config}`. Coordinator writes; worker downloads using the signed URL. | -| `/launch.json` | `{fingerprint, reference}`. Private coordinator replay record containing the signed URL. No worker read. | -| MicroVM `runHookPayload` / ECS `AGENT_PAYLOAD_REF` | JSON string containing the reference below. Never log it. | - -```json -{ - "version": 2, - "task_id": "TASK001", - "bootstrap_s3_uri": "s3://deployment-bucket/bootstrap/.json", - "payload_url": "", - "expires_at": 1789312500000 -} -``` - -The example is descriptive; the manifest digest and signed URL must be produced by the coordinator. `expires_at` is milliseconds since the Unix epoch. - -MicroVM manifests contain the allowlisted environment identifiers built from the coordinator's deployment environment. ECS manifests contain `{}` for `platform_config`: its platform settings already arrive through the trusted task definition and coordinator overrides. Task-specific fields remain in `agent_payload`. - -Manifest and payload limits are **16 KiB** and **8 MiB**, respectively. The serialized MicroVM reference must fit **4,096 bytes**. The entire ECS overrides object must fit **8,192 UTF-8 bytes**; the prompt is no longer duplicated in `TASK_DESCRIPTION`. - -The image rejects the old unsigned inline/S3 envelopes. There is no compatibility branch that accepts settings merely because an image already contains the required environment variables. - -## Permission boundary - -| Principal | Allowed | Denied / not granted | -|---|---|---| -| Worker ambient role | Exact `s3:GetObject` on its bucket's `bootstrap/*` | Explicit `Deny s3:GetObject*` outside that prefix, including other/public buckets; explicit `Deny s3:List*` on its payload bucket. No payload write/delete grant. | -| Trusted coordinator | `PutObject` on manifests and `*/payload.json`, `*/launch.json`; `GetObject`/`DeleteObject` on the two task paths; `ListBucket` on its payload bucket | No task-object version deletion or manifest deletion grant added by this helper. | - -The explicit denial outside `bootstrap/*` is essential. An allow-only policy would not authenticate configuration: an attacker-controlled public bucket could grant the worker access to a fake manifest. A successful read with the deployed deny policy establishes its permitted origin; the hash only detects different bytes, not authorship. - -The coordinator needs `ListBucket` because S3 returns **403 AccessDenied**, rather than **404 NoSuchKey**, for a missing object when the caller cannot list the bucket. The first launch probes for its saved record. Workers do not get this permission. - -AWS documents this in [GetObject permissions](https://docs.aws.amazon.com/AmazonS3/latest/API/API_GetObject.html). Its [conditional-write guide](https://docs.aws.amazon.com/AmazonS3/latest/userguide/conditional-writes.html) specifies that the first completed conditional write wins and later writes receive 412; a concurrent delete can produce 409. The [presigned-URL guide](https://docs.aws.amazon.com/AmazonS3/latest/userguide/using-presigned-url.html) confirms that temporary credential expiry can end a URL's validity before its requested expiry. These documents were checked during this review; deployed behavior remains a separate gate. - -The worker reads the manifest through the attributed `platform_client("s3")` before installing configuration. It then downloads the payload over HTTPS without adding worker credentials. Only the exact bucket/task key on a regional S3 host is accepted; redirects, environment proxies, alternate hosts, credentials in URLs, custom ports, duplicate query parameters and non-HTTPS URLs are rejected. AWS validates the actual signature and expiry; local URL checks are not a cryptographic verifier. - -The runtime needs HTTPS/443 and DNS access to the selected S3 region. The live MicroVM probes verified manifest and signed-task downloads through the deployed runtime connector, including a payload over 1 MiB. Other connectors, ECS and remote-tool connectivity still need their own evidence. Build hooks remain AWS-silent; before configuration installation, runtime bootstrap uses manifest-read and payload-download operations, and diagnostics go to stdout. - -## Retry, expiry and cleanup - -- The coordinator serializes JSON consistently and conditionally creates task objects with `IfNoneMatch: '*'`. A second writer cannot overwrite existing instructions. Read-back handles a write that committed but lost its reply. -- The saved launch record returns the **same URL** after a retry or coordinator restart. Generating a new URL would change the MicroVM Run request while reusing its client token. The private record is deliberately outside TaskTable because the agent can read its own task row. -- A URL is requested for at most **900 seconds**. If the SDK exposes an earlier credential expiry, that shortens the requested lifetime. Initial creation refuses a lifetime under **300 seconds**. AWS can still reject credentials earlier; the provider's actual expiration behavior needs live evidence. -- An expired saved reference fails explicitly; it is not silently re-signed. Recover any existing compute handle before deciding what to do. A timeout or expired ticket does not prove no worker started. MicroVM's saved-start receipt still governs its 120-second application replay window. -- The coordinator refreshes identical manifest bytes at each preparation so the bucket's one-day lifecycle does not remove an old manifest just before a new task reads it. Historical authorized manifests can coexist until lifecycle deletion; this is provenance authentication, not a “latest configuration only” policy. -- Finalization best-effort deletes both `payload.json` and `launch.json`. Deleting the current payload revokes its unversioned download link. Lifecycle deletion remains a backstop and is asynchronous, not an exact one-day alarm. -- ECS removes `AGENT_PAYLOAD_REF` from its environment before importing the pipeline or starting repository subprocesses. Errors redact signed URLs; Python exception chains must not reintroduce them. Control-plane request metadata and the coordinator's private record still contain the capability and require trusted access. - -No new KMS key, SSM parameter, table or bootstrap bundle version is needed. Application IAM policies, coordinator code and worker images all change. - -For P3, `/resume` must continue the restored task, not re-run bootstrap or reuse its expired launch ticket. The task context/workspace and durable approval records supply the information needed after sleep; credential refresh remains a separate resume barrier. - -## Local verification - -Tests exercise real producer/consumer code with fake services. The Python tests use real streaming bodies for truncation/closure cases. The MicroVM recovery tests use the shared transport while simulating lost Run replies and coordinator restart. - -Coverage includes immutable/replayed/conflicting writes, competing launch preparation, committed writes with lost responses, expiry/short-lived credentials, wrong task/configuration, same-account workspace substitution, public-bucket denial before download, cross-region secret preservation, malformed/oversized bytes, redirect/host rejection, ECS environment removal and both cleanup keys. - -A regression reproduced a signed-URL leak through a chained Python traceback. Route logging and sanitized exceptions now prevent that leak. Separate tests cover coordinator-side error redaction and oversized ECS/MicroVM references. - -Permission tests inspect synthesized IAM policies, including the `NotResource` explicit deny and the coordinator's missing-object permission. The constants-checker subprocess rejects invalid bounds/paths and consumers that stop importing the contract. - -An offline compatibility probe generated a URL using the actual JavaScript AWS SDK with dummy credentials and passed it through the Python URL parser. That proves the observed SDK URL shape is supported; it makes no AWS authorization claim. An offline esbuild check using the coordinator's external-module settings also includes the matching S3 client, presigner and constants in the JavaScript bundle. It excludes the separately packaged `pdf-parse` dependency and does not replace full CDK packaging or deployment verification. - -See the [implementation checklist](./645-p3-implementation-plan.md#implementation-progress) for final suite counts. Mock authorization failures are not proof of effective AWS IAM. Conditional S3 writes and the missing-object behavior must also be checked live. - -## Coordinated deployment and rollback - -1. Record the source commit, existing coordinator version, ECS task definitions, MicroVM image/version and effective application policies. Retain the previous deployable artifacts and inspect the CloudFormation change set. Do not delete/recreate storage as an upgrade mechanism. -2. Pause new admissions and scheduled/queue submissions for both affected backends. Drain existing tasks, approvals and pending/retry launches; reconcile uncertain starts and terminate any live test compute. Do not upgrade beneath a running old coordinator. -3. Build the new ECS image and a MicroVM image from the same source/contract. Validate build hooks without task secrets. Retain the old image definitions for rollback. Building the new image before the switch is safe; starting production tasks against mixed versions is not. -4. With admissions paused, deploy the matching coordinator, image/task-definition references and IAM changes. Wait for all resources and permissions to settle. Old producer + new image and new producer + old image both fail the version contract; old workers with new policies cannot use the former broad S3 fetch. -5. Verify effective policies and image identifiers, then perform the live matrix below in the test deployment. Inspect sanitized logs, cleanup and retry records. Resume admissions only after these gates pass. -6. On failure, keep admissions paused. Drain/terminate new-version workers and reconcile unknown launches. Restore the previous code, images/task definitions and application policies together. Old permissions restore the old security exposure; rollback is an availability recovery, not a security fix. Do not reuse a task ID with a conflicting or expired launch record; inspect it and use a fresh task submission once the old compute is accounted for. - -The normal stack already creates distinct payload buckets for the selected backend. Check each backend in its own representative deployment. This runbook does not authorize or perform a deployment. - -## Required AWS evidence - -For **both ECS and MicroVM**, retain source/image identifiers, relevant policy snippets, sanitized error codes, AWS request IDs and timing. Never retain signed URLs, credentials or downloaded customer prompts in evidence. - -**Partial completion, 2026-09-14:** [11 live MicroVM cases](./645-p2-payload-live-20260914.md) -verify own-manifest/download access, task/config/path checks, bad bytes and -signature, URL expiry/revocation, another private bucket's manifest denial, -manifest digest validation and large-file transport. Every passing case also -checks concurrent/repeated producer preparation, changed-input conflict and -immediate identical Run replay. Producer calls use operator credentials; -direct Runs bypass coordinator admission/finalization. The table remains the -full acceptance target, including combinations those probes do not cover. - -The [start/recovery follow-up](./645-p2-start-recovery-live-20260914.md) also -verifies real S3 recovery after committed payload/launch replies are lost or a -local process exits between writes. Production start code preserves the saved -capability through a lost Run reply and a fresh process. These use operator -credentials; deployed durable-Lambda recovery and effective-role tests remain. - -| Check | Expected result | +| Object | Purpose | |---|---| -| New task with no stored launch | Coordinator gets `NoSuchKey`, creates immutable payload/reference, worker starts successfully. | -| Worker reads own deployment manifest | Allowed; hash/backend/config validation passes. | -| Worker lists its payload bucket | Explicitly denied. | -| Worker reads own or another task's payload/launch with ambient credentials | Explicitly denied, including guessed keys. | -| Worker reads a valid-looking manifest in another deployment/public bucket | Explicitly denied before configuration installation or payload download. | -| Worker puts/deletes a manifest, payload or launch record | Denied by effective policies; no alternate grant should permit mutation. | -| Correct signed URL versus modified task path/signature | Own URL succeeds; modified request fails. | -| Expired URL / expired signer credentials | S3 rejects; no configuration or pipeline starts; diagnostic has no bearer URL. | -| Same-account other-workspace secret substitution | Reject; cross-region secret explicitly present in the trusted manifest still works. | -| Two preparations / lost committed S3 reply | Same saved reference or explicit conflict, no overwrite of task instructions. | -| Lost MicroVM Run reply / coordinator restart | Same client token and exact request; existing VM recovered, no replacement. Record actual AWS retention/conflict behavior. | -| Typical/large registry-hydrated task | Manifest and exact S3 download work over runtime DNS/HTTPS; hook/override limits hold. #818 now has local resolution/hydration/storage and hook-to-loader cases; live tool connectivity remains required. | -| Success, failure, cancellation | Expected terminal status; compute terminated; both private task objects deleted or cleanup failure visibly recorded. | -| Attempted public MicroVM ingress | No usable public control path with explicit `NO_INGRESS`. Test while the VM is running. | -| Environment/log inspection | Repository subprocesses do not inherit `AGENT_PAYLOAD_REF`; normal/error logs contain no signed URL. | - -This boundary authenticates the boot path under the deployed policy. It does **not** prove complete isolation from a fully compromised worker: compute roles still select session tags and retain other platform permissions, and a stolen bearer link can be used until it expires or is revoked. See [coordinator metadata verification](./645-coordinator-metadata.md) for the remaining task-reporting and session-tag trust limits. +| `bootstrap/.json` | Non-secret backend/platform configuration; coordinator writes, worker reads with its ambient role. | +| `/payload.json` | Task instructions/configuration; downloaded only through the signed capability. | +| `/launch.json` | Private coordinator replay record containing the exact saved reference. | + +The worker's explicit S3 deny outside its own `bootstrap/*` prevents a foreign +public bucket from supplying fake configuration. A content hash alone does not +establish authorship. Workers cannot list their payload bucket or use ambient +credentials to read task objects. The coordinator needs `ListBucket` to distinguish +missing launch records from access denial. + +Downloaded task identity and configuration must match the authenticated manifest. +Old unsigned envelopes are rejected. All payload sizes use S3; the serialized +MicroVM reference must fit 4,096 bytes. Manifest/payload limits are 16 KiB/8 MiB. +MicroVM receives the reference through `runHookPayload`; ECS uses +`AGENT_PAYLOAD_REF` and removes it before repository subprocesses start. + +A signed URL is a bearer credential. Never log it or store it in a worker-readable +task row. HTTPS downloads reject redirects, environment proxies, alternate hosts +and invalid task paths. The runtime needs regional S3 HTTPS and DNS connectivity. + +## Retry and cleanup + +- Conditional writes keep task instructions immutable. Read-back reconciles a + write that succeeded but lost its response. +- Retries reuse the exact saved URL and Run client token. Re-signing would change + the request under the same token. An expired record fails explicitly. +- Signed lifetime is at most 900 seconds and can end earlier when temporary + credentials expire. Initial creation refuses a known lifetime under 300 seconds. +- A timeout does not prove no worker started. Reconcile the existing handle or + uncertain start before considering replacement; the application replay window + is not a service idempotency-retention guarantee. +- Finalization deletes payload and launch objects on a best-effort basis; bucket + lifecycle deletion is an asynchronous backstop. The coordinator refreshes + identical manifest bytes on preparation so old deployment settings remain usable. +- Resume continues saved task state with refreshed credentials; it does not + download the task again using an expired launch URL. + +## Coordinated upgrades + +Coordinator code, worker images/task definitions and IAM must implement the same +contract. Keep the previous deployable artifacts and inspect the change set. +For an upgrade that cannot support overlapping versions, pause actual admission +sources and drain tasks, approvals and uncertain starts before switching these +components together. A concurrency counter alone does not pause webhook/queue +admission. Do not delete and recreate storage to perform an upgrade. + +For overlapping flat-to-nested migration, preserve old-worker permissions until +those workers drain; follow the [migration prerequisites](./645-p3-nested-stack.md). +Rollback also requires compatible code, images and IAM. Never reuse a task ID +with conflicting or expired stored launch instructions. + +## Verification + +Local producer/consumer tests cover conflicts, lost committed replies, malformed +or oversized input, wrong task/configuration, URL redaction and cleanup. These +cannot prove effective AWS permissions. In a representative deployment, verify: + +- Own manifest and signed payload succeed; foreign manifests, ambient payload + reads, bucket listing and worker mutations are denied. +- Modified signatures/paths, expired credentials/URLs and revoked objects fail + without starting the pipeline or exposing a capability in logs. +- Competing preparation and lost S3/Run replies preserve one exact launch request. +- Finalization removes both task objects and releases only confirmed capacity. + +See [recorded acceptance and remaining checks](./README.md). diff --git a/docs/verification/README.md b/docs/verification/README.md new file mode 100644 index 000000000..74f83cabc --- /dev/null +++ b/docs/verification/README.md @@ -0,0 +1,91 @@ +# Lambda MicroVM verification + +This directory keeps the acceptance criteria and operator notes for +[ADR-021](../decisions/ADR-021-lambda-microvms-compute-backend.md). +Detailed deployment transcripts, temporary worker identifiers and investigation +diaries are archived outside the repository. These documents are not test runners. + +- [Task payload delivery](./645-payload-bootstrap.md): authorization, retries and coordinated upgrades. +- [Nested infrastructure](./645-p3-nested-stack.md): configuration and migration prerequisites. +- [Lifecycle diagnostics](./645-p3-lifecycle-diagnostics.md): locating and interpreting failed wakes. +- [Service-team feedback](./645-lambda-microvm-service-feedback.md): remaining service questions. +- [Continuation design](../design/ORCHESTRATOR.md#retained-microvm-approvals): checkpoint ownership, retirement and replacement. + +## Recorded acceptance + +AWS checks on September 14–18, 2026 exercised the following behavior in the tested +deployments. They do not certify a different image, account, Region or upgrade. + +| Area | Observed result | +|---|---| +| Task delivery | Signed payload downloads, invalid input rejection, expiry/revocation, immutable preparation and lost-reply recovery passed. | +| Approval lifecycle | Approve, deny, explicit expiry and cancellation passed; unanswered requests stayed available without a default deadline. | +| Sleep/wake | Repeated wakes, the 600-second default and the live sleep-off switch passed with compatible image/coordinator versions. | +| Credential renewal | A wait exceeding one hour was followed by successful AWS access with renewed task credentials. | +| Continuation | Conversation and Git/workspace recovery, replacement admission, usage limits and capacity release passed. | +| External integrations | Repository work and remote MCP access passed across sleep; an actual Linear submission exercised MicroVM compute with AgentCore Identity vault. | +| Infrastructure | Fresh nested deployment and a deployment-specific overlapping migration passed, including compatible rollback and old-resource cleanup. | +| Other backends | ECS and AgentCore approval/cancellation and scoped-access checks passed in their tested deployments. | + +The wake correction sends `Connection: close` in lifecycle responses before +freeze. Local transport controls reproduced failure on an old connection; +long-sleep controls and the corrected live flows passed. Service-side traces +for the historical failures remain unavailable, so their exact transport error +is not established for every worker. + +## Open PR checks + +- Finish reusable flat-to-nested migration commands and independently test an + upgrade from current `main` on the same deployment. The earlier bespoke + migration is not a substitute for that acceptance. +- Restore a green full build after integration with `main`. The latest recorded + full run passed 2,170 agent tests and 1,005 CLI tests; CDK had 5,417 passes, + four failures and 56 skips. Three resource-budget cases reached 493 resources + (ECS) or 492 (MicroVM), above the repository's 490-resource cushion. A separate + `github-tags` configuration-load failure passed its isolated rerun; its cause + is unresolved. These are historical run results, not a claim that the latest + commit has been fully revalidated. + +## Reproduce local checks + +From the repository root, with dependencies installed: + +```bash +mise run build +MISE_EXPERIMENTAL=1 mise //cdk:testf -- 'microvm|migration|payload-bootstrap' +``` + +For focused worker checks: + +```bash +cd agent +uv run pytest tests/test_microvm_*.py tests/test_continuation_*.py \ + tests/test_approval_retention.py tests/test_payload_bootstrap.py --no-cov +ABCA_TEST_SDK_CONTINUATION=1 uv run pytest tests/test_continuation_sdk_probe.py --no-cov +``` + +The last command opts into the pinned real SDK/CLI probe with a deterministic +loopback model; it does not launch a cloud worker. Optional DynamoDB Local tests +require their documented local service. Mocks do not establish effective AWS IAM. +The standalone cloud acceptance harness and raw receipts remain outside this PR. + +## Live acceptance for an installation + +Record source commit, image ARN/version, coordinator version, configuration and +UTC test window. Use an owned test repository and identity. Enable suspension +only after verifying that image and coordinator together. + +| Exercise | Required observation | +|---|---| +| Normal task | Repository tools run; terminal status, payload cleanup and capacity release agree. | +| Approval after sleep | The saved decision reaches the exact pending tool; verify guest recovery and tool output, not only the Resume API receipt. | +| Deny, expiry, cancellation | No denied/cancelled tool runs; expiry uses the original deadline; compute and capacity are cleaned up. | +| Default and disabled sleep | Omitted override uses 600 seconds; task-level off and deployment off prevent new suspension while wake/cleanup remain available. | +| Credential expiry | Sleep past the original credential lifetime, then perform actual task-scoped AWS operations. | +| Retirement/replacement | Confirm old-worker shutdown before capacity release; one replacement restores files/conversation and preserves approval identity and usage. | +| Upgrade/rollback | Preserve unrelated resource identities, old in-flight work and recoverable checkpoints; test compatible code/image/policy rollback. | + +Include effective-role checks for own-task access and denial of cross-task data, +foreign bootstrap manifests and ambient payload reads. Keep credentials, signed +URLs, prompts and checkpoints out of diagnostic attachments. Clean up test +workers, executions, task data and owned infrastructure after the run. From f95deccdb48ecea11a3abdd88a01a6698fd4b888 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Fri, 18 Sep 2026 09:46:35 -0400 Subject: [PATCH 080/149] docs(microvm): keep service-team feedback outside the repository --- .../constructs/lambda-microvm-compute.test.ts | 2 +- ...ADR-021-lambda-microvms-compute-backend.md | 4 +- docs/design/SECURITY.md | 2 +- .../src/content/docs/architecture/Security.md | 2 +- ...Adr-021-lambda-microvms-compute-backend.md | 4 +- .../645-lambda-microvm-service-feedback.md | 74 ------------------- .../645-p3-lifecycle-diagnostics.md | 1 - docs/verification/README.md | 7 +- 8 files changed, 11 insertions(+), 85 deletions(-) delete mode 100644 docs/verification/645-lambda-microvm-service-feedback.md diff --git a/cdk/test/constructs/lambda-microvm-compute.test.ts b/cdk/test/constructs/lambda-microvm-compute.test.ts index 10a9ccde8..a97a49223 100644 --- a/cdk/test/constructs/lambda-microvm-compute.test.ts +++ b/cdk/test/constructs/lambda-microvm-compute.test.ts @@ -677,7 +677,7 @@ describe('LambdaMicrovmCompute — image provisioned from a managed base image', // Recorded service calls rejected source-conditioned role trust; removing // those conditions restored connector creation and worker launch. Keep exact // resource grants and verify service support before adding conditions again. - // See ADR-021 §4 and docs/verification/645-lambda-microvm-service-feedback.md. + // See ADR-021 §4 for the tested trust-policy limitations. const roles = Object.entries(template.findResources('AWS::IAM::Role')) .filter(([id]) => id.includes('LambdaMicrovmComputeBuildRole') || id.includes('LambdaMicrovmComputeExecutionRole') diff --git a/docs/decisions/ADR-021-lambda-microvms-compute-backend.md b/docs/decisions/ADR-021-lambda-microvms-compute-backend.md index b4ab48f30..403297552 100644 --- a/docs/decisions/ADR-021-lambda-microvms-compute-backend.md +++ b/docs/decisions/ADR-021-lambda-microvms-compute-backend.md @@ -135,7 +135,7 @@ No ABCA endpoint consumer exists in P1–P3. The platform grants no `CreateMicro The backend adds build/runtime VPC connectors, build artifacts, launch payloads, logs, roles and a managed or external image. Its bootstrap policy is conditional on `ComputeTypes` including `lambda-microvm`. A VPC egress connector requires an operator role. Build egress permits ports 80/443 for package installation; runtime egress permits 443 through the platform VPC. -**Trust and PassRole limitation.** Recorded live checks rejected `aws:SourceAccount`/`aws:SourceArn` conditions on the MicroVM-facing roles and `iam:PassedToService` on the MicroVM PassRole paths. The working roles trust `lambda.amazonaws.com` without those conditions; build/execution roles also allow `sts:TagSession`. IAM simulation with caller-supplied condition values did not prove that the service supplied those values. See the [IAM service questions](../verification/645-lambda-microvm-service-feedback.md#iam-setup-f02). Reintroduce a condition only after verifying service support. +**Trust and PassRole limitation.** Recorded live checks rejected `aws:SourceAccount`/`aws:SourceArn` conditions on the MicroVM-facing roles and `iam:PassedToService` on the MicroVM PassRole paths. The working roles trust `lambda.amazonaws.com` without those conditions; build/execution roles also allow `sts:TagSession`. IAM simulation with caller-supplied condition values did not prove that the service supplied those values. Reintroduce a condition only after verifying service support. | Role/action | Scope and responsibility | |---|---| @@ -195,7 +195,7 @@ Changing the default backend, GPU support, native Slack approval buttons, approv - Suspended workers retain ABCA capacity until confirmed retirement. The service’s account-quota treatment of suspended memory was not established by the recorded probes. - Memory baseline validation and published peak capacity are not workload benchmarks. Sustained heavy builds require measured sizing and may fit ECS better. - A healthy heartbeat does not prove coding progress. Hook failures may cause service termination, but successful hook acceptance does not remove ABCA’s cleanup responsibility. -- Service error wording can hide transport errors. The pooled-hook mitigation has local and live evidence; historical internal dispatch traces remain unavailable. Open questions are tracked in the [service feedback record](../verification/645-lambda-microvm-service-feedback.md). +- Service error wording can hide transport errors. The pooled-hook mitigation has local and live evidence; historical internal dispatch traces remain unavailable. ## Testing diff --git a/docs/design/SECURITY.md b/docs/design/SECURITY.md index 8d9467d70..11f6b3372 100644 --- a/docs/design/SECURITY.md +++ b/docs/design/SECURITY.md @@ -48,7 +48,7 @@ The compute role retains only non-tenant access (Bedrock model invocation — al **Lambda MicroVMs compute-role delta** — The compute role additionally reads the per-workspace channel-OAuth secrets (`bgagent-linear-oauth-*`, `bgagent-jira-oauth-*`) and its payload bucket's `bootstrap/*` manifests. Ambient task-object reads and payload-bucket listing are explicitly denied. It also has read-only `ec2:DescribeAvailabilityZones` for repository CDK synthesis; that API has no resource-level scope. -Recorded live calls rejected source-conditioned MicroVM role trust and service-conditioned PassRole grants. The integration therefore omits those conditions on the affected paths, while restricting which roles can be passed and which resources they may access. Reintroduce conditions only after verifying service support. [ADR-021 §4](../decisions/ADR-021-lambda-microvms-compute-backend.md#4-infra-and-iam-conditional-resources-behind-bootstrap-computetypes) records the role responsibilities and limits; [service feedback](../verification/645-lambda-microvm-service-feedback.md#iam-setup-f02) tracks the unresolved contract questions. +Recorded live calls rejected source-conditioned MicroVM role trust and service-conditioned PassRole grants. The integration therefore omits those conditions on the affected paths, while restricting which roles can be passed and which resources they may access. Reintroduce conditions only after verifying service support. [ADR-021 §4](../decisions/ADR-021-lambda-microvms-compute-backend.md#4-infra-and-iam-conditional-resources-behind-bootstrap-computetypes) records the role responsibilities and limits. **ECS/MicroVM task bootstrap (v2)** — The coordinator publishes non-secret deployment manifests and supplies a short-lived signed URL for exactly one task's payload. The worker reads the manifest with its ambient role, whose explicit deny outside its own `bootstrap/*` prevents a foreign public bucket from authorizing fake configuration. It verifies downloaded task identity and exact manifest/config agreement before installing MicroVM settings. The coordinator privately persists the URL outside TaskTable for retry and deletes both task objects at finalization. Worker roles cannot read/list task objects using their ambient credentials. A signed URL is a bearer capability: redact it in errors and keep it out of task rows and repository subprocess environments. Existing session-tag trust and other platform grants remain separate limitations. The [payload guide](../verification/645-payload-bootstrap.md) describes coordinated image/coordinator/policy upgrades; the [acceptance summary](../verification/README.md) distinguishes recorded AWS checks from remaining validation. diff --git a/docs/src/content/docs/architecture/Security.md b/docs/src/content/docs/architecture/Security.md index ee9cfb42d..bb0830531 100644 --- a/docs/src/content/docs/architecture/Security.md +++ b/docs/src/content/docs/architecture/Security.md @@ -52,7 +52,7 @@ The compute role retains only non-tenant access (Bedrock model invocation — al **Lambda MicroVMs compute-role delta** — The compute role additionally reads the per-workspace channel-OAuth secrets (`bgagent-linear-oauth-*`, `bgagent-jira-oauth-*`) and its payload bucket's `bootstrap/*` manifests. Ambient task-object reads and payload-bucket listing are explicitly denied. It also has read-only `ec2:DescribeAvailabilityZones` for repository CDK synthesis; that API has no resource-level scope. -Recorded live calls rejected source-conditioned MicroVM role trust and service-conditioned PassRole grants. The integration therefore omits those conditions on the affected paths, while restricting which roles can be passed and which resources they may access. Reintroduce conditions only after verifying service support. [ADR-021 §4](/sample-autonomous-cloud-coding-agents/architecture/adr-021-lambda-microvms-compute-backend#4-infra-and-iam-conditional-resources-behind-bootstrap-computetypes) records the role responsibilities and limits; [service feedback](/sample-autonomous-cloud-coding-agents/architecture/645-lambda-microvm-service-feedback#iam-setup-f02) tracks the unresolved contract questions. +Recorded live calls rejected source-conditioned MicroVM role trust and service-conditioned PassRole grants. The integration therefore omits those conditions on the affected paths, while restricting which roles can be passed and which resources they may access. Reintroduce conditions only after verifying service support. [ADR-021 §4](/sample-autonomous-cloud-coding-agents/architecture/adr-021-lambda-microvms-compute-backend#4-infra-and-iam-conditional-resources-behind-bootstrap-computetypes) records the role responsibilities and limits. **ECS/MicroVM task bootstrap (v2)** — The coordinator publishes non-secret deployment manifests and supplies a short-lived signed URL for exactly one task's payload. The worker reads the manifest with its ambient role, whose explicit deny outside its own `bootstrap/*` prevents a foreign public bucket from authorizing fake configuration. It verifies downloaded task identity and exact manifest/config agreement before installing MicroVM settings. The coordinator privately persists the URL outside TaskTable for retry and deletes both task objects at finalization. Worker roles cannot read/list task objects using their ambient credentials. A signed URL is a bearer capability: redact it in errors and keep it out of task rows and repository subprocess environments. Existing session-tag trust and other platform grants remain separate limitations. The [payload guide](/sample-autonomous-cloud-coding-agents/architecture/645-payload-bootstrap) describes coordinated image/coordinator/policy upgrades; the [acceptance summary](/sample-autonomous-cloud-coding-agents/architecture/readme) distinguishes recorded AWS checks from remaining validation. diff --git a/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md b/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md index ba7108892..b9e421f0e 100644 --- a/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md +++ b/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md @@ -139,7 +139,7 @@ No ABCA endpoint consumer exists in P1–P3. The platform grants no `CreateMicro The backend adds build/runtime VPC connectors, build artifacts, launch payloads, logs, roles and a managed or external image. Its bootstrap policy is conditional on `ComputeTypes` including `lambda-microvm`. A VPC egress connector requires an operator role. Build egress permits ports 80/443 for package installation; runtime egress permits 443 through the platform VPC. -**Trust and PassRole limitation.** Recorded live checks rejected `aws:SourceAccount`/`aws:SourceArn` conditions on the MicroVM-facing roles and `iam:PassedToService` on the MicroVM PassRole paths. The working roles trust `lambda.amazonaws.com` without those conditions; build/execution roles also allow `sts:TagSession`. IAM simulation with caller-supplied condition values did not prove that the service supplied those values. See the [IAM service questions](/sample-autonomous-cloud-coding-agents/architecture/645-lambda-microvm-service-feedback#iam-setup-f02). Reintroduce a condition only after verifying service support. +**Trust and PassRole limitation.** Recorded live checks rejected `aws:SourceAccount`/`aws:SourceArn` conditions on the MicroVM-facing roles and `iam:PassedToService` on the MicroVM PassRole paths. The working roles trust `lambda.amazonaws.com` without those conditions; build/execution roles also allow `sts:TagSession`. IAM simulation with caller-supplied condition values did not prove that the service supplied those values. Reintroduce a condition only after verifying service support. | Role/action | Scope and responsibility | |---|---| @@ -199,7 +199,7 @@ Changing the default backend, GPU support, native Slack approval buttons, approv - Suspended workers retain ABCA capacity until confirmed retirement. The service’s account-quota treatment of suspended memory was not established by the recorded probes. - Memory baseline validation and published peak capacity are not workload benchmarks. Sustained heavy builds require measured sizing and may fit ECS better. - A healthy heartbeat does not prove coding progress. Hook failures may cause service termination, but successful hook acceptance does not remove ABCA’s cleanup responsibility. -- Service error wording can hide transport errors. The pooled-hook mitigation has local and live evidence; historical internal dispatch traces remain unavailable. Open questions are tracked in the [service feedback record](/sample-autonomous-cloud-coding-agents/architecture/645-lambda-microvm-service-feedback). +- Service error wording can hide transport errors. The pooled-hook mitigation has local and live evidence; historical internal dispatch traces remain unavailable. ## Testing diff --git a/docs/verification/645-lambda-microvm-service-feedback.md b/docs/verification/645-lambda-microvm-service-feedback.md deleted file mode 100644 index 49324b181..000000000 --- a/docs/verification/645-lambda-microvm-service-feedback.md +++ /dev/null @@ -1,74 +0,0 @@ -# Lambda MicroVM service-team feedback - -These questions come from the P1–P3 integration, most recently checked in -September 2026. They are not confirmed service defects in every Region/build. -Detailed worker timelines and receipts are archived privately for escalation. -No service-side trace confirmation has been obtained for the historical wakes. - -## Hook transport and diagnostics (F01, F03, F05, F08) - -**Observed:** some Resume API calls succeeded, then workers terminated with -“Resume lifecycle hook connection was refused” or the generic hook-failed reason. -In instrumented cases, the guest listener remained present and no new resume -handler entry was logged. Local paused-process controls reproduced failure when -reusing an old HTTP connection; a fresh connection succeeded. Longer-sleep -controls and explicit connection-close responses passed live checks. - -**Application correction:** lifecycle responses send `Connection: close` before -freeze. Approve/deny/expiry/cancellation and repeated wakes passed with that -correction. Passing controls do not prove the exact service error for each -historical worker. - -**Ask:** confirm the deployed hook client's pooling/retry behavior, whether stale -connections can survive suspend, and supported clock/timer semantics on restore. -Consider retiring pooled connections at suspend or retrying a failed dispatch on -a fresh connection. Distinguish refused connections, resets, incomplete responses -and timeouts in customer-visible errors. Expose hook-attempt timestamps, error -classification and correlation to the originating API receipt. A failed hook -that never reaches the guest cannot emit its own application log. - -## IAM setup (F02) - -**Observed:** tested source-conditioned role trust and `iam:PassedToService` -conditions rejected operations that worked without those conditions. A target-role -assumption problem also appeared as a caller-side PassRole denial. These are -historical integration findings, not a fresh comparison of every service build. - -**Ask:** document supported keys/values for build, execution and connector-role -assumption and both PassRole paths. Distinguish caller authorization failures -from target-role trust failures. Recheck service support before tightening the -working exact-role grants; IAM simulation alone does not establish which -condition values the service sends. - -## Restore state and connector schema (F04, F06) - -A worker was observed going `SUSPENDED → PENDING → RUNNING`. ABCA now distinguishes -restore from first startup using saved lifecycle intent. Publish the complete -transition/timestamp contract or expose a restoring state. - -A `VPC_EGRESS` connector required an operator role despite the generated property -being optional. Add conditional validation and a complete role/permission example; -this does not imply other connector types need the same role. - -## Lost Run response and idempotency (F07) - -ABCA preserves the exact Run request/token and bounds automatic replay. A lost -successful response can still leave a running worker with no saved handle. -Matching an image, role and start window is not a unique task identity. Accepted -guest identity logs allowed manual recovery in a tested case. - -**Ask:** specify token retention, behavior after expiry/termination/image changes, -and a supported token or request-ID lookup that returns the original worker. -Clarify required CloudTrail event configuration and retention. ABCA's 120-second -replay policy and observed successful replays are not service guarantees. - -## CloudFormation refactoring (F09) - -An image refactor preview succeeded, but execution rejected -`AWS::Lambda::MicrovmImage` because of an unsupported tag schema. Rollback -preserved the original image/resources. - -**Ask:** support image ownership refactoring or reject unsupported moves during -preview; publish a supported migration procedure. Until verified otherwise, use -the [overlap migration prerequisites](./645-p3-nested-stack.md), not a direct -flat-to-nested template switch. diff --git a/docs/verification/645-p3-lifecycle-diagnostics.md b/docs/verification/645-p3-lifecycle-diagnostics.md index acdec8a7f..eb7ceb4b4 100644 --- a/docs/verification/645-p3-lifecycle-diagnostics.md +++ b/docs/verification/645-p3-lifecycle-diagnostics.md @@ -67,7 +67,6 @@ Preserve the raw service reason for diagnosis. The wording “connection was refused” alone does not prove a closed listener: historical guest observations support a stale pooled-connection race, while service-side dispatch traces remain unavailable. Lifecycle responses explicitly close connections before freeze. -See the [service questions](./645-lambda-microvm-service-feedback.md). Diagnostics omit hook bodies, tool arguments, approval contents, credentials, raw exception messages and SDK response bodies. Do not attach signed payload diff --git a/docs/verification/README.md b/docs/verification/README.md index 74f83cabc..11120346c 100644 --- a/docs/verification/README.md +++ b/docs/verification/README.md @@ -1,14 +1,15 @@ # Lambda MicroVM verification -This directory keeps the acceptance criteria and operator notes for -[ADR-021](../decisions/ADR-021-lambda-microvms-compute-backend.md). +For maintainers reviewing/testing the MicroVM backend and operators deploying, +migrating or diagnosing it. [ADR-021](../decisions/ADR-021-lambda-microvms-compute-backend.md) +explains the design; the [user guide](../guides/USER_GUIDE.md#approval-gates-cedar-hitl) +explains approval and sleep options for people submitting tasks. Detailed deployment transcripts, temporary worker identifiers and investigation diaries are archived outside the repository. These documents are not test runners. - [Task payload delivery](./645-payload-bootstrap.md): authorization, retries and coordinated upgrades. - [Nested infrastructure](./645-p3-nested-stack.md): configuration and migration prerequisites. - [Lifecycle diagnostics](./645-p3-lifecycle-diagnostics.md): locating and interpreting failed wakes. -- [Service-team feedback](./645-lambda-microvm-service-feedback.md): remaining service questions. - [Continuation design](../design/ORCHESTRATOR.md#retained-microvm-approvals): checkpoint ownership, retirement and replacement. ## Recorded acceptance From 15e171d57a45b3570eaab42035e965f77e0142a1 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Fri, 18 Sep 2026 10:35:09 -0400 Subject: [PATCH 081/149] refactor(approvals): share authenticated decision paths across channels --- cdk/src/handlers/approve-task.ts | 16 +++++++++++++--- cdk/src/handlers/deny-task.ts | 16 +++++++++++++--- 2 files changed, 26 insertions(+), 6 deletions(-) diff --git a/cdk/src/handlers/approve-task.ts b/cdk/src/handlers/approve-task.ts index 0d3bdd860..5419eb4b9 100644 --- a/cdk/src/handlers/approve-task.ts +++ b/cdk/src/handlers/approve-task.ts @@ -68,26 +68,36 @@ const AUDIT_EVENT_RETENTION_DAYS = Number(process.env.TASK_RETENTION_DAYS ?? '90 */ export async function handler( event: APIGatewayProxyEvent, context?: Pick, +): Promise { + return recordApprovalForUser({ + userId: extractUserId(event), taskId: event.pathParameters?.task_id, body: event.body, + }, context); +} + +/** Shared decision path. Callers must authenticate and map the platform user first. */ +export async function recordApprovalForUser( + input: { userId: string | null; taskId?: string; body?: string | null }, + context?: Pick, ): Promise { const invocationStartedMs = Date.now(); const requestId = ulid(); try { // 1. Auth - const callerUserId = extractUserId(event); + const callerUserId = input.userId; if (!callerUserId) { return errorResponse(401, ErrorCode.UNAUTHORIZED, 'Missing or invalid authentication.', requestId); } // 2. Path + body - const taskId = event.pathParameters?.task_id; + const taskId = input.taskId; if (!taskId) { return errorResponse(400, ErrorCode.VALIDATION_ERROR, 'Missing task_id path parameter.', requestId); } let parsed: ApprovalRequest | null = null; try { - parsed = event.body ? JSON.parse(event.body) as ApprovalRequest : null; + parsed = input.body ? JSON.parse(input.body) as ApprovalRequest : null; } catch { return errorResponse(400, ErrorCode.VALIDATION_ERROR, 'Request body must be valid JSON.', requestId); } diff --git a/cdk/src/handlers/deny-task.ts b/cdk/src/handlers/deny-task.ts index eed55e5df..37b753cf8 100644 --- a/cdk/src/handlers/deny-task.ts +++ b/cdk/src/handlers/deny-task.ts @@ -66,26 +66,36 @@ const AUDIT_EVENT_RETENTION_DAYS = Number(process.env.TASK_RETENTION_DAYS ?? '90 */ export async function handler( event: APIGatewayProxyEvent, context?: Pick, +): Promise { + return recordDenialForUser({ + userId: extractUserId(event), taskId: event.pathParameters?.task_id, body: event.body, + }, context); +} + +/** Shared decision path. Callers must authenticate and map the platform user first. */ +export async function recordDenialForUser( + input: { userId: string | null; taskId?: string; body?: string | null }, + context?: Pick, ): Promise { const invocationStartedMs = Date.now(); const requestId = ulid(); try { // 1. Auth - const callerUserId = extractUserId(event); + const callerUserId = input.userId; if (!callerUserId) { return errorResponse(401, ErrorCode.UNAUTHORIZED, 'Missing or invalid authentication.', requestId); } // 2. Path + body - const taskId = event.pathParameters?.task_id; + const taskId = input.taskId; if (!taskId) { return errorResponse(400, ErrorCode.VALIDATION_ERROR, 'Missing task_id path parameter.', requestId); } let parsed: DenyRequest | null = null; try { - parsed = event.body ? JSON.parse(event.body) as DenyRequest : null; + parsed = input.body ? JSON.parse(input.body) as DenyRequest : null; } catch { return errorResponse(400, ErrorCode.VALIDATION_ERROR, 'Request body must be valid JSON.', requestId); } From 48ec3e7efcd721d6758d90159eb3ea2da9b8c1e4 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Fri, 18 Sep 2026 10:45:04 -0400 Subject: [PATCH 082/149] feat(approvals): bind Linear comments to retry-safe decisions --- cdk/src/handlers/approve-task.ts | 6 +- cdk/src/handlers/deny-task.ts | 6 +- .../handlers/shared/linear-approval-thread.ts | 100 ++++++++++++++++++ cdk/src/handlers/shared/linear-feedback.ts | 29 +++++ cdk/test/handlers/approve-task.test.ts | 26 ++++- cdk/test/handlers/deny-task.test.ts | 26 ++++- .../shared/linear-approval-thread.test.ts | 77 ++++++++++++++ .../handlers/shared/linear-feedback.test.ts | 39 +++++++ 8 files changed, 303 insertions(+), 6 deletions(-) create mode 100644 cdk/src/handlers/shared/linear-approval-thread.ts create mode 100644 cdk/test/handlers/shared/linear-approval-thread.test.ts diff --git a/cdk/src/handlers/approve-task.ts b/cdk/src/handlers/approve-task.ts index 5419eb4b9..be32195ef 100644 --- a/cdk/src/handlers/approve-task.ts +++ b/cdk/src/handlers/approve-task.ts @@ -76,7 +76,7 @@ export async function handler( /** Shared decision path. Callers must authenticate and map the platform user first. */ export async function recordApprovalForUser( - input: { userId: string | null; taskId?: string; body?: string | null }, + input: { userId: string | null; taskId?: string; body?: string | null; decisionSource?: string }, context?: Pick, ): Promise { const invocationStartedMs = Date.now(); @@ -183,7 +183,8 @@ export async function recordApprovalForUser( TableName: TASK_APPROVALS_TABLE_NAME, Key: { task_id: taskId, request_id }, UpdateExpression: - 'SET #status = :approved, decided_at = :now, #scope = :scope', + 'SET #status = :approved, decided_at = :now, #scope = :scope' + + (input.decisionSource ? ', decision_source = :source' : ''), ConditionExpression: 'attribute_exists(request_id) AND #status = :pending AND user_id = :caller ' + 'AND (attribute_not_exists(deadline_epoch) OR deadline_epoch > :epoch)', @@ -198,6 +199,7 @@ export async function recordApprovalForUser( ':scope': scope, ':caller': callerUserId, ':epoch': nowEpoch, + ...(input.decisionSource ? { ':source': input.decisionSource } : {}), }, }, }, diff --git a/cdk/src/handlers/deny-task.ts b/cdk/src/handlers/deny-task.ts index 37b753cf8..4bac0452c 100644 --- a/cdk/src/handlers/deny-task.ts +++ b/cdk/src/handlers/deny-task.ts @@ -74,7 +74,7 @@ export async function handler( /** Shared decision path. Callers must authenticate and map the platform user first. */ export async function recordDenialForUser( - input: { userId: string | null; taskId?: string; body?: string | null }, + input: { userId: string | null; taskId?: string; body?: string | null; decisionSource?: string }, context?: Pick, ): Promise { const invocationStartedMs = Date.now(); @@ -164,7 +164,8 @@ export async function recordDenialForUser( TableName: TASK_APPROVALS_TABLE_NAME, Key: { task_id: taskId, request_id }, UpdateExpression: - 'SET #status = :denied, decided_at = :now, deny_reason = :reason', + 'SET #status = :denied, decided_at = :now, deny_reason = :reason' + + (input.decisionSource ? ', decision_source = :source' : ''), ConditionExpression: 'attribute_exists(request_id) AND #status = :pending AND user_id = :caller ' + 'AND (attribute_not_exists(deadline_epoch) OR deadline_epoch > :epoch)', @@ -176,6 +177,7 @@ export async function recordDenialForUser( ':reason': sanitizedReason, ':caller': callerUserId, ':epoch': nowEpoch, + ...(input.decisionSource ? { ':source': input.decisionSource } : {}), }, }, }, diff --git a/cdk/src/handlers/shared/linear-approval-thread.ts b/cdk/src/handlers/shared/linear-approval-thread.ts new file mode 100644 index 000000000..e9a19581f --- /dev/null +++ b/cdk/src/handlers/shared/linear-approval-thread.ts @@ -0,0 +1,100 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import { createHash } from 'node:crypto'; +import { GetCommand, UpdateCommand, type DynamoDBDocumentClient } from '@aws-sdk/lib-dynamodb'; + +const RETENTION_DAYS = 90; +const RETENTION_SECONDS = RETENTION_DAYS * 24 * 60 * 60; + +export interface LinearApprovalThread { + readonly workspaceId: string; + readonly issueId: string; + readonly taskId: string; + readonly requestId: string; + readonly userId: string; +} + +/** Stable UUID makes a retry after a lost Linear response target the same comment. */ +export function linearApprovalCommentId(thread: LinearApprovalThread): string { + const hash = createHash('sha256').update(JSON.stringify([ + 'abca-linear-approval-v1', thread.workspaceId, thread.issueId, thread.taskId, thread.requestId, + ])).digest('hex'); + const groups = hash.match(/^(.{8})(.{4}).(.{3}).(.{3})(.{12})/)!; + return `${groups[1]}-${groups[2]}-5${groups[3]}-a${groups[4]}-${groups[5]}`; +} + +function threadKey(workspaceId: string, commentId: string): { task_id: string; request_id: string } { + return { task_id: `LINEAR_COMMENT#${workspaceId}#${commentId}`, request_id: 'APPROVAL' }; +} + +/** Mapping is coordinator-owned: task-scoped worker credentials cannot write this key. */ +export async function saveLinearApprovalThread( + ddb: DynamoDBDocumentClient, tableName: string, thread: LinearApprovalThread, +): Promise { + const commentId = linearApprovalCommentId(thread); + await ddb.send(new UpdateCommand({ + TableName: tableName, + Key: threadKey(thread.workspaceId, commentId), + UpdateExpression: 'SET #kind = :kind, #thread = :thread', + ConditionExpression: 'attribute_not_exists(task_id) OR (#kind = :kind AND #thread = :thread)', + ExpressionAttributeNames: { '#kind': 'kind', '#thread': 'thread' }, + ExpressionAttributeValues: { ':kind': 'linear_approval_thread', ':thread': thread }, + })); + return commentId; +} + +export async function readLinearApprovalThread( + ddb: DynamoDBDocumentClient, tableName: string, workspaceId: string, commentId: string, +): Promise { + const result = await ddb.send(new GetCommand({ + TableName: tableName, Key: threadKey(workspaceId, commentId), ConsistentRead: true, + })); + const thread = result.Item?.thread as LinearApprovalThread | undefined; + if (result.Item?.kind !== 'linear_approval_thread' || !thread + || ![thread.workspaceId, thread.issueId, thread.taskId, thread.requestId, thread.userId] + .every(value => typeof value === 'string' && value.length > 0) + || thread.workspaceId !== workspaceId || linearApprovalCommentId(thread) !== commentId) return null; + return thread; +} + +/** Pending mappings have no TTL, like pending approvals; closure starts retention. */ +export async function closeLinearApprovalThread( + ddb: DynamoDBDocumentClient, tableName: string, thread: LinearApprovalThread, +): Promise { + try { + await ddb.send(new UpdateCommand({ + TableName: tableName, + Key: threadKey(thread.workspaceId, linearApprovalCommentId(thread)), + UpdateExpression: 'SET #ttl = if_not_exists(#ttl, :ttl)', + ConditionExpression: 'attribute_exists(task_id)', + ExpressionAttributeNames: { '#ttl': 'ttl' }, + ExpressionAttributeValues: { ':ttl': Math.floor(Date.now() / 1000) + RETENTION_SECONDS }, + })); + } catch (error) { + if ((error as { name?: string }).name !== 'ConditionalCheckFailedException') throw error; + } +} + +/** Only an explicit answer in an approval thread is a decision, never prose or an edit. */ +export function parseLinearApprovalReply(body: unknown): 'approve' | 'deny' | null { + if (typeof body !== 'string') return null; + const match = /^(approve|deny)[.!]?$/i.exec(body.trim()); + return match ? match[1]!.toLowerCase() as 'approve' | 'deny' : null; +} diff --git a/cdk/src/handlers/shared/linear-feedback.ts b/cdk/src/handlers/shared/linear-feedback.ts index 83de57b4d..6e331a844 100644 --- a/cdk/src/handlers/shared/linear-feedback.ts +++ b/cdk/src/handlers/shared/linear-feedback.ts @@ -408,6 +408,35 @@ export async function postIssueComment( return graphqlRequest(token, COMMENT_CREATE_MUTATION, { issueId, body }); } +/** Retry-safe posting for approval prompts and acknowledgements. */ +export async function postIdentifiedComment( + ctx: LinearFeedbackContext, + input: { id: string; issueId: string; body: string; parentId?: string }, +): Promise { + const token = await resolveToken(ctx); + if (!token) return { ok: false, retryable: false }; + const created = await graphqlData(token, ` + mutation ApprovalComment($input: CommentCreateInput!) { + commentCreate(input: $input) { success comment { id } } + }`, { input }); + if (created.ok && (created.value.commentCreate as { success?: boolean } | undefined)?.success) { + return { ok: true }; + } + // A successful write can lose its response. A duplicate ID is acceptable only + // when the saved comment exactly matches this destination and content. + const existing = await graphqlData(token, ` + query ApprovalComment($id: String!) { + comment(id: $id) { body issue { id } parent { id } } + }`, { id: input.id }); + if (!existing.ok) return { ok: false, retryable: true }; + const comment = existing.value.comment as { + body?: string; issue?: { id?: string }; parent?: { id?: string }; + } | undefined; + return comment?.body === input.body && comment.issue?.id === input.issueId + && comment.parent?.id === input.parentId + ? { ok: true } : { ok: false, retryable: false }; +} + /** * Upsert the orchestration's live status block — ONE comment on the parent epic * that is rewritten as the run progresses, rather than a new comment per diff --git a/cdk/test/handlers/approve-task.test.ts b/cdk/test/handlers/approve-task.test.ts index b94dacbaf..c8c12f411 100644 --- a/cdk/test/handlers/approve-task.test.ts +++ b/cdk/test/handlers/approve-task.test.ts @@ -61,7 +61,7 @@ process.env.TASK_APPROVALS_TABLE_NAME = 'Approvals'; process.env.TASK_EVENTS_TABLE_NAME = 'Events'; process.env.APPROVE_RATE_LIMIT_PER_MINUTE = '30'; -import { handler } from '../../src/handlers/approve-task'; +import { handler, recordApprovalForUser } from '../../src/handlers/approve-task'; function makeEvent(overrides: Partial = {}): APIGatewayProxyEvent { return { @@ -295,3 +295,27 @@ describe('postcommit MicroVM wake', () => { expect(mockWake).toHaveBeenCalledTimes(1); }); }); + +describe('trusted channel decision source', () => { + test('stores the source inside the guarded decision transaction', async () => { + mockSend.mockResolvedValue({}); + const result = await recordApprovalForUser({ + userId: 'user-alice', + taskId: 'task-1', + body: JSON.stringify({ request_id: 'gate', decision: 'approve' }), + decisionSource: 'linear-source', + }); + expect(result.statusCode).toBe(202); + const update = mockSend.mock.calls.find(([cmd]) => cmd._type === 'TransactWrite')![0].input.TransactItems[0].Update; + expect(update.UpdateExpression).toContain('decision_source = :source'); + expect(update.ExpressionAttributeValues[':source']).toBe('linear-source'); + expect(update.ConditionExpression).toContain('#status = :pending'); + expect(update.ConditionExpression).toContain('deadline_epoch > :epoch'); + }); + test('does not trust a source supplied in the HTTP body', async () => { + mockSend.mockResolvedValue({}); + await handler(makeEvent({ body: JSON.stringify({ request_id: 'gate', decision: 'approve', decisionSource: 'forged' }) })); + const update = mockSend.mock.calls.find(([cmd]) => cmd._type === 'TransactWrite')![0].input.TransactItems[0].Update; + expect(update.UpdateExpression).not.toContain('decision_source'); + }); +}); diff --git a/cdk/test/handlers/deny-task.test.ts b/cdk/test/handlers/deny-task.test.ts index ca0f23604..96004f043 100644 --- a/cdk/test/handlers/deny-task.test.ts +++ b/cdk/test/handlers/deny-task.test.ts @@ -53,7 +53,7 @@ process.env.TASK_TABLE_NAME = 'Tasks'; process.env.TASK_APPROVALS_TABLE_NAME = 'Approvals'; process.env.TASK_EVENTS_TABLE_NAME = 'Events'; -import { handler } from '../../src/handlers/deny-task'; +import { handler, recordDenialForUser } from '../../src/handlers/deny-task'; // Secret fixtures assembled at runtime so the source file itself // never holds a contiguous secret literal (Code Defender pre-commit @@ -264,3 +264,27 @@ describe('postcommit MicroVM wake', () => { expect(mockWake).toHaveBeenCalledTimes(1); }); }); + +describe('trusted channel decision source', () => { + test('stores the source inside the guarded decision transaction', async () => { + mockSend.mockResolvedValue({}); + const result = await recordDenialForUser({ + userId: 'user-alice', + taskId: 'task-1', + body: JSON.stringify({ request_id: 'gate', decision: 'deny' }), + decisionSource: 'linear-source', + }); + expect(result.statusCode).toBe(202); + const update = mockSend.mock.calls.find(([cmd]) => cmd._type === 'TransactWrite')![0].input.TransactItems[0].Update; + expect(update.UpdateExpression).toContain('decision_source = :source'); + expect(update.ExpressionAttributeValues[':source']).toBe('linear-source'); + expect(update.ConditionExpression).toContain('#status = :pending'); + expect(update.ConditionExpression).toContain('deadline_epoch > :epoch'); + }); + test('does not trust a source supplied in the HTTP body', async () => { + mockSend.mockResolvedValue({}); + await handler(makeEvent({ body: JSON.stringify({ request_id: 'gate', decision: 'deny', decisionSource: 'forged' }) })); + const update = mockSend.mock.calls.find(([cmd]) => cmd._type === 'TransactWrite')![0].input.TransactItems[0].Update; + expect(update.UpdateExpression).not.toContain('decision_source'); + }); +}); diff --git a/cdk/test/handlers/shared/linear-approval-thread.test.ts b/cdk/test/handlers/shared/linear-approval-thread.test.ts new file mode 100644 index 000000000..6b25e779e --- /dev/null +++ b/cdk/test/handlers/shared/linear-approval-thread.test.ts @@ -0,0 +1,77 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import type { DynamoDBDocumentClient } from '@aws-sdk/lib-dynamodb'; +import { + closeLinearApprovalThread, linearApprovalCommentId, parseLinearApprovalReply, + readLinearApprovalThread, saveLinearApprovalThread, +} from '../../../src/handlers/shared/linear-approval-thread'; +const thread = { workspaceId: 'ws', issueId: 'issue', taskId: 'task', requestId: 'gate', userId: 'owner' }; +const send = jest.fn(); +const ddb = { send } as unknown as DynamoDBDocumentClient; +beforeEach(() => send.mockReset().mockResolvedValue({})); + +test('stable UUID binds distinct workspace, issue, task and request identities', () => { + const id = linearApprovalCommentId(thread); + expect(id).toMatch(/^[0-9a-f]{8}-[0-9a-f]{4}-5[0-9a-f]{3}-a[0-9a-f]{3}-[0-9a-f]{12}$/); + expect(linearApprovalCommentId({ ...thread })).toBe(id); + for (const field of ['workspaceId', 'issueId', 'taskId', 'requestId']) { + expect(linearApprovalCommentId({ ...thread, [field]: 'other' })).not.toBe(id); + } +}); + +test('mapping writes preserve existing TTL and reject changed bindings', async () => { + await saveLinearApprovalThread(ddb, 'Approvals', thread); + const input = send.mock.calls[0][0].input; + expect(input.Key.task_id).toContain('LINEAR_COMMENT#ws#'); + expect(input.UpdateExpression).not.toContain('ttl'); + expect(input.ConditionExpression).toContain('#thread = :thread'); + expect(input.ExpressionAttributeNames).not.toHaveProperty('user_id'); +}); + +test('reads strongly and validates stored scope and shape', async () => { + const id = linearApprovalCommentId(thread); + send.mockResolvedValue({ Item: { kind: 'linear_approval_thread', thread } }); + expect(await readLinearApprovalThread(ddb, 'Approvals', 'ws', id)).toEqual(thread); + expect(send.mock.calls[0][0].input.ConsistentRead).toBe(true); + expect(await readLinearApprovalThread(ddb, 'Approvals', 'other', id)).toBeNull(); + send.mockResolvedValue({ Item: { kind: 'linear_approval_thread', thread: { ...thread, userId: undefined } } }); + expect(await readLinearApprovalThread(ddb, 'Approvals', 'ws', id)).toBeNull(); +}); + +test('closure starts retention once and never recreates a missing mapping', async () => { + await closeLinearApprovalThread(ddb, 'Approvals', thread); + const input = send.mock.calls[0][0].input; + expect(input.UpdateExpression).toContain('if_not_exists'); + expect(input.ConditionExpression).toBe('attribute_exists(task_id)'); + send.mockRejectedValue({ name: 'ConditionalCheckFailedException' }); + await expect(closeLinearApprovalThread(ddb, 'Approvals', thread)).resolves.toBeUndefined(); + send.mockRejectedValue(new Error('throttle')); + await expect(closeLinearApprovalThread(ddb, 'Approvals', thread)).rejects.toThrow('throttle'); +}); + +test.each(['approve', 'APPROVE!', ' approve. '])('accepts explicit answer %s', text => { + expect(parseLinearApprovalReply(text)).toBe('approve'); +}); +test.each(['deny', ' Deny! '])('accepts denial %s', text => { + expect(parseLinearApprovalReply(text)).toBe('deny'); +}); +test.each([undefined, 'I approve', '`approve`', 'approve\ndeny', 'approved'])('rejects ambiguous input %s', text => { + expect(parseLinearApprovalReply(text)).toBeNull(); +}); diff --git a/cdk/test/handlers/shared/linear-feedback.test.ts b/cdk/test/handlers/shared/linear-feedback.test.ts index 68f124349..791fa5bdd 100644 --- a/cdk/test/handlers/shared/linear-feedback.test.ts +++ b/cdk/test/handlers/shared/linear-feedback.test.ts @@ -33,6 +33,7 @@ import { deleteComment, fetchRecentComments, postIssueComment, + postIdentifiedComment, reactToComment, replyToComment, reportIssueFailure, @@ -73,6 +74,44 @@ describe('linear-feedback', () => { fetchMock.mockResolvedValue(jsonResponse({ data: { commentCreate: { success: true } } })); }); + describe('postIdentifiedComment', () => { + const input = { id: 'stable', issueId: ISSUE_ID, body: 'Approval needed', parentId: 'root' }; + test('passes the stable identity and exact thread to Linear', async () => { + expect(await postIdentifiedComment(CTX, input)).toEqual({ ok: true }); + expect(JSON.parse(fetchMock.mock.calls[0][1].body).variables).toEqual({ input }); + }); + test('recovers a lost creation response by verifying saved content and destination', async () => { + fetchMock.mockRejectedValueOnce(new Error('lost response')); + fetchMock.mockResolvedValueOnce(jsonResponse({ + data: { + comment: { + body: input.body, issue: { id: ISSUE_ID }, parent: { id: 'root' }, + }, + }, + })); + expect(await postIdentifiedComment(CTX, input)).toEqual({ ok: true }); + }); + test('does not accept an ID collision in another issue or thread', async () => { + fetchMock.mockResolvedValueOnce(jsonResponse({ errors: ['already exists'] })); + fetchMock.mockResolvedValueOnce(jsonResponse({ + data: { + comment: { + body: input.body, issue: { id: 'other' }, parent: { id: 'root' }, + }, + }, + })); + expect(await postIdentifiedComment(CTX, input)).toEqual({ ok: false, retryable: false }); + }); + test('does not treat success=false as delivery', async () => { + fetchMock.mockResolvedValue(jsonResponse({ data: { commentCreate: { success: false } } })); + expect(await postIdentifiedComment(CTX, input)).toEqual({ ok: false, retryable: false }); + }); + test('retries when creation and verification are both unavailable', async () => { + fetchMock.mockRejectedValue(new Error('outage')); + expect(await postIdentifiedComment(CTX, input)).toEqual({ ok: false, retryable: true }); + }); + }); + describe('postIssueComment', () => { test('POSTs the commentCreate mutation with the issue id and body', async () => { const result = await postIssueComment(CTX, ISSUE_ID, '❌ blocked'); From b3b164e8ec84880729b75598c7c8d7ef20ff5c42 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Fri, 18 Sep 2026 10:45:58 -0400 Subject: [PATCH 083/149] feat(linear): accept owner approval replies and wake waiting workers --- cdk/src/constructs/linear-integration.ts | 29 +++- cdk/src/handlers/fanout-task-events.ts | 23 ++- cdk/src/handlers/linear-webhook-processor.ts | 11 ++ .../handlers/shared/approval-notifications.ts | 4 + .../handlers/shared/linear-approval-reply.ts | 120 ++++++++++++++ cdk/src/stacks/agent.ts | 3 + .../constructs/linear-integration.test.ts | 51 ++++++ cdk/test/handlers/fanout-task-events.test.ts | 17 +- .../shared/linear-approval-reply.test.ts | 156 ++++++++++++++++++ 9 files changed, 401 insertions(+), 13 deletions(-) create mode 100644 cdk/src/handlers/shared/linear-approval-reply.ts create mode 100644 cdk/test/handlers/shared/linear-approval-reply.test.ts diff --git a/cdk/src/constructs/linear-integration.ts b/cdk/src/constructs/linear-integration.ts index 4b63cbe19..21b8c1398 100644 --- a/cdk/src/constructs/linear-integration.ts +++ b/cdk/src/constructs/linear-integration.ts @@ -18,7 +18,7 @@ */ import * as path from 'path'; -import { ArnFormat, Aspects, Duration, RemovalPolicy, Stack } from 'aws-cdk-lib'; +import { ArnFormat, Aspects, Duration, Fn, RemovalPolicy, Stack } from 'aws-cdk-lib'; import * as apigw from 'aws-cdk-lib/aws-apigateway'; import * as cognito from 'aws-cdk-lib/aws-cognito'; import * as dynamodb from 'aws-cdk-lib/aws-dynamodb'; @@ -78,6 +78,11 @@ export interface LinearIntegrationProps { /** The DynamoDB task events table. */ readonly taskEventsTable: dynamodb.ITable; + /** Enables task-owner decisions from replies to Linear approval comments. */ + readonly taskApprovalsTable?: dynamodb.ITable; + readonly lambdaMicrovmImageArn?: string; + readonly continuationBucketName?: string; + /** Monthly user/team budget configuration and spend table. */ readonly budgetTable?: dynamodb.ITable; @@ -244,6 +249,12 @@ export class LinearIntegration extends Construct { TASK_EVENTS_TABLE_NAME: props.taskEventsTable.tableName, TASK_RETENTION_DAYS: String(props.taskRetentionDays ?? DEFAULT_TASK_RETENTION_DAYS), }; + if (props.taskApprovalsTable) { + createTaskEnv.TASK_APPROVALS_TABLE_NAME = props.taskApprovalsTable.tableName; + } + if (props.continuationBucketName) { + createTaskEnv.CONTINUATION_BUCKET_NAME = props.continuationBucketName; + } if (props.repoTable) { createTaskEnv.REPO_TABLE_NAME = props.repoTable.tableName; } @@ -303,7 +314,7 @@ export class LinearIntegration extends Construct { }), // Throttle the seed-time root release to the free concurrency // budget (see prop doc). Only wired when both tables are present. - ...(props.orchestrationTable && props.userConcurrencyTable && { + ...((props.orchestrationTable || props.continuationBucketName) && props.userConcurrencyTable && { USER_CONCURRENCY_TABLE_NAME: props.userConcurrencyTable.tableName, MAX_CONCURRENT_TASKS_PER_USER: String(props.maxConcurrentTasksPerUser ?? 10), }), @@ -388,6 +399,20 @@ export class LinearIntegration extends Construct { } props.taskTable.grantReadWriteData(webhookProcessorFn); props.taskEventsTable.grantReadWriteData(webhookProcessorFn); + props.taskApprovalsTable?.grantReadWriteData(webhookProcessorFn); + if (props.taskApprovalsTable && props.lambdaMicrovmImageArn) { + webhookProcessorFn.addToRolePolicy(new iam.PolicyStatement({ + actions: ['lambda:GetMicrovm', 'lambda:ResumeMicrovm'], + resources: [props.lambdaMicrovmImageArn, `${props.lambdaMicrovmImageArn}:*`], + })); + } + if (props.continuationBucketName && props.orchestratorFunctionArn) { + props.userConcurrencyTable?.grantReadWriteData(webhookProcessorFn); + const coordinatorArn = Fn.join(':', Array.from({ length: 7 }, (_, i) => Fn.select(i, Fn.split(':', props.orchestratorFunctionArn!)))); + webhookProcessorFn.addToRolePolicy(new iam.PolicyStatement({ + actions: ['lambda:InvokeFunction'], resources: [`${coordinatorArn}:*`], + })); + } if (props.repoTable) { props.repoTable.grantReadData(webhookProcessorFn); } diff --git a/cdk/src/handlers/fanout-task-events.ts b/cdk/src/handlers/fanout-task-events.ts index da612296c..686cf7bb9 100644 --- a/cdk/src/handlers/fanout-task-events.ts +++ b/cdk/src/handlers/fanout-task-events.ts @@ -63,7 +63,8 @@ import { renderJiraFinishedPointer, type JiraFinishedPointerKind, } from './shared/jira-status-comment'; -import { EMOJI_FAILURE, EMOJI_NEEDS_INPUT, EMOJI_SUCCESS, postIssueComment, swapCommentReaction, upsertThreadedReply } from './shared/linear-feedback'; +import { closeLinearApprovalThread, saveLinearApprovalThread } from './shared/linear-approval-thread'; +import { postIdentifiedComment, EMOJI_FAILURE, EMOJI_NEEDS_INPUT, EMOJI_SUCCESS, postIssueComment, swapCommentReaction, upsertThreadedReply } from './shared/linear-feedback'; import { logger } from './shared/logger'; import { coerceNumericOrNull } from './shared/numeric'; import { loadRepoConfig } from './shared/repo-config'; @@ -1158,11 +1159,20 @@ async function dispatchToLinear(event: FanOutEvent): Promise { if (isApprovalNotification(effectiveType)) { const notification = await loadApprovalNotification(ddb, task, effectiveType, event.metadata ?? {}, 'linear'); if (!notification) return; - const result = await postIssueComment( - { linearWorkspaceId: workspaceId, registryTableName }, + const thread = { + workspaceId, issueId, - approvalNotificationMarkdown(notification), - ); + taskId: task.task_id, + requestId: notification.requestId, + userId: notification.userId, + }; + const ctx = { linearWorkspaceId: workspaceId, registryTableName }; + const body = approvalNotificationMarkdown(notification); + const result = effectiveType === 'approval_requested' + ? await postIdentifiedComment(ctx, { + id: await saveLinearApprovalThread(ddb, process.env.TASK_APPROVALS_TABLE_NAME!, thread), issueId, body, + }) + : await postIssueComment(ctx, issueId, body); if (!result.ok) { logger.warn('Linear approval notification failed', { event: 'fanout.linear.approval_post_failed', @@ -1173,6 +1183,9 @@ async function dispatchToLinear(event: FanOutEvent): Promise { if (result.retryable) throw new Error('Retryable Linear approval notification failure'); return; } + if (effectiveType !== 'approval_requested') { + await closeLinearApprovalThread(ddb, process.env.TASK_APPROVALS_TABLE_NAME!, thread); + } await markApprovalNotificationDelivered(ddb, notification); logger.info('Linear approval notification delivered', { event: 'fanout.linear.approval_dispatched', task_id: task.task_id, request_id: notification.requestId, diff --git a/cdk/src/handlers/linear-webhook-processor.ts b/cdk/src/handlers/linear-webhook-processor.ts index 72422565d..0e0b86e8d 100644 --- a/cdk/src/handlers/linear-webhook-processor.ts +++ b/cdk/src/handlers/linear-webhook-processor.ts @@ -26,6 +26,7 @@ import type { ScreeningConfig } from './shared/attachment-screening'; import { buildClarifyResumeDescription, isClarifyHold } from './shared/clarify-resume'; import { createTaskCore } from './shared/create-task-core'; import { renderMaturingReply } from './shared/iteration-reply'; +import { handleLinearApprovalReply } from './shared/linear-approval-reply'; import { cleanupPreScreenedAttachments, downloadScreenAndStoreLinearAttachments, LinearAttachmentError } from './shared/linear-attachments'; import { deleteComment, @@ -1817,6 +1818,15 @@ async function handleNearMissMention(payload: LinearCommentEvent): Promise * a clean no-op (no failure comment — comments are conversational). */ async function handleCommentTrigger(payload: LinearCommentEvent): Promise { + if (process.env.TASK_APPROVALS_TABLE_NAME && WORKSPACE_REGISTRY_TABLE + && await handleLinearApprovalReply(payload, { + ddb, + approvalsTable: process.env.TASK_APPROVALS_TABLE_NAME, + taskTable: process.env.TASK_TABLE_NAME!, + registryTable: WORKSPACE_REGISTRY_TABLE, + lookupUser: lookupPlatformUser, + })) return; + // Orchestration must be enabled + a workspace token resolvable. if (!ORCHESTRATION_TABLE || !WORKSPACE_REGISTRY_TABLE) { return; @@ -3271,6 +3281,7 @@ async function lookupPlatformUser(workspaceId: string, userId: string): Promise< const result = await ddb.send(new GetCommand({ TableName: USER_MAPPING_TABLE, Key: { linear_identity: key }, + ConsistentRead: true, })); if (!result.Item || result.Item.status === 'pending') return null; return (result.Item.platform_user_id as string) ?? null; diff --git a/cdk/src/handlers/shared/approval-notifications.ts b/cdk/src/handlers/shared/approval-notifications.ts index 8158287b7..3c8170ebf 100644 --- a/cdk/src/handlers/shared/approval-notifications.ts +++ b/cdk/src/handlers/shared/approval-notifications.ts @@ -112,6 +112,10 @@ export async function loadApprovalNotification( if (Number.isFinite(deadline) && Number(row.timeout_s) > 0) { lines.push(`Decision deadline: ${new Date(deadline).toISOString()}`); } + if (channel === 'linear') { + lines.push('Reply approve or deny to this comment while signed in as the task owner.', + 'Approval allows this action once. You can also use the CLI below.'); + } // Never interpolate untrusted text into suggested shell commands. if ([task.task_id, requestId].every(id => /^[A-Za-z0-9_-]{1,128}$/.test(id))) { lines.push('Respond using the CLI while signed in as the task owner:', diff --git a/cdk/src/handlers/shared/linear-approval-reply.ts b/cdk/src/handlers/shared/linear-approval-reply.ts new file mode 100644 index 000000000..ce9181771 --- /dev/null +++ b/cdk/src/handlers/shared/linear-approval-reply.ts @@ -0,0 +1,120 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import { GetCommand, type DynamoDBDocumentClient } from '@aws-sdk/lib-dynamodb'; +import { + closeLinearApprovalThread, linearApprovalCommentId, parseLinearApprovalReply, readLinearApprovalThread, +} from './linear-approval-thread'; +import { postIdentifiedComment } from './linear-feedback'; +import { logger } from './logger'; + +interface ApprovalReplyEvent { + action: string; + organizationId?: string; + actor?: { id?: string }; + data: { id: string; body?: string; parentId?: string; issueId?: string; issue?: { id?: string } }; +} + +interface ApprovalReplyDependencies { + ddb: DynamoDBDocumentClient; + approvalsTable: string; + taskTable: string; + registryTable: string; + lookupUser: (workspaceId: string, actorId: string) => Promise; +} + +/** A verified webhook may decide only the exact gate bound to its thread. */ +export async function handleLinearApprovalReply( + event: ApprovalReplyEvent, deps: ApprovalReplyDependencies, +): Promise { + const decision = parseLinearApprovalReply(event.data.body); + const workspaceId = event.organizationId; + const parentId = event.data.parentId; + if (event.action !== 'create' || !decision || !workspaceId || !parentId || !event.data.id) return false; + const thread = await readLinearApprovalThread(deps.ddb, deps.approvalsTable, workspaceId, parentId); + if (!thread) return false; + const issueId = event.data.issueId ?? event.data.issue?.id; + if (issueId !== thread.issueId) return true; + const ctx = { linearWorkspaceId: workspaceId, registryTableName: deps.registryTable }; + const respond = async (body: string): Promise => { + const posted = await postIdentifiedComment(ctx, { + id: linearApprovalCommentId({ ...thread, requestId: `${thread.requestId}#reply#${event.data.id}` }), + issueId, + parentId, + body, + }); + if (!posted.ok) { + logger.warn('Linear approval acknowledgement failed', { + task_id: thread.taskId, request_id: thread.requestId, comment_id: event.data.id, retryable: posted.retryable, + }); + if (posted.retryable) throw new Error('Retryable Linear approval acknowledgement failure'); + } + }; + const userId = event.actor?.id ? await deps.lookupUser(workspaceId, event.actor.id) : null; + if (!userId || userId !== thread.userId) { + await respond('Only the task owner can decide this request. Link your Linear account to ABCA, then reply again.'); + return true; + } + const task = (await deps.ddb.send(new GetCommand({ + TableName: deps.taskTable, Key: { task_id: thread.taskId }, ConsistentRead: true, + }))).Item; + if (!task || task.user_id !== userId || task.channel_source !== 'linear' + || task.channel_metadata?.linear_workspace_id !== workspaceId + || task.channel_metadata?.linear_issue_id !== issueId) { + await respond('This approval is no longer available for this task. No decision was recorded.'); + return true; + } + const source = JSON.stringify(['linear', workspaceId, event.data.id]); + const expectedStatus = decision === 'approve' ? 'APPROVED' : 'DENIED'; + const readApproval = async () => (await deps.ddb.send(new GetCommand({ + TableName: deps.approvalsTable, Key: { task_id: thread.taskId, request_id: thread.requestId }, ConsistentRead: true, + }))).Item; + const alreadyRecorded = (row: Awaited>) => + row?.user_id === userId && row.status === expectedStatus && row.decision_source === source; + let recorded = alreadyRecorded(await readApproval()); + if (!recorded) { + // Loaded only for configured approval integrations; ordinary comment paths + // do not require the decision handlers' environment variables. + const record = decision === 'approve' + ? (await import('../approve-task.js')).recordApprovalForUser + : (await import('../deny-task.js')).recordDenialForUser; + const result = await record({ + userId, + taskId: thread.taskId, + decisionSource: source, + body: JSON.stringify({ request_id: thread.requestId, decision }), + }); + recorded = result.statusCode === 202 || alreadyRecorded(await readApproval()); + if (!recorded) { + if (result.statusCode >= 500) throw new Error(`Linear approval decision failed: HTTP ${result.statusCode}`); + await respond(result.statusCode === 429 + ? 'Too many approval decisions. Wait a minute, then send a new reply.' + : 'This request is already closed, expired, or no longer waiting. No new decision was recorded.'); + return true; + } + } + await closeLinearApprovalThread(deps.ddb, deps.approvalsTable, thread); + await respond(decision === 'approve' + ? 'Approved for this action once. The decision is saved; the agent will continue when its worker is ready.' + : 'Denied. The decision is saved; the agent will receive your denial when its worker is ready.'); + logger.info('Linear approval reply recorded', { + task_id: thread.taskId, request_id: thread.requestId, comment_id: event.data.id, decision, + }); + return true; +} diff --git a/cdk/src/stacks/agent.ts b/cdk/src/stacks/agent.ts index cf4665b01..508d1dd72 100644 --- a/cdk/src/stacks/agent.ts +++ b/cdk/src/stacks/agent.ts @@ -1575,6 +1575,9 @@ export class AgentStack extends Stack { userPool: taskApi.userPool, taskTable: taskTable.table, taskEventsTable: taskEventsTable.table, + taskApprovalsTable: taskApprovalsTable.table, + lambdaMicrovmImageArn: lambdaMicrovm?.imageArn, + continuationBucketName: continuationBucket?.bucket.bucketName, budgetTable: budgetTable.table, repoTable: repoTable.table, // Enables the webhook processor's orchestration path diff --git a/cdk/test/constructs/linear-integration.test.ts b/cdk/test/constructs/linear-integration.test.ts index 9638d24e1..98a719aac 100644 --- a/cdk/test/constructs/linear-integration.test.ts +++ b/cdk/test/constructs/linear-integration.test.ts @@ -404,3 +404,54 @@ describe('revoked-authorization recording (#812)', () => { expect(JSON.stringify(registryStatements.map((s) => s.Action))).toContain('dynamodb:UpdateItem'); }); }); + +describe('Linear approval decision permissions', () => { + let template: Template; + beforeAll(() => { + const app = new App(); + const stack = new Stack(app, 'ApprovalStack'); + const table = (id: string) => new dynamodb.Table(stack, id, { + partitionKey: { name: 'id', type: dynamodb.AttributeType.STRING }, + }); + new LinearIntegration(stack, 'Linear', { + api: new apigw.RestApi(stack, 'Api'), + userPool: new cognito.UserPool(stack, 'Users'), + taskTable: table('Tasks'), + taskEventsTable: table('Events'), + taskApprovalsTable: table('Approvals'), + userConcurrencyTable: table('Counters'), + continuationBucketName: 'continuations', + orchestratorFunctionArn: 'arn:aws:lambda:us-east-1:123456789012:function:coordinator:live', + lambdaMicrovmImageArn: 'arn:aws:lambda:us-east-1:123456789012:microvm-image:test', + }); + template = Template.fromStack(stack); + }); + test('configures the processor even without orchestration', () => { + template.hasResourceProperties('AWS::Lambda::Function', { + Environment: { + Variables: Match.objectLike({ + TASK_APPROVALS_TABLE_NAME: Match.anyValue(), + USER_CONCURRENCY_TABLE_NAME: Match.anyValue(), + CONTINUATION_BUCKET_NAME: 'continuations', + }), + }, + }); + }); + test('grants wake only for the configured image and continuation dispatch for its coordinator', () => { + const functions = template.findResources('AWS::Lambda::Function'); + const processor = Object.values(functions).find(fn => fn.Properties.Environment?.Variables?.TASK_APPROVALS_TABLE_NAME); + const role = processor!.Properties.Role['Fn::GetAtt'][0]; + const policies = Object.values(template.findResources('AWS::IAM::Policy')) + .filter(p => p.Properties.Roles?.some((r: { Ref?: string }) => r.Ref === role)); + const statements = policies.flatMap(p => p.Properties.PolicyDocument.Statement); + const wake = statements.find(s => Array.isArray(s.Action) && s.Action.includes('lambda:ResumeMicrovm')); + expect(wake.Action).toEqual(['lambda:GetMicrovm', 'lambda:ResumeMicrovm']); + expect(wake.Resource).toEqual(['arn:aws:lambda:us-east-1:123456789012:microvm-image:test', + 'arn:aws:lambda:us-east-1:123456789012:microvm-image:test:*']); + expect(statements).toContainEqual(expect.objectContaining({ + Action: 'lambda:InvokeFunction', Resource: 'arn:aws:lambda:us-east-1:123456789012:function:coordinator:*', + })); + expect(statements.some(s => (Array.isArray(s.Action) ? s.Action : [s.Action]).includes('dynamodb:UpdateItem') + && JSON.stringify(s.Resource).includes('Counters'))).toBe(true); + }); +}); diff --git a/cdk/test/handlers/fanout-task-events.test.ts b/cdk/test/handlers/fanout-task-events.test.ts index ce180a0cb..7cc08557f 100644 --- a/cdk/test/handlers/fanout-task-events.test.ts +++ b/cdk/test/handlers/fanout-task-events.test.ts @@ -100,6 +100,7 @@ jest.mock('../../src/handlers/slack-notify', () => { // + GraphQL path. Default ``{ ok: true }`` so a test that forgets to // script the mock still drives the happy path (postIssueComment returns // a LinearPostResult, not a bare boolean). +const mockPostIdentifiedComment: jest.Mock = jest.fn().mockResolvedValue({ ok: true }); const mockPostIssueComment: jest.Mock = jest.fn().mockResolvedValue({ ok: true }); // Standalone comment-triggered iterations get a threaded reply to // the human's @bgagent comment, on top of the metrics comment. replyToComment @@ -110,6 +111,7 @@ const mockReplyToComment: jest.Mock = jest.fn().mockResolvedValue('reply-id'); // rather than posting a fresh replyToComment. const mockUpsertThreadedReply: jest.Mock = jest.fn().mockResolvedValue('reply-id'); jest.mock('../../src/handlers/shared/linear-feedback', () => ({ + postIdentifiedComment: (...args: unknown[]) => mockPostIdentifiedComment(...args), postIssueComment: ( ctx: { linearWorkspaceId: string; registryTableName: string }, issueId: string, @@ -780,6 +782,7 @@ describe('fanout-task-events: GitHub dispatcher (Chunk J)', () => { // task short-circuits inside the dispatcher (channel_source === // 'api' / 'github'). Pre-existing tests don't assert on it. mockPostIssueComment.mockReset().mockResolvedValue({ ok: true }); + mockPostIdentifiedComment.mockReset().mockResolvedValue({ ok: true }); }); test('first terminal event POSTs a new comment and persists the comment_id to TaskTable', async () => { @@ -1474,6 +1477,7 @@ describe('fanout-task-events: Linear dispatcher', () => { beforeEach(() => { mockDdbSend.mockReset().mockResolvedValue({ Item: undefined }); mockPostIssueComment.mockReset().mockResolvedValue({ ok: true }); + mockPostIdentifiedComment.mockReset().mockResolvedValue({ ok: true }); mockReplyToComment.mockReset().mockResolvedValue('reply-id'); mockUpsertThreadedReply.mockReset().mockResolvedValue('reply-id'); // Slack/GitHub mocks aren't asserted here but leaving them @@ -1525,7 +1529,7 @@ describe('fanout-task-events: Linear dispatcher', () => { }, }; }); - if (fails) mockPostIssueComment.mockResolvedValueOnce({ ok: false, retryable: true }); + if (fails) mockPostIdentifiedComment.mockResolvedValueOnce({ ok: false, retryable: true }); const outcome = await routeEvent({ task_id: 't-lin', event_id: 'approval-event', @@ -1533,17 +1537,17 @@ describe('fanout-task-events: Linear dispatcher', () => { timestamp: '2026-09-16T12:00:00Z', metadata: { milestone: 'approval_requested', request_id: 'g1' }, }); - expect(mockPostIssueComment).toHaveBeenCalledTimes(1); - expect(mockPostIssueComment.mock.calls[0][2]).toContain('bgagent approve t-lin g1 --scope this_call'); + expect(mockPostIdentifiedComment).toHaveBeenCalledTimes(1); + expect(mockPostIdentifiedComment.mock.calls[0][1].body).toContain('bgagent approve t-lin g1 --scope this_call'); expect(mockUpsertThreadedReply).not.toHaveBeenCalled(); const updates = mockDdbSend.mock.calls.filter(([c]) => c._type === 'Update'); if (fails) { expect(outcome.infraRejections).toHaveLength(1); - expect(updates).toHaveLength(0); + expect(updates).toHaveLength(1); } else { expect(outcome.infraRejections).toHaveLength(0); - expect(updates).toHaveLength(1); - expect(updates[0][0].input.ExpressionAttributeNames).toEqual({ '#marker': 'notified_linear_approval_requested' }); + expect(updates).toHaveLength(2); + expect(updates[1][0].input.ExpressionAttributeNames).toEqual({ '#marker': 'notified_linear_approval_requested' }); } }); @@ -2296,6 +2300,7 @@ describe('fanout-task-events: Jira dispatcher', () => { mockLoadRepoConfig.mockReset().mockResolvedValue(null); mockResolveGitHubToken.mockReset().mockResolvedValue('ghp_fake'); mockPostIssueComment.mockReset().mockResolvedValue({ ok: true }); + mockPostIdentifiedComment.mockReset().mockResolvedValue({ ok: true }); }); const mockGet = (item: unknown) => { diff --git a/cdk/test/handlers/shared/linear-approval-reply.test.ts b/cdk/test/handlers/shared/linear-approval-reply.test.ts new file mode 100644 index 000000000..fb87ebe3f --- /dev/null +++ b/cdk/test/handlers/shared/linear-approval-reply.test.ts @@ -0,0 +1,156 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import type { DynamoDBDocumentClient } from '@aws-sdk/lib-dynamodb'; +import { handleLinearApprovalReply } from '../../../src/handlers/shared/linear-approval-reply'; +import { linearApprovalCommentId } from '../../../src/handlers/shared/linear-approval-thread'; + +const mockApprove = jest.fn(); +const mockDeny = jest.fn(); +const mockPost = jest.fn(); +jest.mock('../../../src/handlers/approve-task.js', () => ({ recordApprovalForUser: mockApprove }), { virtual: true }); +jest.mock('../../../src/handlers/deny-task.js', () => ({ recordDenialForUser: mockDeny }), { virtual: true }); +jest.mock('../../../src/handlers/shared/linear-feedback', () => ({ + postIdentifiedComment: (...args: unknown[]) => mockPost(...args), +})); +const thread = { workspaceId: 'ws', issueId: 'issue', taskId: 'task', requestId: 'gate', userId: 'owner' }; +const root = linearApprovalCommentId(thread); +const event = { + action: 'create', + organizationId: 'ws', + actor: { id: 'actor' }, + data: { id: 'reply', body: 'approve', parentId: root, issueId: 'issue' }, +}; +const source = JSON.stringify(['linear', 'ws', 'reply']); +let approval: Record; +let task: Record; +const send = jest.fn(); +const lookupUser = jest.fn(); +const deps = { + ddb: { send } as unknown as DynamoDBDocumentClient, + approvalsTable: 'Approvals', + taskTable: 'Tasks', + registryTable: 'Registry', + lookupUser, +}; + +beforeEach(() => { + jest.clearAllMocks(); + approval = { user_id: 'owner', status: 'PENDING' }; + task = { + user_id: 'owner', + channel_source: 'linear', + status: 'AWAITING_APPROVAL', + channel_metadata: { linear_workspace_id: 'ws', linear_issue_id: 'issue' }, + }; + lookupUser.mockResolvedValue('owner'); + mockPost.mockResolvedValue({ ok: true }); + mockApprove.mockResolvedValue({ statusCode: 202 }); + mockDeny.mockResolvedValue({ statusCode: 202 }); + send.mockImplementation(async command => { + if (command.input.UpdateExpression) return {}; + if (command.input.Key.task_id.startsWith('LINEAR_COMMENT#')) { + return { Item: { kind: 'linear_approval_thread', thread } }; + } + return { Item: command.input.TableName === 'Tasks' ? task : approval }; + }); +}); + +test.each(['approve', 'deny'])('records %s for the bound request and mapped owner', async body => { + expect(await handleLinearApprovalReply({ ...event, data: { ...event.data, body } }, deps)).toBe(true); + const record = body === 'approve' ? mockApprove : mockDeny; + expect(record).toHaveBeenCalledWith({ + userId: 'owner', + taskId: 'task', + decisionSource: source, + body: JSON.stringify({ request_id: 'gate', decision: body }), + }); + expect(mockPost).toHaveBeenCalledWith(expect.anything(), expect.objectContaining({ issueId: 'issue', parentId: root })); +}); + +test.each(['update', 'remove'])('ignores %s events', async action => { + expect(await handleLinearApprovalReply({ ...event, action }, deps)).toBe(false); + expect(send).not.toHaveBeenCalled(); +}); + +test.each(['I approve', '> approve', 'approve and delete it', '@bgagent approve'])('ignores prose: %s', async body => { + expect(await handleLinearApprovalReply({ ...event, data: { ...event.data, body } }, deps)).toBe(false); + expect(send).not.toHaveBeenCalled(); +}); + +test('ignores top-level replies and unknown threads', async () => { + expect(await handleLinearApprovalReply({ ...event, data: { ...event.data, parentId: undefined } }, deps)).toBe(false); + send.mockResolvedValue({}); + expect(await handleLinearApprovalReply(event, deps)).toBe(false); + expect(mockApprove).not.toHaveBeenCalled(); +}); + +test.each([null, 'different-owner'])('rejects unmapped or different owner %s', async user => { + lookupUser.mockResolvedValue(user); + expect(await handleLinearApprovalReply(event, deps)).toBe(true); + expect(mockApprove).not.toHaveBeenCalled(); + expect(mockPost.mock.calls[0][1].body).toContain('Only the task owner'); +}); + +test('does not act across workspaces or issues', async () => { + expect(await handleLinearApprovalReply({ ...event, organizationId: 'other' }, deps)).toBe(false); + expect(await handleLinearApprovalReply({ ...event, data: { ...event.data, issueId: 'other' } }, deps)).toBe(true); + expect(mockApprove).not.toHaveBeenCalled(); + expect(mockPost).not.toHaveBeenCalled(); +}); + +test.each(['user_id', 'channel_source', 'channel_metadata'])('rejects changed task binding: %s', async field => { + task[field] = 'changed'; + await handleLinearApprovalReply(event, deps); + expect(mockApprove).not.toHaveBeenCalled(); + expect(mockPost.mock.calls[0][1].body).toContain('no longer available'); +}); + +test('acknowledges duplicate deliveries without recording again even after the task advances', async () => { + approval = { user_id: 'owner', status: 'APPROVED', decision_source: source }; + task.status = 'SUCCEEDED'; + await handleLinearApprovalReply(event, deps); + await handleLinearApprovalReply(event, deps); + expect(mockApprove).not.toHaveBeenCalled(); + expect(mockPost.mock.calls[0]).toEqual(mockPost.mock.calls[1]); +}); + +test.each([404, 409])('reports a closed or superseded gate (HTTP %s)', async statusCode => { + mockApprove.mockResolvedValue({ statusCode }); + await handleLinearApprovalReply(event, deps); + expect(mockPost.mock.calls[0][1].body).toContain('No new decision'); + expect(JSON.parse(mockApprove.mock.calls[0][0].body).request_id).toBe('gate'); +}); + +test('recognizes a duplicate that committed concurrently', async () => { + mockApprove.mockImplementation(async () => { + approval = { user_id: 'owner', status: 'APPROVED', decision_source: source }; + return { statusCode: 404 }; + }); + await handleLinearApprovalReply(event, deps); + expect(mockPost.mock.calls[0][1].body).toContain('Approved for this action once'); +}); + +test('retries transient decision and acknowledgement failures', async () => { + mockApprove.mockResolvedValue({ statusCode: 503 }); + await expect(handleLinearApprovalReply(event, deps)).rejects.toThrow('HTTP 503'); + mockApprove.mockResolvedValue({ statusCode: 202 }); + mockPost.mockResolvedValue({ ok: false, retryable: true }); + await expect(handleLinearApprovalReply(event, deps)).rejects.toThrow('acknowledgement failure'); +}); From 5ca974a22caafd765e63e4c416900366a9e8a79e Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Fri, 18 Sep 2026 10:47:54 -0400 Subject: [PATCH 084/149] docs(approvals): explain native Linear decisions and backend behavior --- docs/design/CEDAR_HITL_GATES.md | 19 +++++++++++++++---- docs/guides/USER_GUIDE.md | 19 ++++++++++++++++++- .../docs/architecture/Cedar-hitl-gates.md | 19 +++++++++++++++---- .../docs/using/Approval-gates-cedar-hitl.md | 19 ++++++++++++++++++- 4 files changed, 66 insertions(+), 10 deletions(-) diff --git a/docs/design/CEDAR_HITL_GATES.md b/docs/design/CEDAR_HITL_GATES.md index f503a3d4c..797a8f529 100644 --- a/docs/design/CEDAR_HITL_GATES.md +++ b/docs/design/CEDAR_HITL_GATES.md @@ -1526,9 +1526,8 @@ Approval events flow to the fan-out Lambda via TaskEventsTable Streams. The P3 implementation adds Slack and Linear notifications with the saved action, reason, decision deadline and exact CLI approve/deny commands. Recorded decisions, cancellations, timeouts and stranded waits also produce messages. The response -path uses the CLI owner's authentication. Native Slack approval buttons and Linear -approval replies are not implemented by this notification change; the OAuth/button -design below remains proposed. Email remains a log-only stub and GitHub does not +path supports the CLI and native Linear thread replies. Slack approval buttons +and the Slack OAuth/button design below remain proposed. Email remains a log-only stub and GitHub does not receive approval messages. Deployment status is recorded in the [P3 verification record](../verification/README.md). @@ -1539,7 +1538,19 @@ Slack and Linear route `approval_requested`, `approval_decision_recorded`, reads the current approval row and owning task before displaying a pending request, and records successful delivery per request/channel. Delivery failure does not mark the message delivered. A post that succeeds just before receipt persistence fails -can still produce a duplicate on retry. +can still produce a duplicate on retry, except for Linear approval prompts and +reply acknowledgements, which use deterministic comment IDs. + +For Linear, the platform saves a thread binding to the exact workspace, issue, +task, request and owner before posting the prompt. A verified Comment/create +webhook containing only `approve` or `deny` in that thread resolves the commenter's +linked platform identity. The owner then uses the same atomic decision function, +rate limit, deadline and current-request guards as the API. Approval grants +`this_call`. A trusted source comment ID saved in the decision transaction makes +webhook retries recognize the original result. Top-level comments, edits and +unbound threads never select a pending request. Pending bindings have no TTL; +closure starts 90-day retention. MicroVM decisions use the existing wake or +continuation path; other backends keep their existing polling behavior. **Proposed notification rate limit:** 10 approval-related messages per user per minute. This dispatcher limit is not implemented. Existing gate-creation caps and diff --git a/docs/guides/USER_GUIDE.md b/docs/guides/USER_GUIDE.md index 991b692a8..4128b7339 100644 --- a/docs/guides/USER_GUIDE.md +++ b/docs/guides/USER_GUIDE.md @@ -845,7 +845,24 @@ When a rule marked `@tier("soft")` matches a tool call: 4. The task waits for your decision without an automatic deadline by default. An explicit per-task or policy-rule timeout can limit that window. 5. On approval, the agent proceeds; on denial, the deny reason is best-effort injected back into the agent's context so it can adapt; on timeout, the gate is treated as a denial with `timed_out` as the reason. -A decision is recorded at most once per request. Replaying approve/deny on the same `(task_id, request_id)` is idempotent. +A decision is recorded at most once per request. A repeated decision cannot change an already closed request. + +### Responding in Linear + +For a Linear task, the bot posts an **Approval needed** comment with the action +and reason. Reply **approve** or **deny** in that comment's thread. You do not +need a task ID, request ID, or bot mention. Your Linear account must be linked to +the ABCA account that submitted the task. + +`approve` allows the displayed action once. The bot acknowledges the saved +decision. An old thread cannot approve a newer request; replies to closed requests +explain that no new decision was recorded. Use the CLI for broader approval scopes +or a denial reason. Editing an existing comment does not submit a decision. + +This works on every compute backend. A sleeping MicroVM wakes to receive the +decision; AgentCore and ECS receive it through their existing approval wait. +Sleep does not set the approval deadline. Requests have no automatic deadline by +default, and an explicitly configured deadline still applies. ### Listing pending approvals diff --git a/docs/src/content/docs/architecture/Cedar-hitl-gates.md b/docs/src/content/docs/architecture/Cedar-hitl-gates.md index 2e709bcb4..34d2db75b 100644 --- a/docs/src/content/docs/architecture/Cedar-hitl-gates.md +++ b/docs/src/content/docs/architecture/Cedar-hitl-gates.md @@ -1530,9 +1530,8 @@ Approval events flow to the fan-out Lambda via TaskEventsTable Streams. The P3 implementation adds Slack and Linear notifications with the saved action, reason, decision deadline and exact CLI approve/deny commands. Recorded decisions, cancellations, timeouts and stranded waits also produce messages. The response -path uses the CLI owner's authentication. Native Slack approval buttons and Linear -approval replies are not implemented by this notification change; the OAuth/button -design below remains proposed. Email remains a log-only stub and GitHub does not +path supports the CLI and native Linear thread replies. Slack approval buttons +and the Slack OAuth/button design below remain proposed. Email remains a log-only stub and GitHub does not receive approval messages. Deployment status is recorded in the [P3 verification record](/sample-autonomous-cloud-coding-agents/architecture/readme). @@ -1543,7 +1542,19 @@ Slack and Linear route `approval_requested`, `approval_decision_recorded`, reads the current approval row and owning task before displaying a pending request, and records successful delivery per request/channel. Delivery failure does not mark the message delivered. A post that succeeds just before receipt persistence fails -can still produce a duplicate on retry. +can still produce a duplicate on retry, except for Linear approval prompts and +reply acknowledgements, which use deterministic comment IDs. + +For Linear, the platform saves a thread binding to the exact workspace, issue, +task, request and owner before posting the prompt. A verified Comment/create +webhook containing only `approve` or `deny` in that thread resolves the commenter's +linked platform identity. The owner then uses the same atomic decision function, +rate limit, deadline and current-request guards as the API. Approval grants +`this_call`. A trusted source comment ID saved in the decision transaction makes +webhook retries recognize the original result. Top-level comments, edits and +unbound threads never select a pending request. Pending bindings have no TTL; +closure starts 90-day retention. MicroVM decisions use the existing wake or +continuation path; other backends keep their existing polling behavior. **Proposed notification rate limit:** 10 approval-related messages per user per minute. This dispatcher limit is not implemented. Existing gate-creation caps and diff --git a/docs/src/content/docs/using/Approval-gates-cedar-hitl.md b/docs/src/content/docs/using/Approval-gates-cedar-hitl.md index 0d29c63de..b78570952 100644 --- a/docs/src/content/docs/using/Approval-gates-cedar-hitl.md +++ b/docs/src/content/docs/using/Approval-gates-cedar-hitl.md @@ -18,7 +18,24 @@ When a rule marked `@tier("soft")` matches a tool call: 4. The task waits for your decision without an automatic deadline by default. An explicit per-task or policy-rule timeout can limit that window. 5. On approval, the agent proceeds; on denial, the deny reason is best-effort injected back into the agent's context so it can adapt; on timeout, the gate is treated as a denial with `timed_out` as the reason. -A decision is recorded at most once per request. Replaying approve/deny on the same `(task_id, request_id)` is idempotent. +A decision is recorded at most once per request. A repeated decision cannot change an already closed request. + +### Responding in Linear + +For a Linear task, the bot posts an **Approval needed** comment with the action +and reason. Reply **approve** or **deny** in that comment's thread. You do not +need a task ID, request ID, or bot mention. Your Linear account must be linked to +the ABCA account that submitted the task. + +`approve` allows the displayed action once. The bot acknowledges the saved +decision. An old thread cannot approve a newer request; replies to closed requests +explain that no new decision was recorded. Use the CLI for broader approval scopes +or a denial reason. Editing an existing comment does not submit a decision. + +This works on every compute backend. A sleeping MicroVM wakes to receive the +decision; AgentCore and ECS receive it through their existing approval wait. +Sleep does not set the approval deadline. Requests have no automatic deadline by +default, and an explicitly configured deadline still applies. ### Listing pending approvals From d921441b94f5954f0ab65a7499a78004123771a4 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Fri, 18 Sep 2026 10:59:15 -0400 Subject: [PATCH 085/149] test(linear): cover approval reply routing and retry propagation --- cdk/src/handlers/linear-webhook-processor.ts | 5 ++- .../handlers/linear-webhook-processor.test.ts | 35 +++++++++++++++++++ 2 files changed, 37 insertions(+), 3 deletions(-) diff --git a/cdk/src/handlers/linear-webhook-processor.ts b/cdk/src/handlers/linear-webhook-processor.ts index 0e0b86e8d..0b1b179d2 100644 --- a/cdk/src/handlers/linear-webhook-processor.ts +++ b/cdk/src/handlers/linear-webhook-processor.ts @@ -673,9 +673,8 @@ export async function handler(event: ProcessorEvent): Promise { return; } - // A Comment with an @bgagent mention on an orchestrated sub-issue - // re-iterates that sub-issue's PR (the reconciler then cascades the - // re-stack). Handled on a separate path from Issue → task creation. + // Comments route approval replies first, then @bgagent task/iteration + // requests. Issue events use the separate task-creation path below. if (payload.type === 'Comment') { await handleCommentTrigger(payload as LinearCommentEvent); return; diff --git a/cdk/test/handlers/linear-webhook-processor.test.ts b/cdk/test/handlers/linear-webhook-processor.test.ts index 45f8b8c61..7d60cc254 100644 --- a/cdk/test/handlers/linear-webhook-processor.test.ts +++ b/cdk/test/handlers/linear-webhook-processor.test.ts @@ -20,6 +20,11 @@ import * as fs from 'fs'; import * as path from 'path'; +const approvalReplyMock = jest.fn(); +jest.mock('../../src/handlers/shared/linear-approval-reply', () => ({ + handleLinearApprovalReply: (...args: unknown[]) => approvalReplyMock(...args), +})); + const ddbSend = jest.fn(); jest.mock('@aws-sdk/client-dynamodb', () => ({ DynamoDBClient: jest.fn(() => ({})) })); jest.mock('@aws-sdk/lib-dynamodb', () => ({ @@ -1189,3 +1194,33 @@ describe('every channel_metadata builder carries the vault fields', () => { expect(count).toBeGreaterThanOrEqual(4); }); }); + +describe('native approval reply routing', () => { + const oldTable = process.env.TASK_APPROVALS_TABLE_NAME; + afterEach(() => { + if (oldTable === undefined) delete process.env.TASK_APPROVALS_TABLE_NAME; + else process.env.TASK_APPROVALS_TABLE_NAME = oldTable; + approvalReplyMock.mockReset(); + }); + test('consumes approval replies before mention or new-task routing', async () => { + process.env.TASK_APPROVALS_TABLE_NAME = 'Approvals'; + approvalReplyMock.mockResolvedValue(true); + createTaskCoreMock.mockClear(); + const payload = { + type: 'Comment', + action: 'create', + organizationId: 'org-1', + actor: { id: 'user-1' }, + data: { id: 'reply', parentId: 'approval-root', issueId: 'issue-1', body: 'approve' }, + }; + await handler(eventWith(payload)); + expect(approvalReplyMock).toHaveBeenCalledWith(payload, expect.objectContaining({ approvalsTable: 'Approvals' })); + expect(createTaskCoreMock).not.toHaveBeenCalled(); + }); + test('lets transient approval failures reach the async retry mechanism', async () => { + process.env.TASK_APPROVALS_TABLE_NAME = 'Approvals'; + approvalReplyMock.mockRejectedValue(new Error('approval unavailable')); + await expect(handler(eventWith({ type: 'Comment', action: 'create', data: { id: 'reply', body: 'approve' } }))) + .rejects.toThrow('approval unavailable'); + }); +}); From 2e2f1ae400ff75df48b5aaa3fcaba6b947102ea8 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Fri, 18 Sep 2026 11:11:31 -0400 Subject: [PATCH 086/149] fix(approvals): retain closed Linear bindings independently of delivery --- cdk/src/handlers/fanout-task-events.ts | 7 +++--- cdk/test/handlers/fanout-task-events.test.ts | 23 ++++++++++++++++++++ 2 files changed, 27 insertions(+), 3 deletions(-) diff --git a/cdk/src/handlers/fanout-task-events.ts b/cdk/src/handlers/fanout-task-events.ts index 686cf7bb9..496c3ed5c 100644 --- a/cdk/src/handlers/fanout-task-events.ts +++ b/cdk/src/handlers/fanout-task-events.ts @@ -1168,6 +1168,10 @@ async function dispatchToLinear(event: FanOutEvent): Promise { }; const ctx = { linearWorkspaceId: workspaceId, registryTableName }; const body = approvalNotificationMarkdown(notification); + if (effectiveType !== 'approval_requested') { + // Retention follows the saved closure even if Linear cannot receive its notice. + await closeLinearApprovalThread(ddb, process.env.TASK_APPROVALS_TABLE_NAME!, thread); + } const result = effectiveType === 'approval_requested' ? await postIdentifiedComment(ctx, { id: await saveLinearApprovalThread(ddb, process.env.TASK_APPROVALS_TABLE_NAME!, thread), issueId, body, @@ -1183,9 +1187,6 @@ async function dispatchToLinear(event: FanOutEvent): Promise { if (result.retryable) throw new Error('Retryable Linear approval notification failure'); return; } - if (effectiveType !== 'approval_requested') { - await closeLinearApprovalThread(ddb, process.env.TASK_APPROVALS_TABLE_NAME!, thread); - } await markApprovalNotificationDelivered(ddb, notification); logger.info('Linear approval notification delivered', { event: 'fanout.linear.approval_dispatched', task_id: task.task_id, request_id: notification.requestId, diff --git a/cdk/test/handlers/fanout-task-events.test.ts b/cdk/test/handlers/fanout-task-events.test.ts index 7cc08557f..9a0ea81fa 100644 --- a/cdk/test/handlers/fanout-task-events.test.ts +++ b/cdk/test/handlers/fanout-task-events.test.ts @@ -1551,6 +1551,29 @@ describe('fanout-task-events: Linear dispatcher', () => { } }); + test('starts binding retention when a closed-request notice cannot be delivered', async () => { + mockDdbSend.mockImplementation(async command => { + if (command._type !== 'Get') return {}; + return { + Item: command.input.TableName === 'Approvals' + ? { user_id: 'u-1', status: 'CANCELLED', cancellation_reason: 'Cancelled by owner' } + : { ...TASK_RECORD_LINEAR, status: 'CANCELLED', awaiting_approval_request_id: 'g1' }, + }; + }); + mockPostIssueComment.mockResolvedValueOnce({ ok: false, retryable: false }); + await routeEvent({ + task_id: 't-lin', + event_id: 'closed', + event_type: 'approval_cancelled', + timestamp: '2026-09-18T12:00:00Z', + metadata: { request_id: 'g1' }, + }); + const updates = mockDdbSend.mock.calls.filter(([c]) => c._type === 'Update'); + expect(updates).toHaveLength(1); + expect(updates[0][0].input.Key.task_id).toContain('LINEAR_COMMENT#'); + expect(updates[0][0].input.UpdateExpression).toContain('if_not_exists(#ttl'); + }); + test('task_completed posts ✅ comment with cost / turns / duration on linked Linear issue', async () => { mockGet(TASK_RECORD_LINEAR); From 5c3552e9e5aafa49e6cb40f851f89e85bd28b3f3 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Fri, 18 Sep 2026 11:49:03 -0400 Subject: [PATCH 087/149] fix(linear): generate UUIDv4 approval comment identifiers --- cdk/src/handlers/shared/linear-approval-thread.ts | 4 ++-- cdk/test/handlers/shared/linear-approval-thread.test.ts | 4 ++-- 2 files changed, 4 insertions(+), 4 deletions(-) diff --git a/cdk/src/handlers/shared/linear-approval-thread.ts b/cdk/src/handlers/shared/linear-approval-thread.ts index e9a19581f..d84006626 100644 --- a/cdk/src/handlers/shared/linear-approval-thread.ts +++ b/cdk/src/handlers/shared/linear-approval-thread.ts @@ -31,13 +31,13 @@ export interface LinearApprovalThread { readonly userId: string; } -/** Stable UUID makes a retry after a lost Linear response target the same comment. */ +/** Linear requires UUIDv4; stable hash-derived bits make comment retries address the same ID. */ export function linearApprovalCommentId(thread: LinearApprovalThread): string { const hash = createHash('sha256').update(JSON.stringify([ 'abca-linear-approval-v1', thread.workspaceId, thread.issueId, thread.taskId, thread.requestId, ])).digest('hex'); const groups = hash.match(/^(.{8})(.{4}).(.{3}).(.{3})(.{12})/)!; - return `${groups[1]}-${groups[2]}-5${groups[3]}-a${groups[4]}-${groups[5]}`; + return `${groups[1]}-${groups[2]}-4${groups[3]}-a${groups[4]}-${groups[5]}`; } function threadKey(workspaceId: string, commentId: string): { task_id: string; request_id: string } { diff --git a/cdk/test/handlers/shared/linear-approval-thread.test.ts b/cdk/test/handlers/shared/linear-approval-thread.test.ts index 6b25e779e..8b6494510 100644 --- a/cdk/test/handlers/shared/linear-approval-thread.test.ts +++ b/cdk/test/handlers/shared/linear-approval-thread.test.ts @@ -27,9 +27,9 @@ const send = jest.fn(); const ddb = { send } as unknown as DynamoDBDocumentClient; beforeEach(() => send.mockReset().mockResolvedValue({})); -test('stable UUID binds distinct workspace, issue, task and request identities', () => { +test('Linear-compatible UUIDv4 binds distinct workspace, issue, task and request identities', () => { const id = linearApprovalCommentId(thread); - expect(id).toMatch(/^[0-9a-f]{8}-[0-9a-f]{4}-5[0-9a-f]{3}-a[0-9a-f]{3}-[0-9a-f]{12}$/); + expect(id).toMatch(/^[0-9a-f]{8}-[0-9a-f]{4}-4[0-9a-f]{3}-a[0-9a-f]{3}-[0-9a-f]{12}$/); expect(linearApprovalCommentId({ ...thread })).toBe(id); for (const field of ['workspaceId', 'issueId', 'taskId', 'requestId']) { expect(linearApprovalCommentId({ ...thread, [field]: 'other' })).not.toBe(id); From 24b52a6bb6cbd73b772eda35b051a692820c60bf Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Fri, 18 Sep 2026 12:01:26 -0400 Subject: [PATCH 088/149] fix(linear): bundle SDK clients for approval wake and recovery --- cdk/src/constructs/linear-integration.ts | 12 ++++++++++++ 1 file changed, 12 insertions(+) diff --git a/cdk/src/constructs/linear-integration.ts b/cdk/src/constructs/linear-integration.ts index 21b8c1398..0afb843f3 100644 --- a/cdk/src/constructs/linear-integration.ts +++ b/cdk/src/constructs/linear-integration.ts @@ -240,6 +240,18 @@ export class LinearIntegration extends Construct { // the task-orchestrator. Used by the webhook processor's PDF attachment path. const attachmentScreeningBundling: lambda.BundlingOptions = { ...commonBundling, + // Approval replies need the current MicroVM client and durable Invoke + // fields, which cannot depend on the SDK version supplied by Lambda. + ...(props.taskApprovalsTable && { + externalModules: [ + '@aws-sdk/client-dynamodb', + '@aws-sdk/client-ecs', + '@aws-sdk/client-bedrock-runtime', + '@aws-sdk/client-secrets-manager', + '@aws-sdk/lib-dynamodb', + '@aws-sdk/util-dynamodb', + ], + }), nodeModules: ['pdf-parse'], }; From 818750951a314bb0c2d73cfa5555d8227e81c66e Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Fri, 18 Sep 2026 12:17:33 -0400 Subject: [PATCH 089/149] docs(verification): record native Linear approval acceptance --- docs/verification/README.md | 11 +++++++---- 1 file changed, 7 insertions(+), 4 deletions(-) diff --git a/docs/verification/README.md b/docs/verification/README.md index 11120346c..65e3a4fd3 100644 --- a/docs/verification/README.md +++ b/docs/verification/README.md @@ -25,6 +25,7 @@ deployments. They do not certify a different image, account, Region or upgrade. | Credential renewal | A wait exceeding one hour was followed by successful AWS access with renewed task credentials. | | Continuation | Conversation and Git/workspace recovery, replacement admission, usage limits and capacity release passed. | | External integrations | Repository work and remote MCP access passed across sleep; an actual Linear submission exercised MicroVM compute with AgentCore Identity vault. | +| Linear decisions | Native threaded `approve` and `deny` replies passed on MicroVM and AgentCore. Both MicroVMs were suspended before the replies; exact decisions, tool results, thread acknowledgements and eventual capacity release were verified. | | Infrastructure | Fresh nested deployment and a deployment-specific overlapping migration passed, including compatible rollback and old-resource cleanup. | | Other backends | ECS and AgentCore approval/cancellation and scoped-access checks passed in their tested deployments. | @@ -40,12 +41,13 @@ is not established for every worker. upgrade from current `main` on the same deployment. The earlier bespoke migration is not a substitute for that acceptance. - Restore a green full build after integration with `main`. The latest recorded - full run passed 2,170 agent tests and 1,005 CLI tests; CDK had 5,417 passes, + full run passed 2,170 agent tests and 1,005 CLI tests; CDK had 5,545 passes, four failures and 56 skips. Three resource-budget cases reached 493 resources (ECS) or 492 (MicroVM), above the repository's 490-resource cushion. A separate - `github-tags` configuration-load failure passed its isolated rerun; its cause - is unresolved. These are historical run results, not a claim that the latest - commit has been fully revalidated. + VPC test failed while reading a temporary `tree.json`; its isolated rerun + passed all 18 tests without a code change. The earlier `github-tags` failure + did not recur. These are recorded run results, not a claim that the latest + commit has passed a full build. ## Reproduce local checks @@ -80,6 +82,7 @@ only after verifying that image and coordinator together. |---|---| | Normal task | Repository tools run; terminal status, payload cleanup and capacity release agree. | | Approval after sleep | The saved decision reaches the exact pending tool; verify guest recovery and tool output, not only the Resume API receipt. | +| Linear replies | Submit real issues on each backend. Reply `approve` or `deny` to the exact approval comment as its linked owner; verify the saved decision source, same-thread acknowledgement and allowed/blocked tool result. For MicroVM, observe suspension before replying. | | Deny, expiry, cancellation | No denied/cancelled tool runs; expiry uses the original deadline; compute and capacity are cleaned up. | | Default and disabled sleep | Omitted override uses 600 seconds; task-level off and deployment off prevent new suspension while wake/cleanup remain available. | | Credential expiry | Sleep past the original credential lifetime, then perform actual task-scoped AWS operations. | From 4ffe36eee0d31692230d7038a531e3d20a67c001 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Fri, 18 Sep 2026 15:29:54 -0400 Subject: [PATCH 090/149] fix(approvals): explain the action before technical details --- .../handlers/shared/approval-notifications.ts | 74 +++++++++++++++++-- .../shared/approval-notifications.test.ts | 62 ++++++++++++++++ 2 files changed, 128 insertions(+), 8 deletions(-) diff --git a/cdk/src/handlers/shared/approval-notifications.ts b/cdk/src/handlers/shared/approval-notifications.ts index 3c8170ebf..a475f1fb3 100644 --- a/cdk/src/handlers/shared/approval-notifications.ts +++ b/cdk/src/handlers/shared/approval-notifications.ts @@ -49,10 +49,57 @@ export interface ApprovalNotification { function text(value: unknown, max = 500): string { if (typeof value !== 'string') return ''; // Redact before truncation so cutting a token cannot hide its recognizable shape. - return scanDenyReason(value) + const clean = scanDenyReason(value) .replace(/\u001b\[[0-?]*[ -/]*[@-~]/g, '') - .replace(/[\u0000-\u0008\u000b-\u001f\u007f\u200e\u200f\u202a-\u202e\u2066-\u2069]/g, '') - .slice(0, max); + .replace(/[\u0000-\u0008\u000b-\u001f\u007f\u200e\u200f\u202a-\u202e\u2066-\u2069]/g, ''); + return clean.length > max ? `${clean.slice(0, max)}… [shortened]` : clean; +} + +/** Describe only recorded arguments; tool names cannot establish intent or safety. */ +function describeAction(row: Record, task: TaskRecord): string[] { + const tool = text(row.tool_name, 100); + const preview = typeof row.tool_input_preview === 'string' ? row.tool_input_preview : ''; + let input: Record | undefined; + try { + const parsed: unknown = JSON.parse(preview); + if (parsed && typeof parsed === 'object' && !Array.isArray(parsed)) input = parsed as Record; + } catch { + // The guest stores a bounded preview, which may end partway through JSON. + } + const quote = (value: unknown) => JSON.stringify(text(value)); + const path = typeof input?.file_path === 'string' ? input.file_path : ''; + const workspace = `/workspace/${task.task_id}/`; + const target = quote(path.startsWith(workspace) ? path.slice(workspace.length) : path); + let action: string; + if (tool === 'Read' && path) { + action = `The agent wants to read the file ${target}.`; + } else if (tool === 'Write' && path) { + action = `The agent wants to write to ${target}, creating the file or replacing its contents.`; + } else if (tool === 'Edit' && path) { + action = `The agent wants to change text in ${target}.`; + } else if (tool === 'Bash') { + action = 'The agent wants to run a shell command.'; + } else if (tool === 'WebFetch' && typeof input?.url === 'string') { + action = `The agent wants to fetch content from ${quote(input.url)}.`; + } else if (tool === 'Glob' || tool === 'Grep') { + action = `The agent wants to search ${tool === 'Glob' ? 'file names' : 'file contents'}.`; + } else { + action = `The agent wants to call the tool ${quote(tool)}.`; + } + const lines = [action]; + if (task.repo) lines.push(`Repository: ${text(task.repo)}`); + if (typeof input?.description === 'string' && input.description.trim()) { + lines.push(`Agent's explanation: ${text(input.description)}`); + } + if (tool === 'Bash' && typeof input?.command === 'string') { + lines.push(`Command: ${quote(input.command)}`); + } + // Always retain the preview: summaries must not hide flags, ranges or edits. + lines.push(`Saved arguments: ${text(preview) || '(not available)'}`); + if (!input || preview.endsWith('...') || preview.length > 500) { + lines.push('The saved arguments are incomplete or could not be interpreted. They may not show the full action.'); + } + return lines; } /** Read saved state instead of showing an old, delayed "please approve" event. */ @@ -104,18 +151,26 @@ export async function loadApprovalNotification( : status === 'DENIED' ? 'Denial recorded' : status === 'TIMED_OUT' ? 'Approval request timed out' : status === 'CANCELLED' ? 'Approval request cancelled' : 'Approval wait could not continue'; - const lines = [title, `Task: ${task.task_id}`, `Request: ${requestId}`]; + const lines = [title]; if (status === 'PENDING') { - lines.push(`Tool: ${text(row.tool_name, 100)}`, `Severity: ${text(row.severity, SEVERITY_MAX_LENGTH)}`, - `Reason: ${text(row.reason)}`, `Action preview: ${text(row.tool_input_preview)}`); + lines.push('', ...describeAction(row, task), ''); + const reason = text(row.reason); + lines.push(reason.startsWith('Soft-deny:') + ? 'Why approval is required: your configured policy requires a human decision for this action.' + : `Why approval is required: ${reason || 'No explanation was saved with this request.'}`); + lines.push('Approve: allow this action once.', + 'Deny: block this action and return the decision to the agent.'); const deadline = Date.parse(row.created_at) + Number(row.timeout_s) * 1000; if (Number.isFinite(deadline) && Number(row.timeout_s) > 0) { lines.push(`Decision deadline: ${new Date(deadline).toISOString()}`); } if (channel === 'linear') { - lines.push('Reply approve or deny to this comment while signed in as the task owner.', - 'Approval allows this action once. You can also use the CLI below.'); + lines.push('', 'Reply approve or deny to this comment while signed in as the task owner.'); } + lines.push('', 'Technical details / CLI alternative', + `Task: ${task.task_id}`, `Request: ${requestId}`, + `Tool: ${text(row.tool_name, 100)}`, `Policy severity: ${text(row.severity, SEVERITY_MAX_LENGTH)}`, + `Policy detail: ${reason}`); // Never interpolate untrusted text into suggested shell commands. if ([task.task_id, requestId].every(id => /^[A-Za-z0-9_-]{1,128}$/.test(id))) { lines.push('Respond using the CLI while signed in as the task owner:', @@ -124,16 +179,19 @@ export async function loadApprovalNotification( } lines.push('Run bgagent pending to see currently open requests.'); } else if (status === 'APPROVED') { + lines.push(`Task: ${task.task_id}`, `Request: ${requestId}`); lines.push(`Scope: ${text(row.scope, SCOPE_MAX_LENGTH)}`); lines.push(TERMINAL_STATUSES.includes(task.status) ? `The decision is saved. Task status: ${task.status}. The task has already ended.` : 'The decision is saved. The agent will continue when its worker is ready.'); } else if (status === 'DENIED') { + lines.push(`Task: ${task.task_id}`, `Request: ${requestId}`); lines.push(`Reason: ${text(row.deny_reason) || 'No reason supplied.'}`); lines.push(TERMINAL_STATUSES.includes(task.status) ? `The decision is saved. Task status: ${task.status}. The task has already ended.` : 'The decision is saved. The agent will receive the denial when its worker is ready.'); } else { + lines.push(`Task: ${task.task_id}`, `Request: ${requestId}`); // reason is the original policy explanation, not the cause of closure. // The guest stores a polling failure in deny_reason on its timeout path. const reason = status === 'TIMED_OUT' diff --git a/cdk/test/handlers/shared/approval-notifications.test.ts b/cdk/test/handlers/shared/approval-notifications.test.ts index 532fc87d3..a99200238 100644 --- a/cdk/test/handlers/shared/approval-notifications.test.ts +++ b/cdk/test/handlers/shared/approval-notifications.test.ts @@ -55,6 +55,68 @@ test('shows the saved action and exact deadline, with one-call CLI approval', as expect(send.mock.calls[0][0].input.ConsistentRead).toBe(true); }); +test('explains the README decision before exposing internal policy and request IDs', async () => { + send.mockResolvedValue({ + Item: { + ...row, + tool_name: 'Read', + reason: 'Soft-deny: legacy_extra_0', + tool_input_preview: JSON.stringify({ file_path: '/workspace/task/README.md' }), + }, + }); + const message = (await loadApprovalNotification(ddb, { ...task, repo: 'owner/repo' }, 'approval_requested', { request_id: 'gate' }, 'linear'))!; + expect(message.text).toContain('The agent wants to read the file "README.md".'); + expect(message.text).toContain('Repository: owner/repo'); + expect(message.text).toContain('your configured policy requires a human decision'); + expect(message.text).toContain('Approve: allow this action once.'); + expect(message.text).toContain('Deny: block this action and return the decision to the agent.'); + const primary = message.text.split('Technical details / CLI alternative')[0]; + expect(primary).toContain('Reply approve or deny to this comment'); + expect(primary).not.toMatch(/legacy_extra|Task:|Request:|severity:/); + expect(message.text).toContain('Policy detail: Soft-deny: legacy_extra_0'); +}); + +test.each([ + ['Write', { file_path: '/etc/config', content: 'replacement' }, 'creating the file or replacing its contents.'], + ['Edit', { file_path: '/workspace/task/app.py', old_string: 'before', new_string: 'after' }, 'change text in "app.py".'], + ['Bash', { command: 'git push --force', description: 'Publish the branch' }, 'Command: "git push --force"'], + ['WebFetch', { url: 'https://example.com' }, 'fetch content from "https://example.com".'], + ['Grep', { pattern: 'password', path: '/etc' }, 'search file contents.'], + ['mcp__custom__publish', { target: 'production' }, 'call the tool "mcp__custom__publish".'], +])('describes %s without hiding its saved arguments', async (tool, input, expected) => { + const preview = JSON.stringify(input); + send.mockResolvedValue({ Item: { ...row, tool_name: tool, tool_input_preview: preview } }); + const message = (await loadApprovalNotification(ddb, task, 'approval_requested', { request_id: 'gate' }, 'linear'))!; + expect(message.text).toContain(expected); + expect(message.text).toContain(`Saved arguments: ${preview}`); + if (tool === 'Bash') expect(message.text).toContain("Agent's explanation: Publish the branch"); + expect(message.text).not.toContain('incomplete'); +}); + +test('does not invent a read target from truncated arguments', async () => { + send.mockResolvedValue({ Item: { ...row, tool_name: 'Read', tool_input_preview: '{"file_path": "/workspace/task/secret...' } }); + const message = (await loadApprovalNotification(ddb, task, 'approval_requested', { request_id: 'gate' }, 'linear'))!; + expect(message.text).toContain('saved arguments are incomplete'); + expect(message.text).not.toContain('wants to read the file'); +}); + +test('redacts extracted command descriptions and makes display truncation explicit', async () => { + const token = `ghp_${'a'.repeat(36)}`; + send.mockResolvedValue({ + Item: { + ...row, + tool_input_preview: JSON.stringify({ + command: `echo ${token} ${'x'.repeat(600)}`, description: `Use ${token}`, + }), + }, + }); + const message = (await loadApprovalNotification(ddb, task, 'approval_requested', { request_id: 'gate' }, 'linear'))!; + expect(message.text).not.toContain(token); + expect(message.text).toContain('[REDACTED-GITHUB_TOKEN]'); + expect(message.text).toContain('[shortened]'); + expect(message.text).toContain('may not show the full action'); +}); + test.each([ ['cancelled task', { ...task, status: TaskStatus.CANCELLED }, row], ['new gate', { ...task, awaiting_approval_request_id: 'new' }, row], From edee8a72c6926bbe95a23de77ec8117d91439d00 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Fri, 18 Sep 2026 18:10:46 -0400 Subject: [PATCH 091/149] fix(cdk): nest concurrency maintenance to restore stack budget --- cdk/src/stacks/agent.ts | 7 ++++-- cdk/test/stacks/agent.test.ts | 46 +++++++++++++++++++++++++++-------- 2 files changed, 41 insertions(+), 12 deletions(-) diff --git a/cdk/src/stacks/agent.ts b/cdk/src/stacks/agent.ts index 508d1dd72..5607b2db9 100644 --- a/cdk/src/stacks/agent.ts +++ b/cdk/src/stacks/agent.ts @@ -19,7 +19,7 @@ import * as path from 'path'; import * as bedrock from '@aws-cdk/aws-bedrock-alpha'; -import { ArnFormat, AspectPriority, Aspects, Stack, StackProps, RemovalPolicy, CfnOutput, CfnResource, Duration, Fn, Lazy } from 'aws-cdk-lib'; +import { ArnFormat, AspectPriority, Aspects, Stack, StackProps, NestedStack, RemovalPolicy, CfnOutput, CfnResource, Duration, Fn, Lazy } from 'aws-cdk-lib'; import * as agentcore from 'aws-cdk-lib/aws-bedrockagentcore'; import * as ec2 from 'aws-cdk-lib/aws-ec2'; import * as ecr_assets from 'aws-cdk-lib/aws-ecr-assets'; @@ -1416,7 +1416,10 @@ export class AgentStack extends Stack { agentMemory.grantReadWrite(orchestrator.fn); // --- Concurrency counter reconciler (drift correction) --- - new ConcurrencyReconciler(this, 'ConcurrencyReconciler', { + // Keep this stateless repair job out of the parent resource budget. + // Existing deployments recreate its function/schedule; the tables stay put. + const concurrencyMaintenance = new NestedStack(this, 'ConcurrencyMaintenance'); + new ConcurrencyReconciler(concurrencyMaintenance, 'ConcurrencyReconciler', { taskTable: taskTable.table, userConcurrencyTable: userConcurrencyTable.table, }); diff --git a/cdk/test/stacks/agent.test.ts b/cdk/test/stacks/agent.test.ts index 1419d9060..880e56e74 100644 --- a/cdk/test/stacks/agent.test.ts +++ b/cdk/test/stacks/agent.test.ts @@ -33,6 +33,7 @@ import { AgentStack } from '../../src/stacks/agent'; describe('AgentStack', () => { let template: Template; + let concurrencyMaintenance: Template; beforeAll(() => { const app = new App(); @@ -40,12 +41,31 @@ describe('AgentStack', () => { env: { account: '123456789012', region: 'us-east-1' }, }); template = Template.fromStack(stack); + concurrencyMaintenance = Template.fromStack(stack.node.findChild('ConcurrencyMaintenance') as NestedStack); }); test('synthesizes without errors', () => { expect(template).toBeDefined(); }); + test('nests the concurrency repair job while retaining its tables in the parent', () => { + const parentFunctions = Object.keys(template.findResources('AWS::Lambda::Function')); + expect(parentFunctions.some(id => id.startsWith('ConcurrencyReconciler'))).toBe(false); + concurrencyMaintenance.resourceCountIs('AWS::Lambda::Function', 1); + concurrencyMaintenance.resourceCountIs('AWS::DynamoDB::Table', 0); + concurrencyMaintenance.hasResourceProperties('AWS::Events::Rule', { + ScheduleExpression: 'rate(15 minutes)', + }); + const fn = Object.values(concurrencyMaintenance.findResources('AWS::Lambda::Function'))[0]; + const variables = fn.Properties.Environment.Variables; + const nested = Object.entries(template.findResources('AWS::CloudFormation::Stack')) + .find(([id]) => id.startsWith('ConcurrencyMaintenanceNestedStack'))![1]; + expect(nested.Properties.Parameters[variables.TASK_TABLE_NAME.Ref]) + .toEqual({ Ref: expect.stringMatching(/^TaskTable/) }); + expect(nested.Properties.Parameters[variables.USER_CONCURRENCY_TABLE_NAME.Ref]) + .toEqual({ Ref: expect.stringMatching(/^UserConcurrencyTable/) }); + }); + test('retains the guardrail version referenced by pinned durable environments', () => { template.resourceCountIs('AWS::Bedrock::GuardrailVersion', 1); template.hasResource('AWS::Bedrock::GuardrailVersion', { @@ -1698,7 +1718,7 @@ describe('AgentStack MicroVM image ARN invariant', () => { }); describe('AgentStack solution attribution (#319): AWS_SDK_UA_APP_ID via stack-level aspect', () => { - let template: Template; + let templates: Template[]; beforeAll(() => { const app = new App(); @@ -1712,7 +1732,11 @@ describe('AgentStack solution attribution (#319): AWS_SDK_UA_APP_ID via stack-le Aspects.of(stack).add(new SolutionUaAspect(buildAppId('UaAgentStack')), { priority: AspectPriority.MUTATING, }); - template = Template.fromStack(stack); + templates = [ + Template.fromStack(stack), + ...stack.node.findAll().filter((child): child is NestedStack => NestedStack.isNestedStack(child)) + .map(child => Template.fromStack(child)), + ]; }); // CDK synthesizes its own framework-owned Lambdas that are NOT part of the @@ -1722,22 +1746,24 @@ describe('AgentStack solution attribution (#319): AWS_SDK_UA_APP_ID via stack-le // `AWS679f53fac002430cb0da5b7982bd2287…` `cr.AwsCustomResource` singleton // (which CDK happens to give the env var today, but whose attribution we do // not want to depend on across CDK upgrades). Every framework-owned id is - // enumerated explicitly so the coverage assertion below cannot silently + // enumerated explicitly, including the registry Provider's framework handlers, + // so the coverage assertion below cannot silently // stop covering an ABCA Lambda by relabelling it as "framework". const FRAMEWORK_LAMBDA_ID = - /^(CustomResourceProviderHandler|CustomS3AutoDeleteObjects|CustomVpcRestrictDefaultSG|AWS679f53fac002430cb0da5b7982bd2287)/; + /^(CustomResourceProviderHandler|CustomS3AutoDeleteObjects|CustomVpcRestrictDefaultSG|AWS679f53fac002430cb0da5b7982bd2287|AgentRegistryProviderframework)/; test('every solution Lambda carries AWS_SDK_UA_APP_ID (traverses nested scope)', () => { - const functions = template.findResources('AWS::Lambda::Function'); - const abcaLambdas = Object.entries(functions).filter( + const functions = templates.flatMap(template => Object.entries(template.findResources('AWS::Lambda::Function'))); + const abcaLambdas = functions.filter( ([id]) => !FRAMEWORK_LAMBDA_ID.test(id), ); // exact count — update when adding/removing a Lambda construct (#319). // A loose `toBeGreaterThan` let a whole integration construct disappear // unnoticed; the exact count fails if a Lambda is dropped OR if a new one // is added without being attributed below. - // 47 = 46 on main + RemoveWorkspaceFn (DELETE /v1/linear/workspaces/{slug}). - expect(abcaLambdas.length).toBe(47); + // 47 existing handlers (including concurrency repair) + 2 registry + // provisioning handlers + 4 registry API handlers in nested stacks. + expect(abcaLambdas.length).toBe(53); // Every ABCA-authored Lambda must carry the canonical `#` app-id. Collect // any offenders so a failure names the exact logical id(s) that are naked. const unattributed = abcaLambdas @@ -1754,8 +1780,8 @@ describe('AgentStack solution attribution (#319): AWS_SDK_UA_APP_ID via stack-le // The trap: these functions live inside integration constructs several // scopes below the stack. The env-var still resolves the canonical `#` // form (not the mangled `-` variant). - const functions = template.findResources('AWS::Lambda::Function'); - const nested = Object.entries(functions).filter(([id]) => + const functions = templates.flatMap(template => Object.entries(template.findResources('AWS::Lambda::Function'))); + const nested = functions.filter(([id]) => /Jira|Slack|Linear/.test(id), ); expect(nested.length).toBeGreaterThan(0); From 1a03d9342a9168eccda68d857a7253d8c5f5c70e Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Fri, 18 Sep 2026 18:19:11 -0400 Subject: [PATCH 092/149] docs(verification): record green build and stack budget fix --- docs/verification/README.md | 16 ++++++++-------- 1 file changed, 8 insertions(+), 8 deletions(-) diff --git a/docs/verification/README.md b/docs/verification/README.md index 65e3a4fd3..6c0cb9f5a 100644 --- a/docs/verification/README.md +++ b/docs/verification/README.md @@ -35,19 +35,19 @@ long-sleep controls and the corrected live flows passed. Service-side traces for the historical failures remain unavailable, so their exact transport error is not established for every worker. +The full local build after the resource-budget fix passed: 5,560 CDK tests, +2,170 agent tests and 1,005 CLI tests, plus compile, lint, contracts, docs and +synthesis. The widest parent stacks use 489 resources for ECS and 488 for +MicroVM, including synth metadata, within the unchanged 490-resource budget. +Concurrency maintenance now has its own nested stack. Upgrading recreates that +stateless repair function and schedule; task and concurrency tables stay in the +parent stack. + ## Open PR checks - Finish reusable flat-to-nested migration commands and independently test an upgrade from current `main` on the same deployment. The earlier bespoke migration is not a substitute for that acceptance. -- Restore a green full build after integration with `main`. The latest recorded - full run passed 2,170 agent tests and 1,005 CLI tests; CDK had 5,545 passes, - four failures and 56 skips. Three resource-budget cases reached 493 resources - (ECS) or 492 (MicroVM), above the repository's 490-resource cushion. A separate - VPC test failed while reading a temporary `tree.json`; its isolated rerun - passed all 18 tests without a code change. The earlier `github-tags` failure - did not recur. These are recorded run results, not a claim that the latest - commit has passed a full build. ## Reproduce local checks From 95591d44a92ee72759b34b4163b928c783e2831d Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Fri, 18 Sep 2026 19:12:17 -0400 Subject: [PATCH 093/149] fix(linear): verify approval replies against authoritative comments --- .../handlers/shared/linear-approval-reply.ts | 13 ++++++- cdk/src/handlers/shared/linear-feedback.ts | 30 +++++++++++++++ .../shared/linear-approval-reply.test.ts | 37 +++++++++++++++++++ .../handlers/shared/linear-feedback.test.ts | 36 ++++++++++++++++++ 4 files changed, 114 insertions(+), 2 deletions(-) diff --git a/cdk/src/handlers/shared/linear-approval-reply.ts b/cdk/src/handlers/shared/linear-approval-reply.ts index ce9181771..11274639a 100644 --- a/cdk/src/handlers/shared/linear-approval-reply.ts +++ b/cdk/src/handlers/shared/linear-approval-reply.ts @@ -21,7 +21,7 @@ import { GetCommand, type DynamoDBDocumentClient } from '@aws-sdk/lib-dynamodb'; import { closeLinearApprovalThread, linearApprovalCommentId, parseLinearApprovalReply, readLinearApprovalThread, } from './linear-approval-thread'; -import { postIdentifiedComment } from './linear-feedback'; +import { postIdentifiedComment, readLinearApprovalComment } from './linear-feedback'; import { logger } from './logger'; interface ApprovalReplyEvent { @@ -52,6 +52,15 @@ export async function handleLinearApprovalReply( const issueId = event.data.issueId ?? event.data.issue?.id; if (issueId !== thread.issueId) return true; const ctx = { linearWorkspaceId: workspaceId, registryTableName: deps.registryTable }; + const comment = await readLinearApprovalComment(ctx, event.data.id); + if (!comment || comment.id !== event.data.id || comment.botActor || !comment.user?.id + || comment.issue?.id !== issueId || comment.parent?.id !== parentId + || parseLinearApprovalReply(comment.body) !== decision) { + logger.warn('Linear approval webhook does not match a human comment', { + task_id: thread.taskId, request_id: thread.requestId, comment_id: event.data.id, + }); + return true; + } const respond = async (body: string): Promise => { const posted = await postIdentifiedComment(ctx, { id: linearApprovalCommentId({ ...thread, requestId: `${thread.requestId}#reply#${event.data.id}` }), @@ -66,7 +75,7 @@ export async function handleLinearApprovalReply( if (posted.retryable) throw new Error('Retryable Linear approval acknowledgement failure'); } }; - const userId = event.actor?.id ? await deps.lookupUser(workspaceId, event.actor.id) : null; + const userId = await deps.lookupUser(workspaceId, comment.user.id); if (!userId || userId !== thread.userId) { await respond('Only the task owner can decide this request. Link your Linear account to ABCA, then reply again.'); return true; diff --git a/cdk/src/handlers/shared/linear-feedback.ts b/cdk/src/handlers/shared/linear-feedback.ts index 6e331a844..9f09b41e0 100644 --- a/cdk/src/handlers/shared/linear-feedback.ts +++ b/cdk/src/handlers/shared/linear-feedback.ts @@ -408,6 +408,36 @@ export async function postIssueComment( return graphqlRequest(token, COMMENT_CREATE_MUTATION, { issueId, body }); } +/** + * Read consent from Linear itself. Webhook authentication alone is insufficient: + * legacy worker credentials can read the OAuth bundle containing the HMAC key. + * Lookup failures must retry, never become permission to use webhook fields. + */ +export async function readLinearApprovalComment( + ctx: LinearFeedbackContext, + id: string, +): Promise<{ + id: string; + body: string; + user?: { id: string } | null; + botActor?: { id: string } | null; + issue: { id: string }; + parent?: { id: string } | null; +} | null> { + const token = await resolveToken(ctx); + if (!token) throw new Error('Linear approval verification token unavailable'); + const result = await graphqlData(token, ` + query VerifyApprovalComment($id: String!) { + organization { id } + comment(id: $id) { id body user { id } botActor { id } issue { id } parent { id } } + }`, { id }); + if (!result.ok) throw new Error('Linear approval comment verification unavailable'); + if ((result.value.organization as { id?: string } | undefined)?.id !== ctx.linearWorkspaceId) { + throw new Error('Linear approval verification workspace mismatch'); + } + return result.value.comment as Awaited> ?? null; +} + /** Retry-safe posting for approval prompts and acknowledgements. */ export async function postIdentifiedComment( ctx: LinearFeedbackContext, diff --git a/cdk/test/handlers/shared/linear-approval-reply.test.ts b/cdk/test/handlers/shared/linear-approval-reply.test.ts index fb87ebe3f..2fda92995 100644 --- a/cdk/test/handlers/shared/linear-approval-reply.test.ts +++ b/cdk/test/handlers/shared/linear-approval-reply.test.ts @@ -24,10 +24,12 @@ import { linearApprovalCommentId } from '../../../src/handlers/shared/linear-app const mockApprove = jest.fn(); const mockDeny = jest.fn(); const mockPost = jest.fn(); +const mockReadComment = jest.fn(); jest.mock('../../../src/handlers/approve-task.js', () => ({ recordApprovalForUser: mockApprove }), { virtual: true }); jest.mock('../../../src/handlers/deny-task.js', () => ({ recordDenialForUser: mockDeny }), { virtual: true }); jest.mock('../../../src/handlers/shared/linear-feedback', () => ({ postIdentifiedComment: (...args: unknown[]) => mockPost(...args), + readLinearApprovalComment: (...args: unknown[]) => mockReadComment(...args), })); const thread = { workspaceId: 'ws', issueId: 'issue', taskId: 'task', requestId: 'gate', userId: 'owner' }; const root = linearApprovalCommentId(thread); @@ -61,6 +63,13 @@ beforeEach(() => { }; lookupUser.mockResolvedValue('owner'); mockPost.mockResolvedValue({ ok: true }); + mockReadComment.mockResolvedValue({ + id: 'reply', + body: 'approve', + user: { id: 'real-author' }, + issue: { id: 'issue' }, + parent: { id: root }, + }); mockApprove.mockResolvedValue({ statusCode: 202 }); mockDeny.mockResolvedValue({ statusCode: 202 }); send.mockImplementation(async command => { @@ -73,6 +82,9 @@ beforeEach(() => { }); test.each(['approve', 'deny'])('records %s for the bound request and mapped owner', async body => { + mockReadComment.mockResolvedValue({ + id: 'reply', body, user: { id: 'real-author' }, issue: { id: 'issue' }, parent: { id: root }, + }); expect(await handleLinearApprovalReply({ ...event, data: { ...event.data, body } }, deps)).toBe(true); const record = body === 'approve' ? mockApprove : mockDeny; expect(record).toHaveBeenCalledWith({ @@ -82,6 +94,31 @@ test.each(['approve', 'deny'])('records %s for the bound request and mapped owne body: JSON.stringify({ request_id: 'gate', decision: body }), }); expect(mockPost).toHaveBeenCalledWith(expect.anything(), expect.objectContaining({ issueId: 'issue', parentId: root })); + expect(lookupUser).toHaveBeenCalledWith('ws', 'real-author'); +}); + +test.each([ + null, + { body: 'deny' }, + { user: null }, + { botActor: { id: 'app' } }, + { issue: { id: 'elsewhere' } }, + { parent: { id: 'other-gate' } }, + { id: 'other-reply' }, +])('rejects forged, edited, deleted or bot consent: %j', async change => { + const original = await mockReadComment(); + mockReadComment.mockResolvedValue(change === null ? null : { ...original, ...change }); + expect(await handleLinearApprovalReply(event, deps)).toBe(true); + expect(mockApprove).not.toHaveBeenCalled(); + expect(mockDeny).not.toHaveBeenCalled(); + expect(mockPost).not.toHaveBeenCalled(); +}); + +test('retries unavailable verification without trusting the webhook actor', async () => { + mockReadComment.mockRejectedValue(new Error('Linear unavailable')); + await expect(handleLinearApprovalReply(event, deps)).rejects.toThrow('Linear unavailable'); + expect(lookupUser).not.toHaveBeenCalled(); + expect(mockApprove).not.toHaveBeenCalled(); }); test.each(['update', 'remove'])('ignores %s events', async action => { diff --git a/cdk/test/handlers/shared/linear-feedback.test.ts b/cdk/test/handlers/shared/linear-feedback.test.ts index 791fa5bdd..6a9b13281 100644 --- a/cdk/test/handlers/shared/linear-feedback.test.ts +++ b/cdk/test/handlers/shared/linear-feedback.test.ts @@ -34,6 +34,7 @@ import { fetchRecentComments, postIssueComment, postIdentifiedComment, + readLinearApprovalComment, reactToComment, replyToComment, reportIssueFailure, @@ -74,6 +75,41 @@ describe('linear-feedback', () => { fetchMock.mockResolvedValue(jsonResponse({ data: { commentCreate: { success: true } } })); }); + describe('readLinearApprovalComment', () => { + test('reads the workspace, human author, decision and exact thread from Linear', async () => { + const comment = { + id: 'reply', + body: 'approve', + user: { id: 'human' }, + botActor: null, + issue: { id: ISSUE_ID }, + parent: { id: 'root' }, + }; + fetchMock.mockResolvedValue(jsonResponse({ data: { organization: { id: CTX.linearWorkspaceId }, comment } })); + expect(await readLinearApprovalComment(CTX, 'reply')).toEqual(comment); + const request = JSON.parse(fetchMock.mock.calls[0][1].body); + expect(request.variables).toEqual({ id: 'reply' }); + expect(request.query).toContain('user { id }'); + expect(request.query).toContain('botActor { id }'); + }); + test('returns no consent for a deleted comment', async () => { + fetchMock.mockResolvedValue(jsonResponse({ data: { organization: { id: CTX.linearWorkspaceId }, comment: null } })); + expect(await readLinearApprovalComment(CTX, 'deleted')).toBeNull(); + }); + test.each([ + { errors: [{ message: 'Unavailable' }] }, + { data: { organization: { id: 'other-workspace' }, comment: {} } }, + ])('fails closed on lookup errors or incorrect workspace: %j', async response => { + fetchMock.mockResolvedValue(jsonResponse(response)); + await expect(readLinearApprovalComment(CTX, 'reply')).rejects.toThrow('verification'); + }); + test('fails closed without credentials', async () => { + resolveLinearOauthTokenMock.mockResolvedValue(null); + await expect(readLinearApprovalComment(CTX, 'reply')).rejects.toThrow('token unavailable'); + expect(fetchMock).not.toHaveBeenCalled(); + }); + }); + describe('postIdentifiedComment', () => { const input = { id: 'stable', issueId: ISSUE_ID, body: 'Approval needed', parentId: 'root' }; test('passes the stable identity and exact thread to Linear', async () => { From 94c975b1222207f4e3722f8f5fcfa374d7014798 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Fri, 18 Sep 2026 19:17:37 -0400 Subject: [PATCH 094/149] fix(cdk): preserve flat MicroVM layout unless nesting is explicit --- cdk/src/stacks/agent.ts | 6 +++--- cdk/test/stacks/agent.test.ts | 16 ++++++++++------ docs/verification/645-p3-nested-stack.md | 2 +- 3 files changed, 14 insertions(+), 10 deletions(-) diff --git a/cdk/src/stacks/agent.ts b/cdk/src/stacks/agent.ts index 5607b2db9..51d73d8fe 100644 --- a/cdk/src/stacks/agent.ts +++ b/cdk/src/stacks/agent.ts @@ -356,9 +356,9 @@ export class AgentStack extends Stack { if (microvmNestedContext !== undefined && ![true, false, 'true', 'false'].includes(microvmNestedContext)) { throw new Error('microvm_nested_stack must be true or false'); } - // Existing flat deployments retain their resource paths with false until - // their reviewed resource-migration procedure is complete. - const microvmNested = microvmNestedContext !== false && microvmNestedContext !== 'false'; + // Omission preserves existing flat resource identities. New installations + // may opt in; existing stacks must complete the reviewed migration first. + const microvmNested = microvmNestedContext === true || microvmNestedContext === 'true'; const microvmResourceNamePrefix = this.node.tryGetContext('microvm_resource_name_prefix'); if (microvmResourceNamePrefix !== undefined && (!microvmNested || typeof microvmResourceNamePrefix !== 'string')) { diff --git a/cdk/test/stacks/agent.test.ts b/cdk/test/stacks/agent.test.ts index 880e56e74..7ea173057 100644 --- a/cdk/test/stacks/agent.test.ts +++ b/cdk/test/stacks/agent.test.ts @@ -1295,6 +1295,7 @@ describe('AgentStack with the Lambda MicroVMs substrate gate (--context compute_ const app = new App({ context: { compute_type: 'lambda-microvm', + microvm_nested_stack: true, microvm_base_image_arn: BASE_IMAGE_ARN, microvm_base_image_version: '1', microvm_artifact_sha256: 'a'.repeat(64), @@ -1548,6 +1549,7 @@ describe('AgentStack with the Lambda MicroVMs substrate gate (--context compute_ const app = new App({ context: { compute_type: 'lambda-microvm', + microvm_nested_stack: true, microvm_region_override: true, microvm_base_image_arn: BASE_IMAGE_ARN, microvm_base_image_version: '1', @@ -1562,7 +1564,7 @@ describe('AgentStack with the Lambda MicroVMs substrate gate (--context compute_ }); test('fails synth when the stack Region has no Lambda MicroVMs', () => { - const app = new App({ context: { compute_type: 'lambda-microvm' } }); + const app = new App({ context: { compute_type: 'lambda-microvm', microvm_nested_stack: true } }); expect(() => new AgentStack(app, 'TestAgentStackMicrovmBadRegion', { env: { account: '123456789012', region: 'eu-central-1' }, })).toThrow(/AWS Lambda MicroVMs are not available in eu-central-1/); @@ -1617,7 +1619,7 @@ describe('AgentStack with the MicroVM gate on but no image configured (first dep // but no image yet. Exercises the false branch of the shared // `isLambdaMicrovmImageConfigured` predicate that gates BOTH the // orchestrator's MICROVM_* wiring and the cancel Lambda's grant. - const app = new App({ context: { compute_type: 'lambda-microvm' } }); + const app = new App({ context: { compute_type: 'lambda-microvm', microvm_nested_stack: true } }); const stack = new AgentStack(app, 'TestAgentStackMicrovmNoImage', { env: { account: '123456789012', region: 'us-east-1' }, }); @@ -1661,7 +1663,6 @@ describe('AgentStack MicroVM flat-layout migration compatibility', () => { const app = new App({ context: { compute_type: 'lambda-microvm', - microvm_nested_stack: false, microvm_base_image_arn: 'arn:aws:lambda:us-east-1:aws:microvm-image:al2023-1', microvm_base_image_version: '1', microvm_artifact_sha256: 'a'.repeat(64), @@ -1672,7 +1673,7 @@ describe('AgentStack MicroVM flat-layout migration compatibility', () => { })); }); - test('retains original resource paths and unnamed IAM roles with nesting disabled', () => { + test('preserves flat resource identities when nesting is not configured', () => { template.resourceCountIs('AWS::Lambda::MicrovmImage', 1); template.resourceCountIs('AWS::Lambda::NetworkConnector', 2); expect(Object.keys(template.findResources('AWS::Lambda::MicrovmImage'))[0]) @@ -1696,7 +1697,7 @@ describe('AgentStack MicroVM image ARN invariant', () => { const configuredSpy = jest.spyOn(lambdaMicrovmCompute, 'isLambdaMicrovmImageConfigured') .mockReturnValue(true); try { - const app = new App({ context: { compute_type: 'lambda-microvm' } }); + const app = new App({ context: { compute_type: 'lambda-microvm', microvm_nested_stack: true } }); const stack = new AgentStack(app, 'TestAgentStackMicrovmInvariant', { env: { account: '123456789012', region: 'us-east-1' }, }); @@ -2047,6 +2048,7 @@ describe('AgentStack Linear identity vault gate (#809)', () => { context: { enableLinearIdentityVault: true, compute_type: 'lambda-microvm', + microvm_nested_stack: true, microvm_base_image_arn: 'arn:aws:lambda:us-east-1:aws:microvm-image:al2023-1', microvm_base_image_version: '1', microvm_artifact_sha256: 'a'.repeat(64), @@ -2212,11 +2214,12 @@ describe('AgentStack CloudFormation resource budget 500 with cushion', () => { const CONFIGURATIONS = [ { name: 'agentcore', context: { compute_type: 'agentcore' } }, { name: 'ecs', context: { compute_type: 'ecs' } }, - { name: 'microvm-bootstrap', context: { compute_type: 'lambda-microvm' } }, + { name: 'microvm-bootstrap', context: { compute_type: 'lambda-microvm', microvm_nested_stack: true } }, { name: 'microvm-imported', context: { compute_type: 'lambda-microvm', + microvm_nested_stack: true, microvm_image_identifier: 'arn:aws:lambda:us-east-1:123456789012:microvm-image:existing-agent', microvm_image_version: '6.0', }, @@ -2225,6 +2228,7 @@ describe('AgentStack CloudFormation resource budget 500 with cushion', () => { name: 'microvm-managed', context: { compute_type: 'lambda-microvm', + microvm_nested_stack: true, microvm_base_image_arn: 'arn:aws:lambda:us-east-1:aws:microvm-image:al2023-1', microvm_base_image_version: '1', microvm_artifact_sha256: 'a'.repeat(64), diff --git a/docs/verification/645-p3-nested-stack.md b/docs/verification/645-p3-nested-stack.md index 3d0992442..30e2e7b97 100644 --- a/docs/verification/645-p3-nested-stack.md +++ b/docs/verification/645-p3-nested-stack.md @@ -19,7 +19,7 @@ Names derive from the concrete parent deployment name, not the child stack token | Setting | Behavior | |---|---| -| `microvm_nested_stack` | Defaults to `true` for MicroVM compute. Set explicitly to `false` for an existing flat deployment until migration. | +| `microvm_nested_stack` | Defaults to `false`, preserving flat resource identities. Set `true` for a new installation or after completing the reviewed migration. | | `microvm_resource_name_prefix` | Supplies distinct names for overlapping nested resources; preserve it after migration. It does not retain old resources or permissions by itself. | | `microvm_managed_image_version` | Pins new tasks to an explicitly verified image version. Without a pin, selection follows the latest active version. | | `microvm_approval_suspend_enabled` | Defaults to `false`; enable only after testing the deployed image/coordinator. Disabling new sleep preserves wake and cleanup. | From 71adb3fa2c98d69462e2a035125f0a6857a4ee22 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Fri, 18 Sep 2026 19:17:37 -0400 Subject: [PATCH 095/149] fix(agent): distinguish terminal status races from revoked worker leases --- agent/src/task_state.py | 15 +++++++------- agent/tests/test_task_state.py | 37 ++++++++++++++++++++++++++++++++++ 2 files changed, 45 insertions(+), 7 deletions(-) diff --git a/agent/src/task_state.py b/agent/src/task_state.py index 22bdde2fe..669bf3889 100644 --- a/agent/src/task_state.py +++ b/agent/src/task_state.py @@ -357,12 +357,7 @@ def write_terminal(task_id: str, status: str, result: dict | None = None) -> Non ExpressionAttributeValues=expr_values, ) except Exception as e: - from botocore.exceptions import ClientError - - if ( - isinstance(e, ClientError) - and e.response.get("Error", {}).get("Code") == "ConditionalCheckFailedException" - ): + if _task_status_conflict(e): log( "INFO", "[task_state] write_terminal skipped: " @@ -406,7 +401,13 @@ def write_terminal(task_id: str, status: str, result: dict | None = None) -> Non f"(terminal-state race).", ) return - log("WARN", f"[task_state] write_terminal failed (best-effort): {type(e).__name__}") + # Include DynamoDB's cancellation reasons: losing worker ownership is + # not a benign status race and must never trigger trace self-healing. + log_error_cw( + f"[task_state] write_terminal failed (best-effort): {type(e).__name__}: {e}; " + f"CancellationReasons={_extract_cancellation_reasons(e)}", + task_id=task_id, + ) def write_trace_uri_conditional(task_id: str, uri: str) -> bool: diff --git a/agent/tests/test_task_state.py b/agent/tests/test_task_state.py index 6461d9a58..36f5bc923 100644 --- a/agent/tests/test_task_state.py +++ b/agent/tests/test_task_state.py @@ -50,6 +50,43 @@ def test_status_race_logging_preserves_worker_fence_failures( assert any(call.args[0] == "WARN" for call in log.call_args_list) is expected_warning +@pytest.mark.parametrize( + ("codes", "can_heal"), + [ + (["ConditionalCheckFailed", "None"], True), + (["None", "ConditionalCheckFailed"], False), + (["ConditionalCheckFailed", "ConditionalCheckFailed"], False), + (["TransactionConflict", "None"], False), + ([], False), + ], +) +def test_terminal_transaction_heals_trace_only_when_worker_lease_passed( + monkeypatch, codes, can_heal +): + from botocore.exceptions import ClientError + + monkeypatch.setattr(task_state, "_get_table", MagicMock()) + error = ClientError( + { + "Error": {"Code": "TransactionCanceledException", "Message": "cancelled"}, + "CancellationReasons": [{"Code": code} for code in codes], + }, + "TransactWriteItems", + ) + monkeypatch.setattr(task_state, "_update_task", MagicMock(side_effect=error)) + heal = MagicMock(return_value=True) + report = MagicMock() + monkeypatch.setattr(task_state, "write_trace_uri_conditional", heal) + monkeypatch.setattr(task_state, "log_error_cw", report) + task_state.write_terminal("task", "COMPLETED", {"trace_s3_uri": "s3://bucket/trace"}) + assert heal.called is can_heal + if can_heal: + heal.assert_called_once_with("task", "s3://bucket/trace") + else: + assert "TransactionCanceledException" in report.call_args.args[0] + assert "CancellationReasons" in report.call_args.args[0] + + class TestAgentWriteContract: def test_current_task_writers_fit_the_deployed_attribute_allowlist(self, monkeypatch): """Exercise real writers; detect a new field before IAM rejects it live. From 1d6b678a1c08f080c9f969291fe6c70627db0842 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Fri, 18 Sep 2026 19:17:38 -0400 Subject: [PATCH 096/149] fix(microvm): detect checkpoint SDK drift and report unavailable recovery --- agent/src/hooks.py | 7 +++++++ agent/tests/test_hooks.py | 18 ++++++++++++++++++ cdk/test/scripts/check-constants-sync.test.ts | 9 +++++++++ scripts/check-constants-sync.ts | 5 +++++ 4 files changed, 39 insertions(+) diff --git a/agent/src/hooks.py b/agent/src/hooks.py index 44a92a9dd..5722af3f8 100644 --- a/agent/src/hooks.py +++ b/agent/src/hooks.py @@ -825,6 +825,13 @@ async def _handle_require_approval( f"Continuation checkpoint unavailable task_id={task_id} request_id={request_id} " f"code={getattr(exc, 'code', 'checkpoint_failed')} error_type={type(exc).__name__}", ) + _try_progress( + progress, + "write_agent_milestone", + milestone="continuation_unavailable", + details="Could not save recovery state. This worker must remain available " + f"until the decision is received. code={getattr(exc, 'code', 'checkpoint_failed')}", + ) try: outcome = await _poll_for_decision( task_id=task_id, diff --git a/agent/tests/test_hooks.py b/agent/tests/test_hooks.py index 9bd85ecc3..3ebd3fa9d 100644 --- a/agent/tests/test_hooks.py +++ b/agent/tests/test_hooks.py @@ -1456,6 +1456,24 @@ def resume(*args, **kwargs): ) ) assert phases == ["checkpointing", "checkpoint-ready", "checkpoint-ready"] + unavailable = [ + data + for method, data in progress.calls + if method == "write_agent_milestone" + and data.get("milestone") == "continuation_unavailable" + ] + assert len(unavailable) == int(publish_failed) + if publish_failed: + # Use the real signature so an incorrectly named keyword cannot + # pass through the permissive recording double unnoticed. + from progress_writer import _ProgressWriter + + writer = MagicMock(spec=_ProgressWriter) + import inspect + + inspect.signature(_ProgressWriter.write_agent_milestone).bind( + writer, **unavailable[0] + ) expected = "deny" if transferred else "allow" assert result["hookSpecificOutput"]["permissionDecision"] == expected assert lifecycle.diagnostic_snapshot()["phase"] == ( diff --git a/cdk/test/scripts/check-constants-sync.test.ts b/cdk/test/scripts/check-constants-sync.test.ts index 0b16bc921..17bbd2378 100644 --- a/cdk/test/scripts/check-constants-sync.test.ts +++ b/cdk/test/scripts/check-constants-sync.test.ts @@ -58,6 +58,7 @@ const SCRIPT_REL = 'scripts/check-constants-sync.ts'; const FIXTURE_FILES = [ SCRIPT_REL, 'contracts/constants.json', + 'agent/pyproject.toml', 'agent/src/policy.py', 'agent/src/jira_reactions.py', 'agent/src/server.py', @@ -127,6 +128,14 @@ function patchContract(root: string, mutate: (json: Record) => void } describe('check-constants-sync', () => { + test('rejects an SDK upgrade without checkpoint compatibility verification', () => { + const result = runInMutatedRepo(root => { + write(root, 'agent/pyproject.toml', read(root, 'agent/pyproject.toml') + .replace(/claude-agent-sdk==[0-9.]+/, 'claude-agent-sdk==99.0.0')); + }); + expect(result.status).toBe(1); + expect(result.stderr).toContain('claude-agent-sdk pin must match'); + }); // Node's type-stripping runs the script from source; the suite is a handful of // subprocess spawns, so give it room on a cold cache. jest.setTimeout(60_000); diff --git a/scripts/check-constants-sync.ts b/scripts/check-constants-sync.ts index ff76aeef2..5a6a5bdf0 100644 --- a/scripts/check-constants-sync.ts +++ b/scripts/check-constants-sync.ts @@ -406,6 +406,11 @@ function main(): number { invariantErrors.push('microvm_lifecycle requires a positive protocol version, valid hook port, duration within 1–28800 seconds and ABCA_MICROVM_ marker name'); } const continuation = json.microvm_continuation; + const sdkPins = [...fs.readFileSync(path.join(REPO_ROOT, 'agent/pyproject.toml'), 'utf8') + .matchAll(/^\s*"claude-agent-sdk==([^"]+)"/gm)]; + if (sdkPins.length !== 1 || sdkPins[0][1] !== continuation?.verified_sdk_version) { + invariantErrors.push('claude-agent-sdk pin must match microvm_continuation.verified_sdk_version; verify checkpoint compatibility before upgrading both'); + } if (!continuation || !Number.isSafeInteger(continuation.version) || continuation.version <= 0 || continuation.lease_key_prefix !== 'worker-lease#' || continuation.object_key_prefix !== 'continuations/' || !Number.isSafeInteger(continuation.max_manifest_bytes) || continuation.max_manifest_bytes <= 0 From 70dea63ef20a20879b6d26b7f1e89dadd92c15db Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Fri, 18 Sep 2026 19:22:49 -0400 Subject: [PATCH 097/149] test(microvm): require real DynamoDB transactions in CI --- .github/workflows/build.yml | 7 +++++ agent/tests/test_microvm_checkpoint.py | 2 ++ .../shared/microvm-approval-wake.test.ts | 27 +++++++++++++++++++ .../shared/microvm-lifecycle-local.test.ts | 3 +++ .../shared/task-concurrency-local.test.ts | 3 +++ 5 files changed, 42 insertions(+) diff --git a/.github/workflows/build.yml b/.github/workflows/build.yml index a3ee7a2f8..d2a4a3c6c 100644 --- a/.github/workflows/build.yml +++ b/.github/workflows/build.yml @@ -52,6 +52,12 @@ jobs: compute_type: [agentcore] outputs: self_mutation_happened: ${{ steps.self_mutation.outputs.self_mutation_happened }} + services: + dynamodb: + # Pin the multi-architecture image used for real transaction assertions. + image: amazon/dynamodb-local@sha256:ff89bd48ff32cd8d9be5fee8873b65b8854dc408f1afe881be6eb00247bc0dab + ports: + - 8000/tcp env: CI: "true" MISE_EXPERIMENTAL: "1" @@ -293,6 +299,7 @@ jobs: - name: build env: TMPDIR: ${{ runner.temp }} + ABCA_DDB_LOCAL_ENDPOINT: http://127.0.0.1:${{ job.services.dynamodb.ports[8000] }} run: | echo "::notice::Runner: $(nproc) cores, $(free -h | awk '/Mem:/{print $2}') RAM" SECONDS=0 diff --git a/agent/tests/test_microvm_checkpoint.py b/agent/tests/test_microvm_checkpoint.py index 7f1c3550d..ce8e87c42 100644 --- a/agent/tests/test_microvm_checkpoint.py +++ b/agent/tests/test_microvm_checkpoint.py @@ -293,6 +293,8 @@ def test_failed_refresh_prevents_any_aws_reconciliation(self, state, monkeypatch LOCAL_ENDPOINT = os.environ.get("ABCA_DDB_LOCAL_ENDPOINT", "") +if os.environ.get("CI") == "true" and not LOCAL_ENDPOINT: + raise RuntimeError("CI requires ABCA_DDB_LOCAL_ENDPOINT; checkpoint transaction tests must not skip") @pytest.fixture diff --git a/cdk/test/handlers/shared/microvm-approval-wake.test.ts b/cdk/test/handlers/shared/microvm-approval-wake.test.ts index bab59cd39..43fb6b3ce 100644 --- a/cdk/test/handlers/shared/microvm-approval-wake.test.ts +++ b/cdk/test/handlers/shared/microvm-approval-wake.test.ts @@ -23,6 +23,10 @@ const mockRead = jest.fn(); const mockSave = jest.fn(); const mockSend = jest.fn(); const mockEmit = jest.fn(); +const mockDispatch = jest.fn(); +jest.mock('../../../src/handlers/shared/microvm-continuation-dispatch', () => ({ + dispatchMicrovmContinuation: (...args: unknown[]) => mockDispatch(...args), +})); const mockLogger = { warn: jest.fn(), info: jest.fn() }; jest.mock('../../../src/handlers/shared/microvm-lifecycle', () => ({ readMicrovmLifecycleSnapshot: (...args: unknown[]) => mockRead(...args), @@ -53,6 +57,7 @@ function wake(decision: 'APPROVED' | 'DENIED' = 'APPROVED') { } beforeEach(() => { jest.clearAllMocks(); + mockDispatch.mockReset().mockResolvedValue(false); jest.spyOn(Date, 'now').mockReturnValue(NOW); controller = new AbortController(); row = { @@ -91,6 +96,28 @@ beforeEach(() => { }); afterEach(() => jest.restoreAllMocks()); +test('dispatches a retired continuation without touching the fenced worker', async () => { + mockDispatch.mockResolvedValue(true); + await wake(); + expect(mockDispatch).toHaveBeenCalledWith('task', 'user', 'gate', expect.objectContaining({ + abortSignal: controller.signal, + })); + expect(mockRead).not.toHaveBeenCalled(); + expect(mockSave).not.toHaveBeenCalled(); + expect(mockSend).not.toHaveBeenCalled(); +}); + +test('does not fall back to the old worker when continuation dispatch is uncertain', async () => { + mockDispatch.mockRejectedValue(new Error('dispatch response lost')); + await wake(); + expect(mockRead).not.toHaveBeenCalled(); + expect(mockSave).not.toHaveBeenCalled(); + expect(mockSend).not.toHaveBeenCalled(); + expect(mockEmit).toHaveBeenCalledWith('microvm_resume_orphan', expect.objectContaining({ + reason: 'wake-reconciliation-failed', + }), expect.anything()); +}); + test('ties the saved decision generation to the accepted AWS request without logging its body', async () => { mockSend.mockResolvedValueOnce({ state: 'SUSPENDED' }).mockResolvedValueOnce({ $metadata: { requestId: 'aws-inline-123' }, private: 'secret-response', diff --git a/cdk/test/handlers/shared/microvm-lifecycle-local.test.ts b/cdk/test/handlers/shared/microvm-lifecycle-local.test.ts index 6e9230c62..4431ae213 100644 --- a/cdk/test/handlers/shared/microvm-lifecycle-local.test.ts +++ b/cdk/test/handlers/shared/microvm-lifecycle-local.test.ts @@ -23,6 +23,9 @@ import { CreateTableCommand, DeleteTableCommand, DynamoDBClient } from '@aws-sdk import { DeleteCommand, DynamoDBDocumentClient, GetCommand, PutCommand, ScanCommand, TransactWriteCommand, UpdateCommand } from '@aws-sdk/lib-dynamodb'; const endpoint = process.env.ABCA_DDB_LOCAL_ENDPOINT; +if (process.env.CI === 'true' && !endpoint) { + throw new Error('CI requires ABCA_DDB_LOCAL_ENDPOINT; lifecycle transaction tests must not skip'); +} if (endpoint && (new URL(endpoint).hostname !== '127.0.0.1' || new URL(endpoint).protocol !== 'http:')) { throw new Error('Lifecycle integration tests require an http://127.0.0.1 DynamoDB Local endpoint'); } diff --git a/cdk/test/handlers/shared/task-concurrency-local.test.ts b/cdk/test/handlers/shared/task-concurrency-local.test.ts index e669fca56..c5949ae57 100644 --- a/cdk/test/handlers/shared/task-concurrency-local.test.ts +++ b/cdk/test/handlers/shared/task-concurrency-local.test.ts @@ -27,6 +27,9 @@ import { CreateTableCommand, DeleteTableCommand, DynamoDBClient } from '@aws-sdk import { DeleteCommand, DynamoDBDocumentClient, GetCommand, PutCommand, ScanCommand, UpdateCommand } from '@aws-sdk/lib-dynamodb'; const endpoint = process.env.ABCA_DDB_LOCAL_ENDPOINT; +if (process.env.CI === 'true' && !endpoint) { + throw new Error('CI requires ABCA_DDB_LOCAL_ENDPOINT; concurrency transaction tests must not skip'); +} if (endpoint && (new URL(endpoint).hostname !== '127.0.0.1' || new URL(endpoint).protocol !== 'http:')) { throw new Error('Capacity integration tests require an http://127.0.0.1 DynamoDB Local endpoint'); } From 7b28ff020a669594057704c9ce787d610e79adc3 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Fri, 18 Sep 2026 19:33:53 -0400 Subject: [PATCH 098/149] fix(microvm): close the retained launch lease on terminal cleanup --- .../reconcile-microvm-continuations.ts | 29 +++++++++- .../reconcile-microvm-continuations.test.ts | 54 +++++++++++++++++++ 2 files changed, 81 insertions(+), 2 deletions(-) diff --git a/cdk/src/handlers/reconcile-microvm-continuations.ts b/cdk/src/handlers/reconcile-microvm-continuations.ts index def3d5022..7e566edde 100644 --- a/cdk/src/handlers/reconcile-microvm-continuations.ts +++ b/cdk/src/handlers/reconcile-microvm-continuations.ts @@ -66,16 +66,41 @@ export async function reconcileMicrovmContinuation(task: ContinuableTask): Promi } if (Number.isFinite(started) && Date.now() < started + MICROVM_MAX_DURATION_SECONDS * 1000) return; } + let attempt = task.microvm_start?.clientToken; + const withoutStart = !task.microvm_start; + let leaseState: string | undefined; + if (withoutStart) { + // Replacement admission removes the old start receipt. Its new launch + // token lives on the coordinator-owned slot and lease, not the task id. + const lease = (await ddb.send(new GetCommand({ + TableName: TABLE, Key: workerLeaseKey(task.task_id), ConsistentRead: true, + }), options)).Item; + if (lease) { + const notLaunched = lease.lease_state === 'ACTIVE' && !lease.lease_microvm_id + && task.concurrency_slot?.state === 'held' + && task.concurrency_slot.attempt_id === lease.lease_attempt_id; + if (lease.lease_user_id !== task.user_id + || (!['PARKED', 'CLOSED'].includes(lease.lease_state) && !notLaunched) + || typeof lease.lease_attempt_id !== 'string' || !lease.lease_attempt_id) { + throw new Error('MICROVM_CONTINUATION_LEASE_INVALID: missing start does not prove worker shutdown'); + } + attempt = lease.lease_attempt_id; + leaseState = lease.lease_state; + } + } await ddb.send(new UpdateCommand({ TableName: TABLE, Key: workerLeaseKey(task.task_id), UpdateExpression: 'SET #ttl = :ttl, lease_state = :closed, lease_user_id = :user, lease_attempt_id = :attempt', - ConditionExpression: 'attribute_not_exists(task_id) OR (lease_user_id = :user AND lease_attempt_id = :attempt)', + ConditionExpression: 'attribute_not_exists(task_id) OR (lease_user_id = :user AND lease_attempt_id = :attempt' + + (withoutStart ? ' AND lease_state = :observedState' : '') + + (leaseState === 'ACTIVE' ? ' AND attribute_not_exists(lease_microvm_id)' : '') + ')', ExpressionAttributeNames: { '#ttl': 'ttl' }, ExpressionAttributeValues: { ':user': task.user_id, ':closed': 'CLOSED', - ':attempt': task.microvm_start?.clientToken ?? task.task_id, + ':attempt': attempt ?? task.task_id, + ...(withoutStart ? { ':observedState': leaseState ?? 'CLOSED' } : {}), ':ttl': Math.floor(Date.now() / 1000) + Number(process.env.TASK_RETENTION_DAYS ?? '90') * 86400, }, }), options); diff --git a/cdk/test/handlers/reconcile-microvm-continuations.test.ts b/cdk/test/handlers/reconcile-microvm-continuations.test.ts index 53c22b50d..3d6e078d0 100644 --- a/cdk/test/handlers/reconcile-microvm-continuations.test.ts +++ b/cdk/test/handlers/reconcile-microvm-continuations.test.ts @@ -109,6 +109,60 @@ test('terminal scan rows cannot stop a task whose current record is active', asy expect(mockClose).not.toHaveBeenCalled(); }); +test('closes a retired source using its preserved lease token after microvm_start was removed', async () => { + task.status = 'CANCELLED'; + delete task.microvm_start; + mockSend.mockImplementation(async command => { + if (command.constructor.name === 'UpdateCommand') return {}; + return { + Item: command.input.Key.task_id === 'task' ? task : { + lease_user_id: 'user', lease_attempt_id: 'retired-launch-token', lease_state: 'PARKED', + }, + }; + }); + await reconcileMicrovmContinuation(task); + const closed = mockSend.mock.calls.find(([command]) => command.constructor.name === 'UpdateCommand')![0].input; + expect(closed.ExpressionAttributeValues[':attempt']).toBe('retired-launch-token'); + expect(closed.ExpressionAttributeValues[':observedState']).toBe('PARKED'); + expect(mockRelease).toHaveBeenCalledWith('task', 'user'); + expect(mockDelete).toHaveBeenCalled(); +}); + +test('releases a cancelled replacement admitted before any start receipt or AWS call', async () => { + task.status = 'CANCELLED'; + delete task.microvm_start; + task.continuation = { state: 'STARTING', attempt_id: 'new-token' }; + task.concurrency_slot = { state: 'held', attempt_id: 'new-token' }; + mockSend.mockImplementation(async command => { + if (command.constructor.name === 'UpdateCommand') return {}; + return { + Item: command.input.Key.task_id === 'task' ? task : { + lease_user_id: 'user', lease_attempt_id: 'new-token', lease_state: 'ACTIVE', + }, + }; + }); + await reconcileMicrovmContinuation(task); + const closed = mockSend.mock.calls.find(([command]) => command.constructor.name === 'UpdateCommand')![0].input; + expect(closed.ExpressionAttributeValues[':attempt']).toBe('new-token'); + expect(closed.ExpressionAttributeValues[':observedState']).toBe('ACTIVE'); + expect(closed.ConditionExpression).toContain('attribute_not_exists(lease_microvm_id)'); + expect(mockRelease).toHaveBeenCalledWith('task', 'user'); + expect(mockDelete).toHaveBeenCalled(); +}); + +test('does not release a live lease just because the start record is absent', async () => { + task.status = 'CANCELLED'; + delete task.microvm_start; + mockSend.mockImplementation(async command => ({ + Item: command.input.Key.task_id === 'task' ? task : { + lease_user_id: 'user', lease_attempt_id: 'live-token', lease_state: 'ACTIVE', + }, + })); + await expect(reconcileMicrovmContinuation(task)).rejects.toThrow('LEASE_INVALID'); + expect(mockRelease).not.toHaveBeenCalled(); + expect(mockDelete).not.toHaveBeenCalled(); +}); + test('invalid unknown-start timestamp cannot be treated as proof of shutdown', async () => { task.status = 'FAILED'; task.microvm_start.createdAt = 'invalid'; From 80977e77cf873819d4a4e00b4e000952e86fc4e6 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Fri, 18 Sep 2026 19:33:53 -0400 Subject: [PATCH 099/149] refactor(microvm): share receipt serialization without changing stored bytes --- cdk/src/handlers/shared/canonical-json.ts | 26 ++++++++++++++++ .../shared/microvm-continuation-retirement.ts | 7 ++--- .../shared/microvm-continuation-storage.ts | 6 ++-- cdk/src/handlers/shared/microvm-start.ts | 6 ++-- cdk/src/handlers/shared/payload-bootstrap.ts | 16 +++------- .../handlers/shared/canonical-json.test.ts | 31 +++++++++++++++++++ 6 files changed, 68 insertions(+), 24 deletions(-) create mode 100644 cdk/src/handlers/shared/canonical-json.ts create mode 100644 cdk/test/handlers/shared/canonical-json.test.ts diff --git a/cdk/src/handlers/shared/canonical-json.ts b/cdk/src/handlers/shared/canonical-json.ts new file mode 100644 index 000000000..f89bab658 --- /dev/null +++ b/cdk/src/handlers/shared/canonical-json.ts @@ -0,0 +1,26 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +/** Stable JSON bytes for ABCA receipts; preserve this ordering for stored hashes. */ +export function canonicalJson(value: unknown): string { + return JSON.stringify(value, (_key, item: unknown) => + item && typeof item === 'object' && !Array.isArray(item) + ? Object.fromEntries(Object.entries(item).sort(([a], [b]) => a.localeCompare(b))) + : item); +} diff --git a/cdk/src/handlers/shared/microvm-continuation-retirement.ts b/cdk/src/handlers/shared/microvm-continuation-retirement.ts index 8192b3efa..c87a9af05 100644 --- a/cdk/src/handlers/shared/microvm-continuation-retirement.ts +++ b/cdk/src/handlers/shared/microvm-continuation-retirement.ts @@ -19,6 +19,7 @@ import { randomUUID } from 'node:crypto'; import { GetCommand, TransactWriteCommand } from '@aws-sdk/lib-dynamodb'; +import { canonicalJson } from './canonical-json'; import type { ComputeStrategy, SessionControlOptions } from './compute-strategy'; import { logger } from './logger'; import { verifyContinuationCheckpoint } from './microvm-continuation-storage'; @@ -56,11 +57,7 @@ async function readTask(taskId: string, options: SessionControlOptions): Promise } function sameRecord(left: unknown, right: unknown): boolean { - const canonical = (value: unknown) => JSON.stringify(value, (_key, item: unknown) => - item && typeof item === 'object' && !Array.isArray(item) - ? Object.fromEntries(Object.entries(item).sort(([a], [b]) => a.localeCompare(b))) - : item); - return canonical(left) === canonical(right); + return canonicalJson(left) === canonicalJson(right); } async function leaseMatches( diff --git a/cdk/src/handlers/shared/microvm-continuation-storage.ts b/cdk/src/handlers/shared/microvm-continuation-storage.ts index d12333f24..c3c7473a7 100644 --- a/cdk/src/handlers/shared/microvm-continuation-storage.ts +++ b/cdk/src/handlers/shared/microvm-continuation-storage.ts @@ -21,6 +21,7 @@ import { createHash } from 'node:crypto'; import { Readable } from 'node:stream'; import { DeleteObjectsCommand, GetObjectCommand, HeadObjectCommand, ListObjectVersionsCommand, PutObjectCommand, S3Client } from '@aws-sdk/client-s3'; import { GetCommand, UpdateCommand } from '@aws-sdk/lib-dynamodb'; +import { canonicalJson } from './canonical-json'; import type { SessionControlOptions } from './compute-strategy'; import { CONTINUATION_IO_TIMEOUT_MS } from './microvm-continuation-timing'; import { @@ -50,10 +51,7 @@ interface LaunchInputs { } function canonical(value: unknown): Buffer { - return Buffer.from(JSON.stringify(value, (_key, item: unknown) => - item && typeof item === 'object' && !Array.isArray(item) - ? Object.fromEntries(Object.entries(item).sort(([a], [b]) => a.localeCompare(b))) - : item)); + return Buffer.from(canonicalJson(value)); } function hash(value: Buffer): string { diff --git a/cdk/src/handlers/shared/microvm-start.ts b/cdk/src/handlers/shared/microvm-start.ts index 9553dc704..c259e9d01 100644 --- a/cdk/src/handlers/shared/microvm-start.ts +++ b/cdk/src/handlers/shared/microvm-start.ts @@ -19,6 +19,7 @@ import { createHash } from 'node:crypto'; import { GetCommand, TransactWriteCommand, UpdateCommand } from '@aws-sdk/lib-dynamodb'; +import { canonicalJson } from './canonical-json'; import type { SessionHandle } from './compute-strategy'; import type { ContinuationRecord } from './microvm-continuation-types'; import { MICROVM_IMAGE_CAPABILITY_REQUEST_TIMEOUT_MS, readMicrovmImageMetadata, supportsMicrovmLifecycle } from './microvm-image-capability'; @@ -68,10 +69,7 @@ const ACTIVE = new Set([TaskStatus.HYDRATING, TaskStatus.RUNNING, TaskSt /** Include the full S3 content, not just its URI; ignore object-key ordering. */ export function microvmStartRequestHash(request: unknown, payload: unknown): string { - const canonical = JSON.stringify([request, payload], (_key, value: unknown) => - value && typeof value === 'object' && !Array.isArray(value) - ? Object.fromEntries(Object.entries(value).sort(([a], [b]) => a.localeCompare(b))) - : value); + const canonical = canonicalJson([request, payload]); return createHash('sha256').update(canonical).digest('hex'); } diff --git a/cdk/src/handlers/shared/payload-bootstrap.ts b/cdk/src/handlers/shared/payload-bootstrap.ts index fbb88c84c..c0dc95c94 100644 --- a/cdk/src/handlers/shared/payload-bootstrap.ts +++ b/cdk/src/handlers/shared/payload-bootstrap.ts @@ -20,6 +20,7 @@ import { createHash } from 'node:crypto'; import { DeleteObjectCommand, GetObjectCommand, PutObjectCommand, S3Client } from '@aws-sdk/client-s3'; import { getSignedUrl } from '@aws-sdk/s3-request-presigner'; +import { canonicalJson } from './canonical-json'; import { logger } from './logger'; import { makeClient } from './ua'; import constants from '../../../../contracts/constants.json'; @@ -46,13 +47,6 @@ function s3(): S3Client { return client ??= makeClient(S3Client); } -function canonical(value: unknown): string { - return JSON.stringify(value, (_key, item: unknown) => - item && typeof item === 'object' && !Array.isArray(item) - ? Object.fromEntries(Object.entries(item).sort(([a], [b]) => a.localeCompare(b))) - : item); -} - function sha256(value: string): string { return createHash('sha256').update(value).digest('hex'); } @@ -111,11 +105,11 @@ export async function preparePayloadReference(input: { || payload.attempt_id !== input.attemptId )) throw new Error('PAYLOAD_BOOTSTRAP_INVALID: worker attempt does not match the payload'); const objectPrefix = input.attemptId ? `${taskId}/${input.attemptId}` : taskId; - const manifest = canonical({ + const manifest = canonicalJson({ version: PAYLOAD_BOOTSTRAP.version, backend, platform_config: input.platformConfig ?? {}, }); const manifestKey = `${PAYLOAD_BOOTSTRAP.manifest_prefix}${sha256(manifest)}.json`; - const payloadBody = canonical({ + const payloadBody = canonicalJson({ version: PAYLOAD_BOOTSTRAP.version, task_id: taskId, agent_payload: payload, @@ -125,7 +119,7 @@ export async function preparePayloadReference(input: { || Buffer.byteLength(payloadBody) > PAYLOAD_BOOTSTRAP.max_payload_bytes) { throw new Error('PAYLOAD_BOOTSTRAP_TOO_LARGE: bootstrap manifest or task payload exceeds its byte limit'); } - const fingerprint = sha256(canonical({ bucket, backend, manifest, payloadBody })); + const fingerprint = sha256(canonicalJson({ bucket, backend, manifest, payloadBody })); const launchKey = `${objectPrefix}/${PAYLOAD_BOOTSTRAP.launch_filename}`; const accept = (saved: string): PayloadReference => { const record = JSON.parse(saved) as LaunchRecord; @@ -174,7 +168,7 @@ export async function preparePayloadReference(input: { payload_url: url, expires_at: now + lifetime * 1000, }; - const record = canonical({ fingerprint, reference }); + const record = canonicalJson({ fingerprint, reference }); return accept(await createOnce(bucket, launchKey, record)); } diff --git a/cdk/test/handlers/shared/canonical-json.test.ts b/cdk/test/handlers/shared/canonical-json.test.ts new file mode 100644 index 000000000..58d62f656 --- /dev/null +++ b/cdk/test/handlers/shared/canonical-json.test.ts @@ -0,0 +1,31 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import { canonicalJson } from '../../../src/handlers/shared/canonical-json'; + +test('preserves the existing receipt byte format across nested key insertion orders', () => { + const expected = '{"a":[{"b":2,"z":1},3],"z":null}'; + expect(canonicalJson({ z: null, a: [{ z: 1, b: 2 }, 3] })).toBe(expected); + expect(canonicalJson({ a: [{ b: 2, z: 1 }, 3], z: null })).toBe(expected); +}); + +test('keeps array order and distinct primitive values significant', () => { + expect(canonicalJson([1, '1', false, null])).toBe('[1,"1",false,null]'); + expect(canonicalJson([1, 2])).not.toBe(canonicalJson([2, 1])); +}); From 036beae7fcfd6b42ce3b56751c40d0bf2e6c6514 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Fri, 18 Sep 2026 19:33:53 -0400 Subject: [PATCH 100/149] test(microvm): exercise manifest rejection and restore rollback --- agent/tests/test_continuation_workspace.py | 24 +++++++++++++++++++ agent/tests/test_microvm_checkpoint.py | 4 +++- cdk/package.json | 5 +++- .../shared/linear-approval-reply.test.ts | 4 ++-- .../microvm-continuation-storage.test.ts | 23 ++++++++++++++++++ cdk/test/handlers/shared/orchestrator.test.ts | 1 + .../stacks/microvm-managed-image-nag.test.ts | 1 + 7 files changed, 58 insertions(+), 4 deletions(-) diff --git a/agent/tests/test_continuation_workspace.py b/agent/tests/test_continuation_workspace.py index 24bcf723d..e234b6c52 100644 --- a/agent/tests/test_continuation_workspace.py +++ b/agent/tests/test_continuation_workspace.py @@ -345,6 +345,30 @@ def race(source, target): class TestRestoreBoundaries: + def test_partial_publish_is_removed_after_a_move_failure(self, repo, monkeypatch): + archive, receipt = saved(repo) + remove_original(repo) + real_rename = os.rename + moved = [] + + def fail_second_move(source, target): + if Path(target).parent == repo: + if moved: + raise OSError("injected publish failure") + moved.append(Path(target)) + return real_rename(source, target) + + monkeypatch.setattr(workspace.os, "rename", fail_second_move) + with pytest.raises(workspace.WorkspaceCheckpointError) as error: + workspace.restore_workspace(archive, repo, IDENTITY, expected_sha256=receipt.sha256) + assert error.value.code == "restore_failed" + assert isinstance(error.value.__cause__, OSError) + assert "injected publish failure" in str(error.value.__cause__) + assert len(moved) == 1 + assert not repo.exists() + assert not list(repo.parent.glob(".workspace-restore-*")) + assert archive.exists() + def test_existing_workspace_is_never_cleared(self, repo): archive, receipt = saved(repo) before = inventory(repo) diff --git a/agent/tests/test_microvm_checkpoint.py b/agent/tests/test_microvm_checkpoint.py index ce8e87c42..12c0e7d4d 100644 --- a/agent/tests/test_microvm_checkpoint.py +++ b/agent/tests/test_microvm_checkpoint.py @@ -294,7 +294,9 @@ def test_failed_refresh_prevents_any_aws_reconciliation(self, state, monkeypatch LOCAL_ENDPOINT = os.environ.get("ABCA_DDB_LOCAL_ENDPOINT", "") if os.environ.get("CI") == "true" and not LOCAL_ENDPOINT: - raise RuntimeError("CI requires ABCA_DDB_LOCAL_ENDPOINT; checkpoint transaction tests must not skip") + raise RuntimeError( + "CI requires ABCA_DDB_LOCAL_ENDPOINT; checkpoint transaction tests must not skip" + ) @pytest.fixture diff --git a/cdk/package.json b/cdk/package.json index 538826403..bf838e8ab 100644 --- a/cdk/package.json +++ b/cdk/package.json @@ -4,7 +4,7 @@ "private": true, "description": "ABCA CDK application", "license": "MIT-0", - "_cedar_parity_pin": "DO NOT BUMP @cedar-policy/cedar-wasm IN ISOLATION. The Python cedarpy binding (agent/pyproject.toml) and this WASM binding share a Rust core and must move together. Drift between bindings can produce divergent (decision, matching_rule_ids) on the same (policy, input). See docs/design/CEDAR_HITL_GATES.md §15.6 (decision #23). The contracts/cedar-parity/ golden fixtures are how CI catches divergence; bumping either side requires bumping the other AND refreshing the fixtures in the same commit.", + "_cedar_parity_pin": "DO NOT BUMP @cedar-policy/cedar-wasm IN ISOLATION. The Python cedarpy binding (agent/pyproject.toml) and this WASM binding share a Rust core and must move together. Drift between bindings can produce divergent (decision, matching_rule_ids) on the same (policy, input). See docs/design/CEDAR_HITL_GATES.md \u00a715.6 (decision #23). The contracts/cedar-parity/ golden fixtures are how CI catches divergence; bumping either side requires bumping the other AND refreshing the fixtures in the same commit.", "scripts": { "compile": "tsc --build tsconfig.json", "watch": "tsc --build -w tsconfig.json", @@ -138,6 +138,9 @@ "tsconfig": "tsconfig.jest.json" } ] + }, + "moduleNameMapper": { + "^(\\.{1,2}/.*)\\.js$": "$1" } } } diff --git a/cdk/test/handlers/shared/linear-approval-reply.test.ts b/cdk/test/handlers/shared/linear-approval-reply.test.ts index 2fda92995..fe48da903 100644 --- a/cdk/test/handlers/shared/linear-approval-reply.test.ts +++ b/cdk/test/handlers/shared/linear-approval-reply.test.ts @@ -25,8 +25,8 @@ const mockApprove = jest.fn(); const mockDeny = jest.fn(); const mockPost = jest.fn(); const mockReadComment = jest.fn(); -jest.mock('../../../src/handlers/approve-task.js', () => ({ recordApprovalForUser: mockApprove }), { virtual: true }); -jest.mock('../../../src/handlers/deny-task.js', () => ({ recordDenialForUser: mockDeny }), { virtual: true }); +jest.mock('../../../src/handlers/approve-task', () => ({ recordApprovalForUser: mockApprove })); +jest.mock('../../../src/handlers/deny-task', () => ({ recordDenialForUser: mockDeny })); jest.mock('../../../src/handlers/shared/linear-feedback', () => ({ postIdentifiedComment: (...args: unknown[]) => mockPost(...args), readLinearApprovalComment: (...args: unknown[]) => mockReadComment(...args), diff --git a/cdk/test/handlers/shared/microvm-continuation-storage.test.ts b/cdk/test/handlers/shared/microvm-continuation-storage.test.ts index 1cd47252a..7da549f89 100644 --- a/cdk/test/handlers/shared/microvm-continuation-storage.test.ts +++ b/cdk/test/handlers/shared/microvm-continuation-storage.test.ts @@ -152,6 +152,29 @@ test('verifies both archive versions and checksums before permitting retirement' await expect(verifyContinuationCheckpoint(record)).rejects.toThrow('version, length or checksum'); }); +test.each(['checksum', 'length', 'identity', 'version'])('rejects a manifest %s mismatch before checking archives', async mismatch => { + const identity = { task_id: 'task', user_id: 'user', repo: 'owner/repo', attempt_id: 'vm', request_id: 'request' }; + const bytes = Buffer.from(JSON.stringify({ + version: mismatch === 'version' ? 999 : 1, + identity: mismatch === 'identity' ? { ...identity, user_id: 'other' } : identity, + })); + const record: ContinuationRecord = { + version: 1, + state: 'READY', + identity, + manifest: { + kind: 'manifest', + key: 'manifest', + version_id: 'version-1', + sha256: mismatch === 'checksum' ? '0'.repeat(64) : digest(bytes), + size_bytes: bytes.length + (mismatch === 'length' ? 1 : 0), + }, + }; + mockS3.mockResolvedValueOnce(object(bytes)); + await expect(verifyContinuationCheckpoint(record)).rejects.toThrow('MICROVM_CONTINUATION_STORAGE_INVALID'); + expect(mockS3).toHaveBeenCalledTimes(1); +}); + test('cleanup preserves pending data and removes every closed-task version before clearing its pointer', async () => { const options = { abortSignal: AbortSignal.timeout(1000) }; mockDdb.mockResolvedValueOnce({ Item: { user_id: 'user', status: 'AWAITING_APPROVAL' } }); diff --git a/cdk/test/handlers/shared/orchestrator.test.ts b/cdk/test/handlers/shared/orchestrator.test.ts index 74c547d27..71578f6f2 100644 --- a/cdk/test/handlers/shared/orchestrator.test.ts +++ b/cdk/test/handlers/shared/orchestrator.test.ts @@ -218,6 +218,7 @@ describe('MicroVM terminal finalization', () => { primeReread(status); await finish({ status: 'completed' }); expect(commandsOfType('Get')[0].input.ConsistentRead).toBe(true); + expect(commandsOfType('Update')).toHaveLength(1); for (const command of commandsOfType('Update')) { expect(command.input.ConditionExpression).toBeUndefined(); // terminal TTL stamp only } diff --git a/cdk/test/stacks/microvm-managed-image-nag.test.ts b/cdk/test/stacks/microvm-managed-image-nag.test.ts index 127575248..ea53e1cbd 100644 --- a/cdk/test/stacks/microvm-managed-image-nag.test.ts +++ b/cdk/test/stacks/microvm-managed-image-nag.test.ts @@ -36,6 +36,7 @@ describe.each(configurations)('managed MicroVM security checks (vault=$enableLin appProps: { context: { compute_type: 'lambda-microvm', + microvm_nested_stack: true, enableLinearIdentityVault, enableToolGateway: true, microvm_base_image_arn: 'arn:aws:lambda:us-west-2:aws:microvm-image:al2023-1', From f21881a2048eed3535fff54a95bc40a34974fbb8 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Fri, 18 Sep 2026 19:33:54 -0400 Subject: [PATCH 101/149] test(bootstrap): cover managed MicroVM and nested resource permissions --- cdk/src/bootstrap/resource-action-map.ts | 1 + cdk/test/bootstrap/synth-coverage.test.ts | 35 +++++++++++++++++++++-- 2 files changed, 34 insertions(+), 2 deletions(-) diff --git a/cdk/src/bootstrap/resource-action-map.ts b/cdk/src/bootstrap/resource-action-map.ts index 9d0603d9d..8b580da53 100644 --- a/cdk/src/bootstrap/resource-action-map.ts +++ b/cdk/src/bootstrap/resource-action-map.ts @@ -49,6 +49,7 @@ export const CFN_TYPES_WITHOUT_EXEC_ROLE_IAM = new Set([ * Parent issue: #350. Full synth-time aspect tracked in #125. */ export const RESOURCE_ACTION_MAP: Record = { + 'AWS::SSM::Parameter': ['ssm:PutParameter', 'ssm:GetParameters', 'ssm:DeleteParameter'], 'AWS::ApiGateway::Authorizer': ['apigateway:POST'], 'AWS::ApiGateway::Method': ['apigateway:POST'], 'AWS::ApiGateway::RequestValidator': ['apigateway:POST'], diff --git a/cdk/test/bootstrap/synth-coverage.test.ts b/cdk/test/bootstrap/synth-coverage.test.ts index 5e3e08b66..3a675ef34 100644 --- a/cdk/test/bootstrap/synth-coverage.test.ts +++ b/cdk/test/bootstrap/synth-coverage.test.ts @@ -34,6 +34,7 @@ describe('Bootstrap policy synth coverage', () => { let template: Template; let registryTemplate: Template; let allowedActions: Set; + let allTemplates: Template[]; beforeAll(() => { const app = new App(); @@ -41,6 +42,9 @@ describe('Bootstrap policy synth coverage', () => { env: { account: '123456789012', region: 'us-east-1' }, }); template = Template.fromStack(stack); + allTemplates = [template, ...stack.node.findAll() + .filter((node): node is NestedStack => node instanceof NestedStack) + .map(child => Template.fromStack(child))]; const registryStack = stack.node.tryFindChild('AgentRegistryStack') as AgentRegistryStack | undefined; if (!registryStack) { @@ -64,8 +68,8 @@ describe('Bootstrap policy synth coverage', () => { }); it('maps every synthesized CFN type (that needs IAM) to bootstrap actions', () => { - const resources = template.toJSON().Resources as Record; - const typesInTemplate = new Set(Object.values(resources).map((r) => r.Type)); + const typesInTemplate = new Set(allTemplates.flatMap(child => + Object.values(child.toJSON().Resources as Record).map(resource => resource.Type))); const unmapped: string[] = []; const missingByType: Record = {}; @@ -88,6 +92,33 @@ describe('Bootstrap policy synth coverage', () => { expect(missingByType).toEqual({}); }); + it('covers the managed MicroVM image, suspend parameter and every nested resource', () => { + const app = new App({ + context: { + compute_type: 'lambda-microvm', + microvm_nested_stack: true, + microvm_base_image_arn: 'arn:aws:lambda:us-east-1:aws:microvm-image:al2023-1', + microvm_base_image_version: '1', + microvm_artifact_sha256: 'a'.repeat(64), + }, + }); + const stack = new AgentStack(app, 'backgroundagent-microvm-coverage', { + env: { account: '123456789012', region: 'us-east-1' }, + }); + const root = Template.fromStack(stack); + const children = stack.node.findAll().filter((node): node is NestedStack => node instanceof NestedStack); + expect(children.length).toBeGreaterThan(0); + const types = new Set([root, ...children.map(child => Template.fromStack(child))].flatMap(child => + Object.values(child.toJSON().Resources as Record).map(resource => resource.Type))); + expect(types).toContain('AWS::SSM::Parameter'); + expect(types).toContain('AWS::Lambda::MicrovmImage'); + for (const type of types) { + if (CFN_TYPES_WITHOUT_EXEC_ROLE_IAM.has(type)) continue; + expect({ type, mapped: type in RESOURCE_ACTION_MAP }).toEqual({ type, mapped: true }); + expect({ type, missing: findMissingBootstrapActions(type, allowedActions) }).toEqual({ type, missing: [] }); + } + }); + it('maps the context-gated tool-gateway CFN types (ADR-019, not in default synth)', () => { // The default synth above never instantiates the ToolGateway construct, so // its two CFN types would slip past the coverage loop. Synthesize the gated From 421146a59ddda9799ff6a8eb8d282de53e744076 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Fri, 18 Sep 2026 19:33:54 -0400 Subject: [PATCH 102/149] docs(microvm): repair runbook links and clarify approval lifetime limits --- agent/AGENTS.md | 3 + cdk/AGENTS.md | 3 + cdk/scripts/README.md | 1 + .../handlers/shared/approval-notifications.ts | 2 +- .../shared/microvm-lifecycle-policy.ts | 1 - cdk/src/handlers/shared/orchestrator.ts | 6 +- docs/design/CEDAR_HITL_GATES.md | 11 +- docs/scripts/sync-starlight.mjs | 14 +- .../docs/architecture/Cedar-hitl-gates.md | 17 +- docs/src/content/docs/architecture/Compute.md | 4 +- .../docs/architecture/Deployment-roles.md | 2 +- .../content/docs/architecture/Orchestrator.md | 2 +- .../src/content/docs/architecture/Security.md | 4 +- ...Adr-021-lambda-microvms-compute-backend.md | 10 +- .../docs/using/Approval-gates-cedar-hitl.md | 2 +- .../645-nesting-probe-results.json | 793 ------------------ 16 files changed, 49 insertions(+), 826 deletions(-) delete mode 100644 docs/verification/645-nesting-probe-results.json diff --git a/agent/AGENTS.md b/agent/AGENTS.md index bb13fa087..67fc8e20f 100644 --- a/agent/AGENTS.md +++ b/agent/AGENTS.md @@ -27,6 +27,8 @@ Root `mise run build` includes `//agent:quality` in parallel with `//cdk:build`. | `progress_writer.py` | `agent/tests/test_progress_writer.py` | | `hooks.py`, `policy.py` | `agent/tests/test_hooks.py`, `test_policy.py` | | `pipeline.py`, `runner.py` | `agent/tests/test_pipeline.py`, etc. | +| `microvm_lifecycle.py`, `microvm_checkpoint.py` | `test_microvm_lifecycle.py`, `test_microvm_checkpoint.py`; checkpoint transaction tests require `ABCA_DDB_LOCAL_ENDPOINT` in CI | +| `continuation_*.py` | Matching `test_continuation_*.py`: capture, storage, safe restore, SDK session and usage recovery | Use `@pytest.fixture(autouse=True)` to reset shared module state between tests when handlers use circuit breakers or caches. @@ -91,4 +93,5 @@ def test_a(): - **Cedar parity** — `cedarpy==4.8.4` (agent) and `@cedar-policy/cedar-wasm` 4.8.2 (cdk) must move together. See [cdk/AGENTS.md](../cdk/AGENTS.md) and `docs/design/CEDAR_HITL_GATES.md` §15.6. - **Forgotten consumer** — Progress event schema changes need `cli/src/commands/watch.ts` and `test_progress_writer.py` updates. - **Image bundle** — CDK deploys this tree; root `mise run build` always runs agent quality. +- **Continuation compatibility** — the SDK pin must match `contracts/constants.json` → `microvm_continuation.verified_sdk_version`. Verify real SDK session recovery before updating both. Persisted checkpoint identities are versioned wire data; renaming a Python field can invalidate existing checkpoints. - **Un-attributed AWS SDK client (#319)** — build clients via `aws_session.tenant_client`/`tenant_resource` (tenant-scoped) or `aws_session.platform_client` (unscoped, still attributed); a naked `boto3.client(...)` silently drops solution attribution. diff --git a/cdk/AGENTS.md b/cdk/AGENTS.md index e328711e1..11c0c1f1c 100644 --- a/cdk/AGENTS.md +++ b/cdk/AGENTS.md @@ -29,6 +29,9 @@ mise //cdk:destroy # destroy stack | Shared handler logic | `cdk/test/handlers/shared/*.test.ts` | | Handler entrypoints | `cdk/test/handlers/orchestrate-task.test.ts`, `create-task.test.ts`, `webhook-create-task.test.ts` | | Constructs | `cdk/test/constructs/task-orchestrator.test.ts`, `task-api.test.ts` | +| MicroVM lifecycle/capacity transactions | `test/handlers/shared/*-local.test.ts`; use a loopback DynamoDB Local endpoint in `ABCA_DDB_LOCAL_ENDPOINT` (mandatory in CI) | +| Staged flat-to-nested migration helpers | `src/migration/`, `test/migration/`; these are not a deployable migration command | +| Live verification harnesses | `test/live/`; explicitly invoked against an owned fixture, excluded from normal Jest collection | Construct tests: synthesize each distinct stack config once in `beforeAll`, assert against cached `Template` — do not re-synth per test. Bundling is disabled globally via `test/setup/disable-bundling.ts` (see Common mistakes). diff --git a/cdk/scripts/README.md b/cdk/scripts/README.md index c42b4b20c..b8cc92bee 100644 --- a/cdk/scripts/README.md +++ b/cdk/scripts/README.md @@ -7,6 +7,7 @@ Bundling for Lambda assets is handled at synth time; the **`bundle`** task in ** | `generate-bootstrap-artifacts.ts` | Regenerates `cdk/bootstrap/policies/*.json`, `BOOTSTRAP_VERSION`, `BOOTSTRAP_HASH` from the typed policies in `src/bootstrap/policies/` | `mise //cdk:bootstrap:generate` | | `generate-bootstrap-template.ts` | Regenerates `cdk/bootstrap/bootstrap-template.yaml` (least-privilege CDK bootstrap, `ComputeTypes`-gated compute policies) | `mise //cdk:bootstrap:generate` | | `package-microvm-artifact.sh` | Packages `agent/` + `contracts/` + `Dockerfile` into the zip artifact an `AWS::Lambda::MicrovmImage` builds from, and uploads it to the CDK-created artifact bucket (ADR-021) | run directly — see the script header for the full bootstrap sequence | +| `build-microvm-artifact.py` | Builds the deterministic zip and digest consumed by the packaging script | Called by `package-microvm-artifact.sh`; `python3 cdk/scripts/build-microvm-artifact.py --help` for local packaging | `package-microvm-artifact.sh` exists because CloudFormation cannot produce its own MicroVM `codeArtifact`: the image resource consumes a zip that must already be in S3, and there is no CDK asset type for "zip + Dockerfile a MicroVM image builds from". Everything else on that backend (buckets, roles, network connectors, log group, the image resource itself) is CDK-managed by `src/constructs/lambda-microvm-compute.ts`, normally inside `lambda-microvm-stack.ts`. diff --git a/cdk/src/handlers/shared/approval-notifications.ts b/cdk/src/handlers/shared/approval-notifications.ts index a475f1fb3..4382d7a0d 100644 --- a/cdk/src/handlers/shared/approval-notifications.ts +++ b/cdk/src/handlers/shared/approval-notifications.ts @@ -227,7 +227,7 @@ export async function markApprovalNotificationDelivered( } } -/** Keep previews literal: repository text must not create mentions or Markdown links. */ +/** Escape preview Markdown, including backticks, so repository text cannot break out of its code fence. */ export function approvalNotificationMarkdown(notification: ApprovalNotification): string { return `\`\`\`text\n${notification.text.replace(/`/g, 'ˋ')}\n\`\`\``; } diff --git a/cdk/src/handlers/shared/microvm-lifecycle-policy.ts b/cdk/src/handlers/shared/microvm-lifecycle-policy.ts index e60aaef07..1a9a25785 100644 --- a/cdk/src/handlers/shared/microvm-lifecycle-policy.ts +++ b/cdk/src/handlers/shared/microvm-lifecycle-policy.ts @@ -24,7 +24,6 @@ import { MICROVM_SLEEP_AFTER_S_DEFAULT, MICROVM_SLEEP_AFTER_S_MAX } from './type import { TaskStatus, TERMINAL_STATUSES } from '../../constructs/task-status'; // Initial policy values, not service limits. Live timings must validate them. -export const MICROVM_SUSPEND_GRACE_MS = MICROVM_SLEEP_AFTER_S_DEFAULT * 1000; export const MICROVM_WAKE_MARGIN_MS = 60_000; export const MICROVM_MIN_USEFUL_SLEEP_MS = 30_000; export const MICROVM_TRANSITION_POLL_MS = 5_000; diff --git a/cdk/src/handlers/shared/orchestrator.ts b/cdk/src/handlers/shared/orchestrator.ts index 0f0fd4afd..b6afd9fe7 100644 --- a/cdk/src/handlers/shared/orchestrator.ts +++ b/cdk/src/handlers/shared/orchestrator.ts @@ -1118,7 +1118,8 @@ async function finalizeTaskOutcome(taskId: string, pollState: PollState, task: T // If still RUNNING / FINALIZING / AWAITING_APPROVAL after the poll // window closes, terminate the task through an allowed transition. Approval - // waits permit FAILED, not TIMED_OUT; their approval row remains agent-owned. + // waits permit FAILED, not TIMED_OUT. Task closure cancels pending approvals; + // reaching the execution limit is not a human denial. if ( currentStatus === TaskStatus.RUNNING || currentStatus === TaskStatus.FINALIZING @@ -1129,7 +1130,8 @@ async function finalizeTaskOutcome(taskId: string, pollState: PollState, task: T await transitionTask(taskId, currentStatus, terminalStatus, { completed_at: new Date().toISOString(), error_message: currentStatus === TaskStatus.AWAITING_APPROVAL - ? 'Orchestrator poll timeout exceeded while awaiting approval' + ? 'Task execution limit reached while waiting for approval. ' + + 'The task has closed; submit a new task to continue.' : 'Orchestrator poll timeout exceeded', }); } catch (err) { diff --git a/docs/design/CEDAR_HITL_GATES.md b/docs/design/CEDAR_HITL_GATES.md index 797a8f529..874cb737a 100644 --- a/docs/design/CEDAR_HITL_GATES.md +++ b/docs/design/CEDAR_HITL_GATES.md @@ -1358,20 +1358,23 @@ t=45m: Task #1 completes. count → 9. Bob can submit task #11. AgentCore Runtime's `maxLifetime = 28800s` (8h) is an absolute timer from session start. It does NOT pause during `AWAITING_APPROVAL`. -This has a concrete implication: the hook computes an `effective_timeout` bounded by `maxLifetime - remaining - CLEANUP_MARGIN_120S`. If the task has been running 7h55m and hits a soft-deny gate, the effective timeout might be clamped to a much shorter value than the task default. Below the 30s floor → immediate DENY with reason `"insufficient lifetime"`. +An explicitly timed approval is bounded by the remaining worker lifetime minus the 120-second cleanup margin. Below the 30-second floor, the hook denies the action with reason `"insufficient lifetime"`. + +The default approval timeout is `0`: no decision deadline. Worker lifetime does not turn silence into a human denial. ECS and AgentCore do not currently restore a waiting agent into a replacement worker; when their task execution limit is reached, the task closes and its pending approval is cancelled. MicroVM can retain a verified checkpoint and continue on a replacement worker. Setting its sleep delay to `0` disables early sleep/retirement, but the coordinator still attempts retirement before the service lifetime ends. This option trades idle cost for faster replies. ### 9.6 Stranded-approval reconciliation `reconcile-stranded-tasks.ts` has an AWAITING_APPROVAL-aware branch: -- Uses `APPROVAL_STRANDED_TIMEOUT_SECONDS`, default 7,200 seconds, measured from +- Uses `APPROVAL_STRANDED_TIMEOUT_SECONDS`, default 30,600 seconds (8.5 hours), measured from entry into the current status. It does not calculate twice each row's timeout. - Conditionally changes the task to `FAILED` if it is still awaiting approval, recording the elapsed wait and a recovery suggestion. - Emits `task_stranded`, `task_failed` and a wrapped `approval_stranded` milestone. The legacy milestone has no request ID. -- Leaves the approval row unchanged. The pending endpoint hides it because its - owning task is terminal; late approval is rejected by the task-state guard. +- Closes pending approval rows as `CANCELLED`. Late approval is also rejected by + the task-state guard. A saved MicroVM continuation is handled by its dedicated + coordinator instead of this timeout. The notification helper can recover that legacy milestone's request identity from the consistently read failed task and its saved stranded cause. It verifies diff --git a/docs/scripts/sync-starlight.mjs b/docs/scripts/sync-starlight.mjs index 037180aee..3a60dfeda 100644 --- a/docs/scripts/sync-starlight.mjs +++ b/docs/scripts/sync-starlight.mjs @@ -32,6 +32,12 @@ function rewriteDocsLinkTarget(target) { } const normalizedPath = pathPart.replaceAll('\\', '/'); + // Verification runbooks remain repository documents, not Starlight pages. + // Preserve their real destination instead of inventing an architecture route. + const verification = normalizedPath.match(/(?:^|\/)verification\/(.+)$/); + if (verification) { + return `https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/docs/verification/${verification[1]}${anchor ? `#${anchor}` : ''}`; + } const stem = path.basename(normalizedPath, '.md'); const slug = normalizeFileStem(stem).toLowerCase(); const anchorSuffix = anchor ? `#${anchor}` : ''; @@ -111,12 +117,8 @@ function ensureFrontmatter(content, title) { // must carry that prefix — otherwise they resolve to the domain root // and 404. Starlight prefixes its own nav links automatically, but our // rewritten body links are raw markdown and need it added explicitly - // (same reason the image rewrites above include docsBase). Every - // non-undefined return from rewriteDocsLinkTarget is a `/…` route (bare - // `#…` anchors and external links return undefined and keep their - // original text above), so the prefix always applies. - // (Fixes the broken in-body design-doc links.) - return `[${label}](${docsBase}${rewritten})`; + // Repository-only documents instead retain their absolute GitHub URL. + return `[${label}](${rewritten.startsWith('/') ? docsBase : ''}${rewritten})`; }); const trimmed = normalized.trimStart(); diff --git a/docs/src/content/docs/architecture/Cedar-hitl-gates.md b/docs/src/content/docs/architecture/Cedar-hitl-gates.md index 34d2db75b..60725cd9f 100644 --- a/docs/src/content/docs/architecture/Cedar-hitl-gates.md +++ b/docs/src/content/docs/architecture/Cedar-hitl-gates.md @@ -24,7 +24,7 @@ title: Cedar hitl gates > [continuation protocol](/sample-autonomous-cloud-coding-agents/architecture/orchestrator#retained-microvm-approvals). > The normal deployment passed retained-request, ten-minute sleep, explicit-expiry > and sleep-off/rollback acceptance; see the -> [deployment record](/sample-autonomous-cloud-coding-agents/architecture/readme). +> [deployment record](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/docs/verification/README.md). --- @@ -1043,7 +1043,7 @@ event. A later decision is rejected: the existing API returns `404 REQUEST_NOT_FOUND` for missing, foreign or already-decided approval rows, including a cancelled row. If approval committed first, cancellation preserves the recorded decision while cancelling the task. See the -[P3 approval verification record](/sample-autonomous-cloud-coding-agents/architecture/readme) +[P3 approval verification record](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/docs/verification/README.md) for source versus deployment status. ### 7.2 `POST /v1/tasks/{task_id}/deny` @@ -1362,20 +1362,23 @@ t=45m: Task #1 completes. count → 9. Bob can submit task #11. AgentCore Runtime's `maxLifetime = 28800s` (8h) is an absolute timer from session start. It does NOT pause during `AWAITING_APPROVAL`. -This has a concrete implication: the hook computes an `effective_timeout` bounded by `maxLifetime - remaining - CLEANUP_MARGIN_120S`. If the task has been running 7h55m and hits a soft-deny gate, the effective timeout might be clamped to a much shorter value than the task default. Below the 30s floor → immediate DENY with reason `"insufficient lifetime"`. +An explicitly timed approval is bounded by the remaining worker lifetime minus the 120-second cleanup margin. Below the 30-second floor, the hook denies the action with reason `"insufficient lifetime"`. + +The default approval timeout is `0`: no decision deadline. Worker lifetime does not turn silence into a human denial. ECS and AgentCore do not currently restore a waiting agent into a replacement worker; when their task execution limit is reached, the task closes and its pending approval is cancelled. MicroVM can retain a verified checkpoint and continue on a replacement worker. Setting its sleep delay to `0` disables early sleep/retirement, but the coordinator still attempts retirement before the service lifetime ends. This option trades idle cost for faster replies. ### 9.6 Stranded-approval reconciliation `reconcile-stranded-tasks.ts` has an AWAITING_APPROVAL-aware branch: -- Uses `APPROVAL_STRANDED_TIMEOUT_SECONDS`, default 7,200 seconds, measured from +- Uses `APPROVAL_STRANDED_TIMEOUT_SECONDS`, default 30,600 seconds (8.5 hours), measured from entry into the current status. It does not calculate twice each row's timeout. - Conditionally changes the task to `FAILED` if it is still awaiting approval, recording the elapsed wait and a recovery suggestion. - Emits `task_stranded`, `task_failed` and a wrapped `approval_stranded` milestone. The legacy milestone has no request ID. -- Leaves the approval row unchanged. The pending endpoint hides it because its - owning task is terminal; late approval is rejected by the task-state guard. +- Closes pending approval rows as `CANCELLED`. Late approval is also rejected by + the task-state guard. A saved MicroVM continuation is handled by its dedicated + coordinator instead of this timeout. The notification helper can recover that legacy milestone's request identity from the consistently read failed task and its saved stranded cause. It verifies @@ -1533,7 +1536,7 @@ cancellations, timeouts and stranded waits also produce messages. The response path supports the CLI and native Linear thread replies. Slack approval buttons and the Slack OAuth/button design below remain proposed. Email remains a log-only stub and GitHub does not receive approval messages. Deployment status is recorded in the -[P3 verification record](/sample-autonomous-cloud-coding-agents/architecture/readme). +[P3 verification record](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/docs/verification/README.md). **TaskApprovalsTable Streams are not consumed by the fan-out Lambda**. The approval row is working state; the audit trail is in TaskEventsTable. Enabling Streams on TaskApprovalsTable would be redundant and add noise. Final design: TaskApprovalsTable DOES NOT have Streams enabled. (Retains the `stream` attribute commented out for future use if needed.) diff --git a/docs/src/content/docs/architecture/Compute.md b/docs/src/content/docs/architecture/Compute.md index 7b6d30c24..d836f302d 100644 --- a/docs/src/content/docs/architecture/Compute.md +++ b/docs/src/content/docs/architecture/Compute.md @@ -85,11 +85,11 @@ Lambda MicroVMs are an opt-in third backend, selected per repository with `compu For managed images, run `cdk/scripts/package-microvm-artifact.sh --stack-name ` after each agent change. It packages the Dockerfile's local inputs into a deterministic ZIP and uploads it under `microvm-images/agent-artifact-.zip`. The checksum is a fingerprint of the uploaded bytes: identical inputs reuse the same verified object, and changed inputs produce a new filename. Deploy with the printed `--context microvm_artifact_sha256=` alongside `microvm_base_image_arn` and `microvm_base_image_version`, and retain these inputs for later deployments. The changed S3 URI tells CloudFormation to update the existing image; overwriting the old fixed filename alone does not. Missing or malformed digests fail synthesis. An initial deployment without an image must create the bucket first. The script's explicit `--create-image` alternative retains the fixed base key and calls the image API directly. -Because a snapshot freezes its build-time environment, current deployment identifiers arrive through the v2 payload bootstrap. The coordinator publishes a non-secret deployment manifest and sends a single-object signed download URL for the task. The worker reads only its deployment's `bootstrap/*` with ambient credentials; other object reads and payload-bucket listing are explicitly denied. The downloaded task identity and configuration must match the authenticated manifest before configuration installation. The serialized reference fits the verified 4,096-byte hook limit; all payload sizes use S3. ECS shares this transport through `AGENT_PAYLOAD_REF`. [ADR-021 §3](/sample-autonomous-cloud-coding-agents/architecture/adr-021-lambda-microvms-compute-backend#3-packaging-same-agent-image-source-new-build-path) defines the wire format and compatibility requirements; [live payload checks](/sample-autonomous-cloud-coding-agents/architecture/readme) record validation. +Because a snapshot freezes its build-time environment, current deployment identifiers arrive through the v2 payload bootstrap. The coordinator publishes a non-secret deployment manifest and sends a single-object signed download URL for the task. The worker reads only its deployment's `bootstrap/*` with ambient credentials; other object reads and payload-bucket listing are explicitly denied. The downloaded task identity and configuration must match the authenticated manifest before configuration installation. The serialized reference fits the verified 4,096-byte hook limit; all payload sizes use S3. ECS shares this transport through `AGENT_PAYLOAD_REF`. [ADR-021 §3](/sample-autonomous-cloud-coding-agents/architecture/adr-021-lambda-microvms-compute-backend#3-packaging-same-agent-image-source-new-build-path) defines the wire format and compatibility requirements; [live payload checks](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/docs/verification/README.md) record validation. Networking separates image build from execution: the build-only connector permits TCP 80 and 443 because the Dockerfile uses `apt-get`, while running MicroVMs retain 443-only egress through the platform VPC. Every launch explicitly passes the Lambda-managed `NO_INGRESS` connector; omission would select the service's public-ingress default. Images declare and serve `/ready` and `/validate` at build time and `/run`, `/terminate`, `/suspend` and `/resume` at runtime. Automatic suspension is a separate deployment opt-in. Registry HTTP/SSE tools therefore need reachable HTTPS/443 endpoints; remote non-443 tools are unsupported under the default policies of AgentCore and ECS as well. Local `stdio` tools can run, with their outbound traffic subject to the same restriction. Asset resolution/loading does not probe connectivity; see [registry network support](/sample-autonomous-cloud-coding-agents/architecture/registry#2-asset-kinds-for-mvp). -P3 connects the guest checkpoint/credential hooks, actual image-version capability, durable supervisor and post-commit approval wake. The `microvm_approval_suspend_enabled` deployment context defaults false and controls both a static opt-in and a live Parameter Store switch. Existing durable executions reread the live switch before new suspension because their original Lambda environment is pinned. Recovery and the original service lifetime survive supervisor replay; the API preserves accepted decisions when optional wake fails. AgentCore/ECS retain explicit unsupported pause/wake results. Explicit approval deadlines use the original UTC/monotonic deadline. See the [lifecycle diagnostics](/sample-autonomous-cloud-coding-agents/architecture/645-p3-lifecycle-diagnostics) for failure investigation, and the [acceptance status](/sample-autonomous-cloud-coding-agents/architecture/readme) for deployment evidence. +P3 connects the guest checkpoint/credential hooks, actual image-version capability, durable supervisor and post-commit approval wake. The `microvm_approval_suspend_enabled` deployment context defaults false and controls both a static opt-in and a live Parameter Store switch. Existing durable executions reread the live switch before new suspension because their original Lambda environment is pinned. Recovery and the original service lifetime survive supervisor replay; the API preserves accepted decisions when optional wake fails. AgentCore/ECS retain explicit unsupported pause/wake results. Explicit approval deadlines use the original UTC/monotonic deadline. See the [lifecycle diagnostics](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/docs/verification/645-p3-lifecycle-diagnostics.md) for failure investigation, and the [acceptance status](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/docs/verification/README.md) for deployment evidence. Tasks can set `microvm_sleep_after_s` (CLI: `--microvm-sleep-after `). The default is 600 seconds of waiting for each approval; zero disables sleep. diff --git a/docs/src/content/docs/architecture/Deployment-roles.md b/docs/src/content/docs/architecture/Deployment-roles.md index a0e4e714a..7657444c8 100644 --- a/docs/src/content/docs/architecture/Deployment-roles.md +++ b/docs/src/content/docs/architecture/Deployment-roles.md @@ -857,7 +857,7 @@ exact names in addition to the legacy flat-layout prefixes. The execution role stays in the parent and is still excluded. Re-bootstrap before deploying the child stack. Existing flat deployments must keep `microvm_nested_stack=false` until their resource migration is reviewed; changing ownership is not an ordinary -in-place update. See the [nested-stack runbook](/sample-autonomous-cloud-coding-agents/architecture/645-p3-nested-stack). +in-place update. See the [nested-stack runbook](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/docs/verification/645-p3-nested-stack.md). For a reviewed migration that keeps old and new resources side by side, `microvm_resource_name_prefix` gives the nested image, network connectors and log diff --git a/docs/src/content/docs/architecture/Orchestrator.md b/docs/src/content/docs/architecture/Orchestrator.md index e5f6d60f1..f54421ac9 100644 --- a/docs/src/content/docs/architecture/Orchestrator.md +++ b/docs/src/content/docs/architecture/Orchestrator.md @@ -300,7 +300,7 @@ When the session is unhealthy, the task transitions to `FAILED` with "Agent sess - A terminal substrate report paired with a non-terminal task first checks for a complete, acknowledged approval checkpoint. Such a checkpoint can retire the old attempt and retain the task for a replacement. Without one, finalization strongly re-reads the task row before classifying a substrate failure. - Substrate state detects a dead VM; heartbeat staleness detects loss of the heartbeat writer inside a VM that still reports `RUNNING`. The independent heartbeat thread can continue during a pipeline hang, so a fresh timestamp is not proof of progress. -The P3 supervisor saves intent before control calls and rechecks the gate before and after them. Its durable state retains an absolute service lifetime, consecutive failures, recovery start time and next delay. Three failed cycles or 120 seconds of unconfirmed wake cannot become an indefinite wait. AWS RUNNING does not end recovery while the guest remains stuck on a decided/expired approval; fresh guest liveness is required. API approve/deny commit first, then attempt a bounded wake without changing the decision response. Automatic suspension defaults off via `microvm_approval_suspend_enabled`; disabling new sleep preserves wake and cleanup. See the [lifecycle diagnostics](/sample-autonomous-cloud-coding-agents/architecture/645-p3-lifecycle-diagnostics). +The P3 supervisor saves intent before control calls and rechecks the gate before and after them. Its durable state retains an absolute service lifetime, consecutive failures, recovery start time and next delay. Three failed cycles or 120 seconds of unconfirmed wake cannot become an indefinite wait. AWS RUNNING does not end recovery while the guest remains stuck on a decided/expired approval; fresh guest liveness is required. API approve/deny commit first, then attempt a bounded wake without changing the decision response. Automatic suspension defaults off via `microvm_approval_suspend_enabled`; disabling new sleep preserves wake and cleanup. See the [lifecycle diagnostics](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/docs/verification/645-p3-lifecycle-diagnostics.md). Unanswered approvals have no deadline by default. A checkpointed MicroVM wait can retire after an hour, or before that worker's lifetime ends. Retirement and diff --git a/docs/src/content/docs/architecture/Security.md b/docs/src/content/docs/architecture/Security.md index bb0830531..6bb431540 100644 --- a/docs/src/content/docs/architecture/Security.md +++ b/docs/src/content/docs/architecture/Security.md @@ -48,13 +48,13 @@ Three authentication mechanisms protect the platform, matching its input channel - **DynamoDB**: item access on the four `task_id`-partitioned tables (task, events, approvals, nudges) is gated by a `dynamodb:LeadingKeys` condition equal to `${aws:PrincipalTag/task_id}` and requires that context key to be present. `Scan` is not granted. The main task table permits reads and only attribute-scoped `UpdateItem` writes: the agent can report status, heartbeat, results and approval details, but cannot replace/delete the row or edit coordinator-owned start receipts, capacity reservations, owner identity or compute handles. The reviewed write list is `cdk/src/constructs/agent-task-write-attributes.json`; Python tests exercise the current writers against it. Supporting tables retain task-scoped item writes. `LeadingKeys` binds the base-table partition key, not a GSI key such as `user_id`. - **S3**: trace writes and attachment reads are scoped to the `/${aws:PrincipalTag/user_id}/` object prefix. -The compute role retains only non-tenant access (Bedrock model invocation — already ARN-scoped; CloudWatch Logs; the GitHub PAT secret, read once before the SessionRole is assumed; AgentCore Memory) plus `sts:AssumeRole`/`sts:TagSession` on the SessionRole. Because the agent runs under credentials that are themselves an assumed role, its `AssumeRole` is *role chaining* — capped at one hour regardless of the role's max session duration — so the agent uses a **refreshable** credential provider that re-assumes before expiry (tasks can run up to the 8-hour `maxLifetime`). The design is backend-agnostic: the same SessionRole and agent code serve all three backends (AgentCore Runtime, ECS Fargate, Lambda MicroVMs). Existing scoped credentials are constrained to their tags, but the compute role chooses those tags when assuming the role. The trust policy does not independently bind a worker to one task; this is not complete isolation against a compromised worker with ambient credentials. No worker needs direct access to the shared capacity counter. ECS's legacy configuration without a session role retains unscoped task reads/reporting updates, with the same attribute restriction. The policy structure and Python writer compatibility are tested locally. A historical IAM simulator check covered the original tag conditions; it does not verify the new attribute restriction or deployed transaction authorization. The [acceptance checklist](/sample-autonomous-cloud-coding-agents/architecture/readme#live-acceptance-for-an-installation) includes effective-role authorization checks. +The compute role retains only non-tenant access (Bedrock model invocation — already ARN-scoped; CloudWatch Logs; the GitHub PAT secret, read once before the SessionRole is assumed; AgentCore Memory) plus `sts:AssumeRole`/`sts:TagSession` on the SessionRole. Because the agent runs under credentials that are themselves an assumed role, its `AssumeRole` is *role chaining* — capped at one hour regardless of the role's max session duration — so the agent uses a **refreshable** credential provider that re-assumes before expiry (tasks can run up to the 8-hour `maxLifetime`). The design is backend-agnostic: the same SessionRole and agent code serve all three backends (AgentCore Runtime, ECS Fargate, Lambda MicroVMs). Existing scoped credentials are constrained to their tags, but the compute role chooses those tags when assuming the role. The trust policy does not independently bind a worker to one task; this is not complete isolation against a compromised worker with ambient credentials. No worker needs direct access to the shared capacity counter. ECS's legacy configuration without a session role retains unscoped task reads/reporting updates, with the same attribute restriction. The policy structure and Python writer compatibility are tested locally. A historical IAM simulator check covered the original tag conditions; it does not verify the new attribute restriction or deployed transaction authorization. The [acceptance checklist](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/docs/verification/README.md#live-acceptance-for-an-installation) includes effective-role authorization checks. **Lambda MicroVMs compute-role delta** — The compute role additionally reads the per-workspace channel-OAuth secrets (`bgagent-linear-oauth-*`, `bgagent-jira-oauth-*`) and its payload bucket's `bootstrap/*` manifests. Ambient task-object reads and payload-bucket listing are explicitly denied. It also has read-only `ec2:DescribeAvailabilityZones` for repository CDK synthesis; that API has no resource-level scope. Recorded live calls rejected source-conditioned MicroVM role trust and service-conditioned PassRole grants. The integration therefore omits those conditions on the affected paths, while restricting which roles can be passed and which resources they may access. Reintroduce conditions only after verifying service support. [ADR-021 §4](/sample-autonomous-cloud-coding-agents/architecture/adr-021-lambda-microvms-compute-backend#4-infra-and-iam-conditional-resources-behind-bootstrap-computetypes) records the role responsibilities and limits. -**ECS/MicroVM task bootstrap (v2)** — The coordinator publishes non-secret deployment manifests and supplies a short-lived signed URL for exactly one task's payload. The worker reads the manifest with its ambient role, whose explicit deny outside its own `bootstrap/*` prevents a foreign public bucket from authorizing fake configuration. It verifies downloaded task identity and exact manifest/config agreement before installing MicroVM settings. The coordinator privately persists the URL outside TaskTable for retry and deletes both task objects at finalization. Worker roles cannot read/list task objects using their ambient credentials. A signed URL is a bearer capability: redact it in errors and keep it out of task rows and repository subprocess environments. Existing session-tag trust and other platform grants remain separate limitations. The [payload guide](/sample-autonomous-cloud-coding-agents/architecture/645-payload-bootstrap) describes coordinated image/coordinator/policy upgrades; the [acceptance summary](/sample-autonomous-cloud-coding-agents/architecture/readme) distinguishes recorded AWS checks from remaining validation. +**ECS/MicroVM task bootstrap (v2)** — The coordinator publishes non-secret deployment manifests and supplies a short-lived signed URL for exactly one task's payload. The worker reads the manifest with its ambient role, whose explicit deny outside its own `bootstrap/*` prevents a foreign public bucket from authorizing fake configuration. It verifies downloaded task identity and exact manifest/config agreement before installing MicroVM settings. The coordinator privately persists the URL outside TaskTable for retry and deletes both task objects at finalization. Worker roles cannot read/list task objects using their ambient credentials. A signed URL is a bearer capability: redact it in errors and keep it out of task rows and repository subprocess environments. Existing session-tag trust and other platform grants remain separate limitations. The [payload guide](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/docs/verification/645-payload-bootstrap.md) describes coordinated image/coordinator/policy upgrades; the [acceptance summary](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/docs/verification/README.md) distinguishes recorded AWS checks from remaining validation. > Out of scope for this control and tracked separately as GitHub issues: replacing the shared GitHub PAT (GitHub App / Token Vault), binding credentials to the MicroVM via attestation, and scoping AgentCore Memory (namespace isolation by `actorId`/`sessionId` remains its boundary). diff --git a/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md b/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md index b9e421f0e..2011e6737 100644 --- a/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md +++ b/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md @@ -4,7 +4,7 @@ title: Adr 021 lambda microvms compute backend # ADR-021: AWS Lambda MicroVMs as a third ComputeStrategy backend -> **Implementation status (2026-09-18): P1 and P2 are merged; P3 is in draft review.** Approval sleep/wake, retained requests, conversation/workspace recovery and nested infrastructure have live acceptance evidence. Reusable migration and final integration checks remain open; see [verification status](/sample-autonomous-cloud-coding-agents/architecture/readme). This ADR defines P1–P3, not an official P4. +> **Implementation status (2026-09-18): P1 and P2 are merged; P3 is in draft review.** Approval sleep/wake, retained requests, conversation/workspace recovery and nested infrastructure have live acceptance evidence. Reusable migration and final integration checks remain open; see [verification status](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/docs/verification/README.md). This ADR defines P1–P3, not an official P4. **Status:** proposed **Date:** 2026-07-29 @@ -74,7 +74,7 @@ Unanswered approvals have no deadline by default (`approval_timeout_s=0`). Expli Automatic suspension also requires the deployment’s `microvm_approval_suspend_enabled` opt-in, which defaults false for new deployments. A live Parameter Store switch lets existing durable executions stop initiating new suspensions without changing their pinned Lambda environment. The verified normal deployment has this opt-in enabled. Turning it off does not abandon already-suspended workers. -The coordinator observes the exact pending gate and waits for the sleep delay. It skips sleep when a timed gate has too little time left before its wake margin. The guest holds a coding barrier, drains acknowledged progress and commits a checkpoint before accepting `/suspend`. Lifecycle HTTP responses explicitly close their connections before freeze to avoid reuse of a stale pooled connection; see the [transport evidence summary](/sample-autonomous-cloud-coding-agents/architecture/readme#recorded-acceptance). +The coordinator observes the exact pending gate and waits for the sleep delay. It skips sleep when a timed gate has too little time left before its wake margin. The guest holds a coding barrier, drains acknowledged progress and commits a checkpoint before accepting `/suspend`. Lifecycle HTTP responses explicitly close their connections before freeze to avoid reuse of a stale pooled connection; see the [transport evidence summary](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/docs/verification/README.md#recorded-acceptance). Approval and denial handlers commit the decision first. They then read the current handle consistently, persist wake intent and request resume best-effort. A wake failure records diagnostics and does not undo the accepted decision. The durable supervisor retries and observes both service state and guest consumption of the decision; RUNNING alone does not prove the tool was released. @@ -131,7 +131,7 @@ Normative requirements (EARS): - Signed URLs shall not appear in ordinary logs, agent-readable task rows or repository subprocess environments. Downloads shall use the exact regional S3 HTTPS object without redirects/proxies and with bounded response sizes. - Producers, images and IAM shall be upgraded together; incompatible workers must be drained before switching transport. -The [payload contract](/sample-autonomous-cloud-coding-agents/architecture/645-payload-bootstrap) and [live checks](/sample-autonomous-cloud-coding-agents/architecture/readme) record the implementation and validation. This transport does not establish complete hostile-worker isolation: other platform grants remain, and a stolen signed URL is usable until expiry or revocation. +The [payload contract](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/docs/verification/645-payload-bootstrap.md) and [live checks](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/docs/verification/README.md) record the implementation and validation. This transport does not establish complete hostile-worker isolation: other platform grants remain, and a stolen signed URL is usable until expiry or revocation. No ABCA endpoint consumer exists in P1–P3. The platform grants no `CreateMicrovmAuthToken` permission and mints no JWE tokens. `NO_INGRESS` can still return an endpoint URL; an unauthenticated 403 verifies the authentication boundary, not valid-token reachability. @@ -153,7 +153,7 @@ The backend adds build/runtime VPC connectors, build artifacts, launch payloads, `lambda:PassNetworkConnector` has no resource-level authorization support and therefore uses `Resource: *`. The operator role also has wildcard ENI permissions; some mutations can be scoped in IAM, so their current tested wildcard is not evidence that narrower permissions are impossible. DescribeAvailabilityZones needs a wildcard for fresh CDK repository lookups. These exceptions and namespace wildcards are documented in the construct’s cdk-nag suppressions. -**Nested infrastructure.** `microvm_nested_stack=true` puts MicroVM resources in a nested stack. The shared execution role stays in the parent to avoid a role-trust dependency cycle. Bootstrap 1.9.0 covers nested deployment roles. Existing flat installations need the reviewed overlap/drain migration, explicit image/coordinator pins and rollback checks described in the [migration prerequisites](/sample-autonomous-cloud-coding-agents/architecture/645-p3-nested-stack). Reusable migration commands are still being completed. Moving construct paths alone is not a safe migration. Preserve `microvm_resource_name_prefix` after a migrated deployment. +**Nested infrastructure.** `microvm_nested_stack=true` puts MicroVM resources in a nested stack. The shared execution role stays in the parent to avoid a role-trust dependency cycle. Bootstrap 1.9.0 covers nested deployment roles. Existing flat installations need the reviewed overlap/drain migration, explicit image/coordinator pins and rollback checks described in the [migration prerequisites](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/docs/verification/645-p3-nested-stack.md). Reusable migration commands are still being completed. Moving construct paths alone is not a safe migration. Preserve `microvm_resource_name_prefix` after a migrated deployment. #### Security bar vs existing backends ([#645](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/645) acceptance criterion) @@ -203,7 +203,7 @@ Changing the default backend, GPU support, native Slack approval buttons, approv ## Testing -The [verification summary](/sample-autonomous-cloud-coding-agents/architecture/readme) distinguishes recorded live acceptance from open PR checks. Required coverage includes: +The [verification summary](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/docs/verification/README.md) distinguishes recorded live acceptance from open PR checks. Required coverage includes: - Strategy state mapping, explicit unsupported results, ARN/Region validation, NO_INGRESS, omitted idlePolicy and bounded uncertain-start recovery. - Hook readiness/warm-up, AWS-silent build hooks, authenticated payload installation, arbitrary terminate bodies and lifecycle connection closure. diff --git a/docs/src/content/docs/using/Approval-gates-cedar-hitl.md b/docs/src/content/docs/using/Approval-gates-cedar-hitl.md index b78570952..8a68216f6 100644 --- a/docs/src/content/docs/using/Approval-gates-cedar-hitl.md +++ b/docs/src/content/docs/using/Approval-gates-cedar-hitl.md @@ -141,4 +141,4 @@ pauses can cost more than staying awake. The API equivalent is `microvm_sleep_after_s` (zero means off); task details return the saved setting. Automatic suspension is disabled by default for new deployments. An operator enables it after [verifying the deployed image and -coordinator](/sample-autonomous-cloud-coding-agents/architecture/readme#live-acceptance-for-an-installation). \ No newline at end of file +coordinator](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/docs/verification/README.md#live-acceptance-for-an-installation). \ No newline at end of file diff --git a/docs/verification/645-nesting-probe-results.json b/docs/verification/645-nesting-probe-results.json deleted file mode 100644 index e2eee3f56..000000000 --- a/docs/verification/645-nesting-probe-results.json +++ /dev/null @@ -1,793 +0,0 @@ -{ - "source_commit": "5e10038c7e28179b302ac4de78b709795aeba3ce", - "date": "2026-09-13", - "method": "Offline TypeScript loader transforms only; no production nesting edits; bundling disabled; vault guard disabled only in probe; parent-role mode also derives names from parent stack; resource cycle validation uses Template.fromStack", - "results": [ - { - "name": "default", - "mode": "baseline", - "templates": [ - { - "file": "backgroundagent-dev.template.json", - "resources": 454, - "bytes": 476779, - "parameters": 1, - "outputs": 41, - "names": [], - "roles": [] - }, - { - "file": "backgroundagentdevAgentRegistryStackFF9DF145.nested.template.json", - "resources": 19, - "bytes": 36474, - "parameters": 0, - "outputs": 2, - "names": [], - "roles": [] - }, - { - "file": "backgroundagentdevRegistryApiF3ADADE2.nested.template.json", - "resources": 35, - "bytes": 47684, - "parameters": 3, - "outputs": 3, - "names": [], - "roles": [] - } - ] - }, - { - "name": "microvm", - "mode": "baseline", - "templates": [ - { - "file": "backgroundagent-dev.template.json", - "resources": 471, - "bytes": 507859, - "parameters": 1, - "outputs": 48, - "names": [ - { - "type": "AWS::Lambda::NetworkConnector", - "name": "backgroundagent-dev-microvm-egress" - }, - { - "type": "AWS::Lambda::NetworkConnector", - "name": "backgroundagent-dev-microvm-build-egress" - } - ], - "roles": [ - "LambdaMicrovmComputeConnectorOperatorRoleE206977E", - "LambdaMicrovmComputeBuildRoleF0A13AC7", - "LambdaMicrovmComputeExecutionRoleAA0C4A0D" - ] - }, - { - "file": "backgroundagentdevAgentRegistryStackFF9DF145.nested.template.json", - "resources": 19, - "bytes": 36474, - "parameters": 0, - "outputs": 2, - "names": [], - "roles": [] - }, - { - "file": "backgroundagentdevRegistryApiF3ADADE2.nested.template.json", - "resources": 35, - "bytes": 47684, - "parameters": 3, - "outputs": 3, - "names": [], - "roles": [] - } - ] - }, - { - "name": "microvm-image", - "mode": "baseline", - "templates": [ - { - "file": "backgroundagent-dev.template.json", - "resources": 472, - "bytes": 510894, - "parameters": 1, - "outputs": 48, - "names": [ - { - "type": "AWS::Lambda::NetworkConnector", - "name": "backgroundagent-dev-microvm-egress" - }, - { - "type": "AWS::Lambda::NetworkConnector", - "name": "backgroundagent-dev-microvm-build-egress" - }, - { - "type": "AWS::Lambda::MicrovmImage", - "name": "backgroundagent-dev-abca-agent" - } - ], - "roles": [ - "LambdaMicrovmComputeConnectorOperatorRoleE206977E", - "LambdaMicrovmComputeBuildRoleF0A13AC7", - "LambdaMicrovmComputeExecutionRoleAA0C4A0D" - ] - }, - { - "file": "backgroundagentdevAgentRegistryStackFF9DF145.nested.template.json", - "resources": 19, - "bytes": 36474, - "parameters": 0, - "outputs": 2, - "names": [], - "roles": [] - }, - { - "file": "backgroundagentdevRegistryApiF3ADADE2.nested.template.json", - "resources": 35, - "bytes": 47684, - "parameters": 3, - "outputs": 3, - "names": [], - "roles": [] - } - ] - }, - { - "name": "microvm-image-gateway", - "mode": "baseline", - "templates": [ - { - "file": "backgroundagent-dev.template.json", - "resources": 479, - "bytes": 516182, - "parameters": 1, - "outputs": 48, - "names": [ - { - "type": "AWS::Lambda::NetworkConnector", - "name": "backgroundagent-dev-microvm-egress" - }, - { - "type": "AWS::Lambda::NetworkConnector", - "name": "backgroundagent-dev-microvm-build-egress" - }, - { - "type": "AWS::Lambda::MicrovmImage", - "name": "backgroundagent-dev-abca-agent" - } - ], - "roles": [ - "LambdaMicrovmComputeConnectorOperatorRoleE206977E", - "LambdaMicrovmComputeBuildRoleF0A13AC7", - "LambdaMicrovmComputeExecutionRoleAA0C4A0D" - ] - }, - { - "file": "backgroundagentdevAgentRegistryStackFF9DF145.nested.template.json", - "resources": 19, - "bytes": 36474, - "parameters": 0, - "outputs": 2, - "names": [], - "roles": [] - }, - { - "file": "backgroundagentdevRegistryApiF3ADADE2.nested.template.json", - "resources": 35, - "bytes": 47684, - "parameters": 3, - "outputs": 3, - "names": [], - "roles": [] - } - ] - }, - { - "name": "microvm-image-gateway-vault", - "mode": "baseline", - "templates": [ - { - "file": "backgroundagent-dev.template.json", - "resources": 489, - "bytes": 536098, - "parameters": 1, - "outputs": 50, - "names": [ - { - "type": "AWS::Lambda::NetworkConnector", - "name": "backgroundagent-dev-microvm-egress" - }, - { - "type": "AWS::Lambda::NetworkConnector", - "name": "backgroundagent-dev-microvm-build-egress" - }, - { - "type": "AWS::Lambda::MicrovmImage", - "name": "backgroundagent-dev-abca-agent" - } - ], - "roles": [ - "LambdaMicrovmComputeConnectorOperatorRoleE206977E", - "LambdaMicrovmComputeBuildRoleF0A13AC7", - "LambdaMicrovmComputeExecutionRoleAA0C4A0D" - ] - }, - { - "file": "backgroundagentdevAgentRegistryStackFF9DF145.nested.template.json", - "resources": 19, - "bytes": 36474, - "parameters": 0, - "outputs": 2, - "names": [], - "roles": [] - }, - { - "file": "backgroundagentdevLinearVaultConsentPageStack671B3D9F.nested.template.json", - "resources": 12, - "bytes": 17737, - "parameters": 0, - "outputs": 1, - "names": [], - "roles": [] - }, - { - "file": "backgroundagentdevRegistryApiF3ADADE2.nested.template.json", - "resources": 35, - "bytes": 47684, - "parameters": 3, - "outputs": 3, - "names": [], - "roles": [] - } - ] - }, - { - "name": "default", - "mode": "raw-nested", - "templates": [ - { - "file": "backgroundagent-dev.template.json", - "resources": 454, - "bytes": 476779, - "parameters": 1, - "outputs": 41, - "names": [], - "roles": [] - }, - { - "file": "backgroundagentdevAgentRegistryStackFF9DF145.nested.template.json", - "resources": 19, - "bytes": 36474, - "parameters": 0, - "outputs": 2, - "names": [], - "roles": [] - }, - { - "file": "backgroundagentdevRegistryApiF3ADADE2.nested.template.json", - "resources": 35, - "bytes": 47684, - "parameters": 3, - "outputs": 3, - "names": [], - "roles": [] - } - ] - }, - { - "name": "microvm", - "mode": "raw-nested", - "templates": [ - { - "file": "backgroundagent-dev.template.json", - "resources": 455, - "bytes": 480295, - "parameters": 1, - "outputs": 48, - "names": [], - "roles": [] - }, - { - "file": "backgroundagentdevAgentRegistryStackFF9DF145.nested.template.json", - "resources": 19, - "bytes": 36474, - "parameters": 0, - "outputs": 2, - "names": [], - "roles": [] - }, - { - "file": "backgroundagentdevMicrovmNestedCD7454C8.nested.template.json", - "resources": 19, - "bytes": 41875, - "parameters": 7, - "outputs": 6, - "names": [ - { - "type": "AWS::Lambda::NetworkConnector", - "name": "--Token-TOKEN-5450---microvm-egress" - }, - { - "type": "AWS::Lambda::NetworkConnector", - "name": "--Token-TOKEN-5450---microvm-build-egress" - } - ], - "roles": [ - "LambdaMicrovmComputeConnectorOperatorRoleE206977E", - "LambdaMicrovmComputeBuildRoleF0A13AC7", - "LambdaMicrovmComputeExecutionRoleAA0C4A0D" - ] - }, - { - "file": "backgroundagentdevRegistryApiF3ADADE2.nested.template.json", - "resources": 35, - "bytes": 47684, - "parameters": 3, - "outputs": 3, - "names": [], - "roles": [] - } - ], - "cycleError": "Template is undeployable, these resources have a dependency cycle: TaskApiDeployment093F4F13e3bbe4ed71a6615db6cbe11f92059df4 -> TaskApijirawebhookPOST96251C09 -> JiraIntegrationWebhookFn7CE96A20 -> JiraIntegrationWebhookFnServiceRoleDefaultPolicy40DD76AF -> JiraIntegrationWebhookProcessorFnE5EB0D16 -> JiraIntegrationWebhookProcessorFnServiceRoleDefaultPolicy4E7A0586 -> TaskOrchestratorOrchestratorFnCurrentVersionAliaslive08F80EEB -> TaskOrchestratorOrchestratorFn8CE22C41 -> TaskOrchestratorOrchestratorFnServiceRoleDefaultPolicyDECF0D43 -> Runtime99E3DDFA -> RuntimeExecutionRoleDefaultPolicy2B020CFC -> AgentSessionRoleB6C61074 -> MicrovmNestedNestedStackMicrovmNestedNestedStackResource6BAFC965 -> AgentSessionRoleB6C61074:" - }, - { - "name": "microvm-image", - "mode": "raw-nested", - "templates": [ - { - "file": "backgroundagent-dev.template.json", - "resources": 455, - "bytes": 482894, - "parameters": 1, - "outputs": 48, - "names": [], - "roles": [] - }, - { - "file": "backgroundagentdevAgentRegistryStackFF9DF145.nested.template.json", - "resources": 19, - "bytes": 36474, - "parameters": 0, - "outputs": 2, - "names": [], - "roles": [] - }, - { - "file": "backgroundagentdevMicrovmNestedCD7454C8.nested.template.json", - "resources": 20, - "bytes": 44561, - "parameters": 7, - "outputs": 9, - "names": [ - { - "type": "AWS::Lambda::NetworkConnector", - "name": "--Token-TOKEN-8353---microvm-egress" - }, - { - "type": "AWS::Lambda::NetworkConnector", - "name": "--Token-TOKEN-8353---microvm-build-egress" - }, - { - "type": "AWS::Lambda::MicrovmImage", - "name": "--Token-TOKEN-8353---abca-agent" - } - ], - "roles": [ - "LambdaMicrovmComputeConnectorOperatorRoleE206977E", - "LambdaMicrovmComputeBuildRoleF0A13AC7", - "LambdaMicrovmComputeExecutionRoleAA0C4A0D" - ] - }, - { - "file": "backgroundagentdevRegistryApiF3ADADE2.nested.template.json", - "resources": 35, - "bytes": 47684, - "parameters": 3, - "outputs": 3, - "names": [], - "roles": [] - } - ], - "cycleError": "Template is undeployable, these resources have a dependency cycle: TaskApiDeployment093F4F13e3bbe4ed71a6615db6cbe11f92059df4 -> TaskApijirawebhookPOST96251C09 -> JiraIntegrationWebhookFn7CE96A20 -> JiraIntegrationWebhookFnServiceRoleDefaultPolicy40DD76AF -> JiraIntegrationWebhookProcessorFnE5EB0D16 -> JiraIntegrationWebhookProcessorFnServiceRoleDefaultPolicy4E7A0586 -> TaskOrchestratorOrchestratorFnCurrentVersionAliaslive08F80EEB -> TaskOrchestratorOrchestratorFn8CE22C41 -> TaskOrchestratorOrchestratorFnServiceRoleDefaultPolicyDECF0D43 -> MicrovmNestedNestedStackMicrovmNestedNestedStackResource6BAFC965 -> AgentSessionRoleB6C61074 -> MicrovmNestedNestedStackMicrovmNestedNestedStackResource6BAFC965:" - }, - { - "name": "microvm-image-gateway", - "mode": "raw-nested", - "templates": [ - { - "file": "backgroundagent-dev.template.json", - "resources": 462, - "bytes": 488182, - "parameters": 1, - "outputs": 48, - "names": [], - "roles": [] - }, - { - "file": "backgroundagentdevAgentRegistryStackFF9DF145.nested.template.json", - "resources": 19, - "bytes": 36474, - "parameters": 0, - "outputs": 2, - "names": [], - "roles": [] - }, - { - "file": "backgroundagentdevMicrovmNestedCD7454C8.nested.template.json", - "resources": 20, - "bytes": 44565, - "parameters": 7, - "outputs": 9, - "names": [ - { - "type": "AWS::Lambda::NetworkConnector", - "name": "--Token-TOKEN-11313---microvm-egress" - }, - { - "type": "AWS::Lambda::NetworkConnector", - "name": "--Token-TOKEN-11313---microvm-build-egress" - }, - { - "type": "AWS::Lambda::MicrovmImage", - "name": "--Token-TOKEN-11313---abca-agent" - } - ], - "roles": [ - "LambdaMicrovmComputeConnectorOperatorRoleE206977E", - "LambdaMicrovmComputeBuildRoleF0A13AC7", - "LambdaMicrovmComputeExecutionRoleAA0C4A0D" - ] - }, - { - "file": "backgroundagentdevRegistryApiF3ADADE2.nested.template.json", - "resources": 35, - "bytes": 47684, - "parameters": 3, - "outputs": 3, - "names": [], - "roles": [] - } - ], - "cycleError": "Template is undeployable, these resources have a dependency cycle: TaskApiDeployment093F4F13e3bbe4ed71a6615db6cbe11f92059df4 -> TaskApijirawebhookPOST96251C09 -> JiraIntegrationWebhookFn7CE96A20 -> JiraIntegrationWebhookFnServiceRoleDefaultPolicy40DD76AF -> JiraIntegrationWebhookProcessorFnE5EB0D16 -> JiraIntegrationWebhookProcessorFnServiceRoleDefaultPolicy4E7A0586 -> TaskOrchestratorOrchestratorFnCurrentVersionAliaslive08F80EEB -> TaskOrchestratorOrchestratorFn8CE22C41 -> TaskOrchestratorOrchestratorFnServiceRoleDefaultPolicyDECF0D43 -> MicrovmNestedNestedStackMicrovmNestedNestedStackResource6BAFC965 -> AgentSessionRoleB6C61074 -> MicrovmNestedNestedStackMicrovmNestedNestedStackResource6BAFC965:" - }, - { - "name": "microvm-image-gateway-vault", - "mode": "raw-nested", - "templates": [ - { - "file": "backgroundagent-dev.template.json", - "resources": 472, - "bytes": 506765, - "parameters": 1, - "outputs": 50, - "names": [], - "roles": [] - }, - { - "file": "backgroundagentdevAgentRegistryStackFF9DF145.nested.template.json", - "resources": 19, - "bytes": 36474, - "parameters": 0, - "outputs": 2, - "names": [], - "roles": [] - }, - { - "file": "backgroundagentdevLinearVaultConsentPageStack671B3D9F.nested.template.json", - "resources": 12, - "bytes": 17737, - "parameters": 0, - "outputs": 1, - "names": [], - "roles": [] - }, - { - "file": "backgroundagentdevMicrovmNestedCD7454C8.nested.template.json", - "resources": 20, - "bytes": 46955, - "parameters": 7, - "outputs": 9, - "names": [ - { - "type": "AWS::Lambda::NetworkConnector", - "name": "--Token-TOKEN-14382---microvm-egress" - }, - { - "type": "AWS::Lambda::NetworkConnector", - "name": "--Token-TOKEN-14382---microvm-build-egress" - }, - { - "type": "AWS::Lambda::MicrovmImage", - "name": "--Token-TOKEN-14382---abca-agent" - } - ], - "roles": [ - "LambdaMicrovmComputeConnectorOperatorRoleE206977E", - "LambdaMicrovmComputeBuildRoleF0A13AC7", - "LambdaMicrovmComputeExecutionRoleAA0C4A0D" - ] - }, - { - "file": "backgroundagentdevRegistryApiF3ADADE2.nested.template.json", - "resources": 35, - "bytes": 47684, - "parameters": 3, - "outputs": 3, - "names": [], - "roles": [] - } - ], - "cycleError": "Template is undeployable, these resources have a dependency cycle: TaskApiDeployment093F4F13e3bbe4ed71a6615db6cbe11f92059df4 -> TaskApijirawebhookPOST96251C09 -> JiraIntegrationWebhookFn7CE96A20 -> JiraIntegrationWebhookFnServiceRoleDefaultPolicy40DD76AF -> JiraIntegrationWebhookProcessorFnE5EB0D16 -> JiraIntegrationWebhookProcessorFnServiceRoleDefaultPolicy4E7A0586 -> TaskOrchestratorOrchestratorFnCurrentVersionAliaslive08F80EEB -> TaskOrchestratorOrchestratorFn8CE22C41 -> TaskOrchestratorOrchestratorFnServiceRoleDefaultPolicyDECF0D43 -> MicrovmNestedNestedStackMicrovmNestedNestedStackResource6BAFC965 -> AgentSessionRoleB6C61074 -> MicrovmNestedNestedStackMicrovmNestedNestedStackResource6BAFC965:" - }, - { - "name": "default", - "mode": "parent-role", - "templates": [ - { - "file": "backgroundagent-dev.template.json", - "resources": 454, - "bytes": 476779, - "parameters": 1, - "outputs": 41, - "names": [], - "roles": [] - }, - { - "file": "backgroundagentdevAgentRegistryStackFF9DF145.nested.template.json", - "resources": 19, - "bytes": 36474, - "parameters": 0, - "outputs": 2, - "names": [], - "roles": [] - }, - { - "file": "backgroundagentdevRegistryApiF3ADADE2.nested.template.json", - "resources": 35, - "bytes": 47684, - "parameters": 3, - "outputs": 3, - "names": [], - "roles": [] - } - ] - }, - { - "name": "microvm", - "mode": "parent-role", - "templates": [ - { - "file": "backgroundagent-dev.template.json", - "resources": 457, - "bytes": 488288, - "parameters": 1, - "outputs": 48, - "names": [], - "roles": [] - }, - { - "file": "backgroundagentdevAgentRegistryStackFF9DF145.nested.template.json", - "resources": 19, - "bytes": 36474, - "parameters": 0, - "outputs": 2, - "names": [], - "roles": [] - }, - { - "file": "backgroundagentdevMicrovmNestedCD7454C8.nested.template.json", - "resources": 17, - "bytes": 29503, - "parameters": 3, - "outputs": 6, - "names": [ - { - "type": "AWS::Lambda::NetworkConnector", - "name": "backgroundagent-dev-microvm-egress" - }, - { - "type": "AWS::Lambda::NetworkConnector", - "name": "backgroundagent-dev-microvm-build-egress" - } - ], - "roles": [ - "LambdaMicrovmComputeConnectorOperatorRoleE206977E", - "LambdaMicrovmComputeBuildRoleF0A13AC7" - ] - }, - { - "file": "backgroundagentdevRegistryApiF3ADADE2.nested.template.json", - "resources": 35, - "bytes": 47684, - "parameters": 3, - "outputs": 3, - "names": [], - "roles": [] - } - ] - }, - { - "name": "microvm-image", - "mode": "parent-role", - "templates": [ - { - "file": "backgroundagent-dev.template.json", - "resources": 457, - "bytes": 490655, - "parameters": 1, - "outputs": 48, - "names": [], - "roles": [] - }, - { - "file": "backgroundagentdevAgentRegistryStackFF9DF145.nested.template.json", - "resources": 19, - "bytes": 36474, - "parameters": 0, - "outputs": 2, - "names": [], - "roles": [] - }, - { - "file": "backgroundagentdevMicrovmNestedCD7454C8.nested.template.json", - "resources": 18, - "bytes": 31884, - "parameters": 3, - "outputs": 8, - "names": [ - { - "type": "AWS::Lambda::NetworkConnector", - "name": "backgroundagent-dev-microvm-egress" - }, - { - "type": "AWS::Lambda::NetworkConnector", - "name": "backgroundagent-dev-microvm-build-egress" - }, - { - "type": "AWS::Lambda::MicrovmImage", - "name": "backgroundagent-dev-abca-agent" - } - ], - "roles": [ - "LambdaMicrovmComputeConnectorOperatorRoleE206977E", - "LambdaMicrovmComputeBuildRoleF0A13AC7" - ] - }, - { - "file": "backgroundagentdevRegistryApiF3ADADE2.nested.template.json", - "resources": 35, - "bytes": 47684, - "parameters": 3, - "outputs": 3, - "names": [], - "roles": [] - } - ] - }, - { - "name": "microvm-image-gateway", - "mode": "parent-role", - "templates": [ - { - "file": "backgroundagent-dev.template.json", - "resources": 464, - "bytes": 495943, - "parameters": 1, - "outputs": 48, - "names": [], - "roles": [] - }, - { - "file": "backgroundagentdevAgentRegistryStackFF9DF145.nested.template.json", - "resources": 19, - "bytes": 36474, - "parameters": 0, - "outputs": 2, - "names": [], - "roles": [] - }, - { - "file": "backgroundagentdevMicrovmNestedCD7454C8.nested.template.json", - "resources": 18, - "bytes": 31884, - "parameters": 3, - "outputs": 8, - "names": [ - { - "type": "AWS::Lambda::NetworkConnector", - "name": "backgroundagent-dev-microvm-egress" - }, - { - "type": "AWS::Lambda::NetworkConnector", - "name": "backgroundagent-dev-microvm-build-egress" - }, - { - "type": "AWS::Lambda::MicrovmImage", - "name": "backgroundagent-dev-abca-agent" - } - ], - "roles": [ - "LambdaMicrovmComputeConnectorOperatorRoleE206977E", - "LambdaMicrovmComputeBuildRoleF0A13AC7" - ] - }, - { - "file": "backgroundagentdevRegistryApiF3ADADE2.nested.template.json", - "resources": 35, - "bytes": 47684, - "parameters": 3, - "outputs": 3, - "names": [], - "roles": [] - } - ] - }, - { - "name": "microvm-image-gateway-vault", - "mode": "parent-role", - "templates": [ - { - "file": "backgroundagent-dev.template.json", - "resources": 474, - "bytes": 515859, - "parameters": 1, - "outputs": 50, - "names": [], - "roles": [] - }, - { - "file": "backgroundagentdevAgentRegistryStackFF9DF145.nested.template.json", - "resources": 19, - "bytes": 36474, - "parameters": 0, - "outputs": 2, - "names": [], - "roles": [] - }, - { - "file": "backgroundagentdevLinearVaultConsentPageStack671B3D9F.nested.template.json", - "resources": 12, - "bytes": 17737, - "parameters": 0, - "outputs": 1, - "names": [], - "roles": [] - }, - { - "file": "backgroundagentdevMicrovmNestedCD7454C8.nested.template.json", - "resources": 18, - "bytes": 31884, - "parameters": 3, - "outputs": 8, - "names": [ - { - "type": "AWS::Lambda::NetworkConnector", - "name": "backgroundagent-dev-microvm-egress" - }, - { - "type": "AWS::Lambda::NetworkConnector", - "name": "backgroundagent-dev-microvm-build-egress" - }, - { - "type": "AWS::Lambda::MicrovmImage", - "name": "backgroundagent-dev-abca-agent" - } - ], - "roles": [ - "LambdaMicrovmComputeConnectorOperatorRoleE206977E", - "LambdaMicrovmComputeBuildRoleF0A13AC7" - ] - }, - { - "file": "backgroundagentdevRegistryApiF3ADADE2.nested.template.json", - "resources": 35, - "bytes": 47684, - "parameters": 3, - "outputs": 3, - "names": [], - "roles": [] - } - ] - } - ] -} From 09b79411eff66c5d76f62c9ea908378de71af6ee Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Fri, 18 Sep 2026 19:55:49 -0400 Subject: [PATCH 103/149] fix(approvals): move worker writes behind a task-scoped trusted service --- agent/src/approval_requests.py | 116 ++++++++++ agent/src/task_state.py | 34 +++ agent/tests/test_approval_requests.py | 96 +++++++++ agent/tests/test_task_state.py | 4 +- cdk/src/constructs/agent-session-role.ts | 32 ++- .../constructs/approval-request-service.ts | 103 +++++++++ cdk/src/constructs/ecs-agent-cluster.ts | 12 +- cdk/src/constructs/task-orchestrator.ts | 4 + cdk/src/handlers/request-approval.ts | 200 ++++++++++++++++++ cdk/src/handlers/shared/response.ts | 2 + cdk/src/stacks/agent.ts | 14 +- .../constructs/agent-session-role.test.ts | 30 ++- .../approval-request-service.test.ts | 66 ++++++ cdk/test/constructs/ecs-agent-cluster.test.ts | 14 +- .../constructs/lambda-microvm-compute.test.ts | 3 + .../constructs/lambda-microvm-stack.test.ts | 3 + .../handlers/request-approval-local.test.ts | 179 ++++++++++++++++ cdk/test/handlers/request-approval.test.ts | 131 ++++++++++++ cdk/test/stacks/agent.test.ts | 2 +- contracts/constants.json | 1 + docs/design/CEDAR_HITL_GATES.md | 42 +++- docs/guides/DEPLOYMENT_GUIDE.md | 26 +++ .../docs/architecture/Cedar-hitl-gates.md | 42 +++- .../docs/getting-started/Deployment-guide.md | 26 +++ 24 files changed, 1136 insertions(+), 46 deletions(-) create mode 100644 agent/src/approval_requests.py create mode 100644 agent/tests/test_approval_requests.py create mode 100644 cdk/src/constructs/approval-request-service.ts create mode 100644 cdk/src/handlers/request-approval.ts create mode 100644 cdk/test/constructs/approval-request-service.test.ts create mode 100644 cdk/test/handlers/request-approval-local.test.ts create mode 100644 cdk/test/handlers/request-approval.test.ts diff --git a/agent/src/approval_requests.py b/agent/src/approval_requests.py new file mode 100644 index 000000000..5c36ecc2e --- /dev/null +++ b/agent/src/approval_requests.py @@ -0,0 +1,116 @@ +# Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. +# SPDX-License-Identifier: MIT-0 + +"""IAM-signed requests to the control-plane approval writer.""" + +import json +import os +import re +from http import HTTPStatus +from urllib.parse import urlsplit + +import requests +from botocore.auth import SigV4Auth +from botocore.awsrequest import AWSRequest +from botocore.exceptions import ClientError + +from aws_session import get_session +from microvm_lifecycle import get_context +from ua import sanitize_ua_value, static_user_agent_extra + +API_ENV = "APPROVAL_REQUESTS_API_URL" + + +def configured() -> bool: + return bool(os.environ.get(API_ENV)) + + +def record_request( + operation: str, + task_id: str, + request_id: str, + *, + approval: dict | None = None, + reason: str | None = None, +) -> None: + """Sign the task path with scoped credentials; never follow redirects.""" + endpoint = os.environ.get(API_ENV, "").rstrip("/") + parsed = urlsplit(endpoint) + match = re.fullmatch( + r"[a-z0-9]+\.execute-api\.([a-z0-9-]+)\.amazonaws\.com(?:\.cn)?", parsed.hostname or "" + ) + if ( + parsed.scheme != "https" + or not match + or parsed.username + or parsed.password + or parsed.port not in (None, 443) + or parsed.query + or parsed.fragment + or not re.fullmatch(r"/[A-Za-z0-9_-]+", parsed.path) + or not re.fullmatch(r"[A-Za-z0-9_-]{1,128}", task_id) + or operation not in {"create", "timeout"} + ): + raise ValueError("Invalid approval request endpoint, task id or operation") + payload = {"operation": operation, "task_id": task_id, "request_id": request_id} + if approval is not None: + payload["approval"] = approval + if reason is not None: + payload["reason"] = reason + lifecycle = get_context(task_id) + if lifecycle is not None: + payload["worker_attempt_id"] = lifecycle.attempt_id + body = json.dumps(payload, separators=(",", ":")).encode() + user_agent = static_user_agent_extra() + app_id = os.environ.get("AWS_SDK_UA_APP_ID") + if app_id: + user_agent += f" app/{sanitize_ua_value(app_id)}" + url = f"{endpoint}/tasks/{task_id}" + request = AWSRequest( + method="POST", + url=url, + data=body, + headers={"Content-Type": "application/json", "User-Agent": user_agent}, + ) + credentials = get_session().get_credentials() + if credentials is None: + raise RuntimeError("Approval request credentials unavailable") + SigV4Auth(credentials.get_frozen_credentials(), "execute-api", match.group(1)).add_auth(request) + response = requests.post( + url, + data=body, + headers=dict(request.headers), + timeout=(5, 20), + allow_redirects=False, + ) + try: + result = response.json() + except ValueError as exc: + raise RuntimeError( + f"Approval service returned invalid JSON (HTTP {response.status_code})" + ) from exc + data = result.get("data") if isinstance(result, dict) else None + if response.status_code == HTTPStatus.OK and isinstance(data, dict) and data.get("ok") is True: + return + error = result.get("error") if isinstance(result, dict) else None + code = ( + error.get("code", "ApprovalServiceUnavailable") + if isinstance(error, dict) + else "ApprovalServiceUnavailable" + ) + details = error.get("details") if isinstance(error, dict) else None + reasons = details.get("cancellation_reasons", []) if isinstance(details, dict) else [] + raise ClientError( + { + "Error": { + "Code": code, + "Message": "Control plane did not acknowledge the approval write", + }, + "CancellationReasons": reasons, + "ResponseMetadata": { + "HTTPStatusCode": response.status_code, + "RequestId": error.get("request_id") if isinstance(error, dict) else None, + }, + }, + "RecordApprovalRequest", + ) diff --git a/agent/src/task_state.py b/agent/src/task_state.py index 669bf3889..f4f499167 100644 --- a/agent/src/task_state.py +++ b/agent/src/task_state.py @@ -688,6 +688,23 @@ def transact_write_approval_request( DDB-layer exceptions propagate so the hook's outer try/except can fail-closed with a specific reason. """ + import approval_requests + + if approval_requests.configured(): + try: + approval_requests.record_request( + "create", task_id, request_id, approval=dict(approval_row) + ) + return + except Exception as exc: + if _extract_error_code(exc) == "TransactionCanceledException": + reasons = _extract_cancellation_reasons(exc) + raise ApprovalWriteError( + f"approval write cancelled: reasons={reasons}", cancellation_reasons=reasons + ) from exc + raise + # Compatibility with older deployments. New stacks grant no direct writes, + # so a missing service URL fails closed there rather than bypassing the broker. task_table, approvals_table = _require_tables() ddb = _get_ddb_client(client=client) @@ -1003,6 +1020,23 @@ def best_effort_update_approval_status( Returns ``True`` on successful write, ``False`` on ``ConditionalCheckFailedException``. All other errors propagate. """ + import approval_requests + + if new_status != "TIMED_OUT": + raise ValueError("Workers may record only non-human timeouts") + if approval_requests.configured(): + try: + approval_requests.record_request("timeout", task_id, request_id, reason=reason) + return True + except Exception as exc: + reasons = _extract_cancellation_reasons(exc) + if _extract_error_code(exc) == "TransactionCanceledException" and ( + reasons + and reasons[0].get("Code") == "ConditionalCheckFailed" + and all(reason.get("Code") == "None" for reason in reasons[1:]) + ): + return False + raise _, approvals_table = _require_tables() ddb = _get_ddb_client(client=client) diff --git a/agent/tests/test_approval_requests.py b/agent/tests/test_approval_requests.py new file mode 100644 index 000000000..f58cc28be --- /dev/null +++ b/agent/tests/test_approval_requests.py @@ -0,0 +1,96 @@ +# Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. +# SPDX-License-Identifier: MIT-0 + +import json +from types import SimpleNamespace +from unittest.mock import MagicMock + +import pytest +from botocore.credentials import Credentials +from botocore.exceptions import ClientError + +import approval_requests as broker +import task_state + + +@pytest.fixture +def transport(monkeypatch): + monkeypatch.setenv(broker.API_ENV, "https://fixture.execute-api.us-east-1.amazonaws.com/v1/") + session = MagicMock() + session.get_credentials.return_value = Credentials( + "testing", "testing-secret", "testing-session" + ) + monkeypatch.setattr(broker, "get_session", lambda: session) + post = MagicMock( + return_value=SimpleNamespace(status_code=200, json=lambda: {"data": {"ok": True}}) + ) + monkeypatch.setattr(broker.requests, "post", post) + return post + + +def test_scoped_signature_binds_the_task_path_and_does_not_redirect(transport): + broker.record_request("create", "task", "request", approval={"status": "PENDING"}) + args, kwargs = transport.call_args + assert args == ("https://fixture.execute-api.us-east-1.amazonaws.com/v1/tasks/task",) + assert "execute-api/aws4_request" in kwargs["headers"]["Authorization"] + assert kwargs["headers"]["X-Amz-Security-Token"] == "testing-session" + assert "md/uksb-wt64nei4u6#agent" in kwargs["headers"]["User-Agent"] + assert kwargs["allow_redirects"] is False + assert json.loads(kwargs["data"])["task_id"] == "task" + + +@pytest.mark.parametrize( + "url", + [ + "http://fixture.execute-api.us-east-1.amazonaws.com/v1", + "https://attacker.example/v1", + "https://fixture.execute-api.us-east-1.amazonaws.com/v1?redirect=elsewhere", + ], +) +def test_rejects_untrusted_endpoints_before_signing(transport, monkeypatch, url): + monkeypatch.setenv(broker.API_ENV, url) + with pytest.raises(ValueError): + broker.record_request("create", "task", "request") + transport.assert_not_called() + + +def test_new_writers_use_service_instead_of_direct_dynamodb(transport, monkeypatch): + direct = MagicMock() + monkeypatch.setattr(task_state, "_get_ddb_client", direct) + task_state.transact_write_approval_request("task", "request", {"status": "PENDING"}) + assert task_state.best_effort_update_approval_status("task", "request", "TIMED_OUT") + assert transport.call_count == 2 + direct.assert_not_called() + + +@pytest.mark.parametrize("status", ["APPROVED", "DENIED", "PENDING", "CANCELLED"]) +def test_worker_api_cannot_record_a_human_decision(transport, status): + with pytest.raises(ValueError, match="non-human timeouts"): + task_state.best_effort_update_approval_status("task", "request", status) + transport.assert_not_called() + + +def test_uncertain_service_write_does_not_fall_back_to_direct_dynamodb(transport, monkeypatch): + direct = MagicMock() + monkeypatch.setattr(task_state, "_get_ddb_client", direct) + transport.side_effect = TimeoutError("reply lost") + with pytest.raises(TimeoutError): + task_state.transact_write_approval_request("task", "request", {"status": "PENDING"}) + direct.assert_not_called() + + +def test_timeout_race_preserves_human_winner_but_lease_loss_is_not_benign(transport): + reasons = [{"Code": "ConditionalCheckFailed"}, {"Code": "None"}, {"Code": "None"}] + transport.return_value = SimpleNamespace( + status_code=409, + json=lambda: { + "error": { + "code": "TransactionCanceledException", + "details": {"cancellation_reasons": reasons}, + }, + }, + ) + assert not task_state.best_effort_update_approval_status("task", "request", "TIMED_OUT") + reasons[-1] = {"Code": "ConditionalCheckFailed"} + with pytest.raises(ClientError): + task_state.best_effort_update_approval_status("task", "request", "TIMED_OUT") diff --git a/agent/tests/test_task_state.py b/agent/tests/test_task_state.py index 36f5bc923..5d3a254b7 100644 --- a/agent/tests/test_task_state.py +++ b/agent/tests/test_task_state.py @@ -1051,12 +1051,12 @@ def test_reason_optional_attached(self, approval_tables_env): client.update_item.return_value = {} task_state.best_effort_update_approval_status( - "01KTASK", "01KREQ", "DENIED", reason="no prod pushes", client=client + "01KTASK", "01KREQ", "TIMED_OUT", reason="polling failed", client=client ) call = client.update_item.call_args assert "deny_reason = :reason" in call.kwargs["UpdateExpression"] - assert call.kwargs["ExpressionAttributeValues"][":reason"] == {"S": "no prod pushes"} + assert call.kwargs["ExpressionAttributeValues"][":reason"] == {"S": "polling failed"} def test_conditional_check_failed_returns_false(self, approval_tables_env): """IMPL-24 — this is the VM-throttle race signal the hook re-reads on.""" diff --git a/cdk/src/constructs/agent-session-role.ts b/cdk/src/constructs/agent-session-role.ts index 030139e7a..1c9453d58 100644 --- a/cdk/src/constructs/agent-session-role.ts +++ b/cdk/src/constructs/agent-session-role.ts @@ -31,7 +31,7 @@ import constants from '../../../contracts/constants.json'; * Task reporting may update only the attributes written by task_state.py. * Keep whole-row replacement/deletion and coordinator metadata out of this * grant. DynamoDB evaluates each transaction item using its item action, so - * approval UpdateItem operations receive the same restriction. + * approval-related TaskTable updates receive the same restriction. * * This protects writes, not reads: agents may read their complete task record. * The JSON list is also checked against actual Python writer requests in tests. @@ -80,6 +80,22 @@ export function grantAgentTaskTableAccess( })); } +/** Workers observe decisions; only the control plane writes approval records. */ +export function grantAgentApprovalReadAccess( + table: dynamodb.ITable, grantee: iam.IGrantable, taskScoped: boolean, +): void { + grantee.grantPrincipal.addToPrincipalPolicy(new iam.PolicyStatement({ + actions: ['dynamodb:GetItem', 'dynamodb:BatchGetItem', 'dynamodb:Query', 'dynamodb:ConditionCheckItem'], + resources: [table.tableArn], + ...(taskScoped ? { + conditions: { + 'ForAllValues:StringEquals': { 'dynamodb:LeadingKeys': ['${aws:PrincipalTag/task_id}'] }, + 'Null': { 'dynamodb:LeadingKeys': 'false' }, + }, + } : {}), + })); +} + /** S3 key prefixes the agent writes/reads, scoped per tenant. */ const TRACE_KEY_PREFIX = 'traces'; const ATTACHMENT_KEY_PREFIX = 'attachments'; @@ -104,11 +120,13 @@ export interface AgentSessionRoleProps { * capacity reservations belong to the coordinator. */ readonly taskTable: dynamodb.ITable; + /** Decisions and notification markers are control-plane-owned. */ + readonly approvalsTable: dynamodb.ITable; /** - * Supporting task-scoped tables (events, approvals, nudges), all partitioned - * by `task_id`. Do not include taskTable here: that would bypass its write - * restriction. The SessionRole receives item-level access constrained by a + * Supporting task-scoped tables (events, nudges), all partitioned + * by `task_id`. Do not include taskTable or approvalsTable here: that would + * bypass their write restrictions. The SessionRole receives access constrained by a * `dynamodb:LeadingKeys` condition on `aws:PrincipalTag/task_id`, so a * session can only touch its own task's rows. Order is irrelevant. */ @@ -202,8 +220,9 @@ export class AgentSessionRole extends Construct { 'AgentSessionRole requires at least one assuming role (the compute role[s] that mint scoped credentials)', ); } - if (props.taskScopedTables.some((table) => table.tableArn === props.taskTable.tableArn)) { - throw new Error('taskTable must not appear in taskScopedTables; it requires restricted writes'); + if (props.taskScopedTables.some((table) => + [props.taskTable.tableArn, props.approvalsTable.tableArn].includes(table.tableArn))) { + throw new Error('taskTable and approvalsTable must not appear in taskScopedTables; they require restricted writes'); } const [firstAssumingRole] = props.assumingRoles; @@ -223,6 +242,7 @@ export class AgentSessionRole extends Construct { }); grantAgentTaskTableAccess(props.taskTable, this.role, true); + grantAgentApprovalReadAccess(props.approvalsTable, this.role, true); // --- Supporting tables: item access gated by task_id leading-key --- // One statement per table keeps the resource ARNs explicit. The condition diff --git a/cdk/src/constructs/approval-request-service.ts b/cdk/src/constructs/approval-request-service.ts new file mode 100644 index 000000000..f53521d90 --- /dev/null +++ b/cdk/src/constructs/approval-request-service.ts @@ -0,0 +1,103 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import * as path from 'node:path'; +import { Duration, NestedStack } from 'aws-cdk-lib'; +import * as apigw from 'aws-cdk-lib/aws-apigateway'; +import type * as dynamodb from 'aws-cdk-lib/aws-dynamodb'; +import * as iam from 'aws-cdk-lib/aws-iam'; +import { Architecture, Runtime } from 'aws-cdk-lib/aws-lambda'; +import { NodejsFunction } from 'aws-cdk-lib/aws-lambda-nodejs'; +import * as logs from 'aws-cdk-lib/aws-logs'; +import { NagSuppressions } from 'cdk-nag'; +import type { Construct } from 'constructs'; + +const REQUEST_TIMEOUT_SECONDS = 15; +/** + * Trusted approval writer. A separate API keeps routes inside this child stack. + * IAM binds a signed POST path to the session's task tag; direct Lambda invoke + * would lose that binding because Lambda does not expose caller session tags. + */ +export class ApprovalRequestService extends NestedStack { + public readonly api: apigw.RestApi; + public readonly fn: NodejsFunction; + + constructor(scope: Construct, id: string, props: { + taskTable: dynamodb.ITable; + approvalsTable: dynamodb.ITable; + }) { + super(scope, id); + this.fn = new NodejsFunction(this, 'RequestFn', { + entry: path.join(__dirname, '..', 'handlers', 'request-approval.ts'), + handler: 'handler', + runtime: Runtime.NODEJS_24_X, + architecture: Architecture.ARM_64, + timeout: Duration.seconds(REQUEST_TIMEOUT_SECONDS), + memorySize: 256, + environment: { + ABCA_COMPONENT: 'approval', + TASK_TABLE_NAME: props.taskTable.tableName, + TASK_APPROVALS_TABLE_NAME: props.approvalsTable.tableName, + }, + }); + this.fn.addToRolePolicy(new iam.PolicyStatement({ + actions: ['dynamodb:GetItem', 'dynamodb:UpdateItem', 'dynamodb:ConditionCheckItem'], + resources: [props.taskTable.tableArn], + })); + this.fn.addToRolePolicy(new iam.PolicyStatement({ + actions: ['dynamodb:PutItem', 'dynamodb:UpdateItem'], + resources: [props.approvalsTable.tableArn], + })); + const accessLogs = new logs.LogGroup(this, 'AccessLogs', { retention: logs.RetentionDays.ONE_MONTH }); + this.api = new apigw.RestApi(this, 'Api', { + description: 'IAM-authenticated worker approval requests; no human decision endpoint.', + // The parent TaskApi configures the regional API Gateway logging role. + cloudWatchRole: false, + deployOptions: { + stageName: 'v1', + accessLogDestination: new apigw.LogGroupLogDestination(accessLogs), + accessLogFormat: apigw.AccessLogFormat.jsonWithStandardFields(), + loggingLevel: apigw.MethodLoggingLevel.INFO, + throttlingRateLimit: 60, + throttlingBurstLimit: 100, + }, + }); + this.api.root.addResource('tasks').addResource('{task_id}').addMethod( + 'POST', new apigw.LambdaIntegration(this.fn), + { authorizationType: apigw.AuthorizationType.IAM }, + ); + NagSuppressions.addResourceSuppressions(this.fn, [{ + id: 'AwsSolutions-IAM4', reason: 'AWSLambdaBasicExecutionRole provides Lambda runtime logging.', + }], true); + NagSuppressions.addResourceSuppressions(this.api, [ + { id: 'AwsSolutions-APIG2', reason: 'The handler validates the operation and an explicit approval-field allowlist before every write.' }, + { id: 'AwsSolutions-APIG3', reason: 'Machine-only IAM-signed API; session policy restricts POST to its tagged task path, with stage throttling.' }, + { id: 'AwsSolutions-COG4', reason: 'Workers authenticate with task-scoped AWS credentials, not human Cognito credentials.' }, + ], true); + } + + /** No wildcard task path on production session credentials. */ + public grantRequests(grantee: iam.IGrantable): void { + grantee.grantPrincipal.addToPrincipalPolicy(new iam.PolicyStatement({ + actions: ['execute-api:Invoke'], + resources: [this.api.arnForExecuteApi('POST', '/tasks/${aws:PrincipalTag/task_id}', 'v1')], + conditions: { Null: { 'aws:PrincipalTag/task_id': 'false' } }, + })); + } +} diff --git a/cdk/src/constructs/ecs-agent-cluster.ts b/cdk/src/constructs/ecs-agent-cluster.ts index b487a3c49..2c793f375 100644 --- a/cdk/src/constructs/ecs-agent-cluster.ts +++ b/cdk/src/constructs/ecs-agent-cluster.ts @@ -29,7 +29,7 @@ import * as secretsmanager from 'aws-cdk-lib/aws-secretsmanager'; import { NagSuppressions } from 'cdk-nag'; import { Construct, type Node } from 'constructs'; import { AgentMemory } from './agent-memory'; -import { AgentSessionRole, grantAgentTaskTableAccess } from './agent-session-role'; +import { AgentSessionRole, grantAgentTaskTableAccess, grantAgentApprovalReadAccess } from './agent-session-role'; import { PLATFORM_DEFAULT_AUX_MODEL_ID, PLATFORM_DEFAULT_MODEL_ID, @@ -99,6 +99,7 @@ export interface EcsAgentClusterProps { * retains the direct grants. */ readonly agentSessionRole?: AgentSessionRole; + readonly approvalRequestsApiUrl?: string; /** * AgentCore Memory for cross-task learning. When provided, the ECS task role @@ -239,6 +240,7 @@ export interface EcsTaskSizing { * override key should fail synth, not disable an isolation control. */ const RESERVED_BUILD_ENV_KEYS = new Set([ + 'APPROVAL_REQUESTS_API_URL', 'TASK_TABLE_NAME', 'TASK_EVENTS_TABLE_NAME', 'TASK_APPROVALS_TABLE_NAME', @@ -399,6 +401,7 @@ export class EcsAgentCluster extends Construct { ...(props.taskApprovalsTable && { TASK_APPROVALS_TABLE_NAME: props.taskApprovalsTable.tableName, }), + ...(props.approvalRequestsApiUrl && { APPROVAL_REQUESTS_API_URL: props.approvalRequestsApiUrl }), USER_CONCURRENCY_TABLE_NAME: props.userConcurrencyTable.tableName, LOG_GROUP_NAME: logGroup.logGroupName, GITHUB_TOKEN_SECRET_ARN: props.githubTokenSecret.secretArn, @@ -510,12 +513,7 @@ export class EcsAgentCluster extends Construct { } else { grantAgentTaskTableAccess(props.taskTable, taskRole, false); props.taskEventsTable.grantReadWriteData(taskRole); - props.taskApprovalsTable?.grant( - taskRole, - 'dynamodb:GetItem', - 'dynamodb:PutItem', - 'dynamodb:UpdateItem', - ); + if (props.taskApprovalsTable) grantAgentApprovalReadAccess(props.taskApprovalsTable, taskRole, false); } // Capacity counters are coordinator-owned. The agent never accesses them. diff --git a/cdk/src/constructs/task-orchestrator.ts b/cdk/src/constructs/task-orchestrator.ts index b3e0c0bd5..ad75bd349 100644 --- a/cdk/src/constructs/task-orchestrator.ts +++ b/cdk/src/constructs/task-orchestrator.ts @@ -242,6 +242,7 @@ export interface TaskOrchestratorProps { * fails closed with `approval_write_failed`. */ readonly taskApprovalsTableName: string; + readonly approvalRequestsApiUrl?: string; /** Nudges table (`NUDGES_TABLE_NAME`) the agent polls for mid-task nudges. */ readonly nudgesTableName: string; /** Application log group (`LOG_GROUP_NAME`) the agent writes progress logs to. */ @@ -526,6 +527,9 @@ export class TaskOrchestrator extends Construct { // backends. NO IAM grant accompanies any of these (see the prop docs). ...(props.agentPlatformConfig && { TASK_APPROVALS_TABLE_NAME: props.microvmConfig?.approvalsTable.tableName ?? props.agentPlatformConfig.taskApprovalsTableName, + ...(props.agentPlatformConfig.approvalRequestsApiUrl && { + APPROVAL_REQUESTS_API_URL: props.agentPlatformConfig.approvalRequestsApiUrl, + }), NUDGES_TABLE_NAME: props.agentPlatformConfig.nudgesTableName, LOG_GROUP_NAME: props.agentPlatformConfig.logGroupName, ARTIFACTS_BUCKET_NAME: props.agentPlatformConfig.artifactsBucketName, diff --git a/cdk/src/handlers/request-approval.ts b/cdk/src/handlers/request-approval.ts new file mode 100644 index 000000000..627296a06 --- /dev/null +++ b/cdk/src/handlers/request-approval.ts @@ -0,0 +1,200 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import { GetCommand, TransactWriteCommand } from '@aws-sdk/lib-dynamodb'; +import type { APIGatewayProxyEvent, APIGatewayProxyResult } from 'aws-lambda'; +import { logger } from './shared/logger'; +import { workerLeaseKey } from './shared/microvm-continuation-types'; +import { errorResponse, successResponse } from './shared/response'; +import { makeDocClient } from './shared/ua'; +import constants from '../../../contracts/constants.json'; + +const ddb = makeDocClient(); +const TASKS = process.env.TASK_TABLE_NAME!; +const APPROVALS = process.env.TASK_APPROVALS_TABLE_NAME!; +const MAX_ID_LENGTH = 128; +const MAX_TEXT_LENGTH = 8192; +const REQUEST_FIELDS = new Set([ + 'task_id', 'request_id', 'tool_name', 'tool_input_preview', 'tool_input_sha256', + 'reason', 'severity', 'matching_rule_ids', 'status', 'created_at', 'timeout_s', + 'deadline_epoch', 'user_id', 'repo', +]); + +interface RequestInput { + readonly operation: 'create' | 'timeout'; + readonly task_id: string; + readonly request_id: string; + readonly worker_attempt_id?: string; + readonly approval?: Record; + readonly reason?: string; +} + +function validId(value: unknown): value is string { + return typeof value === 'string' && value.length > 0 && value.length <= MAX_ID_LENGTH; +} + +/** Never copy decision, notification, retention or arbitrary caller fields. */ +function validateRequest(input: RequestInput, task: Record): Record { + const row = input.approval; + if (!row || Object.keys(row).some(key => !REQUEST_FIELDS.has(key)) + || row.task_id !== input.task_id || row.request_id !== input.request_id + || row.user_id !== task.user_id || row.repo !== (task.repo ?? '') + || row.status !== 'PENDING' + || !validId(row.tool_name) || typeof row.tool_input_preview !== 'string' || row.tool_input_preview.length > MAX_TEXT_LENGTH + || typeof row.tool_input_sha256 !== 'string' || !/^[a-f0-9]{64}$/.test(row.tool_input_sha256) + || typeof row.reason !== 'string' || row.reason.length > MAX_TEXT_LENGTH + || !['low', 'medium', 'high'].includes(row.severity as string) + || !Array.isArray(row.matching_rule_ids) || row.matching_rule_ids.length > 500 + || !row.matching_rule_ids.every(validId) + || typeof row.created_at !== 'string' || !Number.isFinite(Date.parse(row.created_at)) + || typeof row.timeout_s !== 'number' || !Number.isInteger(row.timeout_s) + || row.timeout_s < 0 || row.timeout_s > constants.approval_timeout_s.max + || (row.timeout_s === 0 ? row.deadline_epoch !== undefined + : row.deadline_epoch !== Math.floor(Date.parse(row.created_at) / 1000) + row.timeout_s)) { + throw new Error('APPROVAL_REQUEST_INVALID'); + } + return row; +} + +/** + * Worker-callable writer: creates pending requests or records a non-human timeout. + * Human decisions remain exclusively in approve/deny handlers. Workers have no + * direct approvals-table write permission, including whole-row replacement. + */ +export async function recordWorkerRequest(input: RequestInput): Promise> { + if (!input || !validId(input.task_id) || !validId(input.request_id) + || !['create', 'timeout'].includes(input.operation)) { + return { ok: false, code: 'APPROVAL_REQUEST_INVALID' }; + } + try { + const task = (await ddb.send(new GetCommand({ + TableName: TASKS, Key: { task_id: input.task_id }, ConsistentRead: true, + }))).Item; + if (!task || typeof task.user_id !== 'string') return { ok: false, code: 'APPROVAL_TASK_MISSING' }; + const lease = task.compute_type === 'lambda-microvm' ? [{ + ConditionCheck: { + TableName: TASKS, + Key: workerLeaseKey(input.task_id), + ConditionExpression: 'lease_state = :active AND lease_attempt_id = :attempt AND lease_user_id = :user', + ExpressionAttributeValues: { + ':active': 'ACTIVE', ':attempt': input.worker_attempt_id ?? '', ':user': task.user_id, + }, + }, + }] : []; + if (input.operation === 'create') { + const row = validateRequest(input, task); + await ddb.send(new TransactWriteCommand({ + TransactItems: [{ + Put: { + TableName: APPROVALS, + Item: row, + ConditionExpression: 'attribute_not_exists(request_id)', + }, + }, { + Update: { + TableName: TASKS, + Key: { task_id: input.task_id }, + UpdateExpression: 'SET #status = :awaiting, awaiting_approval_request_id = :request', + ConditionExpression: '#status = :running AND user_id = :user', + ExpressionAttributeNames: { '#status': 'status' }, + ExpressionAttributeValues: { + ':awaiting': 'AWAITING_APPROVAL', + ':running': 'RUNNING', + ':request': input.request_id, + ':user': task.user_id, + }, + }, + }, ...lease], + })); + } else { + // A worker may fail closed after a deadline or polling failure; it cannot + // turn either into human DENIED/APPROVED or replace an existing decision. + if (input.approval !== undefined || (input.reason !== undefined + && (typeof input.reason !== 'string' || input.reason.length > MAX_TEXT_LENGTH))) { + return { ok: false, code: 'APPROVAL_REQUEST_INVALID' }; + } + await ddb.send(new TransactWriteCommand({ + TransactItems: [{ + Update: { + TableName: APPROVALS, + Key: { task_id: input.task_id, request_id: input.request_id }, + UpdateExpression: 'SET #status = :timeout, decided_at = :now' + + (input.reason !== undefined ? ', deny_reason = :reason' : ''), + ConditionExpression: '#status = :pending AND user_id = :user', + ExpressionAttributeNames: { '#status': 'status' }, + ExpressionAttributeValues: { + ':timeout': 'TIMED_OUT', + ':pending': 'PENDING', + ':user': task.user_id, + ':now': new Date().toISOString(), + ...(input.reason !== undefined ? { ':reason': input.reason } : {}), + }, + }, + }, { + ConditionCheck: { + TableName: TASKS, + Key: { task_id: input.task_id }, + ConditionExpression: '#status = :awaiting AND awaiting_approval_request_id = :request AND user_id = :user', + ExpressionAttributeNames: { '#status': 'status' }, + ExpressionAttributeValues: { ':awaiting': 'AWAITING_APPROVAL', ':request': input.request_id, ':user': task.user_id }, + }, + }, ...lease], + })); + } + return { ok: true }; + } catch (error) { + const failure = error as { name?: string; message?: string; CancellationReasons?: { Code?: string }[] }; + if (failure.message === 'APPROVAL_REQUEST_INVALID') return { ok: false, code: failure.message }; + const reasons = failure.CancellationReasons?.map(reason => ({ Code: reason.Code })); + logger.warn('Worker approval request was not recorded', { + task_id: input.task_id, + request_id: input.request_id, + operation: input.operation, + error_type: failure.name, + cancellation_reasons: reasons, + }); + return { ok: false, code: failure.name ?? 'APPROVAL_WRITE_FAILED', cancellation_reasons: reasons }; + } +} + +/** IAM authorizes the signed task path before invoking this Lambda. */ +export async function handler(event: APIGatewayProxyEvent): Promise { + const requestId = event.requestContext?.requestId ?? 'unknown'; + if (!event.requestContext?.identity?.userArn || !validId(event.pathParameters?.task_id)) { + return errorResponse(403, 'APPROVAL_CALLER_UNAUTHENTICATED', 'IAM authentication required', requestId); + } + let input: RequestInput; + try { + input = JSON.parse(event.isBase64Encoded + ? Buffer.from(event.body ?? '', 'base64').toString('utf8') : event.body ?? ''); + } catch { + return errorResponse(400, 'APPROVAL_REQUEST_INVALID', 'Invalid JSON request', requestId); + } + if (!input || input.task_id !== event.pathParameters!.task_id) { + return errorResponse(400, 'APPROVAL_TASK_MISMATCH', 'Task must match the signed path', requestId); + } + const result = await recordWorkerRequest(input); + if (result.ok) return successResponse(200, result, requestId); + const code = String(result.code); + const statusCode = code === 'TransactionCanceledException' ? 409 + : code === 'APPROVAL_TASK_MISSING' ? 404 : code === 'APPROVAL_REQUEST_INVALID' ? 400 : 503; + return errorResponse(statusCode, code, 'Approval write was not acknowledged', requestId, { + cancellation_reasons: result.cancellation_reasons ?? [], + }); +} diff --git a/cdk/src/handlers/shared/response.ts b/cdk/src/handlers/shared/response.ts index 76bd0234e..9a9235c40 100644 --- a/cdk/src/handlers/shared/response.ts +++ b/cdk/src/handlers/shared/response.ts @@ -132,6 +132,7 @@ export function errorResponse( code: string, message: string, requestId: string, + details?: Record, ): APIGatewayProxyResult { return { statusCode, @@ -141,6 +142,7 @@ export function errorResponse( code, message, request_id: requestId, + ...(details ? { details } : {}), }, }), }; diff --git a/cdk/src/stacks/agent.ts b/cdk/src/stacks/agent.ts index 51d73d8fe..92218b26c 100644 --- a/cdk/src/stacks/agent.ts +++ b/cdk/src/stacks/agent.ts @@ -35,6 +35,7 @@ import { AgentSessionRole } from '../constructs/agent-session-role'; import { AgentVpc } from '../constructs/agent-vpc'; import { ApiKeyTable } from '../constructs/api-key-table'; import { ApprovalMetricsPublisherConsumer } from '../constructs/approval-metrics-publisher-consumer'; +import { ApprovalRequestService } from '../constructs/approval-request-service'; import { AttachmentsBucket } from '../constructs/attachments-bucket'; import { PLATFORM_DEFAULT_AUX_MODEL_ID, @@ -556,6 +557,13 @@ export class AgentStack extends Stack { ...(microvmImageConfigured && { lambdaMicrovmImageArn: lazyMicrovmImageArn }), }); + const approvalRequests = new ApprovalRequestService(this, 'ApprovalRequests', { + taskTable: taskTable.table, approvalsTable: taskApprovalsTable.table, + }); + // Reuse the parent's regional API Gateway logging configuration. + const apiLoggingAccount = taskApi.api.node.tryFindChild('Account'); + if (apiLoggingAccount) approvalRequests.node.addDependency(apiLoggingAccount); + // Agent asset registry API (#246) in its own NestedStack + RestApi so its // ~35 resources don't count against this root stack's 500-resource limit. // It authorizes against the SHARED Cognito user pool, so a caller's JWT works @@ -629,6 +637,7 @@ export class AgentStack extends Stack { // AWAITING_APPROVAL; absent → hook fails closed with // ``approval_write_failed`` (the `ApprovalTablesUnavailable` path). TASK_APPROVALS_TABLE_NAME: taskApprovalsTable.table.tableName, + APPROVAL_REQUESTS_API_URL: approvalRequests.api.url, // Hint for the hook's remaining-maxLifetime calculation (§6.5 // pseudocode line 793). Kept in sync with the AgentCore // lifecycle configuration below so drift is visible. 8 hours. @@ -845,9 +854,9 @@ export class AgentStack extends Stack { const agentSessionRole = new AgentSessionRole(this, 'AgentSessionRole', { assumingRoles: [runtime.role], taskTable: taskTable.table, + approvalsTable: taskApprovalsTable.table, taskScopedTables: [ taskEventsTable.table, - taskApprovalsTable.table, taskNudgesTable.table, ], traceArtifactsBucket: traceArtifactsBucket.bucket, @@ -858,6 +867,7 @@ export class AgentStack extends Stack { invokableModels: invokableBedrockModels, }); sessionRoleArnHolder = agentSessionRole.role.roleArn; + approvalRequests.grantRequests(agentSessionRole.role); // X-Ray tracing disabled — requires account-level UpdateTraceSegmentDestination // which needs CloudWatch Logs resource policy propagation. Re-enable via @@ -1126,6 +1136,7 @@ export class AgentStack extends Stack { taskTable: taskTable.table, taskEventsTable: taskEventsTable.table, taskApprovalsTable: taskApprovalsTable.table, + approvalRequestsApiUrl: approvalRequests.api.url, userConcurrencyTable: userConcurrencyTable.table, githubTokenSecret, memoryId: agentMemory.memory.memoryId, @@ -1323,6 +1334,7 @@ export class AgentStack extends Stack { // of the resources they identify. agentPlatformConfig: { taskApprovalsTableName: taskApprovalsTable.table.tableName, + approvalRequestsApiUrl: approvalRequests.api.url, nudgesTableName: taskNudgesTable.table.tableName, logGroupName: applicationLogGroup.logGroupName, // INTENTIONAL, not a wiring bug: both keys resolve to the SAME bucket diff --git a/cdk/test/constructs/agent-session-role.test.ts b/cdk/test/constructs/agent-session-role.test.ts index 323a38719..a84e86aa9 100644 --- a/cdk/test/constructs/agent-session-role.test.ts +++ b/cdk/test/constructs/agent-session-role.test.ts @@ -53,9 +53,9 @@ function createStack() { const sessionRole = new AgentSessionRole(stack, 'AgentSessionRole', { assumingRoles: [computeRole], taskTable, + approvalsTable: taskApprovalsTable, taskScopedTables: [ taskEventsTable, - taskApprovalsTable, taskNudgesTable, ], traceArtifactsBucket, @@ -108,11 +108,12 @@ describe('AgentSessionRole construct', () => { const actions = Array.isArray(s.Action) ? s.Action : [s.Action]; return actions.some((a: string) => a.startsWith('dynamodb:')); }); - // Main task read/update grants plus three supporting tables. + // Main task read/update, approval read, and two supporting table grants. expect(ddbStatements).toHaveLength(5); for (const s of ddbStatements) { const actions = Array.isArray(s.Action) ? s.Action : [s.Action]; - const readonlyTask = actions.includes('dynamodb:GetItem') && !actions.includes('dynamodb:PutItem'); + const readonlyTask = JSON.stringify(s.Resource).includes('TaskTable') + && !JSON.stringify(s.Resource).includes('Approvals') && actions.includes('dynamodb:GetItem'); expect(s.Condition['ForAllValues:StringEquals']['dynamodb:LeadingKeys']) .toEqual(readonlyTask ? ['${aws:PrincipalTag/task_id}', 'worker-lease#${aws:PrincipalTag/task_id}'] @@ -143,6 +144,18 @@ describe('AgentSessionRole construct', () => { expect(tracePut).toBeDefined(); }); + test('workers cannot forge decisions or notification flags through any approval-table write', () => { + const statements = Object.values(template.findResources('AWS::IAM::Policy')) + .flatMap(policy => policy.Properties.PolicyDocument.Statement) + .filter((statement: { Resource: unknown }) => JSON.stringify(statement.Resource).includes('TaskApprovalsTable')); + expect(statements).toHaveLength(1); + expect(statements[0].Action).toEqual([ + 'dynamodb:GetItem', 'dynamodb:BatchGetItem', 'dynamodb:Query', 'dynamodb:ConditionCheckItem', + ]); + expect(statements[0].Condition['ForAllValues:StringEquals']['dynamodb:LeadingKeys']) + .toEqual(['${aws:PrincipalTag/task_id}']); + }); + test('task records cannot be replaced, deleted or updated without an attribute allowlist', () => { const policy = Object.entries(template.findResources('AWS::IAM::Policy')) .find(([id]) => id.includes('AgentSessionRole'))![1]; @@ -181,10 +194,13 @@ describe('AgentSessionRole construct', () => { expect(() => new AgentSessionRole(stack, 'Session', { assumingRoles: [computeRole], taskTable: table, + approvalsTable: new dynamodb.Table(stack, 'ApprovalReadTable', { + partitionKey: { name: 'task_id', type: dynamodb.AttributeType.STRING }, + }), taskScopedTables: [table], traceArtifactsBucket: new s3.Bucket(stack, 'Traces'), attachmentsBucket: new s3.Bucket(stack, 'Attachments'), - })).toThrow('taskTable must not appear in taskScopedTables'); + })).toThrow('taskTable and approvalsTable must not appear in taskScopedTables'); }); test('S3 artifact writes are scoped to the per-task_id prefix (#248 Phase 3)', () => { @@ -256,6 +272,9 @@ describe('AgentSessionRole construct', () => { new AgentSessionRole(stack, 'SR', { assumingRoles: [computeRole], taskTable: table, + approvalsTable: new dynamodb.Table(stack, 'ApprovalReadTable', { + partitionKey: { name: 'task_id', type: dynamodb.AttributeType.STRING }, + }), taskScopedTables: [], traceArtifactsBucket: new s3.Bucket(stack, 'TB'), attachmentsBucket: new s3.Bucket(stack, 'AB'), @@ -302,6 +321,9 @@ describe('AgentSessionRole construct', () => { const sessionRole = new AgentSessionRole(stack, 'SR', { assumingRoles: [agentcoreRole], taskTable: table, + approvalsTable: new dynamodb.Table(stack, 'ApprovalReadTable', { + partitionKey: { name: 'task_id', type: dynamodb.AttributeType.STRING }, + }), taskScopedTables: [], traceArtifactsBucket: new s3.Bucket(stack, 'TB'), attachmentsBucket: new s3.Bucket(stack, 'AB'), diff --git a/cdk/test/constructs/approval-request-service.test.ts b/cdk/test/constructs/approval-request-service.test.ts new file mode 100644 index 000000000..9a89bf32e --- /dev/null +++ b/cdk/test/constructs/approval-request-service.test.ts @@ -0,0 +1,66 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import { App, Stack } from 'aws-cdk-lib'; +import { Template } from 'aws-cdk-lib/assertions'; +import * as dynamodb from 'aws-cdk-lib/aws-dynamodb'; +import * as iam from 'aws-cdk-lib/aws-iam'; +import { ApprovalRequestService } from '../../src/constructs/approval-request-service'; + +let root: Template; +let child: Template; +beforeAll(() => { + const stack = new Stack(new App(), 'ApprovalFixture', { + env: { account: '123456789012', region: 'us-east-1' }, + }); + const table = (id: string) => new dynamodb.Table(stack, id, { + partitionKey: { name: 'task_id', type: dynamodb.AttributeType.STRING }, + }); + const service = new ApprovalRequestService(stack, 'ApprovalRequests', { + taskTable: table('Tasks'), approvalsTable: table('Approvals'), + }); + const worker = new iam.Role(stack, 'WorkerSession', { + assumedBy: new iam.ServicePrincipal('ecs-tasks.amazonaws.com'), + }); + service.grantRequests(worker); + root = Template.fromStack(stack); + child = Template.fromStack(service); +}); + +test('requires IAM authentication on the only writable endpoint', () => { + child.resourceCountIs('AWS::ApiGateway::Method', 1); + child.hasResourceProperties('AWS::ApiGateway::Method', { + HttpMethod: 'POST', AuthorizationType: 'AWS_IAM', + }); +}); + +test('binds invocation to the session task tag and grants no direct Lambda invocation', () => { + const statements = Object.values(root.findResources('AWS::IAM::Policy')) + .flatMap(policy => policy.Properties.PolicyDocument.Statement); + expect(statements).toHaveLength(1); + expect(statements[0].Action).toBe('execute-api:Invoke'); + expect(JSON.stringify(statements[0].Resource)).toContain('/v1/POST/tasks/${aws:PrincipalTag/task_id}'); + expect(statements[0].Condition).toEqual({ Null: { 'aws:PrincipalTag/task_id': 'false' } }); +}); + +test('keeps the trusted writer and API infrastructure out of the parent resource budget', () => { + root.resourceCountIs('AWS::Lambda::Function', 0); + child.resourceCountIs('AWS::Lambda::Function', 1); + child.resourceCountIs('AWS::DynamoDB::Table', 0); +}); diff --git a/cdk/test/constructs/ecs-agent-cluster.test.ts b/cdk/test/constructs/ecs-agent-cluster.test.ts index 3c01b5d85..3be3da253 100644 --- a/cdk/test/constructs/ecs-agent-cluster.test.ts +++ b/cdk/test/constructs/ecs-agent-cluster.test.ts @@ -756,7 +756,8 @@ describe('EcsAgentCluster construct', () => { }), ], taskTable, - taskScopedTables: [taskEventsTable, taskApprovalsTable], + approvalsTable: taskApprovalsTable, + taskScopedTables: [taskEventsTable], traceArtifactsBucket: new s3.Bucket(stack, 'TraceBucket'), attachmentsBucket: new s3.Bucket(stack, 'AttachmentsBucket'), }); @@ -848,7 +849,7 @@ describe('EcsAgentCluster approval wiring without a SessionRole', () => { let template: Template; beforeAll(() => { template = createStack({ withApprovals: true }).template; }); - test('grants only the approval item operations used by the agent', () => { + test('grants approval reads even without a SessionRole, never direct writes', () => { const approvalId = Object.keys(template.findResources('AWS::DynamoDB::Table')) .find(id => id.startsWith('TaskApprovalsTable')); const statements = Object.values(template.findResources('AWS::IAM::Policy')) @@ -858,7 +859,7 @@ describe('EcsAgentCluster approval wiring without a SessionRole', () => { ); expect(grants).toEqual([expect.objectContaining({ Effect: 'Allow', - Action: ['dynamodb:GetItem', 'dynamodb:PutItem', 'dynamodb:UpdateItem'], + Action: ['dynamodb:GetItem', 'dynamodb:BatchGetItem', 'dynamodb:Query', 'dynamodb:ConditionCheckItem'], })]); }); @@ -868,6 +869,13 @@ describe('EcsAgentCluster approval wiring without a SessionRole', () => { }), 'S').node; expect(() => resolveEcsTaskSizing(node)).toThrow('TASK_APPROVALS_TABLE_NAME'); }); + + test('rejects a build setting that would replace the approval service', () => { + const node = new Stack(new App({ + context: { ecsExtraBuildEnv: { APPROVAL_REQUESTS_API_URL: 'https://other.example' } }, + }), 'S').node; + expect(() => resolveEcsTaskSizing(node)).toThrow('APPROVAL_REQUESTS_API_URL'); + }); }); describe('EcsAgentCluster payload bucket (#502)', () => { diff --git a/cdk/test/constructs/lambda-microvm-compute.test.ts b/cdk/test/constructs/lambda-microvm-compute.test.ts index a97a49223..b39990f4a 100644 --- a/cdk/test/constructs/lambda-microvm-compute.test.ts +++ b/cdk/test/constructs/lambda-microvm-compute.test.ts @@ -120,6 +120,9 @@ function instantiate(options: BuildOptions = {}): Omit { taskTable: new dynamodb.Table(stack, 'TaskTable', { partitionKey: { name: 'task_id', type: dynamodb.AttributeType.STRING }, }), + approvalsTable: new dynamodb.Table(stack, 'ApprovalReadTable', { + partitionKey: { name: 'task_id', type: dynamodb.AttributeType.STRING }, + }), taskScopedTables: [], traceArtifactsBucket: new s3.Bucket(stack, 'TraceBucket'), attachmentsBucket: new s3.Bucket(stack, 'AttachmentsBucket'), diff --git a/cdk/test/constructs/lambda-microvm-stack.test.ts b/cdk/test/constructs/lambda-microvm-stack.test.ts index 5db13e15c..6d3d581cf 100644 --- a/cdk/test/constructs/lambda-microvm-stack.test.ts +++ b/cdk/test/constructs/lambda-microvm-stack.test.ts @@ -61,6 +61,9 @@ describe.each(IMAGE_INPUTS)('LambdaMicrovmStack %s', (mode, imageInputs) => { taskTable: new dynamodb.Table(stack, 'Tasks', { partitionKey: { name: 'task_id', type: dynamodb.AttributeType.STRING }, }), + approvalsTable: new dynamodb.Table(stack, 'ApprovalReadTable', { + partitionKey: { name: 'task_id', type: dynamodb.AttributeType.STRING }, + }), taskScopedTables: [], traceArtifactsBucket: new s3.Bucket(stack, 'TraceBucket'), attachmentsBucket: new s3.Bucket(stack, 'AttachmentsBucket'), diff --git a/cdk/test/handlers/request-approval-local.test.ts b/cdk/test/handlers/request-approval-local.test.ts new file mode 100644 index 000000000..ee15a5aa0 --- /dev/null +++ b/cdk/test/handlers/request-approval-local.test.ts @@ -0,0 +1,179 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +/** Real transaction conditions; loopback endpoint and dummy credentials only. */ +import { randomUUID } from 'node:crypto'; +import { CreateTableCommand, DeleteTableCommand, DynamoDBClient } from '@aws-sdk/client-dynamodb'; +import { DynamoDBDocumentClient, GetCommand, PutCommand, TransactWriteCommand, UpdateCommand } from '@aws-sdk/lib-dynamodb'; + +const endpoint = process.env.ABCA_DDB_LOCAL_ENDPOINT; +if (process.env.CI === 'true' && !endpoint) throw new Error('CI requires ABCA_DDB_LOCAL_ENDPOINT'); +if (endpoint && (new URL(endpoint).hostname !== '127.0.0.1' || new URL(endpoint).protocol !== 'http:')) { + throw new Error('Approval integration tests require an http://127.0.0.1 endpoint'); +} +const mockBeforeTransaction = jest.fn(); +let mockClient: DynamoDBDocumentClient; +jest.mock('../../src/handlers/shared/ua', () => ({ + makeDocClient: () => ({ + send: async (command: unknown) => { + if (command instanceof TransactWriteCommand) await mockBeforeTransaction(); + return mockClient.send(command as GetCommand); + }, + }), +})); +const suffix = randomUUID(); +const tasks = `approval-tasks-${suffix}`; +const approvals = `approval-requests-${suffix}`; +Object.assign(process.env, { TASK_TABLE_NAME: tasks, TASK_APPROVALS_TABLE_NAME: approvals }); +import { recordWorkerRequest } from '../../src/handlers/request-approval'; + +const raw = new DynamoDBClient({ + endpoint: endpoint ?? 'http://127.0.0.1:1', + region: 'us-east-1', + credentials: { accessKeyId: 'local', secretAccessKey: 'local' }, +}); +mockClient = DynamoDBDocumentClient.from(raw); +const local = endpoint ? describe : describe.skip; +jest.setTimeout(30_000); + +local('Trusted approval writer against DynamoDB Local', () => { + let taskId: string; + const gate = 'gate'; + const row = () => ({ + task_id: taskId, + request_id: gate, + user_id: 'owner', + repo: 'owner/repo', + tool_name: 'Bash', + tool_input_preview: '{"command":"git push"}', + tool_input_sha256: 'a'.repeat(64), + reason: 'Protected operation', + severity: 'high', + matching_rule_ids: ['protected'], + status: 'PENDING', + created_at: new Date().toISOString(), + timeout_s: 0, + }); + const create = () => recordWorkerRequest({ + operation: 'create', task_id: taskId, request_id: gate, worker_attempt_id: 'worker', approval: row(), + }); + const timeout = () => recordWorkerRequest({ + operation: 'timeout', task_id: taskId, request_id: gate, worker_attempt_id: 'worker', + }); + const read = async (table: string) => (await mockClient.send(new GetCommand({ + TableName: table, + Key: { task_id: taskId, ...(table === approvals ? { request_id: gate } : {}) }, + ConsistentRead: true, + }))).Item; + const change = async (table: string, field: string, value: string, lease = false) => + mockClient.send(new UpdateCommand({ + TableName: table, + Key: { task_id: lease ? `worker-lease#${taskId}` : taskId, ...(table === approvals ? { request_id: gate } : {}) }, + UpdateExpression: 'SET #field = :value', + ExpressionAttributeNames: { '#field': field }, + ExpressionAttributeValues: { ':value': value }, + })); + beforeAll(async () => { + for (const table of [tasks, approvals]) { + await raw.send(new CreateTableCommand({ + TableName: table, + BillingMode: 'PAY_PER_REQUEST', + AttributeDefinitions: [{ AttributeName: 'task_id', AttributeType: 'S' }, + ...(table === approvals ? [{ AttributeName: 'request_id', AttributeType: 'S' as const }] : [])], + KeySchema: [{ AttributeName: 'task_id', KeyType: 'HASH' }, + ...(table === approvals ? [{ AttributeName: 'request_id', KeyType: 'RANGE' as const }] : [])], + })); + } + }); + afterAll(async () => { + try { for (const table of [tasks, approvals]) await raw.send(new DeleteTableCommand({ TableName: table })); } finally { raw.destroy(); } + }); + beforeEach(async () => { + taskId = randomUUID(); + mockBeforeTransaction.mockReset(); + await mockClient.send(new PutCommand({ + TableName: tasks, + Item: { + task_id: taskId, status: 'RUNNING', user_id: 'owner', repo: 'owner/repo', compute_type: 'lambda-microvm', + }, + })); + await mockClient.send(new PutCommand({ + TableName: tasks, + Item: { + task_id: `worker-lease#${taskId}`, lease_state: 'ACTIVE', lease_attempt_id: 'worker', lease_user_id: 'owner', + }, + })); + }); + + test('creates a durable pending request and its task pointer atomically', async () => { + expect(await create()).toEqual({ ok: true }); + expect(await read(approvals)).toMatchObject({ status: 'PENDING', timeout_s: 0 }); + expect(await read(approvals)).not.toHaveProperty('ttl'); + expect(await read(tasks)).toMatchObject({ status: 'AWAITING_APPROVAL', awaiting_approval_request_id: gate }); + }); + test.each(['APPROVED', 'DENIED', 'CANCELLED'])('timeout preserves an existing %s decision', async status => { + await create(); + await change(approvals, 'status', status); + const before = await read(approvals); + expect(await timeout()).toMatchObject({ + ok: false, + code: 'TransactionCanceledException', + cancellation_reasons: [{ Code: 'ConditionalCheckFailed' }, { Code: 'None' }, { Code: 'None' }], + }); + expect(await read(approvals)).toEqual(before); + }); + test('only pending requests can transition to a non-human timeout', async () => { + await create(); + expect(await timeout()).toEqual({ ok: true }); + expect(await read(approvals)).toMatchObject({ status: 'TIMED_OUT' }); + expect(await read(approvals)).not.toHaveProperty('decision_source'); + }); + test('cancellation between read and write prevents both request and task changes', async () => { + mockBeforeTransaction.mockImplementationOnce(() => change(tasks, 'status', 'CANCELLED')); + expect(await create()).toMatchObject({ ok: false, code: 'TransactionCanceledException' }); + expect(await read(approvals)).toBeUndefined(); + expect(await read(tasks)).toMatchObject({ status: 'CANCELLED' }); + }); + test.each(['create', 'timeout'])('retirement revokes a stale worker during %s', async operation => { + if (operation === 'timeout') await create(); + const before = await read(approvals); + mockBeforeTransaction.mockImplementationOnce(() => change(tasks, 'lease_state', 'PARKED', true)); + const result = await (operation === 'create' ? create() : timeout()); + expect(result).toMatchObject({ ok: false, code: 'TransactionCanceledException' }); + expect(result.cancellation_reasons).toEqual([ + { Code: 'None' }, { Code: 'None' }, { Code: 'ConditionalCheckFailed' }, + ]); + expect(await read(approvals)).toEqual(before); + }); + test('a second request cannot replace the first while the task waits', async () => { + await create(); + const result = await recordWorkerRequest({ + operation: 'create', + task_id: taskId, + request_id: 'replacement', + worker_attempt_id: 'worker', + approval: { ...row(), request_id: 'replacement' }, + }); + expect(result).toMatchObject({ ok: false, code: 'TransactionCanceledException' }); + expect(await read(tasks)).toMatchObject({ awaiting_approval_request_id: gate }); + expect((await mockClient.send(new GetCommand({ + TableName: approvals, Key: { task_id: taskId, request_id: 'replacement' }, + }))).Item).toBeUndefined(); + }); +}); diff --git a/cdk/test/handlers/request-approval.test.ts b/cdk/test/handlers/request-approval.test.ts new file mode 100644 index 000000000..8981a1a46 --- /dev/null +++ b/cdk/test/handlers/request-approval.test.ts @@ -0,0 +1,131 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import type { APIGatewayProxyEvent } from 'aws-lambda'; + +const send = jest.fn(); +jest.mock('../../src/handlers/shared/ua', () => ({ makeDocClient: () => ({ send }) })); +import { handler, recordWorkerRequest } from '../../src/handlers/request-approval'; + +const approval = { + task_id: 'task', + request_id: 'gate', + user_id: 'owner', + repo: 'owner/repo', + tool_name: 'Bash', + tool_input_preview: '{"command":"git push"}', + tool_input_sha256: 'a'.repeat(64), + reason: 'Protected operation', + severity: 'high', + matching_rule_ids: ['protected'], + status: 'PENDING', + created_at: '2026-09-18T00:00:00Z', + timeout_s: 0, +}; +const input = { operation: 'create' as const, task_id: 'task', request_id: 'gate', approval }; +let task: Record; +beforeEach(() => { + task = { user_id: 'owner', repo: 'owner/repo', compute_type: 'ecs', status: 'RUNNING' }; + send.mockReset().mockImplementation(async command => + command.constructor.name === 'GetCommand' ? { Item: task } : {}); +}); + +test.each([ + { status: 'APPROVED' }, { status: 'DENIED' }, + { notified_linear_approval_requested: true }, { decision_source: 'forged' }, + { decided_at: 'now' }, { scope: 'all' }, { ttl: 1 }, { user_id: 'other' }, { repo: 'other/repo' }, +])('rejects forged decision/notification fields and ownership: %j', async changes => { + expect(await recordWorkerRequest({ ...input, approval: { ...approval, ...changes } })) + .toMatchObject({ ok: false, code: 'APPROVAL_REQUEST_INVALID' }); + expect(send).toHaveBeenCalledTimes(1); +}); + +test('records only the pending request, guarded by current task ownership and status', async () => { + expect(await recordWorkerRequest(input)).toEqual({ ok: true }); + const items = send.mock.calls[1][0].input.TransactItems; + expect(items[0].Put.Item).toEqual(approval); + expect(items[0].Put.ConditionExpression).toBe('attribute_not_exists(request_id)'); + expect(items[1].Update.ConditionExpression).toBe('#status = :running AND user_id = :user'); + expect(items[1].Update.ExpressionAttributeValues[':user']).toBe('owner'); +}); + +test('fences stale MicroVM writers in the same transaction', async () => { + task.compute_type = 'lambda-microvm'; + await recordWorkerRequest({ ...input, worker_attempt_id: 'worker-token' }); + const lease = send.mock.calls[1][0].input.TransactItems[2].ConditionCheck; + expect(lease.Key).toEqual({ task_id: 'worker-lease#task' }); + expect(lease.ExpressionAttributeValues).toEqual({ + ':active': 'ACTIVE', ':attempt': 'worker-token', ':user': 'owner', + }); +}); + +test('timeout cannot overwrite a human decision or resume the task', async () => { + await recordWorkerRequest({ operation: 'timeout', task_id: 'task', request_id: 'gate' }); + const items = send.mock.calls[1][0].input.TransactItems; + expect(items[0].Update.ExpressionAttributeValues[':timeout']).toBe('TIMED_OUT'); + expect(items[0].Update.ConditionExpression).toBe('#status = :pending AND user_id = :user'); + expect(items[1].ConditionCheck.ConditionExpression).toContain('awaiting_approval_request_id = :request'); + expect(items[1].Update).toBeUndefined(); +}); + +test('preserves cancellation reasons without returning request contents', async () => { + send.mockRejectedValueOnce({ + name: 'TransactionCanceledException', + CancellationReasons: [ + { Code: 'ConditionalCheckFailed', Item: { secret: 'must not return' } }, + ], + }); + expect(await recordWorkerRequest(input)).toEqual({ + ok: false, code: 'TransactionCanceledException', cancellation_reasons: [{ Code: 'ConditionalCheckFailed' }], + }); +}); + +test('requires IAM caller identity and refuses body/path task substitution', async () => { + const event = { + body: JSON.stringify(input), + pathParameters: { task_id: 'other-task' }, + requestContext: { identity: { userArn: 'arn:aws:sts::123456789012:assumed-role/Session/task' } }, + } as unknown as APIGatewayProxyEvent; + expect((await handler(event)).statusCode).toBe(400); + expect(send).not.toHaveBeenCalled(); + expect((await handler({ ...event, requestContext: {} } as APIGatewayProxyEvent)).statusCode).toBe(403); +}); + +test('accepts the signed path for the matching request', async () => { + const result = await handler({ + body: JSON.stringify(input), + pathParameters: { task_id: 'task' }, + requestContext: { identity: { userArn: 'arn:aws:sts::123456789012:assumed-role/Session/task' } }, + } as unknown as APIGatewayProxyEvent); + expect(result.statusCode).toBe(200); + expect(JSON.parse(result.body)).toEqual({ data: { ok: true } }); +}); + +test('reports service failures as unavailable with a request ID, not invalid input', async () => { + send.mockRejectedValueOnce({ name: 'ProvisionedThroughputExceededException' }); + const result = await handler({ + body: JSON.stringify(input), + pathParameters: { task_id: 'task' }, + requestContext: { requestId: 'api-request', identity: { userArn: 'worker' } }, + } as unknown as APIGatewayProxyEvent); + expect(result.statusCode).toBe(503); + expect(JSON.parse(result.body)).toMatchObject({ + error: { code: 'ProvisionedThroughputExceededException', request_id: 'api-request' }, + }); +}); diff --git a/cdk/test/stacks/agent.test.ts b/cdk/test/stacks/agent.test.ts index 7ea173057..a50aaabe3 100644 --- a/cdk/test/stacks/agent.test.ts +++ b/cdk/test/stacks/agent.test.ts @@ -1764,7 +1764,7 @@ describe('AgentStack solution attribution (#319): AWS_SDK_UA_APP_ID via stack-le // is added without being attributed below. // 47 existing handlers (including concurrency repair) + 2 registry // provisioning handlers + 4 registry API handlers in nested stacks. - expect(abcaLambdas.length).toBe(53); + expect(abcaLambdas.length).toBe(54); // Every ABCA-authored Lambda must carry the canonical `#` app-id. Collect // any offenders so a failure names the exact logical id(s) that are naked. const unattributed = abcaLambdas diff --git a/contracts/constants.json b/contracts/constants.json index 235221b7a..f9c4d254f 100644 --- a/contracts/constants.json +++ b/contracts/constants.json @@ -41,6 +41,7 @@ "task_table_name": "TASK_TABLE_NAME", "task_events_table_name": "TASK_EVENTS_TABLE_NAME", "task_approvals_table_name": "TASK_APPROVALS_TABLE_NAME", + "approval_requests_api_url": "APPROVAL_REQUESTS_API_URL", "nudges_table_name": "NUDGES_TABLE_NAME", "log_group_name": "LOG_GROUP_NAME", "artifacts_bucket_name": "ARTIFACTS_BUCKET_NAME", diff --git a/docs/design/CEDAR_HITL_GATES.md b/docs/design/CEDAR_HITL_GATES.md index 874cb737a..4f8abee22 100644 --- a/docs/design/CEDAR_HITL_GATES.md +++ b/docs/design/CEDAR_HITL_GATES.md @@ -6,7 +6,7 @@ > **Rev:** 5 (2026-05-06 — fold in parallel adversarial + advocate review of the timeout design: late-approval re-read on TIMED_OUT ConditionCheckFailed; user-visible timeout-cap milestones; ceiling-shrink milestone; Runtime JWT bound verified as auto-refreshed IAM; three new tuning metrics; explicit off-hours trade-off section; notification-delivery-failure boundary. IMPL-24 through IMPL-28 added.). > **Implementation:** Core shipped. The 3-outcome engine (`agent/src/policy.py`), default policy sets (`agent/policies/hard_deny.cedar`, `agent/policies/soft_deny.cedar`), approval Lambdas (`cdk/src/handlers/{approve-task,deny-task,get-pending,get-policies}.ts`) wired into `cdk/src/constructs/task-api.ts` (routes `/tasks/{id}/approve`, `/deny`, `/pending`, `/repos/{repo_id}/policies`), the cross-engine parity fixtures (`contracts/cedar-parity/`), and the exact engine pins are all on `main`. §15's task list is preserved as a historical implementation record; see the note at the top of §15 for what (if anything) remains unbuilt. > -> **Current source behavior (2026-09-17):** the task default is `approval_timeout_s=0`, +> **Current source behavior (2026-09-18):** the task default is `approval_timeout_s=0`, > meaning no decision deadline. Explicit task settings are 30–3,600 seconds; > positive policy-rule deadlines still apply. Pending rows have no DynamoDB TTL, > and `expires_at` is nullable. Task closure cancels unanswered requests and adds @@ -14,8 +14,9 @@ > conditions return `404 REQUEST_NOT_FOUND`; task-only conflicts return 409. > MicroVM checkpoint/retirement/replacement separates human waiting from worker > lifetime and capacity; other backends retain their existing runtime limits. -> CLI response instructions are implemented for Slack/Linear notifications; -> native channel decisions remain separate work. See the current +> Linear accepts an owner’s `approve` or `deny` reply to the approval comment; +> the handler verifies the actual comment through Linear’s API. Slack uses CLI +> response instructions. See the current > [user guide](../guides/USER_GUIDE.md#approval-gates-cedar-hitl) and > [continuation protocol](./ORCHESTRATOR.md#retained-microvm-approvals). > The normal deployment passed retained-request, ten-minute sleep, explicit-expiry @@ -232,7 +233,9 @@ Narrative walk-through of the happy path. Sequence diagrams in the round-trip Me "repo": "my-org/my-app" } ``` -15. **Atomic transition** — hook issues `TransactWriteItems` with two operations: +15. **Atomic transition** — the hook sends an IAM-signed request to the approval + service, which issues `TransactWriteItems` with these operations (and a + worker-lease condition for MicroVM): - Put on `TaskApprovalsTable` (new row with status=PENDING) - ConditionalUpdate on `TaskTable`: `status = :awaiting, awaiting_approval_request_id = :rid WHERE status = :running` Both succeed or both fail. On `TransactionCanceledException` (most likely the TaskTable condition fails because another process moved the status), the hook emits `approval_write_failed` and returns DENY. @@ -1311,18 +1314,18 @@ TaskApprovalsTable rows are **terminal on first decision** — a row never re-op ```mermaid stateDiagram-v2 - [*] --> PENDING: Agent writes row
(TransactWriteItems with
TaskTable → AWAITING_APPROVAL) + [*] --> PENDING: Approval service creates row
(transaction with
TaskTable → AWAITING_APPROVAL) PENDING --> APPROVED: ApproveTaskFn
(cross-table transaction) PENDING --> DENIED: DenyTaskFn
(cross-table transaction) - PENDING --> TIMED_OUT: Agent poll timeout
(best-effort update) + PENDING --> TIMED_OUT: Approval service records
worker timeout PENDING --> CANCELLED: Task owner cancels
(cross-table transaction) APPROVED --> [*]: terminal DENIED --> [*]: terminal TIMED_OUT --> [*]: terminal CANCELLED --> [*]: terminal note right of APPROVED - TTL = created_at + timeout_s + 120s - DDB reaps row after TTL + Terminal retention TTL + never expires a pending request end note ``` @@ -1683,11 +1686,25 @@ These alarms transition to `ALARM` state in CloudWatch and appear in the console ### 12.1 Trust boundaries -- **Agent container ↔ TaskApprovalsTable**: IAM role on the runtime has `GetItem` / `PutItem` / conditional `UpdateItem` on the table. Agent writes pending, reads decisions, writes TIMED_OUT on internal timeout. +- **Agent container ↔ TaskApprovalsTable**: workers have task-scoped reads and + transaction condition checks, with no direct item writes or deletion. +- **Agent container ↔ approval request service**: IAM restricts signed + `POST /v1/tasks/{task_id}` to the session's `task_id` tag. The service creates + `PENDING` rows or conditionally records a non-human `TIMED_OUT`; it rejects + human decisions, notification markers, retention TTL and other extra fields. + It checks current task ownership/state and, for MicroVM, the active worker + lease in the same transaction. Worker-provided action descriptions remain + untrusted. This protects approval records; it does not sandbox code running + inside the agent or bind ambient compute credentials to one task. - **User CLI ↔ API Gateway**: Cognito JWT (same authorizer as `/tasks/*`). Cognito `sub` is the canonical caller identity, used **verbatim** in DDB `ConditionExpression` (§7.1, finding #6). - **ApproveTaskFn/DenyTaskFn ↔ TaskApprovalsTable + TaskTable**: Lambda IAM policy allows `UpdateItem` on both tables under `TransactWriteItems`. Authorization is in the ConditionExpression (ownership AND state), not in a separate IAM boundary. - **Blueprint origin**: blueprints are CDK-deployed constructs (see `cdk/src/constructs/blueprint.ts`). Platform operators deploy them. Users cannot upload arbitrary blueprint.yaml from the target repo. This property is load-bearing for the security model — if blueprint origin ever becomes user-uploaded, the blueprint-injection section (§12.4) must be re-evaluated. The 64 KB text cap (§5.1, finding #12) and `disable:` hard-deny rejection (finding #9) are applied regardless of origin as defense in depth. -- **Slack → ApproveTaskFn**: mediated by the fan-out Lambda + `SlackUserMappingTable` (§11.2). Slack admin cannot forge mappings; Slack approvals capped at `severity: low|medium` (finding #4). +- **Linear → decision handlers**: the mapped task owner must author the actual + reply returned by Linear's API. A webhook signature alone is insufficient. + The Slack button/proxy design in §11.2 remains future work. + +See [upgrading approval permissions](../guides/DEPLOYMENT_GUIDE.md#upgrading-approval-permissions) +before updating an existing deployment. ### 12.2 Ownership encoded in ConditionExpression @@ -1696,7 +1713,10 @@ No TOCTOU window. The `TransactWriteItems` (§7.1) encodes across two tables: - TaskApprovalsTable: `#status = :pending AND user_id = :caller` - TaskTable: `#status = :awaiting AND awaiting_approval_request_id = :rid` -Authorization + approvals-state + task-state transition all atomic. A compromised internal caller (Lambda with raw DDB access) or a logic bug in a future refactor that forgets the ownership check still can't flip rows without matching the `user_id`. The task-state guard additionally prevents the "approve succeeds on a cancelled task" race (finding #7). +The handlers check authorization, approval state and task state atomically. +These conditions prevent races; they do not constrain a compromised Lambda with +raw table-write permission, which could omit them. Only trusted control-plane +handlers receive that permission. `user_id` comparison is against Cognito `sub` **verbatim** — byte-for-byte equality. Any future identity transformation (per-tenant prefixing, namespacing) must apply to BOTH the write path (agent-side row write) AND the compare path (Lambda ConditionExpression) simultaneously, or the comparison silently fails under the new format. A unit test (§15.3) enforces this: given a sample JWT, extract `sub`, write a row, then assert the stored `user_id` equals `sub` byte-for-byte. diff --git a/docs/guides/DEPLOYMENT_GUIDE.md b/docs/guides/DEPLOYMENT_GUIDE.md index c47bd0691..f481ee608 100644 --- a/docs/guides/DEPLOYMENT_GUIDE.md +++ b/docs/guides/DEPLOYMENT_GUIDE.md @@ -243,6 +243,32 @@ Triggers via `workflow_run` when `build.yml` completes successfully. The pipelin ## Known deployment issues +### Upgrading approval permissions + +PR #904 moves worker approval creation and timeout writes into an IAM-authenticated +service. Deploy the matching agent image and CDK together: old workers write +directly to DynamoDB and cannot create new gates after those permissions are removed. + +1. Pause submissions from the CLI, integrations and schedules during the upgrade. + Let existing tasks finish, or have their owners cancel them. Include tasks + awaiting approval, suspended MicroVMs and retained continuations; an empty + running-container list does not prove the deployment has drained. +2. Build the agent from the same revision as the CDK. For an externally managed + MicroVM image, publish that build and select its new version before resuming + submissions. A suspended VM keeps its old code. +3. Deploy the stack. Check that the SessionRole has approval-table reads and + condition checks only, plus `execute-api:Invoke` restricted to its task tag. + CDK supplies `APPROVAL_REQUESTS_API_URL` to all three compute backends. +4. Submit a test task that triggers a known approval rule on each enabled backend. + Verify that the request appears, an owner decision resumes it, and an explicit + deadline records `TIMED_OUT` without overwriting a human decision. Then resume + normal submissions. + +If an old worker survives the upgrade, its next approval write fails closed. +Existing rows remain readable; do not restore direct writes to work around a stale +image. Roll forward with the matching image. Rolling back IAM restores the original +approval-record vulnerability and requires a deliberate operator decision. + ### AgentCore unsupported Availability Zones **Affects:** Fresh deploys in accounts whose default Availability Zones don't line up with the zones AgentCore supports for the region. diff --git a/docs/src/content/docs/architecture/Cedar-hitl-gates.md b/docs/src/content/docs/architecture/Cedar-hitl-gates.md index 60725cd9f..dc5024724 100644 --- a/docs/src/content/docs/architecture/Cedar-hitl-gates.md +++ b/docs/src/content/docs/architecture/Cedar-hitl-gates.md @@ -10,7 +10,7 @@ title: Cedar hitl gates > **Rev:** 5 (2026-05-06 — fold in parallel adversarial + advocate review of the timeout design: late-approval re-read on TIMED_OUT ConditionCheckFailed; user-visible timeout-cap milestones; ceiling-shrink milestone; Runtime JWT bound verified as auto-refreshed IAM; three new tuning metrics; explicit off-hours trade-off section; notification-delivery-failure boundary. IMPL-24 through IMPL-28 added.). > **Implementation:** Core shipped. The 3-outcome engine (`agent/src/policy.py`), default policy sets (`agent/policies/hard_deny.cedar`, `agent/policies/soft_deny.cedar`), approval Lambdas (`cdk/src/handlers/{approve-task,deny-task,get-pending,get-policies}.ts`) wired into `cdk/src/constructs/task-api.ts` (routes `/tasks/{id}/approve`, `/deny`, `/pending`, `/repos/{repo_id}/policies`), the cross-engine parity fixtures (`contracts/cedar-parity/`), and the exact engine pins are all on `main`. §15's task list is preserved as a historical implementation record; see the note at the top of §15 for what (if anything) remains unbuilt. > -> **Current source behavior (2026-09-17):** the task default is `approval_timeout_s=0`, +> **Current source behavior (2026-09-18):** the task default is `approval_timeout_s=0`, > meaning no decision deadline. Explicit task settings are 30–3,600 seconds; > positive policy-rule deadlines still apply. Pending rows have no DynamoDB TTL, > and `expires_at` is nullable. Task closure cancels unanswered requests and adds @@ -18,8 +18,9 @@ title: Cedar hitl gates > conditions return `404 REQUEST_NOT_FOUND`; task-only conflicts return 409. > MicroVM checkpoint/retirement/replacement separates human waiting from worker > lifetime and capacity; other backends retain their existing runtime limits. -> CLI response instructions are implemented for Slack/Linear notifications; -> native channel decisions remain separate work. See the current +> Linear accepts an owner’s `approve` or `deny` reply to the approval comment; +> the handler verifies the actual comment through Linear’s API. Slack uses CLI +> response instructions. See the current > [user guide](/sample-autonomous-cloud-coding-agents/using/overview#approval-gates-cedar-hitl) and > [continuation protocol](/sample-autonomous-cloud-coding-agents/architecture/orchestrator#retained-microvm-approvals). > The normal deployment passed retained-request, ten-minute sleep, explicit-expiry @@ -236,7 +237,9 @@ Narrative walk-through of the happy path. Sequence diagrams in the round-trip Me "repo": "my-org/my-app" } ``` -15. **Atomic transition** — hook issues `TransactWriteItems` with two operations: +15. **Atomic transition** — the hook sends an IAM-signed request to the approval + service, which issues `TransactWriteItems` with these operations (and a + worker-lease condition for MicroVM): - Put on `TaskApprovalsTable` (new row with status=PENDING) - ConditionalUpdate on `TaskTable`: `status = :awaiting, awaiting_approval_request_id = :rid WHERE status = :running` Both succeed or both fail. On `TransactionCanceledException` (most likely the TaskTable condition fails because another process moved the status), the hook emits `approval_write_failed` and returns DENY. @@ -1315,18 +1318,18 @@ TaskApprovalsTable rows are **terminal on first decision** — a row never re-op ```mermaid stateDiagram-v2 - [*] --> PENDING: Agent writes row
(TransactWriteItems with
TaskTable → AWAITING_APPROVAL) + [*] --> PENDING: Approval service creates row
(transaction with
TaskTable → AWAITING_APPROVAL) PENDING --> APPROVED: ApproveTaskFn
(cross-table transaction) PENDING --> DENIED: DenyTaskFn
(cross-table transaction) - PENDING --> TIMED_OUT: Agent poll timeout
(best-effort update) + PENDING --> TIMED_OUT: Approval service records
worker timeout PENDING --> CANCELLED: Task owner cancels
(cross-table transaction) APPROVED --> [*]: terminal DENIED --> [*]: terminal TIMED_OUT --> [*]: terminal CANCELLED --> [*]: terminal note right of APPROVED - TTL = created_at + timeout_s + 120s - DDB reaps row after TTL + Terminal retention TTL + never expires a pending request end note ``` @@ -1687,11 +1690,25 @@ These alarms transition to `ALARM` state in CloudWatch and appear in the console ### 12.1 Trust boundaries -- **Agent container ↔ TaskApprovalsTable**: IAM role on the runtime has `GetItem` / `PutItem` / conditional `UpdateItem` on the table. Agent writes pending, reads decisions, writes TIMED_OUT on internal timeout. +- **Agent container ↔ TaskApprovalsTable**: workers have task-scoped reads and + transaction condition checks, with no direct item writes or deletion. +- **Agent container ↔ approval request service**: IAM restricts signed + `POST /v1/tasks/{task_id}` to the session's `task_id` tag. The service creates + `PENDING` rows or conditionally records a non-human `TIMED_OUT`; it rejects + human decisions, notification markers, retention TTL and other extra fields. + It checks current task ownership/state and, for MicroVM, the active worker + lease in the same transaction. Worker-provided action descriptions remain + untrusted. This protects approval records; it does not sandbox code running + inside the agent or bind ambient compute credentials to one task. - **User CLI ↔ API Gateway**: Cognito JWT (same authorizer as `/tasks/*`). Cognito `sub` is the canonical caller identity, used **verbatim** in DDB `ConditionExpression` (§7.1, finding #6). - **ApproveTaskFn/DenyTaskFn ↔ TaskApprovalsTable + TaskTable**: Lambda IAM policy allows `UpdateItem` on both tables under `TransactWriteItems`. Authorization is in the ConditionExpression (ownership AND state), not in a separate IAM boundary. - **Blueprint origin**: blueprints are CDK-deployed constructs (see `cdk/src/constructs/blueprint.ts`). Platform operators deploy them. Users cannot upload arbitrary blueprint.yaml from the target repo. This property is load-bearing for the security model — if blueprint origin ever becomes user-uploaded, the blueprint-injection section (§12.4) must be re-evaluated. The 64 KB text cap (§5.1, finding #12) and `disable:` hard-deny rejection (finding #9) are applied regardless of origin as defense in depth. -- **Slack → ApproveTaskFn**: mediated by the fan-out Lambda + `SlackUserMappingTable` (§11.2). Slack admin cannot forge mappings; Slack approvals capped at `severity: low|medium` (finding #4). +- **Linear → decision handlers**: the mapped task owner must author the actual + reply returned by Linear's API. A webhook signature alone is insufficient. + The Slack button/proxy design in §11.2 remains future work. + +See [upgrading approval permissions](/sample-autonomous-cloud-coding-agents/getting-started/deployment-guide#upgrading-approval-permissions) +before updating an existing deployment. ### 12.2 Ownership encoded in ConditionExpression @@ -1700,7 +1717,10 @@ No TOCTOU window. The `TransactWriteItems` (§7.1) encodes across two tables: - TaskApprovalsTable: `#status = :pending AND user_id = :caller` - TaskTable: `#status = :awaiting AND awaiting_approval_request_id = :rid` -Authorization + approvals-state + task-state transition all atomic. A compromised internal caller (Lambda with raw DDB access) or a logic bug in a future refactor that forgets the ownership check still can't flip rows without matching the `user_id`. The task-state guard additionally prevents the "approve succeeds on a cancelled task" race (finding #7). +The handlers check authorization, approval state and task state atomically. +These conditions prevent races; they do not constrain a compromised Lambda with +raw table-write permission, which could omit them. Only trusted control-plane +handlers receive that permission. `user_id` comparison is against Cognito `sub` **verbatim** — byte-for-byte equality. Any future identity transformation (per-tenant prefixing, namespacing) must apply to BOTH the write path (agent-side row write) AND the compare path (Lambda ConditionExpression) simultaneously, or the comparison silently fails under the new format. A unit test (§15.3) enforces this: given a sample JWT, extract `sub`, write a row, then assert the stored `user_id` equals `sub` byte-for-byte. diff --git a/docs/src/content/docs/getting-started/Deployment-guide.md b/docs/src/content/docs/getting-started/Deployment-guide.md index 654c438a8..260339881 100644 --- a/docs/src/content/docs/getting-started/Deployment-guide.md +++ b/docs/src/content/docs/getting-started/Deployment-guide.md @@ -247,6 +247,32 @@ Triggers via `workflow_run` when `build.yml` completes successfully. The pipelin ## Known deployment issues +### Upgrading approval permissions + +PR #904 moves worker approval creation and timeout writes into an IAM-authenticated +service. Deploy the matching agent image and CDK together: old workers write +directly to DynamoDB and cannot create new gates after those permissions are removed. + +1. Pause submissions from the CLI, integrations and schedules during the upgrade. + Let existing tasks finish, or have their owners cancel them. Include tasks + awaiting approval, suspended MicroVMs and retained continuations; an empty + running-container list does not prove the deployment has drained. +2. Build the agent from the same revision as the CDK. For an externally managed + MicroVM image, publish that build and select its new version before resuming + submissions. A suspended VM keeps its old code. +3. Deploy the stack. Check that the SessionRole has approval-table reads and + condition checks only, plus `execute-api:Invoke` restricted to its task tag. + CDK supplies `APPROVAL_REQUESTS_API_URL` to all three compute backends. +4. Submit a test task that triggers a known approval rule on each enabled backend. + Verify that the request appears, an owner decision resumes it, and an explicit + deadline records `TIMED_OUT` without overwriting a human decision. Then resume + normal submissions. + +If an old worker survives the upgrade, its next approval write fails closed. +Existing rows remain readable; do not restore direct writes to work around a stale +image. Roll forward with the matching image. Rolling back IAM restores the original +approval-record vulnerability and requires a deliberate operator decision. + ### AgentCore unsupported Availability Zones **Affects:** Fresh deploys in accounts whose default Availability Zones don't line up with the zones AgentCore supports for the region. From eda5cb5fd77298c261a5af6a771f3fcef282a3f3 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Fri, 18 Sep 2026 19:58:03 -0400 Subject: [PATCH 104/149] test(approvals): use complete pending requests in transport fixtures --- agent/src/task_state.py | 10 ++++++---- agent/tests/test_approval_requests.py | 22 ++++++++++++++++++++-- 2 files changed, 26 insertions(+), 6 deletions(-) diff --git a/agent/src/task_state.py b/agent/src/task_state.py index f4f499167..c381fd364 100644 --- a/agent/src/task_state.py +++ b/agent/src/task_state.py @@ -533,10 +533,9 @@ def get_task(task_id: str, *, consistent_read: bool = False) -> dict | None: # - ``get_approval_row`` — strongly-consistent GetItem; default # ``consistent_read=True`` because the race fix relies on it. # -# Errors beyond the structural conditions (unreachable DDB, IAM drift, -# missing env var) raise ``ApprovalTablesUnavailable`` so the hook can -# fail CLOSED without guessing. The hook maps that to DENY so a deploy -# without the approvals table cannot silently bypass gates. +# New deployments route creation and timeout writes through the trusted approval +# service. Missing table configuration raises ``ApprovalTablesUnavailable``; +# transport/IAM errors propagate. The hook fails closed in either case. TASK_APPROVALS_TABLE_ENV = "TASK_APPROVALS_TABLE_NAME" TASK_TABLE_ENV = "TASK_TABLE_NAME" @@ -672,6 +671,9 @@ def transact_write_approval_request( ) -> None: """Atomically record a pending approval + transition the task to AWAITING_APPROVAL. + The configured approval service performs the transaction. Direct DynamoDB + is retained only for older deployments with the legacy permission model. + Two items: 1. Put on ``TaskApprovalsTable`` with ``ConditionExpression: attribute_not_exists(request_id)`` — guards against ULID collisions diff --git a/agent/tests/test_approval_requests.py b/agent/tests/test_approval_requests.py index f58cc28be..84c332a4b 100644 --- a/agent/tests/test_approval_requests.py +++ b/agent/tests/test_approval_requests.py @@ -13,6 +13,24 @@ import task_state +def pending_request() -> task_state.ApprovalRow: + return { + "task_id": "task", + "request_id": "request", + "tool_name": "Bash", + "tool_input_preview": '{"command":"git push"}', + "tool_input_sha256": "a" * 64, + "reason": "Protected operation", + "severity": "high", + "matching_rule_ids": ["protected"], + "status": "PENDING", + "created_at": "2026-09-18T00:00:00Z", + "timeout_s": 0, + "user_id": "owner", + "repo": "owner/repo", + } + + @pytest.fixture def transport(monkeypatch): monkeypatch.setenv(broker.API_ENV, "https://fixture.execute-api.us-east-1.amazonaws.com/v1/") @@ -57,7 +75,7 @@ def test_rejects_untrusted_endpoints_before_signing(transport, monkeypatch, url) def test_new_writers_use_service_instead_of_direct_dynamodb(transport, monkeypatch): direct = MagicMock() monkeypatch.setattr(task_state, "_get_ddb_client", direct) - task_state.transact_write_approval_request("task", "request", {"status": "PENDING"}) + task_state.transact_write_approval_request("task", "request", pending_request()) assert task_state.best_effort_update_approval_status("task", "request", "TIMED_OUT") assert transport.call_count == 2 direct.assert_not_called() @@ -75,7 +93,7 @@ def test_uncertain_service_write_does_not_fall_back_to_direct_dynamodb(transport monkeypatch.setattr(task_state, "_get_ddb_client", direct) transport.side_effect = TimeoutError("reply lost") with pytest.raises(TimeoutError): - task_state.transact_write_approval_request("task", "request", {"status": "PENDING"}) + task_state.transact_write_approval_request("task", "request", pending_request()) direct.assert_not_called() From a45922a150415122235ea20feaf5c67a3126d490 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Fri, 18 Sep 2026 20:03:15 -0400 Subject: [PATCH 105/149] test(microvm): include trusted approval service in platform contract --- agent/tests/test_server.py | 1 + 1 file changed, 1 insertion(+) diff --git a/agent/tests/test_server.py b/agent/tests/test_server.py index 11ed0859a..d867d33c4 100644 --- a/agent/tests/test_server.py +++ b/agent/tests/test_server.py @@ -1776,6 +1776,7 @@ def test_wire_contract_is_exactly_the_documented_key_set(self): "task_table_name": "TASK_TABLE_NAME", "task_events_table_name": "TASK_EVENTS_TABLE_NAME", "task_approvals_table_name": "TASK_APPROVALS_TABLE_NAME", + "approval_requests_api_url": "APPROVAL_REQUESTS_API_URL", "nudges_table_name": "NUDGES_TABLE_NAME", "log_group_name": "LOG_GROUP_NAME", "artifacts_bucket_name": "ARTIFACTS_BUCKET_NAME", From 7fc02a56fef7b8a71a30228bd13f5a50c473c33b Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Sat, 19 Sep 2026 10:46:07 -0400 Subject: [PATCH 106/149] test(cdk): keep MicroVM security synthesis independent of EC2 --- cdk/src/main.ts | 3 ++- cdk/test/stacks/microvm-managed-image-nag.test.ts | 6 ++++++ 2 files changed, 8 insertions(+), 1 deletion(-) diff --git a/cdk/src/main.ts b/cdk/src/main.ts index 7637e5fea..c3371ae05 100644 --- a/cdk/src/main.ts +++ b/cdk/src/main.ts @@ -58,7 +58,8 @@ export interface BuildAppOptions { * Async because AgentCore-supported availability zones are resolved from the * account's zone mapping at synth time (live `DescribeAvailabilityZones` + * `sts:GetCallerIdentity`) when a concrete account/region is bound. Env-agnostic - * synth and the validated context override never touch AWS. + * synth does not call AWS; explicit overrides in supported regions are checked + * against the account's EC2 zone mapping. */ export async function buildApp(options: BuildAppOptions = {}): Promise { const app = new App(options.appProps); diff --git a/cdk/test/stacks/microvm-managed-image-nag.test.ts b/cdk/test/stacks/microvm-managed-image-nag.test.ts index ea53e1cbd..1ede3ec24 100644 --- a/cdk/test/stacks/microvm-managed-image-nag.test.ts +++ b/cdk/test/stacks/microvm-managed-image-nag.test.ts @@ -33,6 +33,12 @@ describe.each(configurations)('managed MicroVM security checks (vault=$enableLin const app = await buildApp({ account: '123456789012', region: 'us-west-2', + // A concrete region verifies even explicit AZ overrides. Keep this + // template-security test independent of credentials and live EC2. + describeAzs: async () => [ + { zoneName: 'us-west-2a', zoneId: 'usw2-az1' }, + { zoneName: 'us-west-2b', zoneId: 'usw2-az2' }, + ], appProps: { context: { compute_type: 'lambda-microvm', From 235a6fbdaed582ffe2bcf93c775ac2e3f4d37579 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Sat, 19 Sep 2026 10:46:35 -0400 Subject: [PATCH 107/149] test(microvm): verify approval API transport in complete platform config --- .../shared/strategies/lambda-microvm-strategy.test.ts | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/cdk/test/handlers/shared/strategies/lambda-microvm-strategy.test.ts b/cdk/test/handlers/shared/strategies/lambda-microvm-strategy.test.ts index 02be0eb76..ad977417f 100644 --- a/cdk/test/handlers/shared/strategies/lambda-microvm-strategy.test.ts +++ b/cdk/test/handlers/shared/strategies/lambda-microvm-strategy.test.ts @@ -1116,6 +1116,7 @@ describe('buildMicrovmPlatformConfig — the MicroVM substitute for a deploy-tim TASK_TABLE_NAME: 'tasks', TASK_EVENTS_TABLE_NAME: 'events', TASK_APPROVALS_TABLE_NAME: 'approvals', + APPROVAL_REQUESTS_API_URL: 'https://fixture.execute-api.us-east-1.amazonaws.com/v1/', NUDGES_TABLE_NAME: 'nudges', LOG_GROUP_NAME: '/aws/abca/application', ARTIFACTS_BUCKET_NAME: 'artifacts-bucket', @@ -1161,6 +1162,7 @@ describe('buildMicrovmPlatformConfig — the MicroVM substitute for a deploy-tim 'task_table_name', 'task_events_table_name', 'task_approvals_table_name', + 'approval_requests_api_url', 'nudges_table_name', 'log_group_name', 'artifacts_bucket_name', @@ -1182,7 +1184,6 @@ describe('buildMicrovmPlatformConfig — the MicroVM substitute for a deploy-tim // `global`, and wrong in a way that surfaces only as AccessDenied at turn 0. 'anthropic_model', ]); - expect(MICROVM_PLATFORM_CONFIG_KEYS).toHaveLength(17); // snake_case on the wire, matching every other key in the /run envelope. for (const key of MICROVM_PLATFORM_CONFIG_KEYS) { expect(key).toMatch(/^[a-z][a-z0-9_]*$/); @@ -1207,6 +1208,7 @@ describe('buildMicrovmPlatformConfig — the MicroVM substitute for a deploy-tim const config = buildMicrovmPlatformConfig(FULL_ENV); expect(Object.keys(config)).toEqual([...MICROVM_PLATFORM_CONFIG_KEYS]); expect(config.task_table_name).toBe('tasks'); + expect(config.approval_requests_api_url).toBe(FULL_ENV.APPROVAL_REQUESTS_API_URL); expect(config.nudges_table_name).toBe('nudges'); expect(config.continuation_bucket_name).toBe('continuation-bucket'); expect(config.linear_vault_enabled).toBe('true'); @@ -1349,7 +1351,7 @@ describe('buildMicrovmPlatformConfig — the MicroVM substitute for a deploy-tim test('does NOT throw for a missing OPTIONAL key', () => { const env = { ...FULL_ENV }; for (const optional of [ - 'TASK_APPROVALS_TABLE_NAME', 'NUDGES_TABLE_NAME', 'LOG_GROUP_NAME', + 'TASK_APPROVALS_TABLE_NAME', 'APPROVAL_REQUESTS_API_URL', 'NUDGES_TABLE_NAME', 'LOG_GROUP_NAME', 'ARTIFACTS_BUCKET_NAME', 'TRACE_ARTIFACTS_BUCKET_NAME', 'LINEAR_OAUTH_SECRET_ARN', 'LINEAR_VAULT_ENABLED', 'LINEAR_WORKLOAD_IDENTITY_NAME', 'CONTINUATION_BUCKET_NAME', From de73239d19da967410e60ef6eb13361e999428f4 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Sat, 19 Sep 2026 10:49:35 -0400 Subject: [PATCH 108/149] docs(security): describe trusted approval writes and updated request flow --- docs/design/CEDAR_HITL_GATES.md | 6 ++++-- docs/design/SECURITY.md | 3 ++- docs/src/content/docs/architecture/Cedar-hitl-gates.md | 6 ++++-- docs/src/content/docs/architecture/Security.md | 3 ++- 4 files changed, 12 insertions(+), 6 deletions(-) diff --git a/docs/design/CEDAR_HITL_GATES.md b/docs/design/CEDAR_HITL_GATES.md index 4f8abee22..ba78fee78 100644 --- a/docs/design/CEDAR_HITL_GATES.md +++ b/docs/design/CEDAR_HITL_GATES.md @@ -317,6 +317,7 @@ sequenceDiagram participant Engine as PolicyEngine participant Events as TaskEventsTable participant Approvals as TaskApprovalsTable + participant Requests as Approval request service participant CLI participant User participant Lambda as ApproveTaskFn @@ -325,8 +326,9 @@ sequenceDiagram Agent->>Hook: tool call (Bash git push --force) Hook->>Engine: evaluate_tool_use Engine-->>Hook: REQUIRE_APPROVAL (soft-deny force_push_any) - Hook->>Approvals: TransactWriteItems - Note right of Hook: Put approval row PENDING
plus TaskTable status
to AWAITING_APPROVAL + Hook->>Requests: IAM-signed create for this task + Requests->>Approvals: TransactWriteItems + Note right of Requests: Put approval row PENDING
plus TaskTable status
to AWAITING_APPROVAL Hook->>Events: approval_requested milestone Events-->>CLI: live stream with approval_requested CLI-->>User: bgagent approve TASK REQ diff --git a/docs/design/SECURITY.md b/docs/design/SECURITY.md index 11f6b3372..eb64589a2 100644 --- a/docs/design/SECURITY.md +++ b/docs/design/SECURITY.md @@ -41,7 +41,8 @@ Three authentication mechanisms protect the platform, matching its input channel **Per-session IAM scoping** - The agent does not use its long-lived compute role (the AgentCore Runtime `ExecutionRole`, ECS Fargate task role, or Lambda MicroVMs execution role) for tenant data. Instead, at task startup it assumes a per-task **SessionRole** via `sts:AssumeRole` with session tags `{user_id, repo, task_id}`, and uses the resulting short-lived credentials for all DynamoDB and S3 tenant-data access. The SessionRole's policies self-constrain on those tags: -- **DynamoDB**: item access on the four `task_id`-partitioned tables (task, events, approvals, nudges) is gated by a `dynamodb:LeadingKeys` condition equal to `${aws:PrincipalTag/task_id}` and requires that context key to be present. `Scan` is not granted. The main task table permits reads and only attribute-scoped `UpdateItem` writes: the agent can report status, heartbeat, results and approval details, but cannot replace/delete the row or edit coordinator-owned start receipts, capacity reservations, owner identity or compute handles. The reviewed write list is `cdk/src/constructs/agent-task-write-attributes.json`; Python tests exercise the current writers against it. Supporting tables retain task-scoped item writes. `LeadingKeys` binds the base-table partition key, not a GSI key such as `user_id`. +- **DynamoDB**: item access on the four `task_id`-partitioned tables (task, events, approvals, nudges) is gated by a `dynamodb:LeadingKeys` condition equal to `${aws:PrincipalTag/task_id}` and requires that context key to be present. `Scan` is not granted. The main task table permits reads and only attribute-scoped `UpdateItem` writes: the agent can report status, heartbeat, results and approval details, but cannot replace/delete the row or edit coordinator-owned start receipts, capacity reservations, owner identity or compute handles. The reviewed write list is `cdk/src/constructs/agent-task-write-attributes.json`; Python tests exercise the current writers against it. Events and nudges retain task-scoped item writes. Approval records permit reads and transaction condition checks only. `LeadingKeys` binds the base-table partition key, not a GSI key such as `user_id`. +- **Approval requests**: workers call an IAM-signed API path restricted to their task tag. Its trusted handler can create `PENDING` or conditionally record `TIMED_OUT`; human decisions, notification markers and retention fields are unavailable to the worker. Task ownership/state and MicroVM worker leases are checked transactionally. See the [approval trust boundary](./CEDAR_HITL_GATES.md#121-trust-boundaries) and [upgrade procedure](../guides/DEPLOYMENT_GUIDE.md#upgrading-approval-permissions). - **S3**: trace writes and attachment reads are scoped to the `/${aws:PrincipalTag/user_id}/` object prefix. The compute role retains only non-tenant access (Bedrock model invocation — already ARN-scoped; CloudWatch Logs; the GitHub PAT secret, read once before the SessionRole is assumed; AgentCore Memory) plus `sts:AssumeRole`/`sts:TagSession` on the SessionRole. Because the agent runs under credentials that are themselves an assumed role, its `AssumeRole` is *role chaining* — capped at one hour regardless of the role's max session duration — so the agent uses a **refreshable** credential provider that re-assumes before expiry (tasks can run up to the 8-hour `maxLifetime`). The design is backend-agnostic: the same SessionRole and agent code serve all three backends (AgentCore Runtime, ECS Fargate, Lambda MicroVMs). Existing scoped credentials are constrained to their tags, but the compute role chooses those tags when assuming the role. The trust policy does not independently bind a worker to one task; this is not complete isolation against a compromised worker with ambient credentials. No worker needs direct access to the shared capacity counter. ECS's legacy configuration without a session role retains unscoped task reads/reporting updates, with the same attribute restriction. The policy structure and Python writer compatibility are tested locally. A historical IAM simulator check covered the original tag conditions; it does not verify the new attribute restriction or deployed transaction authorization. The [acceptance checklist](../verification/README.md#live-acceptance-for-an-installation) includes effective-role authorization checks. diff --git a/docs/src/content/docs/architecture/Cedar-hitl-gates.md b/docs/src/content/docs/architecture/Cedar-hitl-gates.md index dc5024724..bc4eeac0b 100644 --- a/docs/src/content/docs/architecture/Cedar-hitl-gates.md +++ b/docs/src/content/docs/architecture/Cedar-hitl-gates.md @@ -321,6 +321,7 @@ sequenceDiagram participant Engine as PolicyEngine participant Events as TaskEventsTable participant Approvals as TaskApprovalsTable + participant Requests as Approval request service participant CLI participant User participant Lambda as ApproveTaskFn @@ -329,8 +330,9 @@ sequenceDiagram Agent->>Hook: tool call (Bash git push --force) Hook->>Engine: evaluate_tool_use Engine-->>Hook: REQUIRE_APPROVAL (soft-deny force_push_any) - Hook->>Approvals: TransactWriteItems - Note right of Hook: Put approval row PENDING
plus TaskTable status
to AWAITING_APPROVAL + Hook->>Requests: IAM-signed create for this task + Requests->>Approvals: TransactWriteItems + Note right of Requests: Put approval row PENDING
plus TaskTable status
to AWAITING_APPROVAL Hook->>Events: approval_requested milestone Events-->>CLI: live stream with approval_requested CLI-->>User: bgagent approve TASK REQ diff --git a/docs/src/content/docs/architecture/Security.md b/docs/src/content/docs/architecture/Security.md index 6bb431540..4e850ca0c 100644 --- a/docs/src/content/docs/architecture/Security.md +++ b/docs/src/content/docs/architecture/Security.md @@ -45,7 +45,8 @@ Three authentication mechanisms protect the platform, matching its input channel **Per-session IAM scoping** - The agent does not use its long-lived compute role (the AgentCore Runtime `ExecutionRole`, ECS Fargate task role, or Lambda MicroVMs execution role) for tenant data. Instead, at task startup it assumes a per-task **SessionRole** via `sts:AssumeRole` with session tags `{user_id, repo, task_id}`, and uses the resulting short-lived credentials for all DynamoDB and S3 tenant-data access. The SessionRole's policies self-constrain on those tags: -- **DynamoDB**: item access on the four `task_id`-partitioned tables (task, events, approvals, nudges) is gated by a `dynamodb:LeadingKeys` condition equal to `${aws:PrincipalTag/task_id}` and requires that context key to be present. `Scan` is not granted. The main task table permits reads and only attribute-scoped `UpdateItem` writes: the agent can report status, heartbeat, results and approval details, but cannot replace/delete the row or edit coordinator-owned start receipts, capacity reservations, owner identity or compute handles. The reviewed write list is `cdk/src/constructs/agent-task-write-attributes.json`; Python tests exercise the current writers against it. Supporting tables retain task-scoped item writes. `LeadingKeys` binds the base-table partition key, not a GSI key such as `user_id`. +- **DynamoDB**: item access on the four `task_id`-partitioned tables (task, events, approvals, nudges) is gated by a `dynamodb:LeadingKeys` condition equal to `${aws:PrincipalTag/task_id}` and requires that context key to be present. `Scan` is not granted. The main task table permits reads and only attribute-scoped `UpdateItem` writes: the agent can report status, heartbeat, results and approval details, but cannot replace/delete the row or edit coordinator-owned start receipts, capacity reservations, owner identity or compute handles. The reviewed write list is `cdk/src/constructs/agent-task-write-attributes.json`; Python tests exercise the current writers against it. Events and nudges retain task-scoped item writes. Approval records permit reads and transaction condition checks only. `LeadingKeys` binds the base-table partition key, not a GSI key such as `user_id`. +- **Approval requests**: workers call an IAM-signed API path restricted to their task tag. Its trusted handler can create `PENDING` or conditionally record `TIMED_OUT`; human decisions, notification markers and retention fields are unavailable to the worker. Task ownership/state and MicroVM worker leases are checked transactionally. See the [approval trust boundary](/sample-autonomous-cloud-coding-agents/architecture/cedar-hitl-gates#121-trust-boundaries) and [upgrade procedure](/sample-autonomous-cloud-coding-agents/getting-started/deployment-guide#upgrading-approval-permissions). - **S3**: trace writes and attachment reads are scoped to the `/${aws:PrincipalTag/user_id}/` object prefix. The compute role retains only non-tenant access (Bedrock model invocation — already ARN-scoped; CloudWatch Logs; the GitHub PAT secret, read once before the SessionRole is assumed; AgentCore Memory) plus `sts:AssumeRole`/`sts:TagSession` on the SessionRole. Because the agent runs under credentials that are themselves an assumed role, its `AssumeRole` is *role chaining* — capped at one hour regardless of the role's max session duration — so the agent uses a **refreshable** credential provider that re-assumes before expiry (tasks can run up to the 8-hour `maxLifetime`). The design is backend-agnostic: the same SessionRole and agent code serve all three backends (AgentCore Runtime, ECS Fargate, Lambda MicroVMs). Existing scoped credentials are constrained to their tags, but the compute role chooses those tags when assuming the role. The trust policy does not independently bind a worker to one task; this is not complete isolation against a compromised worker with ambient credentials. No worker needs direct access to the shared capacity counter. ECS's legacy configuration without a session role retains unscoped task reads/reporting updates, with the same attribute restriction. The policy structure and Python writer compatibility are tested locally. A historical IAM simulator check covered the original tag conditions; it does not verify the new attribute restriction or deployed transaction authorization. The [acceptance checklist](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/docs/verification/README.md#live-acceptance-for-an-installation) includes effective-role authorization checks. From 2ae8f1165e458c826b77f389800ee1c47ad74dd6 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Mon, 21 Sep 2026 09:04:11 -0400 Subject: [PATCH 109/149] fix(645): keep flat MicroVM deployments within resource budget --- cdk/src/stacks/agent.ts | 15 ++++---- cdk/test/bootstrap/synth-coverage.test.ts | 4 +-- cdk/test/stacks/agent.test.ts | 34 +++++++++++++------ .../stacks/microvm-managed-image-nag.test.ts | 7 ++-- docs/guides/DEPLOYMENT_GUIDE.md | 16 ++++++++- .../docs/getting-started/Deployment-guide.md | 16 ++++++++- 6 files changed, 68 insertions(+), 24 deletions(-) diff --git a/cdk/src/stacks/agent.ts b/cdk/src/stacks/agent.ts index 92218b26c..463801690 100644 --- a/cdk/src/stacks/agent.ts +++ b/cdk/src/stacks/agent.ts @@ -1409,11 +1409,15 @@ export class AgentStack extends Stack { // Now that the orchestrator exists, resolve the Lazy used by TaskApi at synth. orchestratorArnHolder = orchestrator.alias.functionArn; + // Stateless scheduled jobs share a nested stack to leave room in both + // flat and nested MicroVM layouts. Upgrades replace their generated-name + // functions/schedules; task tables, buckets and compute resources stay put. + const concurrencyMaintenance = new NestedStack(this, 'ConcurrencyMaintenance'); if (continuationBucket && lambdaMicrovm?.imageArn) { taskApi.enableMicrovmContinuations( continuationBucket.bucket.bucketName, orchestrator.fn.functionArn, userConcurrencyTable.table, ); - new MicrovmContinuationManager(this, 'MicrovmContinuationManager', { + new MicrovmContinuationManager(concurrencyMaintenance, 'MicrovmContinuationManager', { taskTable: taskTable.table, approvalsTable: taskApprovalsTable.table, userConcurrencyTable: userConcurrencyTable.table, @@ -1428,9 +1432,6 @@ export class AgentStack extends Stack { agentMemory.grantReadWrite(orchestrator.fn); // --- Concurrency counter reconciler (drift correction) --- - // Keep this stateless repair job out of the parent resource budget. - // Existing deployments recreate its function/schedule; the tables stay put. - const concurrencyMaintenance = new NestedStack(this, 'ConcurrencyMaintenance'); new ConcurrencyReconciler(concurrencyMaintenance, 'ConcurrencyReconciler', { taskTable: taskTable.table, userConcurrencyTable: userConcurrencyTable.table, @@ -1441,7 +1442,7 @@ export class AgentStack extends Stack { // concurrency cap is hit) in FIFO order as slots free up: flips // QUEUED -> SUBMITTED and re-invokes the orchestrator, whose atomic // admissionControl remains the single writer of the counter. - new AdmissionQueuePickup(this, 'AdmissionQueuePickup', { + new AdmissionQueuePickup(concurrencyMaintenance, 'AdmissionQueuePickup', { taskTable: taskTable.table, taskEventsTable: taskEventsTable.table, userConcurrencyTable: userConcurrencyTable.table, @@ -1453,7 +1454,7 @@ export class AgentStack extends Stack { // (orchestrator Lambda crash between TaskTable write and InvokeAgentRuntime, // container crash during startup, etc.). Transitions to FAILED with a // `task_stranded` event. - new StrandedTaskReconciler(this, 'StrandedTaskReconciler', { + new StrandedTaskReconciler(concurrencyMaintenance, 'StrandedTaskReconciler', { taskTable: taskTable.table, taskEventsTable: taskEventsTable.table, taskApprovalsTable: taskApprovalsTable.table, @@ -1464,7 +1465,7 @@ export class AgentStack extends Stack { // Auto-cancels PENDING_UPLOADS tasks that were never confirmed within // 30 minutes (client crash, abandoned session, network failure). // Cleans up orphaned S3 objects under the task's attachment prefix. - new PendingUploadCleanup(this, 'PendingUploadCleanup', { + new PendingUploadCleanup(concurrencyMaintenance, 'PendingUploadCleanup', { taskTable: taskTable.table, taskEventsTable: taskEventsTable.table, attachmentsBucket: attachmentsBucket.bucket, diff --git a/cdk/test/bootstrap/synth-coverage.test.ts b/cdk/test/bootstrap/synth-coverage.test.ts index 3a675ef34..8c1c760f6 100644 --- a/cdk/test/bootstrap/synth-coverage.test.ts +++ b/cdk/test/bootstrap/synth-coverage.test.ts @@ -92,11 +92,11 @@ describe('Bootstrap policy synth coverage', () => { expect(missingByType).toEqual({}); }); - it('covers the managed MicroVM image, suspend parameter and every nested resource', () => { + it.each([false, true])('covers managed MicroVM and nested resources (microvm_nested_stack=%s)', microvmNested => { const app = new App({ context: { compute_type: 'lambda-microvm', - microvm_nested_stack: true, + microvm_nested_stack: microvmNested, microvm_base_image_arn: 'arn:aws:lambda:us-east-1:aws:microvm-image:al2023-1', microvm_base_image_version: '1', microvm_artifact_sha256: 'a'.repeat(64), diff --git a/cdk/test/stacks/agent.test.ts b/cdk/test/stacks/agent.test.ts index a50aaabe3..c5b3f42ea 100644 --- a/cdk/test/stacks/agent.test.ts +++ b/cdk/test/stacks/agent.test.ts @@ -48,15 +48,22 @@ describe('AgentStack', () => { expect(template).toBeDefined(); }); - test('nests the concurrency repair job while retaining its tables in the parent', () => { + test('nests scheduled maintenance while retaining its data in the parent', () => { const parentFunctions = Object.keys(template.findResources('AWS::Lambda::Function')); - expect(parentFunctions.some(id => id.startsWith('ConcurrencyReconciler'))).toBe(false); - concurrencyMaintenance.resourceCountIs('AWS::Lambda::Function', 1); + for (const prefix of ['ConcurrencyReconciler', 'AdmissionQueuePickup', 'StrandedTaskReconciler', 'PendingUploadCleanup']) { + expect(parentFunctions.some(id => id.startsWith(prefix))).toBe(false); + expect(Object.keys(concurrencyMaintenance.findResources('AWS::Lambda::Function')) + .filter(id => id.startsWith(prefix))).toHaveLength(1); + } + concurrencyMaintenance.resourceCountIs('AWS::Lambda::Function', 4); + concurrencyMaintenance.resourceCountIs('AWS::Events::Rule', 4); + concurrencyMaintenance.resourceCountIs('AWS::S3::Bucket', 0); concurrencyMaintenance.resourceCountIs('AWS::DynamoDB::Table', 0); concurrencyMaintenance.hasResourceProperties('AWS::Events::Rule', { ScheduleExpression: 'rate(15 minutes)', }); - const fn = Object.values(concurrencyMaintenance.findResources('AWS::Lambda::Function'))[0]; + const fn = Object.entries(concurrencyMaintenance.findResources('AWS::Lambda::Function')) + .find(([id]) => id.startsWith('ConcurrencyReconciler'))![1]; const variables = fn.Properties.Environment.Variables; const nested = Object.entries(template.findResources('AWS::CloudFormation::Stack')) .find(([id]) => id.startsWith('ConcurrencyMaintenanceNestedStack'))![1]; @@ -2211,15 +2218,12 @@ describe('AgentStack CloudFormation resource budget 500 with cushion', () => { // `AWS::CDK::Metadata`. Budget the synthesized number, so add that resource back. const SYNTH_ONLY_RESOURCES = 1; - const CONFIGURATIONS = [ - { name: 'agentcore', context: { compute_type: 'agentcore' } }, - { name: 'ecs', context: { compute_type: 'ecs' } }, - { name: 'microvm-bootstrap', context: { compute_type: 'lambda-microvm', microvm_nested_stack: true } }, + const MICROVM_CONFIGURATIONS = [ + { name: 'microvm-bootstrap', context: { compute_type: 'lambda-microvm' } }, { name: 'microvm-imported', context: { compute_type: 'lambda-microvm', - microvm_nested_stack: true, microvm_image_identifier: 'arn:aws:lambda:us-east-1:123456789012:microvm-image:existing-agent', microvm_image_version: '6.0', }, @@ -2228,13 +2232,23 @@ describe('AgentStack CloudFormation resource budget 500 with cushion', () => { name: 'microvm-managed', context: { compute_type: 'lambda-microvm', - microvm_nested_stack: true, microvm_base_image_arn: 'arn:aws:lambda:us-east-1:aws:microvm-image:al2023-1', microvm_base_image_version: '1', microvm_artifact_sha256: 'a'.repeat(64), }, }, ]; + const CONFIGURATIONS = [ + { name: 'agentcore', context: { compute_type: 'agentcore' } }, + { name: 'ecs', context: { compute_type: 'ecs' } }, + ...MICROVM_CONFIGURATIONS.flatMap(configuration => [ + { ...configuration, name: `${configuration.name}-default-flat` }, + { + name: `${configuration.name}-nested`, + context: { ...configuration.context, microvm_nested_stack: true }, + }, + ]), + ]; const CELLS = CONFIGURATIONS.flatMap(configuration => [false, true].flatMap(enableToolGateway => [false, true].map(enableLinearIdentityVault => ({ diff --git a/cdk/test/stacks/microvm-managed-image-nag.test.ts b/cdk/test/stacks/microvm-managed-image-nag.test.ts index 1ede3ec24..791889f58 100644 --- a/cdk/test/stacks/microvm-managed-image-nag.test.ts +++ b/cdk/test/stacks/microvm-managed-image-nag.test.ts @@ -24,9 +24,10 @@ import { TaskOrchestrator } from '../../src/constructs/task-orchestrator'; import { buildApp } from '../../src/main'; const configurations = [false, true].flatMap(enableLinearIdentityVault => - [false, true].map(extraWildcard => ({ enableLinearIdentityVault, extraWildcard }))); + [false, true].flatMap(extraWildcard => + [false, true].map(microvmNested => ({ enableLinearIdentityVault, extraWildcard, microvmNested })))); -describe.each(configurations)('managed MicroVM security checks (vault=$enableLinearIdentityVault, extra wildcard=$extraWildcard)', ({ enableLinearIdentityVault, extraWildcard }) => { +describe.each(configurations)('managed MicroVM security checks (vault=$enableLinearIdentityVault, extra wildcard=$extraWildcard, nested=$microvmNested)', ({ enableLinearIdentityVault, extraWildcard, microvmNested }) => { let errors: string[]; beforeAll(async () => { @@ -42,7 +43,7 @@ describe.each(configurations)('managed MicroVM security checks (vault=$enableLin appProps: { context: { compute_type: 'lambda-microvm', - microvm_nested_stack: true, + microvm_nested_stack: microvmNested, enableLinearIdentityVault, enableToolGateway: true, microvm_base_image_arn: 'arn:aws:lambda:us-west-2:aws:microvm-image:al2023-1', diff --git a/docs/guides/DEPLOYMENT_GUIDE.md b/docs/guides/DEPLOYMENT_GUIDE.md index f481ee608..6567b8224 100644 --- a/docs/guides/DEPLOYMENT_GUIDE.md +++ b/docs/guides/DEPLOYMENT_GUIDE.md @@ -245,7 +245,7 @@ Triggers via `workflow_run` when `build.yml` completes successfully. The pipelin ### Upgrading approval permissions -PR #904 moves worker approval creation and timeout writes into an IAM-authenticated +Worker approval creation and timeout writes now use an IAM-authenticated service. Deploy the matching agent image and CDK together: old workers write directly to DynamoDB and cannot create new gates after those permissions are removed. @@ -269,6 +269,20 @@ Existing rows remain readable; do not restore direct writes to work around a sta image. Roll forward with the matching image. Rolling back IAM restores the original approval-record vulnerability and requires a deliberate operator decision. +### Scheduled maintenance stack + +Concurrency repair, admission-queue pickup, stranded-task repair and pending-upload +cleanup run in the `ConcurrencyMaintenance` nested stack. MicroVM continuation +recovery also runs there when an image is configured. An upgrade recreates the +stateless functions, roles and schedules; their task tables and storage stay in +the parent stack. Review those replacements in the change set after draining tasks +as described above. Both flat and nested MicroVM layouts support the optional +tool gateway and Linear Identity vault without exceeding the template budget. + +This does not migrate existing MicroVM compute resources. Keep an existing flat +deployment on `microvm_nested_stack=false` until its separate resource migration +has been reviewed. + ### AgentCore unsupported Availability Zones **Affects:** Fresh deploys in accounts whose default Availability Zones don't line up with the zones AgentCore supports for the region. diff --git a/docs/src/content/docs/getting-started/Deployment-guide.md b/docs/src/content/docs/getting-started/Deployment-guide.md index 260339881..dd056918d 100644 --- a/docs/src/content/docs/getting-started/Deployment-guide.md +++ b/docs/src/content/docs/getting-started/Deployment-guide.md @@ -249,7 +249,7 @@ Triggers via `workflow_run` when `build.yml` completes successfully. The pipelin ### Upgrading approval permissions -PR #904 moves worker approval creation and timeout writes into an IAM-authenticated +Worker approval creation and timeout writes now use an IAM-authenticated service. Deploy the matching agent image and CDK together: old workers write directly to DynamoDB and cannot create new gates after those permissions are removed. @@ -273,6 +273,20 @@ Existing rows remain readable; do not restore direct writes to work around a sta image. Roll forward with the matching image. Rolling back IAM restores the original approval-record vulnerability and requires a deliberate operator decision. +### Scheduled maintenance stack + +Concurrency repair, admission-queue pickup, stranded-task repair and pending-upload +cleanup run in the `ConcurrencyMaintenance` nested stack. MicroVM continuation +recovery also runs there when an image is configured. An upgrade recreates the +stateless functions, roles and schedules; their task tables and storage stay in +the parent stack. Review those replacements in the change set after draining tasks +as described above. Both flat and nested MicroVM layouts support the optional +tool gateway and Linear Identity vault without exceeding the template budget. + +This does not migrate existing MicroVM compute resources. Keep an existing flat +deployment on `microvm_nested_stack=false` until its separate resource migration +has been reviewed. + ### AgentCore unsupported Availability Zones **Affects:** Fresh deploys in accounts whose default Availability Zones don't line up with the zones AgentCore supports for the region. From 12525191589d6bc64085953f24301979a979b9ea Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Mon, 21 Sep 2026 09:04:20 -0400 Subject: [PATCH 110/149] test(645): pin manifest guards and approval lease fencing --- cdk/test/handlers/request-approval.test.ts | 1 + .../microvm-continuation-storage.test.ts | 18 +++++++++++++++++- 2 files changed, 18 insertions(+), 1 deletion(-) diff --git a/cdk/test/handlers/request-approval.test.ts b/cdk/test/handlers/request-approval.test.ts index 8981a1a46..b33cde8dd 100644 --- a/cdk/test/handlers/request-approval.test.ts +++ b/cdk/test/handlers/request-approval.test.ts @@ -70,6 +70,7 @@ test('fences stale MicroVM writers in the same transaction', async () => { await recordWorkerRequest({ ...input, worker_attempt_id: 'worker-token' }); const lease = send.mock.calls[1][0].input.TransactItems[2].ConditionCheck; expect(lease.Key).toEqual({ task_id: 'worker-lease#task' }); + expect(lease.ConditionExpression).toBe('lease_state = :active AND lease_attempt_id = :attempt AND lease_user_id = :user'); expect(lease.ExpressionAttributeValues).toEqual({ ':active': 'ACTIVE', ':attempt': 'worker-token', ':user': 'owner', }); diff --git a/cdk/test/handlers/shared/microvm-continuation-storage.test.ts b/cdk/test/handlers/shared/microvm-continuation-storage.test.ts index 7da549f89..615e7018e 100644 --- a/cdk/test/handlers/shared/microvm-continuation-storage.test.ts +++ b/cdk/test/handlers/shared/microvm-continuation-storage.test.ts @@ -157,6 +157,18 @@ test.each(['checksum', 'length', 'identity', 'version'])('rejects a manifest %s const bytes = Buffer.from(JSON.stringify({ version: mismatch === 'version' ? 999 : 1, identity: mismatch === 'identity' ? { ...identity, user_id: 'other' } : identity, + conversation: { + key: `continuations/task/vm/request/${'a'.repeat(64)}.json`, + sha256: 'a'.repeat(64), + size_bytes: 100, + version_id: 'conversation-v', + }, + workspace: { + key: `continuations/task/vm/request/workspace/${'b'.repeat(64)}.tar`, + sha256: 'b'.repeat(64), + size_bytes: 512, + version_id: 'workspace-v', + }, })); const record: ContinuationRecord = { version: 1, @@ -171,7 +183,11 @@ test.each(['checksum', 'length', 'identity', 'version'])('rejects a manifest %s }, }; mockS3.mockResolvedValueOnce(object(bytes)); - await expect(verifyContinuationCheckpoint(record)).rejects.toThrow('MICROVM_CONTINUATION_STORAGE_INVALID'); + await expect(verifyContinuationCheckpoint(record)).rejects.toThrow( + mismatch === 'checksum' || mismatch === 'length' + ? 'checkpoint manifest checksum does not match' + : 'checkpoint manifest identity does not match', + ); expect(mockS3).toHaveBeenCalledTimes(1); }); From 1443e3dd3ab8442f871eb5e281ea8a2051c3b59c Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Mon, 21 Sep 2026 09:04:29 -0400 Subject: [PATCH 111/149] fix(645): reject Linear consent from worker OAuth identity --- cdk/src/handlers/shared/linear-feedback.ts | 10 +++++++- .../handlers/shared/linear-feedback.test.ts | 23 +++++++++++++++++-- 2 files changed, 30 insertions(+), 3 deletions(-) diff --git a/cdk/src/handlers/shared/linear-feedback.ts b/cdk/src/handlers/shared/linear-feedback.ts index 9f09b41e0..442631251 100644 --- a/cdk/src/handlers/shared/linear-feedback.ts +++ b/cdk/src/handlers/shared/linear-feedback.ts @@ -429,13 +429,21 @@ export async function readLinearApprovalComment( const result = await graphqlData(token, ` query VerifyApprovalComment($id: String!) { organization { id } + viewer { id } comment(id: $id) { id body user { id } botActor { id } issue { id } parent { id } } }`, { id }); if (!result.ok) throw new Error('Linear approval comment verification unavailable'); if ((result.value.organization as { id?: string } | undefined)?.id !== ctx.linearWorkspaceId) { throw new Error('Linear approval verification workspace mismatch'); } - return result.value.comment as Awaited> ?? null; + const viewerId = (result.value.viewer as { id?: string } | undefined)?.id; + if (!viewerId) throw new Error('Linear approval verification identity unavailable'); + const comment = result.value.comment as Awaited> ?? null; + // Diagnostic user-mode OAuth tokens can post genuine human comments. Workers + // holding that token must not manufacture consent from its own identity. + // Read the identity from Linear, including for installations predating this check. + if (comment?.user?.id === viewerId) return null; + return comment; } /** Retry-safe posting for approval prompts and acknowledgements. */ diff --git a/cdk/test/handlers/shared/linear-feedback.test.ts b/cdk/test/handlers/shared/linear-feedback.test.ts index 6a9b13281..bf2e7bffd 100644 --- a/cdk/test/handlers/shared/linear-feedback.test.ts +++ b/cdk/test/handlers/shared/linear-feedback.test.ts @@ -85,19 +85,38 @@ describe('linear-feedback', () => { issue: { id: ISSUE_ID }, parent: { id: 'root' }, }; - fetchMock.mockResolvedValue(jsonResponse({ data: { organization: { id: CTX.linearWorkspaceId }, comment } })); + fetchMock.mockResolvedValue(jsonResponse({ data: { organization: { id: CTX.linearWorkspaceId }, viewer: { id: 'app-identity' }, comment } })); expect(await readLinearApprovalComment(CTX, 'reply')).toEqual(comment); const request = JSON.parse(fetchMock.mock.calls[0][1].body); expect(request.variables).toEqual({ id: 'reply' }); expect(request.query).toContain('user { id }'); expect(request.query).toContain('botActor { id }'); + expect(request.query).toContain('viewer { id }'); + }); + test('rejects a genuine human comment made as the saved OAuth token identity', async () => { + fetchMock.mockResolvedValue(jsonResponse({ + data: { + organization: { id: CTX.linearWorkspaceId }, + viewer: { id: 'task-owner' }, + comment: { + id: 'reply', + body: 'approve', + user: { id: 'task-owner' }, + botActor: null, + issue: { id: ISSUE_ID }, + parent: { id: 'root' }, + }, + }, + })); + expect(await readLinearApprovalComment(CTX, 'reply')).toBeNull(); }); test('returns no consent for a deleted comment', async () => { - fetchMock.mockResolvedValue(jsonResponse({ data: { organization: { id: CTX.linearWorkspaceId }, comment: null } })); + fetchMock.mockResolvedValue(jsonResponse({ data: { organization: { id: CTX.linearWorkspaceId }, viewer: { id: 'app-identity' }, comment: null } })); expect(await readLinearApprovalComment(CTX, 'deleted')).toBeNull(); }); test.each([ { errors: [{ message: 'Unavailable' }] }, + { data: { organization: { id: CTX.linearWorkspaceId }, comment: { user: { id: 'human' } } } }, { data: { organization: { id: 'other-workspace' }, comment: {} } }, ])('fails closed on lookup errors or incorrect workspace: %j', async response => { fetchMock.mockResolvedValue(jsonResponse(response)); From e59cda4ee03e26e7e94cad75f15c3d87d1c97a05 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Mon, 21 Sep 2026 09:08:50 -0400 Subject: [PATCH 112/149] fix(645): surface terminal write failure and retry transaction conflicts --- agent/src/pipeline.py | 16 ++++- agent/src/task_state.py | 61 +++++++++++++---- agent/tests/test_pipeline.py | 27 ++++++++ agent/tests/test_task_state.py | 119 ++++++++++++++++++++++++++++++++- 4 files changed, 207 insertions(+), 16 deletions(-) diff --git a/agent/src/pipeline.py b/agent/src/pipeline.py index 482259155..07dfff374 100644 --- a/agent/src/pipeline.py +++ b/agent/src/pipeline.py @@ -466,10 +466,22 @@ def _run_repoless_task( print_metrics(result_dict) terminal_status = "COMPLETED" if overall_status == "success" else "FAILED" - task_state.write_terminal(config.task_id, terminal_status, result_dict) + _persist_finished_task(config.task_id, terminal_status, result_dict) return result_dict +def _persist_finished_task(task_id: str, status: str, result: dict) -> None: + outcome = task_state.write_terminal(task_id, status, result) + if outcome in ( + task_state.TerminalWriteOutcome.FAILED, + task_state.TerminalWriteOutcome.SUPERSEDED, + ): + raise task_state.TerminalWriteError( + f"Task result was not committed ({outcome.value}); " + "inspect the task record and worker lease" + ) + + def _apply_post_hook_gates( workflow: Workflow | None, *, @@ -1824,7 +1836,7 @@ def _on_trace_truncated(max_bytes: int, first_dropped: int) -> None: # Persist terminal state to DynamoDB terminal_status = "COMPLETED" if overall_status == "success" else "FAILED" - task_state.write_terminal(config.task_id, terminal_status, result_dict) + _persist_finished_task(config.task_id, terminal_status, result_dict) return result_dict diff --git a/agent/src/task_state.py b/agent/src/task_state.py index c381fd364..09fcda2ff 100644 --- a/agent/src/task_state.py +++ b/agent/src/task_state.py @@ -7,7 +7,9 @@ """ import os +import random import time +from enum import StrEnum from typing import NotRequired, TypedDict from shell import log, log_error_cw @@ -96,9 +98,31 @@ def _lease_check(task_id: str, *, low_level: bool, identity=None) -> dict | None } +def _transact_with_conflict_retry(client, items: list): + """Retry only uncommitted transaction conflicts, never a failed ownership check.""" + from botocore.exceptions import ClientError + + max_attempts = 3 + for attempt in range(max_attempts): + try: + return client.transact_write_items(TransactItems=items) + except ClientError as error: + code = error.response.get("Error", {}).get("Code") + reasons = _extract_cancellation_reasons(error) + codes = {reason.get("Code") for reason in reasons} + conflict = code == "TransactionConflictException" or ( + code == "TransactionCanceledException" + and "TransactionConflict" in codes + and codes <= {"None", "TransactionConflict"} + ) + if not conflict or attempt == max_attempts - 1: + raise + time.sleep(random.SystemRandom().uniform(0.025, 0.075) * (2**attempt)) + + def _transact_task(client, task_id: str, *, TransactItems: list, identity=None): lease = _lease_check(task_id, low_level=True, identity=identity) - return client.transact_write_items(TransactItems=[*TransactItems, *([lease] if lease else [])]) + return _transact_with_conflict_retry(client, [*TransactItems, *([lease] if lease else [])]) def _update_task(table, task_id: str, *, low_level: bool = False, **operation): @@ -108,7 +132,7 @@ def _update_task(table, task_id: str, *, low_level: bool = False, **operation): if not low_level: operation["TableName"] = table.name client = table if low_level else table.meta.client - return client.transact_write_items(TransactItems=[{"Update": operation}, lease]) + return _transact_with_conflict_retry(client, [{"Update": operation}, lease]) def _task_status_conflict(error: Exception) -> bool: @@ -236,16 +260,29 @@ def write_running(task_id: str) -> None: log("WARN", f"[task_state] write_running failed (best-effort): {type(e).__name__}") -def write_terminal(task_id: str, status: str, result: dict | None = None) -> None: +class TerminalWriteOutcome(StrEnum): + WRITTEN = "written" + SUPERSEDED = "superseded" + FAILED = "failed" + DISABLED = "disabled" + + +class TerminalWriteError(RuntimeError): + """The worker finished but could not commit its result to the task record.""" + + +def write_terminal(task_id: str, status: str, result: dict | None = None) -> TerminalWriteOutcome: """Transition a task to a terminal state (COMPLETED or FAILED). Updates ``status_created_at`` alongside ``status`` — see - :func:`write_running` for why. + :func:`write_running` for why. Callers must not report successful completion + when the result is FAILED or SUPERSEDED. DISABLED supports local runs without + a configured task table; crash-path callers already report failure. """ try: table = _get_table() if table is None: - return + return TerminalWriteOutcome.DISABLED now = _now_iso() expr_names = {"#s": "status"} # Mixed value types: most are strings, but build_passed/lint_passed are @@ -356,6 +393,7 @@ def write_terminal(task_id: str, status: str, result: dict | None = None) -> Non ExpressionAttributeNames=expr_names, ExpressionAttributeValues=expr_values, ) + return TerminalWriteOutcome.WRITTEN except Exception as e: if _task_status_conflict(e): log( @@ -400,7 +438,7 @@ def write_terminal(task_id: str, status: str, result: dict | None = None) -> Non f"task_id={task_id!r} after ConditionalCheckFailed " f"(terminal-state race).", ) - return + return TerminalWriteOutcome.SUPERSEDED # Include DynamoDB's cancellation reasons: losing worker ownership is # not a benign status race and must never trigger trace self-healing. log_error_cw( @@ -408,6 +446,7 @@ def write_terminal(task_id: str, status: str, result: dict | None = None) -> Non f"CancellationReasons={_extract_cancellation_reasons(e)}", task_id=task_id, ) + return TerminalWriteOutcome.FAILED def write_trace_uri_conditional(task_id: str, uri: str) -> bool: @@ -447,12 +486,7 @@ def write_trace_uri_conditional(task_id: str, uri: str) -> bool: ) return True except Exception as e: - from botocore.exceptions import ClientError - - if ( - isinstance(e, ClientError) - and e.response.get("Error", {}).get("Code") == "ConditionalCheckFailedException" - ): + if _task_status_conflict(e): # Benign: URI was already persisted, or status isn't terminal yet. log( "INFO", @@ -464,7 +498,8 @@ def write_trace_uri_conditional(task_id: str, uri: str) -> bool: log( "WARN", f"[task_state] write_trace_uri_conditional failed for " - f"task_id={task_id!r}: {type(e).__name__}: {e}", + f"task_id={task_id!r}: {type(e).__name__}: {e}; " + f"CancellationReasons={_extract_cancellation_reasons(e)}", ) return False diff --git a/agent/tests/test_pipeline.py b/agent/tests/test_pipeline.py index e6b26ea6d..f8a3862a5 100644 --- a/agent/tests/test_pipeline.py +++ b/agent/tests/test_pipeline.py @@ -2340,3 +2340,30 @@ def test_setup_failure_swaps_eyes_to_cross( _args, kwargs = m_finished.call_args assert kwargs.get("success") is False assert kwargs.get("started_reaction_id") == "reaction-42" + + +@pytest.mark.parametrize("outcome", ["failed", "superseded"]) +def test_finished_pipeline_cannot_report_success_without_committing_result(monkeypatch, outcome): + import task_state + from pipeline import _persist_finished_task + + monkeypatch.setattr( + task_state, + "write_terminal", + MagicMock(return_value=task_state.TerminalWriteOutcome(outcome)), + ) + with pytest.raises(task_state.TerminalWriteError, match="Task result was not committed"): + _persist_finished_task("task", "COMPLETED", {"status": "success"}) + + +@pytest.mark.parametrize("outcome", ["written", "disabled"]) +def test_finished_pipeline_accepts_persisted_result_or_local_run(monkeypatch, outcome): + import task_state + from pipeline import _persist_finished_task + + monkeypatch.setattr( + task_state, + "write_terminal", + MagicMock(return_value=task_state.TerminalWriteOutcome(outcome)), + ) + _persist_finished_task("task", "COMPLETED", {"status": "success"}) diff --git a/agent/tests/test_task_state.py b/agent/tests/test_task_state.py index 5d3a254b7..0ce93921b 100644 --- a/agent/tests/test_task_state.py +++ b/agent/tests/test_task_state.py @@ -87,6 +87,121 @@ def test_terminal_transaction_heals_trace_only_when_worker_lease_passed( assert "CancellationReasons" in report.call_args.args[0] +@pytest.mark.parametrize( + ("codes", "benign"), + [ + (["ConditionalCheckFailed", "None"], True), + (["None", "ConditionalCheckFailed"], False), + (["ConditionalCheckFailed", "ConditionalCheckFailed"], False), + ], +) +def test_trace_transaction_classifies_status_race_separately_from_lease_loss( + monkeypatch, codes, benign +): + from botocore.exceptions import ClientError + + error = ClientError( + { + "Error": {"Code": "TransactionCanceledException"}, + "CancellationReasons": [{"Code": code} for code in codes], + }, + "TransactWriteItems", + ) + monkeypatch.setattr(task_state, "_get_table", MagicMock()) + monkeypatch.setattr(task_state, "_update_task", MagicMock(side_effect=error)) + report = MagicMock() + monkeypatch.setattr(task_state, "log", report) + assert not task_state.write_trace_uri_conditional("task", "s3://bucket/trace") + assert report.call_args.args[0] == ("INFO" if benign else "WARN") + if not benign: + assert "CancellationReasons=" in report.call_args.args[1] + + +@pytest.mark.parametrize( + "codes", [["TransactionConflict", "None"], ["None", "TransactionConflict"]] +) +def test_task_transaction_retries_conflict_with_identical_ownership_fence(monkeypatch, codes): + from botocore.exceptions import ClientError + + error = ClientError( + { + "Error": {"Code": "TransactionCanceledException"}, + "CancellationReasons": [{"Code": code} for code in codes], + }, + "TransactWriteItems", + ) + client = MagicMock() + client.transact_write_items.side_effect = [error, {"committed": True}] + monkeypatch.setattr(task_state.time, "sleep", MagicMock()) + lease = { + "ConditionCheck": { + "ConditionExpression": "lease_attempt_id = :attempt", + "ExpressionAttributeValues": {":attempt": "original-worker"}, + } + } + monkeypatch.setattr(task_state, "_lease_check", lambda *args, **kwargs: lease) + operation = {"TableName": "tasks", "Key": {"task_id": {"S": "task"}}} + assert task_state._update_task(client, "task", low_level=True, **operation) == { + "committed": True + } + assert client.transact_write_items.call_count == 2 + for call in client.transact_write_items.call_args_list: + assert call.kwargs["TransactItems"] == [{"Update": operation}, lease] + + +@pytest.mark.parametrize( + ("codes", "attempts"), + [ + (["TransactionConflict", "None"], 3), + (["TransactionConflict", "ConditionalCheckFailed"], 1), + (["None", "ConditionalCheckFailed"], 1), + ([], 1), + ], +) +def test_transaction_retry_is_bounded_and_never_retries_ownership_denial( + monkeypatch, codes, attempts +): + from botocore.exceptions import ClientError + + error = ClientError( + { + "Error": {"Code": "TransactionCanceledException"}, + "CancellationReasons": [{"Code": code} for code in codes], + }, + "TransactWriteItems", + ) + client = MagicMock() + client.transact_write_items.side_effect = error + monkeypatch.setattr(task_state.time, "sleep", MagicMock()) + with pytest.raises(ClientError): + task_state._transact_with_conflict_retry(client, []) + assert client.transact_write_items.call_count == attempts + + +@pytest.mark.parametrize( + ("failure", "expected"), + [ + (None, task_state.TerminalWriteOutcome.WRITTEN), + ("ConditionalCheckFailedException", task_state.TerminalWriteOutcome.SUPERSEDED), + ("AccessDeniedException", task_state.TerminalWriteOutcome.FAILED), + ], +) +def test_terminal_returns_persistence_outcome(monkeypatch, failure, expected): + from botocore.exceptions import ClientError + + monkeypatch.setattr(task_state, "_get_table", MagicMock()) + monkeypatch.setattr( + task_state, + "_update_task", + MagicMock( + side_effect=( + ClientError({"Error": {"Code": failure}}, "UpdateItem") if failure else None + ) + ), + ) + assert task_state.write_terminal("task", "COMPLETED") is expected + + class TestAgentWriteContract: def test_current_task_writers_fit_the_deployed_attribute_allowlist(self, monkeypatch): """Exercise real writers; detect a new field before IAM rejects it live. @@ -207,13 +322,15 @@ def test_write_inventory_requires_review_when_a_new_writer_is_added(self): ) or ( isinstance(child.func, ast.Name) - and child.func.id in {"_update_task", "_transact_task"} + and child.func.id + in {"_update_task", "_transact_task", "_transact_with_conflict_retry"} ) ) for child in ast.walk(node) ) } assert writers == { + "_transact_with_conflict_retry", "_update_task", "_transact_task", "write_running", From f82704d1d9ad77616252df6dabdda7673065cd33 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Mon, 21 Sep 2026 09:20:02 -0400 Subject: [PATCH 113/149] fix(645): validate approval writer configuration before execution --- agent/README.md | 3 +++ agent/tests/test_payload_bootstrap.py | 1 + agent/tests/test_server.py | 3 ++- cdk/src/constructs/ecs-agent-cluster.ts | 7 +++++-- cdk/src/handlers/request-approval.ts | 11 +++++++--- cdk/test/constructs/ecs-agent-cluster.test.ts | 18 +++-------------- cdk/test/handlers/request-approval.test.ts | 7 +++++++ .../lambda-microvm-strategy.test.ts | 20 +++++++++++-------- contracts/constants.json | 1 + 9 files changed, 42 insertions(+), 29 deletions(-) diff --git a/agent/README.md b/agent/README.md index 4d040f34b..349a47007 100644 --- a/agent/README.md +++ b/agent/README.md @@ -286,6 +286,7 @@ Each snake_case key installs into its UPPER_SNAKE env var, and a payload value * |---|---|---| | `task_table_name` | `TASK_TABLE_NAME` | ✅ | | `task_events_table_name` | `TASK_EVENTS_TABLE_NAME` | ✅ | +| `approval_requests_api_url` | `APPROVAL_REQUESTS_API_URL` | ✅ | | `github_token_secret_arn` | `GITHUB_TOKEN_SECRET_ARN` | ✅ | | `agent_session_role_arn` | `AGENT_SESSION_ROLE_ARN` | ✅ | | `task_approvals_table_name` | `TASK_APPROVALS_TABLE_NAME` | | @@ -302,6 +303,8 @@ Each snake_case key installs into its UPPER_SNAKE env var, and a payload value * | `anthropic_default_haiku_model` | `ANTHROPIC_DEFAULT_HAIKU_MODEL` | | | `anthropic_model` | `ANTHROPIC_MODEL` | Main model profile for the deployment's configured geography | +`APPROVAL_REQUESTS_API_URL` identifies the IAM-authenticated approval writer. All cloud backends receive it from CDK; workers create pending requests through this service and have read-only approval-table access. A MicroVM manifest missing the URL is rejected before execution. Deploy matching worker images and infrastructure together. + Values are **non-secret configuration only**. Credentials are fetched at task startup from Secrets Manager or AgentCore Identity. Linear vault use requires `linear_vault_enabled="true"` and the workload identity name; the task's channel metadata identifies the workspace grant. The image carries no task credentials. The allowlist **fails closed**: an unrecognised key rejects the whole block. Blank optional values are skipped; blank required values, control characters and inconsistent ARN account/partition fields are rejected. Deployment authentication comes from the IAM-read manifest and exact configuration comparison, including same-account workspace identifiers ([#817](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/817)). Rejections are structured so they are readable in the MicroVM log group: `400 MICROVM_RUN_PAYLOAD_INVALID` (unusable envelope — retrying the same body cannot help), `500 MICROVM_RUN_PAYLOAD_UNREADABLE` (manifest/payload read or stored bytes failed), `400 MICROVM_RUN_PLATFORM_CONFIG_INVALID` (key off the allowlist, non-object block, or non-string value — fix the producer), `400 MICROVM_RUN_PLATFORM_CONFIG_INCOMPLETE` (a required key missing or blank — fix the deployment wiring), `400 TASK_RECORD_INCOMPLETE` (same validator and vocabulary as `/invocations`). diff --git a/agent/tests/test_payload_bootstrap.py b/agent/tests/test_payload_bootstrap.py index b87fda334..12f9b465c 100644 --- a/agent/tests/test_payload_bootstrap.py +++ b/agent/tests/test_payload_bootstrap.py @@ -44,6 +44,7 @@ def signed_url(task_id="task-1", bucket="payload-bucket", region="us-east-1"): def transport(monkeypatch): config = { "task_table_name": "Tasks", + "approval_requests_api_url": "https://approval.execute-api.us-east-1.amazonaws.com/v1", "task_events_table_name": "Events", "agent_session_role_arn": "arn:aws:iam::123456789012:role/AgentSession", "github_token_secret_arn": ( diff --git a/agent/tests/test_server.py b/agent/tests/test_server.py index d867d33c4..182eda420 100644 --- a/agent/tests/test_server.py +++ b/agent/tests/test_server.py @@ -1799,10 +1799,11 @@ def test_wire_contract_is_exactly_the_documented_key_set(self): "anthropic_model": "ANTHROPIC_MODEL", } - def test_required_subset_is_exactly_the_four_run_blocking_keys(self): + def test_required_subset_includes_the_trusted_approval_service(self): assert ( frozenset( { + "approval_requests_api_url", "task_table_name", "task_events_table_name", "github_token_secret_arn", diff --git a/cdk/src/constructs/ecs-agent-cluster.ts b/cdk/src/constructs/ecs-agent-cluster.ts index 2c793f375..1cdbf78cb 100644 --- a/cdk/src/constructs/ecs-agent-cluster.ts +++ b/cdk/src/constructs/ecs-agent-cluster.ts @@ -29,7 +29,7 @@ import * as secretsmanager from 'aws-cdk-lib/aws-secretsmanager'; import { NagSuppressions } from 'cdk-nag'; import { Construct, type Node } from 'constructs'; import { AgentMemory } from './agent-memory'; -import { AgentSessionRole, grantAgentTaskTableAccess, grantAgentApprovalReadAccess } from './agent-session-role'; +import { AgentSessionRole, grantAgentTaskTableAccess } from './agent-session-role'; import { PLATFORM_DEFAULT_AUX_MODEL_ID, PLATFORM_DEFAULT_MODEL_ID, @@ -322,6 +322,10 @@ export class EcsAgentCluster extends Construct { constructor(scope: Construct, id: string, props: EcsAgentClusterProps) { super(scope, id); + if ((props.taskApprovalsTable || props.approvalRequestsApiUrl) + && (!props.agentSessionRole || !props.approvalRequestsApiUrl)) { + throw new Error('ECS approvals require agentSessionRole and approvalRequestsApiUrl with a task-scoped invocation grant'); + } this.containerName = 'AgentContainer'; // ECS Cluster with Fargate capacity provider and container insights @@ -513,7 +517,6 @@ export class EcsAgentCluster extends Construct { } else { grantAgentTaskTableAccess(props.taskTable, taskRole, false); props.taskEventsTable.grantReadWriteData(taskRole); - if (props.taskApprovalsTable) grantAgentApprovalReadAccess(props.taskApprovalsTable, taskRole, false); } // Capacity counters are coordinator-owned. The agent never accesses them. diff --git a/cdk/src/handlers/request-approval.ts b/cdk/src/handlers/request-approval.ts index 627296a06..31a534d5a 100644 --- a/cdk/src/handlers/request-approval.ts +++ b/cdk/src/handlers/request-approval.ts @@ -45,10 +45,14 @@ interface RequestInput { readonly reason?: string; } -function validId(value: unknown): value is string { +function validShortText(value: unknown): value is string { return typeof value === 'string' && value.length > 0 && value.length <= MAX_ID_LENGTH; } +function validId(value: unknown): value is string { + return validShortText(value) && /^[A-Za-z0-9][A-Za-z0-9_-]*$/.test(value); +} + /** Never copy decision, notification, retention or arbitrary caller fields. */ function validateRequest(input: RequestInput, task: Record): Record { const row = input.approval; @@ -56,12 +60,12 @@ function validateRequest(input: RequestInput, task: Record): Record || row.task_id !== input.task_id || row.request_id !== input.request_id || row.user_id !== task.user_id || row.repo !== (task.repo ?? '') || row.status !== 'PENDING' - || !validId(row.tool_name) || typeof row.tool_input_preview !== 'string' || row.tool_input_preview.length > MAX_TEXT_LENGTH + || !validShortText(row.tool_name) || typeof row.tool_input_preview !== 'string' || row.tool_input_preview.length > MAX_TEXT_LENGTH || typeof row.tool_input_sha256 !== 'string' || !/^[a-f0-9]{64}$/.test(row.tool_input_sha256) || typeof row.reason !== 'string' || row.reason.length > MAX_TEXT_LENGTH || !['low', 'medium', 'high'].includes(row.severity as string) || !Array.isArray(row.matching_rule_ids) || row.matching_rule_ids.length > 500 - || !row.matching_rule_ids.every(validId) + || !row.matching_rule_ids.every(validShortText) || typeof row.created_at !== 'string' || !Number.isFinite(Date.parse(row.created_at)) || typeof row.timeout_s !== 'number' || !Number.isInteger(row.timeout_s) || row.timeout_s < 0 || row.timeout_s > constants.approval_timeout_s.max @@ -79,6 +83,7 @@ function validateRequest(input: RequestInput, task: Record): Record */ export async function recordWorkerRequest(input: RequestInput): Promise> { if (!input || !validId(input.task_id) || !validId(input.request_id) + || (input.worker_attempt_id !== undefined && !validId(input.worker_attempt_id)) || !['create', 'timeout'].includes(input.operation)) { return { ok: false, code: 'APPROVAL_REQUEST_INVALID' }; } diff --git a/cdk/test/constructs/ecs-agent-cluster.test.ts b/cdk/test/constructs/ecs-agent-cluster.test.ts index 3be3da253..77b285131 100644 --- a/cdk/test/constructs/ecs-agent-cluster.test.ts +++ b/cdk/test/constructs/ecs-agent-cluster.test.ts @@ -771,6 +771,7 @@ describe('EcsAgentCluster construct', () => { userConcurrencyTable, githubTokenSecret, agentSessionRole: sessionRole, + approvalRequestsApiUrl: 'https://approval.execute-api.us-east-1.amazonaws.com/v1', }); return Template.fromStack(stack); } @@ -846,21 +847,8 @@ describe('EcsAgentCluster construct', () => { }); describe('EcsAgentCluster approval wiring without a SessionRole', () => { - let template: Template; - beforeAll(() => { template = createStack({ withApprovals: true }).template; }); - - test('grants approval reads even without a SessionRole, never direct writes', () => { - const approvalId = Object.keys(template.findResources('AWS::DynamoDB::Table')) - .find(id => id.startsWith('TaskApprovalsTable')); - const statements = Object.values(template.findResources('AWS::IAM::Policy')) - .flatMap(policy => policy.Properties.PolicyDocument.Statement); - const grants = statements.filter(statement => - JSON.stringify(statement.Resource).includes(approvalId!), - ); - expect(grants).toEqual([expect.objectContaining({ - Effect: 'Allow', - Action: ['dynamodb:GetItem', 'dynamodb:BatchGetItem', 'dynamodb:Query', 'dynamodb:ConditionCheckItem'], - })]); + test('rejects approval wiring without task-scoped credentials before synthesis', () => { + expect(() => createStack({ withApprovals: true })).toThrow('ECS approvals require agentSessionRole and approvalRequestsApiUrl'); }); test('rejects a build setting that would erase approval-table wiring', () => { diff --git a/cdk/test/handlers/request-approval.test.ts b/cdk/test/handlers/request-approval.test.ts index b33cde8dd..80d682958 100644 --- a/cdk/test/handlers/request-approval.test.ts +++ b/cdk/test/handlers/request-approval.test.ts @@ -130,3 +130,10 @@ test('reports service failures as unavailable with a request ID, not invalid inp error: { code: 'ProvisionedThroughputExceededException', request_id: 'api-request' }, }); }); + +test.each(['*', 'task/*', '../task', 'task?x', 'task#lease', 'task\n'])('rejects unsafe task/request/worker identifiers %j before database access', async id => { + for (const key of ['task_id', 'request_id', 'worker_attempt_id']) { + expect(await recordWorkerRequest({ ...input, [key]: id })).toEqual({ ok: false, code: 'APPROVAL_REQUEST_INVALID' }); + } + expect(send).not.toHaveBeenCalled(); +}); diff --git a/cdk/test/handlers/shared/strategies/lambda-microvm-strategy.test.ts b/cdk/test/handlers/shared/strategies/lambda-microvm-strategy.test.ts index ad977417f..27d2b62ec 100644 --- a/cdk/test/handlers/shared/strategies/lambda-microvm-strategy.test.ts +++ b/cdk/test/handlers/shared/strategies/lambda-microvm-strategy.test.ts @@ -32,11 +32,9 @@ const MICROVM_ID = 'mvm-0123456789abcdef'; const ENDPOINT = 'https://mvm-0123456789abcdef.microvm.lambda.us-east-1.amazonaws.com'; // --- platform_config (ADR-021 P2) --- -// The FOUR required identifiers, and only those, are set for the main describes, -// so the default `platform_config` block is small and its exact serialized size is -// known — which the 4 KB boundary probes below depend on. The nine optional keys -// get their own describe (and are deleted here so a leaked env var from another -// suite cannot silently change the envelope's byte length). +// Set only the required platform fields for the main cases. Optional fields +// are tested separately so ambient environment cannot change payload size. +const APPROVAL_REQUESTS_API_URL = 'https://approval.execute-api.us-east-1.amazonaws.com/v1'; const TASK_TABLE_NAME = 'abca-task-table'; const TASK_EVENTS_TABLE_NAME = 'abca-task-events-table'; const GITHUB_TOKEN_SECRET_ARN = @@ -58,6 +56,7 @@ process.env.MICROVM_PAYLOAD_BUCKET = PAYLOAD_BUCKET; process.env.AWS_REGION = 'us-east-1'; delete process.env.MICROVM_INGRESS_CONNECTOR_ARNS; +process.env.APPROVAL_REQUESTS_API_URL = APPROVAL_REQUESTS_API_URL; process.env.TASK_TABLE_NAME = TASK_TABLE_NAME; process.env.TASK_EVENTS_TABLE_NAME = TASK_EVENTS_TABLE_NAME; process.env.GITHUB_TOKEN_SECRET_ARN = GITHUB_TOKEN_SECRET_ARN; @@ -145,6 +144,7 @@ const BLUEPRINT: BlueprintConfig = { compute_type: 'lambda-microvm', runtime_arn const EXPECTED_PLATFORM_CONFIG = { task_table_name: TASK_TABLE_NAME, task_events_table_name: TASK_EVENTS_TABLE_NAME, + approval_requests_api_url: APPROVAL_REQUESTS_API_URL, github_token_secret_arn: GITHUB_TOKEN_SECRET_ARN, agent_session_role_arn: AGENT_SESSION_ROLE_ARN, }; @@ -1190,10 +1190,11 @@ describe('buildMicrovmPlatformConfig — the MicroVM substitute for a deploy-tim } }); - test('pins the REQUIRED subset — these four are what a task cannot start without', () => { + test('pins the required platform fields including the trusted approval service', () => { expect([...MICROVM_PLATFORM_CONFIG_REQUIRED_KEYS]).toEqual([ 'task_table_name', 'task_events_table_name', + 'approval_requests_api_url', 'github_token_secret_arn', 'agent_session_role_arn', ]); @@ -1226,6 +1227,7 @@ describe('buildMicrovmPlatformConfig — the MicroVM substitute for a deploy-tim const config = buildMicrovmPlatformConfig({ TASK_TABLE_NAME: 'tasks', TASK_EVENTS_TABLE_NAME: 'events', + APPROVAL_REQUESTS_API_URL, GITHUB_TOKEN_SECRET_ARN: 'arn:aws:secretsmanager:us-east-1:123456789012:secret:gh-AbCdEf', AGENT_SESSION_ROLE_ARN: 'arn:aws:iam::123456789012:role/SessionRole', }); @@ -1235,6 +1237,7 @@ describe('buildMicrovmPlatformConfig — the MicroVM substitute for a deploy-tim expect(Object.keys(config)).toEqual([ 'task_table_name', 'task_events_table_name', + 'approval_requests_api_url', 'github_token_secret_arn', 'agent_session_role_arn', ]); @@ -1269,6 +1272,7 @@ describe('buildMicrovmPlatformConfig — the MicroVM substitute for a deploy-tim }); test.each([ + ['APPROVAL_REQUESTS_API_URL', 'approval_requests_api_url'], ['TASK_TABLE_NAME', 'task_table_name'], ['TASK_EVENTS_TABLE_NAME', 'task_events_table_name'], ['GITHUB_TOKEN_SECRET_ARN', 'github_token_secret_arn'], @@ -1286,7 +1290,7 @@ describe('buildMicrovmPlatformConfig — the MicroVM substitute for a deploy-tim test('names EVERY missing required key at once, not just the first', () => { // One redeploy should fix all of them; reporting one per attempt turns a - // misconfiguration into four round-trips. + // misconfiguration into repeated round-trips. expect(() => buildMicrovmPlatformConfig({})).toThrow( /task_table_name.*task_events_table_name.*github_token_secret_arn.*agent_session_role_arn/s, ); @@ -1351,7 +1355,7 @@ describe('buildMicrovmPlatformConfig — the MicroVM substitute for a deploy-tim test('does NOT throw for a missing OPTIONAL key', () => { const env = { ...FULL_ENV }; for (const optional of [ - 'TASK_APPROVALS_TABLE_NAME', 'APPROVAL_REQUESTS_API_URL', 'NUDGES_TABLE_NAME', 'LOG_GROUP_NAME', + 'TASK_APPROVALS_TABLE_NAME', 'NUDGES_TABLE_NAME', 'LOG_GROUP_NAME', 'ARTIFACTS_BUCKET_NAME', 'TRACE_ARTIFACTS_BUCKET_NAME', 'LINEAR_OAUTH_SECRET_ARN', 'LINEAR_VAULT_ENABLED', 'LINEAR_WORKLOAD_IDENTITY_NAME', 'CONTINUATION_BUCKET_NAME', diff --git a/contracts/constants.json b/contracts/constants.json index f9c4d254f..f6ce84503 100644 --- a/contracts/constants.json +++ b/contracts/constants.json @@ -60,6 +60,7 @@ "required": [ "task_table_name", "task_events_table_name", + "approval_requests_api_url", "github_token_secret_arn", "agent_session_role_arn" ], From 22d8bae90894fce0943b090c53461a9646062d2a Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Mon, 21 Sep 2026 09:21:30 -0400 Subject: [PATCH 114/149] fix(645): classify checkpoint failures and exercise SDK recovery in CI --- agent/src/continuation_session.py | 27 ++++++++++++++++++---- agent/src/continuation_storage.py | 3 +-- agent/src/continuation_usage.py | 5 ++-- agent/src/continuation_workspace.py | 3 +-- agent/tests/test_continuation_sdk_probe.py | 4 ++-- agent/tests/test_continuation_session.py | 11 ++++++++- agent/tests/test_continuation_usage.py | 3 ++- 7 files changed, 41 insertions(+), 15 deletions(-) diff --git a/agent/src/continuation_session.py b/agent/src/continuation_session.py index c7d3ebf45..16e6f3ae0 100644 --- a/agent/src/continuation_session.py +++ b/agent/src/continuation_session.py @@ -40,12 +40,18 @@ class ContinuationCheckpointError(RuntimeError): """A checkpoint cannot be acknowledged or safely restored.""" + def __init__(self, message: str, *, code: str = "checkpoint_invalid") -> None: + super().__init__(message) + self.code = code + def _encode(value: Any) -> bytes: try: return json.dumps(value, sort_keys=True, separators=(",", ":"), allow_nan=False).encode() except (TypeError, ValueError, RecursionError) as exc: - raise ContinuationCheckpointError("Checkpoint contains invalid JSON data") from exc + raise ContinuationCheckpointError( + "Checkpoint contains invalid JSON data", code="checkpoint_invalid_json" + ) from exc def _copy(value: Any) -> Any: @@ -74,6 +80,12 @@ def _action_hash(tool_input: dict) -> str: @dataclass(frozen=True) class CheckpointIdentity: + """Version-1 wire identity: attempt_id is the source physical MicroVM ID. + + The coordinator's continuation.attempt_id is a separate logical launch token. + Renaming this serialized field requires a checkpoint-version migration. + """ + task_id: str attempt_id: str request_id: str @@ -169,7 +181,9 @@ def decode_checkpoint(body: bytes, identity: CheckpointIdentity) -> dict: try: envelope = json.loads(body) except (ValueError, UnicodeError, RecursionError) as exc: - raise ContinuationCheckpointError("Checkpoint is not valid JSON") from exc + raise ContinuationCheckpointError( + "Checkpoint is not valid JSON", code="checkpoint_invalid_json" + ) from exc if ( not isinstance(envelope, dict) or set(envelope) @@ -323,7 +337,7 @@ async def checkpoint_pending( await self._changed.wait() except TimeoutError as exc: raise ContinuationCheckpointError( - "SDK mirror did not acknowledge the pending action" + "SDK mirror did not acknowledge the pending action", code="checkpoint_sdk_timeout" ) from exc @classmethod @@ -422,7 +436,8 @@ def save(self, body: bytes, identity: CheckpointIdentity) -> CheckpointReceipt: except Exception as exc: cause = write_error if write_error is not None else exc raise ContinuationCheckpointError( - "Checkpoint could not be verified; keep the current worker available" + "Checkpoint could not be verified; keep the current worker available", + code="checkpoint_storage_unverified", ) from cause receipt = CheckpointReceipt(key, version, digest, len(body)) receipt.validate(identity) @@ -433,7 +448,9 @@ def load(self, receipt: CheckpointReceipt, identity: CheckpointIdentity) -> byte try: body, _ = self._read(receipt.key, version_id=receipt.version_id) except Exception as exc: - raise ContinuationCheckpointError("Saved checkpoint could not be read") from exc + raise ContinuationCheckpointError( + "Saved checkpoint could not be read", code="checkpoint_storage_unavailable" + ) from exc if len(body) != receipt.size_bytes or hashlib.sha256(body).hexdigest() != receipt.sha256: raise ContinuationCheckpointError("Saved checkpoint does not match its receipt") decode_checkpoint(body, identity) diff --git a/agent/src/continuation_storage.py b/agent/src/continuation_storage.py index 0cc40ad43..b4bc4dc2e 100644 --- a/agent/src/continuation_storage.py +++ b/agent/src/continuation_storage.py @@ -46,8 +46,7 @@ class ContinuationStorageError(ContinuationCheckpointError): """Content-free failure classification for lifecycle diagnostics.""" def __init__(self, code: str, message: str) -> None: - self.code = code - super().__init__(message) + super().__init__(message, code=code) @dataclass(frozen=True) diff --git a/agent/src/continuation_usage.py b/agent/src/continuation_usage.py index a43a5eced..242290876 100644 --- a/agent/src/continuation_usage.py +++ b/agent/src/continuation_usage.py @@ -46,14 +46,15 @@ async def read_usage(client: Any) -> UsageSnapshot: SDK 0.2.110 bundles CLI 2.1.191, whose experimental ``get_usage`` control request exposes exact session dollars and per-model token counters. Python has no public wrapper yet. Keep this dependency isolated and covered by the - opt-in real SDK probe; an upgrade must verify it before publishing checkpoints. + real SDK probe in CI; an upgrade must verify it before publishing checkpoints. """ if ( importlib.metadata.version("claude-agent-sdk") != SHARED_CONSTANTS["microvm_continuation"]["verified_sdk_version"] ): raise ContinuationCheckpointError( - "Continuation accounting requires the verified SDK version" + "Continuation accounting requires the verified SDK version", + code="checkpoint_sdk_unverified", ) query = getattr(client, "_query", None) send = getattr(query, "_send_control_request", None) diff --git a/agent/src/continuation_workspace.py b/agent/src/continuation_workspace.py index a15767a31..d86614e5f 100644 --- a/agent/src/continuation_workspace.py +++ b/agent/src/continuation_workspace.py @@ -50,8 +50,7 @@ class WorkspaceCheckpointError(ContinuationCheckpointError): """A content-free stage code for lifecycle feedback.""" def __init__(self, code: str, message: str) -> None: - self.code = code - super().__init__(message) + super().__init__(message, code=code) @dataclass(frozen=True) diff --git a/agent/tests/test_continuation_sdk_probe.py b/agent/tests/test_continuation_sdk_probe.py index 5e58c74b3..a09b53f1b 100644 --- a/agent/tests/test_continuation_sdk_probe.py +++ b/agent/tests/test_continuation_sdk_probe.py @@ -204,8 +204,8 @@ async def post(data, tool_id, context): @pytest.mark.skipif( - os.environ.get("ABCA_TEST_SDK_CONTINUATION") != "1", - reason="Opt-in pinned SDK/CLI subprocess diagnostic", + os.environ.get("ABCA_TEST_SDK_CONTINUATION") != "1" and os.environ.get("CI") != "true", + reason="Pinned SDK/CLI subprocess diagnostic runs in CI or by local opt-in", ) @pytest.mark.parametrize("decision", ["approve", "deny"]) def test_sdk_resumes_from_checkpoint_without_original_config(tmp_path, decision): diff --git a/agent/tests/test_continuation_session.py b/agent/tests/test_continuation_session.py index 09f758dc0..5da28d722 100644 --- a/agent/tests/test_continuation_session.py +++ b/agent/tests/test_continuation_session.py @@ -315,8 +315,11 @@ def test_uncommitted_write_failure_never_returns_a_receipt(self, storage, body): def test_unverifiable_storage_never_acknowledges(self, storage, body, failure): store, client = storage setattr(client, failure, failure != "versioned") - with pytest.raises(checkpoint.ContinuationCheckpointError, match="could not be verified"): + with pytest.raises( + checkpoint.ContinuationCheckpointError, match="could not be verified" + ) as error: store.save(body, IDENTITY) + assert error.value.code == "checkpoint_storage_unverified" assert all(stream.closed for stream in client.streams) def test_load_keeps_the_original_version_even_if_current_key_changes(self, storage, body): @@ -377,3 +380,9 @@ def test_invalid_json_fails(self, body): def test_path_components_cannot_escape_task_prefix(self, value): with pytest.raises(checkpoint.ContinuationCheckpointError): replace(IDENTITY, task_id=value) + + +def test_invalid_json_reports_a_specific_checkpoint_code(): + with pytest.raises(checkpoint.ContinuationCheckpointError) as error: + checkpoint._encode({"not_json": object()}) + assert error.value.code == "checkpoint_invalid_json" diff --git a/agent/tests/test_continuation_usage.py b/agent/tests/test_continuation_usage.py index 1c49a3f03..2302c2045 100644 --- a/agent/tests/test_continuation_usage.py +++ b/agent/tests/test_continuation_usage.py @@ -76,5 +76,6 @@ def test_incomplete_or_invalid_accounting_cannot_publish_a_checkpoint(mutation): def test_unverified_sdk_upgrade_requires_explicit_accounting_validation(monkeypatch): monkeypatch.setattr("continuation_usage.importlib.metadata.version", lambda _: "0.3.0") - with pytest.raises(ContinuationCheckpointError, match="verified SDK"): + with pytest.raises(ContinuationCheckpointError, match="verified SDK") as error: asyncio.run(read_usage(None)) + assert error.value.code == "checkpoint_sdk_unverified" From 02ad141ea193a333912f0345900d9494e5003e26 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Mon, 21 Sep 2026 09:21:45 -0400 Subject: [PATCH 115/149] fix(645): tolerate transport line endings in Linear comment recovery --- cdk/src/handlers/shared/linear-feedback.ts | 6 ++++-- cdk/test/handlers/shared/linear-feedback.test.ts | 13 +++++++++++++ 2 files changed, 17 insertions(+), 2 deletions(-) diff --git a/cdk/src/handlers/shared/linear-feedback.ts b/cdk/src/handlers/shared/linear-feedback.ts index 442631251..6b88cc06c 100644 --- a/cdk/src/handlers/shared/linear-feedback.ts +++ b/cdk/src/handlers/shared/linear-feedback.ts @@ -461,7 +461,7 @@ export async function postIdentifiedComment( return { ok: true }; } // A successful write can lose its response. A duplicate ID is acceptable only - // when the saved comment exactly matches this destination and content. + // when destination and content match, ignoring transport line endings only. const existing = await graphqlData(token, ` query ApprovalComment($id: String!) { comment(id: $id) { body issue { id } parent { id } } @@ -470,7 +470,9 @@ export async function postIdentifiedComment( const comment = existing.value.comment as { body?: string; issue?: { id?: string }; parent?: { id?: string }; } | undefined; - return comment?.body === input.body && comment.issue?.id === input.issueId + const normalizeBody = (body: string) => body.replace(/\r\n?/g, '\n').replace(/\n+$/, ''); + return typeof comment?.body === 'string' && normalizeBody(comment.body) === normalizeBody(input.body) + && comment.issue?.id === input.issueId && comment.parent?.id === input.parentId ? { ok: true } : { ok: false, retryable: false }; } diff --git a/cdk/test/handlers/shared/linear-feedback.test.ts b/cdk/test/handlers/shared/linear-feedback.test.ts index bf2e7bffd..d831c110e 100644 --- a/cdk/test/handlers/shared/linear-feedback.test.ts +++ b/cdk/test/handlers/shared/linear-feedback.test.ts @@ -146,6 +146,19 @@ describe('linear-feedback', () => { })); expect(await postIdentifiedComment(CTX, input)).toEqual({ ok: true }); }); + test.each([ + ['Approval needed\r\nReply here\r\n', true], + ['Approval needed\nApprove a DIFFERENT action', false], + ])('compares replay content conservatively: %j', async (body, accepted) => { + fetchMock.mockRejectedValueOnce(new Error('lost response')); + fetchMock.mockResolvedValueOnce(jsonResponse({ + data: { + comment: { body, issue: { id: ISSUE_ID }, parent: { id: 'root' } }, + }, + })); + expect(await postIdentifiedComment(CTX, { ...input, body: 'Approval needed\nReply here' })) + .toEqual(accepted ? { ok: true } : { ok: false, retryable: false }); + }); test('does not accept an ID collision in another issue or thread', async () => { fetchMock.mockResolvedValueOnce(jsonResponse({ errors: ['already exists'] })); fetchMock.mockResolvedValueOnce(jsonResponse({ From 26682580016ea4ce03b3fdec420ebc5c8a2362e5 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Mon, 21 Sep 2026 09:22:09 -0400 Subject: [PATCH 116/149] docs(645): clarify approval trust boundaries and verify rendered links --- CONTRIBUTING.md | 13 +++ agent/AGENTS.md | 1 + agent/src/hooks.py | 8 +- cdk/AGENTS.md | 3 +- .../shared/microvm-continuation-types.ts | 1 + cdk/src/migration/template.ts | 1 + docs/AGENTS.md | 2 + .../ADR-004-tabula-rasa-documentation.md | 2 +- .../ADR-023-trusted-approval-writer.md | 68 +++++++++++++ docs/design/API_CONTRACT.md | 2 +- docs/design/CEDAR_HITL_GATES.md | 22 +++-- docs/design/SECURITY.md | 4 +- docs/guides/DEPLOYMENT_GUIDE.md | 3 + docs/mise.toml | 6 +- docs/package.json | 2 +- docs/scripts/check-site-links.mjs | 40 ++++++++ docs/scripts/link-check.sh | 2 +- docs/scripts/sync-starlight.mjs | 59 +++++++---- .../content/docs/architecture/Api-contract.md | 2 +- .../docs/architecture/Cedar-hitl-gates.md | 30 +++--- docs/src/content/docs/architecture/Compute.md | 6 +- .../docs/architecture/Deployment-roles.md | 2 +- .../docs/architecture/Identity-and-auth.md | 6 +- .../content/docs/architecture/Orchestrator.md | 2 +- .../docs/architecture/Repo-onboarding.md | 6 +- .../src/content/docs/architecture/Security.md | 10 +- docs/src/content/docs/architecture/Vision.md | 8 +- .../content/docs/architecture/Workflows.md | 20 ++-- .../docs/customizing/Cedar-policies.md | 12 +-- ...-002-least-privilege-bootstrap-policies.md | 2 +- .../Adr-004-tabula-rasa-documentation.md | 2 +- .../Adr-012-operational-knowledge-stack.md | 2 +- .../Adr-014-workflow-driven-tasks.md | 4 +- .../Adr-016-pluggable-identity-and-auth.md | 4 +- ...r-019-agentcore-gateway-tool-federation.md | 18 ++-- ...Adr-021-lambda-microvms-compute-backend.md | 12 +-- .../decisions/Adr-022-agent-asset-registry.md | 18 ++-- .../Adr-023-trusted-approval-writer.md | 72 ++++++++++++++ .../docs/developer-guide/Contributing.md | 19 +++- .../docs/developer-guide/Installation.md | 4 +- .../docs/developer-guide/Introduction.md | 2 +- .../developer-guide/Repository-preparation.md | 6 +- .../developer-guide/Where-to-make-changes.md | 2 +- .../docs/getting-started/Deployment-guide.md | 13 ++- .../docs/getting-started/Quick-start.mdx | 4 +- .../docs/using/Approval-gates-cedar-hitl.md | 2 +- .../content/docs/using/Jira-setup-guide.md | 2 +- .../content/docs/using/Linear-setup-guide.md | 2 +- docs/src/content/docs/using/Roles.md | 4 +- docs/src/content/docs/using/Task-lifecycle.md | 2 +- docs/src/content/docs/using/Using-the-cli.md | 2 +- .../645-p3-lifecycle-diagnostics.md | 77 +++++++++++++++ .../docs/verification/645-p3-nested-stack.md | 71 +++++++++++++ .../verification/645-payload-bootstrap.md | 83 ++++++++++++++++ docs/src/content/docs/verification/Readme.md | 99 +++++++++++++++++++ 55 files changed, 730 insertions(+), 141 deletions(-) create mode 100644 docs/decisions/ADR-023-trusted-approval-writer.md create mode 100644 docs/scripts/check-site-links.mjs create mode 100644 docs/src/content/docs/decisions/Adr-023-trusted-approval-writer.md create mode 100644 docs/src/content/docs/verification/645-p3-lifecycle-diagnostics.md create mode 100644 docs/src/content/docs/verification/645-p3-nested-stack.md create mode 100644 docs/src/content/docs/verification/645-payload-bootstrap.md create mode 100644 docs/src/content/docs/verification/Readme.md diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 672bdc1a9..8d2bffe1c 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -98,6 +98,19 @@ PRs labeled `auto-approve` are approved automatically by the `auto-approve` work If `prek install` fails with "refusing to install hooks with `core.hooksPath` set", another tool owns your hooks. Either unset it (`git config --unset-all core.hooksPath`) or integrate these checks into your hook manager. +The build's transaction tests use DynamoDB Local through +`ABCA_DDB_LOCAL_ENDPOINT` (a loopback HTTP endpoint). CI starts a Docker service +pinned by image digest and supplies its mapped port to both CDK and Python tests; +those tests fail if `CI=true` without an endpoint. To reproduce locally, start +DynamoDB Local with `-inMemory -sharedDb`, bind port 8000 to loopback, and run +`ABCA_DDB_LOCAL_ENDPOINT=http://127.0.0.1:8000 mise run build`. + +The pinned SDK continuation probe also runs in CI. Locally enable it with +`ABCA_TEST_SDK_CONTINUATION=1` when running +`agent/tests/test_continuation_sdk_probe.py`. It uses a local simulated Bedrock +endpoint and synthetic credentials; it does not require an AWS account or model +billing. + ## Versioning The project uses semantic versioning based on [Conventional Commits](https://www.conventionalcommits.org/en/v1.0.0/): diff --git a/agent/AGENTS.md b/agent/AGENTS.md index 67fc8e20f..2f24f83c4 100644 --- a/agent/AGENTS.md +++ b/agent/AGENTS.md @@ -28,6 +28,7 @@ Root `mise run build` includes `//agent:quality` in parallel with `//cdk:build`. | `hooks.py`, `policy.py` | `agent/tests/test_hooks.py`, `test_policy.py` | | `pipeline.py`, `runner.py` | `agent/tests/test_pipeline.py`, etc. | | `microvm_lifecycle.py`, `microvm_checkpoint.py` | `test_microvm_lifecycle.py`, `test_microvm_checkpoint.py`; checkpoint transaction tests require `ABCA_DDB_LOCAL_ENDPOINT` in CI | +| `approval_requests.py`, `task_state.py` | `test_approval_requests.py`, `test_task_state.py`; SigV4 writer protocol and persistence outcomes | | `continuation_*.py` | Matching `test_continuation_*.py`: capture, storage, safe restore, SDK session and usage recovery | Use `@pytest.fixture(autouse=True)` to reset shared module state between tests when handlers use circuit breakers or caches. diff --git a/agent/src/hooks.py b/agent/src/hooks.py index 5722af3f8..9039bfcff 100644 --- a/agent/src/hooks.py +++ b/agent/src/hooks.py @@ -2046,10 +2046,10 @@ async def _stop( # Empty dict == allow stop. SyncHookJSONOutput(**{}) is fine. return SyncHookJSONOutput(**result) - # The callback transport must outlive the approval loop. A frozen MicroVM - # may wake after the gate deadline during supervisor recovery, so keep that - # callback alive for the bounded VM lifetime. Explicit gate deadlines remain - # unchanged; a retained request can outlive this callback through continuation. + # All backends share this bounded SDK callback window (eight hours plus + # cleanup). It is separate from the human decision deadline and also covers + # MicroVM freeze/recovery. A retained request can outlive a callback only + # through a verified continuation; this does not extend compute lifetime. callback_window_s = SHARED_CONSTANTS["microvm_lifecycle"]["maximum_duration_seconds"] matchers = { "PreToolUse": [ diff --git a/cdk/AGENTS.md b/cdk/AGENTS.md index 11c0c1f1c..44ddf5899 100644 --- a/cdk/AGENTS.md +++ b/cdk/AGENTS.md @@ -27,9 +27,10 @@ mise //cdk:destroy # destroy stack | Code | Test location | |------|---------------| | Shared handler logic | `cdk/test/handlers/shared/*.test.ts` | +| Trusted approval writer | `src/handlers/request-approval.ts`, `src/constructs/approval-request-service.ts`; handler/unit, DynamoDB Local and construct tests | | Handler entrypoints | `cdk/test/handlers/orchestrate-task.test.ts`, `create-task.test.ts`, `webhook-create-task.test.ts` | | Constructs | `cdk/test/constructs/task-orchestrator.test.ts`, `task-api.test.ts` | -| MicroVM lifecycle/capacity transactions | `test/handlers/shared/*-local.test.ts`; use a loopback DynamoDB Local endpoint in `ABCA_DDB_LOCAL_ENDPOINT` (mandatory in CI) | +| MicroVM lifecycle/capacity transactions | `test/handlers/shared/*-local.test.ts` and `test/handlers/request-approval-local.test.ts`; use a loopback DynamoDB Local endpoint in `ABCA_DDB_LOCAL_ENDPOINT` (mandatory in CI) | | Staged flat-to-nested migration helpers | `src/migration/`, `test/migration/`; these are not a deployable migration command | | Live verification harnesses | `test/live/`; explicitly invoked against an owned fixture, excluded from normal Jest collection | diff --git a/cdk/src/handlers/shared/microvm-continuation-types.ts b/cdk/src/handlers/shared/microvm-continuation-types.ts index 30426761b..07da23233 100644 --- a/cdk/src/handlers/shared/microvm-continuation-types.ts +++ b/cdk/src/handlers/shared/microvm-continuation-types.ts @@ -42,6 +42,7 @@ export interface ContinuationReceipt { export interface ContinuationRecord { readonly version: number; + /** PARKED means the source worker has retired; the guest's local parked phase only means a safe approval wait. */ readonly state: 'READY' | 'FENCED' | 'PARKED' | 'STARTING' | 'RESTORING' | 'CONSUMED'; readonly identity: ContinuationIdentity; readonly manifest: ContinuationReceipt; diff --git a/cdk/src/migration/template.ts b/cdk/src/migration/template.ts index 1ec63bd31..4c28cd0cf 100644 --- a/cdk/src/migration/template.ts +++ b/cdk/src/migration/template.ts @@ -36,6 +36,7 @@ export interface MigrationTemplate { [key: string]: any; } +/** Template digests use code-unit key order; runtime receipts use localeCompare. Do not interchange persisted hashes. */ export function canonical(value: unknown): string { const sort = (item: any): any => { if (Array.isArray(item)) return item.map(sort); diff --git a/docs/AGENTS.md b/docs/AGENTS.md index 896e0b166..3af7434b4 100644 --- a/docs/AGENTS.md +++ b/docs/AGENTS.md @@ -17,6 +17,7 @@ Pre-commit hook `docs-sync` runs sync automatically when prek hooks are installe ## Testing - **Sync + build:** `mise //docs:build` (required before PR if you touched guides, design, ADRs, or `CONTRIBUTING.md`) +- **Rendered links:** `mise //docs:build` checks all internal page and anchor links in `dist/`; external URLs are not checked offline. - **Sync only:** `mise //docs:sync` then `git diff docs/src/content/docs/` — commit mirror changes alongside sources - **Astro check:** `mise //docs:check` (or `cd docs && npm run docs:check`) @@ -29,6 +30,7 @@ CI **"Fail build on mutation"** rejects PRs where committed Starlight mirrors do | `docs/guides/` | WRITE | User and developer guides | | `docs/design/` | WRITE | Architecture and design docs | | `docs/decisions/` | WRITE | ADRs | +| `docs/verification/` | WRITE | Operator runbooks mirrored to `/verification/` | | `docs/imgs/` | WRITE | Static images | | `CONTRIBUTING.md` (repo root) | WRITE | Mirrored to Starlight | | `docs/src/content/docs/` | READ only | Generated — never edit by hand | diff --git a/docs/decisions/ADR-004-tabula-rasa-documentation.md b/docs/decisions/ADR-004-tabula-rasa-documentation.md index 579a476f3..8a6689448 100644 --- a/docs/decisions/ADR-004-tabula-rasa-documentation.md +++ b/docs/decisions/ADR-004-tabula-rasa-documentation.md @@ -44,7 +44,7 @@ Never force a novice to read expert material to proceed. Never force an expert t ### Self-contained references When referencing another document: -- State what the reader gets from it: "See [Deployment Guide](link) for AWS account setup (required before this step)" +- State what the reader gets from it: "See [Deployment Guide](../guides/DEPLOYMENT_GUIDE.md) for AWS account setup (required before this step)" - Never assume the reader has read it - Never use "as mentioned above" — each section must stand alone after context compaction diff --git a/docs/decisions/ADR-023-trusted-approval-writer.md b/docs/decisions/ADR-023-trusted-approval-writer.md new file mode 100644 index 000000000..73daa07e4 --- /dev/null +++ b/docs/decisions/ADR-023-trusted-approval-writer.md @@ -0,0 +1,68 @@ +# ADR-023: Trusted approval writer and retained human decisions + +**Status:** proposed +**Date:** 2026-09-21 + +## Context + +Workers previously had direct write access to approval records. Restricting a +worker to its own task did not prevent it from replacing a pending request with +an approved record. MicroVM continuation makes the distinction between worker +authority and human consent especially important, but the same defect affects +ECS and AgentCore. + +A short approval deadline also couples human response time to compute lifetime. +Someone taking time to consider a request should not lose it merely because the +worker should stop consuming resources. + +## Decision + +Use a small IAM-authenticated API and Lambda as the worker-facing approval +writer. Its interface creates a pending request or closes one after a worker +timeout/failure. It cannot record human approval or denial, notification markers, +or retention TTLs. Human decisions remain in the owner-authenticated decision +handlers. Workers retain approval reads and transaction condition checks. + +The session role can invoke only the task path named by its `task_id` tag. +Creation atomically writes the request and transitions its task from running to +awaiting approval. The transaction checks the task owner/state and, for MicroVM, +the coordinator-owned worker lease. A stale worker cannot use its old lease to +create or close a request. + +Unanswered requests have no decision deadline by default. Explicit positive +timeouts remain available. Task cancellation, terminal failure and invalidated +worker execution can close requests independently. Compute lifetime is separate: +MicroVM can checkpoint and release a worker while retaining a request; ECS and +AgentCore currently cannot restore that waiting execution into a replacement. +Their task execution limits still apply. + +This implementation is included in the P3 review because retaining approvals +without protecting their decision records would preserve the security defect. +The ADR remains proposed for maintainer review. + +## Consequences + +- Existing deployments must pause submissions, drain old workers and deploy + matching infrastructure and images. Old images still attempt direct writes, + which the new IAM policy denies. See the + [upgrade procedure](../guides/DEPLOYMENT_GUIDE.md#upgrading-approval-permissions). +- The service adds one signed request per creation/closure and becomes an + availability dependency. Failure denies permission to proceed; it never + becomes human consent. +- The service validates record shape, not the truth of worker-supplied policy + descriptions. A preview can be truncated and is not a proof of the full tool + input. `TIMED_OUT` currently also represents a worker polling failure; it is + not proof that a human deadline elapsed. Human `DENIED` remains distinct. +- The compute role still chooses session tags when assuming the session role. + This API protects the decision-writing boundary but does not provide full + isolation from a compromised worker retaining ambient compute credentials. +- Linear consent is read back from Linear and must come from the mapped owner, + excluding bots and the saved OAuth token's own identity. Other webhook paths + still need a separate review of the legacy shared OAuth/signing-secret bundle. + +## References + +- [MicroVM backend, issue #645](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/645) +- [Implementation and security review, PR #904](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/904) +- [Approval trust boundaries](../design/CEDAR_HITL_GATES.md#121-trust-boundaries) +- [ADR-021: Lambda MicroVM compute backend](./ADR-021-lambda-microvms-compute-backend.md) diff --git a/docs/design/API_CONTRACT.md b/docs/design/API_CONTRACT.md index a62e30bd3..fc26f9485 100644 --- a/docs/design/API_CONTRACT.md +++ b/docs/design/API_CONTRACT.md @@ -254,7 +254,7 @@ Returns full details of a task. Users can only access their own tasks. } ``` -`agent_heartbeat_at` is the agent's last in-guest liveness beat, or `null`. The agent writes it every 45 s on the `agentcore` and `lambda-microvm` backends; on `ecs` it is written once at start, because that backend runs the pipeline directly instead of serving HTTP. The orchestrator reads the same field to detect a hung agent inside a healthy compute environment ([ORCHESTRATOR.md](./ORCHESTRATOR.md#dynamodb-heartbeat-agentcore-and-lambda-microvms)), so a value that is minutes old on a `RUNNING` task is the signal, not the timestamp itself. `null` on records written before the field existed. +`agent_heartbeat_at` is the agent's last in-guest liveness beat, or `null`. The agent writes it every 45 s on the `agentcore` and `lambda-microvm` backends; on `ecs` it is written once at start, because that backend runs the pipeline directly instead of serving HTTP. The orchestrator reads the same field to detect a hung agent inside a healthy compute environment ([ORCHESTRATOR.md](./ORCHESTRATOR.md#liveness-monitoring)), so a value that is minutes old on a `RUNNING` task is the signal, not the timestamp itself. `null` on records written before the field existed. `error_classification` is a derived field computed at response time from `error_message`. When `error_message` is `null`, `error_classification` is `null`. When present, it contains: diff --git a/docs/design/CEDAR_HITL_GATES.md b/docs/design/CEDAR_HITL_GATES.md index ba78fee78..007c6cb48 100644 --- a/docs/design/CEDAR_HITL_GATES.md +++ b/docs/design/CEDAR_HITL_GATES.md @@ -1363,7 +1363,7 @@ t=45m: Task #1 completes. count → 9. Bob can submit task #11. AgentCore Runtime's `maxLifetime = 28800s` (8h) is an absolute timer from session start. It does NOT pause during `AWAITING_APPROVAL`. -An explicitly timed approval is bounded by the remaining worker lifetime minus the 120-second cleanup margin. Below the 30-second floor, the hook denies the action with reason `"insufficient lifetime"`. +When the worker supplies a remaining-lifetime estimate, the hook refuses to open a new gate if fewer than 30 seconds remain after the 120-second cleanup margin. This check applies to timed and untimed approvals and reports `"insufficient maxLifetime remaining (s) for approval"`. A positive approval timeout is also capped by that remaining budget. The default approval timeout is `0`: no decision deadline. Worker lifetime does not turn silence into a human denial. ECS and AgentCore do not currently restore a waiting agent into a replacement worker; when their task execution limit is reached, the task closes and its pending approval is cancelled. MicroVM can retain a verified checkpoint and continue on a replacement worker. Setting its sleep delay to `0` disables early sleep/retirement, but the coordinator still attempts retirement before the service lifetime ends. This option trades idle cost for faster replies. @@ -1694,18 +1694,26 @@ These alarms transition to `ALARM` state in CloudWatch and appear in the console `POST /v1/tasks/{task_id}` to the session's `task_id` tag. The service creates `PENDING` rows or conditionally records a non-human `TIMED_OUT`; it rejects human decisions, notification markers, retention TTL and other extra fields. - It checks current task ownership/state and, for MicroVM, the active worker - lease in the same transaction. Worker-provided action descriptions remain - untrusted. This protects approval records; it does not sandbox code running + IAM binds the caller to a task path; the transaction prevents concurrent task + ownership/state changes and, for MicroVM, checks the active worker lease. + The tool preview, its hash, severity, reason and matching rules are worker + assertions, not independently evaluated policy results. The preview can be + truncated, so its bytes cannot be used to verify the full-input hash. This protects approval records; it does not sandbox code running inside the agent or bind ambient compute credentials to one task. - **User CLI ↔ API Gateway**: Cognito JWT (same authorizer as `/tasks/*`). Cognito `sub` is the canonical caller identity, used **verbatim** in DDB `ConditionExpression` (§7.1, finding #6). - **ApproveTaskFn/DenyTaskFn ↔ TaskApprovalsTable + TaskTable**: Lambda IAM policy allows `UpdateItem` on both tables under `TransactWriteItems`. Authorization is in the ConditionExpression (ownership AND state), not in a separate IAM boundary. - **Blueprint origin**: blueprints are CDK-deployed constructs (see `cdk/src/constructs/blueprint.ts`). Platform operators deploy them. Users cannot upload arbitrary blueprint.yaml from the target repo. This property is load-bearing for the security model — if blueprint origin ever becomes user-uploaded, the blueprint-injection section (§12.4) must be re-evaluated. The 64 KB text cap (§5.1, finding #12) and `disable:` hard-deny rejection (finding #9) are applied regardless of origin as defense in depth. - **Linear → decision handlers**: the mapped task owner must author the actual - reply returned by Linear's API. A webhook signature alone is insufficient. + reply returned by Linear's API. Bot comments and comments authored as the + saved OAuth token identity are rejected. For approvals, install with the + default `actor=app`; diagnostic user-mode credentials cannot approve on + behalf of their authorizing user. Use the authenticated CLI in that case. + A webhook signature alone is insufficient: legacy workers can read a bundle + containing OAuth credentials and the webhook signing key. The authoritative + readback protects this approval path, not every other webhook action. The Slack button/proxy design in §11.2 remains future work. -See [upgrading approval permissions](../guides/DEPLOYMENT_GUIDE.md#upgrading-approval-permissions) +See [ADR-023](../decisions/ADR-023-trusted-approval-writer.md) for the decision and limits, and [upgrading approval permissions](../guides/DEPLOYMENT_GUIDE.md#upgrading-approval-permissions) before updating an existing deployment. ### 12.2 Ownership encoded in ConditionExpression @@ -1833,7 +1841,7 @@ Tracked as IMPL-22. Without these telemetry-driven re-evaluations, 50 will ossif ### 12.10 JWT replay -Cognito JWT with signature + expiry validation on API Gateway. Approval row conditional-update prevents replay from mutating state. Slack button replays similarly mediated by `SlackUserMappingTable` (§11.2) + Slack's own request signing. +Cognito JWT with signature + expiry validation on API Gateway. Approval row conditional-update prevents replay from mutating state. Native Slack approval buttons are not implemented; §11.2 describes the proposed flow. --- diff --git a/docs/design/SECURITY.md b/docs/design/SECURITY.md index eb64589a2..00dee4cd3 100644 --- a/docs/design/SECURITY.md +++ b/docs/design/SECURITY.md @@ -42,10 +42,10 @@ Three authentication mechanisms protect the platform, matching its input channel **Per-session IAM scoping** - The agent does not use its long-lived compute role (the AgentCore Runtime `ExecutionRole`, ECS Fargate task role, or Lambda MicroVMs execution role) for tenant data. Instead, at task startup it assumes a per-task **SessionRole** via `sts:AssumeRole` with session tags `{user_id, repo, task_id}`, and uses the resulting short-lived credentials for all DynamoDB and S3 tenant-data access. The SessionRole's policies self-constrain on those tags: - **DynamoDB**: item access on the four `task_id`-partitioned tables (task, events, approvals, nudges) is gated by a `dynamodb:LeadingKeys` condition equal to `${aws:PrincipalTag/task_id}` and requires that context key to be present. `Scan` is not granted. The main task table permits reads and only attribute-scoped `UpdateItem` writes: the agent can report status, heartbeat, results and approval details, but cannot replace/delete the row or edit coordinator-owned start receipts, capacity reservations, owner identity or compute handles. The reviewed write list is `cdk/src/constructs/agent-task-write-attributes.json`; Python tests exercise the current writers against it. Events and nudges retain task-scoped item writes. Approval records permit reads and transaction condition checks only. `LeadingKeys` binds the base-table partition key, not a GSI key such as `user_id`. -- **Approval requests**: workers call an IAM-signed API path restricted to their task tag. Its trusted handler can create `PENDING` or conditionally record `TIMED_OUT`; human decisions, notification markers and retention fields are unavailable to the worker. Task ownership/state and MicroVM worker leases are checked transactionally. See the [approval trust boundary](./CEDAR_HITL_GATES.md#121-trust-boundaries) and [upgrade procedure](../guides/DEPLOYMENT_GUIDE.md#upgrading-approval-permissions). +- **Approval requests**: workers call an IAM-signed API path restricted to their task tag. Its trusted handler can create `PENDING` or conditionally record `TIMED_OUT`; human decisions, notification markers and retention fields are unavailable to the worker. IAM binds the caller to the tagged task path. The transaction guards against concurrent task ownership/state changes and checks MicroVM worker leases. See the [approval trust boundary](./CEDAR_HITL_GATES.md#121-trust-boundaries) and [upgrade procedure](../guides/DEPLOYMENT_GUIDE.md#upgrading-approval-permissions). - **S3**: trace writes and attachment reads are scoped to the `/${aws:PrincipalTag/user_id}/` object prefix. -The compute role retains only non-tenant access (Bedrock model invocation — already ARN-scoped; CloudWatch Logs; the GitHub PAT secret, read once before the SessionRole is assumed; AgentCore Memory) plus `sts:AssumeRole`/`sts:TagSession` on the SessionRole. Because the agent runs under credentials that are themselves an assumed role, its `AssumeRole` is *role chaining* — capped at one hour regardless of the role's max session duration — so the agent uses a **refreshable** credential provider that re-assumes before expiry (tasks can run up to the 8-hour `maxLifetime`). The design is backend-agnostic: the same SessionRole and agent code serve all three backends (AgentCore Runtime, ECS Fargate, Lambda MicroVMs). Existing scoped credentials are constrained to their tags, but the compute role chooses those tags when assuming the role. The trust policy does not independently bind a worker to one task; this is not complete isolation against a compromised worker with ambient credentials. No worker needs direct access to the shared capacity counter. ECS's legacy configuration without a session role retains unscoped task reads/reporting updates, with the same attribute restriction. The policy structure and Python writer compatibility are tested locally. A historical IAM simulator check covered the original tag conditions; it does not verify the new attribute restriction or deployed transaction authorization. The [acceptance checklist](../verification/README.md#live-acceptance-for-an-installation) includes effective-role authorization checks. +The compute role retains only non-tenant access (Bedrock model invocation — already ARN-scoped; CloudWatch Logs; the GitHub PAT secret, read once before the SessionRole is assumed; AgentCore Memory) plus `sts:AssumeRole`/`sts:TagSession` on the SessionRole. Because the agent runs under credentials that are themselves an assumed role, its `AssumeRole` is *role chaining* — capped at one hour regardless of the role's max session duration — so the agent uses a **refreshable** credential provider that re-assumes before expiry (tasks can run up to the 8-hour `maxLifetime`). The design is backend-agnostic: the same SessionRole and agent code serve all three backends (AgentCore Runtime, ECS Fargate, Lambda MicroVMs). Existing scoped credentials are constrained to their tags, but the compute role chooses those tags when assuming the role. The trust policy does not independently bind a worker to one task; this is not complete isolation against a compromised worker with ambient credentials. No worker needs direct access to the shared capacity counter. ECS's configuration without a session role retains unscoped task reads/reporting updates, with the same attribute restriction, but cannot enable approval gates; the construct rejects approval wiring without the session role and service URL. The policy structure and Python writer compatibility are tested locally. A historical IAM simulator check covered the original tag conditions; it does not verify the new attribute restriction or deployed transaction authorization. The [acceptance checklist](../verification/README.md#live-acceptance-for-an-installation) includes effective-role authorization checks. **Lambda MicroVMs compute-role delta** — The compute role additionally reads the per-workspace channel-OAuth secrets (`bgagent-linear-oauth-*`, `bgagent-jira-oauth-*`) and its payload bucket's `bootstrap/*` manifests. Ambient task-object reads and payload-bucket listing are explicitly denied. It also has read-only `ec2:DescribeAvailabilityZones` for repository CDK synthesis; that API has no resource-level scope. diff --git a/docs/guides/DEPLOYMENT_GUIDE.md b/docs/guides/DEPLOYMENT_GUIDE.md index 6567b8224..31c9ad1a8 100644 --- a/docs/guides/DEPLOYMENT_GUIDE.md +++ b/docs/guides/DEPLOYMENT_GUIDE.md @@ -259,6 +259,9 @@ directly to DynamoDB and cannot create new gates after those permissions are rem 3. Deploy the stack. Check that the SessionRole has approval-table reads and condition checks only, plus `execute-api:Invoke` restricted to its task tag. CDK supplies `APPROVAL_REQUESTS_API_URL` to all three compute backends. + Custom ECS constructs must provide both the SessionRole and service URL; + approval wiring without them is rejected before deployment. A MicroVM + manifest missing the URL is rejected before the worker starts. 4. Submit a test task that triggers a known approval rule on each enabled backend. Verify that the request appears, an owner decision resumes it, and an explicit deadline records `TIMED_OUT` without overwriting a human decision. Then resume diff --git a/docs/mise.toml b/docs/mise.toml index 191fab83e..2a6639dc5 100644 --- a/docs/mise.toml +++ b/docs/mise.toml @@ -8,13 +8,13 @@ description = "Install from monorepo root (Yarn workspaces)" run = "cd .. && yarn install --check-files" [tasks.sync] -description = "Sync docs content from guides/design/CONTRIBUTING" +description = "Sync docs content from guides/design/decisions/verification/CONTRIBUTING" run = "node scripts/sync-starlight.mjs" [tasks.build] description = "Starlight docs site build" depends = [":sync"] -run = "./node_modules/.bin/astro build" +run = "./node_modules/.bin/astro build && node scripts/check-site-links.mjs" [tasks.dev] description = "Starlight dev server" @@ -26,7 +26,7 @@ depends = [":sync"] run = "./node_modules/.bin/astro sync && ./node_modules/.bin/astro check" [tasks.link-check] -# Scope: docs/guides/, docs/design/, docs/decisions/, and root *.md files. +# Scope: docs/guides/, docs/design/, docs/decisions/, docs/verification/, and root *.md files. # Not scanned: docs/README.md, agent/, cli/, contracts/, docs/abca-plugin/. description = "Check for broken links in Markdown sources" run = "bash scripts/link-check.sh" diff --git a/docs/package.json b/docs/package.json index 2e24f0e8d..64e13c7ef 100644 --- a/docs/package.json +++ b/docs/package.json @@ -5,7 +5,7 @@ "dev": "astro dev", "start": "astro dev", "docs:check": "npm run sync && astro check", - "docs:build": "npm run sync && astro build", + "docs:build": "npm run sync && astro build && node scripts/check-site-links.mjs", "build": "npm run docs:build", "preview": "astro preview", "security:retire": "retire --path . --severity high", diff --git a/docs/scripts/check-site-links.mjs b/docs/scripts/check-site-links.mjs new file mode 100644 index 000000000..2665d57ca --- /dev/null +++ b/docs/scripts/check-site-links.mjs @@ -0,0 +1,40 @@ +import fs from 'node:fs'; +import path from 'node:path'; + +const root = path.resolve(import.meta.dirname, '..', 'dist'); +const base = '/sample-autonomous-cloud-coding-agents/'; +const pages = fs.readdirSync(root, { recursive: true }).filter(file => file.endsWith('.html')); +if (pages.length < 10) throw new Error('Expected a built documentation site before checking links'); + +const errors = new Set(); +const contents = new Map(); +function read(file) { + if (!contents.has(file)) contents.set(file, fs.readFileSync(file, 'utf8')); + return contents.get(file); +} + +for (const file of pages) { + const origin = `https://local${base}${file.replace(/index\.html$/, '')}`; + for (const match of read(path.join(root, file)).matchAll(/]*\bhref="([^"]*)"/g)) { + const href = match[1].replaceAll('&', '&'); + // This check is offline. External availability is outside its scope. + if (/^(?:[a-z]+:|\/\/)/i.test(href)) continue; + const url = new URL(href, origin); + if (!url.pathname.startsWith(base)) { + errors.add(`${file}: link escapes the site's base: ${href}`); + continue; + } + let target = path.join(root, decodeURIComponent(url.pathname.slice(base.length))); + if (fs.existsSync(target) && fs.statSync(target).isDirectory()) { + target = path.join(target, 'index.html'); + } + if (!fs.existsSync(target)) { + errors.add(`${file}: missing page: ${href}`); + } else if (url.hash && !read(target).includes(`id="${decodeURIComponent(url.hash.slice(1))}"`)) { + errors.add(`${file}: missing anchor: ${href}`); + } + } +} + +if (errors.size) throw new Error(`Broken documentation links:\n${[...errors].join('\n')}`); +console.log(`Checked internal page and anchor links in ${pages.length} rendered pages.`); diff --git a/docs/scripts/link-check.sh b/docs/scripts/link-check.sh index 23f3632a8..0ca170cef 100644 --- a/docs/scripts/link-check.sh +++ b/docs/scripts/link-check.sh @@ -4,7 +4,7 @@ set -euo pipefail filelist=$(mktemp) trap 'rm -f "$filelist"' EXIT -{ find guides design decisions -name '*.md' -print0; find .. -maxdepth 1 -name '*.md' -print0; } > "$filelist" +{ find guides design decisions verification -name '*.md' -print0; find .. -maxdepth 1 -name '*.md' -print0; } > "$filelist" count=$(tr '\0' '\n' < "$filelist" | grep -c .) if [ "$count" -lt 10 ]; then diff --git a/docs/scripts/sync-starlight.mjs b/docs/scripts/sync-starlight.mjs index 3a60dfeda..dd895329f 100644 --- a/docs/scripts/sync-starlight.mjs +++ b/docs/scripts/sync-starlight.mjs @@ -8,7 +8,7 @@ const docsBase = '/sample-autonomous-cloud-coding-agents'; function normalizeFileStem(input) { const cleaned = input - .replace(/\.md$/i, '') + .replace(/\.mdx?$/i, '') .replace(/[^a-zA-Z0-9]+/g, '-') .replace(/^-+|-+$/g, '') .toLowerCase(); @@ -18,7 +18,15 @@ function normalizeFileStem(input) { return `${cleaned.charAt(0).toUpperCase()}${cleaned.slice(1)}`; } -function rewriteDocsLinkTarget(target) { +function rewriteDocsLinkTarget(target, sourcePath) { + if (target?.startsWith('#') && path.basename(sourcePath) === 'USER_GUIDE.md') { + const route = { + '#joining-an-existing-deployment': '/using/authentication#joining-an-existing-deployment', + '#get-stack-outputs': '/using/authentication#get-stack-outputs', + '#approval-gates-cedar-hitl': '/using/approval-gates-cedar-hitl', + }[target]; + if (route) return route; + } if (!target || target.startsWith('#') || target.startsWith('/')) { return undefined; } @@ -27,18 +35,28 @@ function rewriteDocsLinkTarget(target) { } const [pathPart, anchor] = target.split('#'); - if (!pathPart.toLowerCase().endsWith('.md')) { - return undefined; - } const normalizedPath = pathPart.replaceAll('\\', '/'); - // Verification runbooks remain repository documents, not Starlight pages. - // Preserve their real destination instead of inventing an architecture route. - const verification = normalizedPath.match(/(?:^|\/)verification\/(.+)$/); - if (verification) { - return `https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/docs/verification/${verification[1]}${anchor ? `#${anchor}` : ''}`; + // Resolve relative links in the source tree before assigning a site route. + // In particular, ./README.md inside verification is not architecture/readme. + const sourceTarget = path.resolve(path.dirname(sourcePath), normalizedPath); + if (sourceTarget === path.join(targetRoot, 'index.md')) return '/'; + if (sourceTarget === path.join(docsRoot, 'decisions')) return '/decisions/readme'; + const relativeSource = path.relative(repoRoot, sourceTarget).replaceAll('\\', '/'); + if (!relativeSource.startsWith('../') && fs.existsSync(sourceTarget) + && (!sourceTarget.startsWith(docsRoot + path.sep) || sourceTarget === path.join(docsRoot, 'README.md') + || sourceTarget.startsWith(path.join(docsRoot, 'abca-plugin') + path.sep))) { + const kind = fs.statSync(sourceTarget).isDirectory() ? 'tree' : 'blob'; + return `https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/${kind}/main/${relativeSource}${anchor ? `#${anchor}` : ''}`; } - const stem = path.basename(normalizedPath, '.md'); + if (!/\.mdx?$/i.test(pathPart)) return undefined; + for (const directory of ['verification', 'decisions']) { + if (path.dirname(sourceTarget) === path.join(docsRoot, directory)) { + const slug = normalizeFileStem(path.basename(sourceTarget)).toLowerCase(); + return `/${directory}/${slug}${anchor ? `#${anchor}` : ''}`; + } + } + const stem = path.basename(normalizedPath).replace(/\.mdx?$/i, ''); const slug = normalizeFileStem(stem).toLowerCase(); const anchorSuffix = anchor ? `#${anchor}` : ''; @@ -77,6 +95,9 @@ function rewriteDocsLinkTarget(target) { const userGuideAnchorRoutes = { overview: '/using/overview', authentication: '/using/authentication', + 'joining-an-existing-deployment': '/using/authentication#joining-an-existing-deployment', + 'get-stack-outputs': '/using/authentication#get-stack-outputs', + 'operator-commands-stack-admin': '/using/using-the-cli#operator-commands-stack-admin', 'repository-onboarding': '/customizing/repository-onboarding', 'per-repo-overrides': '/customizing/per-repo-overrides', 'monthly-user-and-team-budgets': '/customizing/per-repo-overrides#monthly-user-and-team-budgets', @@ -86,6 +107,7 @@ function rewriteDocsLinkTarget(target) { 'webhook-integration': '/using/webhook-integration', 'task-lifecycle': '/using/task-lifecycle', 'what-the-agent-does': '/using/what-the-agent-does', + 'approval-gates-cedar-hitl': '/using/approval-gates-cedar-hitl', 'tips-for-being-a-good-citizen': '/using/tips-for-being-a-good-citizen', }; if (stem === 'USER_GUIDE' && anchor) { @@ -105,11 +127,11 @@ function rewriteDocsLinkTarget(target) { return `/architecture/${slug}${anchorSuffix}`; } -function ensureFrontmatter(content, title) { +function ensureFrontmatter(content, title, sourcePath) { const normalized = content .replaceAll('../imgs/', `${docsBase}/imgs/`) .replace(/\[([^\]]+)\]\(([^)]+)\)/g, (match, label, target) => { - const rewritten = rewriteDocsLinkTarget(target); + const rewritten = rewriteDocsLinkTarget(target, sourcePath); if (!rewritten) { return match; } @@ -154,7 +176,7 @@ function mirrorMarkdownFile(sourcePath, targetRelativePath) { const ext = path.extname(sourcePath); const stem = path.basename(sourcePath, ext); const fallbackTitle = normalizeFileStem(stem).replace(/-/g, ' '); - const out = ensureFrontmatter(raw, fallbackTitle); + const out = ensureFrontmatter(raw, fallbackTitle, sourcePath); writeFile(path.join(docsRoot, targetRelativePath), out); } @@ -170,7 +192,7 @@ function mirrorDirectory(sourceDir, targetDirRelative) { const sourcePath = path.join(sourceDir, file); const raw = fs.readFileSync(sourcePath, 'utf8'); const fallbackTitle = normalizeFileStem(file).replace(/-/g, ' '); - const out = ensureFrontmatter(raw, fallbackTitle); + const out = ensureFrontmatter(raw, fallbackTitle, sourcePath); const normalizedName = `${normalizeFileStem(file)}.md`; writeFile(path.join(docsRoot, targetDirRelative, normalizedName), out); } @@ -202,7 +224,7 @@ function splitGuide(sourcePath, targetDirRelative, introTitle) { const raw = fs.readFileSync(sourcePath, 'utf8'); const parts = raw.split(/\n##\s+/g); const intro = parts.shift() ?? ''; - const introOut = ensureFrontmatter(intro.trim(), introTitle); + const introOut = ensureFrontmatter(intro.trim(), introTitle, sourcePath); writeFile(path.join(docsRoot, targetDirRelative, 'Introduction.md'), introOut); for (const part of parts) { @@ -210,7 +232,7 @@ function splitGuide(sourcePath, targetDirRelative, introTitle) { const heading = (firstNewline === -1 ? part : part.slice(0, firstNewline)).trim(); const body = firstNewline === -1 ? '' : part.slice(firstNewline + 1).trim(); const filename = `${normalizeFileStem(heading)}.md`; - const out = ensureFrontmatter(body, heading); + const out = ensureFrontmatter(body, heading, sourcePath); writeFile(path.join(docsRoot, targetDirRelative, filename), out); } } @@ -318,6 +340,9 @@ mirrorDirectory(path.join(docsRoot, 'design'), path.join('src', 'content', 'docs // --- Decision records (ADRs): mirror to decisions/ --- mirrorDirectory(path.join(docsRoot, 'decisions'), path.join('src', 'content', 'docs', 'decisions')); +// Verification runbooks ship with the same revision as their referring pages. +mirrorDirectory(path.join(docsRoot, 'verification'), path.join('src', 'content', 'docs', 'verification')); + // --- Static assets: copy source image dir into the site's public/ --- // Guides reference images as `../imgs/foo.png`; ensureFrontmatter() turns // those into absolute `//imgs/foo.png` URLs, which Astro serves from diff --git a/docs/src/content/docs/architecture/Api-contract.md b/docs/src/content/docs/architecture/Api-contract.md index b8adba636..885253094 100644 --- a/docs/src/content/docs/architecture/Api-contract.md +++ b/docs/src/content/docs/architecture/Api-contract.md @@ -258,7 +258,7 @@ Returns full details of a task. Users can only access their own tasks. } ``` -`agent_heartbeat_at` is the agent's last in-guest liveness beat, or `null`. The agent writes it every 45 s on the `agentcore` and `lambda-microvm` backends; on `ecs` it is written once at start, because that backend runs the pipeline directly instead of serving HTTP. The orchestrator reads the same field to detect a hung agent inside a healthy compute environment ([ORCHESTRATOR.md](/sample-autonomous-cloud-coding-agents/architecture/orchestrator#dynamodb-heartbeat-agentcore-and-lambda-microvms)), so a value that is minutes old on a `RUNNING` task is the signal, not the timestamp itself. `null` on records written before the field existed. +`agent_heartbeat_at` is the agent's last in-guest liveness beat, or `null`. The agent writes it every 45 s on the `agentcore` and `lambda-microvm` backends; on `ecs` it is written once at start, because that backend runs the pipeline directly instead of serving HTTP. The orchestrator reads the same field to detect a hung agent inside a healthy compute environment ([ORCHESTRATOR.md](/sample-autonomous-cloud-coding-agents/architecture/orchestrator#liveness-monitoring)), so a value that is minutes old on a `RUNNING` task is the signal, not the timestamp itself. `null` on records written before the field existed. `error_classification` is a derived field computed at response time from `error_message`. When `error_message` is `null`, `error_classification` is `null`. When present, it contains: diff --git a/docs/src/content/docs/architecture/Cedar-hitl-gates.md b/docs/src/content/docs/architecture/Cedar-hitl-gates.md index bc4eeac0b..793a80364 100644 --- a/docs/src/content/docs/architecture/Cedar-hitl-gates.md +++ b/docs/src/content/docs/architecture/Cedar-hitl-gates.md @@ -21,11 +21,11 @@ title: Cedar hitl gates > Linear accepts an owner’s `approve` or `deny` reply to the approval comment; > the handler verifies the actual comment through Linear’s API. Slack uses CLI > response instructions. See the current -> [user guide](/sample-autonomous-cloud-coding-agents/using/overview#approval-gates-cedar-hitl) and +> [user guide](/sample-autonomous-cloud-coding-agents/using/approval-gates-cedar-hitl) and > [continuation protocol](/sample-autonomous-cloud-coding-agents/architecture/orchestrator#retained-microvm-approvals). > The normal deployment passed retained-request, ten-minute sleep, explicit-expiry > and sleep-off/rollback acceptance; see the -> [deployment record](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/docs/verification/README.md). +> [deployment record](/sample-autonomous-cloud-coding-agents/verification/readme). --- @@ -1048,7 +1048,7 @@ event. A later decision is rejected: the existing API returns `404 REQUEST_NOT_FOUND` for missing, foreign or already-decided approval rows, including a cancelled row. If approval committed first, cancellation preserves the recorded decision while cancelling the task. See the -[P3 approval verification record](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/docs/verification/README.md) +[P3 approval verification record](/sample-autonomous-cloud-coding-agents/verification/readme) for source versus deployment status. ### 7.2 `POST /v1/tasks/{task_id}/deny` @@ -1367,7 +1367,7 @@ t=45m: Task #1 completes. count → 9. Bob can submit task #11. AgentCore Runtime's `maxLifetime = 28800s` (8h) is an absolute timer from session start. It does NOT pause during `AWAITING_APPROVAL`. -An explicitly timed approval is bounded by the remaining worker lifetime minus the 120-second cleanup margin. Below the 30-second floor, the hook denies the action with reason `"insufficient lifetime"`. +When the worker supplies a remaining-lifetime estimate, the hook refuses to open a new gate if fewer than 30 seconds remain after the 120-second cleanup margin. This check applies to timed and untimed approvals and reports `"insufficient maxLifetime remaining (s) for approval"`. A positive approval timeout is also capped by that remaining budget. The default approval timeout is `0`: no decision deadline. Worker lifetime does not turn silence into a human denial. ECS and AgentCore do not currently restore a waiting agent into a replacement worker; when their task execution limit is reached, the task closes and its pending approval is cancelled. MicroVM can retain a verified checkpoint and continue on a replacement worker. Setting its sleep delay to `0` disables early sleep/retirement, but the coordinator still attempts retirement before the service lifetime ends. This option trades idle cost for faster replies. @@ -1541,7 +1541,7 @@ cancellations, timeouts and stranded waits also produce messages. The response path supports the CLI and native Linear thread replies. Slack approval buttons and the Slack OAuth/button design below remain proposed. Email remains a log-only stub and GitHub does not receive approval messages. Deployment status is recorded in the -[P3 verification record](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/docs/verification/README.md). +[P3 verification record](/sample-autonomous-cloud-coding-agents/verification/readme). **TaskApprovalsTable Streams are not consumed by the fan-out Lambda**. The approval row is working state; the audit trail is in TaskEventsTable. Enabling Streams on TaskApprovalsTable would be redundant and add noise. Final design: TaskApprovalsTable DOES NOT have Streams enabled. (Retains the `stream` attribute commented out for future use if needed.) @@ -1698,18 +1698,26 @@ These alarms transition to `ALARM` state in CloudWatch and appear in the console `POST /v1/tasks/{task_id}` to the session's `task_id` tag. The service creates `PENDING` rows or conditionally records a non-human `TIMED_OUT`; it rejects human decisions, notification markers, retention TTL and other extra fields. - It checks current task ownership/state and, for MicroVM, the active worker - lease in the same transaction. Worker-provided action descriptions remain - untrusted. This protects approval records; it does not sandbox code running + IAM binds the caller to a task path; the transaction prevents concurrent task + ownership/state changes and, for MicroVM, checks the active worker lease. + The tool preview, its hash, severity, reason and matching rules are worker + assertions, not independently evaluated policy results. The preview can be + truncated, so its bytes cannot be used to verify the full-input hash. This protects approval records; it does not sandbox code running inside the agent or bind ambient compute credentials to one task. - **User CLI ↔ API Gateway**: Cognito JWT (same authorizer as `/tasks/*`). Cognito `sub` is the canonical caller identity, used **verbatim** in DDB `ConditionExpression` (§7.1, finding #6). - **ApproveTaskFn/DenyTaskFn ↔ TaskApprovalsTable + TaskTable**: Lambda IAM policy allows `UpdateItem` on both tables under `TransactWriteItems`. Authorization is in the ConditionExpression (ownership AND state), not in a separate IAM boundary. - **Blueprint origin**: blueprints are CDK-deployed constructs (see `cdk/src/constructs/blueprint.ts`). Platform operators deploy them. Users cannot upload arbitrary blueprint.yaml from the target repo. This property is load-bearing for the security model — if blueprint origin ever becomes user-uploaded, the blueprint-injection section (§12.4) must be re-evaluated. The 64 KB text cap (§5.1, finding #12) and `disable:` hard-deny rejection (finding #9) are applied regardless of origin as defense in depth. - **Linear → decision handlers**: the mapped task owner must author the actual - reply returned by Linear's API. A webhook signature alone is insufficient. + reply returned by Linear's API. Bot comments and comments authored as the + saved OAuth token identity are rejected. For approvals, install with the + default `actor=app`; diagnostic user-mode credentials cannot approve on + behalf of their authorizing user. Use the authenticated CLI in that case. + A webhook signature alone is insufficient: legacy workers can read a bundle + containing OAuth credentials and the webhook signing key. The authoritative + readback protects this approval path, not every other webhook action. The Slack button/proxy design in §11.2 remains future work. -See [upgrading approval permissions](/sample-autonomous-cloud-coding-agents/getting-started/deployment-guide#upgrading-approval-permissions) +See [ADR-023](/sample-autonomous-cloud-coding-agents/decisions/adr-023-trusted-approval-writer) for the decision and limits, and [upgrading approval permissions](/sample-autonomous-cloud-coding-agents/getting-started/deployment-guide#upgrading-approval-permissions) before updating an existing deployment. ### 12.2 Ownership encoded in ConditionExpression @@ -1837,7 +1845,7 @@ Tracked as IMPL-22. Without these telemetry-driven re-evaluations, 50 will ossif ### 12.10 JWT replay -Cognito JWT with signature + expiry validation on API Gateway. Approval row conditional-update prevents replay from mutating state. Slack button replays similarly mediated by `SlackUserMappingTable` (§11.2) + Slack's own request signing. +Cognito JWT with signature + expiry validation on API Gateway. Approval row conditional-update prevents replay from mutating state. Native Slack approval buttons are not implemented; §11.2 describes the proposed flow. --- diff --git a/docs/src/content/docs/architecture/Compute.md b/docs/src/content/docs/architecture/Compute.md index d836f302d..0d9786fba 100644 --- a/docs/src/content/docs/architecture/Compute.md +++ b/docs/src/content/docs/architecture/Compute.md @@ -25,7 +25,7 @@ The default runtime is **Amazon Bedrock AgentCore Runtime**, which runs each ses | **Cost model** | vCPU-hrs + GB-hrs | vCPU + mem/sec | Baseline compute (8 GiB / 4 vCPU) plus additional burst usage (up to 32 GiB / 16 vCPU); no suspended compute charge, but snapshot storage and read/write charges remain | EC2 + EBS | EKS control + EC2 | Underlying compute | Request + duration | EC2 metal + your ops | | **Fit** | **Default choice** | Repos > 2 GB image | Suspend/resume economics; approval-wait-heavy workloads; default-sized repos. Heavy sustained-memory builds stay on ECS | GPU, heavy toolchains | Max flexibility | Queued batch jobs | **Poor** (15 min cap) | Best potential, highest cost | -> **Lambda MicroVMs are not Lambda functions.** They are a different compute primitive, so the functions column's 15-minute cap and poor-fit verdict do not apply. See [ADR-021](/sample-autonomous-cloud-coding-agents/architecture/adr-021-lambda-microvms-compute-backend). +> **Lambda MicroVMs are not Lambda functions.** They are a different compute primitive, so the functions column's 15-minute cap and poor-fit verdict do not apply. See [ADR-021](/sample-autonomous-cloud-coding-agents/decisions/adr-021-lambda-microvms-compute-backend). The backend is selected per repo via `compute_type` in the Blueprint config. The orchestrator resolves the strategy and delegates session start, polling, and termination to the strategy implementation. See [REPO_ONBOARDING.md](/sample-autonomous-cloud-coding-agents/architecture/repo-onboarding) for the `ComputeStrategy` interface. @@ -85,11 +85,11 @@ Lambda MicroVMs are an opt-in third backend, selected per repository with `compu For managed images, run `cdk/scripts/package-microvm-artifact.sh --stack-name ` after each agent change. It packages the Dockerfile's local inputs into a deterministic ZIP and uploads it under `microvm-images/agent-artifact-.zip`. The checksum is a fingerprint of the uploaded bytes: identical inputs reuse the same verified object, and changed inputs produce a new filename. Deploy with the printed `--context microvm_artifact_sha256=` alongside `microvm_base_image_arn` and `microvm_base_image_version`, and retain these inputs for later deployments. The changed S3 URI tells CloudFormation to update the existing image; overwriting the old fixed filename alone does not. Missing or malformed digests fail synthesis. An initial deployment without an image must create the bucket first. The script's explicit `--create-image` alternative retains the fixed base key and calls the image API directly. -Because a snapshot freezes its build-time environment, current deployment identifiers arrive through the v2 payload bootstrap. The coordinator publishes a non-secret deployment manifest and sends a single-object signed download URL for the task. The worker reads only its deployment's `bootstrap/*` with ambient credentials; other object reads and payload-bucket listing are explicitly denied. The downloaded task identity and configuration must match the authenticated manifest before configuration installation. The serialized reference fits the verified 4,096-byte hook limit; all payload sizes use S3. ECS shares this transport through `AGENT_PAYLOAD_REF`. [ADR-021 §3](/sample-autonomous-cloud-coding-agents/architecture/adr-021-lambda-microvms-compute-backend#3-packaging-same-agent-image-source-new-build-path) defines the wire format and compatibility requirements; [live payload checks](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/docs/verification/README.md) record validation. +Because a snapshot freezes its build-time environment, current deployment identifiers arrive through the v2 payload bootstrap. The coordinator publishes a non-secret deployment manifest and sends a single-object signed download URL for the task. The worker reads only its deployment's `bootstrap/*` with ambient credentials; other object reads and payload-bucket listing are explicitly denied. The downloaded task identity and configuration must match the authenticated manifest before configuration installation. The serialized reference fits the verified 4,096-byte hook limit; all payload sizes use S3. ECS shares this transport through `AGENT_PAYLOAD_REF`. [ADR-021 §3](/sample-autonomous-cloud-coding-agents/decisions/adr-021-lambda-microvms-compute-backend#3-packaging-same-agent-image-source-new-build-path) defines the wire format and compatibility requirements; [live payload checks](/sample-autonomous-cloud-coding-agents/verification/readme) record validation. Networking separates image build from execution: the build-only connector permits TCP 80 and 443 because the Dockerfile uses `apt-get`, while running MicroVMs retain 443-only egress through the platform VPC. Every launch explicitly passes the Lambda-managed `NO_INGRESS` connector; omission would select the service's public-ingress default. Images declare and serve `/ready` and `/validate` at build time and `/run`, `/terminate`, `/suspend` and `/resume` at runtime. Automatic suspension is a separate deployment opt-in. Registry HTTP/SSE tools therefore need reachable HTTPS/443 endpoints; remote non-443 tools are unsupported under the default policies of AgentCore and ECS as well. Local `stdio` tools can run, with their outbound traffic subject to the same restriction. Asset resolution/loading does not probe connectivity; see [registry network support](/sample-autonomous-cloud-coding-agents/architecture/registry#2-asset-kinds-for-mvp). -P3 connects the guest checkpoint/credential hooks, actual image-version capability, durable supervisor and post-commit approval wake. The `microvm_approval_suspend_enabled` deployment context defaults false and controls both a static opt-in and a live Parameter Store switch. Existing durable executions reread the live switch before new suspension because their original Lambda environment is pinned. Recovery and the original service lifetime survive supervisor replay; the API preserves accepted decisions when optional wake fails. AgentCore/ECS retain explicit unsupported pause/wake results. Explicit approval deadlines use the original UTC/monotonic deadline. See the [lifecycle diagnostics](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/docs/verification/645-p3-lifecycle-diagnostics.md) for failure investigation, and the [acceptance status](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/docs/verification/README.md) for deployment evidence. +P3 connects the guest checkpoint/credential hooks, actual image-version capability, durable supervisor and post-commit approval wake. The `microvm_approval_suspend_enabled` deployment context defaults false and controls both a static opt-in and a live Parameter Store switch. Existing durable executions reread the live switch before new suspension because their original Lambda environment is pinned. Recovery and the original service lifetime survive supervisor replay; the API preserves accepted decisions when optional wake fails. AgentCore/ECS retain explicit unsupported pause/wake results. Explicit approval deadlines use the original UTC/monotonic deadline. See the [lifecycle diagnostics](/sample-autonomous-cloud-coding-agents/verification/645-p3-lifecycle-diagnostics) for failure investigation, and the [acceptance status](/sample-autonomous-cloud-coding-agents/verification/readme) for deployment evidence. Tasks can set `microvm_sleep_after_s` (CLI: `--microvm-sleep-after `). The default is 600 seconds of waiting for each approval; zero disables sleep. diff --git a/docs/src/content/docs/architecture/Deployment-roles.md b/docs/src/content/docs/architecture/Deployment-roles.md index 7657444c8..ebba8cfbb 100644 --- a/docs/src/content/docs/architecture/Deployment-roles.md +++ b/docs/src/content/docs/architecture/Deployment-roles.md @@ -857,7 +857,7 @@ exact names in addition to the legacy flat-layout prefixes. The execution role stays in the parent and is still excluded. Re-bootstrap before deploying the child stack. Existing flat deployments must keep `microvm_nested_stack=false` until their resource migration is reviewed; changing ownership is not an ordinary -in-place update. See the [nested-stack runbook](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/docs/verification/645-p3-nested-stack.md). +in-place update. See the [nested-stack runbook](/sample-autonomous-cloud-coding-agents/verification/645-p3-nested-stack). For a reviewed migration that keeps old and new resources side by side, `microvm_resource_name_prefix` gives the nested image, network connectors and log diff --git a/docs/src/content/docs/architecture/Identity-and-auth.md b/docs/src/content/docs/architecture/Identity-and-auth.md index c91ede25f..eef62f87f 100644 --- a/docs/src/content/docs/architecture/Identity-and-auth.md +++ b/docs/src/content/docs/architecture/Identity-and-auth.md @@ -4,10 +4,10 @@ title: Identity and auth # Identity and authentication — worked examples -ABCA today carries its own inbound identity (Amazon Cognito for the CLI and REST API, HMAC-SHA256 for webhooks) and resolves outbound credentials per integration through hand-rolled Secrets Manager resolvers (`resolve_github_token`, `resolve_linear_api_token`, and the Jira OAuth/Forge resolver). This doc maps each of those integration shapes onto the [Amazon Bedrock AgentCore Identity](https://docs.aws.amazon.com/bedrock-agentcore/latest/devguide/identity.html) primitives: workload identity, the token vault, and the three outbound flows. A contributor adding a new integration can then see which flow to pick and what the wire looks like at each hop. Every code block below is grounded in the [AgentCore developer guide](https://docs.aws.amazon.com/bedrock-agentcore/latest/devguide/identity.html) and the [CreateOauth2CredentialProvider API reference](https://docs.aws.amazon.com/bedrock-agentcore-control/latest/APIReference/API_CreateOauth2CredentialProvider.html); the binding decision for ABCA is captured in [ADR-016](/sample-autonomous-cloud-coding-agents/architecture/adr-016-pluggable-identity-and-auth). +ABCA today carries its own inbound identity (Amazon Cognito for the CLI and REST API, HMAC-SHA256 for webhooks) and resolves outbound credentials per integration through hand-rolled Secrets Manager resolvers (`resolve_github_token`, `resolve_linear_api_token`, and the Jira OAuth/Forge resolver). This doc maps each of those integration shapes onto the [Amazon Bedrock AgentCore Identity](https://docs.aws.amazon.com/bedrock-agentcore/latest/devguide/identity.html) primitives: workload identity, the token vault, and the three outbound flows. A contributor adding a new integration can then see which flow to pick and what the wire looks like at each hop. Every code block below is grounded in the [AgentCore developer guide](https://docs.aws.amazon.com/bedrock-agentcore/latest/devguide/identity.html) and the [CreateOauth2CredentialProvider API reference](https://docs.aws.amazon.com/bedrock-agentcore-control/latest/APIReference/API_CreateOauth2CredentialProvider.html); the binding decision for ABCA is captured in [ADR-016](/sample-autonomous-cloud-coding-agents/decisions/adr-016-pluggable-identity-and-auth). - **Use this doc for:** picking the right outbound flow (USER_FEDERATION / M2M / OBO) for a new integration, reading the principal at each hop of a webhook-triggered async agent, and seeing the Linear before/after the token vault replaces. -- **Related docs:** [SECURITY.md](/sample-autonomous-cloud-coding-agents/architecture/security) for the security boundaries and the shared-PAT limitation, [ADR-016](/sample-autonomous-cloud-coding-agents/architecture/adr-016-pluggable-identity-and-auth) for the two-seam pluggable-auth decision, and the Authentication page (`/using/authentication`) for ABCA's current inbound auth. +- **Related docs:** [SECURITY.md](/sample-autonomous-cloud-coding-agents/architecture/security) for the security boundaries and the shared-PAT limitation, [ADR-016](/sample-autonomous-cloud-coding-agents/decisions/adr-016-pluggable-identity-and-auth) for the two-seam pluggable-auth decision, and the Authentication page (`/using/authentication`) for ABCA's current inbound auth. ## Design principle @@ -188,7 +188,7 @@ The config field is `grantType`, singular. The nested actor field is `actorToken A human creates a ticket and adds a label (`agent:triage` on Jira; the same shape covers the GitHub-issue-label trigger the ABCA team is discussing in #abca). Automation fires a webhook, the agent triages, comments, and maybe closes, acting on the user's behalf where write-back matters. This is the realistic trigger pattern: no interactive browser, identity arrives as webhook payload, and consent happens later when the agent calls back. -> **Illustrative target, not the shipped Jira write path.** The flow below shows a future user-bound token-vault design. Current ABCA Jira uses 3LO only for inbound reads and human lookup. Outbound writes go to an HMAC-signed Forge web trigger, which calls Jira with `api.asApp()` so the actor is the installed `bgagent` app. The human Jira account remains task-attribution data, not the outbound credential. See [ADR-015](/sample-autonomous-cloud-coding-agents/architecture/adr-015-jira-integration). +> **Illustrative target, not the shipped Jira write path.** The flow below shows a future user-bound token-vault design. Current ABCA Jira uses 3LO only for inbound reads and human lookup. Outbound writes go to an HMAC-signed Forge web trigger, which calls Jira with `api.asApp()` so the actor is the installed `bgagent` app. The human Jira account remains task-attribution data, not the outbound credential. See [ADR-015](/sample-autonomous-cloud-coding-agents/decisions/adr-015-jira-integration). The architecture chain: diff --git a/docs/src/content/docs/architecture/Orchestrator.md b/docs/src/content/docs/architecture/Orchestrator.md index f54421ac9..bd55896b7 100644 --- a/docs/src/content/docs/architecture/Orchestrator.md +++ b/docs/src/content/docs/architecture/Orchestrator.md @@ -300,7 +300,7 @@ When the session is unhealthy, the task transitions to `FAILED` with "Agent sess - A terminal substrate report paired with a non-terminal task first checks for a complete, acknowledged approval checkpoint. Such a checkpoint can retire the old attempt and retain the task for a replacement. Without one, finalization strongly re-reads the task row before classifying a substrate failure. - Substrate state detects a dead VM; heartbeat staleness detects loss of the heartbeat writer inside a VM that still reports `RUNNING`. The independent heartbeat thread can continue during a pipeline hang, so a fresh timestamp is not proof of progress. -The P3 supervisor saves intent before control calls and rechecks the gate before and after them. Its durable state retains an absolute service lifetime, consecutive failures, recovery start time and next delay. Three failed cycles or 120 seconds of unconfirmed wake cannot become an indefinite wait. AWS RUNNING does not end recovery while the guest remains stuck on a decided/expired approval; fresh guest liveness is required. API approve/deny commit first, then attempt a bounded wake without changing the decision response. Automatic suspension defaults off via `microvm_approval_suspend_enabled`; disabling new sleep preserves wake and cleanup. See the [lifecycle diagnostics](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/docs/verification/645-p3-lifecycle-diagnostics.md). +The P3 supervisor saves intent before control calls and rechecks the gate before and after them. Its durable state retains an absolute service lifetime, consecutive failures, recovery start time and next delay. Three failed cycles or 120 seconds of unconfirmed wake cannot become an indefinite wait. AWS RUNNING does not end recovery while the guest remains stuck on a decided/expired approval; fresh guest liveness is required. API approve/deny commit first, then attempt a bounded wake without changing the decision response. Automatic suspension defaults off via `microvm_approval_suspend_enabled`; disabling new sleep preserves wake and cleanup. See the [lifecycle diagnostics](/sample-autonomous-cloud-coding-agents/verification/645-p3-lifecycle-diagnostics). Unanswered approvals have no deadline by default. A checkpointed MicroVM wait can retire after an hour, or before that worker's lifetime ends. Retirement and diff --git a/docs/src/content/docs/architecture/Repo-onboarding.md b/docs/src/content/docs/architecture/Repo-onboarding.md index 03d04de47..2da230aec 100644 --- a/docs/src/content/docs/architecture/Repo-onboarding.md +++ b/docs/src/content/docs/architecture/Repo-onboarding.md @@ -7,7 +7,7 @@ title: Repo onboarding Before users can submit tasks for a repository, that repository must be onboarded to the platform. Onboarding registers the repo and produces a per-repo configuration that the orchestrator uses at task time: compute strategy, model, credentials, networking, and pipeline customizations. If a user submits a task for a non-onboarded repo, the API returns `422 REPO_NOT_ONBOARDED`. - **Use this doc for:** the Blueprint construct interface, RepoConfig schema, override precedence, compute strategy interface, and pipeline customization model. -- **For practical usage:** see [Quick Start](../guides/QUICK_START.mdx) for onboarding your first repo and [User Guide](/sample-autonomous-cloud-coding-agents/using/overview) for per-repo overrides. +- **For practical usage:** see [Quick Start](/sample-autonomous-cloud-coding-agents/getting-started/quick-start) for onboarding your first repo and [User Guide](/sample-autonomous-cloud-coding-agents/using/overview) for per-repo overrides. - **Related docs:** [ORCHESTRATOR.md](/sample-autonomous-cloud-coding-agents/architecture/orchestrator) for how the orchestrator consumes blueprint config, [COMPUTE.md](/sample-autonomous-cloud-coding-agents/architecture/compute) for compute backends, [SECURITY.md](/sample-autonomous-cloud-coding-agents/architecture/security) for custom step trust boundaries. ## Why onboarding? @@ -36,11 +36,11 @@ For operators with IAM access to the deployed stack, `bgagent repo onboard` and | **Soft-delete** | `status=removed` + 30-day TTL on stack removal | Same semantics on `offboard` | | **Custom runtime / token IAM** | `additionalRuntimeArns` / `additionalSecretArns` in CDK | Stored in the row, but orchestrator IAM still requires a CDK deploy | | **Cedar, egress, pipeline steps** | Supported via construct props | Not exposed — use CDK | -| **Audit trail** | CloudFormation change set + deploy logs | CLI stdout only today (see [ADR-017](/sample-autonomous-cloud-coding-agents/architecture/adr-017-operator-cli-repo-onboarding)) | +| **Audit trail** | CloudFormation change set + deploy logs | CLI stdout only today (see [ADR-017](/sample-autonomous-cloud-coding-agents/decisions/adr-017-operator-cli-repo-onboarding)) | Use the CLI path for quick day-2 registration with platform defaults (runtime ARN, GitHub token secret). Use CDK when the repo needs durable infrastructure, custom IAM, Cedar policies, egress rules, or pipeline customization. The onboard command prints notes explaining which platform defaults apply and when a redeploy is still required. -See also: [Using the CLI — operator commands](/sample-autonomous-cloud-coding-agents/using/overview#operator-commands-stack-admin) and [ADR-017](/sample-autonomous-cloud-coding-agents/architecture/adr-017-operator-cli-repo-onboarding). +See also: [Using the CLI — operator commands](/sample-autonomous-cloud-coding-agents/using/using-the-cli#operator-commands-stack-admin) and [ADR-017](/sample-autonomous-cloud-coding-agents/decisions/adr-017-operator-cli-repo-onboarding). ### Blueprint construct diff --git a/docs/src/content/docs/architecture/Security.md b/docs/src/content/docs/architecture/Security.md index 4e850ca0c..e78c0c972 100644 --- a/docs/src/content/docs/architecture/Security.md +++ b/docs/src/content/docs/architecture/Security.md @@ -41,21 +41,21 @@ Three authentication mechanisms protect the platform, matching its input channel **Authorization** is user-scoped: any authenticated user can submit tasks, but users can only view and cancel their own tasks (`user_id` enforcement). Both webhook and API key management enforce ownership with 404 (not 403) to avoid leaking resource existence. A platform API key inherits its creator's `user_id`, so a webhook created via a key is attributed to that owner exactly as an interactive session would be. Key scopes gate which routes a key may call (Phase 1: `webhooks:manage`); reserved scopes (`tasks:read`, `tasks:cancel`) are validated but not yet wired to any route. -**Agent credentials** - GitHub access currently uses a PAT stored in Secrets Manager. The orchestrator reads the secret at hydration time and passes it to the agent runtime. The model never receives the token in its context. Planned: replace the shared PAT with a GitHub App via AgentCore Identity Token Vault, providing per-task, repo-scoped, short-lived tokens (see [GitHub issues](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues), the [ADR-016](/sample-autonomous-cloud-coding-agents/architecture/adr-016-pluggable-identity-and-auth) two-seam design, and the [IDENTITY_AND_AUTH.md](/sample-autonomous-cloud-coding-agents/architecture/identity-and-auth) worked examples). +**Agent credentials** - GitHub access currently uses a PAT stored in Secrets Manager. The orchestrator reads the secret at hydration time and passes it to the agent runtime. The model never receives the token in its context. Planned: replace the shared PAT with a GitHub App via AgentCore Identity Token Vault, providing per-task, repo-scoped, short-lived tokens (see [GitHub issues](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues), the [ADR-016](/sample-autonomous-cloud-coding-agents/decisions/adr-016-pluggable-identity-and-auth) two-seam design, and the [IDENTITY_AND_AUTH.md](/sample-autonomous-cloud-coding-agents/architecture/identity-and-auth) worked examples). **Per-session IAM scoping** - The agent does not use its long-lived compute role (the AgentCore Runtime `ExecutionRole`, ECS Fargate task role, or Lambda MicroVMs execution role) for tenant data. Instead, at task startup it assumes a per-task **SessionRole** via `sts:AssumeRole` with session tags `{user_id, repo, task_id}`, and uses the resulting short-lived credentials for all DynamoDB and S3 tenant-data access. The SessionRole's policies self-constrain on those tags: - **DynamoDB**: item access on the four `task_id`-partitioned tables (task, events, approvals, nudges) is gated by a `dynamodb:LeadingKeys` condition equal to `${aws:PrincipalTag/task_id}` and requires that context key to be present. `Scan` is not granted. The main task table permits reads and only attribute-scoped `UpdateItem` writes: the agent can report status, heartbeat, results and approval details, but cannot replace/delete the row or edit coordinator-owned start receipts, capacity reservations, owner identity or compute handles. The reviewed write list is `cdk/src/constructs/agent-task-write-attributes.json`; Python tests exercise the current writers against it. Events and nudges retain task-scoped item writes. Approval records permit reads and transaction condition checks only. `LeadingKeys` binds the base-table partition key, not a GSI key such as `user_id`. -- **Approval requests**: workers call an IAM-signed API path restricted to their task tag. Its trusted handler can create `PENDING` or conditionally record `TIMED_OUT`; human decisions, notification markers and retention fields are unavailable to the worker. Task ownership/state and MicroVM worker leases are checked transactionally. See the [approval trust boundary](/sample-autonomous-cloud-coding-agents/architecture/cedar-hitl-gates#121-trust-boundaries) and [upgrade procedure](/sample-autonomous-cloud-coding-agents/getting-started/deployment-guide#upgrading-approval-permissions). +- **Approval requests**: workers call an IAM-signed API path restricted to their task tag. Its trusted handler can create `PENDING` or conditionally record `TIMED_OUT`; human decisions, notification markers and retention fields are unavailable to the worker. IAM binds the caller to the tagged task path. The transaction guards against concurrent task ownership/state changes and checks MicroVM worker leases. See the [approval trust boundary](/sample-autonomous-cloud-coding-agents/architecture/cedar-hitl-gates#121-trust-boundaries) and [upgrade procedure](/sample-autonomous-cloud-coding-agents/getting-started/deployment-guide#upgrading-approval-permissions). - **S3**: trace writes and attachment reads are scoped to the `/${aws:PrincipalTag/user_id}/` object prefix. -The compute role retains only non-tenant access (Bedrock model invocation — already ARN-scoped; CloudWatch Logs; the GitHub PAT secret, read once before the SessionRole is assumed; AgentCore Memory) plus `sts:AssumeRole`/`sts:TagSession` on the SessionRole. Because the agent runs under credentials that are themselves an assumed role, its `AssumeRole` is *role chaining* — capped at one hour regardless of the role's max session duration — so the agent uses a **refreshable** credential provider that re-assumes before expiry (tasks can run up to the 8-hour `maxLifetime`). The design is backend-agnostic: the same SessionRole and agent code serve all three backends (AgentCore Runtime, ECS Fargate, Lambda MicroVMs). Existing scoped credentials are constrained to their tags, but the compute role chooses those tags when assuming the role. The trust policy does not independently bind a worker to one task; this is not complete isolation against a compromised worker with ambient credentials. No worker needs direct access to the shared capacity counter. ECS's legacy configuration without a session role retains unscoped task reads/reporting updates, with the same attribute restriction. The policy structure and Python writer compatibility are tested locally. A historical IAM simulator check covered the original tag conditions; it does not verify the new attribute restriction or deployed transaction authorization. The [acceptance checklist](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/docs/verification/README.md#live-acceptance-for-an-installation) includes effective-role authorization checks. +The compute role retains only non-tenant access (Bedrock model invocation — already ARN-scoped; CloudWatch Logs; the GitHub PAT secret, read once before the SessionRole is assumed; AgentCore Memory) plus `sts:AssumeRole`/`sts:TagSession` on the SessionRole. Because the agent runs under credentials that are themselves an assumed role, its `AssumeRole` is *role chaining* — capped at one hour regardless of the role's max session duration — so the agent uses a **refreshable** credential provider that re-assumes before expiry (tasks can run up to the 8-hour `maxLifetime`). The design is backend-agnostic: the same SessionRole and agent code serve all three backends (AgentCore Runtime, ECS Fargate, Lambda MicroVMs). Existing scoped credentials are constrained to their tags, but the compute role chooses those tags when assuming the role. The trust policy does not independently bind a worker to one task; this is not complete isolation against a compromised worker with ambient credentials. No worker needs direct access to the shared capacity counter. ECS's configuration without a session role retains unscoped task reads/reporting updates, with the same attribute restriction, but cannot enable approval gates; the construct rejects approval wiring without the session role and service URL. The policy structure and Python writer compatibility are tested locally. A historical IAM simulator check covered the original tag conditions; it does not verify the new attribute restriction or deployed transaction authorization. The [acceptance checklist](/sample-autonomous-cloud-coding-agents/verification/readme#live-acceptance-for-an-installation) includes effective-role authorization checks. **Lambda MicroVMs compute-role delta** — The compute role additionally reads the per-workspace channel-OAuth secrets (`bgagent-linear-oauth-*`, `bgagent-jira-oauth-*`) and its payload bucket's `bootstrap/*` manifests. Ambient task-object reads and payload-bucket listing are explicitly denied. It also has read-only `ec2:DescribeAvailabilityZones` for repository CDK synthesis; that API has no resource-level scope. -Recorded live calls rejected source-conditioned MicroVM role trust and service-conditioned PassRole grants. The integration therefore omits those conditions on the affected paths, while restricting which roles can be passed and which resources they may access. Reintroduce conditions only after verifying service support. [ADR-021 §4](/sample-autonomous-cloud-coding-agents/architecture/adr-021-lambda-microvms-compute-backend#4-infra-and-iam-conditional-resources-behind-bootstrap-computetypes) records the role responsibilities and limits. +Recorded live calls rejected source-conditioned MicroVM role trust and service-conditioned PassRole grants. The integration therefore omits those conditions on the affected paths, while restricting which roles can be passed and which resources they may access. Reintroduce conditions only after verifying service support. [ADR-021 §4](/sample-autonomous-cloud-coding-agents/decisions/adr-021-lambda-microvms-compute-backend#4-infra-and-iam-conditional-resources-behind-bootstrap-computetypes) records the role responsibilities and limits. -**ECS/MicroVM task bootstrap (v2)** — The coordinator publishes non-secret deployment manifests and supplies a short-lived signed URL for exactly one task's payload. The worker reads the manifest with its ambient role, whose explicit deny outside its own `bootstrap/*` prevents a foreign public bucket from authorizing fake configuration. It verifies downloaded task identity and exact manifest/config agreement before installing MicroVM settings. The coordinator privately persists the URL outside TaskTable for retry and deletes both task objects at finalization. Worker roles cannot read/list task objects using their ambient credentials. A signed URL is a bearer capability: redact it in errors and keep it out of task rows and repository subprocess environments. Existing session-tag trust and other platform grants remain separate limitations. The [payload guide](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/docs/verification/645-payload-bootstrap.md) describes coordinated image/coordinator/policy upgrades; the [acceptance summary](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/docs/verification/README.md) distinguishes recorded AWS checks from remaining validation. +**ECS/MicroVM task bootstrap (v2)** — The coordinator publishes non-secret deployment manifests and supplies a short-lived signed URL for exactly one task's payload. The worker reads the manifest with its ambient role, whose explicit deny outside its own `bootstrap/*` prevents a foreign public bucket from authorizing fake configuration. It verifies downloaded task identity and exact manifest/config agreement before installing MicroVM settings. The coordinator privately persists the URL outside TaskTable for retry and deletes both task objects at finalization. Worker roles cannot read/list task objects using their ambient credentials. A signed URL is a bearer capability: redact it in errors and keep it out of task rows and repository subprocess environments. Existing session-tag trust and other platform grants remain separate limitations. The [payload guide](/sample-autonomous-cloud-coding-agents/verification/645-payload-bootstrap) describes coordinated image/coordinator/policy upgrades; the [acceptance summary](/sample-autonomous-cloud-coding-agents/verification/readme) distinguishes recorded AWS checks from remaining validation. > Out of scope for this control and tracked separately as GitHub issues: replacing the shared GitHub PAT (GitHub App / Token Vault), binding credentials to the MicroVM via attestation, and scoping AgentCore Memory (namespace isolation by `actorId`/`sessionId` remains its boundary). diff --git a/docs/src/content/docs/architecture/Vision.md b/docs/src/content/docs/architecture/Vision.md index f003616d4..f6d8a5f5c 100644 --- a/docs/src/content/docs/architecture/Vision.md +++ b/docs/src/content/docs/architecture/Vision.md @@ -7,7 +7,7 @@ title: Vision This document states the long-term direction of **ABCA (Autonomous Background Coding Agents on AWS)** and the **tenets** that should guide design, implementation, and review. Use it when evaluating pull requests, RFCs, and ADRs: if a change clearly advances the vision and respects the tenets, it belongs; if it trades tenets away without an explicit, documented rationale, it needs more discussion. - **Use this doc for:** alignment checks in review — “does this fit where we are going?” -- **Not a substitute for:** [ARCHITECTURE.md](/sample-autonomous-cloud-coding-agents/architecture/architecture) (system shape), [GitHub issues](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues) (planned work and priorities), or [docs/decisions/](../decisions/) (specific accepted choices). +- **Not a substitute for:** [ARCHITECTURE.md](/sample-autonomous-cloud-coding-agents/architecture/architecture) (system shape), [GitHub issues](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues) (planned work and priorities), or [docs/decisions/](/sample-autonomous-cloud-coding-agents/decisions/readme) (specific accepted choices). ## Vision @@ -17,7 +17,7 @@ We are building toward **lights-sparse**, **graduated** autonomy (defined below) ### What "lights-sparse" means -**Lights-sparse** is project vocabulary (not general industry jargon): it names the autonomy posture ABCA targets today, drawn from the **software dark factory** analogy in the [introduction](/sample-autonomous-cloud-coding-agents/architecture/index). +**Lights-sparse** is project vocabulary (not general industry jargon): it names the autonomy posture ABCA targets today, drawn from the **software dark factory** analogy in the [introduction](/sample-autonomous-cloud-coding-agents/). - **Lights-out** (the analogy’s end state): humans set goals, policy, and constraints; production runs without people on the floor. - **Lights-sparse** (where teams are now): the **implementation loop** — edit code, run tests, open pull requests — is increasingly **unattended**, while **governance, merge authority, and production release** stay **supervised**. Humans are not at the keyboard for every step; they are still accountable for what ships. @@ -148,7 +148,7 @@ These are out of scope for the project vision. Proposals that primarily serve th | [SECURITY.md](/sample-autonomous-cloud-coding-agents/architecture/security) | Threat model and controls (tenets 4–5 in depth) | | [CEDAR_HITL_GATES.md](/sample-autonomous-cloud-coding-agents/architecture/cedar-hitl-gates) | HITL approval gates, pre-approve scopes, graduated in-run autonomy | | [INTERACTIVE_AGENTS.md](/sample-autonomous-cloud-coding-agents/architecture/interactive-agents) | Async UX, watch/nudge, notification plane, approval state machine | -| [docs/decisions/](../decisions/) | Recorded choices when tenets conflict or ambiguity is resolved | -| [docs/src/content/docs/index.md](/sample-autonomous-cloud-coding-agents/architecture/index) (synced intro) | Public-facing narrative including dark-factory attribute table | +| [docs/decisions/](/sample-autonomous-cloud-coding-agents/decisions/readme) | Recorded choices when tenets conflict or ambiguity is resolved | +| [docs/src/content/docs/index.md](/sample-autonomous-cloud-coding-agents/) (synced intro) | Public-facing narrative including dark-factory attribute table | When tenets and architecture principles overlap, **tenets win for review judgment**; **architecture and ADRs win for implementation detail** once a direction is chosen. diff --git a/docs/src/content/docs/architecture/Workflows.md b/docs/src/content/docs/architecture/Workflows.md index a05ee80c4..2da69fb26 100644 --- a/docs/src/content/docs/architecture/Workflows.md +++ b/docs/src/content/docs/architecture/Workflows.md @@ -10,7 +10,7 @@ The three former task types — `new_task`, `pr_iteration`, `pr_review` — are - **Use this doc for:** the workflow file schema, step-kind catalog, the agent-side step runner model, and how a `workflow_ref` flows from API to agent. - **Related docs:** [ARCHITECTURE.md](/sample-autonomous-cloud-coding-agents/architecture/architecture) for the deterministic-steps-wrapping-one-agentic-step model, [ORCHESTRATOR.md](/sample-autonomous-cloud-coding-agents/architecture/orchestrator) for the durable lifecycle the workflow runs inside, [REPO_ONBOARDING.md](/sample-autonomous-cloud-coding-agents/architecture/repo-onboarding) for the per-repo **Blueprint** (a distinct concept — see [Naming](#naming-workflow-vs-blueprint)), [CEDAR_HITL_GATES.md](/sample-autonomous-cloud-coding-agents/architecture/cedar-hitl-gates) for the policy engine a workflow's `agent_config` feeds, [SECURITY.md](/sample-autonomous-cloud-coding-agents/architecture/security) for tool tiers, and [API_CONTRACT.md](/sample-autonomous-cloud-coding-agents/architecture/api-contract) for the `workflow_ref` wire field. -- **Decision record:** [ADR-014](/sample-autonomous-cloud-coding-agents/architecture/adr-014-workflow-driven-tasks). +- **Decision record:** [ADR-014](/sample-autonomous-cloud-coding-agents/decisions/adr-014-workflow-driven-tasks). - **Tracking issue:** [#248](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/248). Pairs with the agent asset registry ([#246](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/246)) and attribution ([#245](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/245)). Scoped-down, current-architecture track of the broader AKW vision ([#99](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/99)). ## Background: what workflows replaced @@ -206,7 +206,7 @@ This second example is the **target shape** for repo-less execution — the acce ## The agent-side step runner -Per [ADR-014](/sample-autonomous-cloud-coding-agents/architecture/adr-014-workflow-driven-tasks), the runner is **agent-side**: it lives in the container and interprets `workflow.steps`. The orchestrator's durable shape (`admission-control → pre-flight → hydrate-context → start-session → await-agent-completion → finalize`) is unchanged — the workflow drives *what happens inside* the `RUNNING` state, not the platform lifecycle. This keeps the blast radius off durable orchestration and matches the issue's "executes steps in order *inside the container*." +Per [ADR-014](/sample-autonomous-cloud-coding-agents/decisions/adr-014-workflow-driven-tasks), the runner is **agent-side**: it lives in the container and interprets `workflow.steps`. The orchestrator's durable shape (`admission-control → pre-flight → hydrate-context → start-session → await-agent-completion → finalize`) is unchanged — the workflow drives *what happens inside* the `RUNNING` state, not the platform lifecycle. This keeps the blast radius off durable orchestration and matches the issue's "executes steps in order *inside the container*." ```python # agent/src/workflow/runner.py (shape, not final code) @@ -231,7 +231,7 @@ The step runner runs inside the compute substrate, which is **not** a throwaway - **Step completion is checkpointed; resume skips completed steps.** The runner records each step's outcome to a small `workflow_state.json` on the persistent mount (`/mnt/workspace`) as it goes. On resume (orchestrator re-invokes the same session, or — when shipped — a replacement worker rehydrates from the [S3-backed SDK session store](#relationship-to-portable-resume)), the runner reads that checkpoint, **skips already-completed deterministic steps** (`clone_repo` need not re-clone a populated `/workspace`; a completed `verify_build` is not re-run), and **resumes the agent loop** via the persisted SDK session UUID rather than restarting it from turn 0. This is the same property the orchestrator already relies on for session start being idempotent (pre-generated, reused session id). - **Side-effecting steps remain idempotent.** Independent of resume, `clone_repo`, `ensure_pr`, `post_review`, and `deliver_artifact` must tolerate a partial prior run (a resume can re-enter the step that was in flight when the worker died). Each documents its idempotency key — PR branch, review id, artifact S3 key = `task_id` — so re-entry reconciles rather than duplicates (today's `ensure_pr` already does this: it checks `gh pr view` before creating). - **`on_failure: continue` is forbidden after side effects** (validation rule 10). A failed `ensure_pr` (commits pushed, PR-create failed) must not reach a *succeeded* terminal — committed work with no PR and no compensation. `continue` is permitted only for non-side-effecting, advisory steps (e.g. an informational `verify_lint`). `skip_remaining` ends the workflow cleanly and runs terminal-outcome resolution against whatever completed; `fail` (default) is terminal `FAILED`. -- **Granularity boundary.** Resume is *workflow-step granular on the agent side*, not a new orchestrator-side durable checkpoint per step — the orchestrator still treats the whole session as one `await-agent-completion` step, so platform invariants stay agent-external ([ADR-014](/sample-autonomous-cloud-coding-agents/architecture/adr-014-workflow-driven-tasks)). What changes versus today is that the agent-side runner makes its *own* progress recoverable across a stop/resume, which today's monolithic `run_task` does not. +- **Granularity boundary.** Resume is *workflow-step granular on the agent side*, not a new orchestrator-side durable checkpoint per step — the orchestrator still treats the whole session as one `await-agent-completion` step, so platform invariants stay agent-external ([ADR-014](/sample-autonomous-cloud-coding-agents/decisions/adr-014-workflow-driven-tasks)). What changes versus today is that the agent-side runner makes its *own* progress recoverable across a stop/resume, which today's monolithic `run_task` does not. #### Relationship to portable resume @@ -297,7 +297,7 @@ Scope discipline for #248: **`github` is the only implemented provider** — add Read-only is enforced by Cedar hard-deny rules. **As of #248 Phase 2a** these key off the `context.read_only` attribute (`read_only_forbid_write`, `read_only_forbid_edit`), not a principal literal — and `read_only: true` *also* makes the runner drop `Write`/`Edit` from the SDK `allowed_tools` list. Two layers: - **Defense in depth.** `read_only: true` makes the runner drop `Write`/`Edit` from `allowed_tools` *and* sends `context.read_only == true` on every Cedar request — closing the earlier gap where read-only was enforced only by a Cedar string-match on the principal, not by the tool list. -- **Property-keyed enforcement (security-relevant — was precise, not hand-waved).** Read-only enforcement attaches to the *property*, not a per-task-type literal: the principal keeps the legacy `Agent::TaskAgent::""` identity scheme (audit/attribution only), while the two hard-deny rules forbid `Write`/`Edit` **whenever `context.read_only == true`**. So the deny applies uniformly to *every* read-only workflow — not just `coding/pr-review` — and there is no literal a new read-only workflow could fail to match. This was a deliberate, recorded behavior change (see [ADR-014](/sample-autonomous-cloud-coding-agents/architecture/adr-014-workflow-driven-tasks) addendum 2026-06-08), gated by the `contracts/cedar-parity/` fixtures (`read-only-forbid-write`, `read-only-forbid-edit`, `read-only-false-permits-write`) run against *both* the `cedarpy` and `cedar-wasm` engines. +- **Property-keyed enforcement (security-relevant — was precise, not hand-waved).** Read-only enforcement attaches to the *property*, not a per-task-type literal: the principal keeps the legacy `Agent::TaskAgent::""` identity scheme (audit/attribution only), while the two hard-deny rules forbid `Write`/`Edit` **whenever `context.read_only == true`**. So the deny applies uniformly to *every* read-only workflow — not just `coding/pr-review` — and there is no literal a new read-only workflow could fail to match. This was a deliberate, recorded behavior change (see [ADR-014](/sample-autonomous-cloud-coding-agents/decisions/adr-014-workflow-driven-tasks) addendum 2026-06-08), gated by the `contracts/cedar-parity/` fixtures (`read-only-forbid-write`, `read-only-forbid-edit`, `read-only-false-permits-write`) run against *both* the `cedarpy` and `cedar-wasm` engines. This is the migration step where an error *silently weakens* enforcement (the rule stops matching) rather than failing loudly. The original plan was to ship it as an isolated PR ahead of the Phase 2b workflow migrations; because 2b shipped first behind a `read_only ⇒ "pr_review"` principal bridge (so read-only was never unprotected), Phase 2a instead removes that bridge and lands the property-keyed rules + parity fixtures together on the #248 branch. See the ADR-014 addendum and [Phasing](#phasing). @@ -311,7 +311,7 @@ Registry-sourced `cedar_policy_modules` / `mcp_servers` are trusted content load ### Authorship & governance -A workflow file selects the agent's tool surface and policy posture, so **who may publish a `production` workflow is a trust decision, not a convenience**. Per [ADR-003](/sample-autonomous-cloud-coding-agents/architecture/adr-003-contribution-governance), publishing or promoting a first-party workflow follows the same issue → approval → review → merge path as any code change — a workflow YAML in `agent/workflows/**` is reviewed like code, and the synth-time validator (the [validation rules](#validation-rules)) is a required CI gate. When the registry (#246) makes workflows publishable out-of-band, publish/promote ACLs are Cedar-governed per #246 Phase 3; until then, the only way a `production` workflow exists is through a reviewed merge. The `description`/`guidance` discovery fields are author-controlled free text; when they feed an agent's workflow-*selection* context (Phase 4), they are treated as untrusted-external input and screened like other hydrated content. +A workflow file selects the agent's tool surface and policy posture, so **who may publish a `production` workflow is a trust decision, not a convenience**. Per [ADR-003](/sample-autonomous-cloud-coding-agents/decisions/adr-003-contribution-governance), publishing or promoting a first-party workflow follows the same issue → approval → review → merge path as any code change — a workflow YAML in `agent/workflows/**` is reviewed like code, and the synth-time validator (the [validation rules](#validation-rules)) is a required CI gate. When the registry (#246) makes workflows publishable out-of-band, publish/promote ACLs are Cedar-governed per #246 Phase 3; until then, the only way a `production` workflow exists is through a reviewed merge. The `description`/`guidance` discovery fields are author-controlled free text; when they feed an agent's workflow-*selection* context (Phase 4), they are treated as untrusted-external input and screened like other hydrated content. ## Wire contract: `workflow_ref` from API to agent @@ -437,7 +437,7 @@ So: JSON Schema = canonical shape, consumed not copied; cross-field rules = one ## Promotion is earned, not set -`status: production` is not a label an author flips — it is a state a version *earns* by passing its declared `promotion_gate`. This makes the promotion lifecycle (`draft → validated → production → deprecated`) a machine-checked quality gate rather than a human's say-so, and it slots directly onto the existing [tiered validation pyramid](/sample-autonomous-cloud-coding-agents/architecture/adr-013-tiered-validation-pyramid): +`status: production` is not a label an author flips — it is a state a version *earns* by passing its declared `promotion_gate`. This makes the promotion lifecycle (`draft → validated → production → deprecated`) a machine-checked quality gate rather than a human's say-so, and it slots directly onto the existing [tiered validation pyramid](/sample-autonomous-cloud-coding-agents/decisions/adr-013-tiered-validation-pyramid): | Workflow status | Gate that must pass | Validation tier (ADR-013) | |---|---|---| @@ -478,10 +478,10 @@ Adapted from the issue's phases (the issue framed Phase 1 as a `task_type` *alia | Phase | Deliverable | Primary files | |---|---|---| -| 0 | This design doc + [ADR-014](/sample-autonomous-cloud-coding-agents/architecture/adr-014-workflow-driven-tasks) + JSON Schema + step-runner skeleton | `docs/design/WORKFLOWS.md`, `docs/decisions/`, `agent/workflows/schema/` | +| 0 | This design doc + [ADR-014](/sample-autonomous-cloud-coding-agents/decisions/adr-014-workflow-driven-tasks) + JSON Schema + step-runner skeleton | `docs/design/WORKFLOWS.md`, `docs/decisions/`, `agent/workflows/schema/` | | 1 | Step runner + `default/agent-v1` + migrate `new_task` to a workflow file; introduce `workflow_ref` and **remove the `task_type` enum** end-to-end (API/CLI/agent); the single workflow validator + `contracts/workflow-validation/` golden corpus | `agent/src/workflow/`, `agent/workflows/coding/new-task-v1.yaml`, `cdk/src/handlers/`, `cli/src/`, `contracts/workflow-validation/` | | 2b | Migrate `pr_iteration`, `pr_review` onto workflows behind a `read_only ⇒ "pr_review"` principal bridge (read-only stays enforced by the existing literal rules throughout) | `agent/workflows/coding/*`, `agent/tests/` | -| 2a | **Cedar property-keyed read-only migration** — literal `"pr_review"` hard-deny → `context.read_only == true` rules (`read_only_forbid_write/edit`), threaded via `context.read_only`; removes the 2b bridge; adds `read-only-*` `contracts/cedar-parity/` fixtures verified on *both* engines. (Originally planned as an isolated PR ahead of 2b; reordered after 2b shipped first behind the bridge — see [ADR-014](/sample-autonomous-cloud-coding-agents/architecture/adr-014-workflow-driven-tasks) addendum.) | `agent/policies/`, `cdk/src/handlers/shared/builtin-policies.ts`, `contracts/cedar-parity/`, `agent/src/policy.py`, `agent/src/workflow/loader.py` | +| 2a | **Cedar property-keyed read-only migration** — literal `"pr_review"` hard-deny → `context.read_only == true` rules (`read_only_forbid_write/edit`), threaded via `context.read_only`; removes the 2b bridge; adds `read-only-*` `contracts/cedar-parity/` fixtures verified on *both* engines. (Originally planned as an isolated PR ahead of 2b; reordered after 2b shipped first behind the bridge — see [ADR-014](/sample-autonomous-cloud-coding-agents/decisions/adr-014-workflow-driven-tasks) addendum.) | `agent/policies/`, `cdk/src/handlers/shared/builtin-policies.ts`, `contracts/cedar-parity/`, `agent/src/policy.py`, `agent/src/workflow/loader.py` | | 3 | Repo-optional `web_research` workflow (the repo-optional refactor — see [the requires_repo note](#domain--requires_repo)) | `cdk/src/handlers/`, `agent/workflows/knowledge/` | | 4 | Registry-native workflows (#246); Blueprint workflow allow-list + `default_workflow`; inline/repo-local for dev | depends on #246 | @@ -491,7 +491,7 @@ Per #248, the following remain out of scope (deferred to #99 / separate issues): ## Open questions -These are genuine forks; the repo-optional items (1–2) were **prerequisites for Phase 3** and have been **resolved as recorded decisions in the [ADR-014](/sample-autonomous-cloud-coding-agents/architecture/adr-014-workflow-driven-tasks) addendum (2026-06-08)**, with the one implied schema reshape applied — so the Phase-0 schema is now **frozen**. They are kept here (struck-through) for traceability. +These are genuine forks; the repo-optional items (1–2) were **prerequisites for Phase 3** and have been **resolved as recorded decisions in the [ADR-014](/sample-autonomous-cloud-coding-agents/decisions/adr-014-workflow-driven-tasks) addendum (2026-06-08)**, with the one implied schema reshape applied — so the Phase-0 schema is now **frozen**. They are kept here (struck-through) for traceability. 1. ~~**Memory actorId for repo-less tasks.**~~ **RESOLVED (ADR-014 addendum):** per-user `actorId = user:{cognito_sub}` (caller-scoped, no cross-tenant bleed; mirrors the per-user trace prefix). Cross-workflow knowledge pooling is explicitly not adopted. **No schema field added** (fixed platform fallback, not author-configurable) — a Phase-3 `memory.py` change keys on `user:{user_id}` when `repo` is absent. Coordinate with [MEMORY.md](/sample-autonomous-cloud-coding-agents/architecture/memory). 2. ~~**Artifact delivery contract.**~~ **RESOLVED (ADR-014 addendum):** `deliver_artifact.target` is an **open string naming a registered Python deliverer** (`workflow/deliverers.py` → `DELIVERERS`), not a closed enum — new delivery methods are registered deliverers, not schema changes. Shared plumbing is **pinned**: task-scoped key `artifacts/{task_id}/`, a prefix-scoped SessionRole IAM grant, a per-artifact size limit, and `TaskDetail` URL surfacing; the SessionRole `repo` tenant tag gains a `workflow:{id}` repo-less form. Each deliverer declares the outcomes it `produces`; validator rule 11 consults that registry. Implementations land in Phase 3; only the contract is frozen here. @@ -506,4 +506,4 @@ This design is a scoped-down reconciliation of the unmerged AKW port on `origin/ Two refinements layered on top of that port are worth calling out, because they shape the schema: 1. **Discovery is separate from execution.** A workflow carries optional `description` / `guidance` fields — a human- and agent-readable selection surface for registry search and workflow-selection (#246) — kept distinct from the machine-facing `prompt`. -2. **Promotion is earned, not set** — see [Promotion is earned, not set](#promotion-is-earned-not-set). `production` is gated by a declared `promotion_gate`, reusing the [ADR-013](/sample-autonomous-cloud-coding-agents/architecture/adr-013-tiered-validation-pyramid) validation pyramid rather than being a label an author flips. +2. **Promotion is earned, not set** — see [Promotion is earned, not set](#promotion-is-earned-not-set). `production` is gated by a declared `promotion_gate`, reusing the [ADR-013](/sample-autonomous-cloud-coding-agents/decisions/adr-013-tiered-validation-pyramid) validation pyramid rather than being a label an author flips. diff --git a/docs/src/content/docs/customizing/Cedar-policies.md b/docs/src/content/docs/customizing/Cedar-policies.md index 73db677c7..a04540221 100644 --- a/docs/src/content/docs/customizing/Cedar-policies.md +++ b/docs/src/content/docs/customizing/Cedar-policies.md @@ -6,7 +6,7 @@ title: Cedar policy guide This guide is for **blueprint authors** — repo owners writing the Cedar policies that govern what tool calls the agent can make unattended versus which ones pause for human approval. -> **If you are a task submitter** looking for how approvals work at the CLI, see [User guide — Approval gates](/sample-autonomous-cloud-coding-agents/using/overview#approval-gates-cedar-hitl). This guide is about *writing* the rules that cause approvals. +> **If you are a task submitter** looking for how approvals work at the CLI, see [User guide — Approval gates](/sample-autonomous-cloud-coding-agents/using/approval-gates-cedar-hitl). This guide is about *writing* the rules that cause approvals. > > **For the full design** (fail-closed posture, engine internals, concurrency), see [Cedar HITL gates design doc](/sample-autonomous-cloud-coding-agents/architecture/cedar-hitl-gates). @@ -47,7 +47,7 @@ security: approvalGateCap: 50 # optional per-task gate budget (1–500, default 50) ``` -The built-in rule set is documented in [`agent/policies/hard_deny.cedar`](../../agent/policies/hard_deny.cedar) and [`agent/policies/soft_deny.cedar`](../../agent/policies/soft_deny.cedar). Run `bgagent policies list --repo owner/repo` against a deployed stack to see the effective rules for a repo. +The built-in rule set is documented in [`agent/policies/hard_deny.cedar`](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/agent/policies/hard_deny.cedar) and [`agent/policies/soft_deny.cedar`](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/agent/policies/soft_deny.cedar). Run `bgagent policies list --repo owner/repo` against a deployed stack to see the effective rules for a repo. ## Vocabulary @@ -162,12 +162,12 @@ Fix the blueprint, redeploy (or update the `blueprint.yaml` if you're using a pu ## Testing policies before shipping -Every repo blueprint is covered by **cross-engine parity fixtures** in [`contracts/cedar-parity/`](../../contracts/cedar-parity/). Before shipping a non-trivial rule change, drop a golden-file fixture that pins the expected `(decision, matching_rule_ids)` for a representative `(policies, input)` pair. Both the Python `cedarpy` engine and the TypeScript `@cedar-policy/cedar-wasm` engine run it — divergence fails CI. See the directory's README for the fixture schema. +Every repo blueprint is covered by **cross-engine parity fixtures** in [`contracts/cedar-parity/`](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/tree/main/contracts/cedar-parity). Before shipping a non-trivial rule change, drop a golden-file fixture that pins the expected `(decision, matching_rule_ids)` for a representative `(policies, input)` pair. Both the Python `cedarpy` engine and the TypeScript `@cedar-policy/cedar-wasm` engine run it — divergence fails CI. See the directory's README for the fixture schema. -For unit coverage of your own rules without the cross-engine guarantee, add a case to [`agent/tests/test_policy.py`](../../agent/tests/test_policy.py) using `PolicyEngine.evaluate_tool_use(...)`. +For unit coverage of your own rules without the cross-engine guarantee, add a case to [`agent/tests/test_policy.py`](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/agent/tests/test_policy.py) using `PolicyEngine.evaluate_tool_use(...)`. ## Where to look next - [`docs/design/CEDAR_HITL_GATES.md`](/sample-autonomous-cloud-coding-agents/architecture/cedar-hitl-gates) — full design: engine internals, fail-closed posture, late-approval races, concurrency. -- [`agent/policies/hard_deny.cedar`](../../agent/policies/hard_deny.cedar) + [`agent/policies/soft_deny.cedar`](../../agent/policies/soft_deny.cedar) — the built-in rule set, good starting point for copy-paste. -- [User guide — Approval gates](/sample-autonomous-cloud-coding-agents/using/overview#approval-gates-cedar-hitl) — the CLI side (`bgagent pending` / `approve` / `deny` / `policies`). +- [`agent/policies/hard_deny.cedar`](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/agent/policies/hard_deny.cedar) + [`agent/policies/soft_deny.cedar`](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/agent/policies/soft_deny.cedar) — the built-in rule set, good starting point for copy-paste. +- [User guide — Approval gates](/sample-autonomous-cloud-coding-agents/using/approval-gates-cedar-hitl) — the CLI side (`bgagent pending` / `approve` / `deny` / `policies`). diff --git a/docs/src/content/docs/decisions/Adr-002-least-privilege-bootstrap-policies.md b/docs/src/content/docs/decisions/Adr-002-least-privilege-bootstrap-policies.md index befa6db4b..2e39f10d8 100644 --- a/docs/src/content/docs/decisions/Adr-002-least-privilege-bootstrap-policies.md +++ b/docs/src/content/docs/decisions/Adr-002-least-privilege-bootstrap-policies.md @@ -68,7 +68,7 @@ The implementation is decomposed into 8 sub-issues, each independently reviewabl ## References -- [ADR-001](/sample-autonomous-cloud-coding-agents/architecture/adr-001-stacked-pull-requests) — delivery methodology (stacked PRs) +- [ADR-001](/sample-autonomous-cloud-coding-agents/decisions/adr-001-stacked-pull-requests) — delivery methodology (stacked PRs) - RFC #120 — parent issue with full design and sub-issue breakdown - `docs/design/DEPLOYMENT_ROLES.md` — current documentation (will become generated) - PR #46 — original policy derivation and validation methodology diff --git a/docs/src/content/docs/decisions/Adr-004-tabula-rasa-documentation.md b/docs/src/content/docs/decisions/Adr-004-tabula-rasa-documentation.md index 697d4ef34..6f818da23 100644 --- a/docs/src/content/docs/decisions/Adr-004-tabula-rasa-documentation.md +++ b/docs/src/content/docs/decisions/Adr-004-tabula-rasa-documentation.md @@ -48,7 +48,7 @@ Never force a novice to read expert material to proceed. Never force an expert t ### Self-contained references When referencing another document: -- State what the reader gets from it: "See [Deployment Guide](link) for AWS account setup (required before this step)" +- State what the reader gets from it: "See [Deployment Guide](/sample-autonomous-cloud-coding-agents/getting-started/deployment-guide) for AWS account setup (required before this step)" - Never assume the reader has read it - Never use "as mentioned above" — each section must stand alone after context compaction diff --git a/docs/src/content/docs/decisions/Adr-012-operational-knowledge-stack.md b/docs/src/content/docs/decisions/Adr-012-operational-knowledge-stack.md index 5c3399b3b..d12a38a8c 100644 --- a/docs/src/content/docs/decisions/Adr-012-operational-knowledge-stack.md +++ b/docs/src/content/docs/decisions/Adr-012-operational-knowledge-stack.md @@ -155,7 +155,7 @@ Organized by persona: ```markdown # Contributor Workflow -> Operationalizes [ADR-003](/sample-autonomous-cloud-coding-agents/architecture/adr-003-contribution-governance) +> Operationalizes [ADR-003](/sample-autonomous-cloud-coding-agents/decisions/adr-003-contribution-governance) ## For Planners - Issue quality bar (what makes an issue "ready") diff --git a/docs/src/content/docs/decisions/Adr-014-workflow-driven-tasks.md b/docs/src/content/docs/decisions/Adr-014-workflow-driven-tasks.md index 099bbc0e6..bf9a3068f 100644 --- a/docs/src/content/docs/decisions/Adr-014-workflow-driven-tasks.md +++ b/docs/src/content/docs/decisions/Adr-014-workflow-driven-tasks.md @@ -106,5 +106,5 @@ With both resolved and the one schema reshape applied, the Phase-0 schema is **f - [docs/design/REPO_ONBOARDING.md](/sample-autonomous-cloud-coding-agents/architecture/repo-onboarding) — the Blueprint construct and `step_sequence` model - [docs/design/CEDAR_HITL_GATES.md](/sample-autonomous-cloud-coding-agents/architecture/cedar-hitl-gates) — policy engine the `agent_config` feeds - Prior art: `origin/merge/akw-integration` (commit `9d066a8`) — AKW YAML registry and models (reconciled, scoped down) -- [ADR-013](/sample-autonomous-cloud-coding-agents/architecture/adr-013-tiered-validation-pyramid) — the validation pyramid the `promotion_gate` layers onto -- [ADR-005](/sample-autonomous-cloud-coding-agents/architecture/adr-005-feedback-loop) — the feedback loop that workflow trajectory-evolution would extend (future, out of scope) +- [ADR-013](/sample-autonomous-cloud-coding-agents/decisions/adr-013-tiered-validation-pyramid) — the validation pyramid the `promotion_gate` layers onto +- [ADR-005](/sample-autonomous-cloud-coding-agents/decisions/adr-005-feedback-loop) — the feedback loop that workflow trajectory-evolution would extend (future, out of scope) diff --git a/docs/src/content/docs/decisions/Adr-016-pluggable-identity-and-auth.md b/docs/src/content/docs/decisions/Adr-016-pluggable-identity-and-auth.md index 147a62e9d..ee2d596e9 100644 --- a/docs/src/content/docs/decisions/Adr-016-pluggable-identity-and-auth.md +++ b/docs/src/content/docs/decisions/Adr-016-pluggable-identity-and-auth.md @@ -94,7 +94,7 @@ The abstraction is intentionally a contract, not a forklift of credential handli - **Backend-agnostic.** One `resolve__token()` contract serves both the AgentCore Runtime backend (token arrives via the `WorkloadAccessToken` header) and the parked ECS backend (in-process boto3). The backend decides how the token arrives; the seam does not care. - **Incremental.** The rollout is phased and flag-gated, and the shared PAT fallback stays until the vault path is green. No big-bang cutover. -- **Consistent with ADR-014.** This is the credential-plane analog of [ADR-014](/sample-autonomous-cloud-coding-agents/architecture/adr-014-workflow-driven-tasks)'s provider-neutral `VcsProvider` seam: that one named GitHub-specific control-plane operations as instances of generic concepts; this one names the per-integration credential resolvers as instances of one outbound contract. +- **Consistent with ADR-014.** This is the credential-plane analog of [ADR-014](/sample-autonomous-cloud-coding-agents/decisions/adr-014-workflow-driven-tasks)'s provider-neutral `VcsProvider` seam: that one named GitHub-specific control-plane operations as instances of generic concepts; this one names the per-integration credential resolvers as instances of one outbound contract. ## Credential types @@ -246,7 +246,7 @@ revoking a grant — it changes who stores and refreshes the token, not whether - Issue [#215](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/215) — Bedrock billing attribution - Issue [#237](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/237) — governance planes / `abca.audit.v1` correlation block - Issue [#288](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/288) / PR [#302](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/302) — Jira integration (second provider through the resolver seam) -- [ADR-014](/sample-autonomous-cloud-coding-agents/architecture/adr-014-workflow-driven-tasks) — workflow-driven tasks; introduced the provider-neutral `VcsProvider` seam this ADR is the credential-plane analog of +- [ADR-014](/sample-autonomous-cloud-coding-agents/decisions/adr-014-workflow-driven-tasks) — workflow-driven tasks; introduced the provider-neutral `VcsProvider` seam this ADR is the credential-plane analog of - [IDENTITY_AND_AUTH.md](/sample-autonomous-cloud-coding-agents/architecture/identity-and-auth) — the worked use-cases, seams table, decision tree, and Linear before/after - [SECURITY.md](/sample-autonomous-cloud-coding-agents/architecture/security) — current auth posture, the shared-PAT limitation this ADR resolves - GitHub issues — per-repo GitHub credentials, layered credential derivation, delegation chain propagation (priority labels `P0`, `P1`, etc.) diff --git a/docs/src/content/docs/decisions/Adr-019-agentcore-gateway-tool-federation.md b/docs/src/content/docs/decisions/Adr-019-agentcore-gateway-tool-federation.md index a9e061b15..34f70f0af 100644 --- a/docs/src/content/docs/decisions/Adr-019-agentcore-gateway-tool-federation.md +++ b/docs/src/content/docs/decisions/Adr-019-agentcore-gateway-tool-federation.md @@ -9,9 +9,9 @@ title: Adr 019 agentcore gateway tool federation ## Context -ABCA gives the agent tools by writing a per-thread `.mcp.json` into the cloned repo before each run. `agent/src/channel_mcp.py` maps an inbound channel to a hosted MCP server entry. **Today the agent holds zero functional platform-managed MCP servers:** the single `CHANNEL_MCP_BUILDERS` entry is `jira`, and it is a **non-functional placeholder** — the headless agent cannot complete Atlassian's interactive OAuth 2.1 flow, so the live outbound path is the REST shim in `jira_reactions.py` (see [ADR-015](/sample-autonomous-cloud-coding-agents/architecture/adr-015-jira-integration)). Linear is **deterministic by decision** ([ADR-016](/sample-autonomous-cloud-coding-agents/architecture/adr-016-pluggable-identity-and-auth)): there is no Linear MCP entry — it was removed after it proved non-functional, and `strip_linear_mcp_servers()` scrubs any Linear MCP a repo commits to `.mcp.json` before the SDK loads it. When a real MCP tool *is* wired, its entry carries a credential the container holds for the whole task (e.g. a `Bearer ${...}` header), resolved by a per-integration `resolve__token()` in `config.py`. This pattern — for any tool ABCA does adopt — has four structural costs: +ABCA gives the agent tools by writing a per-thread `.mcp.json` into the cloned repo before each run. `agent/src/channel_mcp.py` maps an inbound channel to a hosted MCP server entry. **Today the agent holds zero functional platform-managed MCP servers:** the single `CHANNEL_MCP_BUILDERS` entry is `jira`, and it is a **non-functional placeholder** — the headless agent cannot complete Atlassian's interactive OAuth 2.1 flow, so the live outbound path is the REST shim in `jira_reactions.py` (see [ADR-015](/sample-autonomous-cloud-coding-agents/decisions/adr-015-jira-integration)). Linear is **deterministic by decision** ([ADR-016](/sample-autonomous-cloud-coding-agents/decisions/adr-016-pluggable-identity-and-auth)): there is no Linear MCP entry — it was removed after it proved non-functional, and `strip_linear_mcp_servers()` scrubs any Linear MCP a repo commits to `.mcp.json` before the SDK loads it. When a real MCP tool *is* wired, its entry carries a credential the container holds for the whole task (e.g. a `Bearer ${...}` header), resolved by a per-integration `resolve__token()` in `config.py`. This pattern — for any tool ABCA does adopt — has four structural costs: -1. **The tool credential lives in the container.** Every MCP entry injects a bearer token into the agent's environment. The token is in the blast radius of any prompt-injection or dependency compromise for the whole task. [ADR-016](/sample-autonomous-cloud-coding-agents/architecture/adr-016-pluggable-identity-and-auth) is unifying the *resolution* of these tokens, but not the fact that the resolved token still lands in the container for MCP. +1. **The tool credential lives in the container.** Every MCP entry injects a bearer token into the agent's environment. The token is in the blast radius of any prompt-injection or dependency compromise for the whole task. [ADR-016](/sample-autonomous-cloud-coding-agents/decisions/adr-016-pluggable-identity-and-auth) is unifying the *resolution* of these tokens, but not the fact that the resolved token still lands in the container for MCP. 2. **Tool wiring is bespoke per server.** Adding a tool means editing `channel_mcp.py`, adding a `CHANNEL_MCP_BUILDERS` entry, threading a credential resolver, and redeploying. There is no declarative "add a tool" path for an operator. @@ -21,13 +21,13 @@ ABCA gives the agent tools by writing a per-thread `.mcp.json` into the cloned r **Amazon Bedrock AgentCore Gateway is a fully managed service that addresses all four.** Gateway converts APIs, Lambda functions, Smithy models, OpenAPI specs, and remote MCP servers into MCP-compatible tools; aggregates multiple such **targets** behind **one virtual MCP server** (a single consolidated `tools/list`); and manages **both** inbound authentication (agent → gateway) and outbound authentication (gateway → target) as a managed concern. It is serverless and observable, supports **MCP session reuse** and **semantic tool search**, and — critically for ABCA — its inbound authorizer can be **AWS IAM (SigV4)** or a **CUSTOM_JWT** token, both of which are portable across compute substrates. -**This is the tool-plane complement to [ADR-016](/sample-autonomous-cloud-coding-agents/architecture/adr-016-pluggable-identity-and-auth).** ADR-016 unifies *who is the principal* (inbound) and *how do we resolve an outbound credential* (the `resolve__token()` seam) with AgentCore Identity as one backend. This ADR decides *where the agent's tools live and how it reaches them*: behind a single managed Gateway endpoint whose outbound leg is built **on** AgentCore Identity's credential providers — so the same vault ADR-016 adopts holds the tool credential, and it is injected gateway → target and **never enters the container**. +**This is the tool-plane complement to [ADR-016](/sample-autonomous-cloud-coding-agents/decisions/adr-016-pluggable-identity-and-auth).** ADR-016 unifies *who is the principal* (inbound) and *how do we resolve an outbound credential* (the `resolve__token()` seam) with AgentCore Identity as one backend. This ADR decides *where the agent's tools live and how it reaches them*: behind a single managed Gateway endpoint whose outbound leg is built **on** AgentCore Identity's credential providers — so the same vault ADR-016 adopts holds the tool credential, and it is injected gateway → target and **never enters the container**. **A spike has already de-risked the mechanism.** The reference branch `feat/agentcore-gateway-mcp` (on the upstream remote — not merged to `main`; its `docs/design/AGENTCORE_GATEWAY_MCP_SPIKE.md` records findings F0–F16) took Gateway end-to-end against a hosted MCP server and reached a working federated endpoint — **verdict GO**, live-proven and torn down. That spike is Linear-coupled and, per [#641](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/641), **a reference for the mechanism, not a base** — this ADR does not continue off it. It reuses the proven *mechanism* — per-workspace provisioning shape, M2M inbound token minting, the CDK role/grant pattern, the registry → `.mcp.json` plumbing — while structuring it around a **provider-agnostic** target + config model rather than Linear-specific code, so tools onboard without new platform code. The originating issue is [#641](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/641). **Near-term scope: lead with the simplest, most common target type — not the hardest.** [#641](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/641) is explicit that OAuth is *one option among several* and that most targets ABCA would add never touch it; the prior spike's mistake was exercising only a single 3LO-OAuth MCP-server target. So P1 onboards a **Lambda tool target** — outbound auth is the gateway execution role (IAM), **no stored credential and no consent flow at all** — which proves the end-to-end path (provisioning, substrate-portable inbound, aggregation, agent routing) on the *easiest* outbound leg. Simpler-and-common target types (IAM-signed HTTP/OpenAPI, API-key) follow; the demanding 3LO-OAuth remote-MCP path is exercised **last**, once the general model is proven, not first. -Linear and Jira MCP are **explicitly not in near-term scope.** Linear is deterministic by decision ([ADR-016](/sample-autonomous-cloud-coding-agents/architecture/adr-016-pluggable-identity-and-auth)) and stays that way — this ADR does not re-introduce an agent-side Linear MCP. Jira's live path is the REST shim ([ADR-015](/sample-autonomous-cloud-coding-agents/architecture/adr-015-jira-integration)); whether a gateway that owns the OAuth flow could *unbreak* the non-functional Jira MCP placeholder is recorded below as a **speculative later experiment**, explicitly not part of the deterministic reaction/orchestration paths. +Linear and Jira MCP are **explicitly not in near-term scope.** Linear is deterministic by decision ([ADR-016](/sample-autonomous-cloud-coding-agents/decisions/adr-016-pluggable-identity-and-auth)) and stays that way — this ADR does not re-introduce an agent-side Linear MCP. Jira's live path is the REST shim ([ADR-015](/sample-autonomous-cloud-coding-agents/decisions/adr-015-jira-integration)); whether a gateway that owns the OAuth flow could *unbreak* the non-functional Jira MCP placeholder is recorded below as a **speculative later experiment**, explicitly not part of the deterministic reaction/orchestration paths. > **Not to be confused with the input gateway.** [`INPUT_GATEWAY.md`](/sample-autonomous-cloud-coding-agents/architecture/input-gateway) describes ABCA's *inbound channel normalization* — turning CLI / Slack / webhook payloads into one internal message format on the way **in**. AgentCore Gateway here is the opposite direction: aggregating *outbound tool calls* the agent makes **out**. They share the word "gateway" and nothing else. @@ -117,21 +117,21 @@ Implementation lands as **multiple PRs** after this ADR: | P1 | **Lambda tool behind the gateway.** Context-gated CDK gateway (L2 `Gateway`) + service role, AWS_IAM (SigV4) inbound, one read-only Lambda target (`abca_repo_config`); outbound = gateway execution role, **no credential**. **Agent routing** is an **in-process SigV4 MCP bridge** (`agent/src/gateway_tools.py`), *not* a `.mcp.json` entry: the SDK's remote MCP client attaches only a **static** `Authorization` header, so it cannot sign SigV4 — the bridge signs each request in-process with the compute role's credentials (`service = bedrock-agentcore`) and proxies the model's call. The whole feature (CDK + agent bridge) is gated on `ABCA_TOOL_GATEWAY_URL` (set only under `--context enableToolGateway=true`), so default synth is byte-unchanged and the bridge is inert when the flag is off. | Green on microVM; default synth byte-unchanged; feature inert when gate off. | | P2 | **Substrate parity + a credentialed target.** IAM-SigV4 / JWT inbound proven on **both** microVM and ECS/Fargate; add one credentialed simple target (IAM-signed HTTP/OpenAPI or API-key vaulted via AgentCore Identity). | Both substrates pass; credential never enters the container. | | P3 | **Generalize to N targets.** Provider-agnostic target + config model keyed off the registry ([#246](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/246)); `bgagent gateway add-target/list-targets/sync/remove-target`; remaining simpler target types (Smithy, no-auth, OAuth 2LO). | Onboarding branches green; idempotent re-run. | -| P4 | **Hardening, search, and the hard auth path.** Semantic tool search evaluation; lifecycle + cleanup + `*_PENDING_AUTH` recovery; aggregation of ≥2 target types behind one gateway; the 3LO-OAuth remote-MCP path (the spike's territory). *Speculative experiment, separately gated:* whether a gateway that owns the OAuth flow can make the non-functional Jira MCP placeholder reachable — explicitly **not** re-introducing agent-side Linear/Jira MCP into the deterministic reaction/orchestration paths ([ADR-015](/sample-autonomous-cloud-coding-agents/architecture/adr-015-jira-integration), [ADR-016](/sample-autonomous-cloud-coding-agents/architecture/adr-016-pluggable-identity-and-auth)). | As needed by tool count / ops; Jira-MCP experiment records reachable-or-not. | +| P4 | **Hardening, search, and the hard auth path.** Semantic tool search evaluation; lifecycle + cleanup + `*_PENDING_AUTH` recovery; aggregation of ≥2 target types behind one gateway; the 3LO-OAuth remote-MCP path (the spike's territory). *Speculative experiment, separately gated:* whether a gateway that owns the OAuth flow can make the non-functional Jira MCP placeholder reachable — explicitly **not** re-introducing agent-side Linear/Jira MCP into the deterministic reaction/orchestration paths ([ADR-015](/sample-autonomous-cloud-coding-agents/decisions/adr-015-jira-integration), [ADR-016](/sample-autonomous-cloud-coding-agents/decisions/adr-016-pluggable-identity-and-auth)). | As needed by tool count / ops; Jira-MCP experiment records reachable-or-not. | ## Out of scope (this ADR) - **The provisioning, CLI, and agent-routing code.** This ADR records the decision; the code is the P1–P4 follow-up PRs on [#641](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/641). - **The registry schema and storage.** Owned by [#246](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/246) / PR [#548](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/548); this ADR consumes its MCP-server asset kind, it does not define a second registry. -- **Inbound principal verification and outbound credential resolution.** Owned by [ADR-016](/sample-autonomous-cloud-coding-agents/architecture/adr-016-pluggable-identity-and-auth); this ADR is the tool-plane consumer of both seams. -- **The Linear/Jira deterministic reaction paths.** `linear_reactions.py` / `jira_reactions.py` stay as-is and authoritative. This ADR does **not** re-introduce an agent-side Linear or Jira MCP: Linear stays deterministic ([ADR-016](/sample-autonomous-cloud-coding-agents/architecture/adr-016-pluggable-identity-and-auth), enforced by `strip_linear_mcp_servers()`), and Jira's live path stays the REST shim ([ADR-015](/sample-autonomous-cloud-coding-agents/architecture/adr-015-jira-integration)). The speculative Jira-MCP-via-gateway experiment (P4) is gated separately and touches neither path. +- **Inbound principal verification and outbound credential resolution.** Owned by [ADR-016](/sample-autonomous-cloud-coding-agents/decisions/adr-016-pluggable-identity-and-auth); this ADR is the tool-plane consumer of both seams. +- **The Linear/Jira deterministic reaction paths.** `linear_reactions.py` / `jira_reactions.py` stay as-is and authoritative. This ADR does **not** re-introduce an agent-side Linear or Jira MCP: Linear stays deterministic ([ADR-016](/sample-autonomous-cloud-coding-agents/decisions/adr-016-pluggable-identity-and-auth), enforced by `strip_linear_mcp_servers()`), and Jira's live path stays the REST shim ([ADR-015](/sample-autonomous-cloud-coding-agents/decisions/adr-015-jira-integration)). The speculative Jira-MCP-via-gateway experiment (P4) is gated separately and touches neither path. ## References - Issue [#641](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/641) — the feature this ADR records (AgentCore Gateway tool federation) - Issue [#246](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/246) / PR [#548](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/548) — central agent asset registry (the MCP-server asset kind this ADR's targets are an instance of) -- [ADR-016](/sample-autonomous-cloud-coding-agents/architecture/adr-016-pluggable-identity-and-auth) — pluggable identity and auth (the credential-plane seams this ADR consumes) -- [ADR-015](/sample-autonomous-cloud-coding-agents/architecture/adr-015-jira-integration) — Jira integration (the REST-shim precedent for a non-functional MCP placeholder) +- [ADR-016](/sample-autonomous-cloud-coding-agents/decisions/adr-016-pluggable-identity-and-auth) — pluggable identity and auth (the credential-plane seams this ADR consumes) +- [ADR-015](/sample-autonomous-cloud-coding-agents/decisions/adr-015-jira-integration) — Jira integration (the REST-shim precedent for a non-functional MCP placeholder) - `docs/design/AGENTCORE_GATEWAY_MCP_SPIKE.md` — the GO spike, findings F0–F16, and the provisioning recipe + gotchas. Lives only on the reference branch `feat/agentcore-gateway-mcp` (upstream remote); not merged to `main`. - [`INPUT_GATEWAY.md`](/sample-autonomous-cloud-coding-agents/architecture/input-gateway) — the *inbound channel* gateway, distinct from AgentCore Gateway - AgentCore Gateway — [AWS Bedrock AgentCore developer guide](https://docs.aws.amazon.com/bedrock-agentcore/latest/devguide/gateway.html) and the [gateway target / outbound-auth references](https://docs.aws.amazon.com/bedrock-agentcore/latest/devguide/gateway-target-MCPservers.html) for target types, the inbound/outbound auth matrix, and session configuration diff --git a/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md b/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md index 2011e6737..900ccdbee 100644 --- a/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md +++ b/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md @@ -4,7 +4,7 @@ title: Adr 021 lambda microvms compute backend # ADR-021: AWS Lambda MicroVMs as a third ComputeStrategy backend -> **Implementation status (2026-09-18): P1 and P2 are merged; P3 is in draft review.** Approval sleep/wake, retained requests, conversation/workspace recovery and nested infrastructure have live acceptance evidence. Reusable migration and final integration checks remain open; see [verification status](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/docs/verification/README.md). This ADR defines P1–P3, not an official P4. +> **Implementation status (2026-09-18): P1 and P2 are merged; P3 is in draft review.** Approval sleep/wake, retained requests, conversation/workspace recovery and nested infrastructure have live acceptance evidence. Reusable migration and final integration checks remain open; see [verification status](/sample-autonomous-cloud-coding-agents/verification/readme). This ADR defines P1–P3, not an official P4. **Status:** proposed **Date:** 2026-07-29 @@ -74,7 +74,7 @@ Unanswered approvals have no deadline by default (`approval_timeout_s=0`). Expli Automatic suspension also requires the deployment’s `microvm_approval_suspend_enabled` opt-in, which defaults false for new deployments. A live Parameter Store switch lets existing durable executions stop initiating new suspensions without changing their pinned Lambda environment. The verified normal deployment has this opt-in enabled. Turning it off does not abandon already-suspended workers. -The coordinator observes the exact pending gate and waits for the sleep delay. It skips sleep when a timed gate has too little time left before its wake margin. The guest holds a coding barrier, drains acknowledged progress and commits a checkpoint before accepting `/suspend`. Lifecycle HTTP responses explicitly close their connections before freeze to avoid reuse of a stale pooled connection; see the [transport evidence summary](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/docs/verification/README.md#recorded-acceptance). +The coordinator observes the exact pending gate and waits for the sleep delay. It skips sleep when a timed gate has too little time left before its wake margin. The guest holds a coding barrier, drains acknowledged progress and commits a checkpoint before accepting `/suspend`. Lifecycle HTTP responses explicitly close their connections before freeze to avoid reuse of a stale pooled connection; see the [transport evidence summary](/sample-autonomous-cloud-coding-agents/verification/readme#recorded-acceptance). Approval and denial handlers commit the decision first. They then read the current handle consistently, persist wake intent and request resume best-effort. A wake failure records diagnostics and does not undo the accepted decision. The durable supervisor retries and observes both service state and guest consumption of the decision; RUNNING alone does not prove the tool was released. @@ -131,7 +131,7 @@ Normative requirements (EARS): - Signed URLs shall not appear in ordinary logs, agent-readable task rows or repository subprocess environments. Downloads shall use the exact regional S3 HTTPS object without redirects/proxies and with bounded response sizes. - Producers, images and IAM shall be upgraded together; incompatible workers must be drained before switching transport. -The [payload contract](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/docs/verification/645-payload-bootstrap.md) and [live checks](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/docs/verification/README.md) record the implementation and validation. This transport does not establish complete hostile-worker isolation: other platform grants remain, and a stolen signed URL is usable until expiry or revocation. +The [payload contract](/sample-autonomous-cloud-coding-agents/verification/645-payload-bootstrap) and [live checks](/sample-autonomous-cloud-coding-agents/verification/readme) record the implementation and validation. This transport does not establish complete hostile-worker isolation: other platform grants remain, and a stolen signed URL is usable until expiry or revocation. No ABCA endpoint consumer exists in P1–P3. The platform grants no `CreateMicrovmAuthToken` permission and mints no JWE tokens. `NO_INGRESS` can still return an endpoint URL; an unauthenticated 403 verifies the authentication boundary, not valid-token reachability. @@ -153,7 +153,7 @@ The backend adds build/runtime VPC connectors, build artifacts, launch payloads, `lambda:PassNetworkConnector` has no resource-level authorization support and therefore uses `Resource: *`. The operator role also has wildcard ENI permissions; some mutations can be scoped in IAM, so their current tested wildcard is not evidence that narrower permissions are impossible. DescribeAvailabilityZones needs a wildcard for fresh CDK repository lookups. These exceptions and namespace wildcards are documented in the construct’s cdk-nag suppressions. -**Nested infrastructure.** `microvm_nested_stack=true` puts MicroVM resources in a nested stack. The shared execution role stays in the parent to avoid a role-trust dependency cycle. Bootstrap 1.9.0 covers nested deployment roles. Existing flat installations need the reviewed overlap/drain migration, explicit image/coordinator pins and rollback checks described in the [migration prerequisites](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/docs/verification/645-p3-nested-stack.md). Reusable migration commands are still being completed. Moving construct paths alone is not a safe migration. Preserve `microvm_resource_name_prefix` after a migrated deployment. +**Nested infrastructure.** `microvm_nested_stack=true` puts MicroVM resources in a nested stack. The shared execution role stays in the parent to avoid a role-trust dependency cycle. Bootstrap 1.9.0 covers nested deployment roles. Existing flat installations need the reviewed overlap/drain migration, explicit image/coordinator pins and rollback checks described in the [migration prerequisites](/sample-autonomous-cloud-coding-agents/verification/645-p3-nested-stack). Reusable migration commands are still being completed. Moving construct paths alone is not a safe migration. Preserve `microvm_resource_name_prefix` after a migrated deployment. #### Security bar vs existing backends ([#645](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/645) acceptance criterion) @@ -203,7 +203,7 @@ Changing the default backend, GPU support, native Slack approval buttons, approv ## Testing -The [verification summary](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/docs/verification/README.md) distinguishes recorded live acceptance from open PR checks. Required coverage includes: +The [verification summary](/sample-autonomous-cloud-coding-agents/verification/readme) distinguishes recorded live acceptance from open PR checks. Required coverage includes: - Strategy state mapping, explicit unsupported results, ARN/Region validation, NO_INGRESS, omitted idlePolicy and bounded uncertain-start recovery. - Hook readiness/warm-up, AWS-silent build hooks, authenticated payload installation, arbitrary terminate bodies and lifecycle connection closure. @@ -220,5 +220,5 @@ Live evidence must distinguish service acknowledgment from completed guest recov - [Issue #491](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/491) — unified liveness model - [Issue #641](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/641) — substrate-portable tool plane - [AWS Lambda MicroVMs guide](https://docs.aws.amazon.com/lambda/latest/dg/lambda-microvms-guide.html) and [lifecycle APIs/hooks](https://docs.aws.amazon.com/lambda/latest/dg/microvms-launching.html) -- [ADR-020](/sample-autonomous-cloud-coding-agents/architecture/adr-020-ears-requirements-syntax) — requirement syntax +- [ADR-020](/sample-autonomous-cloud-coding-agents/decisions/adr-020-ears-requirements-syntax) — requirement syntax - [Compute](/sample-autonomous-cloud-coding-agents/architecture/compute), [orchestrator](/sample-autonomous-cloud-coding-agents/architecture/orchestrator), [approval gates](/sample-autonomous-cloud-coding-agents/architecture/cedar-hitl-gates) diff --git a/docs/src/content/docs/decisions/Adr-022-agent-asset-registry.md b/docs/src/content/docs/decisions/Adr-022-agent-asset-registry.md index cc8ac39b3..a8c5a9808 100644 --- a/docs/src/content/docs/decisions/Adr-022-agent-asset-registry.md +++ b/docs/src/content/docs/decisions/Adr-022-agent-asset-registry.md @@ -16,10 +16,10 @@ This has three costs: 1. **Every new tool/skill/policy costs a deploy.** Rolling out a new MCP server to N repos is N Blueprint edits + a CDK deploy. Teams can't publish autonomously. 2. **No pin, no reproducibility.** Because assets aren't versioned, "the tool the agent used on 2026-05-01" can't be reconstructed from the task record. -3. **The vocabulary already anticipates a registry.** [ADR-014](/sample-autonomous-cloud-coding-agents/architecture/adr-014-workflow-driven-tasks) and [WORKFLOWS.md](/sample-autonomous-cloud-coding-agents/architecture/workflows) already: - - Coined `registry://kind/name` refs (grammar in [`agent/src/workflow/validator.py`](../../agent/src/workflow/validator.py) `_REGISTRY_REF`). +3. **The vocabulary already anticipates a registry.** [ADR-014](/sample-autonomous-cloud-coding-agents/decisions/adr-014-workflow-driven-tasks) and [WORKFLOWS.md](/sample-autonomous-cloud-coding-agents/architecture/workflows) already: + - Coined `registry://kind/name` refs (grammar in [`agent/src/workflow/validator.py`](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/agent/src/workflow/validator.py) `_REGISTRY_REF`). - Modeled `agent_config` asset kinds (`mcp_servers`, `skills`, `plugins`, `subagents`, `prompt_fragments`, `cedar_policy_modules`) 1:1 with the vocabulary this ADR needs. - - Designed a resolver interface as a **drop-in swap** — filesystem-backed today, registry-backed later ([`agent/src/workflow/loader.py:107`](../../agent/src/workflow/loader.py), WORKFLOWS.md §"Registry integration (#246)"). + - Designed a resolver interface as a **drop-in swap** — filesystem-backed today, registry-backed later ([`agent/src/workflow/loader.py:107`](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/agent/src/workflow/loader.py), WORKFLOWS.md §"Registry integration (#246)"). - Left validator rule 8 as a deferred check: every asset ref resolves — builtins today, registry refs when the registry lands. Issue [#246](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/246) proposes closing this gap with a **central versioned asset registry**: a catalog of typed, immutable-at-version artifact records that blueprints pin by `registry://kind/name@constraint`, that the orchestrator resolves at task start, and that the agent receives as a resolved bundle. Six acceptance criteria — asset kinds enumerated, publish+resolve with semver+immutability, blueprint reference of at least one kind, agent E2E for one kind, descriptor validation at publish, tests + docs. @@ -32,7 +32,7 @@ Two forces shape the decision: Adjacent decisions the ADR must respect but not re-open: - **[#381](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/381) — ADR↔Persona↔Skill graph.** #381 wires bidirectional frontmatter edges between ADRs, personas, and skills as `docs/` and plugin markdown, enforced by a parity linter. Both #246 and #381 mention "skills," but they mean different things: #381 is *documentation graph consistency*; #246 is a *runtime artifact catalog*. This ADR keeps them cleanly separated. -- **[ADR-014](/sample-autonomous-cloud-coding-agents/architecture/adr-014-workflow-driven-tasks).** Workflows are the registry's first `capability`-kind consumer; the `workflow_ref` field's resolution semantics are the model this ADR generalizes to all asset kinds. Workflows themselves stay filesystem-backed in the MVP — they already ship and are validated; migrating them to the registry is a separate follow-up, not part of #246. +- **[ADR-014](/sample-autonomous-cloud-coding-agents/decisions/adr-014-workflow-driven-tasks).** Workflows are the registry's first `capability`-kind consumer; the `workflow_ref` field's resolution semantics are the model this ADR generalizes to all asset kinds. Workflows themselves stay filesystem-backed in the MVP — they already ship and are validated; migrating them to the registry is a separate follow-up, not part of #246. ## Decision @@ -44,7 +44,7 @@ The ADR fixes the **contract**. The substrate ranking below was written to defer ### Sub-decisions -1. **URI grammar and kinds.** `registry:////@`. MVP kinds: `mcp_server`, `cedar_policy_module`, `skill`. Schema declares — but does not yet load — `plugin`, `subagent`, `prompt_fragment`, `capability` (`capability` = workflow, ADR-014 vocabulary). This grammar *extends* the shape pre-declared for ADR-014 at [`agent/src/workflow/validator.py`](../../agent/src/workflow/validator.py) `_REGISTRY_REF`, which admitted a 2-segment `registry://kind/name` form but not the `@` suffix or `_` (snake_case) in the kind segment. The implementation widens that lenient acceptance check and lands the **authoritative, strict grammar** in a dedicated parser mirrored byte-for-byte across both languages (`cdk/src/handlers/shared/registry/ref.ts` and `agent/src/registry/ref.py`); `validator.py` remains a lenient pre-flight admitting both forms. +1. **URI grammar and kinds.** `registry:////@`. MVP kinds: `mcp_server`, `cedar_policy_module`, `skill`. Schema declares — but does not yet load — `plugin`, `subagent`, `prompt_fragment`, `capability` (`capability` = workflow, ADR-014 vocabulary). This grammar *extends* the shape pre-declared for ADR-014 at [`agent/src/workflow/validator.py`](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/agent/src/workflow/validator.py) `_REGISTRY_REF`, which admitted a 2-segment `registry://kind/name` form but not the `@` suffix or `_` (snake_case) in the kind segment. The implementation widens that lenient acceptance check and lands the **authoritative, strict grammar** in a dedicated parser mirrored byte-for-byte across both languages (`cdk/src/handlers/shared/registry/ref.ts` and `agent/src/registry/ref.py`); `validator.py` remains a lenient pre-flight admitting both forms. **Short vs long forms (migration note).** [WORKFLOWS.md](/sample-autonomous-cloud-coding-agents/architecture/workflows) illustrates refs in a *short* 2-segment form with no constraint (`registry://prompt/web-research-workflow`, `registry://mcp/web-search-v1`, `registry://skill/research-synthesis-v1`). Those are **forward-declarations** — the WORKFLOWS spec explicitly marks `registry://` refs as "declared in the schema now but ignored by the runner until the registry (#246) can resolve them." This ADR's strict grammar is the *long* form: 3 segments (`//`), snake_case kinds (`mcp_server`, not `mcp`; `prompt_fragment`, not `prompt`; `cedar_policy_module`, not `cedar`), and a **mandatory** `@`. Only the long form resolves. The short form stays **lenient-only** — accepted syntactically by `validator.py`'s pre-flight so existing illustrative workflows don't fail validation, but *not resolvable* by the registry until rewritten to the long form. There is no automatic aliasing (`mcp` → `mcp_server`) at resolve time: a ref must be long-form to load. Migrating the WORKFLOWS examples to long form is doc-only cleanup tracked with the workflow/registry integration, not a blocker for #246. @@ -180,12 +180,12 @@ Regardless of substrate, the invariants above (semver, immutability, resolve-at- - Issue [#481](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/481) — capability descriptors. - Issue [#230](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/230) — event-rule packs (defer to registry Phase 3). - Issue [#99](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/99) — ToolBuilderAgent / meta-agent vision (out of scope; registry is a prerequisite). -- [ADR-003](/sample-autonomous-cloud-coding-agents/architecture/adr-003-contribution-governance) — contribution governance (publish/promote follows the same approval path). -- [ADR-014](/sample-autonomous-cloud-coding-agents/architecture/adr-014-workflow-driven-tasks) — workflow-driven tasks (defines the `registry://` grammar, resolver interface, and asset-kind vocabulary this ADR generalizes). +- [ADR-003](/sample-autonomous-cloud-coding-agents/decisions/adr-003-contribution-governance) — contribution governance (publish/promote follows the same approval path). +- [ADR-014](/sample-autonomous-cloud-coding-agents/decisions/adr-014-workflow-driven-tasks) — workflow-driven tasks (defines the `registry://` grammar, resolver interface, and asset-kind vocabulary this ADR generalizes). - [docs/design/WORKFLOWS.md](/sample-autonomous-cloud-coding-agents/architecture/workflows) §"Registry integration (#246)" — the workflow-side spec this ADR closes. -- [`agent/src/workflow/validator.py`](../../agent/src/workflow/validator.py) `_REGISTRY_REF` — the lenient pre-flight ref check (pre-dates this ADR; widened here to admit the strict form). +- [`agent/src/workflow/validator.py`](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/agent/src/workflow/validator.py) `_REGISTRY_REF` — the lenient pre-flight ref check (pre-dates this ADR; widened here to admit the strict form). - `cdk/src/handlers/shared/registry/ref.ts` / `agent/src/registry/ref.py` — the authoritative strict `registry://` grammar (two-language, parity-tested). -- [`agent/src/workflow/loader.py:107`](../../agent/src/workflow/loader.py) — Phase 4 deferral comment this ADR unblocks. +- [`agent/src/workflow/loader.py:107`](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/agent/src/workflow/loader.py) — Phase 4 deferral comment this ADR unblocks. - [AWS Agent Registry documentation](https://docs.aws.amazon.com/bedrock-agentcore/latest/devguide/registry.html) — preferred substrate candidate. - [AWS Agent Registry key capabilities](https://docs.aws.amazon.com/bedrock-agentcore/latest/devguide/registry-key-capabilities.html) — governance lifecycle, hybrid search, EventBridge integration. - [AWS Agent Registry: Migration from public preview](https://docs.aws.amazon.com/bedrock-agentcore/latest/devguide/registry-faq.html) — the 2026-08-06 namespace migration referenced in risks. diff --git a/docs/src/content/docs/decisions/Adr-023-trusted-approval-writer.md b/docs/src/content/docs/decisions/Adr-023-trusted-approval-writer.md new file mode 100644 index 000000000..70fc974d7 --- /dev/null +++ b/docs/src/content/docs/decisions/Adr-023-trusted-approval-writer.md @@ -0,0 +1,72 @@ +--- +title: Adr 023 trusted approval writer +--- + +# ADR-023: Trusted approval writer and retained human decisions + +**Status:** proposed +**Date:** 2026-09-21 + +## Context + +Workers previously had direct write access to approval records. Restricting a +worker to its own task did not prevent it from replacing a pending request with +an approved record. MicroVM continuation makes the distinction between worker +authority and human consent especially important, but the same defect affects +ECS and AgentCore. + +A short approval deadline also couples human response time to compute lifetime. +Someone taking time to consider a request should not lose it merely because the +worker should stop consuming resources. + +## Decision + +Use a small IAM-authenticated API and Lambda as the worker-facing approval +writer. Its interface creates a pending request or closes one after a worker +timeout/failure. It cannot record human approval or denial, notification markers, +or retention TTLs. Human decisions remain in the owner-authenticated decision +handlers. Workers retain approval reads and transaction condition checks. + +The session role can invoke only the task path named by its `task_id` tag. +Creation atomically writes the request and transitions its task from running to +awaiting approval. The transaction checks the task owner/state and, for MicroVM, +the coordinator-owned worker lease. A stale worker cannot use its old lease to +create or close a request. + +Unanswered requests have no decision deadline by default. Explicit positive +timeouts remain available. Task cancellation, terminal failure and invalidated +worker execution can close requests independently. Compute lifetime is separate: +MicroVM can checkpoint and release a worker while retaining a request; ECS and +AgentCore currently cannot restore that waiting execution into a replacement. +Their task execution limits still apply. + +This implementation is included in the P3 review because retaining approvals +without protecting their decision records would preserve the security defect. +The ADR remains proposed for maintainer review. + +## Consequences + +- Existing deployments must pause submissions, drain old workers and deploy + matching infrastructure and images. Old images still attempt direct writes, + which the new IAM policy denies. See the + [upgrade procedure](/sample-autonomous-cloud-coding-agents/getting-started/deployment-guide#upgrading-approval-permissions). +- The service adds one signed request per creation/closure and becomes an + availability dependency. Failure denies permission to proceed; it never + becomes human consent. +- The service validates record shape, not the truth of worker-supplied policy + descriptions. A preview can be truncated and is not a proof of the full tool + input. `TIMED_OUT` currently also represents a worker polling failure; it is + not proof that a human deadline elapsed. Human `DENIED` remains distinct. +- The compute role still chooses session tags when assuming the session role. + This API protects the decision-writing boundary but does not provide full + isolation from a compromised worker retaining ambient compute credentials. +- Linear consent is read back from Linear and must come from the mapped owner, + excluding bots and the saved OAuth token's own identity. Other webhook paths + still need a separate review of the legacy shared OAuth/signing-secret bundle. + +## References + +- [MicroVM backend, issue #645](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/645) +- [Implementation and security review, PR #904](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/pull/904) +- [Approval trust boundaries](/sample-autonomous-cloud-coding-agents/architecture/cedar-hitl-gates#121-trust-boundaries) +- [ADR-021: Lambda MicroVM compute backend](/sample-autonomous-cloud-coding-agents/decisions/adr-021-lambda-microvms-compute-backend) diff --git a/docs/src/content/docs/developer-guide/Contributing.md b/docs/src/content/docs/developer-guide/Contributing.md index 86f5f0b3e..eef135d34 100644 --- a/docs/src/content/docs/developer-guide/Contributing.md +++ b/docs/src/content/docs/developer-guide/Contributing.md @@ -18,9 +18,9 @@ Describe what you intend to contribute. This avoids duplicate work and gives mai ### 2. Set up your environment -Follow the [Quick Start](./docs/guides/QUICK_START.mdx) to clone, install, and build the project. See the [Developer guide](/sample-autonomous-cloud-coding-agents/developer-guide/introduction) for local testing and the development workflow. +Follow the [Quick Start](/sample-autonomous-cloud-coding-agents/getting-started/quick-start) to clone, install, and build the project. See the [Developer guide](/sample-autonomous-cloud-coding-agents/developer-guide/introduction) for local testing and the development workflow. -Use **[AGENTS.md](/sample-autonomous-cloud-coding-agents/architecture/agents)** to understand where to make changes (CDK vs CLI vs agent vs docs), which tests to extend, and common pitfalls (generated docs, mirrored API types, `mise` tasks). Package-specific detail lives in **`AGENTS.md`** under `cdk/`, `cli/`, `agent/`, and `docs/`. +Use **[AGENTS.md](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/AGENTS.md)** to understand where to make changes (CDK vs CLI vs agent vs docs), which tests to extend, and common pitfalls (generated docs, mirrored API types, `mise` tasks). Package-specific detail lives in **`AGENTS.md`** under `cdk/`, `cli/`, `agent/`, and `docs/`. ### 3. Implement your change @@ -32,7 +32,7 @@ Guidelines: - If you change API types in `cdk/src/handlers/shared/types.ts`, update `cli/src/types.ts` to match. - If you change docs sources (`docs/guides/`, `docs/design/`), run `mise //docs:sync` so generated content stays in sync. - For significant features, add a design document to `docs/design/`. -- For cross-cutting or hard-to-reverse decisions, add an ADR to `docs/decisions/` (see [ADR README](/sample-autonomous-cloud-coding-agents/architecture/readme)). +- For cross-cutting or hard-to-reverse decisions, add an ADR to `docs/decisions/` (see [ADR README](/sample-autonomous-cloud-coding-agents/decisions/readme)). ### 4. Commit @@ -102,6 +102,19 @@ PRs labeled `auto-approve` are approved automatically by the `auto-approve` work If `prek install` fails with "refusing to install hooks with `core.hooksPath` set", another tool owns your hooks. Either unset it (`git config --unset-all core.hooksPath`) or integrate these checks into your hook manager. +The build's transaction tests use DynamoDB Local through +`ABCA_DDB_LOCAL_ENDPOINT` (a loopback HTTP endpoint). CI starts a Docker service +pinned by image digest and supplies its mapped port to both CDK and Python tests; +those tests fail if `CI=true` without an endpoint. To reproduce locally, start +DynamoDB Local with `-inMemory -sharedDb`, bind port 8000 to loopback, and run +`ABCA_DDB_LOCAL_ENDPOINT=http://127.0.0.1:8000 mise run build`. + +The pinned SDK continuation probe also runs in CI. Locally enable it with +`ABCA_TEST_SDK_CONTINUATION=1` when running +`agent/tests/test_continuation_sdk_probe.py`. It uses a local simulated Bedrock +endpoint and synthetic credentials; it does not require an AWS account or model +billing. + ## Versioning The project uses semantic versioning based on [Conventional Commits](https://www.conventionalcommits.org/en/v1.0.0/): diff --git a/docs/src/content/docs/developer-guide/Installation.md b/docs/src/content/docs/developer-guide/Installation.md index 5a01813c6..449345a84 100644 --- a/docs/src/content/docs/developer-guide/Installation.md +++ b/docs/src/content/docs/developer-guide/Installation.md @@ -2,7 +2,7 @@ title: Installation --- -Follow the [Quick Start](./QUICK_START.mdx) to clone, install, deploy, and submit your first task. It covers prerequisites, toolchain setup, deployment, PAT configuration, Cognito user creation, and a smoke test. +Follow the [Quick Start](/sample-autonomous-cloud-coding-agents/getting-started/quick-start) to clone, install, deploy, and submit your first task. It covers prerequisites, toolchain setup, deployment, PAT configuration, Cognito user creation, and a smoke test. This section covers what the Quick Start does not: troubleshooting, local testing, and the development workflow. @@ -152,7 +152,7 @@ For the full list, see `agent/README.md`. For how the model default is layered, ### Deployment -Follow the [Quick Start](./QUICK_START.mdx) steps 3-6 for first-time deployment. For subsequent deploys after code changes: +Follow the [Quick Start](/sample-autonomous-cloud-coding-agents/getting-started/quick-start) steps 3-6 for first-time deployment. For subsequent deploys after code changes: ```bash mise run build diff --git a/docs/src/content/docs/developer-guide/Introduction.md b/docs/src/content/docs/developer-guide/Introduction.md index 5b259c4ca..8a81f45b7 100644 --- a/docs/src/content/docs/developer-guide/Introduction.md +++ b/docs/src/content/docs/developer-guide/Introduction.md @@ -14,6 +14,6 @@ The repository is organized around four main pieces: - **Infrastructure as code** in AWS CDK under `cdk/src/` - stacks, constructs, and handlers that define and deploy the platform on AWS. - **Documentation site** under `docs/` - source guides/design docs plus the generated Astro/Starlight documentation site. - **CLI package** under `cli/` - the `bgagent` command-line client used to authenticate, submit tasks, and inspect task status/events. -- **Claude Code plugin** under `docs/abca-plugin/` - a [Claude Code plugin](https://docs.anthropic.com/en/docs/claude-code/plugins) with guided skills and agents for setup, deployment, task submission, and troubleshooting. See the [plugin README](/sample-autonomous-cloud-coding-agents/architecture/readme) for details. +- **Claude Code plugin** under `docs/abca-plugin/` - a [Claude Code plugin](https://docs.anthropic.com/en/docs/claude-code/plugins) with guided skills and agents for setup, deployment, task submission, and troubleshooting. See the [plugin README](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/docs/abca-plugin/README.md) for details. > **Tip:** If you use Claude Code, run `claude --plugin-dir docs/abca-plugin` from the repo root. The plugin's `/setup` skill walks you through the entire setup process interactively. \ No newline at end of file diff --git a/docs/src/content/docs/developer-guide/Repository-preparation.md b/docs/src/content/docs/developer-guide/Repository-preparation.md index 0877d39b1..e267b0656 100644 --- a/docs/src/content/docs/developer-guide/Repository-preparation.md +++ b/docs/src/content/docs/developer-guide/Repository-preparation.md @@ -2,7 +2,7 @@ title: Repository preparation --- -The [Quick Start](./QUICK_START.mdx) covers the basic setup: forking a sample repo, creating a PAT, registering a Blueprint, and storing the token in Secrets Manager. This section covers what you need beyond that. +The [Quick Start](/sample-autonomous-cloud-coding-agents/getting-started/quick-start) covers the basic setup: forking a sample repo, creating a PAT, registering a Blueprint, and storing the token in Secrets Manager. This section covers what you need beyond that. ### Pre-flight checks @@ -13,7 +13,7 @@ Permission requirements vary by task type: - `new_task` and `pr_iteration` require Contents (read/write) and Pull requests (read/write). - `pr_review` only needs Triage or higher since it does not push branches. -Classic PATs with `repo` + `read:org` scopes also work and are required when fine-grained tokens cannot reach the target repo (collaborator access, cross-org repos). See [agent/README.md](/sample-autonomous-cloud-coding-agents/architecture/readme#github-pat--minimal-permissions) for when to use which token type. +Classic PATs with `repo` + `read:org` scopes also work and are required when fine-grained tokens cannot reach the target repo (collaborator access, cross-org repos). See [agent/README.md](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/agent/README.md#github-pat--minimal-permissions) for when to use which token type. ### Quick setup (single repo) @@ -79,7 +79,7 @@ The default image (`agent/Dockerfile`) includes Python, Node 24 (LTS), `git`, `g A blueprint can declare its own `security.cedarPolicies` rules on top of the built-in hard/soft-deny starter set. Hard-deny rules absolutely block a tool call; soft-deny rules pause the agent and ask a human before proceeding. -See the [Cedar policy guide](/sample-autonomous-cloud-coding-agents/customizing/cedar-policies) for the full authoring reference — vocabulary (`execute_bash`, `write_file`, `context.command`, `context.file_path`), annotations (`@rule_id`, `@tier`, `@approval_timeout_s`, `@severity`, `@category`), worked examples, multi-match rules, and cross-engine parity testing with [`contracts/cedar-parity/`](../../contracts/cedar-parity/) fixtures. +See the [Cedar policy guide](/sample-autonomous-cloud-coding-agents/customizing/cedar-policies) for the full authoring reference — vocabulary (`execute_bash`, `write_file`, `context.command`, `context.file_path`), annotations (`@rule_id`, `@tier`, `@approval_timeout_s`, `@severity`, `@category`), worked examples, multi-match rules, and cross-engine parity testing with [`contracts/cedar-parity/`](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/tree/main/contracts/cedar-parity) fixtures. ### Other options diff --git a/docs/src/content/docs/developer-guide/Where-to-make-changes.md b/docs/src/content/docs/developer-guide/Where-to-make-changes.md index 395e15315..7ba8d0713 100644 --- a/docs/src/content/docs/developer-guide/Where-to-make-changes.md +++ b/docs/src/content/docs/developer-guide/Where-to-make-changes.md @@ -12,4 +12,4 @@ Before editing, decide which part of the monorepo owns the behavior. This keeps | Agent runtime | `agent/` | Bundled into the image CDK deploys; run `mise run quality` in `agent/` or root build. | | Docs (source) | `docs/guides/`, `docs/design/` | After edits, run **`mise //docs:sync`** or **`mise //docs:build`**. Do not edit `docs/src/content/docs/` directly. | -For a concise duplicate of this table, common pitfalls, and a CDK test file map, see **[AGENTS.md](/sample-autonomous-cloud-coding-agents/architecture/agents)** at the repo root (oriented toward automation-assisted contributors). Package-specific detail lives in **`AGENTS.md`** under `cdk/`, `cli/`, `agent/`, and `docs/`. \ No newline at end of file +For a concise duplicate of this table, common pitfalls, and a CDK test file map, see **[AGENTS.md](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/AGENTS.md)** at the repo root (oriented toward automation-assisted contributors). Package-specific detail lives in **`AGENTS.md`** under `cdk/`, `cli/`, `agent/`, and `docs/`. \ No newline at end of file diff --git a/docs/src/content/docs/getting-started/Deployment-guide.md b/docs/src/content/docs/getting-started/Deployment-guide.md index dd056918d..2bcd3b1f9 100644 --- a/docs/src/content/docs/getting-started/Deployment-guide.md +++ b/docs/src/content/docs/getting-started/Deployment-guide.md @@ -4,7 +4,7 @@ title: Deployment guide # Deployment guide -This guide covers deploying ABCA into an AWS account, including compute backend choices, scale-to-zero characteristics, and the complete AWS service inventory. For day-to-day development workflow, see the [Developer guide](/sample-autonomous-cloud-coding-agents/developer-guide/introduction). For a quick first deployment, see the [Quick start](./QUICK_START.mdx). For least-privilege IAM deployment roles, see [DEPLOYMENT_ROLES.md](/sample-autonomous-cloud-coding-agents/architecture/deployment-roles). +This guide covers deploying ABCA into an AWS account, including compute backend choices, scale-to-zero characteristics, and the complete AWS service inventory. For day-to-day development workflow, see the [Developer guide](/sample-autonomous-cloud-coding-agents/developer-guide/introduction). For a quick first deployment, see the [Quick start](/sample-autonomous-cloud-coding-agents/getting-started/quick-start). For least-privilege IAM deployment roles, see [DEPLOYMENT_ROLES.md](/sample-autonomous-cloud-coding-agents/architecture/deployment-roles). ## Architecture overview @@ -25,7 +25,7 @@ ECS Fargate is **opt-in**. Deploy with `--context compute_type=ecs`; the stack e ### Lambda MicroVMs backend (experimental) -> **Not for production.** `lambda-microvm` carries no smoke-parity guarantee for an unattended deployment. Keep production repositories on `agentcore` or `ecs`. Synth emits a verification warning whenever a MicroVM image is configured; selecting the backend without an image emits a separate setup warning. Design detail: [COMPUTE.md](/sample-autonomous-cloud-coding-agents/architecture/compute) and [ADR-021](/sample-autonomous-cloud-coding-agents/architecture/adr-021-lambda-microvms-compute-backend). +> **Not for production.** `lambda-microvm` carries no smoke-parity guarantee for an unattended deployment. Keep production repositories on `agentcore` or `ecs`. Synth emits a verification warning whenever a MicroVM image is configured; selecting the backend without an image emits a separate setup warning. Design detail: [COMPUTE.md](/sample-autonomous-cloud-coding-agents/architecture/compute) and [ADR-021](/sample-autonomous-cloud-coding-agents/decisions/adr-021-lambda-microvms-compute-backend). Selecting it is a synth-time context flag: @@ -263,6 +263,9 @@ directly to DynamoDB and cannot create new gates after those permissions are rem 3. Deploy the stack. Check that the SessionRole has approval-table reads and condition checks only, plus `execute-api:Invoke` restricted to its task tag. CDK supplies `APPROVAL_REQUESTS_API_URL` to all three compute backends. + Custom ECS constructs must provide both the SessionRole and service URL; + approval wiring without them is rejected before deployment. A MicroVM + manifest missing the URL is rejected before the worker starts. 4. Submit a test task that triggers a known approval rule on each enabled backend. Verify that the request appears, an owner decision resumes it, and an explicit deadline records `TIMED_OUT` without overwriting a human decision. Then resume @@ -344,7 +347,7 @@ aws ec2 describe-subnets --filters "Name=vpc-id,Values=" \ --query 'Subnets[].[SubnetId,AvailabilityZone,AvailabilityZoneId]' --output text ``` -Be aware that destroying a VPC whose subnets held AgentCore ENIs can take 20–40 minutes while AWS reclaims them (see the `DELETE_FAILED` note in the [quick start](./QUICK_START.mdx) troubleshooting table). +Be aware that destroying a VPC whose subnets held AgentCore ENIs can take 20–40 minutes while AWS reclaims them (see the `DELETE_FAILED` note in the [quick start](/sample-autonomous-cloud-coding-agents/getting-started/quick-start) troubleshooting table). ### DNS Query Log Config replacement cascade (upgrading from pre-v0.5) @@ -407,11 +410,11 @@ For users without AWS CLI access. ## Related docs -- [Quick start](./QUICK_START.mdx) -- Zero-to-first-PR in 6 steps. +- [Quick start](/sample-autonomous-cloud-coding-agents/getting-started/quick-start) -- Zero-to-first-PR in 6 steps. - [Developer guide](/sample-autonomous-cloud-coding-agents/developer-guide/introduction) -- Local development, testing, repository onboarding. - [User guide](/sample-autonomous-cloud-coding-agents/using/overview) -- API reference, CLI usage, task management. - [DEPLOYMENT_ROLES.md](/sample-autonomous-cloud-coding-agents/architecture/deployment-roles) -- Least-privilege IAM policies for CloudFormation execution. - [COST_MODEL.md](/sample-autonomous-cloud-coding-agents/architecture/cost-model) -- Per-task costs, cost guardrails, cost at scale. - [COST_ATTRIBUTION.md](/sample-autonomous-cloud-coding-agents/getting-started/cost-attribution) -- Operator FinOps setup for per-user/per-repo Bedrock chargeback (Cost Explorer / CUR 2.0, invocation-log forensics). - [COMPUTE.md](/sample-autonomous-cloud-coding-agents/architecture/compute) -- Compute backend architecture and trade-offs. -- [ADR-021](/sample-autonomous-cloud-coding-agents/architecture/adr-021-lambda-microvms-compute-backend) -- Lambda MicroVMs backend decision, phased rollout, and live-verification evidence. +- [ADR-021](/sample-autonomous-cloud-coding-agents/decisions/adr-021-lambda-microvms-compute-backend) -- Lambda MicroVMs backend decision, phased rollout, and live-verification evidence. diff --git a/docs/src/content/docs/getting-started/Quick-start.mdx b/docs/src/content/docs/getting-started/Quick-start.mdx index b46a6113a..dde3a92a4 100644 --- a/docs/src/content/docs/getting-started/Quick-start.mdx +++ b/docs/src/content/docs/getting-started/Quick-start.mdx @@ -170,7 +170,7 @@ If you have not yet added `blueprintRepo` to `cdk/cdk.json`, go back to [Registe :::note[Collaborator or cross-org repos?] -Fine-grained tokens only work for repos you own (or orgs that have opted in). If you're a collaborator on someone else's repo, create a **classic PAT** with `repo` + `read:org` scopes instead. See [agent/README.md](/sample-autonomous-cloud-coding-agents/architecture/readme#github-pat--minimal-permissions) for details. +Fine-grained tokens only work for repos you own (or orgs that have opted in). If you're a collaborator on someone else's repo, create a **classic PAT** with `repo` + `read:org` scopes instead. See [agent/README.md](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/agent/README.md#github-pat--minimal-permissions) for details. ::: @@ -492,7 +492,7 @@ node lib/bin/bgagent.js submit --repo owner/repo --issue 42 \ --pre-approve write_path:tests/** ``` -Hard-deny rules (no `@tier("soft")` annotation) are always enforced — `--pre-approve` only short-circuits soft-deny rules. For the full command reference see [User guide — Approval gates](/sample-autonomous-cloud-coding-agents/using/overview#approval-gates-cedar-hitl); for authoring your own rules see the [Cedar policy guide](/sample-autonomous-cloud-coding-agents/customizing/cedar-policies). +Hard-deny rules (no `@tier("soft")` annotation) are always enforced — `--pre-approve` only short-circuits soft-deny rules. For the full command reference see [User guide — Approval gates](/sample-autonomous-cloud-coding-agents/using/approval-gates-cedar-hitl); for authoring your own rules see the [Cedar policy guide](/sample-autonomous-cloud-coding-agents/customizing/cedar-policies). ## What happened behind the scenes diff --git a/docs/src/content/docs/using/Approval-gates-cedar-hitl.md b/docs/src/content/docs/using/Approval-gates-cedar-hitl.md index 8a68216f6..834372cfd 100644 --- a/docs/src/content/docs/using/Approval-gates-cedar-hitl.md +++ b/docs/src/content/docs/using/Approval-gates-cedar-hitl.md @@ -141,4 +141,4 @@ pauses can cost more than staying awake. The API equivalent is `microvm_sleep_after_s` (zero means off); task details return the saved setting. Automatic suspension is disabled by default for new deployments. An operator enables it after [verifying the deployed image and -coordinator](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/docs/verification/README.md#live-acceptance-for-an-installation). \ No newline at end of file +coordinator](/sample-autonomous-cloud-coding-agents/verification/readme#live-acceptance-for-an-installation). \ No newline at end of file diff --git a/docs/src/content/docs/using/Jira-setup-guide.md b/docs/src/content/docs/using/Jira-setup-guide.md index ddc12a4ba..1c621ca6b 100644 --- a/docs/src/content/docs/using/Jira-setup-guide.md +++ b/docs/src/content/docs/using/Jira-setup-guide.md @@ -107,7 +107,7 @@ Comments are advisory and best-effort: network/auth failures are logged and swal > (`mcp.atlassian.com`) requires an interactive, browser-based OAuth 2.1 flow > and cannot connect from a headless agent. Forge provides the supported app > actor through `api.asApp().requestJira(...)`. See -> [ADR-015](/sample-autonomous-cloud-coding-agents/architecture/adr-015-jira-integration). +> [ADR-015](/sample-autonomous-cloud-coding-agents/decisions/adr-015-jira-integration). Inbound admission (webhook → task) is Jira-specific and has no DynamoDB Streams consumer of its own. Ordinary **terminal** status comments are delivered by the shared fan-out plane's DynamoDB Streams consumer (`dispatchToJira`). For comment-triggered iterations, fan-out matures standalone status comments while the orchestration reconciler matures child-iteration comments before restacking dependents. diff --git a/docs/src/content/docs/using/Linear-setup-guide.md b/docs/src/content/docs/using/Linear-setup-guide.md index 3e3dce86c..6647b6924 100644 --- a/docs/src/content/docs/using/Linear-setup-guide.md +++ b/docs/src/content/docs/using/Linear-setup-guide.md @@ -217,7 +217,7 @@ Pick the teammate from the list of human members. You get a one-time code (24h T They need an ABCA account first. If they don't have one: -1. **Admin** runs `bgagent admin invite-user teammate@example.com` to create their Cognito user (see [User guide → Joining an existing deployment](/sample-autonomous-cloud-coding-agents/using/overview#joining-an-existing-deployment) for the full Cognito-side flow). +1. **Admin** runs `bgagent admin invite-user teammate@example.com` to create their Cognito user (see [User guide → Joining an existing deployment](/sample-autonomous-cloud-coding-agents/using/authentication#joining-an-existing-deployment) for the full Cognito-side flow). 2. **Teammate** pastes the bundle + temp password from the admin into: ```bash diff --git a/docs/src/content/docs/using/Roles.md b/docs/src/content/docs/using/Roles.md index f4c45ab7e..7ff440b1b 100644 --- a/docs/src/content/docs/using/Roles.md +++ b/docs/src/content/docs/using/Roles.md @@ -13,6 +13,6 @@ There are four lifecycle roles. They are often the same person early on, but the | **Repo onboarder** | Runs `bgagent linear onboard-project` (or registers a Blueprint via CDK) to wire a repo into the platform | As needed; any authenticated user | | **Teammate** | Runs `bgagent configure` once + `bgagent submit` / Linear or Jira label / Slack mention from then on | Daily user | -If you're a teammate joining an existing deployment, jump to [Joining an existing deployment](#joining-an-existing-deployment) below. +If you're a teammate joining an existing deployment, jump to [Joining an existing deployment](/sample-autonomous-cloud-coding-agents/using/authentication#joining-an-existing-deployment) below. -If you're standing up a new deployment from scratch, see the [Developer guide](/sample-autonomous-cloud-coding-agents/developer-guide/introduction) first, then come back here for the [admin onboarding flow](#get-stack-outputs). \ No newline at end of file +If you're standing up a new deployment from scratch, see the [Developer guide](/sample-autonomous-cloud-coding-agents/developer-guide/introduction) first, then come back here for the [admin onboarding flow](/sample-autonomous-cloud-coding-agents/using/authentication#get-stack-outputs). \ No newline at end of file diff --git a/docs/src/content/docs/using/Task-lifecycle.md b/docs/src/content/docs/using/Task-lifecycle.md index e416e934d..38b444ce0 100644 --- a/docs/src/content/docs/using/Task-lifecycle.md +++ b/docs/src/content/docs/using/Task-lifecycle.md @@ -28,7 +28,7 @@ The orchestrator uses Lambda Durable Functions to manage the lifecycle durably - | `SUBMITTED` | Task accepted; orchestrator invoked asynchronously | | `HYDRATING` | Orchestrator passed admission control; assembling the agent payload | | `RUNNING` | Agent session started and actively working on the task | -| `AWAITING_APPROVAL` | Agent paused at a Cedar HITL gate; waiting for your `approve` or `deny` decision. See [Approval gates](#approval-gates-cedar-hitl). | +| `AWAITING_APPROVAL` | Agent paused at a Cedar HITL gate; waiting for your `approve` or `deny` decision. See [Approval gates](/sample-autonomous-cloud-coding-agents/using/approval-gates-cedar-hitl). | | `FINALIZING` | Agent session ended; task is wrapping up (post-session hooks / PR finalization) before reaching a terminal state | | `COMPLETED` | Agent finished and created a PR (or determined no changes were needed) | | `FAILED` | Something went wrong - pre-flight check failed, concurrency limit reached, guardrail blocked the content, or the agent encountered an error | diff --git a/docs/src/content/docs/using/Using-the-cli.md b/docs/src/content/docs/using/Using-the-cli.md index 0cb9fa8d6..affbfaac9 100644 --- a/docs/src/content/docs/using/Using-the-cli.md +++ b/docs/src/content/docs/using/Using-the-cli.md @@ -114,7 +114,7 @@ Created: 2026-04-01T00:39:51.271Z | `--max-budget` | Maximum cost budget in USD (0.01–100). Overrides per-repo Blueprint default. No default limit. | | `--idempotency-key` | Idempotency key for deduplication. | | `--trace` | Enable detailed tracing: raises progress preview cap to 4 KB and uploads full NDJSON trajectory to S3 on completion. Download with `bgagent trace download`. | -| `--approval-timeout` | Cedar HITL decision window: `0` (default) keeps unanswered requests available; a positive value sets a 30–3,600 second deadline. A matching rule can require a shorter positive deadline. See [Approval gates](#approval-gates-cedar-hitl). | +| `--approval-timeout` | Cedar HITL decision window: `0` (default) keeps unanswered requests available; a positive value sets a 30–3,600 second deadline. A matching rule can require a shorter positive deadline. See [Approval gates](/sample-autonomous-cloud-coding-agents/using/approval-gates-cedar-hitl). | | `--microvm-sleep-after` | Seconds to wait for approval before putting a Lambda MicroVM to sleep (default 600 = 10 minutes; 0–3600 accepted). Use `off` to keep it awake. Requires the deployment's automatic-sleep feature to be enabled; does not change approval deadlines or affect other compute backends. | | `--pre-approve` | Cedar HITL scope to approve up-front (repeatable). Same scope forms as `bgagent approve --scope`. Hard-deny rules are always enforced. | | `--wait` | Poll until the task reaches a terminal status. | diff --git a/docs/src/content/docs/verification/645-p3-lifecycle-diagnostics.md b/docs/src/content/docs/verification/645-p3-lifecycle-diagnostics.md new file mode 100644 index 000000000..6f5d5743b --- /dev/null +++ b/docs/src/content/docs/verification/645-p3-lifecycle-diagnostics.md @@ -0,0 +1,77 @@ +--- +title: 645 p3 lifecycle diagnostics +--- + +# MicroVM lifecycle diagnostics + +An approval being saved, AWS accepting Resume, and the guest resuming work are +three different events. Diagnose each independently; an accepted API response +is not a completed wake. + +## Correlate coordinator and guest logs + +Search coordinator/approval Lambda logs and the MicroVM image log group using +`task_id` and `microvm_id`. `request_id` identifies the approval gate; +`aws_request_id` identifies an AWS call; `hook_id` identifies one guest HTTP hook. + +| Coordinator record | Meaning | +|---|---| +| `MicroVM observed after approval decision` | State and saved lifecycle intent observed after the decision. | +| `MicroVM wake request started/requested after approval decision` | Wake dispatch and acknowledgment, including receipt and elapsed time when available. | +| `MicroVM lifecycle request started/acknowledged/failed` | Suspend/Resume operation and result. | +| `MicroVM supervisor observation changed` | Task/worker/gate state, recovery timing, failure counters and outcome. | +| `MicroVM reached a terminal state with a substrate reason` | Service reason and worker/image identity. | +| `Lambda MicroVM termination requested` | Cleanup request and receipt. | + +Unchanged supervisor observations are deduplicated across durable replay. +Recovery timeouts retain their original start; investigating or retrying must +not reset them. + +For guest logs, select the deployed image's `/aws/lambda-microvms/` +log group and run a CloudWatch Logs Insights query such as: + +```text +fields @timestamp, event, action, stage, callback_stage, code, http_status, + hook_id, request_id, pid, phase, elapsed_ms, late, + error_type, aws_error_code, aws_request_id +| filter microvm_id = "REPLACE_WITH_WORKER_ID" +| sort @timestamp asc +| limit 500 +``` + +| Guest event | Meaning | +|---|---| +| `microvm_hook_started` | Handler entry, before body reading. | +| `microvm_hook_stage` | The next potentially blocking operation. | +| `microvm_hook_stage_finished` / `microvm_hook_stage_failed` | Operation completion or safe error metadata. | +| `microvm_hook_finished` | Selected handler status/code; not proof of service receipt. | + +`stage` records the last entered operation, such as credential refresh or approval +identity reconciliation. `late: true` means a callback finished after the handler +had already returned; it cannot turn a timed-out wake into success. The coding +barrier remains responsible for preventing tools after an uncertain wake. + +## Diagnose a failed wake + +1. Locate the saved decision/intent and actual Resume acknowledgment or failure. + Retain the original generation, deadlines and API request ID. +2. Find guest hook entry. If absent, account for log delivery and retention before + inferring anything about the listener or process. +3. If entered, inspect the final stage/code. Credential-refresh `AccessDenied` + points to renewal permissions; identity-read failures point to task/gate + reconciliation. A timeout identifies the outstanding operation. +4. Check subsequent guest progress, coordinator outcome, worker termination and + capacity release. A saved approval or successful cleanup does not establish + that the approved tool ran. +5. Retain UTC timestamps, task/worker identifiers, exact image/coordinator + versions, service state reason, AWS receipts and relevant sanitized logs. + +`MICROVM_RESUME_HOOK_FAILED` identifies recognized service resume-hook failures. +Preserve the raw service reason for diagnosis. The wording “connection was +refused” alone does not prove a closed listener: historical guest observations +support a stale pooled-connection race, while service-side dispatch traces remain +unavailable. Lifecycle responses explicitly close connections before freeze. + +Diagnostics omit hook bodies, tool arguments, approval contents, credentials, +raw exception messages and SDK response bodies. Do not attach signed payload +URLs or conversation/workspace checkpoints when escalating an incident. diff --git a/docs/src/content/docs/verification/645-p3-nested-stack.md b/docs/src/content/docs/verification/645-p3-nested-stack.md new file mode 100644 index 000000000..691e1c2b2 --- /dev/null +++ b/docs/src/content/docs/verification/645-p3-nested-stack.md @@ -0,0 +1,71 @@ +--- +title: 645 p3 nested stack +--- + +# Nested MicroVM infrastructure + +This is a CloudFormation infrastructure split, not a virtual machine running +inside another virtual machine. Fresh nested deployment and a deployment-specific +migration were verified; reusable migration commands remain unfinished. + +## Resource ownership + +`AgentStack` creates the `Microvm` child (`LambdaMicrovmStack`). It owns the +managed image when configured, build/runtime network connectors and security +groups, artifact/payload buckets, logs, and build/operator roles. + +The execution role remains at `LambdaMicrovmCompute/ExecutionRole` in the parent. +This preserves its logical ID and avoids a dependency cycle through SessionRole +trust. Parent `Microvm*` outputs retain the names used by packaging and consumers. +Names derive from the concrete parent deployment name, not the child stack token. + +## Configuration + +| Setting | Behavior | +|---|---| +| `microvm_nested_stack` | Defaults to `false`, preserving flat resource identities. Set `true` for a new installation or after completing the reviewed migration. | +| `microvm_resource_name_prefix` | Supplies distinct names for overlapping nested resources; preserve it after migration. It does not retain old resources or permissions by itself. | +| `microvm_managed_image_version` | Pins new tasks to an explicitly verified image version. Without a pin, selection follows the latest active version. | +| `microvm_approval_suspend_enabled` | Defaults to `false`; enable only after testing the deployed image/coordinator. Disabling new sleep preserves wake and cleanup. | + +Nested deployment requires bootstrap bundle **1.9.0** or later. See +[deployment roles](/sample-autonomous-cloud-coding-agents/architecture/deployment-roles) and the +[artifact packaging instructions](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/cdk/scripts/README.md). +Build a new image before switching its runtime pin; building alone must not +silently change the image used by a pinned coordinator. Retain a compatible +published coordinator and exact image version for rollback. + +## Existing flat deployments + +Do not deploy the nested template directly over a flat deployment. CloudFormation +sees removed parent resources and newly created child resources; names can collide +and bucket auto-delete handlers can erase artifacts or pending payloads. + +The tested provider rejected native `AWS::Lambda::MicrovmImage` stack refactoring +with an unsupported tag-schema error even though the preview succeeded. Do not +assume resource import/refactoring is supported from a successful preview. + +The required overlap migration has these stages. This checklist is not yet a +runnable migration command: + +1. Preserve the exact deployed template/cloud assembly, image and published + coordinator versions. Capture current resource identities and configuration. + Keep `microvm_nested_stack=false` on ordinary updates before migration. +2. Create the child with distinct names while retaining the old resources and + their permissions. Build and verify the new image before switching consumers. + Reject intermediate templates that exceed CloudFormation's 500-resource limit. +3. Deploy compatible approval handlers/coordinator before producers of retained + requests. Preserve permissions for old workers, including the old payload + access that P3's new bootstrap deny would otherwise block. +4. Switch consumers with an explicit image pin. Verify normal tasks, retained + approvals, sleep/wake and rollback while both resource sets are available. +5. Retire old resources only after old workers, Durable executions and uncertain + starts are accounted for and old tasks/leases are drained. An idle inventory + snapshot alone is not an admission fence. Preserve checkpoint data and any + resources still needed by published coordinator versions. + +Setting a concurrency counter to zero is not a reliable pause for asynchronous +producers. A rollback must restore compatible code, image selection and IAM +without deleting pending requests or saved work. The reusable tool must enforce +these prerequisites; migration template and permission helpers alone do not +constitute that tool. Track remaining acceptance in [verification status](/sample-autonomous-cloud-coding-agents/verification/readme#open-pr-checks). diff --git a/docs/src/content/docs/verification/645-payload-bootstrap.md b/docs/src/content/docs/verification/645-payload-bootstrap.md new file mode 100644 index 000000000..f722d4e98 --- /dev/null +++ b/docs/src/content/docs/verification/645-payload-bootstrap.md @@ -0,0 +1,83 @@ +--- +title: 645 payload bootstrap +--- + +# Trusted task delivery for ECS and MicroVM + +The coordinator delivers each task through an IAM-authenticated deployment +manifest and a signed URL for one task object. This is application startup +configuration, distinct from the CDK infrastructure bootstrap. + +## Contract and permissions + +Version 2 is defined in `contracts/constants.json` under `payload_bootstrap`. +The [ADR wire contract](/sample-autonomous-cloud-coding-agents/decisions/adr-021-lambda-microvms-compute-backend#3-packaging-same-agent-image-source-new-build-path) +and [security design](/sample-autonomous-cloud-coding-agents/architecture/security) describe its trust boundary. + +| Object | Purpose | +|---|---| +| `bootstrap/.json` | Non-secret backend/platform configuration; coordinator writes, worker reads with its ambient role. | +| `/payload.json` | Task instructions/configuration; downloaded only through the signed capability. | +| `/launch.json` | Private coordinator replay record containing the exact saved reference. | + +The worker's explicit S3 deny outside its own `bootstrap/*` prevents a foreign +public bucket from supplying fake configuration. A content hash alone does not +establish authorship. Workers cannot list their payload bucket or use ambient +credentials to read task objects. The coordinator needs `ListBucket` to distinguish +missing launch records from access denial. + +Downloaded task identity and configuration must match the authenticated manifest. +Old unsigned envelopes are rejected. All payload sizes use S3; the serialized +MicroVM reference must fit 4,096 bytes. Manifest/payload limits are 16 KiB/8 MiB. +MicroVM receives the reference through `runHookPayload`; ECS uses +`AGENT_PAYLOAD_REF` and removes it before repository subprocesses start. + +A signed URL is a bearer credential. Never log it or store it in a worker-readable +task row. HTTPS downloads reject redirects, environment proxies, alternate hosts +and invalid task paths. The runtime needs regional S3 HTTPS and DNS connectivity. + +## Retry and cleanup + +- Conditional writes keep task instructions immutable. Read-back reconciles a + write that succeeded but lost its response. +- Retries reuse the exact saved URL and Run client token. Re-signing would change + the request under the same token. An expired record fails explicitly. +- Signed lifetime is at most 900 seconds and can end earlier when temporary + credentials expire. Initial creation refuses a known lifetime under 300 seconds. +- A timeout does not prove no worker started. Reconcile the existing handle or + uncertain start before considering replacement; the application replay window + is not a service idempotency-retention guarantee. +- Finalization deletes payload and launch objects on a best-effort basis; bucket + lifecycle deletion is an asynchronous backstop. The coordinator refreshes + identical manifest bytes on preparation so old deployment settings remain usable. +- Resume continues saved task state with refreshed credentials; it does not + download the task again using an expired launch URL. + +## Coordinated upgrades + +Coordinator code, worker images/task definitions and IAM must implement the same +contract. Keep the previous deployable artifacts and inspect the change set. +For an upgrade that cannot support overlapping versions, pause actual admission +sources and drain tasks, approvals and uncertain starts before switching these +components together. A concurrency counter alone does not pause webhook/queue +admission. Do not delete and recreate storage to perform an upgrade. + +For overlapping flat-to-nested migration, preserve old-worker permissions until +those workers drain; follow the [migration prerequisites](/sample-autonomous-cloud-coding-agents/verification/645-p3-nested-stack). +Rollback also requires compatible code, images and IAM. Never reuse a task ID +with conflicting or expired stored launch instructions. + +## Verification + +Local producer/consumer tests cover conflicts, lost committed replies, malformed +or oversized input, wrong task/configuration, URL redaction and cleanup. These +cannot prove effective AWS permissions. In a representative deployment, verify: + +- Own manifest and signed payload succeed; foreign manifests, ambient payload + reads, bucket listing and worker mutations are denied. +- Modified signatures/paths, expired credentials/URLs and revoked objects fail + without starting the pipeline or exposing a capability in logs. +- Competing preparation and lost S3/Run replies preserve one exact launch request. +- Finalization removes both task objects and releases only confirmed capacity. + +See [recorded acceptance and remaining checks](/sample-autonomous-cloud-coding-agents/verification/readme). diff --git a/docs/src/content/docs/verification/Readme.md b/docs/src/content/docs/verification/Readme.md new file mode 100644 index 000000000..eaac8b951 --- /dev/null +++ b/docs/src/content/docs/verification/Readme.md @@ -0,0 +1,99 @@ +--- +title: Readme +--- + +# Lambda MicroVM verification + +For maintainers reviewing/testing the MicroVM backend and operators deploying, +migrating or diagnosing it. [ADR-021](/sample-autonomous-cloud-coding-agents/decisions/adr-021-lambda-microvms-compute-backend) +explains the design; the [user guide](/sample-autonomous-cloud-coding-agents/using/approval-gates-cedar-hitl) +explains approval and sleep options for people submitting tasks. +Detailed deployment transcripts, temporary worker identifiers and investigation +diaries are archived outside the repository. These documents are not test runners. + +- [Task payload delivery](/sample-autonomous-cloud-coding-agents/verification/645-payload-bootstrap): authorization, retries and coordinated upgrades. +- [Nested infrastructure](/sample-autonomous-cloud-coding-agents/verification/645-p3-nested-stack): configuration and migration prerequisites. +- [Lifecycle diagnostics](/sample-autonomous-cloud-coding-agents/verification/645-p3-lifecycle-diagnostics): locating and interpreting failed wakes. +- [Continuation design](/sample-autonomous-cloud-coding-agents/architecture/orchestrator#retained-microvm-approvals): checkpoint ownership, retirement and replacement. + +## Recorded acceptance + +AWS checks on September 14–18, 2026 exercised the following behavior in the tested +deployments. They do not certify a different image, account, Region or upgrade. + +| Area | Observed result | +|---|---| +| Task delivery | Signed payload downloads, invalid input rejection, expiry/revocation, immutable preparation and lost-reply recovery passed. | +| Approval lifecycle | Approve, deny, explicit expiry and cancellation passed; unanswered requests stayed available without a default deadline. | +| Sleep/wake | Repeated wakes, the 600-second default and the live sleep-off switch passed with compatible image/coordinator versions. | +| Credential renewal | A wait exceeding one hour was followed by successful AWS access with renewed task credentials. | +| Continuation | Conversation and Git/workspace recovery, replacement admission, usage limits and capacity release passed. | +| External integrations | Repository work and remote MCP access passed across sleep; an actual Linear submission exercised MicroVM compute with AgentCore Identity vault. | +| Linear decisions | Native threaded `approve` and `deny` replies passed on MicroVM and AgentCore. Both MicroVMs were suspended before the replies; exact decisions, tool results, thread acknowledgements and eventual capacity release were verified. | +| Infrastructure | Fresh nested deployment and a deployment-specific overlapping migration passed, including compatible rollback and old-resource cleanup. | +| Other backends | ECS and AgentCore approval/cancellation and scoped-access checks passed in their tested deployments. | + +The wake correction sends `Connection: close` in lifecycle responses before +freeze. Local transport controls reproduced failure on an old connection; +long-sleep controls and the corrected live flows passed. Service-side traces +for the historical failures remain unavailable, so their exact transport error +is not established for every worker. + +The full local build after the resource-budget fix passed: 5,560 CDK tests, +2,170 agent tests and 1,005 CLI tests, plus compile, lint, contracts, docs and +synthesis. The widest parent stacks use 489 resources for ECS and 488 for +MicroVM, including synth metadata, within the unchanged 490-resource budget. +Concurrency maintenance now has its own nested stack. Upgrading recreates that +stateless repair function and schedule; task and concurrency tables stay in the +parent stack. + +## Open PR checks + +- Finish reusable flat-to-nested migration commands and independently test an + upgrade from current `main` on the same deployment. The earlier bespoke + migration is not a substitute for that acceptance. + +## Reproduce local checks + +From the repository root, with dependencies installed: + +```bash +mise run build +MISE_EXPERIMENTAL=1 mise //cdk:testf -- 'microvm|migration|payload-bootstrap' +``` + +For focused worker checks: + +```bash +cd agent +uv run pytest tests/test_microvm_*.py tests/test_continuation_*.py \ + tests/test_approval_retention.py tests/test_payload_bootstrap.py --no-cov +ABCA_TEST_SDK_CONTINUATION=1 uv run pytest tests/test_continuation_sdk_probe.py --no-cov +``` + +The last command opts into the pinned real SDK/CLI probe with a deterministic +loopback model; it does not launch a cloud worker. Optional DynamoDB Local tests +require their documented local service. Mocks do not establish effective AWS IAM. +The standalone cloud acceptance harness and raw receipts remain outside this PR. + +## Live acceptance for an installation + +Record source commit, image ARN/version, coordinator version, configuration and +UTC test window. Use an owned test repository and identity. Enable suspension +only after verifying that image and coordinator together. + +| Exercise | Required observation | +|---|---| +| Normal task | Repository tools run; terminal status, payload cleanup and capacity release agree. | +| Approval after sleep | The saved decision reaches the exact pending tool; verify guest recovery and tool output, not only the Resume API receipt. | +| Linear replies | Submit real issues on each backend. Reply `approve` or `deny` to the exact approval comment as its linked owner; verify the saved decision source, same-thread acknowledgement and allowed/blocked tool result. For MicroVM, observe suspension before replying. | +| Deny, expiry, cancellation | No denied/cancelled tool runs; expiry uses the original deadline; compute and capacity are cleaned up. | +| Default and disabled sleep | Omitted override uses 600 seconds; task-level off and deployment off prevent new suspension while wake/cleanup remain available. | +| Credential expiry | Sleep past the original credential lifetime, then perform actual task-scoped AWS operations. | +| Retirement/replacement | Confirm old-worker shutdown before capacity release; one replacement restores files/conversation and preserves approval identity and usage. | +| Upgrade/rollback | Preserve unrelated resource identities, old in-flight work and recoverable checkpoints; test compatible code/image/policy rollback. | + +Include effective-role checks for own-task access and denial of cross-task data, +foreign bootstrap manifests and ambient payload reads. Keep credentials, signed +URLs, prompts and checkpoints out of diagnostic attachments. Clean up test +workers, executions, task data and owned infrastructure after the run. From c00d33c82fbc934c1df73fd021f20c016ffa1855 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Mon, 21 Sep 2026 09:25:47 -0400 Subject: [PATCH 117/149] test(645): supply approval service in MicroVM composition fixtures --- cdk/test/handlers/orchestrate-task-microvm.test.ts | 1 + cdk/test/handlers/shared/microvm-start-recovery.test.ts | 1 + cdk/test/handlers/start-session-composition.test.ts | 1 + 3 files changed, 3 insertions(+) diff --git a/cdk/test/handlers/orchestrate-task-microvm.test.ts b/cdk/test/handlers/orchestrate-task-microvm.test.ts index bb0017c58..6ef344589 100644 --- a/cdk/test/handlers/orchestrate-task-microvm.test.ts +++ b/cdk/test/handlers/orchestrate-task-microvm.test.ts @@ -154,6 +154,7 @@ process.env.MICROVM_EXECUTION_ROLE_ARN = 'arn:aws:iam::123456789012:role/AbcaMic process.env.MICROVM_EGRESS_CONNECTOR_ARNS = 'arn:aws:lambda:us-east-1:123456789012:network-connector/egress-1'; process.env.MICROVM_PAYLOAD_BUCKET = 'test-microvm-payload-bucket'; process.env.TASK_TABLE_NAME = 'Tasks'; +process.env.APPROVAL_REQUESTS_API_URL = 'https://approval.execute-api.us-east-1.amazonaws.com/v1'; process.env.TASK_EVENTS_TABLE_NAME = 'TaskEvents'; process.env.USER_CONCURRENCY_TABLE_NAME = 'UserConcurrency'; process.env.TASK_RETENTION_DAYS = '90'; diff --git a/cdk/test/handlers/shared/microvm-start-recovery.test.ts b/cdk/test/handlers/shared/microvm-start-recovery.test.ts index 7e2192682..9f542c6ab 100644 --- a/cdk/test/handlers/shared/microvm-start-recovery.test.ts +++ b/cdk/test/handlers/shared/microvm-start-recovery.test.ts @@ -48,6 +48,7 @@ jest.mock('../../../src/handlers/shared/logger', () => ({ Object.assign(process.env, { TASK_TABLE_NAME: 'tasks', + APPROVAL_REQUESTS_API_URL: 'https://approval.execute-api.us-east-1.amazonaws.com/v1', TASK_EVENTS_TABLE_NAME: 'events', MICROVM_IMAGE_IDENTIFIER: 'arn:aws:lambda:us-east-1:123456789012:microvm-image:test', MICROVM_IMAGE_VERSION: '1', diff --git a/cdk/test/handlers/start-session-composition.test.ts b/cdk/test/handlers/start-session-composition.test.ts index fac6ea82c..55e32c42e 100644 --- a/cdk/test/handlers/start-session-composition.test.ts +++ b/cdk/test/handlers/start-session-composition.test.ts @@ -91,6 +91,7 @@ let ulidCounter = 0; jest.mock('ulid', () => ({ ulid: jest.fn(() => `ULID${ulidCounter++}`) })); process.env.TASK_TABLE_NAME = 'Tasks'; +process.env.APPROVAL_REQUESTS_API_URL = 'https://approval.execute-api.us-east-1.amazonaws.com/v1'; process.env.TASK_EVENTS_TABLE_NAME = 'TaskEvents'; process.env.USER_CONCURRENCY_TABLE_NAME = 'UserConcurrency'; process.env.RUNTIME_ARN = 'arn:aws:bedrock-agentcore:us-east-1:123456789012:runtime/test'; From b71bef0edf9f5eb1e7c8bb4aa161a752d9030e99 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Mon, 21 Sep 2026 09:43:25 -0400 Subject: [PATCH 118/149] test(645): pin approval endpoint forwarding across compute backends --- cdk/test/constructs/task-orchestrator.test.ts | 8 ++++---- cdk/test/handlers/orchestrate-task-microvm.test.ts | 6 +----- cdk/test/handlers/start-session-composition.test.ts | 6 +----- cdk/test/stacks/agent.test.ts | 5 +++-- 4 files changed, 9 insertions(+), 16 deletions(-) diff --git a/cdk/test/constructs/task-orchestrator.test.ts b/cdk/test/constructs/task-orchestrator.test.ts index 313100c6e..472d8ec6e 100644 --- a/cdk/test/constructs/task-orchestrator.test.ts +++ b/cdk/test/constructs/task-orchestrator.test.ts @@ -965,14 +965,13 @@ describe('TaskOrchestrator agentPlatformConfig (ADR-021 P2 platform_config trans expect(env.ANTHROPIC_MODEL).toBe(MAIN_PROFILE); }); - test('carries the four REQUIRED platform_config sources together (never a partial set)', () => { - // The strategy refuses to start a lambda-microvm session without these four. - // Three come from the orchestrator's own wiring and one from this block, so - // this is the assertion that they are all reachable from ONE deploy. + test('carries all required platform configuration, including the approval service', () => { + // One deployment must supply the complete configuration required at launch. const env = orchestratorEnvVars(createStack({ githubTokenSecretArn: 'arn:aws:secretsmanager:us-east-1:123456789012:secret:github-token-abc123', agentPlatformConfig: { taskApprovalsTableName: 'approvals', + approvalRequestsApiUrl: 'https://approval.execute-api.us-east-1.amazonaws.com/v1', nudgesTableName: 'nudges', logGroupName: '/aws/abca/application', artifactsBucketName: 'artifacts', @@ -982,6 +981,7 @@ describe('TaskOrchestrator agentPlatformConfig (ADR-021 P2 platform_config trans anthropicModel: MAIN_PROFILE, }, }).template); + expect(env.APPROVAL_REQUESTS_API_URL).toBe('https://approval.execute-api.us-east-1.amazonaws.com/v1'); expect(env.TASK_TABLE_NAME).toBeDefined(); expect(env.TASK_EVENTS_TABLE_NAME).toBeDefined(); expect(env.GITHUB_TOKEN_SECRET_ARN).toBeDefined(); diff --git a/cdk/test/handlers/orchestrate-task-microvm.test.ts b/cdk/test/handlers/orchestrate-task-microvm.test.ts index 6ef344589..dfb90358e 100644 --- a/cdk/test/handlers/orchestrate-task-microvm.test.ts +++ b/cdk/test/handlers/orchestrate-task-microvm.test.ts @@ -159,11 +159,7 @@ process.env.TASK_EVENTS_TABLE_NAME = 'TaskEvents'; process.env.USER_CONCURRENCY_TABLE_NAME = 'UserConcurrency'; process.env.TASK_RETENTION_DAYS = '90'; -// platform_config (ADR-021 P2): the four REQUIRED identifiers the MicroVM -// strategy refuses to start a session without — they are the agent's only -// channel for them, since a snapshot must not bake configuration in. Read at -// call time by `buildMicrovmPlatformConfig`, but set here alongside the rest -// for clarity. +// Required platform configuration is supplied at launch, never baked into the image. process.env.GITHUB_TOKEN_SECRET_ARN = 'arn:aws:secretsmanager:us-east-1:123456789012:secret:abca/github-token-AbCdEf'; process.env.AGENT_SESSION_ROLE_ARN = 'arn:aws:iam::123456789012:role/AbcaAgentSessionRole'; diff --git a/cdk/test/handlers/start-session-composition.test.ts b/cdk/test/handlers/start-session-composition.test.ts index 55e32c42e..4591dd4a0 100644 --- a/cdk/test/handlers/start-session-composition.test.ts +++ b/cdk/test/handlers/start-session-composition.test.ts @@ -102,11 +102,7 @@ process.env.MICROVM_EXECUTION_ROLE_ARN = 'arn:aws:iam::123456789012:role/AbcaMic process.env.MICROVM_EGRESS_CONNECTOR_ARNS = 'arn:aws:lambda:us-east-1:123456789012:network-connector/egress-1'; process.env.MICROVM_PAYLOAD_BUCKET = 'test-microvm-payload-bucket'; -// platform_config (ADR-021 P2): the four REQUIRED identifiers the MicroVM -// strategy refuses to start a session without — they are the agent's only -// channel for them, since a snapshot must not bake configuration in. Read at -// call time by `buildMicrovmPlatformConfig`, but set here alongside the rest -// for clarity. +// Required platform configuration is supplied at launch, never baked into the image. process.env.GITHUB_TOKEN_SECRET_ARN = 'arn:aws:secretsmanager:us-east-1:123456789012:secret:abca/github-token-AbCdEf'; process.env.AGENT_SESSION_ROLE_ARN = 'arn:aws:iam::123456789012:role/AbcaAgentSessionRole'; diff --git a/cdk/test/stacks/agent.test.ts b/cdk/test/stacks/agent.test.ts index c5b3f42ea..1cd1d55c7 100644 --- a/cdk/test/stacks/agent.test.ts +++ b/cdk/test/stacks/agent.test.ts @@ -153,6 +153,7 @@ describe('AgentStack', () => { for (const key of [ 'TASK_APPROVALS_TABLE_NAME', + 'APPROVAL_REQUESTS_API_URL', 'NUDGES_TABLE_NAME', 'LOG_GROUP_NAME', 'ARTIFACTS_BUCKET_NAME', @@ -166,8 +167,7 @@ describe('AgentStack', () => { ]) { expect(env[key]).toBeDefined(); } - // Plus the three the orchestrator already carried for its own work — together - // these cover all four identifiers the MicroVM strategy treats as required. + // These existing coordinator values complete the required platform configuration. expect(env.TASK_TABLE_NAME).toBeDefined(); expect(env.TASK_EVENTS_TABLE_NAME).toBeDefined(); expect(env.GITHUB_TOKEN_SECRET_ARN).toBeDefined(); @@ -187,6 +187,7 @@ describe('AgentStack', () => { for (const key of [ 'TASK_APPROVALS_TABLE_NAME', + 'APPROVAL_REQUESTS_API_URL', 'NUDGES_TABLE_NAME', 'LOG_GROUP_NAME', 'ARTIFACTS_BUCKET_NAME', From 359b69bdd273a02e4d6a6d6c23b354609dca9a1a Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Mon, 21 Sep 2026 15:28:39 -0400 Subject: [PATCH 119/149] fix(645): respect terminal cancellation races and test pipeline wiring --- agent/src/pipeline.py | 7 +-- agent/tests/test_pipeline.py | 116 +++++++++++++++++++++++++++++++++-- 2 files changed, 114 insertions(+), 9 deletions(-) diff --git a/agent/src/pipeline.py b/agent/src/pipeline.py index 07dfff374..b5bfdc39c 100644 --- a/agent/src/pipeline.py +++ b/agent/src/pipeline.py @@ -472,14 +472,13 @@ def _run_repoless_task( def _persist_finished_task(task_id: str, status: str, result: dict) -> None: outcome = task_state.write_terminal(task_id, status, result) - if outcome in ( - task_state.TerminalWriteOutcome.FAILED, - task_state.TerminalWriteOutcome.SUPERSEDED, - ): + if outcome == task_state.TerminalWriteOutcome.FAILED: raise task_state.TerminalWriteError( f"Task result was not committed ({outcome.value}); " "inspect the task record and worker lease" ) + # A cancel or another terminal writer won the status race. Its result stays + # authoritative; this is not a worker crash and must not emit a failure reaction. def _apply_post_hook_gates( diff --git a/agent/tests/test_pipeline.py b/agent/tests/test_pipeline.py index f8a3862a5..98d7ab957 100644 --- a/agent/tests/test_pipeline.py +++ b/agent/tests/test_pipeline.py @@ -1,6 +1,7 @@ """Unit tests for pipeline.py — cedar_policies injection and pure helpers.""" import os +from contextlib import ExitStack from unittest.mock import MagicMock, patch import pytest @@ -2342,22 +2343,21 @@ def test_setup_failure_swaps_eyes_to_cross( assert kwargs.get("started_reaction_id") == "reaction-42" -@pytest.mark.parametrize("outcome", ["failed", "superseded"]) -def test_finished_pipeline_cannot_report_success_without_committing_result(monkeypatch, outcome): +def test_finished_pipeline_cannot_report_success_without_committing_result(monkeypatch): import task_state from pipeline import _persist_finished_task monkeypatch.setattr( task_state, "write_terminal", - MagicMock(return_value=task_state.TerminalWriteOutcome(outcome)), + MagicMock(return_value=task_state.TerminalWriteOutcome.FAILED), ) with pytest.raises(task_state.TerminalWriteError, match="Task result was not committed"): _persist_finished_task("task", "COMPLETED", {"status": "success"}) -@pytest.mark.parametrize("outcome", ["written", "disabled"]) -def test_finished_pipeline_accepts_persisted_result_or_local_run(monkeypatch, outcome): +@pytest.mark.parametrize("outcome", ["written", "disabled", "superseded"]) +def test_finished_pipeline_accepts_persistence_local_run_or_supersession(monkeypatch, outcome): import task_state from pipeline import _persist_finished_task @@ -2367,3 +2367,109 @@ def test_finished_pipeline_accepts_persisted_result_or_local_run(monkeypatch, ou MagicMock(return_value=task_state.TerminalWriteOutcome(outcome)), ) _persist_finished_task("task", "COMPLETED", {"status": "success"}) + + +@pytest.mark.parametrize("repo_url", ["owner/repo", ""]) +@pytest.mark.parametrize("outcome", ["failed", "superseded"]) +def test_terminal_persistence_through_task_entry_point(monkeypatch, repo_url, outcome): + """Both actual pipeline paths must distinguish storage failure from cancellation.""" + import task_state + from pipeline import run_task + + monkeypatch.setenv("AWS_REGION", "us-east-1") + monkeypatch.setenv("ARTIFACTS_BUCKET_NAME", "artifacts-bkt") + + async def agent_result(*args, **kwargs): + return AgentResult( + status="success", turns=1, num_turns=1, result_text="The README describes the app." + ) + + finished = MagicMock() + terminal = MagicMock(return_value=task_state.TerminalWriteOutcome(outcome)) + with ExitStack() as stack: + for context in ( + patch("runner.run_agent", side_effect=agent_result), + patch( + "repo.setup_repo", return_value=RepoSetup(repo_dir="/workspace/repo", branch="test") + ), + patch("pipeline.task_span"), + patch("pipeline.discover_project_config"), + patch("pipeline.build_system_prompt"), + patch("pipeline.resolve_linear_api_token"), + patch("pipeline.configure_channel_mcp"), + patch("pipeline.react_task_started", return_value="started-reaction"), + patch("pipeline.comment_task_started"), + patch("pipeline.transition_task_started"), + patch("pipeline.react_task_finished", finished), + patch("pipeline.ensure_committed", return_value=False), + patch("pipeline.verify_build", return_value=VerifyOutcome(passed=True)), + patch("pipeline.verify_lint", return_value=VerifyOutcome(passed=True)), + patch("pipeline.ensure_pr", return_value="https://github.com/owner/repo/pull/1"), + patch("pipeline.get_disk_usage", return_value=0), + patch("pipeline.print_metrics"), + patch("pipeline._maybe_upload_trace", return_value=None), + patch("aws_session.tenant_client", return_value=MagicMock()), + patch.object(task_state, "write_running"), + patch.object(task_state, "write_terminal", terminal), + ): + stack.enter_context(context) + kwargs = dict( + repo_url=repo_url, + task_description="Read the README", + github_token="ghp_test", + aws_region="us-east-1", + task_id="terminal-race", + channel_source="linear", + channel_metadata=_LINEAR_META, + ) + if not repo_url: + kwargs["resolved_workflow"] = {"id": "default/agent-v1", "version": "1.0.0"} + if outcome == "failed": + with pytest.raises( + task_state.TerminalWriteError, match="Task result was not committed" + ): + run_task(**kwargs) + else: + result = run_task(**kwargs) + assert result["status"] == "success" + terminal.assert_called_once() + assert not any(call.kwargs.get("success") is False for call in finished.call_args_list) + assert terminal.call_args_list[0].args[1] == "COMPLETED" + + +def test_normal_terminal_writes_cannot_bypass_outcome_handling(): + """Direct best-effort writes are reserved for the existing crash handler.""" + import ast + from pathlib import Path + + import pipeline + + class TerminalCalls(ast.NodeVisitor): + function = "" + in_exception = False + + def visit_FunctionDef(self, node): + previous = self.function + self.function = node.name + self.generic_visit(node) + self.function = previous + + def visit_ExceptHandler(self, node): + previous = self.in_exception + self.in_exception = True + self.generic_visit(node) + self.in_exception = previous + + def visit_Call(self, node): + if ( + isinstance(node.func, ast.Attribute) + and isinstance(node.func.value, ast.Name) + and node.func.value.id == "task_state" + and node.func.attr == "write_terminal" + ): + assert self.function == "_persist_finished_task" or ( + self.function == "run_task" and self.in_exception + ), f"Unchecked terminal outcome at pipeline.py:{node.lineno}" + self.generic_visit(node) + + TerminalCalls().visit(ast.parse(Path(pipeline.__file__).read_text())) From da8344dbab10f8e8d1690779d78a3be8a4e6190b Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Mon, 21 Sep 2026 15:34:15 -0400 Subject: [PATCH 120/149] fix(645): classify checkpoint failures without assuming corrupt data --- agent/src/continuation_session.py | 6 +++-- agent/src/continuation_usage.py | 11 +++++++-- agent/tests/test_continuation_session.py | 29 ++++++++++++++++++++++-- agent/tests/test_continuation_usage.py | 15 ++++++++++++ 4 files changed, 55 insertions(+), 6 deletions(-) diff --git a/agent/src/continuation_session.py b/agent/src/continuation_session.py index 16e6f3ae0..78d25dd86 100644 --- a/agent/src/continuation_session.py +++ b/agent/src/continuation_session.py @@ -40,7 +40,7 @@ class ContinuationCheckpointError(RuntimeError): """A checkpoint cannot be acknowledged or safely restored.""" - def __init__(self, message: str, *, code: str = "checkpoint_invalid") -> None: + def __init__(self, message: str, *, code: str = "checkpoint_failed") -> None: super().__init__(message) self.code = code @@ -361,7 +361,9 @@ class S3ContinuationCheckpoints: def __init__(self, bucket: str, *, client: Any = None) -> None: if not isinstance(bucket, str) or not bucket or "/" in bucket: - raise ContinuationCheckpointError("Checkpoint bucket is unavailable") + raise ContinuationCheckpointError( + "Checkpoint bucket is unavailable", code="checkpoint_storage_unavailable" + ) if client is None: from botocore.config import Config diff --git a/agent/src/continuation_usage.py b/agent/src/continuation_usage.py index 242290876..cab40900e 100644 --- a/agent/src/continuation_usage.py +++ b/agent/src/continuation_usage.py @@ -59,8 +59,15 @@ async def read_usage(client: Any) -> UsageSnapshot: query = getattr(client, "_query", None) send = getattr(query, "_send_control_request", None) if not callable(send): - raise ContinuationCheckpointError("Continuation accounting client is unavailable") - response = await asyncio.wait_for(send({"subtype": "get_usage"}), timeout=5) + raise ContinuationCheckpointError( + "Continuation accounting client is unavailable", code="checkpoint_sdk_unverified" + ) + try: + response = await asyncio.wait_for(send({"subtype": "get_usage"}), timeout=5) + except TimeoutError as exc: + raise ContinuationCheckpointError( + "Continuation accounting request timed out", code="checkpoint_sdk_timeout" + ) from exc session = response.get("session") if isinstance(response, dict) else None if not isinstance(session, dict) or not valid_cost(session.get("total_cost_usd")): raise ContinuationCheckpointError("Continuation accounting response has no valid cost") diff --git a/agent/tests/test_continuation_session.py b/agent/tests/test_continuation_session.py index 5da28d722..2350f8281 100644 --- a/agent/tests/test_continuation_session.py +++ b/agent/tests/test_continuation_session.py @@ -115,8 +115,11 @@ async def scenario(): def test_missing_mirror_never_certifies_the_checkpoint(self): async def scenario(): - with pytest.raises(checkpoint.ContinuationCheckpointError, match="did not acknowledge"): + with pytest.raises( + checkpoint.ContinuationCheckpointError, match="did not acknowledge" + ) as error: await capture(checkpoint.CheckpointSessionStore(PROJECT)) + assert error.value.code == "checkpoint_sdk_timeout" asyncio.run(scenario()) @@ -259,6 +262,11 @@ def get_object(self, **kwargs): class TestImmutableStorage: + def test_missing_bucket_reports_storage_configuration_failure(self): + with pytest.raises(checkpoint.ContinuationCheckpointError) as error: + checkpoint.S3ContinuationCheckpoints("") + assert error.value.code == "checkpoint_storage_unavailable" + @pytest.fixture def storage(self): client = VersionedS3() @@ -296,6 +304,16 @@ def test_save_requires_read_back_and_returns_a_version_pinned_receipt(self, stor assert client.calls[-1][1]["VersionId"] == receipt.version_id assert all(stream.closed for stream in client.streams) + def test_unreadable_saved_version_reports_storage_failure(self, storage, body): + store, client = storage + receipt = store.save(body, IDENTITY) + client.get_object = Mock( + side_effect=ClientError({"Error": {"Code": "AccessDenied"}}, "GetObject") + ) + with pytest.raises(checkpoint.ContinuationCheckpointError) as error: + store.load(receipt, IDENTITY) + assert error.value.code == "checkpoint_storage_unavailable" + def test_lost_write_reply_recovers_from_exact_read_back(self, storage, body): store, client = storage client.lose_write_reply = True @@ -373,8 +391,11 @@ def test_corrupt_or_incompatible_envelope_fails(self, body, change): @pytest.mark.parametrize("body", [b"{", b"\xff", b"null", b""]) def test_invalid_json_fails(self, body): - with pytest.raises(checkpoint.ContinuationCheckpointError): + with pytest.raises(checkpoint.ContinuationCheckpointError) as error: checkpoint.decode_checkpoint(body, IDENTITY) + assert error.value.code == ( + "checkpoint_failed" if body in (b"null", b"") else "checkpoint_invalid_json" + ) @pytest.mark.parametrize("value", ["../task", "", "task/other", "x" * 129]) def test_path_components_cannot_escape_task_prefix(self, value): @@ -386,3 +407,7 @@ def test_invalid_json_reports_a_specific_checkpoint_code(): with pytest.raises(checkpoint.ContinuationCheckpointError) as error: checkpoint._encode({"not_json": object()}) assert error.value.code == "checkpoint_invalid_json" + + +def test_unspecified_checkpoint_failure_does_not_claim_invalid_data(): + assert checkpoint.ContinuationCheckpointError("Unknown failure").code == "checkpoint_failed" diff --git a/agent/tests/test_continuation_usage.py b/agent/tests/test_continuation_usage.py index 2302c2045..81eb9b046 100644 --- a/agent/tests/test_continuation_usage.py +++ b/agent/tests/test_continuation_usage.py @@ -79,3 +79,18 @@ def test_unverified_sdk_upgrade_requires_explicit_accounting_validation(monkeypa with pytest.raises(ContinuationCheckpointError, match="verified SDK") as error: asyncio.run(read_usage(None)) assert error.value.code == "checkpoint_sdk_unverified" + + +def test_missing_sdk_accounting_interface_is_not_invalid_checkpoint_data(): + with pytest.raises(ContinuationCheckpointError) as error: + asyncio.run(read_usage(SimpleNamespace())) + assert error.value.code == "checkpoint_sdk_unverified" + + +def test_sdk_accounting_timeout_reports_transport_failure(): + client = SimpleNamespace( + _query=SimpleNamespace(_send_control_request=AsyncMock(side_effect=TimeoutError)) + ) + with pytest.raises(ContinuationCheckpointError) as error: + asyncio.run(read_usage(client)) + assert error.value.code == "checkpoint_sdk_timeout" From 7439ecbb41cbd0e3e861a6d7427c910adef33750 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Mon, 21 Sep 2026 15:38:45 -0400 Subject: [PATCH 121/149] fix(645): reject missing cloud approval configuration with actionable errors --- agent/src/approval_requests.py | 8 +++++++- agent/src/task_state.py | 4 ++-- agent/tests/test_approval_requests.py | 14 ++++++++++++++ cdk/src/constructs/approval-request-service.ts | 2 +- cdk/src/constructs/task-orchestrator.ts | 3 +++ cdk/src/handlers/request-approval.ts | 7 ++++++- cdk/test/constructs/task-orchestrator.test.ts | 14 ++++++++++++-- cdk/test/handlers/request-approval.test.ts | 16 +++++++++++----- 8 files changed, 56 insertions(+), 12 deletions(-) diff --git a/agent/src/approval_requests.py b/agent/src/approval_requests.py index 5c36ecc2e..16b4785ad 100644 --- a/agent/src/approval_requests.py +++ b/agent/src/approval_requests.py @@ -22,7 +22,13 @@ def configured() -> bool: - return bool(os.environ.get(API_ENV)) + available = bool(os.environ.get(API_ENV)) + if not available and os.environ.get("AGENT_SESSION_ROLE_ARN"): + raise RuntimeError( + "APPROVAL_REQUESTS_API_URL is required for cloud approval requests; " + "deploy matching CDK and worker images" + ) + return available def record_request( diff --git a/agent/src/task_state.py b/agent/src/task_state.py index 09fcda2ff..caaea02d4 100644 --- a/agent/src/task_state.py +++ b/agent/src/task_state.py @@ -740,8 +740,8 @@ def transact_write_approval_request( f"approval write cancelled: reasons={reasons}", cancellation_reasons=reasons ) from exc raise - # Compatibility with older deployments. New stacks grant no direct writes, - # so a missing service URL fails closed there rather than bypassing the broker. + # Compatibility with unscoped legacy/local deployments. configured() rejects + # a missing service URL for cloud workers that use the session role. task_table, approvals_table = _require_tables() ddb = _get_ddb_client(client=client) diff --git a/agent/tests/test_approval_requests.py b/agent/tests/test_approval_requests.py index 84c332a4b..a09b53295 100644 --- a/agent/tests/test_approval_requests.py +++ b/agent/tests/test_approval_requests.py @@ -81,6 +81,20 @@ def test_new_writers_use_service_instead_of_direct_dynamodb(transport, monkeypat direct.assert_not_called() +@pytest.mark.parametrize("operation", ["create", "timeout"]) +def test_cloud_worker_missing_endpoint_reports_configuration_error(monkeypatch, operation): + monkeypatch.delenv(broker.API_ENV, raising=False) + monkeypatch.setenv("AGENT_SESSION_ROLE_ARN", "arn:aws:iam::123456789012:role/session") + direct = MagicMock() + monkeypatch.setattr(task_state, "_get_ddb_client", direct) + with pytest.raises(RuntimeError, match="APPROVAL_REQUESTS_API_URL.*matching CDK"): + if operation == "create": + task_state.transact_write_approval_request("task", "request", pending_request()) + else: + task_state.best_effort_update_approval_status("task", "request", "TIMED_OUT") + direct.assert_not_called() + + @pytest.mark.parametrize("status", ["APPROVED", "DENIED", "PENDING", "CANCELLED"]) def test_worker_api_cannot_record_a_human_decision(transport, status): with pytest.raises(ValueError, match="non-human timeouts"): diff --git a/cdk/src/constructs/approval-request-service.ts b/cdk/src/constructs/approval-request-service.ts index f53521d90..40784c302 100644 --- a/cdk/src/constructs/approval-request-service.ts +++ b/cdk/src/constructs/approval-request-service.ts @@ -86,7 +86,7 @@ export class ApprovalRequestService extends NestedStack { id: 'AwsSolutions-IAM4', reason: 'AWSLambdaBasicExecutionRole provides Lambda runtime logging.', }], true); NagSuppressions.addResourceSuppressions(this.api, [ - { id: 'AwsSolutions-APIG2', reason: 'The handler validates the operation and an explicit approval-field allowlist before every write.' }, + { id: 'AwsSolutions-APIG2', reason: 'The handler validates request shape and allowed fields. Action descriptions and policy metadata remain worker assertions, not independently verified policy results.' }, { id: 'AwsSolutions-APIG3', reason: 'Machine-only IAM-signed API; session policy restricts POST to its tagged task path, with stage throttling.' }, { id: 'AwsSolutions-COG4', reason: 'Workers authenticate with task-scoped AWS credentials, not human Cognito credentials.' }, ], true); diff --git a/cdk/src/constructs/task-orchestrator.ts b/cdk/src/constructs/task-orchestrator.ts index ad75bd349..94795b79e 100644 --- a/cdk/src/constructs/task-orchestrator.ts +++ b/cdk/src/constructs/task-orchestrator.ts @@ -398,6 +398,9 @@ export class TaskOrchestrator extends Construct { constructor(scope: Construct, id: string, props: TaskOrchestratorProps) { super(scope, id); + if (props.agentPlatformConfig && !props.agentPlatformConfig.approvalRequestsApiUrl) { + throw new Error('agentPlatformConfig requires approvalRequestsApiUrl; deploy the matching approval service'); + } if (props.guardrailId && !props.guardrailVersion) { throw new Error('guardrailVersion is required when guardrailId is provided'); } diff --git a/cdk/src/handlers/request-approval.ts b/cdk/src/handlers/request-approval.ts index 31a534d5a..3d58a4950 100644 --- a/cdk/src/handlers/request-approval.ts +++ b/cdk/src/handlers/request-approval.ts @@ -53,6 +53,11 @@ function validId(value: unknown): value is string { return validShortText(value) && /^[A-Za-z0-9][A-Za-z0-9_-]*$/.test(value); } +/** Service-owned identity, compared as a DynamoDB value; never used as an IAM path. */ +function validWorkerId(value: unknown): value is string { + return validShortText(value) && !/[\u0000-\u001f\u007f]/.test(value); +} + /** Never copy decision, notification, retention or arbitrary caller fields. */ function validateRequest(input: RequestInput, task: Record): Record { const row = input.approval; @@ -83,7 +88,7 @@ function validateRequest(input: RequestInput, task: Record): Record */ export async function recordWorkerRequest(input: RequestInput): Promise> { if (!input || !validId(input.task_id) || !validId(input.request_id) - || (input.worker_attempt_id !== undefined && !validId(input.worker_attempt_id)) + || (input.worker_attempt_id !== undefined && !validWorkerId(input.worker_attempt_id)) || !['create', 'timeout'].includes(input.operation)) { return { ok: false, code: 'APPROVAL_REQUEST_INVALID' }; } diff --git a/cdk/test/constructs/task-orchestrator.test.ts b/cdk/test/constructs/task-orchestrator.test.ts index 472d8ec6e..b59d83d7e 100644 --- a/cdk/test/constructs/task-orchestrator.test.ts +++ b/cdk/test/constructs/task-orchestrator.test.ts @@ -36,6 +36,7 @@ interface StackOverrides { /** ADR-021 P2: the identifiers the orchestrator forwards as `platform_config`. */ agentPlatformConfig?: { taskApprovalsTableName: string; + approvalRequestsApiUrl?: string; nudgesTableName: string; logGroupName: string; artifactsBucketName: string; @@ -891,7 +892,10 @@ describe('TaskOrchestrator agentPlatformConfig (ADR-021 P2 platform_config trans * NAME, grants nothing" property can be asserted against actual logical IDs * rather than string literals. */ - function createPlatformConfigStack(withConfig: boolean): { template: Template } { + function createPlatformConfigStack( + withConfig: boolean, + approvalRequestsApiUrl = 'https://approval.execute-api.us-east-1.amazonaws.com/v1', + ): { template: Template } { const app = new App(); const stack = new Stack(app, 'TestStack', { env: { account: '123456789012', region: 'us-east-1' }, @@ -913,6 +917,7 @@ describe('TaskOrchestrator agentPlatformConfig (ADR-021 P2 platform_config trans ...(withConfig && { agentPlatformConfig: { taskApprovalsTableName: approvalsTable.tableName, + approvalRequestsApiUrl, nudgesTableName: nudgesTable.tableName, logGroupName: '/aws/abca/application', artifactsBucketName: traceBucket.bucketName, @@ -941,7 +946,11 @@ describe('TaskOrchestrator agentPlatformConfig (ADR-021 P2 platform_config trans withoutConfigTemplate = createPlatformConfigStack(false).template; }); - test('injects the eight forwarded identifiers under the names the strategy reads', () => { + test('rejects platform approval configuration without the service URL', () => { + expect(() => createPlatformConfigStack(true, '')).toThrow('requires approvalRequestsApiUrl'); + }); + + test('injects forwarded identifiers under the names the strategy reads', () => { // These names are a CONTRACT with // `handlers/shared/strategies/lambda-microvm-strategy.ts`'s // PLATFORM_CONFIG_ENV_VARS map, and with the AgentCore runtime env block in @@ -949,6 +958,7 @@ describe('TaskOrchestrator agentPlatformConfig (ADR-021 P2 platform_config trans // side silently strips a key from every MicroVM task's platform_config. const env = orchestratorEnvVars(template); expect(env.TASK_APPROVALS_TABLE_NAME).toEqual({ Ref: expect.stringMatching(/^TaskApprovalsTable/) }); + expect(env.APPROVAL_REQUESTS_API_URL).toBe('https://approval.execute-api.us-east-1.amazonaws.com/v1'); expect(env.NUDGES_TABLE_NAME).toEqual({ Ref: expect.stringMatching(/^TaskNudgesTable/) }); expect(env.LOG_GROUP_NAME).toBe('/aws/abca/application'); expect(env.ARTIFACTS_BUCKET_NAME).toEqual({ Ref: expect.stringMatching(/^TraceArtifactsBucket/) }); diff --git a/cdk/test/handlers/request-approval.test.ts b/cdk/test/handlers/request-approval.test.ts index 80d682958..19f746952 100644 --- a/cdk/test/handlers/request-approval.test.ts +++ b/cdk/test/handlers/request-approval.test.ts @@ -65,14 +65,14 @@ test('records only the pending request, guarded by current task ownership and st expect(items[1].Update.ExpressionAttributeValues[':user']).toBe('owner'); }); -test('fences stale MicroVM writers in the same transaction', async () => { +test.each(['worker-token', 'microvm.worker:1'])('fences stale MicroVM writers using opaque identity %s', async workerId => { task.compute_type = 'lambda-microvm'; - await recordWorkerRequest({ ...input, worker_attempt_id: 'worker-token' }); + await recordWorkerRequest({ ...input, worker_attempt_id: workerId }); const lease = send.mock.calls[1][0].input.TransactItems[2].ConditionCheck; expect(lease.Key).toEqual({ task_id: 'worker-lease#task' }); expect(lease.ConditionExpression).toBe('lease_state = :active AND lease_attempt_id = :attempt AND lease_user_id = :user'); expect(lease.ExpressionAttributeValues).toEqual({ - ':active': 'ACTIVE', ':attempt': 'worker-token', ':user': 'owner', + ':active': 'ACTIVE', ':attempt': workerId, ':user': 'owner', }); }); @@ -131,9 +131,15 @@ test('reports service failures as unavailable with a request ID, not invalid inp }); }); -test.each(['*', 'task/*', '../task', 'task?x', 'task#lease', 'task\n'])('rejects unsafe task/request/worker identifiers %j before database access', async id => { - for (const key of ['task_id', 'request_id', 'worker_attempt_id']) { +test.each(['*', 'task/*', '../task', 'task?x', 'task#lease', 'task\n'])('rejects unsafe task/request identifiers %j before database access', async id => { + for (const key of ['task_id', 'request_id']) { expect(await recordWorkerRequest({ ...input, [key]: id })).toEqual({ ok: false, code: 'APPROVAL_REQUEST_INVALID' }); } expect(send).not.toHaveBeenCalled(); }); + +test.each(['', 'worker\n', '\u0000worker', 'x'.repeat(129)])('rejects malformed worker identity %j', async workerId => { + expect(await recordWorkerRequest({ ...input, worker_attempt_id: workerId })) + .toEqual({ ok: false, code: 'APPROVAL_REQUEST_INVALID' }); + expect(send).not.toHaveBeenCalled(); +}); From 5d99bf0fc4ad60f26bd9cf64006cca52a1e38db0 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Mon, 21 Sep 2026 15:39:28 -0400 Subject: [PATCH 122/149] fix(645): fence continuation cursor updates during overlapping sweeps --- .../reconcile-microvm-continuations.ts | 27 +++++++++++++------ .../reconcile-microvm-continuations.test.ts | 20 ++++++++++++++ 2 files changed, 39 insertions(+), 8 deletions(-) diff --git a/cdk/src/handlers/reconcile-microvm-continuations.ts b/cdk/src/handlers/reconcile-microvm-continuations.ts index 7e566edde..dd67835e8 100644 --- a/cdk/src/handlers/reconcile-microvm-continuations.ts +++ b/cdk/src/handlers/reconcile-microvm-continuations.ts @@ -139,15 +139,26 @@ export async function handler(_event: unknown, context: Pick | undefined = saved.Item?.cursor; + const initialCursor: Record | undefined = saved.Item?.cursor; + let lastKey = initialCursor; const saveCursor = async (cursor?: Record) => { - await ddb.send(new UpdateCommand({ - TableName: TABLE, - Key: CURSOR_KEY, - UpdateExpression: cursor ? 'SET #cursor = :cursor' : 'REMOVE #cursor', - ExpressionAttributeNames: { '#cursor': 'cursor' }, - ...(cursor && { ExpressionAttributeValues: { ':cursor': cursor } }), - }), { abortSignal: AbortSignal.timeout(CURSOR_WRITE_TIMEOUT_MS) }); + const values = { + ...(cursor && { ':cursor': cursor }), + ...(initialCursor && { ':previous': initialCursor }), + }; + try { + await ddb.send(new UpdateCommand({ + TableName: TABLE, + Key: CURSOR_KEY, + UpdateExpression: cursor ? 'SET #cursor = :cursor' : 'REMOVE #cursor', + ConditionExpression: initialCursor ? '#cursor = :previous' : 'attribute_not_exists(#cursor)', + ExpressionAttributeNames: { '#cursor': 'cursor' }, + ...(Object.keys(values).length > 0 && { ExpressionAttributeValues: values }), + }), { abortSignal: AbortSignal.timeout(CURSOR_WRITE_TIMEOUT_MS) }); + } catch (error) { + if (!(error instanceof Error) || error.name !== 'ConditionalCheckFailedException') throw error; + logger.info('Another continuation sweep advanced the cursor; preserving its progress'); + } }; let processed = 0; let failures = 0; diff --git a/cdk/test/handlers/reconcile-microvm-continuations.test.ts b/cdk/test/handlers/reconcile-microvm-continuations.test.ts index 3d6e078d0..15f71a547 100644 --- a/cdk/test/handlers/reconcile-microvm-continuations.test.ts +++ b/cdk/test/handlers/reconcile-microvm-continuations.test.ts @@ -184,6 +184,10 @@ test('persists the last completed batch when invocation time is low, then resume expect(mockSend.mock.calls.find(([command]) => command.constructor.name === 'ScanCommand')![0].input.ExclusiveStartKey) .toEqual({ task_id: 'previous' }); expect(mockSend.mock.calls.at(-1)![0].input.ExpressionAttributeValues[':cursor']).toEqual({ task_id: 'task-3' }); + expect(mockSend.mock.calls.at(-1)![0].input).toMatchObject({ + ConditionExpression: '#cursor = :previous', + ExpressionAttributeValues: { ':previous': { task_id: 'previous' } }, + }); }); test('a failed row does not prevent later rows or clearing the cursor after a complete scan', async () => { @@ -193,4 +197,20 @@ test('a failed row does not prevent later rows or clearing the cursor after a co await handler({}, { getRemainingTimeInMillis: () => 100000 }); expect(mockDispatch).toHaveBeenCalledTimes(2); expect(mockSend.mock.calls.at(-1)![0].input.UpdateExpression).toBe('REMOVE #cursor'); + expect(mockSend.mock.calls.at(-1)![0].input.ConditionExpression).toBe('attribute_not_exists(#cursor)'); }); + +test.each(['ConditionalCheckFailedException', 'ProvisionedThroughputExceededException'])( + 'only an overlapping sweep may supersede the saved cursor: %s', + async name => { + const error = Object.assign(new Error('cursor write rejected'), { name }); + mockSend.mockImplementation(async command => { + if (command.constructor.name === 'UpdateCommand') throw error; + return {}; + }); + const run = handler({}, { getRemainingTimeInMillis: () => 100000 }); + if (name === 'ConditionalCheckFailedException') await expect(run).resolves.toBeUndefined(); + else await expect(run).rejects.toBe(error); + expect(mockSend.mock.calls.filter(([command]) => command.constructor.name === 'UpdateCommand')).toHaveLength(1); + }, +); From c3ae2f02bf982c82449983f65b4c39c115b1088b Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Mon, 21 Sep 2026 15:41:25 -0400 Subject: [PATCH 123/149] test(645): pin recovery health and conservative Linear replay handling --- .github/workflows/build.yml | 1 + agent/tests/test_server.py | 14 +++++--------- cdk/test/handlers/shared/linear-feedback.test.ts | 2 ++ 3 files changed, 8 insertions(+), 9 deletions(-) diff --git a/.github/workflows/build.yml b/.github/workflows/build.yml index d2a4a3c6c..303061520 100644 --- a/.github/workflows/build.yml +++ b/.github/workflows/build.yml @@ -60,6 +60,7 @@ jobs: - 8000/tcp env: CI: "true" + ABCA_TEST_SDK_CONTINUATION: "1" MISE_EXPERIMENTAL: "1" GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }} GITHUB_API_TOKEN: ${{ secrets.GITHUB_TOKEN }} diff --git a/agent/tests/test_server.py b/agent/tests/test_server.py index 182eda420..460a186ec 100644 --- a/agent/tests/test_server.py +++ b/agent/tests/test_server.py @@ -73,11 +73,12 @@ def test_ping_healthy_by_default(client): assert r.json() == {"status": "healthy"} -def test_background_thread_failure_503_and_backup_terminal_write(client, monkeypatch): +@pytest.mark.parametrize("outcome", ["written", "failed", "superseded"]) +def test_background_thread_failure_503_and_backup_terminal_write(client, monkeypatch, outcome): def boom(**_kwargs): raise RuntimeError("simulated pipeline crash") - mock_write = MagicMock() + mock_write = MagicMock(return_value=server.task_state.TerminalWriteOutcome(outcome)) monkeypatch.setattr(server, "run_task", boom) monkeypatch.setattr(server.task_state, "write_terminal", mock_write) @@ -115,13 +116,8 @@ def boom(**_kwargs): assert body["status"] == "unhealthy" assert body["reason"] == "background_pipeline_failed" - # Race: /ping flips to 503 as soon as ``_background_pipeline_failed = True`` - # is set in the except block, but ``task_state.write_terminal(...)`` happens - # a few lines later (after ``print()`` + ``traceback.print_exc()``). Wait - # for the mock to actually be invoked before asserting. - deadline2 = time.time() + 5.0 - while time.time() < deadline2 and not mock_write.called: - time.sleep(0.05) + # The thread has exited, including the backup write. Even a failed backup + # must leave /ping unhealthy so the coordinator can recover the task. mock_write.assert_called() call_kw = mock_write.call_args assert call_kw[0][0] == "task-crash-1" diff --git a/cdk/test/handlers/shared/linear-feedback.test.ts b/cdk/test/handlers/shared/linear-feedback.test.ts index d831c110e..526202821 100644 --- a/cdk/test/handlers/shared/linear-feedback.test.ts +++ b/cdk/test/handlers/shared/linear-feedback.test.ts @@ -149,6 +149,8 @@ describe('linear-feedback', () => { test.each([ ['Approval needed\r\nReply here\r\n', true], ['Approval needed\nApprove a DIFFERENT action', false], + ['Approval needed\nReply here', false], + ['approval needed\nReply here', false], ])('compares replay content conservatively: %j', async (body, accepted) => { fetchMock.mockRejectedValueOnce(new Error('lost response')); fetchMock.mockResolvedValueOnce(jsonResponse({ From 9f3d564f9ad6e13b5a1404903bbf86ca006d14f2 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Mon, 21 Sep 2026 15:42:25 -0400 Subject: [PATCH 124/149] docs(645): explain upgrade overlap and checkpoint diagnostics --- agent/src/microvm_lifecycle.py | 2 ++ cdk/AGENTS.md | 1 + docs/design/CEDAR_HITL_GATES.md | 2 +- docs/guides/DEPLOYMENT_GUIDE.md | 17 +++++++++++++++-- docs/scripts/check-site-links.mjs | 10 ++++++++++ .../docs/architecture/Cedar-hitl-gates.md | 2 +- .../docs/getting-started/Deployment-guide.md | 17 +++++++++++++++-- .../645-p3-lifecycle-diagnostics.md | 18 ++++++++++++++++++ .../645-p3-lifecycle-diagnostics.md | 18 ++++++++++++++++++ 9 files changed, 81 insertions(+), 6 deletions(-) diff --git a/agent/src/microvm_lifecycle.py b/agent/src/microvm_lifecycle.py index 8f1f393ff..a519d33f3 100644 --- a/agent/src/microvm_lifecycle.py +++ b/agent/src/microvm_lifecycle.py @@ -169,6 +169,8 @@ def park_approval( # A new gate cannot acknowledge a wake using the previous gate's # cached result. Its own suspend must establish a fresh safe point. self._last_resume_park = None + # Local "parked" keeps this worker alive at a safe tool boundary. + # The coordinator's persisted PARKED continuation means source retirement. self._phase = "parked" return park diff --git a/cdk/AGENTS.md b/cdk/AGENTS.md index 44ddf5899..a9d4cde9a 100644 --- a/cdk/AGENTS.md +++ b/cdk/AGENTS.md @@ -44,6 +44,7 @@ Construct tests: synthesize each distinct stack config once in `beforeAll`, asse | `cdk/src/stacks/` | WRITE | Stack definitions | | `cdk/src/constructs/` | WRITE | Reusable constructs | | `cdk/src/handlers/shared/types.ts` | WRITE | API types (mirror to `cli/src/types.ts`) | +| `cdk/src/handlers/shared/canonical-json.ts` | WRITE | Runtime receipt encoding; preserve stored bytes and keep migration encoding separate | | `cdk/test/` | WRITE | Unit / snapshot tests | | `cli/src/types.ts` | WRITE (sync) | Must match shared types | diff --git a/docs/design/CEDAR_HITL_GATES.md b/docs/design/CEDAR_HITL_GATES.md index 007c6cb48..4a049f40e 100644 --- a/docs/design/CEDAR_HITL_GATES.md +++ b/docs/design/CEDAR_HITL_GATES.md @@ -1883,7 +1883,7 @@ The per-task `approvalGateCap` (decision #13; default 50, configurable) is **per ### 13.7 Insufficient lifetime remaining for approval -If `remaining_maxLifetime - CLEANUP_MARGIN_120S < FLOOR_30S`, hook immediately returns DENY with reason `"insufficient maxLifetime for approval"`. Task continues without a gate — or, if the gate was load-bearing, fails gracefully in RUNNING state. +Without a continuation runtime, if `remaining_maxLifetime - CLEANUP_MARGIN_120S < FLOOR_30S`, the hook immediately denies the tool with reason `"insufficient maxLifetime remaining ({n}s) for approval"`, where `{n}` is the remaining lifetime in seconds. No approval request is created and the tool does not run. The agent receives the denial and may choose another action. ### 13.8 PreToolUse hook itself crashes diff --git a/docs/guides/DEPLOYMENT_GUIDE.md b/docs/guides/DEPLOYMENT_GUIDE.md index 31c9ad1a8..10069fd95 100644 --- a/docs/guides/DEPLOYMENT_GUIDE.md +++ b/docs/guides/DEPLOYMENT_GUIDE.md @@ -262,6 +262,10 @@ directly to DynamoDB and cannot create new gates after those permissions are rem Custom ECS constructs must provide both the SessionRole and service URL; approval wiring without them is rejected before deployment. A MicroVM manifest missing the URL is rejected before the worker starts. + If an AgentCore environment was edited to remove the URL, the worker reports + `APPROVAL_REQUESTS_API_URL is required for cloud approval requests` at its + next gate rather than attempting a direct DynamoDB write. Redeploy the + matching stack and image. 4. Submit a test task that triggers a known approval rule on each enabled backend. Verify that the request appears, an owner decision resumes it, and an explicit deadline records `TIMED_OUT` without overwriting a human decision. Then resume @@ -278,8 +282,17 @@ Concurrency repair, admission-queue pickup, stranded-task repair and pending-upl cleanup run in the `ConcurrencyMaintenance` nested stack. MicroVM continuation recovery also runs there when an image is configured. An upgrade recreates the stateless functions, roles and schedules; their task tables and storage stay in -the parent stack. Review those replacements in the change set after draining tasks -as described above. Both flat and nested MicroVM layouts support the optional +the parent stack. This applies to every compute backend, including AgentCore, +and replaces roughly twenty resources, depending on enabled features. +CloudFormation creates the new schedules before deleting the old ones, so both +can fire during the update. Task mutations and the continuation scan cursor use +conditional writes to tolerate that overlap. The drain above is required for +the approval permission/image upgrade, not a scheduling gap. + +Review the replacements in the change set. Existing Lambda log groups remain +under their old generated names; use the new function's log group for post-upgrade +invocations and retain the old groups when investigating earlier runs. +Both flat and nested MicroVM layouts support the optional tool gateway and Linear Identity vault without exceeding the template budget. This does not migrate existing MicroVM compute resources. Keep an existing flat diff --git a/docs/scripts/check-site-links.mjs b/docs/scripts/check-site-links.mjs index 2665d57ca..6d1ea7b11 100644 --- a/docs/scripts/check-site-links.mjs +++ b/docs/scripts/check-site-links.mjs @@ -2,6 +2,8 @@ import fs from 'node:fs'; import path from 'node:path'; const root = path.resolve(import.meta.dirname, '..', 'dist'); +const repoRoot = path.resolve(import.meta.dirname, '..', '..'); +const repoBlobPrefix = 'https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/'; const base = '/sample-autonomous-cloud-coding-agents/'; const pages = fs.readdirSync(root, { recursive: true }).filter(file => file.endsWith('.html')); if (pages.length < 10) throw new Error('Expected a built documentation site before checking links'); @@ -17,6 +19,14 @@ for (const file of pages) { const origin = `https://local${base}${file.replace(/index\.html$/, '')}`; for (const match of read(path.join(root, file)).matchAll(/]*\bhref="([^"]*)"/g)) { const href = match[1].replaceAll('&', '&'); + if (href.startsWith(repoBlobPrefix)) { + const relative = decodeURIComponent(href.slice(repoBlobPrefix.length).split(/[?#]/)[0]); + const target = path.resolve(repoRoot, relative); + if (!target.startsWith(`${repoRoot}${path.sep}`) || !fs.existsSync(target) || !fs.statSync(target).isFile()) { + errors.add(`${file}: missing repository file: ${href}`); + } + continue; + } // This check is offline. External availability is outside its scope. if (/^(?:[a-z]+:|\/\/)/i.test(href)) continue; const url = new URL(href, origin); diff --git a/docs/src/content/docs/architecture/Cedar-hitl-gates.md b/docs/src/content/docs/architecture/Cedar-hitl-gates.md index 793a80364..2b4755cbb 100644 --- a/docs/src/content/docs/architecture/Cedar-hitl-gates.md +++ b/docs/src/content/docs/architecture/Cedar-hitl-gates.md @@ -1887,7 +1887,7 @@ The per-task `approvalGateCap` (decision #13; default 50, configurable) is **per ### 13.7 Insufficient lifetime remaining for approval -If `remaining_maxLifetime - CLEANUP_MARGIN_120S < FLOOR_30S`, hook immediately returns DENY with reason `"insufficient maxLifetime for approval"`. Task continues without a gate — or, if the gate was load-bearing, fails gracefully in RUNNING state. +Without a continuation runtime, if `remaining_maxLifetime - CLEANUP_MARGIN_120S < FLOOR_30S`, the hook immediately denies the tool with reason `"insufficient maxLifetime remaining ({n}s) for approval"`, where `{n}` is the remaining lifetime in seconds. No approval request is created and the tool does not run. The agent receives the denial and may choose another action. ### 13.8 PreToolUse hook itself crashes diff --git a/docs/src/content/docs/getting-started/Deployment-guide.md b/docs/src/content/docs/getting-started/Deployment-guide.md index 2bcd3b1f9..223202704 100644 --- a/docs/src/content/docs/getting-started/Deployment-guide.md +++ b/docs/src/content/docs/getting-started/Deployment-guide.md @@ -266,6 +266,10 @@ directly to DynamoDB and cannot create new gates after those permissions are rem Custom ECS constructs must provide both the SessionRole and service URL; approval wiring without them is rejected before deployment. A MicroVM manifest missing the URL is rejected before the worker starts. + If an AgentCore environment was edited to remove the URL, the worker reports + `APPROVAL_REQUESTS_API_URL is required for cloud approval requests` at its + next gate rather than attempting a direct DynamoDB write. Redeploy the + matching stack and image. 4. Submit a test task that triggers a known approval rule on each enabled backend. Verify that the request appears, an owner decision resumes it, and an explicit deadline records `TIMED_OUT` without overwriting a human decision. Then resume @@ -282,8 +286,17 @@ Concurrency repair, admission-queue pickup, stranded-task repair and pending-upl cleanup run in the `ConcurrencyMaintenance` nested stack. MicroVM continuation recovery also runs there when an image is configured. An upgrade recreates the stateless functions, roles and schedules; their task tables and storage stay in -the parent stack. Review those replacements in the change set after draining tasks -as described above. Both flat and nested MicroVM layouts support the optional +the parent stack. This applies to every compute backend, including AgentCore, +and replaces roughly twenty resources, depending on enabled features. +CloudFormation creates the new schedules before deleting the old ones, so both +can fire during the update. Task mutations and the continuation scan cursor use +conditional writes to tolerate that overlap. The drain above is required for +the approval permission/image upgrade, not a scheduling gap. + +Review the replacements in the change set. Existing Lambda log groups remain +under their old generated names; use the new function's log group for post-upgrade +invocations and retain the old groups when investigating earlier runs. +Both flat and nested MicroVM layouts support the optional tool gateway and Linear Identity vault without exceeding the template budget. This does not migrate existing MicroVM compute resources. Keep an existing flat diff --git a/docs/src/content/docs/verification/645-p3-lifecycle-diagnostics.md b/docs/src/content/docs/verification/645-p3-lifecycle-diagnostics.md index 6f5d5743b..bda5d7655 100644 --- a/docs/src/content/docs/verification/645-p3-lifecycle-diagnostics.md +++ b/docs/src/content/docs/verification/645-p3-lifecycle-diagnostics.md @@ -51,6 +51,24 @@ identity reconciliation. `late: true` means a callback finished after the handle had already returned; it cannot turn a timed-out wake into success. The coding barrier remains responsible for preventing tools after an uncertain wake. +## Checkpoint failure codes + +The checkpoint diagnostic `code` narrows the failing operation; it does not +prove that saved data is corrupt. + +| Code | Meaning | +|---|---| +| `checkpoint_failed` | No narrower classification; inspect the accompanying stage and message. | +| `checkpoint_invalid_json` | Checkpoint JSON could not be encoded or decoded. | +| `checkpoint_sdk_unverified` | SDK version or required accounting interface is not verified. | +| `checkpoint_sdk_timeout` | SDK transcript acknowledgment or accounting request timed out. | +| `checkpoint_storage_unverified` | A save could not be verified by reading back the exact data; keep the source worker. | +| `checkpoint_storage_unavailable` | Storage configuration is unavailable or the saved version could not be read. | + +Workspace capture and restore can also report specific codes such as +`disk_pressure` or `git_timeout`. Preserve the reported code and stage when +escalating; do not replace them with a generic checkpoint label. + ## Diagnose a failed wake 1. Locate the saved decision/intent and actual Resume acknowledgment or failure. diff --git a/docs/verification/645-p3-lifecycle-diagnostics.md b/docs/verification/645-p3-lifecycle-diagnostics.md index eb7ceb4b4..9288934c7 100644 --- a/docs/verification/645-p3-lifecycle-diagnostics.md +++ b/docs/verification/645-p3-lifecycle-diagnostics.md @@ -47,6 +47,24 @@ identity reconciliation. `late: true` means a callback finished after the handle had already returned; it cannot turn a timed-out wake into success. The coding barrier remains responsible for preventing tools after an uncertain wake. +## Checkpoint failure codes + +The checkpoint diagnostic `code` narrows the failing operation; it does not +prove that saved data is corrupt. + +| Code | Meaning | +|---|---| +| `checkpoint_failed` | No narrower classification; inspect the accompanying stage and message. | +| `checkpoint_invalid_json` | Checkpoint JSON could not be encoded or decoded. | +| `checkpoint_sdk_unverified` | SDK version or required accounting interface is not verified. | +| `checkpoint_sdk_timeout` | SDK transcript acknowledgment or accounting request timed out. | +| `checkpoint_storage_unverified` | A save could not be verified by reading back the exact data; keep the source worker. | +| `checkpoint_storage_unavailable` | Storage configuration is unavailable or the saved version could not be read. | + +Workspace capture and restore can also report specific codes such as +`disk_pressure` or `git_timeout`. Preserve the reported code and stage when +escalating; do not replace them with a generic checkpoint label. + ## Diagnose a failed wake 1. Locate the saved decision/intent and actual Resume acknowledgment or failure. From 39ff08998f72ef8bcd7581656016431721e9da57 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Mon, 21 Sep 2026 15:53:14 -0400 Subject: [PATCH 125/149] test(645): keep pipeline regression fixtures type-safe --- agent/tests/test_approval_requests.py | 2 +- agent/tests/test_pipeline.py | 30 +++++++++++++++------------ 2 files changed, 18 insertions(+), 14 deletions(-) diff --git a/agent/tests/test_approval_requests.py b/agent/tests/test_approval_requests.py index a09b53295..373853db9 100644 --- a/agent/tests/test_approval_requests.py +++ b/agent/tests/test_approval_requests.py @@ -87,7 +87,7 @@ def test_cloud_worker_missing_endpoint_reports_configuration_error(monkeypatch, monkeypatch.setenv("AGENT_SESSION_ROLE_ARN", "arn:aws:iam::123456789012:role/session") direct = MagicMock() monkeypatch.setattr(task_state, "_get_ddb_client", direct) - with pytest.raises(RuntimeError, match="APPROVAL_REQUESTS_API_URL.*matching CDK"): + with pytest.raises(RuntimeError, match=r"APPROVAL_REQUESTS_API_URL.*matching CDK"): if operation == "create": task_state.transact_write_approval_request("task", "request", pending_request()) else: diff --git a/agent/tests/test_pipeline.py b/agent/tests/test_pipeline.py index 98d7ab957..44f94f597 100644 --- a/agent/tests/test_pipeline.py +++ b/agent/tests/test_pipeline.py @@ -2413,24 +2413,28 @@ async def agent_result(*args, **kwargs): patch.object(task_state, "write_terminal", terminal), ): stack.enter_context(context) - kwargs = dict( - repo_url=repo_url, - task_description="Read the README", - github_token="ghp_test", - aws_region="us-east-1", - task_id="terminal-race", - channel_source="linear", - channel_metadata=_LINEAR_META, - ) - if not repo_url: - kwargs["resolved_workflow"] = {"id": "default/agent-v1", "version": "1.0.0"} + + def execute(): + return run_task( + repo_url=repo_url, + task_description="Read the README", + github_token="ghp_test", + aws_region="us-east-1", + task_id="terminal-race", + channel_source="linear", + channel_metadata=_LINEAR_META, + resolved_workflow=None + if repo_url + else {"id": "default/agent-v1", "version": "1.0.0"}, + ) + if outcome == "failed": with pytest.raises( task_state.TerminalWriteError, match="Task result was not committed" ): - run_task(**kwargs) + execute() else: - result = run_task(**kwargs) + result = execute() assert result["status"] == "success" terminal.assert_called_once() assert not any(call.kwargs.get("success") is False for call in finished.call_args_list) From 159576e642dbef0f45d8d113c9bd1eedb70afa89 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Mon, 21 Sep 2026 16:11:15 -0400 Subject: [PATCH 126/149] fix(645): retain validation for platform worker lease tokens --- cdk/src/handlers/request-approval.ts | 7 +------ cdk/test/handlers/request-approval.test.ts | 16 +++++----------- 2 files changed, 6 insertions(+), 17 deletions(-) diff --git a/cdk/src/handlers/request-approval.ts b/cdk/src/handlers/request-approval.ts index 3d58a4950..31a534d5a 100644 --- a/cdk/src/handlers/request-approval.ts +++ b/cdk/src/handlers/request-approval.ts @@ -53,11 +53,6 @@ function validId(value: unknown): value is string { return validShortText(value) && /^[A-Za-z0-9][A-Za-z0-9_-]*$/.test(value); } -/** Service-owned identity, compared as a DynamoDB value; never used as an IAM path. */ -function validWorkerId(value: unknown): value is string { - return validShortText(value) && !/[\u0000-\u001f\u007f]/.test(value); -} - /** Never copy decision, notification, retention or arbitrary caller fields. */ function validateRequest(input: RequestInput, task: Record): Record { const row = input.approval; @@ -88,7 +83,7 @@ function validateRequest(input: RequestInput, task: Record): Record */ export async function recordWorkerRequest(input: RequestInput): Promise> { if (!input || !validId(input.task_id) || !validId(input.request_id) - || (input.worker_attempt_id !== undefined && !validWorkerId(input.worker_attempt_id)) + || (input.worker_attempt_id !== undefined && !validId(input.worker_attempt_id)) || !['create', 'timeout'].includes(input.operation)) { return { ok: false, code: 'APPROVAL_REQUEST_INVALID' }; } diff --git a/cdk/test/handlers/request-approval.test.ts b/cdk/test/handlers/request-approval.test.ts index 19f746952..80d682958 100644 --- a/cdk/test/handlers/request-approval.test.ts +++ b/cdk/test/handlers/request-approval.test.ts @@ -65,14 +65,14 @@ test('records only the pending request, guarded by current task ownership and st expect(items[1].Update.ExpressionAttributeValues[':user']).toBe('owner'); }); -test.each(['worker-token', 'microvm.worker:1'])('fences stale MicroVM writers using opaque identity %s', async workerId => { +test('fences stale MicroVM writers in the same transaction', async () => { task.compute_type = 'lambda-microvm'; - await recordWorkerRequest({ ...input, worker_attempt_id: workerId }); + await recordWorkerRequest({ ...input, worker_attempt_id: 'worker-token' }); const lease = send.mock.calls[1][0].input.TransactItems[2].ConditionCheck; expect(lease.Key).toEqual({ task_id: 'worker-lease#task' }); expect(lease.ConditionExpression).toBe('lease_state = :active AND lease_attempt_id = :attempt AND lease_user_id = :user'); expect(lease.ExpressionAttributeValues).toEqual({ - ':active': 'ACTIVE', ':attempt': workerId, ':user': 'owner', + ':active': 'ACTIVE', ':attempt': 'worker-token', ':user': 'owner', }); }); @@ -131,15 +131,9 @@ test('reports service failures as unavailable with a request ID, not invalid inp }); }); -test.each(['*', 'task/*', '../task', 'task?x', 'task#lease', 'task\n'])('rejects unsafe task/request identifiers %j before database access', async id => { - for (const key of ['task_id', 'request_id']) { +test.each(['*', 'task/*', '../task', 'task?x', 'task#lease', 'task\n'])('rejects unsafe task/request/worker identifiers %j before database access', async id => { + for (const key of ['task_id', 'request_id', 'worker_attempt_id']) { expect(await recordWorkerRequest({ ...input, [key]: id })).toEqual({ ok: false, code: 'APPROVAL_REQUEST_INVALID' }); } expect(send).not.toHaveBeenCalled(); }); - -test.each(['', 'worker\n', '\u0000worker', 'x'.repeat(129)])('rejects malformed worker identity %j', async workerId => { - expect(await recordWorkerRequest({ ...input, worker_attempt_id: workerId })) - .toEqual({ ok: false, code: 'APPROVAL_REQUEST_INVALID' }); - expect(send).not.toHaveBeenCalled(); -}); From 7384f9b0b7d8ec507a42e6f995cc1c087750a01d Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Tue, 22 Sep 2026 09:02:12 -0400 Subject: [PATCH 127/149] fix: retry contended task capacity transactions (#645) --- cdk/src/handlers/shared/task-concurrency.ts | 49 +++++++++++-- .../handlers/shared/task-concurrency.test.ts | 70 ++++++++++++++++++- 2 files changed, 114 insertions(+), 5 deletions(-) diff --git a/cdk/src/handlers/shared/task-concurrency.ts b/cdk/src/handlers/shared/task-concurrency.ts index fcde85569..fdfcdedcb 100644 --- a/cdk/src/handlers/shared/task-concurrency.ts +++ b/cdk/src/handlers/shared/task-concurrency.ts @@ -18,6 +18,7 @@ */ import { randomUUID } from 'node:crypto'; +import { setTimeout as delay } from 'node:timers/promises'; import { GetCommand, TransactWriteCommand } from '@aws-sdk/lib-dynamodb'; import { logger } from './logger'; import { workerLeaseKey } from './microvm-continuation-types'; @@ -61,6 +62,46 @@ function conditionalFailure(error: unknown, index: number): boolean { && failure.CancellationReasons?.[index]?.Code === 'ConditionalCheckFailed'; } +/** + * The SDK does not retry TransactionCanceledException when another task is + * updating the same user's counter. Retry only explicit transaction conflicts; + * conditional failures still belong to the admission/release state machine. + */ +async function sendReservationTransaction( + command: TransactWriteCommand, + taskId: string, + userId: string, +): Promise { + const maxAttempts = 5; + const baseDelayMs = 50; + for (let attempt = 1; attempt <= maxAttempts; attempt++) { + try { + // Keep the exact transaction and client token across retries, including + // the worker-lease condition. A retry cannot weaken the shutdown fence. + await ddb.send(command); + return; + } catch (error) { + const failure = error as { name?: string; CancellationReasons?: { Code?: string }[] }; + const codes = failure?.CancellationReasons?.map(reason => reason.Code); + const conflict = failure?.name === 'TransactionCanceledException' + && codes?.includes('TransactionConflict') + && codes.every(code => code === 'None' || code === 'TransactionConflict'); + if (!conflict) throw error; + if (attempt === maxAttempts) { + logger.warn('Capacity transaction contention exhausted bounded retries', { + task_id: taskId, + user_id: userId, + attempts: attempt, + error_id: 'CONCURRENCY_TRANSACTION_CONFLICT', + cancellation_codes: codes, + }); + throw error; + } + await delay(Math.floor(Math.random() * baseDelayMs * 2 ** (attempt - 1))); + } + } +} + /** Reserve once per task, including after a lost transaction acknowledgement. */ export async function acquireTaskSlot(taskId: string, userId: string, limit: number): Promise { const current = await readTask(taskId, userId); @@ -73,7 +114,7 @@ export async function acquireTaskSlot(taskId: string, userId: string, limit: num const now = new Date().toISOString(); const revision = randomUUID(); try { - await ddb.send(new TransactWriteCommand({ + await sendReservationTransaction(new TransactWriteCommand({ ClientRequestToken: revision, TransactItems: [ { @@ -103,7 +144,7 @@ export async function acquireTaskSlot(taskId: string, userId: string, limit: num }, }, ], - })); + }), taskId, userId); return true; } catch (error) { // A competing invocation or a lost successful response may have reserved it. @@ -142,7 +183,7 @@ export async function releaseTaskSlot(taskId: string, userId: string): Promise ({ + setTimeout: jest.fn().mockResolvedValue(undefined), +})); jest.mock('@aws-sdk/lib-dynamodb', () => ({ GetCommand: jest.fn((input: unknown) => ({ kind: 'get', input })), TransactWriteCommand: jest.fn((input: unknown) => ({ kind: 'transaction', input })), @@ -28,6 +31,7 @@ jest.mock('../../../src/handlers/shared/ua', () => ({ process.env.TASK_TABLE_NAME = 'Tasks'; process.env.USER_CONCURRENCY_TABLE_NAME = 'Counters'; +import { setTimeout as delay } from 'node:timers/promises'; import { acquireTaskSlot, releaseTaskSlot } from '../../../src/handlers/shared/task-concurrency'; const base = { task_id: 'task', user_id: 'user', status: 'SUBMITTED' }; @@ -38,7 +42,71 @@ function cancelled(index: number) { CancellationReasons: [0, 1].map(i => ({ Code: i === index ? 'ConditionalCheckFailed' : 'None' })), }); } -beforeEach(() => mockSend.mockReset()); +beforeEach(() => { + mockSend.mockReset(); + jest.mocked(delay).mockClear(); +}); + +function conflict(codes = ['None', 'TransactionConflict']) { + return Object.assign(new Error('transaction conflict'), { + name: 'TransactionCanceledException', + CancellationReasons: codes.map(Code => ({ Code })), + }); +} + +test.each(['acquire', 'release'])('%s retries counter contention with the identical guarded transaction', async operation => { + mockSend.mockResolvedValueOnce({ + Item: operation === 'acquire' ? base : { ...base, status: 'COMPLETED', concurrency_slot: held }, + }).mockRejectedValueOnce(conflict()).mockRejectedValueOnce(conflict()).mockResolvedValueOnce({}); + const result = operation === 'acquire' + ? await acquireTaskSlot('task', 'user', 10) + : await releaseTaskSlot('task', 'user'); + expect(result).toBe(true); + const transactions = mockSend.mock.calls.filter(([command]) => command.kind === 'transaction'); + expect(transactions).toHaveLength(3); + expect(transactions[1][0]).toBe(transactions[0][0]); + expect(transactions[2][0]).toBe(transactions[0][0]); + expect(delay).toHaveBeenCalledTimes(2); +}); + +test('persistent contention stays bounded and propagates the original error', async () => { + const error = conflict(); + mockSend.mockImplementation(async command => { + if (command.kind === 'get') return { Item: { ...base, status: 'COMPLETED', concurrency_slot: held } }; + throw error; + }); + await expect(releaseTaskSlot('task', 'user')).rejects.toBe(error); + expect(mockSend.mock.calls.filter(([command]) => command.kind === 'transaction')).toHaveLength(5); + expect(delay).toHaveBeenCalledTimes(4); + for (const [index, call] of jest.mocked(delay).mock.calls.entries()) { + expect(call[0]).toBeGreaterThanOrEqual(0); + expect(call[0]).toBeLessThan(50 * 2 ** index); + } +}); + +test('a mixed conflict and capacity condition failure follows normal admission without retrying', async () => { + mockSend.mockResolvedValueOnce({ Item: base }) + .mockRejectedValueOnce(conflict(['TransactionConflict', 'ConditionalCheckFailed'])) + .mockResolvedValueOnce({ Item: base }); + expect(await acquireTaskSlot('task', 'user', 10)).toBe(false); + expect(delay).not.toHaveBeenCalled(); +}); + +test('a worker lease changing during contention cannot release its reservation', async () => { + const task = { ...base, status: 'FAILED', concurrency_slot: held, continuation_launch: {}, microvm_start: { clientToken: 'attempt' } }; + const fenced = conflict(['None', 'None', 'ConditionalCheckFailed']); + mockSend.mockResolvedValueOnce({ Item: task }) + .mockResolvedValueOnce({ Item: { lease_user_id: 'user', lease_attempt_id: 'attempt', lease_state: 'CLOSED' } }) + .mockRejectedValueOnce(conflict(['None', 'TransactionConflict', 'None'])) + .mockRejectedValueOnce(fenced) + .mockResolvedValueOnce({ Item: task }); + await expect(releaseTaskSlot('task', 'user')).rejects.toBe(fenced); + const transactions = mockSend.mock.calls.filter(([command]) => command.kind === 'transaction'); + expect(transactions).toHaveLength(2); + expect(transactions[1][0]).toBe(transactions[0][0]); + expect(transactions[1][0].input.TransactItems[2].ConditionCheck.ConditionExpression).toContain('lease_state = :closed'); + expect(delay).toHaveBeenCalledTimes(1); +}); test.each([undefined, 'ACTIVE', 'FENCED', 'PARKED'])( 'a terminal saved task keeps capacity until shutdown is confirmed (lease %s)', async leaseState => { From 68bdbfaf67731fa2159d4001b51228edcba05e05 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Tue, 22 Sep 2026 10:13:18 -0400 Subject: [PATCH 128/149] docs(645): clarify MicroVM lifetime and active cleanup --- .../shared/strategies/lambda-microvm-strategy.ts | 10 ++++------ 1 file changed, 4 insertions(+), 6 deletions(-) diff --git a/cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts b/cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts index cf07439a0..cea997852 100644 --- a/cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts +++ b/cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts @@ -826,12 +826,10 @@ export class LambdaMicrovmComputeStrategy implements ComputeStrategy { * MicroVMs may be leaking) from "something else" (worth a warning). * * ADR-021: termination is the active cleanup path — it must not rely on - * ``maximumDurationInSeconds`` expiring, which would keep paying for an - * 8-hour reservation after the task is done. Live verification made that - * mandatory rather than belt-and-braces: a hook-less MicroVM reached - * ``RUNNING`` in 12 s and stayed ``RUNNING`` indefinitely with no - * ``stateReason`` — nothing self-terminates, so nothing cleans up if the - * orchestrator does not. + * ``maximumDurationInSeconds`` expiring. The service enforces an eight-hour + * lifetime (including suspended time), but task completion does not itself + * terminate the VM. Explicit termination avoids paying for unused running time + * until that limit. */ async stopSession(handle: SessionHandle, options?: SessionControlOptions): Promise { if (handle.strategyType !== 'lambda-microvm') { From 35f05bbb7e55bf313ac96baec6b45bd36a3a3fe7 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Tue, 22 Sep 2026 10:16:43 -0400 Subject: [PATCH 129/149] feat(645): default MicroVM infrastructure to nested stacks --- cdk/scripts/README.md | 4 ++-- cdk/scripts/package-microvm-artifact.sh | 9 +++++---- cdk/src/stacks/agent.ts | 6 +++--- cdk/test/stacks/agent.test.ts | 14 +++++++------- .../ADR-021-lambda-microvms-compute-backend.md | 2 +- docs/design/COMPUTE.md | 4 ++++ docs/design/DEPLOYMENT_ROLES.md | 6 +++--- docs/guides/DEPLOYMENT_GUIDE.md | 9 ++++++--- docs/src/content/docs/architecture/Compute.md | 4 ++++ .../content/docs/architecture/Deployment-roles.md | 6 +++--- .../Adr-021-lambda-microvms-compute-backend.md | 2 +- .../docs/getting-started/Deployment-guide.md | 9 ++++++--- .../docs/verification/645-p3-nested-stack.md | 7 ++++++- docs/verification/645-p3-nested-stack.md | 7 ++++++- 14 files changed, 57 insertions(+), 32 deletions(-) diff --git a/cdk/scripts/README.md b/cdk/scripts/README.md index b8cc92bee..ac2b55f73 100644 --- a/cdk/scripts/README.md +++ b/cdk/scripts/README.md @@ -11,6 +11,6 @@ Bundling for Lambda assets is handled at synth time; the **`bundle`** task in ** `package-microvm-artifact.sh` exists because CloudFormation cannot produce its own MicroVM `codeArtifact`: the image resource consumes a zip that must already be in S3, and there is no CDK asset type for "zip + Dockerfile a MicroVM image builds from". Everything else on that backend (buckets, roles, network connectors, log group, the image resource itself) is CDK-managed by `src/constructs/lambda-microvm-compute.ts`, normally inside `lambda-microvm-stack.ts`. -The nested layout requires bootstrap bundle 1.9.0. Existing flat deployments must -keep `microvm_nested_stack=false` until completing the +The default nested layout requires bootstrap bundle 1.9.0. Before upgrading an +existing flat deployment, set and retain `microvm_nested_stack=false` until completing the [resource migration](../../docs/verification/645-p3-nested-stack.md). diff --git a/cdk/scripts/package-microvm-artifact.sh b/cdk/scripts/package-microvm-artifact.sh index 579f1cbeb..065a4e644 100755 --- a/cdk/scripts/package-microvm-artifact.sh +++ b/cdk/scripts/package-microvm-artifact.sh @@ -19,8 +19,9 @@ # BOOTSTRAP SEQUENCE (first time) # --------------------------------------------------------------------------- # 0. Use bootstrap policy bundle >= 1.9.0 for the default nested layout. -# Existing flat deployments must retain microvm_nested_stack=false until -# their resources are migrated; see docs/verification/645-p3-nested-stack.md. +# Before upgrading an existing flat deployment, set and retain +# microvm_nested_stack=false (bundle >= 1.8.0) on every deploy below until +# its resources are migrated; see docs/verification/645-p3-nested-stack.md. # # 1. Deploy the MicroVM substrate WITHOUT an image. Synth warns that no image # is configured; that is expected — the artifact bucket must exist before @@ -353,8 +354,8 @@ if [[ "${CREATE_IMAGE}" -eq 0 ]]; then An older bundle may deny iam:PassRole for the image's build role. Updating the source bundle alone does not update the account's installed policies. - Existing flat stacks: keep --context microvm_nested_stack=false on deployment - commands until the resource migration is complete. Changing the layout directly + Existing flat stacks: set --context microvm_nested_stack=false before upgrading + and retain it until the resource migration is complete. Changing the layout directly can replace resources or delete bucket contents. See: docs/verification/645-p3-nested-stack.md diff --git a/cdk/src/stacks/agent.ts b/cdk/src/stacks/agent.ts index 463801690..50c459730 100644 --- a/cdk/src/stacks/agent.ts +++ b/cdk/src/stacks/agent.ts @@ -357,9 +357,9 @@ export class AgentStack extends Stack { if (microvmNestedContext !== undefined && ![true, false, 'true', 'false'].includes(microvmNestedContext)) { throw new Error('microvm_nested_stack must be true or false'); } - // Omission preserves existing flat resource identities. New installations - // may opt in; existing stacks must complete the reviewed migration first. - const microvmNested = microvmNestedContext === true || microvmNestedContext === 'true'; + // New installations use a child stack. Existing flat deployments must set + // false before upgrading and retain it until their resources are migrated. + const microvmNested = microvmNestedContext !== false && microvmNestedContext !== 'false'; const microvmResourceNamePrefix = this.node.tryGetContext('microvm_resource_name_prefix'); if (microvmResourceNamePrefix !== undefined && (!microvmNested || typeof microvmResourceNamePrefix !== 'string')) { diff --git a/cdk/test/stacks/agent.test.ts b/cdk/test/stacks/agent.test.ts index 1cd1d55c7..f47a3a9de 100644 --- a/cdk/test/stacks/agent.test.ts +++ b/cdk/test/stacks/agent.test.ts @@ -1303,7 +1303,6 @@ describe('AgentStack with the Lambda MicroVMs substrate gate (--context compute_ const app = new App({ context: { compute_type: 'lambda-microvm', - microvm_nested_stack: true, microvm_base_image_arn: BASE_IMAGE_ARN, microvm_base_image_version: '1', microvm_artifact_sha256: 'a'.repeat(64), @@ -1317,7 +1316,7 @@ describe('AgentStack with the Lambda MicroVMs substrate gate (--context compute_ childTemplate = Template.fromStack(stack.node.findChild('Microvm') as LambdaMicrovmStack); }); - test('provisions the MicroVM image + BOTH egress network connectors', () => { + test('defaults to a child stack for the MicroVM image and both egress network connectors', () => { template.resourceCountIs('AWS::Lambda::MicrovmImage', 0); template.resourceCountIs('AWS::Lambda::NetworkConnector', 0); childTemplate.resourceCountIs('AWS::Lambda::MicrovmImage', 1); @@ -1664,13 +1663,14 @@ describe('AgentStack with the MicroVM gate on but no image configured (first dep }); }); -describe('AgentStack MicroVM flat-layout migration compatibility', () => { +describe.each([false, 'false'])('AgentStack MicroVM flat-layout escape hatch (%s)', microvmNested => { let template: Template; beforeAll(() => { const app = new App({ context: { compute_type: 'lambda-microvm', + microvm_nested_stack: microvmNested, microvm_base_image_arn: 'arn:aws:lambda:us-east-1:aws:microvm-image:al2023-1', microvm_base_image_version: '1', microvm_artifact_sha256: 'a'.repeat(64), @@ -1681,7 +1681,7 @@ describe('AgentStack MicroVM flat-layout migration compatibility', () => { })); }); - test('preserves flat resource identities when nesting is not configured', () => { + test('preserves flat resource identities when nesting is explicitly disabled', () => { template.resourceCountIs('AWS::Lambda::MicrovmImage', 1); template.resourceCountIs('AWS::Lambda::NetworkConnector', 2); expect(Object.keys(template.findResources('AWS::Lambda::MicrovmImage'))[0]) @@ -2243,10 +2243,10 @@ describe('AgentStack CloudFormation resource budget 500 with cushion', () => { { name: 'agentcore', context: { compute_type: 'agentcore' } }, { name: 'ecs', context: { compute_type: 'ecs' } }, ...MICROVM_CONFIGURATIONS.flatMap(configuration => [ - { ...configuration, name: `${configuration.name}-default-flat` }, + { ...configuration, name: `${configuration.name}-default-nested` }, { - name: `${configuration.name}-nested`, - context: { ...configuration.context, microvm_nested_stack: true }, + name: `${configuration.name}-flat`, + context: { ...configuration.context, microvm_nested_stack: false }, }, ]), ]; diff --git a/docs/decisions/ADR-021-lambda-microvms-compute-backend.md b/docs/decisions/ADR-021-lambda-microvms-compute-backend.md index 403297552..3bb5830aa 100644 --- a/docs/decisions/ADR-021-lambda-microvms-compute-backend.md +++ b/docs/decisions/ADR-021-lambda-microvms-compute-backend.md @@ -149,7 +149,7 @@ The backend adds build/runtime VPC connectors, build artifacts, launch payloads, `lambda:PassNetworkConnector` has no resource-level authorization support and therefore uses `Resource: *`. The operator role also has wildcard ENI permissions; some mutations can be scoped in IAM, so their current tested wildcard is not evidence that narrower permissions are impossible. DescribeAvailabilityZones needs a wildcard for fresh CDK repository lookups. These exceptions and namespace wildcards are documented in the construct’s cdk-nag suppressions. -**Nested infrastructure.** `microvm_nested_stack=true` puts MicroVM resources in a nested stack. The shared execution role stays in the parent to avoid a role-trust dependency cycle. Bootstrap 1.9.0 covers nested deployment roles. Existing flat installations need the reviewed overlap/drain migration, explicit image/coordinator pins and rollback checks described in the [migration prerequisites](../verification/645-p3-nested-stack.md). Reusable migration commands are still being completed. Moving construct paths alone is not a safe migration. Preserve `microvm_resource_name_prefix` after a migrated deployment. +**Nested infrastructure.** `microvm_nested_stack` defaults to `true`, putting MicroVM resources in a nested stack. The shared execution role stays in the parent to avoid a role-trust dependency cycle. Bootstrap 1.9.0 covers nested deployment roles. Before upgrading an existing flat installation, set and retain `microvm_nested_stack=false` until completing the reviewed overlap/drain migration, explicit image/coordinator pins and rollback checks described in the [migration prerequisites](../verification/645-p3-nested-stack.md). Reusable migration commands are still being completed. Moving construct paths alone is not a safe migration. Preserve `microvm_resource_name_prefix` after a migrated deployment. #### Security bar vs existing backends ([#645](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/645) acceptance criterion) diff --git a/docs/design/COMPUTE.md b/docs/design/COMPUTE.md index 1f70a7a12..2cfaf5f4a 100644 --- a/docs/design/COMPUTE.md +++ b/docs/design/COMPUTE.md @@ -79,6 +79,10 @@ See [ORCHESTRATOR.md](./ORCHESTRATOR.md) for how the orchestrator handles these Lambda MicroVMs are an opt-in third backend, selected per repository with `compute_type: lambda-microvm`; AgentCore remains the default. Image configuration has three states: a managed base-image ARN, version and artifact digest create the snapshot image in CDK; an external image identifier uses a snapshot built out of band; and supplying neither image provisions only the roles, buckets, and connectors needed for the bootstrap deploy. Lambda MicroVMs are available in five launch regions (us-east-1, us-east-2, us-west-2, eu-west-1, ap-northeast-1) and will expand; the platform enforces regional availability in layers via a synth-time constant, onboarding live probes, and orchestration-time classification. +MicroVM infrastructure defaults to a nested stack and requires bootstrap bundle 1.9.0. +Before upgrading an existing flat installation, set `microvm_nested_stack=false` +and retain it until completing the [resource migration](../verification/645-p3-nested-stack.md). + For managed images, run `cdk/scripts/package-microvm-artifact.sh --stack-name ` after each agent change. It packages the Dockerfile's local inputs into a deterministic ZIP and uploads it under `microvm-images/agent-artifact-.zip`. The checksum is a fingerprint of the uploaded bytes: identical inputs reuse the same verified object, and changed inputs produce a new filename. Deploy with the printed `--context microvm_artifact_sha256=` alongside `microvm_base_image_arn` and `microvm_base_image_version`, and retain these inputs for later deployments. The changed S3 URI tells CloudFormation to update the existing image; overwriting the old fixed filename alone does not. Missing or malformed digests fail synthesis. An initial deployment without an image must create the bucket first. The script's explicit `--create-image` alternative retains the fixed base key and calls the image API directly. Because a snapshot freezes its build-time environment, current deployment identifiers arrive through the v2 payload bootstrap. The coordinator publishes a non-secret deployment manifest and sends a single-object signed download URL for the task. The worker reads only its deployment's `bootstrap/*` with ambient credentials; other object reads and payload-bucket listing are explicitly denied. The downloaded task identity and configuration must match the authenticated manifest before configuration installation. The serialized reference fits the verified 4,096-byte hook limit; all payload sizes use S3. ECS shares this transport through `AGENT_PAYLOAD_REF`. [ADR-021 §3](../decisions/ADR-021-lambda-microvms-compute-backend.md#3-packaging-same-agent-image-source-new-build-path) defines the wire format and compatibility requirements; [live payload checks](../verification/README.md) record validation. diff --git a/docs/design/DEPLOYMENT_ROLES.md b/docs/design/DEPLOYMENT_ROLES.md index 9c34c2349..8be3aa78a 100644 --- a/docs/design/DEPLOYMENT_ROLES.md +++ b/docs/design/DEPLOYMENT_ROLES.md @@ -846,13 +846,13 @@ The second statement, `MicrovmPassRoles`, is the one exception to the rule that P3 additionally requires **bundle 1.8.0** for `MicrovmSuspendConfiguration`. -The nested MicroVM layout requires **bundle 1.9.0**. Its child stack uses the +The default nested MicroVM layout requires **bundle 1.9.0**. Its child stack uses the explicit parent-derived names `backgroundagent-dev-MicrovmBuildRole` and `backgroundagent-dev-MicrovmConnectorRole`; `MicrovmPassRoles` admits those two exact names in addition to the legacy flat-layout prefixes. The execution role stays in the parent and is still excluded. Re-bootstrap before deploying the -child stack. Existing flat deployments must keep `microvm_nested_stack=false` -until their resource migration is reviewed; changing ownership is not an ordinary +child stack. Before upgrading an existing flat deployment, set and retain +`microvm_nested_stack=false` until its resource migration is complete; changing ownership is not an ordinary in-place update. See the [nested-stack runbook](../verification/645-p3-nested-stack.md). For a reviewed migration that keeps old and new resources side by side, diff --git a/docs/guides/DEPLOYMENT_GUIDE.md b/docs/guides/DEPLOYMENT_GUIDE.md index 10069fd95..a6120dde2 100644 --- a/docs/guides/DEPLOYMENT_GUIDE.md +++ b/docs/guides/DEPLOYMENT_GUIDE.md @@ -295,9 +295,12 @@ invocations and retain the old groups when investigating earlier runs. Both flat and nested MicroVM layouts support the optional tool gateway and Linear Identity vault without exceeding the template budget. -This does not migrate existing MicroVM compute resources. Keep an existing flat -deployment on `microvm_nested_stack=false` until its separate resource migration -has been reviewed. +MicroVM infrastructure now defaults to a nested stack (bootstrap bundle 1.9.0). +Before upgrading an existing flat MicroVM deployment, save +`"microvm_nested_stack": false` in its CDK context or pass +`--context microvm_nested_stack=false` on every deploy. Keep this escape hatch +until completing the [resource migration](../verification/645-p3-nested-stack.md). +Omitting the setting does not automatically migrate existing resources. ### AgentCore unsupported Availability Zones diff --git a/docs/src/content/docs/architecture/Compute.md b/docs/src/content/docs/architecture/Compute.md index 0d9786fba..0c2d2fe4b 100644 --- a/docs/src/content/docs/architecture/Compute.md +++ b/docs/src/content/docs/architecture/Compute.md @@ -83,6 +83,10 @@ See [ORCHESTRATOR.md](/sample-autonomous-cloud-coding-agents/architecture/orches Lambda MicroVMs are an opt-in third backend, selected per repository with `compute_type: lambda-microvm`; AgentCore remains the default. Image configuration has three states: a managed base-image ARN, version and artifact digest create the snapshot image in CDK; an external image identifier uses a snapshot built out of band; and supplying neither image provisions only the roles, buckets, and connectors needed for the bootstrap deploy. Lambda MicroVMs are available in five launch regions (us-east-1, us-east-2, us-west-2, eu-west-1, ap-northeast-1) and will expand; the platform enforces regional availability in layers via a synth-time constant, onboarding live probes, and orchestration-time classification. +MicroVM infrastructure defaults to a nested stack and requires bootstrap bundle 1.9.0. +Before upgrading an existing flat installation, set `microvm_nested_stack=false` +and retain it until completing the [resource migration](/sample-autonomous-cloud-coding-agents/verification/645-p3-nested-stack). + For managed images, run `cdk/scripts/package-microvm-artifact.sh --stack-name ` after each agent change. It packages the Dockerfile's local inputs into a deterministic ZIP and uploads it under `microvm-images/agent-artifact-.zip`. The checksum is a fingerprint of the uploaded bytes: identical inputs reuse the same verified object, and changed inputs produce a new filename. Deploy with the printed `--context microvm_artifact_sha256=` alongside `microvm_base_image_arn` and `microvm_base_image_version`, and retain these inputs for later deployments. The changed S3 URI tells CloudFormation to update the existing image; overwriting the old fixed filename alone does not. Missing or malformed digests fail synthesis. An initial deployment without an image must create the bucket first. The script's explicit `--create-image` alternative retains the fixed base key and calls the image API directly. Because a snapshot freezes its build-time environment, current deployment identifiers arrive through the v2 payload bootstrap. The coordinator publishes a non-secret deployment manifest and sends a single-object signed download URL for the task. The worker reads only its deployment's `bootstrap/*` with ambient credentials; other object reads and payload-bucket listing are explicitly denied. The downloaded task identity and configuration must match the authenticated manifest before configuration installation. The serialized reference fits the verified 4,096-byte hook limit; all payload sizes use S3. ECS shares this transport through `AGENT_PAYLOAD_REF`. [ADR-021 §3](/sample-autonomous-cloud-coding-agents/decisions/adr-021-lambda-microvms-compute-backend#3-packaging-same-agent-image-source-new-build-path) defines the wire format and compatibility requirements; [live payload checks](/sample-autonomous-cloud-coding-agents/verification/readme) record validation. diff --git a/docs/src/content/docs/architecture/Deployment-roles.md b/docs/src/content/docs/architecture/Deployment-roles.md index ebba8cfbb..91725540d 100644 --- a/docs/src/content/docs/architecture/Deployment-roles.md +++ b/docs/src/content/docs/architecture/Deployment-roles.md @@ -850,13 +850,13 @@ The second statement, `MicrovmPassRoles`, is the one exception to the rule that P3 additionally requires **bundle 1.8.0** for `MicrovmSuspendConfiguration`. -The nested MicroVM layout requires **bundle 1.9.0**. Its child stack uses the +The default nested MicroVM layout requires **bundle 1.9.0**. Its child stack uses the explicit parent-derived names `backgroundagent-dev-MicrovmBuildRole` and `backgroundagent-dev-MicrovmConnectorRole`; `MicrovmPassRoles` admits those two exact names in addition to the legacy flat-layout prefixes. The execution role stays in the parent and is still excluded. Re-bootstrap before deploying the -child stack. Existing flat deployments must keep `microvm_nested_stack=false` -until their resource migration is reviewed; changing ownership is not an ordinary +child stack. Before upgrading an existing flat deployment, set and retain +`microvm_nested_stack=false` until its resource migration is complete; changing ownership is not an ordinary in-place update. See the [nested-stack runbook](/sample-autonomous-cloud-coding-agents/verification/645-p3-nested-stack). For a reviewed migration that keeps old and new resources side by side, diff --git a/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md b/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md index 900ccdbee..6e45785ea 100644 --- a/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md +++ b/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md @@ -153,7 +153,7 @@ The backend adds build/runtime VPC connectors, build artifacts, launch payloads, `lambda:PassNetworkConnector` has no resource-level authorization support and therefore uses `Resource: *`. The operator role also has wildcard ENI permissions; some mutations can be scoped in IAM, so their current tested wildcard is not evidence that narrower permissions are impossible. DescribeAvailabilityZones needs a wildcard for fresh CDK repository lookups. These exceptions and namespace wildcards are documented in the construct’s cdk-nag suppressions. -**Nested infrastructure.** `microvm_nested_stack=true` puts MicroVM resources in a nested stack. The shared execution role stays in the parent to avoid a role-trust dependency cycle. Bootstrap 1.9.0 covers nested deployment roles. Existing flat installations need the reviewed overlap/drain migration, explicit image/coordinator pins and rollback checks described in the [migration prerequisites](/sample-autonomous-cloud-coding-agents/verification/645-p3-nested-stack). Reusable migration commands are still being completed. Moving construct paths alone is not a safe migration. Preserve `microvm_resource_name_prefix` after a migrated deployment. +**Nested infrastructure.** `microvm_nested_stack` defaults to `true`, putting MicroVM resources in a nested stack. The shared execution role stays in the parent to avoid a role-trust dependency cycle. Bootstrap 1.9.0 covers nested deployment roles. Before upgrading an existing flat installation, set and retain `microvm_nested_stack=false` until completing the reviewed overlap/drain migration, explicit image/coordinator pins and rollback checks described in the [migration prerequisites](/sample-autonomous-cloud-coding-agents/verification/645-p3-nested-stack). Reusable migration commands are still being completed. Moving construct paths alone is not a safe migration. Preserve `microvm_resource_name_prefix` after a migrated deployment. #### Security bar vs existing backends ([#645](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/645) acceptance criterion) diff --git a/docs/src/content/docs/getting-started/Deployment-guide.md b/docs/src/content/docs/getting-started/Deployment-guide.md index 223202704..13d868aec 100644 --- a/docs/src/content/docs/getting-started/Deployment-guide.md +++ b/docs/src/content/docs/getting-started/Deployment-guide.md @@ -299,9 +299,12 @@ invocations and retain the old groups when investigating earlier runs. Both flat and nested MicroVM layouts support the optional tool gateway and Linear Identity vault without exceeding the template budget. -This does not migrate existing MicroVM compute resources. Keep an existing flat -deployment on `microvm_nested_stack=false` until its separate resource migration -has been reviewed. +MicroVM infrastructure now defaults to a nested stack (bootstrap bundle 1.9.0). +Before upgrading an existing flat MicroVM deployment, save +`"microvm_nested_stack": false` in its CDK context or pass +`--context microvm_nested_stack=false` on every deploy. Keep this escape hatch +until completing the [resource migration](/sample-autonomous-cloud-coding-agents/verification/645-p3-nested-stack). +Omitting the setting does not automatically migrate existing resources. ### AgentCore unsupported Availability Zones diff --git a/docs/src/content/docs/verification/645-p3-nested-stack.md b/docs/src/content/docs/verification/645-p3-nested-stack.md index 691e1c2b2..b35407f65 100644 --- a/docs/src/content/docs/verification/645-p3-nested-stack.md +++ b/docs/src/content/docs/verification/645-p3-nested-stack.md @@ -23,7 +23,7 @@ Names derive from the concrete parent deployment name, not the child stack token | Setting | Behavior | |---|---| -| `microvm_nested_stack` | Defaults to `false`, preserving flat resource identities. Set `true` for a new installation or after completing the reviewed migration. | +| `microvm_nested_stack` | Defaults to `true`. Before upgrading an existing flat deployment, set `false` and retain it until completing the reviewed migration. | | `microvm_resource_name_prefix` | Supplies distinct names for overlapping nested resources; preserve it after migration. It does not retain old resources or permissions by itself. | | `microvm_managed_image_version` | Pins new tasks to an explicitly verified image version. Without a pin, selection follows the latest active version. | | `microvm_approval_suspend_enabled` | Defaults to `false`; enable only after testing the deployed image/coordinator. Disabling new sleep preserves wake and cleanup. | @@ -37,6 +37,11 @@ published coordinator and exact image version for rollback. ## Existing flat deployments +Before the first upgrade, save `"microvm_nested_stack": false` in the deployment's +CDK context or pass `--context microvm_nested_stack=false` on every deploy. +Omitting the setting now selects nested infrastructure; it does not detect or +migrate existing flat resources. Keep `false` until the migration below is complete. + Do not deploy the nested template directly over a flat deployment. CloudFormation sees removed parent resources and newly created child resources; names can collide and bucket auto-delete handlers can erase artifacts or pending payloads. diff --git a/docs/verification/645-p3-nested-stack.md b/docs/verification/645-p3-nested-stack.md index 30e2e7b97..ed0a221b8 100644 --- a/docs/verification/645-p3-nested-stack.md +++ b/docs/verification/645-p3-nested-stack.md @@ -19,7 +19,7 @@ Names derive from the concrete parent deployment name, not the child stack token | Setting | Behavior | |---|---| -| `microvm_nested_stack` | Defaults to `false`, preserving flat resource identities. Set `true` for a new installation or after completing the reviewed migration. | +| `microvm_nested_stack` | Defaults to `true`. Before upgrading an existing flat deployment, set `false` and retain it until completing the reviewed migration. | | `microvm_resource_name_prefix` | Supplies distinct names for overlapping nested resources; preserve it after migration. It does not retain old resources or permissions by itself. | | `microvm_managed_image_version` | Pins new tasks to an explicitly verified image version. Without a pin, selection follows the latest active version. | | `microvm_approval_suspend_enabled` | Defaults to `false`; enable only after testing the deployed image/coordinator. Disabling new sleep preserves wake and cleanup. | @@ -33,6 +33,11 @@ published coordinator and exact image version for rollback. ## Existing flat deployments +Before the first upgrade, save `"microvm_nested_stack": false` in the deployment's +CDK context or pass `--context microvm_nested_stack=false` on every deploy. +Omitting the setting now selects nested infrastructure; it does not detect or +migrate existing flat resources. Keep `false` until the migration below is complete. + Do not deploy the nested template directly over a flat deployment. CloudFormation sees removed parent resources and newly created child resources; names can collide and bucket auto-delete handlers can erase artifacts or pending payloads. From 0da9cf315ea50a96bd7463b1833ac6077b8c8964 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Tue, 22 Sep 2026 11:12:07 -0400 Subject: [PATCH 130/149] fix(645): warn before implicit nested MicroVM deployment --- cdk/src/stacks/agent.ts | 24 ++++- .../stacks/microvm-layout-context.test.ts | 96 +++++++++++++++++++ .../docs/verification/645-p3-nested-stack.md | 5 +- docs/verification/645-p3-nested-stack.md | 5 +- 4 files changed, 124 insertions(+), 6 deletions(-) create mode 100644 cdk/test/stacks/microvm-layout-context.test.ts diff --git a/cdk/src/stacks/agent.ts b/cdk/src/stacks/agent.ts index 50c459730..cdd723958 100644 --- a/cdk/src/stacks/agent.ts +++ b/cdk/src/stacks/agent.ts @@ -19,7 +19,7 @@ import * as path from 'path'; import * as bedrock from '@aws-cdk/aws-bedrock-alpha'; -import { ArnFormat, AspectPriority, Aspects, Stack, StackProps, NestedStack, RemovalPolicy, CfnOutput, CfnResource, Duration, Fn, Lazy } from 'aws-cdk-lib'; +import { Annotations, ArnFormat, AspectPriority, Aspects, Stack, StackProps, NestedStack, RemovalPolicy, CfnOutput, CfnResource, Duration, Fn, Lazy } from 'aws-cdk-lib'; import * as agentcore from 'aws-cdk-lib/aws-bedrockagentcore'; import * as ec2 from 'aws-cdk-lib/aws-ec2'; import * as ecr_assets from 'aws-cdk-lib/aws-ecr-assets'; @@ -360,10 +360,26 @@ export class AgentStack extends Stack { // New installations use a child stack. Existing flat deployments must set // false before upgrading and retain it until their resources are migrated. const microvmNested = microvmNestedContext !== false && microvmNestedContext !== 'false'; + if (lambdaMicrovmEnabled && microvmNestedContext === undefined) { + Annotations.of(this).addWarningV2( + 'abca:microvm-implicit-nested-layout', + 'microvm_nested_stack is unset: MicroVM infrastructure defaults to a nested stack. ' + + 'Before upgrading an existing flat MicroVM deployment, set --context microvm_nested_stack=false ' + + 'and retain it until migration is complete. Applying the nested template directly can cause ' + + 'resource-name collisions and bucket auto-delete handlers can erase artifacts and pending payloads. ' + + 'For a new or already-nested deployment, explicitly set microvm_nested_stack=true to acknowledge ' + + 'the layout. This warning does not inspect deployed resources or perform a migration. ' + + 'See docs/verification/645-p3-nested-stack.md.', + ); + } const microvmResourceNamePrefix = this.node.tryGetContext('microvm_resource_name_prefix'); - if (microvmResourceNamePrefix !== undefined - && (!microvmNested || typeof microvmResourceNamePrefix !== 'string')) { - throw new Error('microvm_resource_name_prefix requires a string and microvm_nested_stack=true'); + if (microvmResourceNamePrefix !== undefined) { + if (typeof microvmResourceNamePrefix !== 'string') { + throw new Error('microvm_resource_name_prefix must be a string'); + } + if (!microvmNested) { + throw new Error('microvm_resource_name_prefix cannot be used with microvm_nested_stack=false'); + } } const suspendContext = this.node.tryGetContext('microvm_approval_suspend_enabled'); if (suspendContext !== undefined && ![true, false, 'true', 'false'].includes(suspendContext)) { diff --git a/cdk/test/stacks/microvm-layout-context.test.ts b/cdk/test/stacks/microvm-layout-context.test.ts new file mode 100644 index 000000000..9109fc274 --- /dev/null +++ b/cdk/test/stacks/microvm-layout-context.test.ts @@ -0,0 +1,96 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +import { App } from 'aws-cdk-lib'; +import { Annotations, Match, Template } from 'aws-cdk-lib/assertions'; +import { AgentStack } from '../../src/stacks/agent'; + +const env = { account: '123456789012', region: 'us-east-1' }; +const imageContext = { + microvm_base_image_arn: 'arn:aws:lambda:us-east-1:aws:microvm-image:al2023-1', + microvm_base_image_version: '1', + microvm_artifact_sha256: 'a'.repeat(64), +}; + +describe.each([ + { name: 'MicrovmBootstrap', context: { compute_type: 'lambda-microvm' }, warns: true }, + { name: 'MicrovmManaged', context: { compute_type: 'lambda-microvm', ...imageContext }, warns: true }, + ...[true, 'true', false, 'false'].map((value, index) => ({ + name: `ExplicitLayout${index}`, + context: { compute_type: 'lambda-microvm', microvm_nested_stack: value }, + warns: false, + })), + { name: 'Agentcore', context: { compute_type: 'agentcore' }, warns: false }, + { name: 'Ecs', context: { compute_type: 'ecs' }, warns: false }, +])('MicroVM layout warning: $name', ({ name, context, warns }) => { + let annotations: Annotations; + + beforeAll(() => { + const stack = new AgentStack(new App({ context }), name, { env }); + Template.fromStack(stack); + annotations = Annotations.fromStack(stack); + }); + + test('warns only when MicroVM layout is implicit, even before an image exists', () => { + const warnings = annotations.findWarning('*', Match.stringLikeRegexp('microvm_nested_stack is unset')); + expect(warnings).toHaveLength(warns ? 1 : 0); + if (warns) { + const message = warnings[0]!.entry.data; + expect(message).toContain('--context microvm_nested_stack=false'); + expect(message).toContain('erase artifacts and pending payloads'); + expect(message).toContain('does not inspect deployed resources or perform a migration'); + expect(message).toContain('docs/verification/645-p3-nested-stack.md'); + } + }); +}); + +describe('MicroVM resource prefix validation', () => { + test.each([false, 'false'])('explains an explicitly disabled nested layout (%s)', value => { + const app = new App({ + context: { + compute_type: 'lambda-microvm', + microvm_nested_stack: value, + microvm_resource_name_prefix: 'migration', + }, + }); + expect(() => new AgentStack(app, 'FlatPrefix', { env })) + .toThrow('microvm_resource_name_prefix cannot be used with microvm_nested_stack=false'); + }); + + test.each([42, true, null])('identifies non-string prefix %s without requiring an explicit true flag', value => { + const app = new App({ + context: { + compute_type: 'lambda-microvm', + microvm_resource_name_prefix: value, + }, + }); + expect(() => new AgentStack(app, 'InvalidPrefix', { env })) + .toThrow('microvm_resource_name_prefix must be a string'); + }); + + test('accepts a valid prefix with the default nested layout', () => { + const app = new App({ + context: { + compute_type: 'lambda-microvm', + microvm_resource_name_prefix: 'migration', + }, + }); + expect(() => new AgentStack(app, 'DefaultNestedPrefix', { env })).not.toThrow(); + }); +}); diff --git a/docs/src/content/docs/verification/645-p3-nested-stack.md b/docs/src/content/docs/verification/645-p3-nested-stack.md index b35407f65..377b82a60 100644 --- a/docs/src/content/docs/verification/645-p3-nested-stack.md +++ b/docs/src/content/docs/verification/645-p3-nested-stack.md @@ -23,7 +23,7 @@ Names derive from the concrete parent deployment name, not the child stack token | Setting | Behavior | |---|---| -| `microvm_nested_stack` | Defaults to `true`. Before upgrading an existing flat deployment, set `false` and retain it until completing the reviewed migration. | +| `microvm_nested_stack` | Defaults to `true`. When MicroVM is enabled, omission emits a synthesis warning. Before upgrading an existing flat deployment, set `false` and retain it until completing the reviewed migration. | | `microvm_resource_name_prefix` | Supplies distinct names for overlapping nested resources; preserve it after migration. It does not retain old resources or permissions by itself. | | `microvm_managed_image_version` | Pins new tasks to an explicitly verified image version. Without a pin, selection follows the latest active version. | | `microvm_approval_suspend_enabled` | Defaults to `false`; enable only after testing the deployed image/coordinator. Disabling new sleep preserves wake and cleanup. | @@ -41,6 +41,9 @@ Before the first upgrade, save `"microvm_nested_stack": false` in the deployment CDK context or pass `--context microvm_nested_stack=false` on every deploy. Omitting the setting now selects nested infrastructure; it does not detect or migrate existing flat resources. Keep `false` until the migration below is complete. +The synthesis warning is advisory: it does not inspect the deployed stack or block +deployment. New and already-nested installations can explicitly set `true` to +acknowledge the layout; setting it is not a migration. Do not deploy the nested template directly over a flat deployment. CloudFormation sees removed parent resources and newly created child resources; names can collide diff --git a/docs/verification/645-p3-nested-stack.md b/docs/verification/645-p3-nested-stack.md index ed0a221b8..73a979779 100644 --- a/docs/verification/645-p3-nested-stack.md +++ b/docs/verification/645-p3-nested-stack.md @@ -19,7 +19,7 @@ Names derive from the concrete parent deployment name, not the child stack token | Setting | Behavior | |---|---| -| `microvm_nested_stack` | Defaults to `true`. Before upgrading an existing flat deployment, set `false` and retain it until completing the reviewed migration. | +| `microvm_nested_stack` | Defaults to `true`. When MicroVM is enabled, omission emits a synthesis warning. Before upgrading an existing flat deployment, set `false` and retain it until completing the reviewed migration. | | `microvm_resource_name_prefix` | Supplies distinct names for overlapping nested resources; preserve it after migration. It does not retain old resources or permissions by itself. | | `microvm_managed_image_version` | Pins new tasks to an explicitly verified image version. Without a pin, selection follows the latest active version. | | `microvm_approval_suspend_enabled` | Defaults to `false`; enable only after testing the deployed image/coordinator. Disabling new sleep preserves wake and cleanup. | @@ -37,6 +37,9 @@ Before the first upgrade, save `"microvm_nested_stack": false` in the deployment CDK context or pass `--context microvm_nested_stack=false` on every deploy. Omitting the setting now selects nested infrastructure; it does not detect or migrate existing flat resources. Keep `false` until the migration below is complete. +The synthesis warning is advisory: it does not inspect the deployed stack or block +deployment. New and already-nested installations can explicitly set `true` to +acknowledge the layout; setting it is not a migration. Do not deploy the nested template directly over a flat deployment. CloudFormation sees removed parent resources and newly created child resources; names can collide From c6f9635db634d5c12b8d2932f82e25f634b441cc Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Tue, 22 Sep 2026 11:24:18 -0400 Subject: [PATCH 131/149] docs(645): align worker and infrastructure comments with P3 --- agent/README.md | 4 +-- agent/src/hooks.py | 22 +++++++-------- agent/src/policy.py | 3 +- cdk/scripts/package-microvm-artifact.sh | 32 ++++++---------------- cdk/src/constructs/task-approvals-table.ts | 28 ++++++------------- cdk/src/handlers/shared/orchestrator.ts | 4 +-- cdk/src/stacks/agent.ts | 25 +++++++---------- 7 files changed, 42 insertions(+), 76 deletions(-) diff --git a/agent/README.md b/agent/README.md index 349a47007..e1ca763b7 100644 --- a/agent/README.md +++ b/agent/README.md @@ -248,7 +248,7 @@ It runs under the **build role**, which has no Bedrock, Secrets Manager or Dynam Baked secrets are **reported, not enforced**: `warnings` lists the names (never values) of any credential-shaped env var present in the snapshot, because the build environment's own credentials may legitimately be in that env and failing here would fail every build. -**`POST /aws/lambda-microvms/runtime/v1/terminate`** — Runtime hook (P2). Best-effort: emits one final structured log line and returns 200 — always, inside the hook budget, even with nothing running, and for **any body**: malformed JSON, a wrong content-type, an empty body or no body at all. That is why the handler takes the raw request instead of a typed body model — FastAPI validates a typed body *before* the handler runs, so a truncated body would answer 422 and report a hook failure for a teardown that actually succeeded. It does **not** join the pipeline thread (that is `lifespan`'s job on graceful shutdown) and it **never writes terminal task status**: the orchestrator finalizes the task and *then* calls `TerminateMicrovm`, so a status write here would race that finalization. `ProgressWriter` writes each event synchronously, but catches and drops failures; returning from an event method is not a durability guarantee. The P3 `/suspend` hook uses its own atomic acknowledged checkpoint after draining tracked activity. +**`POST /aws/lambda-microvms/runtime/v1/terminate`** — Closes the local coding barrier, logs teardown and acknowledges any request body within the hook budget. Raw request parsing avoids a premature FastAPI validation error for malformed input. Each step is best-effort. The hook neither joins the pipeline thread nor writes terminal task status: termination can interrupt active work or retire a worker whose approval remains pending. Task finalization belongs to the coordinator. Acknowledged checkpointing belongs to `/suspend`; ordinary progress logging is not a durability guarantee. `microvmId` is parsed defensively and **arrives empty in practice**: the service sends `""` here, unlike `/run` where it is populated (live-verified, ADR-021 P2-F8). So an empty id is expected-normal, not a degraded read — and this hook therefore **cannot** join the guest's record to the control-plane one. `/run`'s `hook accepted task_id=… microvm_id=…` line carries that correlation; `/terminate`'s value is the pipeline-state snapshot it reports. @@ -510,7 +510,7 @@ agent/ │ ├── repo.py Repository setup: clone, branch, git auth, mise trust/install/build/lint │ ├── shell.py Shell utilities: log(), run_cmd(), redact_secrets(), slugify(), truncate() │ ├── telemetry.py Metrics, disk usage, trajectory writer (_TrajectoryWriter with write_policy_decision) -│ ├── server.py FastAPI — async /invocations (background thread), /ping health check, MicroVM /ready + /run lifecycle hooks, heartbeat daemon; OTEL session correlation +│ ├── server.py FastAPI — async /invocations (background thread), /ping health check, MicroVM build/run/terminate hooks (suspend/resume in microvm_http.py), heartbeat daemon; OTEL session correlation │ ├── task_state.py Best-effort DynamoDB task status and heartbeat writes (no-op if TASK_TABLE_NAME unset) │ ├── observability.py OpenTelemetry helpers (e.g. AgentCore session id) │ ├── memory.py Optional memory / episode integration for the agent diff --git a/agent/src/hooks.py b/agent/src/hooks.py index 9039bfcff..5575ae514 100644 --- a/agent/src/hooks.py +++ b/agent/src/hooks.py @@ -1,8 +1,9 @@ """PreToolUse, PostToolUse, and Stop hook callbacks. - PreToolUse: three-outcome Cedar policy enforcement (ALLOW / DENY / - REQUIRE_APPROVAL). The REQUIRE_APPROVAL path writes a pending approval - row + transitions the task to AWAITING_APPROVAL atomically, polls for a + REQUIRE_APPROVAL). The REQUIRE_APPROVAL path asks the trusted approval + service to create a pending row and atomically transition the task to + AWAITING_APPROVAL, polls for a human decision, then resumes / denies per the user's input. See ``docs/design/CEDAR_HITL_GATES.md``. - PostToolUse: output scanner for secrets/PII. @@ -417,8 +418,9 @@ async def pre_tool_use_hook( - permissionDecision: "allow" or "deny" - permissionDecisionReason: explanation string - The REQUIRE_APPROVAL path pauses here: writes a pending approval row - + transitions the task to AWAITING_APPROVAL atomically, polls for a + The REQUIRE_APPROVAL path pauses here: the trusted service creates a pending + approval row and atomically transitions the task to AWAITING_APPROVAL, then + the worker polls for a human decision with 2s→5s backoff, then returns allow / deny based on the decision. On TIMED_OUT a ConditionCheckFailed from the best-effort status write triggers a re-read — if the user's decision landed between @@ -1195,14 +1197,10 @@ def _compute_effective_timeout( ) -> tuple[int, str | None, int]: """Compute the effective approval timeout. - ``min(rule-annotation timeout, task default, remaining lifetime - - cleanup margin)``, floored at FLOOR_30S. The engine's - ``_merge_annotations`` already applies ``min(rule_annotation, - task_default)`` — decision.timeout_s reaches us pre-clipped against - those two. Here we apply the remaining-lifetime ceiling and report - whichever source pulled the effective timeout below the task - default, so the user sees "your gate was clipped because ..." rather - than silent clipping. + The engine has already merged positive rule/task deadlines. Zero means + no deadline and returns unchanged. For a positive result, apply an optional + remaining-lifetime ceiling and the 30-second floor. Continuation runtimes + omit the lifetime ceiling because the request can outlive this worker. Returns ``(effective, clip_reason, requested)``: - ``requested`` — the user-visible "would have liked" value (task diff --git a/agent/src/policy.py b/agent/src/policy.py index 6cdf62420..3d6abd8bd 100644 --- a/agent/src/policy.py +++ b/agent/src/policy.py @@ -699,7 +699,8 @@ def _merge_annotations( ) -> tuple[list[str], int, str]: """Merge annotations across multiple matching soft-deny policies (§6.3). - Timeout: min across rules (clamped by FLOOR_TIMEOUT_S). Severity: max. + Timeout: shortest positive rule/task value (clamped by FLOOR_TIMEOUT_S), + or zero when no deadline is configured. Severity: max. rule_ids preserved in order of match. If a matching rule has no annotation data (shouldn't happen post-validation), falls back to the policy ID. diff --git a/cdk/scripts/package-microvm-artifact.sh b/cdk/scripts/package-microvm-artifact.sh index 065a4e644..001ac2ab1 100755 --- a/cdk/scripts/package-microvm-artifact.sh +++ b/cdk/scripts/package-microvm-artifact.sh @@ -90,30 +90,14 @@ # Test the selected image/coordinator together before enabling automatic sleep. # See docs/verification/README.md for acceptance criteria and remaining PR checks. # -# ADR-021 sub-decision 3's hook-phasing table (corrected after the live P1 -# verification run, then completed in P2) is now: -# -# /ready, /run declared by the CDK construct AND served by -# the agent in P1 (agent/src/server.py). -# /ready is MANDATORY: create-microvm-image -# refuses any lifecycle hook without it, and an -# image with no hooks at all cannot receive a -# runHookPayload — so "declare /run in P1, -# serve it in P2" was never a reachable state. -# /validate, /terminate declared AND served in P2. /validate is a -# build-time self-check that makes ZERO AWS -# calls (it runs under the build role, which -# holds no Bedrock/Secrets/DynamoDB grants); -# /terminate is a best-effort in-guest -# breadcrumb that must not write terminal task -# status — the orchestrator finalizes the task -# and THEN calls TerminateMicrovm. -# /suspend, /resume declared AND served in P3, with the image -# protocol marker. The coordinator verifies the -# actual launched version before allowing sleep. -# -# The clean P2 deployment used bootstrap policy bundle 1.7.0. The Dockerfile is -# copied unmodified; image build success does not establish full P2 acceptance. +# Managed images declare all six hooks: +# /ready, /validate Local build-time checks and warm-up; no AWS calls. +# /run Authenticated task bootstrap and asynchronous execution. +# /terminate Close the coding barrier and acknowledge teardown without +# finalizing the task; the worker may be retiring mid-task. +# /suspend, /resume Checkpoint and credential/gate reconciliation. The +# coordinator verifies the launched image's protocol marker +# before allowing automatic sleep. # # Requires: awscli v2 with conditional PutObject/checksum support, python3. diff --git a/cdk/src/constructs/task-approvals-table.ts b/cdk/src/constructs/task-approvals-table.ts index 68d4683d3..159bae293 100644 --- a/cdk/src/constructs/task-approvals-table.ts +++ b/cdk/src/constructs/task-approvals-table.ts @@ -60,9 +60,9 @@ export interface TaskApprovalsTableProps { * * Schema: `task_id` (PK, ULID matching TaskTable) + `request_id` (SK, * ULID minted by the agent). Each row represents one human-in-the-loop - * approval gate; the agent writes PENDING, the ApproveTaskFn / - * DenyTaskFn Lambdas (Chunk 5) update to APPROVED / DENIED, and the - * reconciler sweeps STRANDED rows. + * approval gate. The trusted approval request service creates PENDING rows; + * owner-authenticated decision handlers record APPROVED / DENIED. Task closure + * cancels unanswered requests. Workers have no direct approval-row write grant. * * A GSI (`user_id-status-index`) supports the `bgagent pending` access * pattern — `user_id = :caller AND status = :pending` — without @@ -117,10 +117,9 @@ export class TaskApprovalsTable extends Construct { // GSI for GET /v1/pending — user_id PK + status SK (§10.1). // - // Projection is INCLUDE with exactly the non-key attributes the - // pending-list endpoint needs: keeps per-write cost small while - // keeping the list response small enough to render in the CLI - // without additional GetItem round-trips. + // Preserve the existing INCLUDE projection. The pending-list endpoint uses + // this GSI to discover candidates, then strongly reads approval/task rows + // before returning them; GSI results alone may contain stale decisions. this.table.addGlobalSecondaryIndex({ indexName: USER_STATUS_INDEX_NAME, partitionKey: { @@ -141,19 +140,8 @@ export class TaskApprovalsTable extends Construct { 'reason', 'created_at', 'timeout_s', - // Cedar HITL: surface which rule(s) fired on the gate in the - // pending-list response so `bgagent pending` can show _why_ - // without a second read against the base table. Projected - // because the handler reads rows through this GSI. - // - // ARCHITECTURAL NOTE: DynamoDB rejects in-place updates to - // ``nonKeyAttributes`` on an existing GSI. Any future field - // that needs to appear on the pending view must be decided - // here at design time — adding one post-hoc requires a - // destructive migration (delete + recreate the table, or - // create a parallel GSI under a new name with shadow - // backfill). Chunks that extend TaskApprovalsTable should - // audit this list before shipping. + // DynamoDB rejects in-place projection changes. New response fields + // can come from the existing base-table read without changing this GSI. 'matching_rule_ids', ], }); diff --git a/cdk/src/handlers/shared/orchestrator.ts b/cdk/src/handlers/shared/orchestrator.ts index b6afd9fe7..23047fed5 100644 --- a/cdk/src/handlers/shared/orchestrator.ts +++ b/cdk/src/handlers/shared/orchestrator.ts @@ -323,8 +323,8 @@ const MAX_POLL_INTERVAL_MS = 300_000; /** * Build the ``compute_metadata`` map persisted on the task row at session start, * so a later handler can act on the right backend without re-deriving anything: - * ``cancel-task.ts`` reads ``clusterArn``/``taskArn`` from it today, and ADR-021 - * sub-decision 2 has the approve/deny Lambdas read ``microvmId`` from it in P3. + * cancellation uses the backend's worker identity, and MicroVM approval wake + * uses ``microvmId`` plus the saved image identity. * * Kept as an exhaustive switch (not a ternary + spread) so a fourth backend is a * compile error here rather than a silently empty metadata map — the field is diff --git a/cdk/src/stacks/agent.ts b/cdk/src/stacks/agent.ts index cdd723958..0718d35ff 100644 --- a/cdk/src/stacks/agent.ts +++ b/cdk/src/stacks/agent.ts @@ -2319,8 +2319,8 @@ interface PinnedLogResource { * constructed — the hash in each one is not reproducible from the construct path * alone, which is precisely why they have to be written down. * - * An entry stays until its stack is gone. Removing one while the stack still - * exists re-introduces the rename and the failed update that comes with it. + * Removing an entry requires checking the deployed ids and completing any + * necessary log-delivery migration; otherwise the rename can fail the update. */ const PINNED_LOG_DELIVERY_BY_STACK: Record = { 'backgroundagent-dev': [ @@ -2357,7 +2357,7 @@ const PINNED_LOG_DELIVERY_BY_STACK: Record }; /** - * Pin the auto-created log-delivery resources to stable logical ids, ALWAYS. + * Apply captured legacy log-delivery ids for entries in the pin table. * * These resources are created for us by the AgentCore Runtime and named after * whatever construct path the library uses internally, so a library-side rename @@ -2368,15 +2368,12 @@ const PINNED_LOG_DELIVERY_BY_STACK: Record * update rolls the whole stack back. Owning the ids ourselves decouples us from * the library's internal naming. * - * Applied unconditionally rather than behind a flag. Three cases, all safe: - * - * - An existing stack in the account that owns these resources: the ids match - * what CloudFormation already recorded, so it updates them in place. This is - * the case that was broken. - * - A fresh stack or account: nothing owns these names yet, so they create - * normally. The ids are ours rather than the library's, which is the point; - * the values themselves carry no meaning beyond being stable. - * - Any other name: the ids embed the stack name, so each stack gets its own. + * Known limitation (#703): stack name does not identify deployment history. + * An existing same-named stack that already uses the library's current ids can + * collide when these pins rename its resources. Fresh stacks have no existing + * resources to collide with. Other stack names keep the library's naming. + * Removing the pins also requires migration for stacks still using these ids; + * the account-agnostic fix and migration are tracked in PR #705. * * The values were read off a stack deployed before the rename. Do not "tidy" * them — they are a record of what CloudFormation already has, and editing one @@ -2385,9 +2382,7 @@ const PINNED_LOG_DELIVERY_BY_STACK: Record function pinLogDeliveryLogicalIds(runtime: agentcore.Runtime): void { const stack = Stack.of(runtime); const pins = PINNED_LOG_DELIVERY_BY_STACK[stack.stackName]; - // Only the stack these ids were recorded from can use them: they embed that - // stack's name. Any other stack keeps the library's own naming, which is - // correct for it — it has no pre-rename resources to line up with. + // This lookup checks only the name, not the account or deployed ids (#703). if (!pins) return; for (const pin of pins) { From 548bdf48689d104494f298ddf5871ff19d9b7167 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Tue, 22 Sep 2026 11:24:36 -0400 Subject: [PATCH 132/149] docs(645): refresh approval contracts and supported response flows --- ...ADR-021-lambda-microvms-compute-backend.md | 4 +- docs/design/API_CONTRACT.md | 4 +- docs/design/CEDAR_HITL_GATES.md | 365 +++++------------- docs/design/INTERACTIVE_AGENTS.md | 44 ++- .../content/docs/architecture/Api-contract.md | 4 +- .../docs/architecture/Cedar-hitl-gates.md | 365 +++++------------- .../docs/architecture/Interactive-agents.md | 44 ++- ...Adr-021-lambda-microvms-compute-backend.md | 4 +- 8 files changed, 230 insertions(+), 604 deletions(-) diff --git a/docs/decisions/ADR-021-lambda-microvms-compute-backend.md b/docs/decisions/ADR-021-lambda-microvms-compute-backend.md index 3bb5830aa..9fa9bdf2c 100644 --- a/docs/decisions/ADR-021-lambda-microvms-compute-backend.md +++ b/docs/decisions/ADR-021-lambda-microvms-compute-backend.md @@ -174,7 +174,7 @@ The launch-region list is us-east-1, us-east-2, us-west-2, eu-west-1 and ap-nort - CLI onboarding and platform doctor probe `list-managed-microvm-images`. - Runtime regional failures receive a configuration remedy instead of an opaque SDK error. -### 5. Rollout: phased, default unchanged +### 5. Rollout: phased, AgentCore remains the default backend | Phase | Delivered behavior | |---|---| @@ -184,7 +184,7 @@ The launch-region list is us-east-1, us-east-2, us-west-2, eu-west-1 and ap-nort Activate sleep only after verifying the deployed image and coordinator together. Keep a compatible published coordinator and explicit image pin for rollback. Normal acceptance includes the 600-second default, explicit expiry, new and existing task off-switch behavior, replacement, cleanup and preservation of unrelated infrastructure. -Changing the default backend, GPU support, native Slack approval buttons, approval-by-Linear-reply and operator shell access are outside this ADR. CLI responses remain the supported approval path. Future work needs its own scope; “P4” is not an approved phase here. +Approval responses are supported through the authenticated CLI and owner-authored `approve`/`deny` replies to Linear approval comments. Changing the default backend, GPU support, native Slack approval buttons and operator shell access remain outside this ADR. Future work needs its own scope; “P4” is not an approved phase here. ## Consequences diff --git a/docs/design/API_CONTRACT.md b/docs/design/API_CONTRACT.md index fc26f9485..4d9f64e1b 100644 --- a/docs/design/API_CONTRACT.md +++ b/docs/design/API_CONTRACT.md @@ -382,7 +382,7 @@ When a task pauses in `AWAITING_APPROVAL` (Cedar soft-deny gate), the owner appr { "data": { "task_id": "01HYX...", "request_id": "...", "status": "DENIED", "decided_at": "2025-03-15T10:35:00Z" } } ``` -**Errors:** `400 VALIDATION_ERROR`, `401 UNAUTHORIZED`, `404 REQUEST_NOT_FOUND` (collapses "row missing" and "wrong caller"), `409 REQUEST_ALREADY_DECIDED`, `409 TASK_NOT_AWAITING_APPROVAL`. +**Errors:** `400 VALIDATION_ERROR`, `401 UNAUTHORIZED`, `404 REQUEST_NOT_FOUND` (collapses missing, inaccessible, closed or expired approval rows), `409 TASK_NOT_AWAITING_APPROVAL` (task-only state conflict). ### List pending approvals @@ -608,7 +608,7 @@ There is no per-user request-rate or "tasks-per-hour" limiter on task creation. | `RATE_LIMIT_EXCEEDED` | 429 | Rate/concurrency gate exceeded — per-task nudge limit, the application rate limiter on approval endpoints, or the user concurrency limit on confirm-uploads | | `BUDGET_EXCEEDED` | 429 | A configured user or Cognito-team monthly budget reached 100% with hard stop enabled | | `REQUEST_NOT_FOUND` | 404 | Cedar HITL approval request not found (also returned when the caller does not own it) | -| `REQUEST_ALREADY_DECIDED` | 409 | Cedar HITL approval request was already approved or denied | +| `REQUEST_ALREADY_DECIDED` | 409 | Legacy error-code enum; current approve/deny handlers return `404 REQUEST_NOT_FOUND` for closed or inaccessible approval rows | | `TASK_NOT_AWAITING_APPROVAL` | 409 | Task is not in `AWAITING_APPROVAL`, so the approval decision does not apply | | `INTERNAL_ERROR` | 500 | Unexpected server error | | `SERVICE_UNAVAILABLE` | 503 | Downstream dependency unavailable (retry with backoff) | diff --git a/docs/design/CEDAR_HITL_GATES.md b/docs/design/CEDAR_HITL_GATES.md index 4a049f40e..ea442af7d 100644 --- a/docs/design/CEDAR_HITL_GATES.md +++ b/docs/design/CEDAR_HITL_GATES.md @@ -6,8 +6,10 @@ > **Rev:** 5 (2026-05-06 — fold in parallel adversarial + advocate review of the timeout design: late-approval re-read on TIMED_OUT ConditionCheckFailed; user-visible timeout-cap milestones; ceiling-shrink milestone; Runtime JWT bound verified as auto-refreshed IAM; three new tuning metrics; explicit off-hours trade-off section; notification-delivery-failure boundary. IMPL-24 through IMPL-28 added.). > **Implementation:** Core shipped. The 3-outcome engine (`agent/src/policy.py`), default policy sets (`agent/policies/hard_deny.cedar`, `agent/policies/soft_deny.cedar`), approval Lambdas (`cdk/src/handlers/{approve-task,deny-task,get-pending,get-policies}.ts`) wired into `cdk/src/constructs/task-api.ts` (routes `/tasks/{id}/approve`, `/deny`, `/pending`, `/repos/{repo_id}/policies`), the cross-engine parity fixtures (`contracts/cedar-parity/`), and the exact engine pins are all on `main`. §15's task list is preserved as a historical implementation record; see the note at the top of §15 for what (if anything) remains unbuilt. > -> **Current source behavior (2026-09-18):** the task default is `approval_timeout_s=0`, -> meaning no decision deadline. Explicit task settings are 30–3,600 seconds; +> **Current source behavior (2026-09-22):** the task default is `approval_timeout_s=0`, +> meaning no decision deadline. Workers create/close requests through the IAM-authenticated +> [trusted approval writer](../decisions/ADR-023-trusted-approval-writer.md); they cannot +> write approval rows directly. Explicit task settings are 30–3,600 seconds; > positive policy-rule deadlines still apply. Pending rows have no DynamoDB TTL, > and `expires_at` is nullable. Task closure cancels unanswered requests and adds > retention TTL without changing already-recorded decisions. Closed approval-row @@ -154,7 +156,7 @@ Settled during the 2026-04-23 design discussion and extended after the 2026-04-2 | 20 | **`write_path:` scope** | Added so users can pre-approve file writes under specific path patterns (e.g., `write_path:docs/**`) without needing to grant all Writes. Validation uses Python `fnmatch` at runtime; glob semantics are a Cedar-`like` superset (§6.4, §5.5). | | 21 | **`tool_group:file_write` convenience scope** | Resolves to `{Write, Edit}`. Prevents the surprise of pre-approving `Write` and still getting gated on `Edit`. | | 22 | **Pre-implementation spike: cedarpy annotation round-trip** | Day 1 of implementation validates that `policies_to_json_str()` returns annotations in the expected shape. If the API has changed, fall back to policy-ID prefix conventions. | -| 23 | **Cedar engine parity contract (Python `cedarpy` ↔ JS `cedar-wasm`)** | Both engines are pinned in `mise.toml`. A golden-file parity test runs in CI: for each `(policy, input)` fixture the test asserts Python and WASM return the same `decision` and the same set of matching rule IDs. Policy authors who upgrade either engine must refresh the golden file; drift fails the build. See §15.6 and Appendix B. | +| 23 | **Cedar engine parity contract (Python `cedarpy` ↔ JS `cedar-wasm`)** | Both engines are pinned in their package manifests. A golden-file parity test runs in CI: for each `(policy, input)` fixture the test asserts Python and WASM return the same `decision` and the same set of matching rule IDs. Policy authors who upgrade either engine must refresh the golden file; drift fails the build. See §15.6 and Appendix B. | --- @@ -176,7 +178,7 @@ Narrative walk-through of the happy path. Sequence diagrams in the round-trip Me - rejects blueprint whose combined `cedar_policies` text exceeds the 64 KB cap (§12.4) regardless of origin - resolves `approval_gate_cap` = `Blueprint.security.approvalGateCap ?? 50`; rejects if outside `[1, 500]` (decision #13) 5. Task persists. `approval_timeout_s`, `approval_gate_cap`, and `initial_approvals` become DDB attributes on the task row (cap is captured at submit time so mid-task blueprint edits do not shift the cap beneath a running task). -6. Container spawns on Runtime-JWT. `PolicyEngine.__init__` loads: +6. Container starts with IAM runtime credentials. `PolicyEngine.__init__` loads: - `HARD_DENY_POLICIES` (built-in + repo blueprint's `security.cedarPolicies.hard`; blueprint `disable:` may suppress non-built-in rules only, §5.1, §15.4) - `SOFT_DENY_POLICIES` (built-in + repo blueprint's `security.cedarPolicies.soft`; blueprint `disable:` may suppress soft-deny rules freely) - Annotation lookup table: `{policy_id: {annotation: value}}` built from `cedarpy.policies_to_json_str()` once, cached for the task lifetime @@ -228,7 +230,7 @@ Narrative walk-through of the happy path. Sequence diagrams in the round-trip Me "status": "PENDING", "created_at": "2026-04-23T14:00:00Z", "timeout_s": 300, - "ttl": 1734567890, # created_at + timeout_s + CLEANUP_MARGIN_120S; always covers the decision window + "expires_at": "2026-04-23T14:05:00Z", # explicit decision deadline; no retention TTL "user_id": "...", "repo": "my-org/my-app" } @@ -238,7 +240,7 @@ Narrative walk-through of the happy path. Sequence diagrams in the round-trip Me worker-lease condition for MicroVM): - Put on `TaskApprovalsTable` (new row with status=PENDING) - ConditionalUpdate on `TaskTable`: `status = :awaiting, awaiting_approval_request_id = :rid WHERE status = :running` - Both succeed or both fail. On `TransactionCanceledException` (most likely the TaskTable condition fails because another process moved the status), the hook emits `approval_write_failed` and returns DENY. + Both succeed or both fail. If the service rejects a task/lease conflict or the request fails, the hook emits `approval_write_failed` and returns DENY. 16. Hook emits `agent_milestone("approval_requested", {...})` to both `ProgressWriter` (DDB audit) and `sse_adapter` (live stream). Best-effort emission — transactional write has already committed; milestone failure is observability degradation, not state degradation. 17. Terminal A stream renders: ``` @@ -249,37 +251,12 @@ Narrative walk-through of the happy path. Sequence diagrams in the round-trip Me timeout 300s ``` Severity colors the line (respecting `NO_COLOR` env var). -18. Hook enters poll loop with strongly-consistent reads. The deadline is captured with the original approval row, before database writes/notifications: UTC expiry is `created_at + timeout_s`, capped by the original monotonic remaining duration. `deadline.remaining_s()` takes the smaller remainder and clamps at zero, so a frozen guest clock or backward UTC correction cannot restart the window. - ```python - async def _poll_for_decision(task_id, request_id, deadline): - start = time.monotonic() - interval = 2 - consecutive_failures = 0 - while True: - elapsed = time.monotonic() - start - if deadline.remaining_s() <= 0: - return TimedOut() - if elapsed > 30: - interval = 5 # backoff - try: - row = await _ddb_get_approval(task_id, request_id, ConsistentRead=True) - consecutive_failures = 0 - if row is not None and row["status"] in ("APPROVED", "DENIED"): - return Decided(row) - except Exception as exc: - consecutive_failures += 1 - if consecutive_failures == 3: - log("WARN", f"approval poll degraded for {request_id}: {exc}") - emit_milestone("approval_poll_degraded", {...}) - if consecutive_failures >= 10: - return TimedOut(reason="approval poll consecutive failures") - # Missing rows keep waiting within the original deadline; never allow. - remaining = deadline.remaining_s() - if remaining <= 0: - return TimedOut() - await asyncio.sleep(min(interval, remaining)) - ``` -19. The local-timeout path attempts to write the row to TIMED_OUT (best-effort conditional update `status = :pending`) before returning. If that write loses or fails, the hook rereads consistently to honor an already-committed decision. The approval-cap check runs before row creation and has no row to update. +18. Hook enters the poll loop with strongly-consistent reads. For an explicit positive timeout, the deadline is captured with the original approval row, before database writes/notifications: UTC expiry is `created_at + timeout_s`, capped by the original monotonic remaining duration. `deadline.remaining_s()` takes the smaller remainder and clamps at zero, so a frozen guest clock or backward UTC correction cannot restart the window. + An untimed request has no expiry. The executable poll loop in + [hooks.py](../../agent/src/hooks.py) also handles cancelled/closed requests, + read failures and continuation barriers; it must not be replaced by a loop + that checks only APPROVED/DENIED or starts a fresh timer after wake. +19. The local-timeout path asks the trusted approval service to conditionally mark the pending row TIMED_OUT before returning. If that write loses or fails, the hook rereads consistently to honor an already-committed decision. The approval-cap check runs before row creation and has no row to update. ### User responds @@ -340,7 +317,8 @@ sequenceDiagram Lambda-->>CLI: 202 APPROVED Hook->>Approvals: poll with ConsistentRead Approvals-->>Hook: status APPROVED - Hook->>Approvals: TransactWriteItems, TaskTable to RUNNING + Hook->>Approvals: ConditionCheck on recorded decision + Note right of Hook: Same transaction updates TaskTable to RUNNING;
worker does not modify the approval row Hook->>Engine: allowlist.add(scope) if scope is not this_call Hook-->>Agent: permissionDecision allow Note over Stop: (not used on approval path) @@ -693,35 +671,12 @@ The recent-decision cache is a simple `dict[(tool_name, input_sha), (decision, r When multiple soft-deny rules match a single tool call: -```python -def _merge_annotations(self, policy_ids: list[str]) -> dict: - rule_ids, timeouts, severities = [], [], [] - for pid in policy_ids: - ann = self._annotations[pid] - rule_ids.append(ann.get("rule_id", pid)) - if "approval_timeout_s" in ann: - try: - t = int(ann["approval_timeout_s"]) - if t >= FLOOR_30S: - timeouts.append(t) - except ValueError: - log("WARN", f"malformed @approval_timeout_s on {ann.get('rule_id', pid)}") - severities.append(ann.get("severity", "medium")) - - # Task default always eligible - timeouts.append(self._task_default_timeout_s) - - raw_min_timeout = min(timeouts) - return { - "rule_ids": rule_ids, - "timeout_s": max(FLOOR_30S, raw_min_timeout), # floor enforcement - "severity": _max_severity(severities), # "high" > "medium" > "low" - } -``` +The implementation is `_merge_annotations` in [policy.py](../../agent/src/policy.py). +It takes the shortest **positive** matching-rule or task timeout; zero means no +deadline and is excluded from that minimum. If no positive timeout exists, the +result is zero. Positive results retain the 30-second floor. The highest matching +severity governs the displayed severity, and matching rule IDs are preserved. -**Rationale for min/max choices**: -- **Timeout → min (above floor)**: multiple rules matching means multiple concerns. Users should have *less* time to decide when stakes are higher. Floor prevents unusable 5s windows. -- **Severity → max**: the most severe concern governs the UX coloring. ### 6.4 Allowlist data structure @@ -788,160 +743,28 @@ class ApprovalAllowlist: PreToolUse hook (compressed for doc; implementation will be richer): -```python -async def pre_tool_use_hook(hook_input, tool_use_id, ctx, *, - engine, task_id, user_id, progress, sse_adapter, - task_default_timeout_s): - tool_name, tool_input = _extract(hook_input) - decision = engine.evaluate_tool_use(tool_name, tool_input) - - if decision.outcome == Outcome.ALLOW: - return _allow() - if decision.outcome == Outcome.DENY: - return _deny(decision.reason) - - # REQUIRE_APPROVAL path. - # Cap + rate-limit check. Per-minute rate limit is per-container; on - # container restart the counter resets. The per-task approvalGateCap - # (blueprint-configurable, default 50) is persisted and bounds cumulative - # damage across restarts (§13.6). - if engine.approval_gate_count >= engine.approval_gate_cap: - return _deny(f"approval-gate cap exceeded ({engine.approval_gate_cap}/task)") - if engine.approvals_in_last_minute >= APPROVAL_RATE_LIMIT: - return _deny("approval-gate rate limit exceeded (20/min)") - - # Compute effective timeout with floor/ceiling. - remaining = _remaining_maxlifetime_s() - effective_timeout = max( - FLOOR_30S, - min(decision.timeout_s or task_default_timeout_s, - task_default_timeout_s, - remaining - CLEANUP_MARGIN_120S), - ) - if remaining - CLEANUP_MARGIN_120S < FLOOR_30S: - return _deny(f"insufficient maxLifetime remaining ({remaining}s) for approval") - - request_id = _ulid() - engine.approval_gate_count += 1 - - row = { - "task_id": task_id, "request_id": request_id, - "tool_name": tool_name, - "tool_input_preview": _strip_ansi(_preview(tool_input))[:256], - "tool_input_sha256": _sha256(_serialize(tool_input)), - "reason": decision.reason, "severity": decision.severity, - "matching_rule_ids": list(decision.matching_rule_ids), - "status": "PENDING", - "created_at": _iso_now(), - "timeout_s": effective_timeout, - "ttl": int(time.time()) + effective_timeout + CLEANUP_MARGIN_120S, - "user_id": user_id, "repo": engine.repo, - } - deadline = _ApprovalDeadline.from_recorded(row["created_at"], effective_timeout) - - # ATOMIC: put approval row + transition TaskTable status in one transaction. - try: - await _transact_write_approval_request(task_id, request_id, row) - except TransactionCanceledException as exc: - # Either the task was concurrently cancelled, or status wasn't RUNNING. - _emit("approval_write_failed", {"request_id": request_id, "reason": str(exc)}) - return _deny("approval system unavailable") - - _emit("approval_requested", { - "request_id": request_id, "tool_name": tool_name, - "input_preview": row["tool_input_preview"], - "reason": decision.reason, "severity": decision.severity, - "timeout_s": effective_timeout, - "matching_rule_ids": list(decision.matching_rule_ids), - }) - - outcome = await _poll_for_decision(task_id, request_id, deadline) - - # On TIMED_OUT, attempt to write the row to TIMED_OUT so future reads see - # a terminal state (not orphaned PENDING). The conditional write is guarded - # by `status = :pending` — if the user's APPROVE landed between our last - # poll and this write, the condition fails. In that case we MUST re-read - # the row and honor whatever terminal state won the race; otherwise local - # `outcome.status = "TIMED_OUT"` is stale and we would deny a call the user - # just approved ("I approved it" → agent denies). See §13.12 and the - # scenario below. - if outcome.status == "TIMED_OUT": - wrote_timeout = await _best_effort_update_status( - task_id, request_id, "TIMED_OUT", - reason=outcome.reason, - # Returns True on successful write, False on ConditionCheckFailed. - ) - if not wrote_timeout: - # Re-read the row with ConsistentRead — user's decision beat us. - row = await _ddb_get_approval(task_id, request_id, ConsistentRead=True) - if row is not None and row["status"] == "APPROVED": - # Late-approve wins. Honor it. Rebuild the outcome so the - # downstream allow flow (scope propagation, milestone emission, - # resume transaction) runs identically to the normal approve - # path. - outcome = Decided( - status="APPROVED", - scope=row.get("scope"), - decided_by=row.get("user_id"), - decided_at=row.get("decided_at"), - ) - _emit("approval_late_win", { - "request_id": request_id, - "outcome": "APPROVED", - "reason": "user decision landed during TIMED_OUT write", - }) - elif row is not None and row["status"] == "DENIED": - outcome = Decided( - status="DENIED", - reason=row.get("deny_reason") or "denied", - decided_at=row.get("decided_at"), - ) - # If status is still PENDING (rare — concurrent reaper race) or - # the row is gone (TTL reaped before we could read it), fall - # through with the original TIMED_OUT outcome; fail-closed deny. - - # ATOMIC: resume TaskTable status RUNNING, conditional on awaiting_approval_request_id matching. - try: - await _transact_resume(task_id, request_id) - except TransactionCanceledException: - # User cancelled (or some other path) during poll; abandon gracefully. - _emit("approval_resume_failed", {"request_id": request_id}) - return _deny("task no longer awaiting approval") - - if outcome.status == "APPROVED": - if outcome.scope and outcome.scope != "this_call": - engine._allowlist.add(outcome.scope) - _emit("approval_granted", {"request_id": request_id, - "scope": outcome.scope or "this_call", - "decided_at": outcome.decided_at}) - return _allow() - - # DENIED or TIMED_OUT — cache for 60s + queue denial injection. - engine._recent_decisions.record( - tool_name, _sha256(_serialize(tool_input)), - decision="DENIED" if outcome.status == "DENIED" else "TIMED_OUT", - reason=outcome.reason, - ) - # Truncated reason for guaranteed-surface permissionDecisionReason. - # Best-effort richer injection via _denial_between_turns_hook; may be - # pre-empted by _cancel_between_turns_hook on a concurrently-cancelled - # task. See §4 "Denial with steering text" scenario. - permission_decision_reason = _truncate( - outcome.reason or f"User {outcome.status.lower()}", max_len=500 - ) - if outcome.status == "DENIED": - # Queue steering injection via Stop hook's between_turns_hooks. - engine._queue_denial_injection( - request_id=request_id, - reason=outcome.reason, # already sanitized by DenyTaskFn - decided_at=outcome.decided_at, - ) - _emit("approval_denied" if outcome.status == "DENIED" else "approval_timed_out", - {"request_id": request_id, "reason": outcome.reason}) - return _deny(permission_decision_reason) -``` - -`engine._queue_denial_injection` appends to a list consumed by `_denial_between_turns_hook` — registered **after** `_nudge_between_turns_hook` in the `between_turns_hooks` list (which itself runs after `_cancel_between_turns_hook`). At the next Stop hook fire, the denial is emitted as `…` XML (sanitized via `_xml_escape` from the shared utility introduced with Phase 2). If a `bgagent cancel` has landed between the deny and the next Stop seam, `_cancel_between_turns_hook` short-circuits the dispatcher and the denial text is NOT injected — in which case the guaranteed surface is `permissionDecisionReason` on the hook return. See finding #2 scenario in §4 for the cancel-vs-deny race reasoning. +The executable flow lives in [hooks.py](../../agent/src/hooks.py), with signed +request creation/closure in [approval_requests.py](../../agent/src/approval_requests.py) +and task-state transactions in [task_state.py](../../agent/src/task_state.py). + +1. Evaluate policy, existing grants, gate cap and creation-rate limit. +2. Resolve the shortest positive task/rule deadline. Zero remains untimed. + For a positive deadline, apply a remaining-worker-lifetime ceiling when one + is supplied. A continuation runtime separates this deadline from worker life. +3. Ask the trusted service to atomically create the pending request and move the + task to `AWAITING_APPROVAL`; MicroVM requests also check the active worker lease. + Pending rows have no storage TTL. A failed write denies the action. +4. Emit the notification milestone and poll with strongly consistent reads, + preserving the original UTC/monotonic deadline. No deadline means no timer expiry. +5. If polling times out or fails, ask the trusted service to close the pending + request. If a human decision won the race, reread and preserve that decision. +6. Resume only the matching active task/gate and worker lease. Approval allows the + action and applicable scope; denial is returned to the tool hook and queued as + best-effort steering for the next Stop hook. Cancellation prevents resumption. + +The worker's `TIMED_OUT` outcome currently also covers polling failures; it is +not proof that an explicit human deadline elapsed. This limitation is recorded in +[ADR-023](../decisions/ADR-023-trusted-approval-writer.md). **Scenario (§13.12 VM-throttle + late-approval race).** Alice hits a gate with a 300-second window and commits APPROVED at t=294.7s. Scheduling delays prevent the next agent poll until t=300.2s. That poll sees the original deadline has passed and returns TIMED_OUT before reading the row. The conditional TIMED_OUT write then loses because APPROVED is already stored. The hook rereads with `ConsistentRead`, preserves Alice's scope and decision metadata, emits `approval_late_win`, and proceeds through the guarded resume transaction and allow flow. Without that reread it would deny an already-approved call. See IMPL-24, §13.12, and §15.2 task #43 for the race test. @@ -1029,9 +852,9 @@ On `TransactionCanceledException`, `ApproveTaskFn` inspects per-item This is symmetric with the agent-side `TransactWriteItems` pattern (§4 step 25a) used for the resume transition — Lambdas and agent speak the same atomic-update contract. -**Ownership**: `user_id` stored on TaskApprovalsTable and compared against `caller_user_id` in the ConditionExpression is the Cognito `sub` claim **verbatim**. The Lambda extracts `sub` from the validated JWT and uses it as-is: no prefix stripping, no tenant mapping, no format normalization. If we ever introduce per-tenant user ID namespacing, that transformation MUST happen at the **write** path (i.e. before the agent writes the row in §4 step 14) rather than at compare time, so the ConditionExpression always compares identical-shape identifiers. See finding #6 scenario below. +**Ownership**: `user_id` stored on TaskApprovalsTable and compared against `caller_user_id` in the ConditionExpression is the Cognito `sub` claim **verbatim**. The Lambda extracts `sub` from the validated JWT and uses it as-is: no prefix stripping, no tenant mapping, no format normalization. If we ever introduce per-tenant user ID namespacing, that transformation MUST happen at the **write** path (i.e. before the approval service persists the row prepared in §4 step 14) rather than at compare time, so the ConditionExpression always compares identical-shape identifiers. See finding #6 scenario below. -After successful transaction, `ApproveTaskFn` writes an audit event to `TaskEventsTable` (`approval_decision_recorded` event_type), ensuring the 90-day audit trail is owned by the Lambda path — not dependent on the agent's milestone emission. +After a successful transaction, the decision handler attempts the authoritative `approval_decision_recorded` audit event independently of the agent milestone. Audit-delivery failure does not undo the saved decision. **Scenario (finding #6):** Three months from now, a platform engineer adds a multi-tenant mode where Cognito `sub` becomes `tenant-abc:01JXZ...`. They update the agent's row-write path to prefix-strip: `user_id = sub.split(":", 1)[1]`, storing `01JXZ...` on TaskApprovalsTable. They forget to update `ApproveTaskFn`. Now the Lambda reads `sub = "tenant-abc:01JXZ..."` from the JWT and compares it against the stored `01JXZ...` — condition fails, 404 on every approve, all tasks stranded. The fix as written: "the Cognito sub is compared verbatim; any transformation must happen at write time, not at compare time" — if the agent writes the full `sub`, the Lambda compares the full `sub`; if either side transforms, both sides must. The CI assertion is a unit test that extracts `user_id` from a sample row and asserts it matches the `sub` claim of a sample JWT byte-for-byte. This test would fail on the prefix-strip refactor above and force the engineer to update both sides. Without this hard rule, ownership-in-condition silently breaks under any future identity refactor. @@ -1332,8 +1155,8 @@ stateDiagram-v2 ``` `STRANDED` remains a recognized row status in the type contract. The current -reconciler does not write it: it fails the owning task and leaves the row -`PENDING`, as described below. +reconciler does not write it: it fails the owning task and closes pending requests +as `CANCELLED`, as described below. ### 9.3 Orchestrator impact @@ -1398,31 +1221,11 @@ The design assumes a human is watching. For truly unattended tasks (scheduled au ### 10.1 New DynamoDB table: `TaskApprovalsTable` -```typescript -new dynamodb.Table(this, 'Table', { - partitionKey: { name: 'task_id', type: dynamodb.AttributeType.STRING }, - sortKey: { name: 'request_id', type: dynamodb.AttributeType.STRING }, // ULID - billingMode: dynamodb.BillingMode.PAY_PER_REQUEST, - pointInTimeRecovery: true, - timeToLiveAttribute: 'ttl', - stream: dynamodb.StreamViewType.NEW_AND_OLD_IMAGES, // (evaluated — may drop; see §11) - removalPolicy: RemovalPolicy.RETAIN, -}); - -// v1 GSI — backs `GET /v1/pending` and `bgagent pending`. -// Required at v1 ship, not deferred — see finding #8 scenario in §7.7. -table.addGlobalSecondaryIndex({ - indexName: 'user_id-status-index', - partitionKey: { name: 'user_id', type: dynamodb.AttributeType.STRING }, - sortKey: { name: 'status', type: dynamodb.AttributeType.STRING }, - projectionType: dynamodb.ProjectionType.INCLUDE, - nonKeyAttributes: [ - 'task_id', 'request_id', 'tool_name', 'tool_input_preview', - 'severity', 'reason', 'created_at', 'timeout_s', - 'matching_rule_ids', - ], -}); -``` +[TaskApprovalsTable](../../cdk/src/constructs/task-approvals-table.ts) defines the +`task_id` partition key, `request_id` sort key, optional retention TTL attribute +`ttl`, and `user_id-status-index` GSI. Streams are disabled; TaskEventsTable carries +the audit/fan-out stream. The GSI discovers candidates; the pending endpoint then +strongly reads the approval and owning task before returning a request. **Projection is fixed at design time.** DynamoDB rejects in-place updates to a GSI's `nonKeyAttributes` (the CloudFormation error @@ -1452,11 +1255,12 @@ Attributes: | `matching_rule_ids` | L | Yes | List (not Set — can be empty) of soft-deny rule IDs | | `status` | S | Yes | PENDING \| APPROVED \| DENIED \| TIMED_OUT \| STRANDED \| CANCELLED | | `created_at` | S | Yes | ISO8601 | -| `decided_at` | S | No | Set by approve/deny/cancel; the current guest timeout writer can omit it | +| `decided_at` | S | No | Set when the trusted service or platform closes a request | | `scope` | S | No | Set on APPROVED | | `deny_reason` | S | No | Set on DENIED; sanitized user text | | `timeout_s` | N | Yes | Resolved timeout for audit | -| `ttl` | N | Yes | `created_at_epoch + timeout_s + CLEANUP_MARGIN_120S` — always covers the decision window | +| `expires_at` | S/null | Yes | Original decision deadline; null when `timeout_s=0` | +| `ttl` | N | No | Retention cleanup set on task closure; absent while pending | | `user_id` | S | Yes | Cognito `sub` **verbatim**; used in ownership check `ConditionExpression` (§7.1 finding #6) | | `repo` | S | Yes | Denormalized for fan-out | @@ -1539,7 +1343,7 @@ and the Slack OAuth/button design below remain proposed. Email remains a log-onl receive approval messages. Deployment status is recorded in the [P3 verification record](../verification/README.md). -**TaskApprovalsTable Streams are not consumed by the fan-out Lambda**. The approval row is working state; the audit trail is in TaskEventsTable. Enabling Streams on TaskApprovalsTable would be redundant and add noise. Final design: TaskApprovalsTable DOES NOT have Streams enabled. (Retains the `stream` attribute commented out for future use if needed.) +**TaskApprovalsTable Streams are not consumed by the fan-out Lambda**. The approval row is working state; the audit trail is in TaskEventsTable. Enabling Streams on TaskApprovalsTable would be redundant and add noise. Final design: TaskApprovalsTable DOES NOT have Streams enabled. Slack and Linear route `approval_requested`, `approval_decision_recorded`, `approval_timed_out`, `approval_cancelled` and `approval_stranded`. The dispatcher @@ -1728,7 +1532,7 @@ These conditions prevent races; they do not constrain a compromised Lambda with raw table-write permission, which could omit them. Only trusted control-plane handlers receive that permission. -`user_id` comparison is against Cognito `sub` **verbatim** — byte-for-byte equality. Any future identity transformation (per-tenant prefixing, namespacing) must apply to BOTH the write path (agent-side row write) AND the compare path (Lambda ConditionExpression) simultaneously, or the comparison silently fails under the new format. A unit test (§15.3) enforces this: given a sample JWT, extract `sub`, write a row, then assert the stored `user_id` equals `sub` byte-for-byte. +`user_id` comparison is against Cognito `sub` **verbatim** — byte-for-byte equality. Any future identity transformation (per-tenant prefixing, namespacing) must apply to BOTH the write path (trusted approval service) AND the compare path (Lambda ConditionExpression) simultaneously, or the comparison silently fails under the new format. A unit test (§15.3) enforces this: given a sample JWT, extract `sub`, write a row, then assert the stored `user_id` equals `sub` byte-for-byte. ### 12.3 Race prevention @@ -1909,34 +1713,38 @@ Addressed by the parity contract (decision #23, §15.6). Golden-file CI test run ### 13.12 VM-throttle + late-approval race -The agent's poll loop retains the original approval row's UTC expiry and a monotonic cap, using whichever expires first. Slow writes/notifications count toward that same window. This also covers a MicroVM whose monotonic clock stops while suspended; the future `/resume` hook must reuse this deadline and wake the decision loop. If CPU throttling or suspension delays polling beyond expiry, a user's APPROVE transaction may already have committed. The agent then attempts `status = TIMED_OUT WHERE status = :pending`. The condition fails because APPROVED already won. Without a re-read, the agent's local state would remain TIMED_OUT while DDB holds APPROVED, and it would incorrectly deny the approved call. +The agent's poll loop retains the original approval row's UTC expiry and a monotonic cap, using whichever expires first. Slow writes/notifications count toward that same window. This also covers a MicroVM whose monotonic clock stops while suspended; the `/resume` hook reuses this deadline and wakes the decision loop. If CPU throttling or suspension delays polling beyond expiry, a user's APPROVE transaction may already have committed. The agent asks the trusted service to conditionally record `TIMED_OUT` while the row remains `PENDING`. The condition fails because APPROVED already won. Without a re-read, the agent's local state would remain TIMED_OUT while DDB holds APPROVED, and it would incorrectly deny the approved call. -**Mitigation**: the §6.5 pseudocode re-reads the approval row with `ConsistentRead=True` whenever `_best_effort_update_status("TIMED_OUT", ...)` returns ConditionCheckFailed, and honors whatever terminal state the row carries: +**Mitigation**: the hook rereads the approval row with `ConsistentRead=True` when the service reports that another decision won the timeout race, and honors the recorded decision: - If `status == "APPROVED"`: rebuild the local `outcome` to reflect APPROVED, preserving `scope`, `decided_by`, `decided_at`, and proceed through the normal allow flow (scope-propagation, `approval_granted` milestone, resume transaction, return `{"permissionDecision": "allow"}`). Emit a `approval_late_win` milestone so operator telemetry can count races. - If `status == "DENIED"`: honor the denial text the user submitted. Agent returns DENY with the user's sanitized reason as `permissionDecisionReason` (same surface as normal deny). -- If `status` is still PENDING (rare — concurrent reaper race) or the row is gone (TTL reaped): fall through with the original TIMED_OUT outcome; fail-closed deny. +- If `status` is still PENDING (rare — concurrent reaper race) or the row is missing: fall through with the original TIMED_OUT outcome; fail-closed deny. The scenario is bounded by the polling cadence (2-5s ticks) and DDB's strongly-consistent read latency (tens of ms), so the re-read adds at most one extra GetItem to the racing path — acceptable cost for honoring user intent. See IMPL-24 and §15.2 task #43 (race tests). ### 13.13 Runtime JWT expiry during approval wait -**Context in this codebase (verified 2026-05-06).** The AgentCore Runtime container authenticates outbound AWS API calls (DynamoDB, Secrets Manager, etc.) via the container's IAM role, which the SDK resolves through the instance-metadata-service equivalent and auto-refreshes transparently. There is no user-presented JWT with a short rolling expiry consumed by the container's own API calls — `grep -rn -iE 'runtime.jwt|jwt.refresh|token_expiry' agent/src/` returns nothing (only `token_usage` for LLM billing and `GITHUB_TOKEN` for git operations). AgentCore Runtime invocation on the Lambda side uses sigv4 via `InvokeAgentRuntimeCommand` (see `cdk/src/handlers/shared/strategies/agentcore-strategy.ts`) — also auto-refreshed AWS credentials, not a user JWT. The "Runtime-JWT" label in §4 step 6 and the sequence diagrams below refers to the **caller-facing SSE auth** (Terminal A's Cognito ID token presented to API Gateway to stream task events) — it does not authenticate the container's own DDB writes. - -**Therefore, for v1: no separate Runtime JWT expiry term is required in the ceiling computation.** The `maxLifetime` term (AgentCore's hard lifetime of 8h) is the only upper bound we control; IAM credentials refresh automatically within that window. The ceiling definition in decision #6 stands as `min(1h, maxLifetime_remaining - cleanup_margin)`. +Workers authenticate AWS calls with IAM role credentials, not the user's Cognito +JWT. The CLI's Cognito token protects its platform API calls; an expired CLI login +requires reauthentication but does not itself delete the saved approval request. +MicroVM resume refreshes runtime and task-role credentials before releasing coding. -**If the auth model changes** (e.g. a future design introduces a container-held user JWT to authenticate `permissionDecisionReason` attribution, or to carry the caller's Cognito `sub` end-to-end for per-user DDB conditions), the ceiling MUST be extended to `min(1h, maxLifetime_remaining - 120s, runtime_jwt_expiry - 120s)` and this section updated. Tracked as IMPL-27 so the contract is reviewed whenever the auth shape changes. The failure signature if this bound is missed: the container's IAM calls succeed but some JWT-gated channel (e.g. Terminal A's SSE stream) quietly 403s mid-approval-wait; the user's decision lands in DDB but the agent's poll fails to deliver `approval_granted` to the live stream. Today that channel is best-effort observability, not state — but a future state-bearing channel would need the ceiling term. +Credential renewal does not extend a worker's service lifetime. Apply the worker +and explicit-deadline rules in §9.5; do not impose a new approval deadline based on +the user's API token expiry. A future worker-held user-token design would need a +separate review of that boundary. ### 13.14 Notification delivery failure -Fan-out delivery failures (Slack down, email bounce, webhook 5xx) do **NOT** pause the approval timer. The timer runs on the agent's local clock, keyed to the `created_at` timestamp on the DDB row — it is independent of whether any notification channel succeeded in alerting the human. +A failed notification does not change the recorded decision deadline. Timed +requests retain their original UTC/monotonic deadline; untimed requests remain +available until task closure or another valid resolution. Neither case permits +the action without approval. -**Rationale (security):** coupling the timer to notification-plane availability creates a bypass. An adversary who takes down the webhook (or poisons the Slack rate limit) would get an unbounded approval window; worse, a compromised tenant could deliberately suppress their own notifications to escape gates. Fail-closed on timer expiry is invariant; delivery is best-effort observability. +Users can discover unanswered requests with `bgagent pending`, which reads through +the authenticated API independently of notification delivery. A late-discovered +explicitly timed request does not receive a fresh decision window. -**Recovery path for the user:** `bgagent pending` queries `TaskApprovalsTable` directly via the `user_id-status-index` GSI (§7.7, §10.1) — it does not depend on notification delivery. A user who suspects notifications are broken can poll `bgagent pending` at any time to see all live approvals. If the notification never landed and the user finds a gate via `bgagent pending`, they can `bgagent approve/deny` normally; the timer is still running against the original `created_at`, not against when the user found it. - -**Operational signal:** `approval_timed_out` events carry `timeout_s` and the `created_at`/`decided_at` delta. A rising `approval_timed_out` rate with flat `approval_requested` rate (measured via `ApprovalTimeoutClipRate` and `ApprovalDecisionLatency` in §11.3) is the telemetry that indicates notification breakage, not an unresponsive user. - -**The fail-closed posture on timer expiry remains unchanged.** Delivery-availability-aware scheduling is the notification plane's job (see §14.8 and INTERACTIVE_AGENTS.md notification-plane design); the timer does not reason about it. ### 13.15 Fail-closed summary @@ -1997,7 +1805,7 @@ BLOCKED[]: (resource: ) # when a resource is named ### 14.1 Scenario A: force-push with per-rule timeout -Setup: repo `my-org/my-app` blueprint extends soft-deny with `force_push_main` (@approval_timeout_s=600). Task default is 300s. +Setup: repo `my-org/my-app` blueprint extends soft-deny with `force_push_main` (@approval_timeout_s=600). This example explicitly sets the task timeout to 300s. ```bash $ bgagent run --repo my-org/my-app \ @@ -2115,7 +1923,7 @@ Each phase has explicit scope. Matches real-world review workflows. Visible in a ### 14.6 Scenario F: VM-throttle + late-approval race (trace) -Setup: task default 300s; force-push gate fires. User Alice approves at the very edge of the timeout window while the VM is throttled. +Setup: task explicitly configured with a 300-second timeout; force-push gate fires. User Alice approves at the very edge of the timeout window while the VM is throttled. ``` t=0.00s PreToolUse hook fires: Bash "git push --force origin feature-x" @@ -2223,7 +2031,7 @@ See §17.18 for the off-hours escalation future-work primitive, and §13.14 for | # | Package | File | Change | |---|---|---|---| | 1 | agent | Spike | Validate cedarpy.policies_to_json_str() returns annotations. Confirm `diagnostics.reasons` shape for multi-match. If API diverges, update §6 before proceeding. | -| 2 | mise + agent + cdk | `mise.toml`, `agent/pyproject.toml`, `cdk/package.json` | Pin `cedarpy==4.8.0` (agent) and `@cedar-policy/cedar-wasm==4.10.0` (cdk). The two bindings are intentionally on different version lines — verified compatible via the parity fixtures, not required to be equal. Both pinned exactly, not `^` or `~` — decision #23 / finding #1. | +| 2 | mise + agent + cdk | `mise.toml`, `agent/pyproject.toml`, `cdk/package.json` | Pin both Cedar bindings exactly in their package manifests and verify the pair through the shared parity fixtures; see §15.6. | | 3 | agent + cdk | `contracts/cedar-parity/*.json` (shared fixture dir; follows precedent set by `contracts/memory-hash-vectors.json`) | Golden-file parity fixtures: `(policy_set, input) → {decision, matching_rule_ids}`. Agent side loads via `cedarpy`; Lambda side via `cedar-wasm`. Divergence fails CI. | | 4 | agent | `src/policy.py` | Extend `PolicyDecision` (outcome/timeout_s/severity/matching_rule_ids/allowed-property). Split `_DEFAULT_POLICIES` into hard + soft. Add annotation parsing. Implement `ApprovalAllowlist` + `RecentDecisionCache` (50-entry LRU cap, independent of `approvalGateCap`). Load-time validation (rule_id uniqueness, tier mismatch, annotation floor, 64 KB cap, disable-list hard-deny rejection, `approvalGateCap` bounds check `1 ≤ N ≤ 500`). `PolicyEngine.__init__` accepts `approval_gate_cap` sourced from blueprint (default 50). | | 5 | agent | `policies/hard_deny.cedar` (new) | Migrate current hard-deny rules + add DROP TABLE. Annotations. | @@ -2366,12 +2174,13 @@ Rollout steps: ### 15.6 Shared Cedar parsing — cross-engine parity contract -The agent runtime uses Python [`cedarpy@4.8.0`](https://pypi.org/project/cedarpy/); the Lambda side (`CreateTaskFn`, `ApproveTaskFn`, `DenyTaskFn`, `GetPoliciesFn`) uses [`@cedar-policy/cedar-wasm@4.10.0`](https://www.npmjs.com/package/@cedar-policy/cedar-wasm) — AWS's official WASM-compiled Cedar engine. Same Rust core, two bindings. Because these engines evolve independently, we ship a **parity contract** (decision #23, finding #1) to catch drift before deploy. - -**Version pinning.** Both engines are pinned exactly (not `^` or `~`) in the monorepo's canonical manifest files. The two bindings are deliberately on **different version lines** — they are NOT required to be equal. `cedarpy` and `cedar-wasm` follow independent release cadences over the shared Cedar Rust core, and the currently-shipped pins (`cedarpy==4.8.0` ↔ `@cedar-policy/cedar-wasm==4.10.0`) are an intentional, tested-compatible skew: the parity fixtures in `contracts/cedar-parity/` are what certify that this specific pair produces identical `(decision, matching_rule_ids)` on every fixture. The rule is "move together and re-verify parity when you bump either side," not "keep the version strings equal." -- `agent/pyproject.toml`: `cedarpy==4.8.0` -- `cdk/package.json`: `"@cedar-policy/cedar-wasm": "4.10.0"` -- `mise.toml` documents the pinned versions in a comment for operator visibility +The agent uses Python `cedarpy`; the policy Lambdas use +`@cedar-policy/cedar-wasm`. Exact current pins live in +[agent/pyproject.toml](../../agent/pyproject.toml) and +[cdk/package.json](../../cdk/package.json). Binding version strings need not be +identical. Upgrade them as a tested pair and run the shared +[parity fixtures](../../contracts/cedar-parity/README.md), which check decisions +and matching rule IDs across both engines. **Lambda layer packaging** (finding #5). The cedar-wasm package is 4.1 MB unzipped. Shipping it in the deployment bundle of each of the 4 policy Lambdas would consume ~16 MB of unzipped bundle size — manageable on its own but leaves little room for AWS SDK + other deps as the codebase grows, and threatens the Lambda 250 MB unzipped limit under realistic growth. Solution: package cedar-wasm as a **Lambda layer** (`cedar-wasm-layer.ts`, task #10 in §15.2), attached to each policy Lambda. This reduces each Lambda's deployment bundle to just the handler code + thin wrapper around the layer import. Policy Lambdas are configured with ≥ 512 MB memory to accommodate WASM module instantiation under concurrent invocation (measured under 100-concurrent bursts in §15.3 Lambda memory tests). @@ -2418,7 +2227,7 @@ flowchart LR When policy authors upgrade either engine, the parity fixture must be re-generated (a small helper script dumps decisions from both engines; the human confirms the change is intentional). -**Scenario (finding #1, illustrative):** This example uses hypothetical versions (e.g. cedarpy `4.10.1` → `4.11.0`) to show the *class* of bug the parity contract catches; it does not describe the real shipped pins (which are the intentional `cedarpy==4.8.0` ↔ `cedar-wasm==4.10.0` skew documented above). The point is that closeness of version strings — even within the same minor line — is no guarantee of behavioral parity, which is exactly why the golden fixtures, not the version numbers, are the source of truth. A platform engineer runs `mise run deps:update` which bumps cedarpy from 4.10.1 to 4.11.0. They notice cedar-wasm is still 4.10.0 but assume it's fine because both say "4.x". Between these versions, cedarpy added support for a new `context has` operator that cedar-wasm doesn't yet have. A new blueprint soft-deny rule uses `context has "approved_context"`. On deploy: +**Scenario (finding #1, illustrative):** This example uses hypothetical versions (e.g. cedarpy `4.10.1` → `4.11.0`) to show the *class* of bug the parity contract catches; it does not describe the real shipped pins (read the package manifests for the actual pins). The point is that closeness of version strings — even within the same minor line — is no guarantee of behavioral parity, which is exactly why the golden fixtures, not the version numbers, are the source of truth. A platform engineer runs `mise run deps:update` which bumps cedarpy from 4.10.1 to 4.11.0. They notice cedar-wasm is still 4.10.0 but assume it's fine because both say "4.x". Between these versions, cedarpy added support for a new `context has` operator that cedar-wasm doesn't yet have. A new blueprint soft-deny rule uses `context has "approved_context"`. On deploy: - Agent-side `PolicyEngine.__init__` parses the rule successfully; engine loads normally. - `CreateTaskFn` on the Lambda side calls cedar-wasm `policyToJson()` — it throws: `ParseError: unknown operator 'has' at line 3`. - User submits a task against that repo. `CreateTaskFn` crashes mid-validation. Error message: "500 Internal Server Error" (because the Lambda didn't handle the upstream parse error gracefully). @@ -2513,7 +2322,7 @@ Items from the design reviews not captured above as design changes — to be add **IMPL-9** (functional P1-3): Runtime allowlist revocation. Not shipped in v1. Placeholder: `bgagent revoke-approval ` noted in §17. -**IMPL-10** (functional P1-12): `approval_timeout_s` default 300 documented consistently in §3 #6, §7.3 table, §10.2 attribute description. +**IMPL-10** (historical functional P1-12): the original 300-second default was superseded by the September retained-request default of `0` (no decision deadline). **IMPL-11** (functional P2-8): CLI `run.ts` command exists from Phase 1b. `submit.ts` also exists. `--pre-approve` / `--approval-timeout` flags added to both. diff --git a/docs/design/INTERACTIVE_AGENTS.md b/docs/design/INTERACTIVE_AGENTS.md index c17955d97..fa6d98678 100644 --- a/docs/design/INTERACTIVE_AGENTS.md +++ b/docs/design/INTERACTIVE_AGENTS.md @@ -1,6 +1,8 @@ # Interactive Agents: Async Interaction Design -> **Status:** Active design +> **Status:** Historical interaction design with current approval-path corrections (2026-09-22). +> Proposed dispatcher services and Slack buttons below are not all implemented; use the +> [API contract](./API_CONTRACT.md) and [approval guide](../guides/USER_GUIDE.md#approval-gates-cedar-hitl) for supported interfaces. > **Branch:** `feature/interactive-background-agents` > **Last updated:** 2026-04-29 (rev 6) @@ -19,7 +21,7 @@ This document describes the interactivity surfaces layered on top of that model 3. **Watch** — `bgagent watch ` polls `TaskEventsTable` with an adaptive interval (500 ms when events are arriving, back-off to 5 s when idle). Same endpoint used under the hood for foreground-block UX on `ask` and for HITL approval waits. 4. **Nudge** — `bgagent nudge ""` writes a row into `TaskNudgesTable`. The agent reads pending nudges between turns, acknowledges with a `nudge_acknowledged` milestone event, and integrates the nudge on its next turn. 5. **Ask** — `bgagent ask ""` (Phase 2) writes a question row. The agent answers at the next between-turns boundary; the answer surfaces as a `status_response` event. CLI default is foreground block-and-poll with a spinner; task and answer are both durable if the CLI disconnects. -6. **Approval gates** — Phase 3 Cedar-driven hard gates. Agent emits `approval_requested`, waits for a decision from `bgagent approve` / `bgagent deny` or a Slack button-press. Detailed design in [`CEDAR_HITL_GATES.md`](./CEDAR_HITL_GATES.md). +6. **Approval gates** — Phase 3 Cedar-driven hard gates. Agent emits `approval_requested`, waits for a decision from `bgagent approve` / `bgagent deny` or an owner-authored Linear thread reply. Slack notifications provide CLI instructions; buttons remain proposed. Detailed design in [`CEDAR_HITL_GATES.md`](./CEDAR_HITL_GATES.md). ### Core architectural choices @@ -27,7 +29,7 @@ This document describes the interactivity surfaces layered on top of that model - **Durable event table (`TaskEventsTable`)** is the one source of truth for agent progress. Every reader — CLI, Slack/GitHub/email dispatchers, status Lambda — reads from this table, never from the live agent. - **Polling-only CLI.** No SSE, no WebSockets. DDB eventually-consistent reads with an `event_id` cursor are cheap, reliable, and compute-agnostic. - **Notification plane as first-class.** A FanOutConsumer Lambda subscribes to `TaskEventsTable` DDB Streams and routes per-event-type to per-channel dispatcher Lambdas (Slack, email, GitHub comment). Per-channel defaults ship in v1. -- **Agent interaction via the hook mechanism the Claude Agent SDK provides.** Nudges, asks, and approvals all use `Stop` / between-turns hooks; no mechanism outside the SDK's contract is required. +- **Agent interaction via the hook mechanism the Claude Agent SDK provides.** Nudges and asks use `Stop` / between-turns hooks; approval gates pause in `PreToolUse`, with denial steering delivered at a later Stop hook; no mechanism outside the SDK's contract is required. --- @@ -111,7 +113,7 @@ This document describes the interactivity surfaces layered on top of that model │ DELETE /tasks/{id} cancel │ │ POST /tasks/{id}/nudge nudge │ │ POST /tasks/{id}/asks ask (P2) │ - │ POST /tasks/{id}/approvals approve P3 │ + │ POST /tasks/{id}/approve approve │ │ POST /webhooks/tasks GH webhook │ └───────────┬──────────────────────────────────┘ │ @@ -223,17 +225,16 @@ Consumer: agent between-turns hook reads pending nudges, emits `nudge_acknowledg ### 3.7 TaskApprovalsTable (Phase 3) Phase 3 approval-request spine. Detailed schema in [`CEDAR_HITL_GATES.md`](./CEDAR_HITL_GATES.md). Semantics summary: -- Agent writes an approval row with the request context. -- Agent transitions `RUNNING → AWAITING_APPROVAL` and enters a poll loop. -- User responds via REST (`POST /tasks/{id}/approvals/{request_id}`) or via a Slack button dispatched by the notification plane. +- The worker calls the trusted approval service, which atomically creates the pending row and transitions the task `RUNNING → AWAITING_APPROVAL`. The worker then polls for a decision. +- The owner responds through `POST /tasks/{id}/approve` or `/deny` with `request_id` in the body, the equivalent CLI commands, or an `approve`/`deny` reply to the Linear approval comment. - On decision, agent transitions back to `RUNNING`; denial reasons are injected as Stop-hook steering on the next turn. ### 3.8 FanOutConsumer (router) Lambda subscribed to `TaskEventsTable` DDB Streams (relying on the DynamoDB Streams **default** `ParallelizationFactor` of 1, which preserves per-`task_id` ordering by shard — not set explicitly in `fanout-consumer.ts`; see §6.1). Reads per-task notification config (from `TaskTable` metadata or `RepoTable` defaults), filters events by channel subscription, and invokes per-channel dispatcher Lambdas. -- **SlackDispatchFn** — posts to configured channel / DM. Includes action buttons for `approval_required` events. -- **EmailDispatchFn** — SES. +- **SlackDispatchFn** — posts to configured channel / DM. Approval notifications currently contain CLI response instructions; buttons remain proposed. +- **EmailDispatchFn** — proposed SES delivery; current email dispatch is a log-only stub. - **GitHubDispatchFn** — edits a single GitHub issue comment in place via `PATCH /repos/{o}/{r}/issues/comments/{id}`. On 404 (comment deleted upstream) falls back to POSTing a fresh comment. Per-task ordering is guaranteed upstream by the DDB Streams default `ParallelizationFactor` of 1 (see §6.1), so no conditional-request header is needed (and GitHub's REST API does not accept `If-Match` on this endpoint — see §6.4). Detailed routing and default filters in §6. @@ -296,8 +297,8 @@ Authentication: Cognito User Pool ID token in `Authorization` header for all RES | `agent_milestone` | Agent code (pipeline, hooks) | Named checkpoint (`repo_cloned`, `pr_opened`, `nudge_acknowledged`, ...) | | `agent_cost_update` | Runner | Cumulative token + dollar cost | | `agent_error` | Runner | Handled exception | -| `approval_required` (P3) | PreToolUse Cedar hook | Cedar policy requires user decision | -| `approval_decided` (P3) | Approve/Deny Lambda | User responded | +| `approval_requested` (P3) | PreToolUse Cedar hook | Cedar policy requires user decision | +| `approval_decision_recorded` (P3) | Approve/Deny Lambda | User responded | | `status_response` (P2) | Between-turns hook | Agent answered an `ask` | | `nudge_acknowledged` | Between-turns hook | Agent saw a nudge before incorporating it | | `pr_created` | Pipeline | PR opened for the task | @@ -320,7 +321,7 @@ Consumers page `TaskEventsTable` using `event_id` as a cursor: `KeyConditionExpr ### 5.1 `bgagent submit` ``` -$ bgagent submit --repo org/repo "fix the auth timeout bug" +$ bgagent submit --repo org/repo --task "fix the auth timeout bug" task submitted: abc123 ``` @@ -411,9 +412,9 @@ Flags: HITL approval commands. All flows are REST + DDB; no streaming. Detailed design in [`CEDAR_HITL_GATES.md`](./CEDAR_HITL_GATES.md). Summary: -- Agent emits `approval_required` with the tool context. -- Notification plane dispatches the event (Slack with action buttons, email, GitHub). -- User responds via `bgagent approve `, `bgagent deny --reason "…"`, or Slack button click. +- Agent emits `approval_requested` with the tool context. +- The notification plane sends approval messages to Slack and Linear. Email is a stub; GitHub does not receive approval messages. +- The owner responds via `bgagent approve `, `bgagent deny --reason "…"`, or an `approve`/`deny` reply to the Linear approval comment. - Agent's poll loop sees the decision and proceeds or deny-steers. ### 5.7 `bgagent cancel` @@ -444,19 +445,22 @@ TaskEventsTable ──DDB Stream──▶ FanOutConsumer - Router reads per-task notification config (channel enablement + event-type filters), then invokes the relevant dispatcher Lambda(s) per event. - Dispatchers are separate Lambdas so a GitHub API outage doesn't block Slack notifications. -### 6.2 Per-channel defaults (v1) +### 6.2 Original proposed per-channel defaults (v1) | Channel | Default subscribed events | Opt-in via `--verbose` | |---|---|---| -| **Slack** | `task_completed`, `task_failed`, `task_cancelled`, `pr_created`, `agent_error`, `approval_required`, `status_response` | adds `agent_milestone` | -| **Email** | `task_completed`, `task_failed`, `approval_required` | — | +| **Slack** | `task_completed`, `task_failed`, `task_cancelled`, `pr_created`, `agent_error`, `approval_requested`, `status_response` | adds `agent_milestone` | +| **Email** | `task_completed`, `task_failed`, `approval_requested` | — | | **GitHub issue comment** | `pr_created`, terminal status (single edit-in-place comment) | — already minimal | Rationale: if Slack pings on every milestone, users mute the bot within days. Default to the minimal set that surfaces decision-requiring events and completion; power users opt into verbose streams. ### 6.3 Slack approval buttons -`approval_required` events delivered to Slack include `Approve` / `Deny` action buttons. On click, Slack invokes an interaction callback Lambda which writes to `TaskApprovalsTable` via the same `POST /approvals` path the CLI uses. This gives the common case (reviewer in Slack, not at a terminal) a one-click response path. +**Proposed, not implemented.** Current Slack approval messages contain CLI +commands. A future button callback would need to map the Slack user to the task +owner and use the authenticated decision path; a valid Slack signature alone +would not authorize approval. Linear thread replies already support owner decisions. ### 6.4 GitHub issue comment — edit-in-place @@ -482,7 +486,7 @@ Submitted with the task (optional) or resolved from repo defaults: { "notifications": { "slack": { "enabled": true, "channel": "#coding-agents", "events": ["default"] }, - "email": { "enabled": true, "events": ["approval_required", "task_failed"] }, + "email": { "enabled": true, "events": ["approval_requested", "task_failed"] }, "github": { "enabled": true, "events": ["default"] } } } diff --git a/docs/src/content/docs/architecture/Api-contract.md b/docs/src/content/docs/architecture/Api-contract.md index 885253094..dc3e71aa4 100644 --- a/docs/src/content/docs/architecture/Api-contract.md +++ b/docs/src/content/docs/architecture/Api-contract.md @@ -386,7 +386,7 @@ When a task pauses in `AWAITING_APPROVAL` (Cedar soft-deny gate), the owner appr { "data": { "task_id": "01HYX...", "request_id": "...", "status": "DENIED", "decided_at": "2025-03-15T10:35:00Z" } } ``` -**Errors:** `400 VALIDATION_ERROR`, `401 UNAUTHORIZED`, `404 REQUEST_NOT_FOUND` (collapses "row missing" and "wrong caller"), `409 REQUEST_ALREADY_DECIDED`, `409 TASK_NOT_AWAITING_APPROVAL`. +**Errors:** `400 VALIDATION_ERROR`, `401 UNAUTHORIZED`, `404 REQUEST_NOT_FOUND` (collapses missing, inaccessible, closed or expired approval rows), `409 TASK_NOT_AWAITING_APPROVAL` (task-only state conflict). ### List pending approvals @@ -612,7 +612,7 @@ There is no per-user request-rate or "tasks-per-hour" limiter on task creation. | `RATE_LIMIT_EXCEEDED` | 429 | Rate/concurrency gate exceeded — per-task nudge limit, the application rate limiter on approval endpoints, or the user concurrency limit on confirm-uploads | | `BUDGET_EXCEEDED` | 429 | A configured user or Cognito-team monthly budget reached 100% with hard stop enabled | | `REQUEST_NOT_FOUND` | 404 | Cedar HITL approval request not found (also returned when the caller does not own it) | -| `REQUEST_ALREADY_DECIDED` | 409 | Cedar HITL approval request was already approved or denied | +| `REQUEST_ALREADY_DECIDED` | 409 | Legacy error-code enum; current approve/deny handlers return `404 REQUEST_NOT_FOUND` for closed or inaccessible approval rows | | `TASK_NOT_AWAITING_APPROVAL` | 409 | Task is not in `AWAITING_APPROVAL`, so the approval decision does not apply | | `INTERNAL_ERROR` | 500 | Unexpected server error | | `SERVICE_UNAVAILABLE` | 503 | Downstream dependency unavailable (retry with backoff) | diff --git a/docs/src/content/docs/architecture/Cedar-hitl-gates.md b/docs/src/content/docs/architecture/Cedar-hitl-gates.md index 2b4755cbb..63e962ab7 100644 --- a/docs/src/content/docs/architecture/Cedar-hitl-gates.md +++ b/docs/src/content/docs/architecture/Cedar-hitl-gates.md @@ -10,8 +10,10 @@ title: Cedar hitl gates > **Rev:** 5 (2026-05-06 — fold in parallel adversarial + advocate review of the timeout design: late-approval re-read on TIMED_OUT ConditionCheckFailed; user-visible timeout-cap milestones; ceiling-shrink milestone; Runtime JWT bound verified as auto-refreshed IAM; three new tuning metrics; explicit off-hours trade-off section; notification-delivery-failure boundary. IMPL-24 through IMPL-28 added.). > **Implementation:** Core shipped. The 3-outcome engine (`agent/src/policy.py`), default policy sets (`agent/policies/hard_deny.cedar`, `agent/policies/soft_deny.cedar`), approval Lambdas (`cdk/src/handlers/{approve-task,deny-task,get-pending,get-policies}.ts`) wired into `cdk/src/constructs/task-api.ts` (routes `/tasks/{id}/approve`, `/deny`, `/pending`, `/repos/{repo_id}/policies`), the cross-engine parity fixtures (`contracts/cedar-parity/`), and the exact engine pins are all on `main`. §15's task list is preserved as a historical implementation record; see the note at the top of §15 for what (if anything) remains unbuilt. > -> **Current source behavior (2026-09-18):** the task default is `approval_timeout_s=0`, -> meaning no decision deadline. Explicit task settings are 30–3,600 seconds; +> **Current source behavior (2026-09-22):** the task default is `approval_timeout_s=0`, +> meaning no decision deadline. Workers create/close requests through the IAM-authenticated +> [trusted approval writer](/sample-autonomous-cloud-coding-agents/decisions/adr-023-trusted-approval-writer); they cannot +> write approval rows directly. Explicit task settings are 30–3,600 seconds; > positive policy-rule deadlines still apply. Pending rows have no DynamoDB TTL, > and `expires_at` is nullable. Task closure cancels unanswered requests and adds > retention TTL without changing already-recorded decisions. Closed approval-row @@ -158,7 +160,7 @@ Settled during the 2026-04-23 design discussion and extended after the 2026-04-2 | 20 | **`write_path:` scope** | Added so users can pre-approve file writes under specific path patterns (e.g., `write_path:docs/**`) without needing to grant all Writes. Validation uses Python `fnmatch` at runtime; glob semantics are a Cedar-`like` superset (§6.4, §5.5). | | 21 | **`tool_group:file_write` convenience scope** | Resolves to `{Write, Edit}`. Prevents the surprise of pre-approving `Write` and still getting gated on `Edit`. | | 22 | **Pre-implementation spike: cedarpy annotation round-trip** | Day 1 of implementation validates that `policies_to_json_str()` returns annotations in the expected shape. If the API has changed, fall back to policy-ID prefix conventions. | -| 23 | **Cedar engine parity contract (Python `cedarpy` ↔ JS `cedar-wasm`)** | Both engines are pinned in `mise.toml`. A golden-file parity test runs in CI: for each `(policy, input)` fixture the test asserts Python and WASM return the same `decision` and the same set of matching rule IDs. Policy authors who upgrade either engine must refresh the golden file; drift fails the build. See §15.6 and Appendix B. | +| 23 | **Cedar engine parity contract (Python `cedarpy` ↔ JS `cedar-wasm`)** | Both engines are pinned in their package manifests. A golden-file parity test runs in CI: for each `(policy, input)` fixture the test asserts Python and WASM return the same `decision` and the same set of matching rule IDs. Policy authors who upgrade either engine must refresh the golden file; drift fails the build. See §15.6 and Appendix B. | --- @@ -180,7 +182,7 @@ Narrative walk-through of the happy path. Sequence diagrams in the round-trip Me - rejects blueprint whose combined `cedar_policies` text exceeds the 64 KB cap (§12.4) regardless of origin - resolves `approval_gate_cap` = `Blueprint.security.approvalGateCap ?? 50`; rejects if outside `[1, 500]` (decision #13) 5. Task persists. `approval_timeout_s`, `approval_gate_cap`, and `initial_approvals` become DDB attributes on the task row (cap is captured at submit time so mid-task blueprint edits do not shift the cap beneath a running task). -6. Container spawns on Runtime-JWT. `PolicyEngine.__init__` loads: +6. Container starts with IAM runtime credentials. `PolicyEngine.__init__` loads: - `HARD_DENY_POLICIES` (built-in + repo blueprint's `security.cedarPolicies.hard`; blueprint `disable:` may suppress non-built-in rules only, §5.1, §15.4) - `SOFT_DENY_POLICIES` (built-in + repo blueprint's `security.cedarPolicies.soft`; blueprint `disable:` may suppress soft-deny rules freely) - Annotation lookup table: `{policy_id: {annotation: value}}` built from `cedarpy.policies_to_json_str()` once, cached for the task lifetime @@ -232,7 +234,7 @@ Narrative walk-through of the happy path. Sequence diagrams in the round-trip Me "status": "PENDING", "created_at": "2026-04-23T14:00:00Z", "timeout_s": 300, - "ttl": 1734567890, # created_at + timeout_s + CLEANUP_MARGIN_120S; always covers the decision window + "expires_at": "2026-04-23T14:05:00Z", # explicit decision deadline; no retention TTL "user_id": "...", "repo": "my-org/my-app" } @@ -242,7 +244,7 @@ Narrative walk-through of the happy path. Sequence diagrams in the round-trip Me worker-lease condition for MicroVM): - Put on `TaskApprovalsTable` (new row with status=PENDING) - ConditionalUpdate on `TaskTable`: `status = :awaiting, awaiting_approval_request_id = :rid WHERE status = :running` - Both succeed or both fail. On `TransactionCanceledException` (most likely the TaskTable condition fails because another process moved the status), the hook emits `approval_write_failed` and returns DENY. + Both succeed or both fail. If the service rejects a task/lease conflict or the request fails, the hook emits `approval_write_failed` and returns DENY. 16. Hook emits `agent_milestone("approval_requested", {...})` to both `ProgressWriter` (DDB audit) and `sse_adapter` (live stream). Best-effort emission — transactional write has already committed; milestone failure is observability degradation, not state degradation. 17. Terminal A stream renders: ``` @@ -253,37 +255,12 @@ Narrative walk-through of the happy path. Sequence diagrams in the round-trip Me timeout 300s ``` Severity colors the line (respecting `NO_COLOR` env var). -18. Hook enters poll loop with strongly-consistent reads. The deadline is captured with the original approval row, before database writes/notifications: UTC expiry is `created_at + timeout_s`, capped by the original monotonic remaining duration. `deadline.remaining_s()` takes the smaller remainder and clamps at zero, so a frozen guest clock or backward UTC correction cannot restart the window. - ```python - async def _poll_for_decision(task_id, request_id, deadline): - start = time.monotonic() - interval = 2 - consecutive_failures = 0 - while True: - elapsed = time.monotonic() - start - if deadline.remaining_s() <= 0: - return TimedOut() - if elapsed > 30: - interval = 5 # backoff - try: - row = await _ddb_get_approval(task_id, request_id, ConsistentRead=True) - consecutive_failures = 0 - if row is not None and row["status"] in ("APPROVED", "DENIED"): - return Decided(row) - except Exception as exc: - consecutive_failures += 1 - if consecutive_failures == 3: - log("WARN", f"approval poll degraded for {request_id}: {exc}") - emit_milestone("approval_poll_degraded", {...}) - if consecutive_failures >= 10: - return TimedOut(reason="approval poll consecutive failures") - # Missing rows keep waiting within the original deadline; never allow. - remaining = deadline.remaining_s() - if remaining <= 0: - return TimedOut() - await asyncio.sleep(min(interval, remaining)) - ``` -19. The local-timeout path attempts to write the row to TIMED_OUT (best-effort conditional update `status = :pending`) before returning. If that write loses or fails, the hook rereads consistently to honor an already-committed decision. The approval-cap check runs before row creation and has no row to update. +18. Hook enters the poll loop with strongly-consistent reads. For an explicit positive timeout, the deadline is captured with the original approval row, before database writes/notifications: UTC expiry is `created_at + timeout_s`, capped by the original monotonic remaining duration. `deadline.remaining_s()` takes the smaller remainder and clamps at zero, so a frozen guest clock or backward UTC correction cannot restart the window. + An untimed request has no expiry. The executable poll loop in + [hooks.py](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/agent/src/hooks.py) also handles cancelled/closed requests, + read failures and continuation barriers; it must not be replaced by a loop + that checks only APPROVED/DENIED or starts a fresh timer after wake. +19. The local-timeout path asks the trusted approval service to conditionally mark the pending row TIMED_OUT before returning. If that write loses or fails, the hook rereads consistently to honor an already-committed decision. The approval-cap check runs before row creation and has no row to update. ### User responds @@ -344,7 +321,8 @@ sequenceDiagram Lambda-->>CLI: 202 APPROVED Hook->>Approvals: poll with ConsistentRead Approvals-->>Hook: status APPROVED - Hook->>Approvals: TransactWriteItems, TaskTable to RUNNING + Hook->>Approvals: ConditionCheck on recorded decision + Note right of Hook: Same transaction updates TaskTable to RUNNING;
worker does not modify the approval row Hook->>Engine: allowlist.add(scope) if scope is not this_call Hook-->>Agent: permissionDecision allow Note over Stop: (not used on approval path) @@ -697,35 +675,12 @@ The recent-decision cache is a simple `dict[(tool_name, input_sha), (decision, r When multiple soft-deny rules match a single tool call: -```python -def _merge_annotations(self, policy_ids: list[str]) -> dict: - rule_ids, timeouts, severities = [], [], [] - for pid in policy_ids: - ann = self._annotations[pid] - rule_ids.append(ann.get("rule_id", pid)) - if "approval_timeout_s" in ann: - try: - t = int(ann["approval_timeout_s"]) - if t >= FLOOR_30S: - timeouts.append(t) - except ValueError: - log("WARN", f"malformed @approval_timeout_s on {ann.get('rule_id', pid)}") - severities.append(ann.get("severity", "medium")) - - # Task default always eligible - timeouts.append(self._task_default_timeout_s) - - raw_min_timeout = min(timeouts) - return { - "rule_ids": rule_ids, - "timeout_s": max(FLOOR_30S, raw_min_timeout), # floor enforcement - "severity": _max_severity(severities), # "high" > "medium" > "low" - } -``` +The implementation is `_merge_annotations` in [policy.py](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/agent/src/policy.py). +It takes the shortest **positive** matching-rule or task timeout; zero means no +deadline and is excluded from that minimum. If no positive timeout exists, the +result is zero. Positive results retain the 30-second floor. The highest matching +severity governs the displayed severity, and matching rule IDs are preserved. -**Rationale for min/max choices**: -- **Timeout → min (above floor)**: multiple rules matching means multiple concerns. Users should have *less* time to decide when stakes are higher. Floor prevents unusable 5s windows. -- **Severity → max**: the most severe concern governs the UX coloring. ### 6.4 Allowlist data structure @@ -792,160 +747,28 @@ class ApprovalAllowlist: PreToolUse hook (compressed for doc; implementation will be richer): -```python -async def pre_tool_use_hook(hook_input, tool_use_id, ctx, *, - engine, task_id, user_id, progress, sse_adapter, - task_default_timeout_s): - tool_name, tool_input = _extract(hook_input) - decision = engine.evaluate_tool_use(tool_name, tool_input) - - if decision.outcome == Outcome.ALLOW: - return _allow() - if decision.outcome == Outcome.DENY: - return _deny(decision.reason) - - # REQUIRE_APPROVAL path. - # Cap + rate-limit check. Per-minute rate limit is per-container; on - # container restart the counter resets. The per-task approvalGateCap - # (blueprint-configurable, default 50) is persisted and bounds cumulative - # damage across restarts (§13.6). - if engine.approval_gate_count >= engine.approval_gate_cap: - return _deny(f"approval-gate cap exceeded ({engine.approval_gate_cap}/task)") - if engine.approvals_in_last_minute >= APPROVAL_RATE_LIMIT: - return _deny("approval-gate rate limit exceeded (20/min)") - - # Compute effective timeout with floor/ceiling. - remaining = _remaining_maxlifetime_s() - effective_timeout = max( - FLOOR_30S, - min(decision.timeout_s or task_default_timeout_s, - task_default_timeout_s, - remaining - CLEANUP_MARGIN_120S), - ) - if remaining - CLEANUP_MARGIN_120S < FLOOR_30S: - return _deny(f"insufficient maxLifetime remaining ({remaining}s) for approval") - - request_id = _ulid() - engine.approval_gate_count += 1 - - row = { - "task_id": task_id, "request_id": request_id, - "tool_name": tool_name, - "tool_input_preview": _strip_ansi(_preview(tool_input))[:256], - "tool_input_sha256": _sha256(_serialize(tool_input)), - "reason": decision.reason, "severity": decision.severity, - "matching_rule_ids": list(decision.matching_rule_ids), - "status": "PENDING", - "created_at": _iso_now(), - "timeout_s": effective_timeout, - "ttl": int(time.time()) + effective_timeout + CLEANUP_MARGIN_120S, - "user_id": user_id, "repo": engine.repo, - } - deadline = _ApprovalDeadline.from_recorded(row["created_at"], effective_timeout) - - # ATOMIC: put approval row + transition TaskTable status in one transaction. - try: - await _transact_write_approval_request(task_id, request_id, row) - except TransactionCanceledException as exc: - # Either the task was concurrently cancelled, or status wasn't RUNNING. - _emit("approval_write_failed", {"request_id": request_id, "reason": str(exc)}) - return _deny("approval system unavailable") - - _emit("approval_requested", { - "request_id": request_id, "tool_name": tool_name, - "input_preview": row["tool_input_preview"], - "reason": decision.reason, "severity": decision.severity, - "timeout_s": effective_timeout, - "matching_rule_ids": list(decision.matching_rule_ids), - }) - - outcome = await _poll_for_decision(task_id, request_id, deadline) - - # On TIMED_OUT, attempt to write the row to TIMED_OUT so future reads see - # a terminal state (not orphaned PENDING). The conditional write is guarded - # by `status = :pending` — if the user's APPROVE landed between our last - # poll and this write, the condition fails. In that case we MUST re-read - # the row and honor whatever terminal state won the race; otherwise local - # `outcome.status = "TIMED_OUT"` is stale and we would deny a call the user - # just approved ("I approved it" → agent denies). See §13.12 and the - # scenario below. - if outcome.status == "TIMED_OUT": - wrote_timeout = await _best_effort_update_status( - task_id, request_id, "TIMED_OUT", - reason=outcome.reason, - # Returns True on successful write, False on ConditionCheckFailed. - ) - if not wrote_timeout: - # Re-read the row with ConsistentRead — user's decision beat us. - row = await _ddb_get_approval(task_id, request_id, ConsistentRead=True) - if row is not None and row["status"] == "APPROVED": - # Late-approve wins. Honor it. Rebuild the outcome so the - # downstream allow flow (scope propagation, milestone emission, - # resume transaction) runs identically to the normal approve - # path. - outcome = Decided( - status="APPROVED", - scope=row.get("scope"), - decided_by=row.get("user_id"), - decided_at=row.get("decided_at"), - ) - _emit("approval_late_win", { - "request_id": request_id, - "outcome": "APPROVED", - "reason": "user decision landed during TIMED_OUT write", - }) - elif row is not None and row["status"] == "DENIED": - outcome = Decided( - status="DENIED", - reason=row.get("deny_reason") or "denied", - decided_at=row.get("decided_at"), - ) - # If status is still PENDING (rare — concurrent reaper race) or - # the row is gone (TTL reaped before we could read it), fall - # through with the original TIMED_OUT outcome; fail-closed deny. - - # ATOMIC: resume TaskTable status RUNNING, conditional on awaiting_approval_request_id matching. - try: - await _transact_resume(task_id, request_id) - except TransactionCanceledException: - # User cancelled (or some other path) during poll; abandon gracefully. - _emit("approval_resume_failed", {"request_id": request_id}) - return _deny("task no longer awaiting approval") - - if outcome.status == "APPROVED": - if outcome.scope and outcome.scope != "this_call": - engine._allowlist.add(outcome.scope) - _emit("approval_granted", {"request_id": request_id, - "scope": outcome.scope or "this_call", - "decided_at": outcome.decided_at}) - return _allow() - - # DENIED or TIMED_OUT — cache for 60s + queue denial injection. - engine._recent_decisions.record( - tool_name, _sha256(_serialize(tool_input)), - decision="DENIED" if outcome.status == "DENIED" else "TIMED_OUT", - reason=outcome.reason, - ) - # Truncated reason for guaranteed-surface permissionDecisionReason. - # Best-effort richer injection via _denial_between_turns_hook; may be - # pre-empted by _cancel_between_turns_hook on a concurrently-cancelled - # task. See §4 "Denial with steering text" scenario. - permission_decision_reason = _truncate( - outcome.reason or f"User {outcome.status.lower()}", max_len=500 - ) - if outcome.status == "DENIED": - # Queue steering injection via Stop hook's between_turns_hooks. - engine._queue_denial_injection( - request_id=request_id, - reason=outcome.reason, # already sanitized by DenyTaskFn - decided_at=outcome.decided_at, - ) - _emit("approval_denied" if outcome.status == "DENIED" else "approval_timed_out", - {"request_id": request_id, "reason": outcome.reason}) - return _deny(permission_decision_reason) -``` - -`engine._queue_denial_injection` appends to a list consumed by `_denial_between_turns_hook` — registered **after** `_nudge_between_turns_hook` in the `between_turns_hooks` list (which itself runs after `_cancel_between_turns_hook`). At the next Stop hook fire, the denial is emitted as `…` XML (sanitized via `_xml_escape` from the shared utility introduced with Phase 2). If a `bgagent cancel` has landed between the deny and the next Stop seam, `_cancel_between_turns_hook` short-circuits the dispatcher and the denial text is NOT injected — in which case the guaranteed surface is `permissionDecisionReason` on the hook return. See finding #2 scenario in §4 for the cancel-vs-deny race reasoning. +The executable flow lives in [hooks.py](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/agent/src/hooks.py), with signed +request creation/closure in [approval_requests.py](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/agent/src/approval_requests.py) +and task-state transactions in [task_state.py](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/agent/src/task_state.py). + +1. Evaluate policy, existing grants, gate cap and creation-rate limit. +2. Resolve the shortest positive task/rule deadline. Zero remains untimed. + For a positive deadline, apply a remaining-worker-lifetime ceiling when one + is supplied. A continuation runtime separates this deadline from worker life. +3. Ask the trusted service to atomically create the pending request and move the + task to `AWAITING_APPROVAL`; MicroVM requests also check the active worker lease. + Pending rows have no storage TTL. A failed write denies the action. +4. Emit the notification milestone and poll with strongly consistent reads, + preserving the original UTC/monotonic deadline. No deadline means no timer expiry. +5. If polling times out or fails, ask the trusted service to close the pending + request. If a human decision won the race, reread and preserve that decision. +6. Resume only the matching active task/gate and worker lease. Approval allows the + action and applicable scope; denial is returned to the tool hook and queued as + best-effort steering for the next Stop hook. Cancellation prevents resumption. + +The worker's `TIMED_OUT` outcome currently also covers polling failures; it is +not proof that an explicit human deadline elapsed. This limitation is recorded in +[ADR-023](/sample-autonomous-cloud-coding-agents/decisions/adr-023-trusted-approval-writer). **Scenario (§13.12 VM-throttle + late-approval race).** Alice hits a gate with a 300-second window and commits APPROVED at t=294.7s. Scheduling delays prevent the next agent poll until t=300.2s. That poll sees the original deadline has passed and returns TIMED_OUT before reading the row. The conditional TIMED_OUT write then loses because APPROVED is already stored. The hook rereads with `ConsistentRead`, preserves Alice's scope and decision metadata, emits `approval_late_win`, and proceeds through the guarded resume transaction and allow flow. Without that reread it would deny an already-approved call. See IMPL-24, §13.12, and §15.2 task #43 for the race test. @@ -1033,9 +856,9 @@ On `TransactionCanceledException`, `ApproveTaskFn` inspects per-item This is symmetric with the agent-side `TransactWriteItems` pattern (§4 step 25a) used for the resume transition — Lambdas and agent speak the same atomic-update contract. -**Ownership**: `user_id` stored on TaskApprovalsTable and compared against `caller_user_id` in the ConditionExpression is the Cognito `sub` claim **verbatim**. The Lambda extracts `sub` from the validated JWT and uses it as-is: no prefix stripping, no tenant mapping, no format normalization. If we ever introduce per-tenant user ID namespacing, that transformation MUST happen at the **write** path (i.e. before the agent writes the row in §4 step 14) rather than at compare time, so the ConditionExpression always compares identical-shape identifiers. See finding #6 scenario below. +**Ownership**: `user_id` stored on TaskApprovalsTable and compared against `caller_user_id` in the ConditionExpression is the Cognito `sub` claim **verbatim**. The Lambda extracts `sub` from the validated JWT and uses it as-is: no prefix stripping, no tenant mapping, no format normalization. If we ever introduce per-tenant user ID namespacing, that transformation MUST happen at the **write** path (i.e. before the approval service persists the row prepared in §4 step 14) rather than at compare time, so the ConditionExpression always compares identical-shape identifiers. See finding #6 scenario below. -After successful transaction, `ApproveTaskFn` writes an audit event to `TaskEventsTable` (`approval_decision_recorded` event_type), ensuring the 90-day audit trail is owned by the Lambda path — not dependent on the agent's milestone emission. +After a successful transaction, the decision handler attempts the authoritative `approval_decision_recorded` audit event independently of the agent milestone. Audit-delivery failure does not undo the saved decision. **Scenario (finding #6):** Three months from now, a platform engineer adds a multi-tenant mode where Cognito `sub` becomes `tenant-abc:01JXZ...`. They update the agent's row-write path to prefix-strip: `user_id = sub.split(":", 1)[1]`, storing `01JXZ...` on TaskApprovalsTable. They forget to update `ApproveTaskFn`. Now the Lambda reads `sub = "tenant-abc:01JXZ..."` from the JWT and compares it against the stored `01JXZ...` — condition fails, 404 on every approve, all tasks stranded. The fix as written: "the Cognito sub is compared verbatim; any transformation must happen at write time, not at compare time" — if the agent writes the full `sub`, the Lambda compares the full `sub`; if either side transforms, both sides must. The CI assertion is a unit test that extracts `user_id` from a sample row and asserts it matches the `sub` claim of a sample JWT byte-for-byte. This test would fail on the prefix-strip refactor above and force the engineer to update both sides. Without this hard rule, ownership-in-condition silently breaks under any future identity refactor. @@ -1336,8 +1159,8 @@ stateDiagram-v2 ``` `STRANDED` remains a recognized row status in the type contract. The current -reconciler does not write it: it fails the owning task and leaves the row -`PENDING`, as described below. +reconciler does not write it: it fails the owning task and closes pending requests +as `CANCELLED`, as described below. ### 9.3 Orchestrator impact @@ -1402,31 +1225,11 @@ The design assumes a human is watching. For truly unattended tasks (scheduled au ### 10.1 New DynamoDB table: `TaskApprovalsTable` -```typescript -new dynamodb.Table(this, 'Table', { - partitionKey: { name: 'task_id', type: dynamodb.AttributeType.STRING }, - sortKey: { name: 'request_id', type: dynamodb.AttributeType.STRING }, // ULID - billingMode: dynamodb.BillingMode.PAY_PER_REQUEST, - pointInTimeRecovery: true, - timeToLiveAttribute: 'ttl', - stream: dynamodb.StreamViewType.NEW_AND_OLD_IMAGES, // (evaluated — may drop; see §11) - removalPolicy: RemovalPolicy.RETAIN, -}); - -// v1 GSI — backs `GET /v1/pending` and `bgagent pending`. -// Required at v1 ship, not deferred — see finding #8 scenario in §7.7. -table.addGlobalSecondaryIndex({ - indexName: 'user_id-status-index', - partitionKey: { name: 'user_id', type: dynamodb.AttributeType.STRING }, - sortKey: { name: 'status', type: dynamodb.AttributeType.STRING }, - projectionType: dynamodb.ProjectionType.INCLUDE, - nonKeyAttributes: [ - 'task_id', 'request_id', 'tool_name', 'tool_input_preview', - 'severity', 'reason', 'created_at', 'timeout_s', - 'matching_rule_ids', - ], -}); -``` +[TaskApprovalsTable](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/cdk/src/constructs/task-approvals-table.ts) defines the +`task_id` partition key, `request_id` sort key, optional retention TTL attribute +`ttl`, and `user_id-status-index` GSI. Streams are disabled; TaskEventsTable carries +the audit/fan-out stream. The GSI discovers candidates; the pending endpoint then +strongly reads the approval and owning task before returning a request. **Projection is fixed at design time.** DynamoDB rejects in-place updates to a GSI's `nonKeyAttributes` (the CloudFormation error @@ -1456,11 +1259,12 @@ Attributes: | `matching_rule_ids` | L | Yes | List (not Set — can be empty) of soft-deny rule IDs | | `status` | S | Yes | PENDING \| APPROVED \| DENIED \| TIMED_OUT \| STRANDED \| CANCELLED | | `created_at` | S | Yes | ISO8601 | -| `decided_at` | S | No | Set by approve/deny/cancel; the current guest timeout writer can omit it | +| `decided_at` | S | No | Set when the trusted service or platform closes a request | | `scope` | S | No | Set on APPROVED | | `deny_reason` | S | No | Set on DENIED; sanitized user text | | `timeout_s` | N | Yes | Resolved timeout for audit | -| `ttl` | N | Yes | `created_at_epoch + timeout_s + CLEANUP_MARGIN_120S` — always covers the decision window | +| `expires_at` | S/null | Yes | Original decision deadline; null when `timeout_s=0` | +| `ttl` | N | No | Retention cleanup set on task closure; absent while pending | | `user_id` | S | Yes | Cognito `sub` **verbatim**; used in ownership check `ConditionExpression` (§7.1 finding #6) | | `repo` | S | Yes | Denormalized for fan-out | @@ -1543,7 +1347,7 @@ and the Slack OAuth/button design below remain proposed. Email remains a log-onl receive approval messages. Deployment status is recorded in the [P3 verification record](/sample-autonomous-cloud-coding-agents/verification/readme). -**TaskApprovalsTable Streams are not consumed by the fan-out Lambda**. The approval row is working state; the audit trail is in TaskEventsTable. Enabling Streams on TaskApprovalsTable would be redundant and add noise. Final design: TaskApprovalsTable DOES NOT have Streams enabled. (Retains the `stream` attribute commented out for future use if needed.) +**TaskApprovalsTable Streams are not consumed by the fan-out Lambda**. The approval row is working state; the audit trail is in TaskEventsTable. Enabling Streams on TaskApprovalsTable would be redundant and add noise. Final design: TaskApprovalsTable DOES NOT have Streams enabled. Slack and Linear route `approval_requested`, `approval_decision_recorded`, `approval_timed_out`, `approval_cancelled` and `approval_stranded`. The dispatcher @@ -1732,7 +1536,7 @@ These conditions prevent races; they do not constrain a compromised Lambda with raw table-write permission, which could omit them. Only trusted control-plane handlers receive that permission. -`user_id` comparison is against Cognito `sub` **verbatim** — byte-for-byte equality. Any future identity transformation (per-tenant prefixing, namespacing) must apply to BOTH the write path (agent-side row write) AND the compare path (Lambda ConditionExpression) simultaneously, or the comparison silently fails under the new format. A unit test (§15.3) enforces this: given a sample JWT, extract `sub`, write a row, then assert the stored `user_id` equals `sub` byte-for-byte. +`user_id` comparison is against Cognito `sub` **verbatim** — byte-for-byte equality. Any future identity transformation (per-tenant prefixing, namespacing) must apply to BOTH the write path (trusted approval service) AND the compare path (Lambda ConditionExpression) simultaneously, or the comparison silently fails under the new format. A unit test (§15.3) enforces this: given a sample JWT, extract `sub`, write a row, then assert the stored `user_id` equals `sub` byte-for-byte. ### 12.3 Race prevention @@ -1913,34 +1717,38 @@ Addressed by the parity contract (decision #23, §15.6). Golden-file CI test run ### 13.12 VM-throttle + late-approval race -The agent's poll loop retains the original approval row's UTC expiry and a monotonic cap, using whichever expires first. Slow writes/notifications count toward that same window. This also covers a MicroVM whose monotonic clock stops while suspended; the future `/resume` hook must reuse this deadline and wake the decision loop. If CPU throttling or suspension delays polling beyond expiry, a user's APPROVE transaction may already have committed. The agent then attempts `status = TIMED_OUT WHERE status = :pending`. The condition fails because APPROVED already won. Without a re-read, the agent's local state would remain TIMED_OUT while DDB holds APPROVED, and it would incorrectly deny the approved call. +The agent's poll loop retains the original approval row's UTC expiry and a monotonic cap, using whichever expires first. Slow writes/notifications count toward that same window. This also covers a MicroVM whose monotonic clock stops while suspended; the `/resume` hook reuses this deadline and wakes the decision loop. If CPU throttling or suspension delays polling beyond expiry, a user's APPROVE transaction may already have committed. The agent asks the trusted service to conditionally record `TIMED_OUT` while the row remains `PENDING`. The condition fails because APPROVED already won. Without a re-read, the agent's local state would remain TIMED_OUT while DDB holds APPROVED, and it would incorrectly deny the approved call. -**Mitigation**: the §6.5 pseudocode re-reads the approval row with `ConsistentRead=True` whenever `_best_effort_update_status("TIMED_OUT", ...)` returns ConditionCheckFailed, and honors whatever terminal state the row carries: +**Mitigation**: the hook rereads the approval row with `ConsistentRead=True` when the service reports that another decision won the timeout race, and honors the recorded decision: - If `status == "APPROVED"`: rebuild the local `outcome` to reflect APPROVED, preserving `scope`, `decided_by`, `decided_at`, and proceed through the normal allow flow (scope-propagation, `approval_granted` milestone, resume transaction, return `{"permissionDecision": "allow"}`). Emit a `approval_late_win` milestone so operator telemetry can count races. - If `status == "DENIED"`: honor the denial text the user submitted. Agent returns DENY with the user's sanitized reason as `permissionDecisionReason` (same surface as normal deny). -- If `status` is still PENDING (rare — concurrent reaper race) or the row is gone (TTL reaped): fall through with the original TIMED_OUT outcome; fail-closed deny. +- If `status` is still PENDING (rare — concurrent reaper race) or the row is missing: fall through with the original TIMED_OUT outcome; fail-closed deny. The scenario is bounded by the polling cadence (2-5s ticks) and DDB's strongly-consistent read latency (tens of ms), so the re-read adds at most one extra GetItem to the racing path — acceptable cost for honoring user intent. See IMPL-24 and §15.2 task #43 (race tests). ### 13.13 Runtime JWT expiry during approval wait -**Context in this codebase (verified 2026-05-06).** The AgentCore Runtime container authenticates outbound AWS API calls (DynamoDB, Secrets Manager, etc.) via the container's IAM role, which the SDK resolves through the instance-metadata-service equivalent and auto-refreshes transparently. There is no user-presented JWT with a short rolling expiry consumed by the container's own API calls — `grep -rn -iE 'runtime.jwt|jwt.refresh|token_expiry' agent/src/` returns nothing (only `token_usage` for LLM billing and `GITHUB_TOKEN` for git operations). AgentCore Runtime invocation on the Lambda side uses sigv4 via `InvokeAgentRuntimeCommand` (see `cdk/src/handlers/shared/strategies/agentcore-strategy.ts`) — also auto-refreshed AWS credentials, not a user JWT. The "Runtime-JWT" label in §4 step 6 and the sequence diagrams below refers to the **caller-facing SSE auth** (Terminal A's Cognito ID token presented to API Gateway to stream task events) — it does not authenticate the container's own DDB writes. - -**Therefore, for v1: no separate Runtime JWT expiry term is required in the ceiling computation.** The `maxLifetime` term (AgentCore's hard lifetime of 8h) is the only upper bound we control; IAM credentials refresh automatically within that window. The ceiling definition in decision #6 stands as `min(1h, maxLifetime_remaining - cleanup_margin)`. +Workers authenticate AWS calls with IAM role credentials, not the user's Cognito +JWT. The CLI's Cognito token protects its platform API calls; an expired CLI login +requires reauthentication but does not itself delete the saved approval request. +MicroVM resume refreshes runtime and task-role credentials before releasing coding. -**If the auth model changes** (e.g. a future design introduces a container-held user JWT to authenticate `permissionDecisionReason` attribution, or to carry the caller's Cognito `sub` end-to-end for per-user DDB conditions), the ceiling MUST be extended to `min(1h, maxLifetime_remaining - 120s, runtime_jwt_expiry - 120s)` and this section updated. Tracked as IMPL-27 so the contract is reviewed whenever the auth shape changes. The failure signature if this bound is missed: the container's IAM calls succeed but some JWT-gated channel (e.g. Terminal A's SSE stream) quietly 403s mid-approval-wait; the user's decision lands in DDB but the agent's poll fails to deliver `approval_granted` to the live stream. Today that channel is best-effort observability, not state — but a future state-bearing channel would need the ceiling term. +Credential renewal does not extend a worker's service lifetime. Apply the worker +and explicit-deadline rules in §9.5; do not impose a new approval deadline based on +the user's API token expiry. A future worker-held user-token design would need a +separate review of that boundary. ### 13.14 Notification delivery failure -Fan-out delivery failures (Slack down, email bounce, webhook 5xx) do **NOT** pause the approval timer. The timer runs on the agent's local clock, keyed to the `created_at` timestamp on the DDB row — it is independent of whether any notification channel succeeded in alerting the human. +A failed notification does not change the recorded decision deadline. Timed +requests retain their original UTC/monotonic deadline; untimed requests remain +available until task closure or another valid resolution. Neither case permits +the action without approval. -**Rationale (security):** coupling the timer to notification-plane availability creates a bypass. An adversary who takes down the webhook (or poisons the Slack rate limit) would get an unbounded approval window; worse, a compromised tenant could deliberately suppress their own notifications to escape gates. Fail-closed on timer expiry is invariant; delivery is best-effort observability. +Users can discover unanswered requests with `bgagent pending`, which reads through +the authenticated API independently of notification delivery. A late-discovered +explicitly timed request does not receive a fresh decision window. -**Recovery path for the user:** `bgagent pending` queries `TaskApprovalsTable` directly via the `user_id-status-index` GSI (§7.7, §10.1) — it does not depend on notification delivery. A user who suspects notifications are broken can poll `bgagent pending` at any time to see all live approvals. If the notification never landed and the user finds a gate via `bgagent pending`, they can `bgagent approve/deny` normally; the timer is still running against the original `created_at`, not against when the user found it. - -**Operational signal:** `approval_timed_out` events carry `timeout_s` and the `created_at`/`decided_at` delta. A rising `approval_timed_out` rate with flat `approval_requested` rate (measured via `ApprovalTimeoutClipRate` and `ApprovalDecisionLatency` in §11.3) is the telemetry that indicates notification breakage, not an unresponsive user. - -**The fail-closed posture on timer expiry remains unchanged.** Delivery-availability-aware scheduling is the notification plane's job (see §14.8 and INTERACTIVE_AGENTS.md notification-plane design); the timer does not reason about it. ### 13.15 Fail-closed summary @@ -2001,7 +1809,7 @@ BLOCKED[]: (resource: ) # when a resource is named ### 14.1 Scenario A: force-push with per-rule timeout -Setup: repo `my-org/my-app` blueprint extends soft-deny with `force_push_main` (@approval_timeout_s=600). Task default is 300s. +Setup: repo `my-org/my-app` blueprint extends soft-deny with `force_push_main` (@approval_timeout_s=600). This example explicitly sets the task timeout to 300s. ```bash $ bgagent run --repo my-org/my-app \ @@ -2119,7 +1927,7 @@ Each phase has explicit scope. Matches real-world review workflows. Visible in a ### 14.6 Scenario F: VM-throttle + late-approval race (trace) -Setup: task default 300s; force-push gate fires. User Alice approves at the very edge of the timeout window while the VM is throttled. +Setup: task explicitly configured with a 300-second timeout; force-push gate fires. User Alice approves at the very edge of the timeout window while the VM is throttled. ``` t=0.00s PreToolUse hook fires: Bash "git push --force origin feature-x" @@ -2227,7 +2035,7 @@ See §17.18 for the off-hours escalation future-work primitive, and §13.14 for | # | Package | File | Change | |---|---|---|---| | 1 | agent | Spike | Validate cedarpy.policies_to_json_str() returns annotations. Confirm `diagnostics.reasons` shape for multi-match. If API diverges, update §6 before proceeding. | -| 2 | mise + agent + cdk | `mise.toml`, `agent/pyproject.toml`, `cdk/package.json` | Pin `cedarpy==4.8.0` (agent) and `@cedar-policy/cedar-wasm==4.10.0` (cdk). The two bindings are intentionally on different version lines — verified compatible via the parity fixtures, not required to be equal. Both pinned exactly, not `^` or `~` — decision #23 / finding #1. | +| 2 | mise + agent + cdk | `mise.toml`, `agent/pyproject.toml`, `cdk/package.json` | Pin both Cedar bindings exactly in their package manifests and verify the pair through the shared parity fixtures; see §15.6. | | 3 | agent + cdk | `contracts/cedar-parity/*.json` (shared fixture dir; follows precedent set by `contracts/memory-hash-vectors.json`) | Golden-file parity fixtures: `(policy_set, input) → {decision, matching_rule_ids}`. Agent side loads via `cedarpy`; Lambda side via `cedar-wasm`. Divergence fails CI. | | 4 | agent | `src/policy.py` | Extend `PolicyDecision` (outcome/timeout_s/severity/matching_rule_ids/allowed-property). Split `_DEFAULT_POLICIES` into hard + soft. Add annotation parsing. Implement `ApprovalAllowlist` + `RecentDecisionCache` (50-entry LRU cap, independent of `approvalGateCap`). Load-time validation (rule_id uniqueness, tier mismatch, annotation floor, 64 KB cap, disable-list hard-deny rejection, `approvalGateCap` bounds check `1 ≤ N ≤ 500`). `PolicyEngine.__init__` accepts `approval_gate_cap` sourced from blueprint (default 50). | | 5 | agent | `policies/hard_deny.cedar` (new) | Migrate current hard-deny rules + add DROP TABLE. Annotations. | @@ -2370,12 +2178,13 @@ Rollout steps: ### 15.6 Shared Cedar parsing — cross-engine parity contract -The agent runtime uses Python [`cedarpy@4.8.0`](https://pypi.org/project/cedarpy/); the Lambda side (`CreateTaskFn`, `ApproveTaskFn`, `DenyTaskFn`, `GetPoliciesFn`) uses [`@cedar-policy/cedar-wasm@4.10.0`](https://www.npmjs.com/package/@cedar-policy/cedar-wasm) — AWS's official WASM-compiled Cedar engine. Same Rust core, two bindings. Because these engines evolve independently, we ship a **parity contract** (decision #23, finding #1) to catch drift before deploy. - -**Version pinning.** Both engines are pinned exactly (not `^` or `~`) in the monorepo's canonical manifest files. The two bindings are deliberately on **different version lines** — they are NOT required to be equal. `cedarpy` and `cedar-wasm` follow independent release cadences over the shared Cedar Rust core, and the currently-shipped pins (`cedarpy==4.8.0` ↔ `@cedar-policy/cedar-wasm==4.10.0`) are an intentional, tested-compatible skew: the parity fixtures in `contracts/cedar-parity/` are what certify that this specific pair produces identical `(decision, matching_rule_ids)` on every fixture. The rule is "move together and re-verify parity when you bump either side," not "keep the version strings equal." -- `agent/pyproject.toml`: `cedarpy==4.8.0` -- `cdk/package.json`: `"@cedar-policy/cedar-wasm": "4.10.0"` -- `mise.toml` documents the pinned versions in a comment for operator visibility +The agent uses Python `cedarpy`; the policy Lambdas use +`@cedar-policy/cedar-wasm`. Exact current pins live in +[agent/pyproject.toml](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/agent/pyproject.toml) and +[cdk/package.json](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/cdk/package.json). Binding version strings need not be +identical. Upgrade them as a tested pair and run the shared +[parity fixtures](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/contracts/cedar-parity/README.md), which check decisions +and matching rule IDs across both engines. **Lambda layer packaging** (finding #5). The cedar-wasm package is 4.1 MB unzipped. Shipping it in the deployment bundle of each of the 4 policy Lambdas would consume ~16 MB of unzipped bundle size — manageable on its own but leaves little room for AWS SDK + other deps as the codebase grows, and threatens the Lambda 250 MB unzipped limit under realistic growth. Solution: package cedar-wasm as a **Lambda layer** (`cedar-wasm-layer.ts`, task #10 in §15.2), attached to each policy Lambda. This reduces each Lambda's deployment bundle to just the handler code + thin wrapper around the layer import. Policy Lambdas are configured with ≥ 512 MB memory to accommodate WASM module instantiation under concurrent invocation (measured under 100-concurrent bursts in §15.3 Lambda memory tests). @@ -2422,7 +2231,7 @@ flowchart LR When policy authors upgrade either engine, the parity fixture must be re-generated (a small helper script dumps decisions from both engines; the human confirms the change is intentional). -**Scenario (finding #1, illustrative):** This example uses hypothetical versions (e.g. cedarpy `4.10.1` → `4.11.0`) to show the *class* of bug the parity contract catches; it does not describe the real shipped pins (which are the intentional `cedarpy==4.8.0` ↔ `cedar-wasm==4.10.0` skew documented above). The point is that closeness of version strings — even within the same minor line — is no guarantee of behavioral parity, which is exactly why the golden fixtures, not the version numbers, are the source of truth. A platform engineer runs `mise run deps:update` which bumps cedarpy from 4.10.1 to 4.11.0. They notice cedar-wasm is still 4.10.0 but assume it's fine because both say "4.x". Between these versions, cedarpy added support for a new `context has` operator that cedar-wasm doesn't yet have. A new blueprint soft-deny rule uses `context has "approved_context"`. On deploy: +**Scenario (finding #1, illustrative):** This example uses hypothetical versions (e.g. cedarpy `4.10.1` → `4.11.0`) to show the *class* of bug the parity contract catches; it does not describe the real shipped pins (read the package manifests for the actual pins). The point is that closeness of version strings — even within the same minor line — is no guarantee of behavioral parity, which is exactly why the golden fixtures, not the version numbers, are the source of truth. A platform engineer runs `mise run deps:update` which bumps cedarpy from 4.10.1 to 4.11.0. They notice cedar-wasm is still 4.10.0 but assume it's fine because both say "4.x". Between these versions, cedarpy added support for a new `context has` operator that cedar-wasm doesn't yet have. A new blueprint soft-deny rule uses `context has "approved_context"`. On deploy: - Agent-side `PolicyEngine.__init__` parses the rule successfully; engine loads normally. - `CreateTaskFn` on the Lambda side calls cedar-wasm `policyToJson()` — it throws: `ParseError: unknown operator 'has' at line 3`. - User submits a task against that repo. `CreateTaskFn` crashes mid-validation. Error message: "500 Internal Server Error" (because the Lambda didn't handle the upstream parse error gracefully). @@ -2517,7 +2326,7 @@ Items from the design reviews not captured above as design changes — to be add **IMPL-9** (functional P1-3): Runtime allowlist revocation. Not shipped in v1. Placeholder: `bgagent revoke-approval ` noted in §17. -**IMPL-10** (functional P1-12): `approval_timeout_s` default 300 documented consistently in §3 #6, §7.3 table, §10.2 attribute description. +**IMPL-10** (historical functional P1-12): the original 300-second default was superseded by the September retained-request default of `0` (no decision deadline). **IMPL-11** (functional P2-8): CLI `run.ts` command exists from Phase 1b. `submit.ts` also exists. `--pre-approve` / `--approval-timeout` flags added to both. diff --git a/docs/src/content/docs/architecture/Interactive-agents.md b/docs/src/content/docs/architecture/Interactive-agents.md index f968bc830..fa90f8f8c 100644 --- a/docs/src/content/docs/architecture/Interactive-agents.md +++ b/docs/src/content/docs/architecture/Interactive-agents.md @@ -4,7 +4,9 @@ title: Interactive agents # Interactive Agents: Async Interaction Design -> **Status:** Active design +> **Status:** Historical interaction design with current approval-path corrections (2026-09-22). +> Proposed dispatcher services and Slack buttons below are not all implemented; use the +> [API contract](/sample-autonomous-cloud-coding-agents/architecture/api-contract) and [approval guide](/sample-autonomous-cloud-coding-agents/using/approval-gates-cedar-hitl) for supported interfaces. > **Branch:** `feature/interactive-background-agents` > **Last updated:** 2026-04-29 (rev 6) @@ -23,7 +25,7 @@ This document describes the interactivity surfaces layered on top of that model 3. **Watch** — `bgagent watch ` polls `TaskEventsTable` with an adaptive interval (500 ms when events are arriving, back-off to 5 s when idle). Same endpoint used under the hood for foreground-block UX on `ask` and for HITL approval waits. 4. **Nudge** — `bgagent nudge ""` writes a row into `TaskNudgesTable`. The agent reads pending nudges between turns, acknowledges with a `nudge_acknowledged` milestone event, and integrates the nudge on its next turn. 5. **Ask** — `bgagent ask ""` (Phase 2) writes a question row. The agent answers at the next between-turns boundary; the answer surfaces as a `status_response` event. CLI default is foreground block-and-poll with a spinner; task and answer are both durable if the CLI disconnects. -6. **Approval gates** — Phase 3 Cedar-driven hard gates. Agent emits `approval_requested`, waits for a decision from `bgagent approve` / `bgagent deny` or a Slack button-press. Detailed design in [`CEDAR_HITL_GATES.md`](/sample-autonomous-cloud-coding-agents/architecture/cedar-hitl-gates). +6. **Approval gates** — Phase 3 Cedar-driven hard gates. Agent emits `approval_requested`, waits for a decision from `bgagent approve` / `bgagent deny` or an owner-authored Linear thread reply. Slack notifications provide CLI instructions; buttons remain proposed. Detailed design in [`CEDAR_HITL_GATES.md`](/sample-autonomous-cloud-coding-agents/architecture/cedar-hitl-gates). ### Core architectural choices @@ -31,7 +33,7 @@ This document describes the interactivity surfaces layered on top of that model - **Durable event table (`TaskEventsTable`)** is the one source of truth for agent progress. Every reader — CLI, Slack/GitHub/email dispatchers, status Lambda — reads from this table, never from the live agent. - **Polling-only CLI.** No SSE, no WebSockets. DDB eventually-consistent reads with an `event_id` cursor are cheap, reliable, and compute-agnostic. - **Notification plane as first-class.** A FanOutConsumer Lambda subscribes to `TaskEventsTable` DDB Streams and routes per-event-type to per-channel dispatcher Lambdas (Slack, email, GitHub comment). Per-channel defaults ship in v1. -- **Agent interaction via the hook mechanism the Claude Agent SDK provides.** Nudges, asks, and approvals all use `Stop` / between-turns hooks; no mechanism outside the SDK's contract is required. +- **Agent interaction via the hook mechanism the Claude Agent SDK provides.** Nudges and asks use `Stop` / between-turns hooks; approval gates pause in `PreToolUse`, with denial steering delivered at a later Stop hook; no mechanism outside the SDK's contract is required. --- @@ -115,7 +117,7 @@ This document describes the interactivity surfaces layered on top of that model │ DELETE /tasks/{id} cancel │ │ POST /tasks/{id}/nudge nudge │ │ POST /tasks/{id}/asks ask (P2) │ - │ POST /tasks/{id}/approvals approve P3 │ + │ POST /tasks/{id}/approve approve │ │ POST /webhooks/tasks GH webhook │ └───────────┬──────────────────────────────────┘ │ @@ -227,17 +229,16 @@ Consumer: agent between-turns hook reads pending nudges, emits `nudge_acknowledg ### 3.7 TaskApprovalsTable (Phase 3) Phase 3 approval-request spine. Detailed schema in [`CEDAR_HITL_GATES.md`](/sample-autonomous-cloud-coding-agents/architecture/cedar-hitl-gates). Semantics summary: -- Agent writes an approval row with the request context. -- Agent transitions `RUNNING → AWAITING_APPROVAL` and enters a poll loop. -- User responds via REST (`POST /tasks/{id}/approvals/{request_id}`) or via a Slack button dispatched by the notification plane. +- The worker calls the trusted approval service, which atomically creates the pending row and transitions the task `RUNNING → AWAITING_APPROVAL`. The worker then polls for a decision. +- The owner responds through `POST /tasks/{id}/approve` or `/deny` with `request_id` in the body, the equivalent CLI commands, or an `approve`/`deny` reply to the Linear approval comment. - On decision, agent transitions back to `RUNNING`; denial reasons are injected as Stop-hook steering on the next turn. ### 3.8 FanOutConsumer (router) Lambda subscribed to `TaskEventsTable` DDB Streams (relying on the DynamoDB Streams **default** `ParallelizationFactor` of 1, which preserves per-`task_id` ordering by shard — not set explicitly in `fanout-consumer.ts`; see §6.1). Reads per-task notification config (from `TaskTable` metadata or `RepoTable` defaults), filters events by channel subscription, and invokes per-channel dispatcher Lambdas. -- **SlackDispatchFn** — posts to configured channel / DM. Includes action buttons for `approval_required` events. -- **EmailDispatchFn** — SES. +- **SlackDispatchFn** — posts to configured channel / DM. Approval notifications currently contain CLI response instructions; buttons remain proposed. +- **EmailDispatchFn** — proposed SES delivery; current email dispatch is a log-only stub. - **GitHubDispatchFn** — edits a single GitHub issue comment in place via `PATCH /repos/{o}/{r}/issues/comments/{id}`. On 404 (comment deleted upstream) falls back to POSTing a fresh comment. Per-task ordering is guaranteed upstream by the DDB Streams default `ParallelizationFactor` of 1 (see §6.1), so no conditional-request header is needed (and GitHub's REST API does not accept `If-Match` on this endpoint — see §6.4). Detailed routing and default filters in §6. @@ -300,8 +301,8 @@ Authentication: Cognito User Pool ID token in `Authorization` header for all RES | `agent_milestone` | Agent code (pipeline, hooks) | Named checkpoint (`repo_cloned`, `pr_opened`, `nudge_acknowledged`, ...) | | `agent_cost_update` | Runner | Cumulative token + dollar cost | | `agent_error` | Runner | Handled exception | -| `approval_required` (P3) | PreToolUse Cedar hook | Cedar policy requires user decision | -| `approval_decided` (P3) | Approve/Deny Lambda | User responded | +| `approval_requested` (P3) | PreToolUse Cedar hook | Cedar policy requires user decision | +| `approval_decision_recorded` (P3) | Approve/Deny Lambda | User responded | | `status_response` (P2) | Between-turns hook | Agent answered an `ask` | | `nudge_acknowledged` | Between-turns hook | Agent saw a nudge before incorporating it | | `pr_created` | Pipeline | PR opened for the task | @@ -324,7 +325,7 @@ Consumers page `TaskEventsTable` using `event_id` as a cursor: `KeyConditionExpr ### 5.1 `bgagent submit` ``` -$ bgagent submit --repo org/repo "fix the auth timeout bug" +$ bgagent submit --repo org/repo --task "fix the auth timeout bug" task submitted: abc123 ``` @@ -415,9 +416,9 @@ Flags: HITL approval commands. All flows are REST + DDB; no streaming. Detailed design in [`CEDAR_HITL_GATES.md`](/sample-autonomous-cloud-coding-agents/architecture/cedar-hitl-gates). Summary: -- Agent emits `approval_required` with the tool context. -- Notification plane dispatches the event (Slack with action buttons, email, GitHub). -- User responds via `bgagent approve `, `bgagent deny --reason "…"`, or Slack button click. +- Agent emits `approval_requested` with the tool context. +- The notification plane sends approval messages to Slack and Linear. Email is a stub; GitHub does not receive approval messages. +- The owner responds via `bgagent approve `, `bgagent deny --reason "…"`, or an `approve`/`deny` reply to the Linear approval comment. - Agent's poll loop sees the decision and proceeds or deny-steers. ### 5.7 `bgagent cancel` @@ -448,19 +449,22 @@ TaskEventsTable ──DDB Stream──▶ FanOutConsumer - Router reads per-task notification config (channel enablement + event-type filters), then invokes the relevant dispatcher Lambda(s) per event. - Dispatchers are separate Lambdas so a GitHub API outage doesn't block Slack notifications. -### 6.2 Per-channel defaults (v1) +### 6.2 Original proposed per-channel defaults (v1) | Channel | Default subscribed events | Opt-in via `--verbose` | |---|---|---| -| **Slack** | `task_completed`, `task_failed`, `task_cancelled`, `pr_created`, `agent_error`, `approval_required`, `status_response` | adds `agent_milestone` | -| **Email** | `task_completed`, `task_failed`, `approval_required` | — | +| **Slack** | `task_completed`, `task_failed`, `task_cancelled`, `pr_created`, `agent_error`, `approval_requested`, `status_response` | adds `agent_milestone` | +| **Email** | `task_completed`, `task_failed`, `approval_requested` | — | | **GitHub issue comment** | `pr_created`, terminal status (single edit-in-place comment) | — already minimal | Rationale: if Slack pings on every milestone, users mute the bot within days. Default to the minimal set that surfaces decision-requiring events and completion; power users opt into verbose streams. ### 6.3 Slack approval buttons -`approval_required` events delivered to Slack include `Approve` / `Deny` action buttons. On click, Slack invokes an interaction callback Lambda which writes to `TaskApprovalsTable` via the same `POST /approvals` path the CLI uses. This gives the common case (reviewer in Slack, not at a terminal) a one-click response path. +**Proposed, not implemented.** Current Slack approval messages contain CLI +commands. A future button callback would need to map the Slack user to the task +owner and use the authenticated decision path; a valid Slack signature alone +would not authorize approval. Linear thread replies already support owner decisions. ### 6.4 GitHub issue comment — edit-in-place @@ -486,7 +490,7 @@ Submitted with the task (optional) or resolved from repo defaults: { "notifications": { "slack": { "enabled": true, "channel": "#coding-agents", "events": ["default"] }, - "email": { "enabled": true, "events": ["approval_required", "task_failed"] }, + "email": { "enabled": true, "events": ["approval_requested", "task_failed"] }, "github": { "enabled": true, "events": ["default"] } } } diff --git a/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md b/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md index 6e45785ea..4dddb11f0 100644 --- a/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md +++ b/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md @@ -178,7 +178,7 @@ The launch-region list is us-east-1, us-east-2, us-west-2, eu-west-1 and ap-nort - CLI onboarding and platform doctor probe `list-managed-microvm-images`. - Runtime regional failures receive a configuration remedy instead of an opaque SDK error. -### 5. Rollout: phased, default unchanged +### 5. Rollout: phased, AgentCore remains the default backend | Phase | Delivered behavior | |---|---| @@ -188,7 +188,7 @@ The launch-region list is us-east-1, us-east-2, us-west-2, eu-west-1 and ap-nort Activate sleep only after verifying the deployed image and coordinator together. Keep a compatible published coordinator and explicit image pin for rollback. Normal acceptance includes the 600-second default, explicit expiry, new and existing task off-switch behavior, replacement, cleanup and preservation of unrelated infrastructure. -Changing the default backend, GPU support, native Slack approval buttons, approval-by-Linear-reply and operator shell access are outside this ADR. CLI responses remain the supported approval path. Future work needs its own scope; “P4” is not an approved phase here. +Approval responses are supported through the authenticated CLI and owner-authored `approve`/`deny` replies to Linear approval comments. Changing the default backend, GPU support, native Slack approval buttons and operator shell access remain outside this ADR. Future work needs its own scope; “P4” is not an approved phase here. ## Consequences From 5651cccd3918e77c295344c16b2cd16dad368995 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Tue, 22 Sep 2026 18:39:31 -0400 Subject: [PATCH 133/149] fix(microvm): configure CLI approval replacement admission --- cdk/src/constructs/task-api.ts | 3 +++ cdk/src/stacks/agent.ts | 1 + cdk/test/constructs/microvm-continuation-manager.test.ts | 7 ++++++- 3 files changed, 10 insertions(+), 1 deletion(-) diff --git a/cdk/src/constructs/task-api.ts b/cdk/src/constructs/task-api.ts index 5a47f3aa4..6e9da19c2 100644 --- a/cdk/src/constructs/task-api.ts +++ b/cdk/src/constructs/task-api.ts @@ -1459,10 +1459,13 @@ export class TaskApi extends Construct { /** Wire after the coordinator and bucket exist, avoiding a construction-order cycle. */ public enableMicrovmContinuations( bucketName: string, coordinatorArn: string, concurrencyTable: dynamodb.ITable, + maxConcurrentTasksPerUser: number, ): void { for (const fn of this.approvalDecisionFunctions) { fn.addEnvironment('CONTINUATION_BUCKET_NAME', bucketName); fn.addEnvironment('ORCHESTRATOR_FUNCTION_ARN', coordinatorArn); + fn.addEnvironment('USER_CONCURRENCY_TABLE_NAME', concurrencyTable.tableName); + fn.addEnvironment('MAX_CONCURRENT_TASKS_PER_USER', String(maxConcurrentTasksPerUser)); concurrencyTable.grantReadWriteData(fn); fn.addToRolePolicy(new iam.PolicyStatement({ actions: ['lambda:InvokeFunction'], resources: [`${coordinatorArn}:*`], diff --git a/cdk/src/stacks/agent.ts b/cdk/src/stacks/agent.ts index 0718d35ff..fff03fe6d 100644 --- a/cdk/src/stacks/agent.ts +++ b/cdk/src/stacks/agent.ts @@ -1432,6 +1432,7 @@ export class AgentStack extends Stack { if (continuationBucket && lambdaMicrovm?.imageArn) { taskApi.enableMicrovmContinuations( continuationBucket.bucket.bucketName, orchestrator.fn.functionArn, userConcurrencyTable.table, + maxConcurrentTasksPerUser, ); new MicrovmContinuationManager(concurrencyMaintenance, 'MicrovmContinuationManager', { taskTable: taskTable.table, diff --git a/cdk/test/constructs/microvm-continuation-manager.test.ts b/cdk/test/constructs/microvm-continuation-manager.test.ts index a98688bc5..e8a1e5bdc 100644 --- a/cdk/test/constructs/microvm-continuation-manager.test.ts +++ b/cdk/test/constructs/microvm-continuation-manager.test.ts @@ -44,7 +44,7 @@ test('scheduled recovery and decisions invoke retained versions of only the orig imageArn, }); const api = new TaskApi(stack, 'Api', { taskTable, taskEventsTable: table('Events'), taskApprovalsTable: approvalsTable }); - api.enableMicrovmContinuations(bucket.bucket.bucketName, coordinator, userConcurrencyTable); + api.enableMicrovmContinuations(bucket.bucket.bucketName, coordinator, userConcurrencyTable, 7); const template = Template.fromStack(stack); const policies = Object.entries(template.findResources('AWS::IAM::Policy')); for (const name of ['ManagerReconcilerFn', 'ApproveTaskFn', 'DenyTaskFn']) { @@ -60,5 +60,10 @@ test('scheduled recovery and decisions invoke retained versions of only the orig const fn = functions.find(([id]) => id.includes(name))![1]; expect(fn.Properties.Environment.Variables.ORCHESTRATOR_FUNCTION_ARN).toBe(coordinator); expect(fn.Properties.Environment.Variables.CONTINUATION_BUCKET_NAME).toBeDefined(); + expect(fn.Properties.Environment.Variables.USER_CONCURRENCY_TABLE_NAME) + .toEqual(stack.resolve(userConcurrencyTable.tableName)); + if (name !== 'ManagerReconcilerFn') { + expect(fn.Properties.Environment.Variables.MAX_CONCURRENT_TASKS_PER_USER).toBe('7'); + } } }); From 0821220e1996325c76034ffcb54b85888766520e Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Tue, 22 Sep 2026 18:40:36 -0400 Subject: [PATCH 134/149] fix(microvm): clean fenced attempts that never launched --- .../reconcile-microvm-continuations.ts | 13 +++++-- .../reconcile-microvm-continuations.test.ts | 39 +++++++++++++++++++ 2 files changed, 48 insertions(+), 4 deletions(-) diff --git a/cdk/src/handlers/reconcile-microvm-continuations.ts b/cdk/src/handlers/reconcile-microvm-continuations.ts index dd67835e8..7939eba7b 100644 --- a/cdk/src/handlers/reconcile-microvm-continuations.ts +++ b/cdk/src/handlers/reconcile-microvm-continuations.ts @@ -76,9 +76,14 @@ export async function reconcileMicrovmContinuation(task: ContinuableTask): Promi TableName: TABLE, Key: workerLeaseKey(task.task_id), ConsistentRead: true, }), options)).Item; if (lease) { - const notLaunched = lease.lease_state === 'ACTIVE' && !lease.lease_microvm_id - && task.concurrency_slot?.state === 'held' - && task.concurrency_slot.attempt_id === lease.lease_attempt_id; + // claimMicrovmStart persists its receipt before calling AWS and only + // while the task is active. This terminal task has no receipt, so a + // matching admitted attempt cannot still launch. Finalization may have + // already released its slot; that does not erase its attempt identity. + const notLaunched = ['ACTIVE', 'FENCED'].includes(lease.lease_state) && !lease.lease_microvm_id + && ['held', 'released'].includes(task.concurrency_slot?.state ?? '') + && task.concurrency_slot?.attempt_id === lease.lease_attempt_id + && task.continuation?.attempt_id === lease.lease_attempt_id; if (lease.lease_user_id !== task.user_id || (!['PARKED', 'CLOSED'].includes(lease.lease_state) && !notLaunched) || typeof lease.lease_attempt_id !== 'string' || !lease.lease_attempt_id) { @@ -94,7 +99,7 @@ export async function reconcileMicrovmContinuation(task: ContinuableTask): Promi UpdateExpression: 'SET #ttl = :ttl, lease_state = :closed, lease_user_id = :user, lease_attempt_id = :attempt', ConditionExpression: 'attribute_not_exists(task_id) OR (lease_user_id = :user AND lease_attempt_id = :attempt' + (withoutStart ? ' AND lease_state = :observedState' : '') - + (leaseState === 'ACTIVE' ? ' AND attribute_not_exists(lease_microvm_id)' : '') + ')', + + (['ACTIVE', 'FENCED'].includes(leaseState ?? '') ? ' AND attribute_not_exists(lease_microvm_id)' : '') + ')', ExpressionAttributeNames: { '#ttl': 'ttl' }, ExpressionAttributeValues: { ':user': task.user_id, diff --git a/cdk/test/handlers/reconcile-microvm-continuations.test.ts b/cdk/test/handlers/reconcile-microvm-continuations.test.ts index 15f71a547..30a01a351 100644 --- a/cdk/test/handlers/reconcile-microvm-continuations.test.ts +++ b/cdk/test/handlers/reconcile-microvm-continuations.test.ts @@ -163,6 +163,45 @@ test('does not release a live lease just because the start record is absent', as expect(mockDelete).not.toHaveBeenCalled(); }); +test.each(['held', 'released'])('cleans a fenced pre-launch failure with a %s slot', async state => { + task.status = 'FAILED'; + delete task.microvm_start; + task.continuation = { state: 'STARTING', attempt_id: 'new-token' }; + task.concurrency_slot = { state, attempt_id: 'new-token' }; + mockSend.mockImplementation(async command => { + if (command.constructor.name === 'UpdateCommand') return {}; + return { + Item: command.input.Key.task_id === 'task' ? task : { + lease_user_id: 'user', lease_attempt_id: 'new-token', lease_state: 'FENCED', + }, + }; + }); + await reconcileMicrovmContinuation(task); + const closed = mockSend.mock.calls.find(([command]) => command.constructor.name === 'UpdateCommand')![0].input; + expect(closed.ExpressionAttributeValues).toMatchObject({ ':attempt': 'new-token', ':observedState': 'FENCED' }); + expect(closed.ConditionExpression).toContain('attribute_not_exists(lease_microvm_id)'); + expect(mockDelete).toHaveBeenCalled(); +}); + +test.each([ + { lease_microvm_id: 'possibly-live-worker' }, + { lease_user_id: 'other-user' }, + { lease_attempt_id: 'other-attempt' }, +])('retains checkpoints for an inconsistent fenced attempt: %j', async mismatch => { + task.status = 'FAILED'; + delete task.microvm_start; + task.continuation = { state: 'STARTING', attempt_id: 'new-token' }; + task.concurrency_slot = { state: 'released', attempt_id: 'new-token' }; + mockSend.mockImplementation(async command => ({ + Item: command.input.Key.task_id === 'task' ? task : { + lease_user_id: 'user', lease_attempt_id: 'new-token', lease_state: 'FENCED', ...mismatch, + }, + })); + await expect(reconcileMicrovmContinuation(task)).rejects.toThrow('LEASE_INVALID'); + expect(mockRelease).not.toHaveBeenCalled(); + expect(mockDelete).not.toHaveBeenCalled(); +}); + test('invalid unknown-start timestamp cannot be treated as proof of shutdown', async () => { task.status = 'FAILED'; task.microvm_start.createdAt = 'invalid'; From 65797567fb418f970992e6268985ee795c21ae69 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Tue, 22 Sep 2026 18:43:01 -0400 Subject: [PATCH 135/149] fix(microvm): tolerate advisory credential refresh outages --- agent/src/aws_session.py | 5 ++++- agent/tests/test_microvm_credentials.py | 21 +++++++++++++++++++++ 2 files changed, 25 insertions(+), 1 deletion(-) diff --git a/agent/src/aws_session.py b/agent/src/aws_session.py index 8c05da9e8..1f6a3e0ba 100644 --- a/agent/src/aws_session.py +++ b/agent/src/aws_session.py @@ -386,7 +386,10 @@ def _locked_refresh(credentials: Any, *, force: bool) -> dict[str, str]: raise TimeoutError("Credential refresh lock did not become available") try: if force or credentials.refresh_needed(): - credentials._protected_refresh(is_mandatory=True) + credentials._protected_refresh( + is_mandatory=force + or credentials.refresh_needed(credentials._mandatory_refresh_timeout) + ) frozen = credentials._frozen_credentials expiry = credentials._expiry_time if ( diff --git a/agent/tests/test_microvm_credentials.py b/agent/tests/test_microvm_credentials.py index baaf888b1..769d9de35 100644 --- a/agent/tests/test_microvm_credentials.py +++ b/agent/tests/test_microvm_credentials.py @@ -211,6 +211,27 @@ def test_expiration_is_utc_and_keys_are_a_coherent_pair(self): assert 3598 < (expiry - datetime.now(UTC)).total_seconds() <= 3600 assert result["AccessKeyId"] == "PAIR" + @pytest.mark.parametrize( + ("remaining_seconds", "force", "must_fail"), + [(12 * 60, False, False), (5 * 60, False, True), (12 * 60, True, True)], + ) + def test_refresh_outage_preserves_only_advisory_cached_credentials( + self, remaining_seconds, force, must_fail + ): + metadata = _metadata("CACHED") + credentials = DeferredRefreshableCredentials(method="test", refresh_using=lambda: metadata) + aws_session._locked_refresh(credentials, force=False) + credentials._expiry_time = datetime.now(UTC) + timedelta(seconds=remaining_seconds) + credentials._refresh_using = MagicMock(side_effect=RuntimeError("synthetic STS outage")) + if must_fail: + with pytest.raises(RuntimeError, match="synthetic STS outage"): + aws_session._locked_refresh(credentials, force=force) + else: + result = aws_session._locked_refresh(credentials, force=force) + assert result["AccessKeyId"] == "CACHED" + assert result["Token"] == metadata["token"] + credentials._refresh_using.assert_called_once() + class TestScopedBroker: def test_requires_auth_and_scrubs_only_child_environment(self, monkeypatch): From 1a0d527c3affb71ea247afb1c7166b38d8ba9492 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Tue, 22 Sep 2026 18:45:22 -0400 Subject: [PATCH 136/149] fix(microvm): require explicit layout until migration is verified --- cdk/scripts/README.md | 2 +- cdk/scripts/package-microvm-artifact.sh | 19 ++++---- cdk/src/stacks/agent.ts | 22 ++++------ cdk/test/stacks/agent.test.ts | 3 +- .../stacks/microvm-layout-context.test.ts | 43 ++++++++----------- ...ADR-021-lambda-microvms-compute-backend.md | 2 +- docs/design/COMPUTE.md | 3 +- docs/design/DEPLOYMENT_ROLES.md | 2 +- docs/guides/DEPLOYMENT_GUIDE.md | 5 ++- docs/src/content/docs/architecture/Compute.md | 3 +- .../docs/architecture/Deployment-roles.md | 2 +- ...Adr-021-lambda-microvms-compute-backend.md | 2 +- .../docs/getting-started/Deployment-guide.md | 5 ++- .../docs/verification/645-p3-nested-stack.md | 11 +++-- docs/verification/645-p3-nested-stack.md | 11 +++-- 15 files changed, 65 insertions(+), 70 deletions(-) diff --git a/cdk/scripts/README.md b/cdk/scripts/README.md index ac2b55f73..42688a709 100644 --- a/cdk/scripts/README.md +++ b/cdk/scripts/README.md @@ -11,6 +11,6 @@ Bundling for Lambda assets is handled at synth time; the **`bundle`** task in ** `package-microvm-artifact.sh` exists because CloudFormation cannot produce its own MicroVM `codeArtifact`: the image resource consumes a zip that must already be in S3, and there is no CDK asset type for "zip + Dockerfile a MicroVM image builds from". Everything else on that backend (buckets, roles, network connectors, log group, the image resource itself) is CDK-managed by `src/constructs/lambda-microvm-compute.ts`, normally inside `lambda-microvm-stack.ts`. -The default nested layout requires bootstrap bundle 1.9.0. Before upgrading an +New installations must explicitly select `microvm_nested_stack=true`; the nested layout requires bootstrap bundle 1.9.0. Before upgrading an existing flat deployment, set and retain `microvm_nested_stack=false` until completing the [resource migration](../../docs/verification/645-p3-nested-stack.md). diff --git a/cdk/scripts/package-microvm-artifact.sh b/cdk/scripts/package-microvm-artifact.sh index 001ac2ab1..c0dc13153 100755 --- a/cdk/scripts/package-microvm-artifact.sh +++ b/cdk/scripts/package-microvm-artifact.sh @@ -18,7 +18,10 @@ # --------------------------------------------------------------------------- # BOOTSTRAP SEQUENCE (first time) # --------------------------------------------------------------------------- -# 0. Use bootstrap policy bundle >= 1.9.0 for the default nested layout. +# Choose microvm_nested_stack=true for new/already-nested installations, +# or false for existing flat installations, in the commands below. +# +# 0. Use bootstrap policy bundle >= 1.9.0 for the nested layout. # Before upgrading an existing flat deployment, set and retain # microvm_nested_stack=false (bundle >= 1.8.0) on every deploy below until # its resources are migrated; see docs/verification/645-p3-nested-stack.md. @@ -27,7 +30,7 @@ # is configured; that is expected — the artifact bucket must exist before # you can upload to it. # -# MISE_EXPERIMENTAL=1 mise //cdk:deploy -- --context compute_type=lambda-microvm +# MISE_EXPERIMENTAL=1 mise //cdk:deploy -- --context compute_type=lambda-microvm --context microvm_nested_stack=REPLACE_WITH_TRUE_OR_FALSE # # 2. Package + upload the artifact (this script). It reads the bucket name and # base object key from stack outputs, uploads an immutable hash-suffixed @@ -48,7 +51,7 @@ # image resource and injects MICROVM_IMAGE_IDENTIFIER into the orchestrator: # # MISE_EXPERIMENTAL=1 mise //cdk:deploy -- \ -# --context compute_type=lambda-microvm \ +# --context compute_type=lambda-microvm --context microvm_nested_stack=REPLACE_WITH_TRUE_OR_FALSE \ # --context microvm_base_image_arn= \ # --context microvm_base_image_version= \ # --context microvm_artifact_sha256= @@ -69,7 +72,7 @@ # — `run-microvm` rejects a bare name): # # MISE_EXPERIMENTAL=1 mise //cdk:deploy -- \ -# --context compute_type=lambda-microvm \ +# --context compute_type=lambda-microvm --context microvm_nested_stack=REPLACE_WITH_TRUE_OR_FALSE \ # --context microvm_image_identifier= # # --------------------------------------------------------------------------- @@ -231,7 +234,7 @@ error: stack '${STACK_NAME}' has no MicrovmArtifactBucketName/MicrovmArtifactObj That means this stack was not deployed with the lambda-microvm compute backend. Deploy it first: - MISE_EXPERIMENTAL=1 mise //cdk:deploy -- --context compute_type=lambda-microvm + MISE_EXPERIMENTAL=1 mise //cdk:deploy -- --context compute_type=lambda-microvm --context microvm_nested_stack=REPLACE_WITH_TRUE_OR_FALSE EOF exit 1 fi @@ -328,7 +331,7 @@ if [[ "${CREATE_IMAGE}" -eq 0 ]]; then CDK-managed (recommended) — redeploy with the base image and artifact pinned. Keep this digest with the deployment's context; it identifies these exact ZIP bytes. - The default nested layout requires bootstrap policy bundle >= 1.9.0. + The nested layout requires bootstrap policy bundle >= 1.9.0. Flat P3 deployments require >= 1.8.0. Check the installed bundle: aws cloudformation describe-stacks --stack-name CDKToolkit \\ @@ -347,7 +350,7 @@ if [[ "${CREATE_IMAGE}" -eq 0 ]]; then aws lambda-microvms list-managed-microvm-images MISE_EXPERIMENTAL=1 mise //cdk:deploy -- \\ - --context compute_type=lambda-microvm \\ + --context compute_type=lambda-microvm --context microvm_nested_stack=REPLACE_WITH_TRUE_OR_FALSE \\ --context microvm_base_image_arn= \\ --context microvm_base_image_version= \\ --context microvm_artifact_sha256=${ARTIFACT_SHA256} @@ -477,7 +480,7 @@ cat < { { name: 'agentcore', context: { compute_type: 'agentcore' } }, { name: 'ecs', context: { compute_type: 'ecs' } }, ...MICROVM_CONFIGURATIONS.flatMap(configuration => [ - { ...configuration, name: `${configuration.name}-default-nested` }, + { name: `${configuration.name}-nested`, context: { ...configuration.context, microvm_nested_stack: true } }, { name: `${configuration.name}-flat`, context: { ...configuration.context, microvm_nested_stack: false }, diff --git a/cdk/test/stacks/microvm-layout-context.test.ts b/cdk/test/stacks/microvm-layout-context.test.ts index 9109fc274..5595110e8 100644 --- a/cdk/test/stacks/microvm-layout-context.test.ts +++ b/cdk/test/stacks/microvm-layout-context.test.ts @@ -18,7 +18,7 @@ */ import { App } from 'aws-cdk-lib'; -import { Annotations, Match, Template } from 'aws-cdk-lib/assertions'; +import { Template } from 'aws-cdk-lib/assertions'; import { AgentStack } from '../../src/stacks/agent'; const env = { account: '123456789012', region: 'us-east-1' }; @@ -28,35 +28,26 @@ const imageContext = { microvm_artifact_sha256: 'a'.repeat(64), }; +test.each([{}, imageContext])('rejects an omitted layout before synthesizing MicroVM resources: %j', context => { + const app = new App({ context: { compute_type: 'lambda-microvm', ...context } }); + expect(() => new AgentStack(app, 'ImplicitLayout', { env })) + .toThrow('microvm_nested_stack must be explicitly selected'); +}); + describe.each([ - { name: 'MicrovmBootstrap', context: { compute_type: 'lambda-microvm' }, warns: true }, - { name: 'MicrovmManaged', context: { compute_type: 'lambda-microvm', ...imageContext }, warns: true }, ...[true, 'true', false, 'false'].map((value, index) => ({ name: `ExplicitLayout${index}`, context: { compute_type: 'lambda-microvm', microvm_nested_stack: value }, - warns: false, })), - { name: 'Agentcore', context: { compute_type: 'agentcore' }, warns: false }, - { name: 'Ecs', context: { compute_type: 'ecs' }, warns: false }, -])('MicroVM layout warning: $name', ({ name, context, warns }) => { - let annotations: Annotations; - + { name: 'Agentcore', context: { compute_type: 'agentcore' } }, + { name: 'Ecs', context: { compute_type: 'ecs' } }, +])('MicroVM layout selection: $name', ({ name, context }) => { + let template: Template; beforeAll(() => { - const stack = new AgentStack(new App({ context }), name, { env }); - Template.fromStack(stack); - annotations = Annotations.fromStack(stack); + template = Template.fromStack(new AgentStack(new App({ context }), name, { env })); }); - - test('warns only when MicroVM layout is implicit, even before an image exists', () => { - const warnings = annotations.findWarning('*', Match.stringLikeRegexp('microvm_nested_stack is unset')); - expect(warnings).toHaveLength(warns ? 1 : 0); - if (warns) { - const message = warnings[0]!.entry.data; - expect(message).toContain('--context microvm_nested_stack=false'); - expect(message).toContain('erase artifacts and pending payloads'); - expect(message).toContain('does not inspect deployed resources or perform a migration'); - expect(message).toContain('docs/verification/645-p3-nested-stack.md'); - } + test('accepts explicit MicroVM layouts and leaves other backends unaffected', () => { + expect(template.toJSON().Resources).toBeDefined(); }); }); @@ -73,10 +64,11 @@ describe('MicroVM resource prefix validation', () => { .toThrow('microvm_resource_name_prefix cannot be used with microvm_nested_stack=false'); }); - test.each([42, true, null])('identifies non-string prefix %s without requiring an explicit true flag', value => { + test.each([42, true, null])('identifies non-string prefix %s in nested mode', value => { const app = new App({ context: { compute_type: 'lambda-microvm', + microvm_nested_stack: true, microvm_resource_name_prefix: value, }, }); @@ -84,10 +76,11 @@ describe('MicroVM resource prefix validation', () => { .toThrow('microvm_resource_name_prefix must be a string'); }); - test('accepts a valid prefix with the default nested layout', () => { + test('accepts a valid prefix with an explicit nested layout', () => { const app = new App({ context: { compute_type: 'lambda-microvm', + microvm_nested_stack: true, microvm_resource_name_prefix: 'migration', }, }); diff --git a/docs/decisions/ADR-021-lambda-microvms-compute-backend.md b/docs/decisions/ADR-021-lambda-microvms-compute-backend.md index 9fa9bdf2c..2ad7ec195 100644 --- a/docs/decisions/ADR-021-lambda-microvms-compute-backend.md +++ b/docs/decisions/ADR-021-lambda-microvms-compute-backend.md @@ -149,7 +149,7 @@ The backend adds build/runtime VPC connectors, build artifacts, launch payloads, `lambda:PassNetworkConnector` has no resource-level authorization support and therefore uses `Resource: *`. The operator role also has wildcard ENI permissions; some mutations can be scoped in IAM, so their current tested wildcard is not evidence that narrower permissions are impossible. DescribeAvailabilityZones needs a wildcard for fresh CDK repository lookups. These exceptions and namespace wildcards are documented in the construct’s cdk-nag suppressions. -**Nested infrastructure.** `microvm_nested_stack` defaults to `true`, putting MicroVM resources in a nested stack. The shared execution role stays in the parent to avoid a role-trust dependency cycle. Bootstrap 1.9.0 covers nested deployment roles. Before upgrading an existing flat installation, set and retain `microvm_nested_stack=false` until completing the reviewed overlap/drain migration, explicit image/coordinator pins and rollback checks described in the [migration prerequisites](../verification/645-p3-nested-stack.md). Reusable migration commands are still being completed. Moving construct paths alone is not a safe migration. Preserve `microvm_resource_name_prefix` after a migrated deployment. +**Nested infrastructure.** `microvm_nested_stack` must be explicitly selected until flat-to-nested migration is verified. New and already-nested deployments should use `true`, putting MicroVM resources in a nested stack. Omission fails synthesis. The shared execution role stays in the parent to avoid a role-trust dependency cycle. Bootstrap 1.9.0 covers nested deployment roles. Before upgrading an existing flat installation, set and retain `microvm_nested_stack=false` until completing the reviewed overlap/drain migration, explicit image/coordinator pins and rollback checks described in the [migration prerequisites](../verification/645-p3-nested-stack.md). Reusable migration commands are still being completed. Moving construct paths alone is not a safe migration. Preserve `microvm_resource_name_prefix` after a migrated deployment. #### Security bar vs existing backends ([#645](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/645) acceptance criterion) diff --git a/docs/design/COMPUTE.md b/docs/design/COMPUTE.md index 2cfaf5f4a..cba02c388 100644 --- a/docs/design/COMPUTE.md +++ b/docs/design/COMPUTE.md @@ -79,7 +79,8 @@ See [ORCHESTRATOR.md](./ORCHESTRATOR.md) for how the orchestrator handles these Lambda MicroVMs are an opt-in third backend, selected per repository with `compute_type: lambda-microvm`; AgentCore remains the default. Image configuration has three states: a managed base-image ARN, version and artifact digest create the snapshot image in CDK; an external image identifier uses a snapshot built out of band; and supplying neither image provisions only the roles, buckets, and connectors needed for the bootstrap deploy. Lambda MicroVMs are available in five launch regions (us-east-1, us-east-2, us-west-2, eu-west-1, ap-northeast-1) and will expand; the platform enforces regional availability in layers via a synth-time constant, onboarding live probes, and orchestration-time classification. -MicroVM infrastructure defaults to a nested stack and requires bootstrap bundle 1.9.0. +New MicroVM installations should select `microvm_nested_stack=true` and require bootstrap bundle 1.9.0. +The layout setting is temporarily required; omission fails synthesis until migration is verified. Before upgrading an existing flat installation, set `microvm_nested_stack=false` and retain it until completing the [resource migration](../verification/645-p3-nested-stack.md). diff --git a/docs/design/DEPLOYMENT_ROLES.md b/docs/design/DEPLOYMENT_ROLES.md index 8be3aa78a..a3b4c464d 100644 --- a/docs/design/DEPLOYMENT_ROLES.md +++ b/docs/design/DEPLOYMENT_ROLES.md @@ -846,7 +846,7 @@ The second statement, `MicrovmPassRoles`, is the one exception to the rule that P3 additionally requires **bundle 1.8.0** for `MicrovmSuspendConfiguration`. -The default nested MicroVM layout requires **bundle 1.9.0**. Its child stack uses the +The nested MicroVM layout requires **bundle 1.9.0**. Its child stack uses the explicit parent-derived names `backgroundagent-dev-MicrovmBuildRole` and `backgroundagent-dev-MicrovmConnectorRole`; `MicrovmPassRoles` admits those two exact names in addition to the legacy flat-layout prefixes. The execution role diff --git a/docs/guides/DEPLOYMENT_GUIDE.md b/docs/guides/DEPLOYMENT_GUIDE.md index a6120dde2..6a902e3fe 100644 --- a/docs/guides/DEPLOYMENT_GUIDE.md +++ b/docs/guides/DEPLOYMENT_GUIDE.md @@ -295,12 +295,13 @@ invocations and retain the old groups when investigating earlier runs. Both flat and nested MicroVM layouts support the optional tool gateway and Linear Identity vault without exceeding the template budget. -MicroVM infrastructure now defaults to a nested stack (bootstrap bundle 1.9.0). +New MicroVM installations must explicitly select `microvm_nested_stack=true` +(bootstrap bundle 1.9.0). Before upgrading an existing flat MicroVM deployment, save `"microvm_nested_stack": false` in its CDK context or pass `--context microvm_nested_stack=false` on every deploy. Keep this escape hatch until completing the [resource migration](../verification/645-p3-nested-stack.md). -Omitting the setting does not automatically migrate existing resources. +Omitting the setting fails synthesis. Selecting `true` does not migrate existing flat resources. ### AgentCore unsupported Availability Zones diff --git a/docs/src/content/docs/architecture/Compute.md b/docs/src/content/docs/architecture/Compute.md index 0c2d2fe4b..41de3b5f9 100644 --- a/docs/src/content/docs/architecture/Compute.md +++ b/docs/src/content/docs/architecture/Compute.md @@ -83,7 +83,8 @@ See [ORCHESTRATOR.md](/sample-autonomous-cloud-coding-agents/architecture/orches Lambda MicroVMs are an opt-in third backend, selected per repository with `compute_type: lambda-microvm`; AgentCore remains the default. Image configuration has three states: a managed base-image ARN, version and artifact digest create the snapshot image in CDK; an external image identifier uses a snapshot built out of band; and supplying neither image provisions only the roles, buckets, and connectors needed for the bootstrap deploy. Lambda MicroVMs are available in five launch regions (us-east-1, us-east-2, us-west-2, eu-west-1, ap-northeast-1) and will expand; the platform enforces regional availability in layers via a synth-time constant, onboarding live probes, and orchestration-time classification. -MicroVM infrastructure defaults to a nested stack and requires bootstrap bundle 1.9.0. +New MicroVM installations should select `microvm_nested_stack=true` and require bootstrap bundle 1.9.0. +The layout setting is temporarily required; omission fails synthesis until migration is verified. Before upgrading an existing flat installation, set `microvm_nested_stack=false` and retain it until completing the [resource migration](/sample-autonomous-cloud-coding-agents/verification/645-p3-nested-stack). diff --git a/docs/src/content/docs/architecture/Deployment-roles.md b/docs/src/content/docs/architecture/Deployment-roles.md index 91725540d..3e09a9c42 100644 --- a/docs/src/content/docs/architecture/Deployment-roles.md +++ b/docs/src/content/docs/architecture/Deployment-roles.md @@ -850,7 +850,7 @@ The second statement, `MicrovmPassRoles`, is the one exception to the rule that P3 additionally requires **bundle 1.8.0** for `MicrovmSuspendConfiguration`. -The default nested MicroVM layout requires **bundle 1.9.0**. Its child stack uses the +The nested MicroVM layout requires **bundle 1.9.0**. Its child stack uses the explicit parent-derived names `backgroundagent-dev-MicrovmBuildRole` and `backgroundagent-dev-MicrovmConnectorRole`; `MicrovmPassRoles` admits those two exact names in addition to the legacy flat-layout prefixes. The execution role diff --git a/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md b/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md index 4dddb11f0..a928ad2cf 100644 --- a/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md +++ b/docs/src/content/docs/decisions/Adr-021-lambda-microvms-compute-backend.md @@ -153,7 +153,7 @@ The backend adds build/runtime VPC connectors, build artifacts, launch payloads, `lambda:PassNetworkConnector` has no resource-level authorization support and therefore uses `Resource: *`. The operator role also has wildcard ENI permissions; some mutations can be scoped in IAM, so their current tested wildcard is not evidence that narrower permissions are impossible. DescribeAvailabilityZones needs a wildcard for fresh CDK repository lookups. These exceptions and namespace wildcards are documented in the construct’s cdk-nag suppressions. -**Nested infrastructure.** `microvm_nested_stack` defaults to `true`, putting MicroVM resources in a nested stack. The shared execution role stays in the parent to avoid a role-trust dependency cycle. Bootstrap 1.9.0 covers nested deployment roles. Before upgrading an existing flat installation, set and retain `microvm_nested_stack=false` until completing the reviewed overlap/drain migration, explicit image/coordinator pins and rollback checks described in the [migration prerequisites](/sample-autonomous-cloud-coding-agents/verification/645-p3-nested-stack). Reusable migration commands are still being completed. Moving construct paths alone is not a safe migration. Preserve `microvm_resource_name_prefix` after a migrated deployment. +**Nested infrastructure.** `microvm_nested_stack` must be explicitly selected until flat-to-nested migration is verified. New and already-nested deployments should use `true`, putting MicroVM resources in a nested stack. Omission fails synthesis. The shared execution role stays in the parent to avoid a role-trust dependency cycle. Bootstrap 1.9.0 covers nested deployment roles. Before upgrading an existing flat installation, set and retain `microvm_nested_stack=false` until completing the reviewed overlap/drain migration, explicit image/coordinator pins and rollback checks described in the [migration prerequisites](/sample-autonomous-cloud-coding-agents/verification/645-p3-nested-stack). Reusable migration commands are still being completed. Moving construct paths alone is not a safe migration. Preserve `microvm_resource_name_prefix` after a migrated deployment. #### Security bar vs existing backends ([#645](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/issues/645) acceptance criterion) diff --git a/docs/src/content/docs/getting-started/Deployment-guide.md b/docs/src/content/docs/getting-started/Deployment-guide.md index 13d868aec..917322dbd 100644 --- a/docs/src/content/docs/getting-started/Deployment-guide.md +++ b/docs/src/content/docs/getting-started/Deployment-guide.md @@ -299,12 +299,13 @@ invocations and retain the old groups when investigating earlier runs. Both flat and nested MicroVM layouts support the optional tool gateway and Linear Identity vault without exceeding the template budget. -MicroVM infrastructure now defaults to a nested stack (bootstrap bundle 1.9.0). +New MicroVM installations must explicitly select `microvm_nested_stack=true` +(bootstrap bundle 1.9.0). Before upgrading an existing flat MicroVM deployment, save `"microvm_nested_stack": false` in its CDK context or pass `--context microvm_nested_stack=false` on every deploy. Keep this escape hatch until completing the [resource migration](/sample-autonomous-cloud-coding-agents/verification/645-p3-nested-stack). -Omitting the setting does not automatically migrate existing resources. +Omitting the setting fails synthesis. Selecting `true` does not migrate existing flat resources. ### AgentCore unsupported Availability Zones diff --git a/docs/src/content/docs/verification/645-p3-nested-stack.md b/docs/src/content/docs/verification/645-p3-nested-stack.md index 377b82a60..0f4c56866 100644 --- a/docs/src/content/docs/verification/645-p3-nested-stack.md +++ b/docs/src/content/docs/verification/645-p3-nested-stack.md @@ -23,7 +23,7 @@ Names derive from the concrete parent deployment name, not the child stack token | Setting | Behavior | |---|---| -| `microvm_nested_stack` | Defaults to `true`. When MicroVM is enabled, omission emits a synthesis warning. Before upgrading an existing flat deployment, set `false` and retain it until completing the reviewed migration. | +| `microvm_nested_stack` | Required when MicroVM is enabled: use `true` for new or already-nested deployments; retain `false` for existing flat deployments until completing the reviewed migration. Omission fails synthesis. | | `microvm_resource_name_prefix` | Supplies distinct names for overlapping nested resources; preserve it after migration. It does not retain old resources or permissions by itself. | | `microvm_managed_image_version` | Pins new tasks to an explicitly verified image version. Without a pin, selection follows the latest active version. | | `microvm_approval_suspend_enabled` | Defaults to `false`; enable only after testing the deployed image/coordinator. Disabling new sleep preserves wake and cleanup. | @@ -39,11 +39,10 @@ published coordinator and exact image version for rollback. Before the first upgrade, save `"microvm_nested_stack": false` in the deployment's CDK context or pass `--context microvm_nested_stack=false` on every deploy. -Omitting the setting now selects nested infrastructure; it does not detect or -migrate existing flat resources. Keep `false` until the migration below is complete. -The synthesis warning is advisory: it does not inspect the deployed stack or block -deployment. New and already-nested installations can explicitly set `true` to -acknowledge the layout; setting it is not a migration. +Omitting the setting fails synthesis. This temporary requirement protects upgrades +that previously omitted the setting; it does not inspect live resources or perform +a migration. New and already-nested installations should explicitly set `true`. +Do not reuse older synthesized assemblies: regenerate them with this version. Do not deploy the nested template directly over a flat deployment. CloudFormation sees removed parent resources and newly created child resources; names can collide diff --git a/docs/verification/645-p3-nested-stack.md b/docs/verification/645-p3-nested-stack.md index 73a979779..8470fa3bf 100644 --- a/docs/verification/645-p3-nested-stack.md +++ b/docs/verification/645-p3-nested-stack.md @@ -19,7 +19,7 @@ Names derive from the concrete parent deployment name, not the child stack token | Setting | Behavior | |---|---| -| `microvm_nested_stack` | Defaults to `true`. When MicroVM is enabled, omission emits a synthesis warning. Before upgrading an existing flat deployment, set `false` and retain it until completing the reviewed migration. | +| `microvm_nested_stack` | Required when MicroVM is enabled: use `true` for new or already-nested deployments; retain `false` for existing flat deployments until completing the reviewed migration. Omission fails synthesis. | | `microvm_resource_name_prefix` | Supplies distinct names for overlapping nested resources; preserve it after migration. It does not retain old resources or permissions by itself. | | `microvm_managed_image_version` | Pins new tasks to an explicitly verified image version. Without a pin, selection follows the latest active version. | | `microvm_approval_suspend_enabled` | Defaults to `false`; enable only after testing the deployed image/coordinator. Disabling new sleep preserves wake and cleanup. | @@ -35,11 +35,10 @@ published coordinator and exact image version for rollback. Before the first upgrade, save `"microvm_nested_stack": false` in the deployment's CDK context or pass `--context microvm_nested_stack=false` on every deploy. -Omitting the setting now selects nested infrastructure; it does not detect or -migrate existing flat resources. Keep `false` until the migration below is complete. -The synthesis warning is advisory: it does not inspect the deployed stack or block -deployment. New and already-nested installations can explicitly set `true` to -acknowledge the layout; setting it is not a migration. +Omitting the setting fails synthesis. This temporary requirement protects upgrades +that previously omitted the setting; it does not inspect live resources or perform +a migration. New and already-nested installations should explicitly set `true`. +Do not reuse older synthesized assemblies: regenerate them with this version. Do not deploy the nested template directly over a flat deployment. CloudFormation sees removed parent resources and newly created child resources; names can collide From 0352a7c8c869c6c10c4f3bdf756a76cc61ba737e Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Tue, 22 Sep 2026 18:47:00 -0400 Subject: [PATCH 137/149] fix(microvm): retire hook-denied calls from SDK error results --- agent/src/runner.py | 6 ++++++ agent/tests/test_runner.py | 16 ++++++++++++++-- 2 files changed, 20 insertions(+), 2 deletions(-) diff --git a/agent/src/runner.py b/agent/src/runner.py index bb86e897e..ac6138b42 100644 --- a/agent/src/runner.py +++ b/agent/src/runner.py @@ -887,6 +887,12 @@ def _on_stderr(line: str) -> None: if isinstance(message.content, list): for block in message.content: if isinstance(block, ToolResultBlock): + # A later project hook can deny a call our pre-hook + # allowed without emitting either post-tool hook. + # The CLI's error result retires only that call; + # unrelated active/background work remains fenced. + if lifecycle is not None and block.is_error: + lifecycle.tool_finished(block.tool_use_id) status, content = _format_tool_result(block) log("RESULT", f"[{status}] {truncate(content)}") tool_name = tool_use_id_to_name.get( diff --git a/agent/tests/test_runner.py b/agent/tests/test_runner.py index 0041d4620..bb862060e 100644 --- a/agent/tests/test_runner.py +++ b/agent/tests/test_runner.py @@ -50,7 +50,7 @@ def _config(**overrides: Any) -> TaskConfig: class TestClaudeSessionOwnership: @pytest.mark.parametrize("microvm", [False, True]) - @pytest.mark.parametrize("failure", [None, "connect", "query", "receive", "cancel"]) + @pytest.mark.parametrize("failure", [None, "connect", "query", "receive", "cancel", "hook-denied"]) def test_broker_selection_and_cleanup_on_every_session_exit( self, monkeypatch, microvm, failure ): @@ -73,6 +73,18 @@ async def messages(): raise RuntimeError("synthetic receive failure") if failure == "cancel": raise asyncio.CancelledError + if failure == "hook-denied": + if context is not None: + await context.tool_started("denied-call") + await context.tool_started("other-active-call") + yield claude_agent_sdk.UserMessage(content=[ + claude_agent_sdk.ToolResultBlock( + tool_use_id="denied-call", content="Project hook denied", is_error=True, + ), + ]) + if context is not None: + assert context.diagnostic_snapshot()["active_tools"] == 1 + assert "other-active-call" in context._tools yield claude_agent_sdk.ResultMessage( subtype="success", duration_ms=1, @@ -106,7 +118,7 @@ async def messages(): result = asyncio.run( runner.run_agent("probe", "probe", config, trajectory=MagicMock()) ) - assert result.status == ("error" if failure else "success") + assert result.status == ("error" if failure not in {None, "hook-denied"} else "success") options = make_client.call_args.kwargs["options"] if microvm: make_broker.assert_called_once_with(context) From c52d4a1656a6f4429955a1ce5363f91478dec2c9 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Tue, 22 Sep 2026 18:47:00 -0400 Subject: [PATCH 138/149] test(microvm): exercise failed replacement cleanup in DynamoDB --- .../shared/microvm-lifecycle-local.test.ts | 58 +++++++++++++++++++ 1 file changed, 58 insertions(+) diff --git a/cdk/test/handlers/shared/microvm-lifecycle-local.test.ts b/cdk/test/handlers/shared/microvm-lifecycle-local.test.ts index 4431ae213..ab75fa2f0 100644 --- a/cdk/test/handlers/shared/microvm-lifecycle-local.test.ts +++ b/cdk/test/handlers/shared/microvm-lifecycle-local.test.ts @@ -32,6 +32,11 @@ if (endpoint && (new URL(endpoint).hostname !== '127.0.0.1' || new URL(endpoint) const mockBeforeSend = jest.fn(); const mockAfterSend = jest.fn(); const mockClients: DynamoDBDocumentClient[] = []; +const mockDeleteContinuations = jest.fn(); +jest.mock('../../../src/handlers/shared/microvm-continuation-storage', () => ({ + ...jest.requireActual('../../../src/handlers/shared/microvm-continuation-storage'), + deleteClosedTaskContinuations: (...args: unknown[]) => mockDeleteContinuations(...args), +})); jest.mock('../../../src/handlers/shared/microvm-suspend-config', () => ({ readMicrovmSuspendEnabled: async () => true, })); @@ -64,6 +69,9 @@ Object.assign(process.env, { TASK_TABLE_NAME: tasks, TASK_APPROVALS_TABLE_NAME: import { readMicrovmLifecycleSnapshot, saveMicrovmLifecycleIntent } from '../../../src/handlers/shared/microvm-lifecycle'; import { claimMicrovmStart, saveMicrovmImageCapability, saveMicrovmStartHandle } from '../../../src/handlers/shared/microvm-start'; import { superviseMicrovm, type MicrovmSupervisorState } from '../../../src/handlers/shared/microvm-supervisor'; +import { failContinuationAttempt } from '../../../src/handlers/shared/microvm-continuation-runner'; +import { reconcileMicrovmContinuation } from '../../../src/handlers/reconcile-microvm-continuations'; +import { workerLeaseKey } from '../../../src/handlers/shared/microvm-continuation-types'; const raw = new DynamoDBClient({ endpoint: endpoint ?? 'http://127.0.0.1:1', @@ -90,6 +98,7 @@ local('MicroVM lifecycle against DynamoDB Local', () => { beforeEach(async () => { mockBeforeSend.mockReset(); mockAfterSend.mockReset(); + mockDeleteContinuations.mockReset(); for (const name of [tasks, approvals]) { const result = await admin.send(new ScanCommand({ TableName: name })); for (const item of result.Items ?? []) { @@ -145,6 +154,55 @@ local('MicroVM lifecycle against DynamoDB Local', () => { return value; } + test.each([false, true])('cleans an early failed replacement, preserving a concurrent worker ID (%s)', async lateWorker => { + const attempt = randomUUID(); + await admin.send(new PutCommand({ + TableName: tasks, + Item: { + task_id: 'task', user_id: 'user', status: 'AWAITING_APPROVAL', compute_type: 'lambda-microvm', + continuation: { state: 'STARTING', attempt_id: attempt }, + concurrency_slot: { state: 'held', attempt_id: attempt }, + }, + })); + await admin.send(new PutCommand({ + TableName: tasks, + Item: { ...workerLeaseKey('task'), lease_user_id: 'user', lease_attempt_id: attempt, lease_state: 'ACTIVE' }, + })); + await failContinuationAttempt({ + task_id: 'task', continuation_request_id: 'gate', continuation_attempt_id: attempt, + }, 'user', 'MICROVM_CONTINUATION_VERSION_CHANGED'); + expect((await admin.send(new GetCommand({ TableName: tasks, Key: workerLeaseKey('task') }))).Item?.lease_state) + .toBe('FENCED'); + // Finalization can release a no-receipt attempt before the scheduled sweep. + await admin.send(new UpdateCommand({ + TableName: tasks, Key: { task_id: 'task' }, + UpdateExpression: 'SET concurrency_slot.#state = :released', + ExpressionAttributeNames: { '#state': 'state' }, ExpressionAttributeValues: { ':released': 'released' }, + })); + expect(await claimMicrovmStart('task', 'user', 'hash', attempt)).toMatchObject({ closed: true }); + if (lateWorker) { + mockBeforeSend.mockImplementation(async command => { + if (command instanceof UpdateCommand && command.input.UpdateExpression?.includes('lease_state = :closed')) { + await admin.send(new UpdateCommand({ + TableName: tasks, Key: workerLeaseKey('task'), + UpdateExpression: 'SET lease_microvm_id = :id', ExpressionAttributeValues: { ':id': 'late-worker' }, + })); + } + }); + } + const row = (await admin.send(new GetCommand({ TableName: tasks, Key: { task_id: 'task' } }))).Item!; + if (lateWorker) { + await expect(reconcileMicrovmContinuation(row as Parameters[0])) + .rejects.toThrow('The conditional request failed'); + expect(mockDeleteContinuations).not.toHaveBeenCalled(); + } else { + await reconcileMicrovmContinuation(row as Parameters[0]); + expect((await admin.send(new GetCommand({ TableName: tasks, Key: workerLeaseKey('task') }))).Item?.lease_state) + .toBe('CLOSED'); + expect(mockDeleteContinuations).toHaveBeenCalledWith('task', 'user', expect.any(Object)); + } + }); + test('real supervisor and store preserve approval-during-suspend wake across serialized polls', async () => { const handle = (await current()).handle; let observed = 'RUNNING'; From c9c8759e4eab20eb71dc088ca0049b87fa0aa9f9 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Tue, 22 Sep 2026 18:50:53 -0400 Subject: [PATCH 139/149] fix(microvm): default uv installs to checkpoint-safe copies --- agent/README.md | 3 +++ agent/src/server.py | 4 ++++ agent/tests/test_server.py | 2 ++ 3 files changed, 9 insertions(+) diff --git a/agent/README.md b/agent/README.md index e1ca763b7..0a48ff11c 100644 --- a/agent/README.md +++ b/agent/README.md @@ -331,6 +331,9 @@ and untracked/ignored files, with a 1 GiB / 100,000-entry default limit. Restore rebuilds Git configuration and refuses to replace an existing destination. Unsupported filesystem/Git states or detected concurrent writes prevent capture. Repository-free tasks use a private workspace with a local Git baseline. +MicroVM tasks default `UV_LINK_MODE=copy` before repository setup so `uv` does not +hardlink installed packages to its cache. Explicit overrides or commands using +hardlinks can still make the workspace ineligible for capture. `S3ContinuationStorage` uploads and verifies version-pinned, checksummed objects using task-scoped credentials. The continuation bucket and SessionRole grants diff --git a/agent/src/server.py b/agent/src/server.py index 4091f086b..e62b0ab3d 100644 --- a/agent/src/server.py +++ b/agent/src/server.py @@ -471,6 +471,10 @@ def _run_task_background( lifecycle = ( register_task(task_id, microvm_id, attempt_id=attempt_id or task_id) if microvm_id else None ) + if lifecycle is not None: + # Linux uv defaults to cache hardlinks, which checkpoint capture rejects + # because another path can mutate the same inode outside the workspace. + os.environ.setdefault("UV_LINK_MODE", "copy") stop_heartbeat = threading.Event() hb_thread: threading.Thread | None = None try: diff --git a/agent/tests/test_server.py b/agent/tests/test_server.py index 460a186ec..9350ef1cc 100644 --- a/agent/tests/test_server.py +++ b/agent/tests/test_server.py @@ -3037,6 +3037,7 @@ def test_microvm_pipeline_registers_identity_and_always_removes_lifecycle( ): from microvm_lifecycle import get_context + monkeypatch.delenv("UV_LINK_MODE", raising=False) observed = [] def run_task(**kwargs): @@ -3045,6 +3046,7 @@ def run_task(**kwargs): assert context.microvm_id == "microvm-server" assert context.attempt_id == (attempt or "lifecycle-server-task") assert "microvm_id" not in kwargs + assert os.environ["UV_LINK_MODE"] == "copy" observed.append(context) if crash: raise RuntimeError("pipeline crashed") From 564d9a8478b6bdd6a929cfb89a6eac9da4bf3c6e Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Tue, 22 Sep 2026 18:50:53 -0400 Subject: [PATCH 140/149] fix(microvm): confirm worker state after termination conflicts --- cdk/src/handlers/shared/compute-strategy.ts | 2 +- .../strategies/lambda-microvm-strategy.ts | 19 +++++++++++++++++++ .../lambda-microvm-strategy.test.ts | 15 +++++++++++++++ 3 files changed, 35 insertions(+), 1 deletion(-) diff --git a/cdk/src/handlers/shared/compute-strategy.ts b/cdk/src/handlers/shared/compute-strategy.ts index cf6ff4e97..f4070d37c 100644 --- a/cdk/src/handlers/shared/compute-strategy.ts +++ b/cdk/src/handlers/shared/compute-strategy.ts @@ -116,7 +116,7 @@ export type SessionLifecycleResult = /** Optional evidence from best-effort cleanup. A request is not confirmed teardown. */ export type SessionStopResult = - | { readonly outcome: 'requested' | 'not-found' } + | { readonly outcome: 'requested' | 'not-found' | 'terminated' } | { readonly outcome: 'unconfirmed'; readonly error_type: string; readonly aws_request_id?: string }; export interface ComputeStrategy { diff --git a/cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts b/cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts index cea997852..0ce6020ba 100644 --- a/cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts +++ b/cdk/src/handlers/shared/strategies/lambda-microvm-strategy.ts @@ -916,6 +916,25 @@ export class LambdaMicrovmComputeStrategy implements ComputeStrategy { } catch (err) { const identity = microvmErrorIdentity(err); const errName = identity.error_type; + if (errName === 'ConflictException') { + try { + const observed = await getClient().send(new GetMicrovmCommand({ + microvmIdentifier: microvmId, + }), { abortSignal: controlSignal(options) }); + if (observed?.state === 'TERMINATED') { + logger.info('MicroVM already terminated during concurrent cleanup', { microvm_id: microvmId, reason }); + return { outcome: 'terminated' }; + } + } catch (confirmationError) { + if (microvmErrorIdentity(confirmationError).error_type === 'ResourceNotFoundException') { + logger.info('MicroVM no longer found during concurrent cleanup', { microvm_id: microvmId, reason }); + return { outcome: 'not-found' }; + } + logger.warn('MicroVM termination conflict could not be confirmed', { + microvm_id: microvmId, ...microvmErrorIdentity(confirmationError), + }); + } + } if (errName === 'ResourceNotFoundException') { logger.info('MicroVM no longer found during termination', { microvm_id: microvmId, diff --git a/cdk/test/handlers/shared/strategies/lambda-microvm-strategy.test.ts b/cdk/test/handlers/shared/strategies/lambda-microvm-strategy.test.ts index 27d2b62ec..bcef69013 100644 --- a/cdk/test/handlers/shared/strategies/lambda-microvm-strategy.test.ts +++ b/cdk/test/handlers/shared/strategies/lambda-microvm-strategy.test.ts @@ -817,6 +817,21 @@ describe('LambdaMicrovmComputeStrategy', () => { expect(call.input).toEqual({ microvmIdentifier: MICROVM_ID }); }); + test.each(['TERMINATED', 'TERMINATING', 'RUNNING'])('confirms termination conflicts against %s', async state => { + mockSend.mockRejectedValueOnce(Object.assign(new Error('conflict'), { name: 'ConflictException' })) + .mockResolvedValueOnce({ state }); + const result = await new LambdaMicrovmComputeStrategy().stopSession(makeHandle()); + expect(result).toEqual(state === 'TERMINATED' ? { outcome: 'terminated' } + : { outcome: 'unconfirmed', error_type: 'ConflictException' }); + expect(mockSend.mock.calls[1][0]._type).toBe('GetMicrovm'); + }); + + test('confirms an absent worker after a termination conflict', async () => { + mockSend.mockRejectedValueOnce(Object.assign(new Error('conflict'), { name: 'ConflictException' })) + .mockRejectedValueOnce(Object.assign(new Error('gone'), { name: 'ResourceNotFoundException' })); + await expect(new LambdaMicrovmComputeStrategy().stopSession(makeHandle())).resolves.toEqual({ outcome: 'not-found' }); + }); + test.each([ ['ResourceNotFoundException', 'info'], ['ConflictException', 'warn'], From 63ef34dc9e25677052fd6e5d3d9330c9fab32cc4 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Tue, 22 Sep 2026 18:59:03 -0400 Subject: [PATCH 141/149] fix(approvals): explain inactive MicroVM worker leases --- agent/src/approval_requests.py | 14 +++++++++++++- agent/tests/test_approval_requests.py | 3 ++- 2 files changed, 15 insertions(+), 2 deletions(-) diff --git a/agent/src/approval_requests.py b/agent/src/approval_requests.py index 16b4785ad..31f7e2683 100644 --- a/agent/src/approval_requests.py +++ b/agent/src/approval_requests.py @@ -106,11 +106,23 @@ def record_request( ) details = error.get("details") if isinstance(error, dict) else None reasons = details.get("cancellation_reasons", []) if isinstance(details, dict) else [] + lease_failed = False + # The third transaction item authorizes the MicroVM worker lease. + match reasons: + case [_, _, {"Code": "ConditionalCheckFailed"}]: + lease_failed = code == "TransactionCanceledException" + message = ( + "MicroVM approval worker lease is missing or inactive. Check worker ownership; " + "pre-upgrade tasks must be drained and resubmitted with matching worker images " + "and coordinator versions." + if lease_failed + else "Control plane did not acknowledge the approval write" + ) raise ClientError( { "Error": { "Code": code, - "Message": "Control plane did not acknowledge the approval write", + "Message": message, }, "CancellationReasons": reasons, "ResponseMetadata": { diff --git a/agent/tests/test_approval_requests.py b/agent/tests/test_approval_requests.py index 373853db9..1b649ddcf 100644 --- a/agent/tests/test_approval_requests.py +++ b/agent/tests/test_approval_requests.py @@ -124,5 +124,6 @@ def test_timeout_race_preserves_human_winner_but_lease_loss_is_not_benign(transp ) assert not task_state.best_effort_update_approval_status("task", "request", "TIMED_OUT") reasons[-1] = {"Code": "ConditionalCheckFailed"} - with pytest.raises(ClientError): + with pytest.raises(ClientError, match="pre-upgrade tasks must be drained") as failure: task_state.best_effort_update_approval_status("task", "request", "TIMED_OUT") + assert failure.value.response["Error"]["Code"] == "TransactionCanceledException" From e1a099c9c43a9a671f08524311ab9fc2fa4443e7 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Tue, 22 Sep 2026 18:59:03 -0400 Subject: [PATCH 142/149] docs(microvm): correct approval defaults and live verification commands --- cdk/src/handlers/shared/types.ts | 2 +- cdk/test/live/README.md | 33 +++++++++++++++++++ docs/design/CEDAR_HITL_GATES.md | 22 +++++++------ docs/guides/DEPLOYMENT_GUIDE.md | 7 ++-- docs/guides/LINEAR_SETUP_GUIDE.md | 2 +- docs/guides/QUICK_START.mdx | 2 +- .../docs/architecture/Cedar-hitl-gates.md | 22 +++++++------ .../docs/getting-started/Deployment-guide.md | 7 ++-- .../docs/getting-started/Quick-start.mdx | 2 +- .../content/docs/using/Linear-setup-guide.md | 2 +- 10 files changed, 72 insertions(+), 29 deletions(-) create mode 100644 cdk/test/live/README.md diff --git a/cdk/src/handlers/shared/types.ts b/cdk/src/handlers/shared/types.ts index 6d890e817..2faf71e4f 100644 --- a/cdk/src/handlers/shared/types.ts +++ b/cdk/src/handlers/shared/types.ts @@ -1392,7 +1392,7 @@ export interface TimedOutApprovalRecord extends ApprovalRecordBase { } /** STRANDED approval row, when explicitly recorded. The current stranded-task - * reconciler closes the owning task and leaves its approval row PENDING. */ + * reconciler closes the owning task and cancels its unanswered approval rows. */ export interface StrandedApprovalRecord extends ApprovalRecordBase { readonly status: 'STRANDED'; readonly decided_at: string; diff --git a/cdk/test/live/README.md b/cdk/test/live/README.md new file mode 100644 index 000000000..0b9fa8955 --- /dev/null +++ b/cdk/test/live/README.md @@ -0,0 +1,33 @@ +# MicroVM live probes + +These scripts create synthetic task records, S3 objects and short-lived workers +in an existing deployment. They are excluded from Jest. Use an owned, quiet test +deployment with `NO_INGRESS`, a working image version, AWS CLI v2, and credentials +authorized for its resources. Concurrent tasks can invalidate worker attribution. + +From `cdk/`, inspect cases and effects without calling AWS: + +```sh +mise exec -- node -r ts-node/register/transpile-only test/live/verify-microvm-start.live.ts +mise exec -- node -r ts-node/register/transpile-only test/live/verify-microvm-payload.live.ts +mise exec -- node -r ts-node/register/transpile-only test/live/verify-microvm-replay.live.ts +``` + +To execute, supply every target field and a private output directory: + +```sh +mise exec -- node -r ts-node/register/transpile-only test/live/verify-microvm-start.live.ts \ + --execute --account YOUR_ACCOUNT_ID --region YOUR_REGION \ + --stack YOUR_STACK --image-version YOUR_IMAGE_VERSION \ + --output /tmp/microvm-start-verification +``` + +Use the same arguments for the other two scripts, with separate output +directories. Start/payload probes also accept `--cases` with comma-separated +names from their inspection output. The replay probe includes waits beyond five +minutes. An assertion failure is a failed verification, even if some cases pass. +Inspect the output and worker inventory after interruption; do not assume cleanup +completed. Never publish raw output containing task data or payload URLs. + +The scripts test launch recovery, payload handling and service replay behavior. +They do not replace the [approval and replacement acceptance checks](../../../docs/verification/README.md). diff --git a/docs/design/CEDAR_HITL_GATES.md b/docs/design/CEDAR_HITL_GATES.md index ea442af7d..0ffe3b42f 100644 --- a/docs/design/CEDAR_HITL_GATES.md +++ b/docs/design/CEDAR_HITL_GATES.md @@ -162,7 +162,7 @@ Settled during the 2026-04-23 design discussion and extended after the 2026-04-2 ## 4. End-to-end request flow -Narrative walk-through of the happy path. Sequence diagrams in the round-trip Mermaid below. +Narrative walk-through with an explicit 600-second task deadline and a custom `force_push_any` policy annotated with `@approval_timeout_s("300")`. Built-in starter rules do not set deadlines. Sequence diagrams are below. ### Setup (task start) @@ -212,7 +212,7 @@ Narrative walk-through of the happy path. Sequence diagrams in the round-trip Me ) → effective = 300s ``` - If `maxLifetime_remaining_s - CLEANUP_MARGIN_120S < FLOOR_30S`, hook returns DENY immediately with reason `"insufficient lifetime for approval"` (§13.7). + This lifetime ceiling applies to workers without continuation support. A continuation-capable MicroVM omits it because the approval can outlive the worker. With no positive task/rule deadline, the effective timeout is zero (no decision deadline). 12. Hook checks per-task approval-gate cap (default 50, configurable per blueprint via `security.approvalGateCap`; §5.1) and per-minute rate limit (20/task, per-container). If either exceeded → DENY with reason `"approval-gate cap exceeded"` (fail-closed). 13. Hook mints `request_id = _ulid()` (26-char ULID). @@ -449,7 +449,7 @@ forbid (principal, action == Agent::Action::"execute_bash", resource) when { context.command like "*DROP TABLE*" }; ``` -**Gate destructive git ops** (soft-deny — part of the built-in starter set): +**Gate destructive git ops** (custom timed variants of the built-in starter rules): ```cedar @tier("soft") @rule_id("force_push_any") @@ -484,7 +484,7 @@ forbid (principal, action == Agent::Action::"execute_bash", resource) A force-push to any branch needs approval in 300s. A force-push to `main` or `prod` gives the user 600s with elevated severity. A non-force push to a protected branch (`main`/`prod`/`master`/`release/*`) also gates — catches the case where an agent directly pushes rather than opening a PR. If a command matches both `force_push_any` and `force_push_main`, multi-match merging picks `min(300, 600) = 300s` and `max(medium, high) = high`. -**Protect sensitive file paths** (soft-deny — part of the built-in starter set): +**Protect sensitive file paths** (custom timed variants of the built-in starter rules): ```cedar @tier("soft") @rule_id("write_env_files") @@ -2126,11 +2126,13 @@ Built-in policies shipped with the agent: **Hard-deny (absolute, cannot be disabled by blueprint)**: `rm_slash`, `write_git_internals`, `write_git_internals_nested`, `drop_table`. Absolute; no scope bypasses them; blueprint `disable:` cannot remove them (§5.1, finding #9). **Soft-deny starter set (require approval by default, may be disabled by blueprint)**: -- `force_push_any` — `like "*git push --force*"` — medium, 300s -- `push_to_protected_branch` — pushes to `main`/`master`/`prod`/`release/*` (non-force) — medium, 300s -- `force_push_main` — force-push specifically to `main`/`prod` — high, 600s -- `write_env_files` — `like "*.env"` — high, 600s -- `write_credentials` — `like "*credentials*"` — high, 300s +- `force_push_any` — `like "*git push --force*"` — medium +- `push_to_protected_branch` — pushes to `main`/`master`/`prod`/`release/*` (non-force) — medium +- `force_push_main` — force-push specifically to `main`/`prod` — high +- `write_env_files` — `like "*.env"` — high +- `write_credentials` — `like "*credentials*"` — high + +These built-in rules inherit the task deadline: zero/no deadline by default. Custom rule annotations may impose a positive deadline. Users who want fully autonomous execution (no approval gates) pass `--pre-approve all_session --yes` at submit. Repos that want additional gates add them via `Blueprint.security.cedarPolicies.soft`. Repos that want a different policy set can override specific built-in **soft-deny** rules by `@rule_id` via the blueprint's `security.cedarPolicies.disable` list. The `disable:` mechanism is restricted: it may NOT include any built-in hard-deny rule_id, and the blueprint loader rejects such configurations at task start. @@ -2512,7 +2514,7 @@ See §15.2. Net new files: ~15. Net modified files: ~15. Total LOC estimate: ~40 - [ ] Backward compat: Phase 1a/1b tests pass without modification - [ ] ULID length references are 26 chars throughout CLI + docs - [ ] **Re-read approval row on TIMED_OUT ConditionCheckFailed (IMPL-24)**: `_best_effort_update_status("TIMED_OUT")` failure path re-reads with ConsistentRead and honors APPROVED/DENIED if the user's decision beat the agent's timer; emits `approval_late_win` milestone. See §6.5 pseudocode, §13.12 VM-throttle race, §14.6 trace, §15.2 task #43. -- [ ] **Default `--approval-timeout` is 300s** documented consistently in decision #6, §5.2, §7.3 field table, §8.2 CLI flags, and §10.2 TaskTable schema. +- [ ] **Default `--approval-timeout` is 0 (no deadline)** documented consistently in decision #6, §5.2, §7.3 field table, §8.2 CLI flags, and §10.2 TaskTable schema. - [ ] **Sub-120s `@approval_timeout_s` emits WARN (IMPL-25)** at blueprint load; sub-30s still rejected. `bgagent lint-policies` (§17.14) surfaces the same WARN pre-submit. - [ ] **User-visible timeout milestones (IMPL-26)**: `approval_timeout_capped` (per-gate, on SSE stream), `approval_timeout_capped_at_submit` (on `POST /v1/tasks` response), `approval_ceiling_shrinking` (once per task at lifetime threshold). All carry `{requested_timeout_s, effective_timeout_s, reason}`. - [ ] **Runtime JWT ceiling (IMPL-27)**: no separate JWT expiry term required in v1 — container uses auto-refreshed IAM credentials (verified by grep of `agent/src/`). Ceiling stays `min(1h, maxLifetime_remaining - cleanup_margin)`. Review if container auth shape changes (see §13.13). diff --git a/docs/guides/DEPLOYMENT_GUIDE.md b/docs/guides/DEPLOYMENT_GUIDE.md index 6a902e3fe..ab8696654 100644 --- a/docs/guides/DEPLOYMENT_GUIDE.md +++ b/docs/guides/DEPLOYMENT_GUIDE.md @@ -23,12 +23,15 @@ ECS Fargate is **opt-in**. Deploy with `--context compute_type=ecs`; the stack e > **Not for production.** `lambda-microvm` carries no smoke-parity guarantee for an unattended deployment. Keep production repositories on `agentcore` or `ecs`. Synth emits a verification warning whenever a MicroVM image is configured; selecting the backend without an image emits a separate setup warning. Design detail: [COMPUTE.md](../design/COMPUTE.md) and [ADR-021](../decisions/ADR-021-lambda-microvms-compute-backend.md). -Selecting it is a synth-time context flag: +For a new installation, select the backend and nested layout: ```bash -mise //cdk:deploy -- --context compute_type=lambda-microvm +mise //cdk:deploy -- --context compute_type=lambda-microvm --context microvm_nested_stack=true ``` +Existing flat installations must retain `microvm_nested_stack=false` until +completing the [resource migration](../verification/645-p3-nested-stack.md). + **You must re-bootstrap first.** This is the single most common way this backend fails, and the failure does not look like a configuration problem: 1. Check the bootstrap policy bundle already deployed in the account: diff --git a/docs/guides/LINEAR_SETUP_GUIDE.md b/docs/guides/LINEAR_SETUP_GUIDE.md index fe5ebf5cf..0ce9f5ae3 100644 --- a/docs/guides/LINEAR_SETUP_GUIDE.md +++ b/docs/guides/LINEAR_SETUP_GUIDE.md @@ -38,7 +38,7 @@ When a workspace's authorization dies, ABCA records it on the registry row and p #### Using the vault with Lambda MicroVMs -Deploy with both `compute_type=lambda-microvm` and `enableLinearIdentityVault=true`. The coordinator sends the workload identity name through authenticated `platform_config`; the guest uses its compute execution role to obtain a Linear token. Credentials are not baked into the MicroVM image. +Deploy with `compute_type=lambda-microvm`, `enableLinearIdentityVault=true` and an explicit `microvm_nested_stack` value (`true` for new/already-nested installations; retain `false` for existing flat installations until migration). The coordinator sends the workload identity name through authenticated `platform_config`; the guest uses its compute execution role to obtain a Linear token. Credentials are not baked into the MicroVM image. When upgrading an existing MicroVM deployment, rebuild the guest image too: the coordinator and guest must both support the vault configuration fields. diff --git a/docs/guides/QUICK_START.mdx b/docs/guides/QUICK_START.mdx index 6322efc73..8b01d97a8 100644 --- a/docs/guides/QUICK_START.mdx +++ b/docs/guides/QUICK_START.mdx @@ -480,7 +480,7 @@ node lib/bin/bgagent.js deny \ --reason "Don't force-push shared branches; open a revert PR instead" ``` -The task transitions back to `RUNNING` immediately on a decision. The denial reason is injected into the agent's context so it can adapt rather than retry the same tool call. If no decision arrives within the rule's timeout (300 s by default), the gate is treated as a denial with `timed_out` as the reason. +A decision lets the task continue; a sleeping or retired MicroVM first wakes or restores its saved work. The denial reason is passed to the agent so it can adapt. Requests have no decision deadline by default. If an explicit task or rule deadline expires, the gate is treated as a denial with `timed_out` as the reason. Worker lifetime limits remain separate. If you want a task to run without interactive gates (e.g. an unattended overnight job), pre-approve the scopes you trust up-front: diff --git a/docs/src/content/docs/architecture/Cedar-hitl-gates.md b/docs/src/content/docs/architecture/Cedar-hitl-gates.md index 63e962ab7..94c163f08 100644 --- a/docs/src/content/docs/architecture/Cedar-hitl-gates.md +++ b/docs/src/content/docs/architecture/Cedar-hitl-gates.md @@ -166,7 +166,7 @@ Settled during the 2026-04-23 design discussion and extended after the 2026-04-2 ## 4. End-to-end request flow -Narrative walk-through of the happy path. Sequence diagrams in the round-trip Mermaid below. +Narrative walk-through with an explicit 600-second task deadline and a custom `force_push_any` policy annotated with `@approval_timeout_s("300")`. Built-in starter rules do not set deadlines. Sequence diagrams are below. ### Setup (task start) @@ -216,7 +216,7 @@ Narrative walk-through of the happy path. Sequence diagrams in the round-trip Me ) → effective = 300s ``` - If `maxLifetime_remaining_s - CLEANUP_MARGIN_120S < FLOOR_30S`, hook returns DENY immediately with reason `"insufficient lifetime for approval"` (§13.7). + This lifetime ceiling applies to workers without continuation support. A continuation-capable MicroVM omits it because the approval can outlive the worker. With no positive task/rule deadline, the effective timeout is zero (no decision deadline). 12. Hook checks per-task approval-gate cap (default 50, configurable per blueprint via `security.approvalGateCap`; §5.1) and per-minute rate limit (20/task, per-container). If either exceeded → DENY with reason `"approval-gate cap exceeded"` (fail-closed). 13. Hook mints `request_id = _ulid()` (26-char ULID). @@ -453,7 +453,7 @@ forbid (principal, action == Agent::Action::"execute_bash", resource) when { context.command like "*DROP TABLE*" }; ``` -**Gate destructive git ops** (soft-deny — part of the built-in starter set): +**Gate destructive git ops** (custom timed variants of the built-in starter rules): ```cedar @tier("soft") @rule_id("force_push_any") @@ -488,7 +488,7 @@ forbid (principal, action == Agent::Action::"execute_bash", resource) A force-push to any branch needs approval in 300s. A force-push to `main` or `prod` gives the user 600s with elevated severity. A non-force push to a protected branch (`main`/`prod`/`master`/`release/*`) also gates — catches the case where an agent directly pushes rather than opening a PR. If a command matches both `force_push_any` and `force_push_main`, multi-match merging picks `min(300, 600) = 300s` and `max(medium, high) = high`. -**Protect sensitive file paths** (soft-deny — part of the built-in starter set): +**Protect sensitive file paths** (custom timed variants of the built-in starter rules): ```cedar @tier("soft") @rule_id("write_env_files") @@ -2130,11 +2130,13 @@ Built-in policies shipped with the agent: **Hard-deny (absolute, cannot be disabled by blueprint)**: `rm_slash`, `write_git_internals`, `write_git_internals_nested`, `drop_table`. Absolute; no scope bypasses them; blueprint `disable:` cannot remove them (§5.1, finding #9). **Soft-deny starter set (require approval by default, may be disabled by blueprint)**: -- `force_push_any` — `like "*git push --force*"` — medium, 300s -- `push_to_protected_branch` — pushes to `main`/`master`/`prod`/`release/*` (non-force) — medium, 300s -- `force_push_main` — force-push specifically to `main`/`prod` — high, 600s -- `write_env_files` — `like "*.env"` — high, 600s -- `write_credentials` — `like "*credentials*"` — high, 300s +- `force_push_any` — `like "*git push --force*"` — medium +- `push_to_protected_branch` — pushes to `main`/`master`/`prod`/`release/*` (non-force) — medium +- `force_push_main` — force-push specifically to `main`/`prod` — high +- `write_env_files` — `like "*.env"` — high +- `write_credentials` — `like "*credentials*"` — high + +These built-in rules inherit the task deadline: zero/no deadline by default. Custom rule annotations may impose a positive deadline. Users who want fully autonomous execution (no approval gates) pass `--pre-approve all_session --yes` at submit. Repos that want additional gates add them via `Blueprint.security.cedarPolicies.soft`. Repos that want a different policy set can override specific built-in **soft-deny** rules by `@rule_id` via the blueprint's `security.cedarPolicies.disable` list. The `disable:` mechanism is restricted: it may NOT include any built-in hard-deny rule_id, and the blueprint loader rejects such configurations at task start. @@ -2516,7 +2518,7 @@ See §15.2. Net new files: ~15. Net modified files: ~15. Total LOC estimate: ~40 - [ ] Backward compat: Phase 1a/1b tests pass without modification - [ ] ULID length references are 26 chars throughout CLI + docs - [ ] **Re-read approval row on TIMED_OUT ConditionCheckFailed (IMPL-24)**: `_best_effort_update_status("TIMED_OUT")` failure path re-reads with ConsistentRead and honors APPROVED/DENIED if the user's decision beat the agent's timer; emits `approval_late_win` milestone. See §6.5 pseudocode, §13.12 VM-throttle race, §14.6 trace, §15.2 task #43. -- [ ] **Default `--approval-timeout` is 300s** documented consistently in decision #6, §5.2, §7.3 field table, §8.2 CLI flags, and §10.2 TaskTable schema. +- [ ] **Default `--approval-timeout` is 0 (no deadline)** documented consistently in decision #6, §5.2, §7.3 field table, §8.2 CLI flags, and §10.2 TaskTable schema. - [ ] **Sub-120s `@approval_timeout_s` emits WARN (IMPL-25)** at blueprint load; sub-30s still rejected. `bgagent lint-policies` (§17.14) surfaces the same WARN pre-submit. - [ ] **User-visible timeout milestones (IMPL-26)**: `approval_timeout_capped` (per-gate, on SSE stream), `approval_timeout_capped_at_submit` (on `POST /v1/tasks` response), `approval_ceiling_shrinking` (once per task at lifetime threshold). All carry `{requested_timeout_s, effective_timeout_s, reason}`. - [ ] **Runtime JWT ceiling (IMPL-27)**: no separate JWT expiry term required in v1 — container uses auto-refreshed IAM credentials (verified by grep of `agent/src/`). Ceiling stays `min(1h, maxLifetime_remaining - cleanup_margin)`. Review if container auth shape changes (see §13.13). diff --git a/docs/src/content/docs/getting-started/Deployment-guide.md b/docs/src/content/docs/getting-started/Deployment-guide.md index 917322dbd..6dc802e44 100644 --- a/docs/src/content/docs/getting-started/Deployment-guide.md +++ b/docs/src/content/docs/getting-started/Deployment-guide.md @@ -27,12 +27,15 @@ ECS Fargate is **opt-in**. Deploy with `--context compute_type=ecs`; the stack e > **Not for production.** `lambda-microvm` carries no smoke-parity guarantee for an unattended deployment. Keep production repositories on `agentcore` or `ecs`. Synth emits a verification warning whenever a MicroVM image is configured; selecting the backend without an image emits a separate setup warning. Design detail: [COMPUTE.md](/sample-autonomous-cloud-coding-agents/architecture/compute) and [ADR-021](/sample-autonomous-cloud-coding-agents/decisions/adr-021-lambda-microvms-compute-backend). -Selecting it is a synth-time context flag: +For a new installation, select the backend and nested layout: ```bash -mise //cdk:deploy -- --context compute_type=lambda-microvm +mise //cdk:deploy -- --context compute_type=lambda-microvm --context microvm_nested_stack=true ``` +Existing flat installations must retain `microvm_nested_stack=false` until +completing the [resource migration](/sample-autonomous-cloud-coding-agents/verification/645-p3-nested-stack). + **You must re-bootstrap first.** This is the single most common way this backend fails, and the failure does not look like a configuration problem: 1. Check the bootstrap policy bundle already deployed in the account: diff --git a/docs/src/content/docs/getting-started/Quick-start.mdx b/docs/src/content/docs/getting-started/Quick-start.mdx index dde3a92a4..010ce7ef4 100644 --- a/docs/src/content/docs/getting-started/Quick-start.mdx +++ b/docs/src/content/docs/getting-started/Quick-start.mdx @@ -480,7 +480,7 @@ node lib/bin/bgagent.js deny \ --reason "Don't force-push shared branches; open a revert PR instead" ``` -The task transitions back to `RUNNING` immediately on a decision. The denial reason is injected into the agent's context so it can adapt rather than retry the same tool call. If no decision arrives within the rule's timeout (300 s by default), the gate is treated as a denial with `timed_out` as the reason. +A decision lets the task continue; a sleeping or retired MicroVM first wakes or restores its saved work. The denial reason is passed to the agent so it can adapt. Requests have no decision deadline by default. If an explicit task or rule deadline expires, the gate is treated as a denial with `timed_out` as the reason. Worker lifetime limits remain separate. If you want a task to run without interactive gates (e.g. an unattended overnight job), pre-approve the scopes you trust up-front: diff --git a/docs/src/content/docs/using/Linear-setup-guide.md b/docs/src/content/docs/using/Linear-setup-guide.md index 6647b6924..e78572f25 100644 --- a/docs/src/content/docs/using/Linear-setup-guide.md +++ b/docs/src/content/docs/using/Linear-setup-guide.md @@ -42,7 +42,7 @@ When a workspace's authorization dies, ABCA records it on the registry row and p #### Using the vault with Lambda MicroVMs -Deploy with both `compute_type=lambda-microvm` and `enableLinearIdentityVault=true`. The coordinator sends the workload identity name through authenticated `platform_config`; the guest uses its compute execution role to obtain a Linear token. Credentials are not baked into the MicroVM image. +Deploy with `compute_type=lambda-microvm`, `enableLinearIdentityVault=true` and an explicit `microvm_nested_stack` value (`true` for new/already-nested installations; retain `false` for existing flat installations until migration). The coordinator sends the workload identity name through authenticated `platform_config`; the guest uses its compute execution role to obtain a Linear token. Credentials are not baked into the MicroVM image. When upgrading an existing MicroVM deployment, rebuild the guest image too: the coordinator and guest must both support the vault configuration fields. From a42e52822ef8ef3fc85ff0bf26a607e1fd5f7d6c Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Tue, 22 Sep 2026 18:59:03 -0400 Subject: [PATCH 143/149] test(microvm): cover heartbeat boundaries and format review regressions --- agent/tests/test_runner.py | 21 +++++++---- .../handlers/shared/agent-heartbeat.test.ts | 36 +++++++++++++++++++ .../shared/microvm-lifecycle-local.test.ts | 23 +++++++----- 3 files changed, 65 insertions(+), 15 deletions(-) create mode 100644 cdk/test/handlers/shared/agent-heartbeat.test.ts diff --git a/agent/tests/test_runner.py b/agent/tests/test_runner.py index bb862060e..0a9872eba 100644 --- a/agent/tests/test_runner.py +++ b/agent/tests/test_runner.py @@ -50,7 +50,9 @@ def _config(**overrides: Any) -> TaskConfig: class TestClaudeSessionOwnership: @pytest.mark.parametrize("microvm", [False, True]) - @pytest.mark.parametrize("failure", [None, "connect", "query", "receive", "cancel", "hook-denied"]) + @pytest.mark.parametrize( + "failure", [None, "connect", "query", "receive", "cancel", "hook-denied"] + ) def test_broker_selection_and_cleanup_on_every_session_exit( self, monkeypatch, microvm, failure ): @@ -77,11 +79,15 @@ async def messages(): if context is not None: await context.tool_started("denied-call") await context.tool_started("other-active-call") - yield claude_agent_sdk.UserMessage(content=[ - claude_agent_sdk.ToolResultBlock( - tool_use_id="denied-call", content="Project hook denied", is_error=True, - ), - ]) + yield claude_agent_sdk.UserMessage( + content=[ + claude_agent_sdk.ToolResultBlock( + tool_use_id="denied-call", + content="Project hook denied", + is_error=True, + ), + ] + ) if context is not None: assert context.diagnostic_snapshot()["active_tools"] == 1 assert "other-active-call" in context._tools @@ -118,7 +124,8 @@ async def messages(): result = asyncio.run( runner.run_agent("probe", "probe", config, trajectory=MagicMock()) ) - assert result.status == ("error" if failure not in {None, "hook-denied"} else "success") + expected = "error" if failure not in {None, "hook-denied"} else "success" + assert result.status == expected options = make_client.call_args.kwargs["options"] if microvm: make_broker.assert_called_once_with(context) diff --git a/cdk/test/handlers/shared/agent-heartbeat.test.ts b/cdk/test/handlers/shared/agent-heartbeat.test.ts new file mode 100644 index 000000000..17832d6b3 --- /dev/null +++ b/cdk/test/handlers/shared/agent-heartbeat.test.ts @@ -0,0 +1,36 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +// SPDX-License-Identifier: MIT-0 + +import { evaluateAgentHeartbeat } from '../../../src/handlers/shared/agent-heartbeat'; + +test.each([ + [undefined, undefined, 1_000_000, undefined], + [NaN, undefined, 1_000_000, undefined], + [0, undefined, 360_000, undefined], + [0, undefined, 360_001, 'missing'], + [0, -200_000, 120_000, undefined], + [0, -200_000, 120_001, 'stale'], + [0, 120_000, 360_000, undefined], + [0, 120_000, 360_001, 'stale'], + [0, 400_000, 360_001, undefined], +] as const)('heartbeat boundary: start=%s heartbeat=%s now=%s yields %s', (start, heartbeat, now, expected) => { + expect(evaluateAgentHeartbeat(start, heartbeat, now)).toBe(expected); +}); diff --git a/cdk/test/handlers/shared/microvm-lifecycle-local.test.ts b/cdk/test/handlers/shared/microvm-lifecycle-local.test.ts index ab75fa2f0..c2940ba9e 100644 --- a/cdk/test/handlers/shared/microvm-lifecycle-local.test.ts +++ b/cdk/test/handlers/shared/microvm-lifecycle-local.test.ts @@ -66,12 +66,12 @@ const suffix = randomUUID(); const tasks = `lifecycle-tasks-${suffix}`; const approvals = `lifecycle-approvals-${suffix}`; Object.assign(process.env, { TASK_TABLE_NAME: tasks, TASK_APPROVALS_TABLE_NAME: approvals }); +import { reconcileMicrovmContinuation } from '../../../src/handlers/reconcile-microvm-continuations'; +import { failContinuationAttempt } from '../../../src/handlers/shared/microvm-continuation-runner'; +import { workerLeaseKey } from '../../../src/handlers/shared/microvm-continuation-types'; import { readMicrovmLifecycleSnapshot, saveMicrovmLifecycleIntent } from '../../../src/handlers/shared/microvm-lifecycle'; import { claimMicrovmStart, saveMicrovmImageCapability, saveMicrovmStartHandle } from '../../../src/handlers/shared/microvm-start'; import { superviseMicrovm, type MicrovmSupervisorState } from '../../../src/handlers/shared/microvm-supervisor'; -import { failContinuationAttempt } from '../../../src/handlers/shared/microvm-continuation-runner'; -import { reconcileMicrovmContinuation } from '../../../src/handlers/reconcile-microvm-continuations'; -import { workerLeaseKey } from '../../../src/handlers/shared/microvm-continuation-types'; const raw = new DynamoDBClient({ endpoint: endpoint ?? 'http://127.0.0.1:1', @@ -159,7 +159,10 @@ local('MicroVM lifecycle against DynamoDB Local', () => { await admin.send(new PutCommand({ TableName: tasks, Item: { - task_id: 'task', user_id: 'user', status: 'AWAITING_APPROVAL', compute_type: 'lambda-microvm', + task_id: 'task', + user_id: 'user', + status: 'AWAITING_APPROVAL', + compute_type: 'lambda-microvm', continuation: { state: 'STARTING', attempt_id: attempt }, concurrency_slot: { state: 'held', attempt_id: attempt }, }, @@ -175,17 +178,21 @@ local('MicroVM lifecycle against DynamoDB Local', () => { .toBe('FENCED'); // Finalization can release a no-receipt attempt before the scheduled sweep. await admin.send(new UpdateCommand({ - TableName: tasks, Key: { task_id: 'task' }, + TableName: tasks, + Key: { task_id: 'task' }, UpdateExpression: 'SET concurrency_slot.#state = :released', - ExpressionAttributeNames: { '#state': 'state' }, ExpressionAttributeValues: { ':released': 'released' }, + ExpressionAttributeNames: { '#state': 'state' }, + ExpressionAttributeValues: { ':released': 'released' }, })); expect(await claimMicrovmStart('task', 'user', 'hash', attempt)).toMatchObject({ closed: true }); if (lateWorker) { mockBeforeSend.mockImplementation(async command => { if (command instanceof UpdateCommand && command.input.UpdateExpression?.includes('lease_state = :closed')) { await admin.send(new UpdateCommand({ - TableName: tasks, Key: workerLeaseKey('task'), - UpdateExpression: 'SET lease_microvm_id = :id', ExpressionAttributeValues: { ':id': 'late-worker' }, + TableName: tasks, + Key: workerLeaseKey('task'), + UpdateExpression: 'SET lease_microvm_id = :id', + ExpressionAttributeValues: { ':id': 'late-worker' }, })); } }); From a6eeab44f56fa0ff519481b24df38c899ee90481 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Tue, 22 Sep 2026 19:00:10 -0400 Subject: [PATCH 144/149] fix(cdk): preserve unused MicroVM settings on other backends --- cdk/src/stacks/agent.ts | 2 +- cdk/test/stacks/microvm-layout-context.test.ts | 4 ++++ 2 files changed, 5 insertions(+), 1 deletion(-) diff --git a/cdk/src/stacks/agent.ts b/cdk/src/stacks/agent.ts index 705322a13..fecac6c58 100644 --- a/cdk/src/stacks/agent.ts +++ b/cdk/src/stacks/agent.ts @@ -359,7 +359,7 @@ export class AgentStack extends Stack { } // Require an explicit layout until flat-to-nested migration is verified. // A synth-time AWS lookup cannot protect deployments of saved assemblies. - const microvmNested = microvmNestedContext === true || microvmNestedContext === 'true'; + const microvmNested = microvmNestedContext !== false && microvmNestedContext !== 'false'; if (lambdaMicrovmEnabled && microvmNestedContext === undefined) { throw new Error( 'microvm_nested_stack must be explicitly selected: use --context microvm_nested_stack=true ' diff --git a/cdk/test/stacks/microvm-layout-context.test.ts b/cdk/test/stacks/microvm-layout-context.test.ts index 5595110e8..b3a31312f 100644 --- a/cdk/test/stacks/microvm-layout-context.test.ts +++ b/cdk/test/stacks/microvm-layout-context.test.ts @@ -41,6 +41,10 @@ describe.each([ })), { name: 'Agentcore', context: { compute_type: 'agentcore' } }, { name: 'Ecs', context: { compute_type: 'ecs' } }, + ...['agentcore', 'ecs'].map(compute => ({ + name: `${compute}WithUnusedPrefix`, + context: { compute_type: compute, microvm_resource_name_prefix: 'retained-config' }, + })), ])('MicroVM layout selection: $name', ({ name, context }) => { let template: Template; beforeAll(() => { From ffcbe7eac913e947580cc53f28dc66dd556dfd75 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Tue, 22 Sep 2026 19:00:36 -0400 Subject: [PATCH 145/149] docs(microvm): distinguish pending CLI wake acceptance from Linear coverage --- docs/verification/README.md | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/docs/verification/README.md b/docs/verification/README.md index 6c0cb9f5a..0492a2411 100644 --- a/docs/verification/README.md +++ b/docs/verification/README.md @@ -48,6 +48,9 @@ parent stack. - Finish reusable flat-to-nested migration commands and independently test an upgrade from current `main` on the same deployment. The earlier bespoke migration is not a substitute for that acceptance. +- Deploy the latest review fixes and verify CLI/API approve and deny on a + `PARKED` task, including immediate replacement admission. Earlier Linear + acceptance does not exercise the decision API functions' configuration. ## Reproduce local checks @@ -71,6 +74,8 @@ The last command opts into the pinned real SDK/CLI probe with a deterministic loopback model; it does not launch a cloud worker. Optional DynamoDB Local tests require their documented local service. Mocks do not establish effective AWS IAM. The standalone cloud acceptance harness and raw receipts remain outside this PR. +The repository's narrower [launch, payload and replay probes](../../cdk/test/live/README.md) +have separate inspection and execution commands. ## Live acceptance for an installation @@ -87,6 +92,7 @@ only after verifying that image and coordinator together. | Default and disabled sleep | Omitted override uses 600 seconds; task-level off and deployment off prevent new suspension while wake/cleanup remain available. | | Credential expiry | Sleep past the original credential lifetime, then perform actual task-scoped AWS operations. | | Retirement/replacement | Confirm old-worker shutdown before capacity release; one replacement restores files/conversation and preserves approval identity and usage. | +| CLI/API replacement | After the task reaches `PARKED`, approve or deny through the CLI/API. Verify replacement admission starts from that decision without waiting for the scheduled sweep, and the exact pending tool follows the decision. | | Upgrade/rollback | Preserve unrelated resource identities, old in-flight work and recoverable checkpoints; test compatible code/image/policy rollback. | Include effective-role checks for own-task access and denial of cross-task data, From bf7c271d82fbcbe39d7c03c31361d823049668e8 Mon Sep 17 00:00:00 2001 From: Sphia Sadek Date: Tue, 22 Sep 2026 19:00:53 -0400 Subject: [PATCH 146/149] docs: sync verification acceptance mirror --- docs/src/content/docs/verification/Readme.md | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/docs/src/content/docs/verification/Readme.md b/docs/src/content/docs/verification/Readme.md index eaac8b951..7ba3c667f 100644 --- a/docs/src/content/docs/verification/Readme.md +++ b/docs/src/content/docs/verification/Readme.md @@ -52,6 +52,9 @@ parent stack. - Finish reusable flat-to-nested migration commands and independently test an upgrade from current `main` on the same deployment. The earlier bespoke migration is not a substitute for that acceptance. +- Deploy the latest review fixes and verify CLI/API approve and deny on a + `PARKED` task, including immediate replacement admission. Earlier Linear + acceptance does not exercise the decision API functions' configuration. ## Reproduce local checks @@ -75,6 +78,8 @@ The last command opts into the pinned real SDK/CLI probe with a deterministic loopback model; it does not launch a cloud worker. Optional DynamoDB Local tests require their documented local service. Mocks do not establish effective AWS IAM. The standalone cloud acceptance harness and raw receipts remain outside this PR. +The repository's narrower [launch, payload and replay probes](https://github.com/aws-samples/sample-autonomous-cloud-coding-agents/blob/main/cdk/test/live/README.md) +have separate inspection and execution commands. ## Live acceptance for an installation @@ -91,6 +96,7 @@ only after verifying that image and coordinator together. | Default and disabled sleep | Omitted override uses 600 seconds; task-level off and deployment off prevent new suspension while wake/cleanup remain available. | | Credential expiry | Sleep past the original credential lifetime, then perform actual task-scoped AWS operations. | | Retirement/replacement | Confirm old-worker shutdown before capacity release; one replacement restores files/conversation and preserves approval identity and usage. | +| CLI/API replacement | After the task reaches `PARKED`, approve or deny through the CLI/API. Verify replacement admission starts from that decision without waiting for the scheduled sweep, and the exact pending tool follows the decision. | | Upgrade/rollback | Preserve unrelated resource identities, old in-flight work and recoverable checkpoints; test compatible code/image/policy rollback. | Include effective-role checks for own-task access and denial of cross-task data, From 440ffd6b046d09f012871b44e605f0255edd98fe Mon Sep 17 00:00:00 2001 From: t Date: Wed, 23 Sep 2026 16:09:46 +0000 Subject: [PATCH 147/149] chore(cdk): leave region-scoped dashboard/OAC renames to a separate PR Reverts 4ea7e88b. Suffixing the CloudWatch dashboard and CloudFront OAC names with the Region fixes a real multi-Region collision, but it is unrelated to the MicroVM P3 lifecycle and renames (replaces) live resources for every existing deployment, including AgentCore-only ones. Keeping it out of this PR narrows the change to #645; it should ship on its own with its own upgrade note. Refs #645 Co-Authored-By: Claude Fable 5.1 --- cdk/src/constructs/screenshot-bucket.ts | 9 +--- cdk/src/constructs/task-dashboard.ts | 4 +- cdk/test/constructs/screenshot-bucket.test.ts | 49 +------------------ cdk/test/constructs/task-dashboard.test.ts | 43 +++++----------- docs/design/OBSERVABILITY.md | 2 +- .../docs/architecture/Observability.md | 2 +- 6 files changed, 19 insertions(+), 90 deletions(-) diff --git a/cdk/src/constructs/screenshot-bucket.ts b/cdk/src/constructs/screenshot-bucket.ts index 00595220b..19418bb33 100644 --- a/cdk/src/constructs/screenshot-bucket.ts +++ b/cdk/src/constructs/screenshot-bucket.ts @@ -17,7 +17,7 @@ * SOFTWARE. */ -import { Duration, Names, RemovalPolicy, Stack } from 'aws-cdk-lib'; +import { Duration, RemovalPolicy } from 'aws-cdk-lib'; import * as cloudfront from 'aws-cdk-lib/aws-cloudfront'; import * as origins from 'aws-cdk-lib/aws-cloudfront-origins'; import * as s3 from 'aws-cdk-lib/aws-s3'; @@ -97,14 +97,9 @@ export class ScreenshotBucket extends Construct { // and grants `s3:GetObject` to the distribution's CF service principal // only — no anonymous principal in the policy, so account-level BPA // doesn't reject it. - // OAC names are account-global. Keep the construct-path hash and leave - // room for the Region within CloudFront's 64-character name limit. - const originAccessControl = new cloudfront.S3OriginAccessControl(this, 'OriginAccessControl', { - originAccessControlName: `${Names.uniqueResourceName(this, { maxLength: 40 })}-${Stack.of(this).region}`, - }); this.distribution = new cloudfront.Distribution(this, 'Distribution', { defaultBehavior: { - origin: origins.S3BucketOrigin.withOriginAccessControl(this.bucket, { originAccessControl }), + origin: origins.S3BucketOrigin.withOriginAccessControl(this.bucket), viewerProtocolPolicy: cloudfront.ViewerProtocolPolicy.REDIRECT_TO_HTTPS, // Screenshots are immutable per (repo, sha) — long TTL is safe // and minimizes origin S3 requests on hot PRs. diff --git a/cdk/src/constructs/task-dashboard.ts b/cdk/src/constructs/task-dashboard.ts index 78c8ff4c7..6f4e40f7a 100644 --- a/cdk/src/constructs/task-dashboard.ts +++ b/cdk/src/constructs/task-dashboard.ts @@ -64,9 +64,7 @@ export class TaskDashboard extends Construct { const logGroup = props.applicationLogGroup; this.dashboard = new cloudwatch.Dashboard(this, 'Dashboard', { - // Dashboard names are account-global, so identical stack names in - // different Regions must not share one CloudFormation-owned dashboard. - dashboardName: `BackgroundAgent-Tasks-${Stack.of(this).stackName}-${Stack.of(this).region}`, + dashboardName: `BackgroundAgent-Tasks-${Stack.of(this).stackName}`, defaultInterval: Duration.hours(24), }); diff --git a/cdk/test/constructs/screenshot-bucket.test.ts b/cdk/test/constructs/screenshot-bucket.test.ts index 552727b3e..882df1ae9 100644 --- a/cdk/test/constructs/screenshot-bucket.test.ts +++ b/cdk/test/constructs/screenshot-bucket.test.ts @@ -23,59 +23,12 @@ import { ScreenshotBucket } from '../../src/constructs/screenshot-bucket'; describe('ScreenshotBucket', () => { let template: Template; - let regionalTemplates: Template[]; - let longNameTemplate: Template; - beforeAll(() => { + beforeEach(() => { const app = new App(); const stack = new Stack(app, 'TestStack'); new ScreenshotBucket(stack, 'ScreenshotBucket'); template = Template.fromStack(stack); - regionalTemplates = ['us-east-1', 'us-west-2'].map((region) => { - const regionalStack = new Stack(new App(), 'TestStack', { - env: { account: '123456789012', region }, - }); - new ScreenshotBucket(regionalStack, 'ScreenshotBucket'); - return Template.fromStack(regionalStack); - }); - const longNameStack = new Stack(new App(), 'a'.repeat(128), { - env: { account: '123456789012', region: 'us-west-2' }, - }); - new ScreenshotBucket(longNameStack, 'FirstScreenshots'); - new ScreenshotBucket(longNameStack, 'SecondScreenshots'); - longNameTemplate = Template.fromStack(longNameStack); - }); - - function accessControlNames(source: Template): string[] { - return Object.values(source.findResources('AWS::CloudFront::OriginAccessControl')) - .map((resource) => resource.Properties.OriginAccessControlConfig.Name); - } - - test('same stack in different regions has distinct global access-control names', () => { - const names = regionalTemplates.flatMap(accessControlNames); - expect(names).toHaveLength(2); - expect(names[0]).toMatch(/-us-east-1$/); - expect(names[1]).toMatch(/-us-west-2$/); - expect(new Set(names).size).toBe(2); - }); - - test('long stack names preserve distinct access controls within the 64-character limit', () => { - const names = accessControlNames(longNameTemplate); - expect(new Set(names).size).toBe(2); - for (const name of names) { - expect(name.length).toBeLessThanOrEqual(64); - expect(name).toMatch(/-us-west-2$/); - } - }); - - test('access control always signs S3 requests with sigv4', () => { - template.hasResourceProperties('AWS::CloudFront::OriginAccessControl', { - OriginAccessControlConfig: { - OriginAccessControlOriginType: 's3', - SigningBehavior: 'always', - SigningProtocol: 'sigv4', - }, - }); }); // Lock in the screenshot bucket lifecycle defaults. diff --git a/cdk/test/constructs/task-dashboard.test.ts b/cdk/test/constructs/task-dashboard.test.ts index 185b67d28..fc12dba03 100644 --- a/cdk/test/constructs/task-dashboard.test.ts +++ b/cdk/test/constructs/task-dashboard.test.ts @@ -22,11 +22,9 @@ import { Template } from 'aws-cdk-lib/assertions'; import * as logs from 'aws-cdk-lib/aws-logs'; import { TaskDashboard } from '../../src/constructs/task-dashboard'; -function createStack(region?: string): { stack: Stack; template: Template } { +function createStack(): { stack: Stack; template: Template } { const app = new App(); - const stack = new Stack(app, 'TestStack', region - ? { env: { account: '123456789012', region } } - : {}); + const stack = new Stack(app, 'TestStack'); const logGroup = new logs.LogGroup(stack, 'AppLogGroup'); @@ -40,36 +38,16 @@ function createStack(region?: string): { stack: Stack; template: Template } { } describe('TaskDashboard construct', () => { - let template: Template; - let regionalTemplates: Template[]; - - beforeAll(() => { - ({ template } = createStack()); - regionalTemplates = ['us-east-1', 'us-west-2'].map((region) => createStack(region).template); - }); - test('creates a CloudWatch Dashboard', () => { + const { template } = createStack(); template.resourceCountIs('AWS::CloudWatch::Dashboard', 1); }); - test('dashboard name includes stack name and deployment region', () => { + test('dashboard name includes stack name', () => { + const { template } = createStack(); template.hasResourceProperties('AWS::CloudWatch::Dashboard', { - DashboardName: { - 'Fn::Join': ['', ['BackgroundAgent-Tasks-TestStack-', { Ref: 'AWS::Region' }]], - }, - }); - }); - - test('same stack name in two regions creates distinct global dashboard names', () => { - const names = regionalTemplates.map((regionalTemplate) => { - const dashboards = regionalTemplate.findResources('AWS::CloudWatch::Dashboard'); - return Object.values(dashboards)[0].Properties.DashboardName; + DashboardName: 'BackgroundAgent-Tasks-TestStack', }); - expect(names).toEqual([ - 'BackgroundAgent-Tasks-TestStack-us-east-1', - 'BackgroundAgent-Tasks-TestStack-us-west-2', - ]); - expect(new Set(names).size).toBe(2); }); // --- Chunk 8b: Cedar HITL approval widgets (§11.3, IMPL-28) ------------ @@ -77,8 +55,8 @@ describe('TaskDashboard construct', () => { // The dashboard body is serialized as CloudFormation ``Fn::Join`` parts // with CDK tokens for the stack region/account. Use ``Match.serializedJson`` // / substring checks via template rendering. - function dashboardBodyContains(source: Template, needle: string): boolean { - const dashboards = source.findResources('AWS::CloudWatch::Dashboard'); + function dashboardBodyContains(template: Template, needle: string): boolean { + const dashboards = template.findResources('AWS::CloudWatch::Dashboard'); for (const res of Object.values(dashboards)) { const body = (res as any).Properties?.DashboardBody; if (typeof body === 'string') { @@ -94,10 +72,12 @@ describe('TaskDashboard construct', () => { } test('dashboard references ABCA/Cedar-HITL namespace for the new metrics', () => { + const { template } = createStack(); expect(dashboardBodyContains(template, 'ABCA/Cedar-HITL')).toBe(true); }); test('dashboard includes ApprovalTimeoutClipRate widget (MathExpression with IF guard)', () => { + const { template } = createStack(); expect(dashboardBodyContains(template, 'Approval Timeout Clip Rate')).toBe(true); // The IF(requested > 0, ...) guard is critical to avoid divide-by-zero // NaN renders (silent gap). Assert the expression ships in the widget. @@ -109,11 +89,13 @@ describe('TaskDashboard construct', () => { }); test('dashboard references ClippedApprovalCount and ApprovalRequestCount metrics', () => { + const { template } = createStack(); expect(dashboardBodyContains(template, 'ClippedApprovalCount')).toBe(true); expect(dashboardBodyContains(template, 'ApprovalRequestCount')).toBe(true); }); test('dashboard includes ApprovalTimeoutBreakdown widget with p50/p90/p99 on TimedOutEffectiveTimeout', () => { + const { template } = createStack(); expect(dashboardBodyContains(template, 'Approval Timeout Breakdown')).toBe(true); expect(dashboardBodyContains(template, 'TimedOutEffectiveTimeout')).toBe(true); // All three percentiles required by §11.3. @@ -123,6 +105,7 @@ describe('TaskDashboard construct', () => { }); test('dashboard includes ApprovalDecisionLatency widget with outcome dims', () => { + const { template } = createStack(); expect(dashboardBodyContains(template, 'Approval Decision Latency')).toBe(true); expect(dashboardBodyContains(template, 'ApprovalDecisionLatencyMs')).toBe(true); // Three outcome dim values — one series set per outcome per percentile. diff --git a/docs/design/OBSERVABILITY.md b/docs/design/OBSERVABILITY.md index 5f0efbab8..96441db95 100644 --- a/docs/design/OBSERVABILITY.md +++ b/docs/design/OBSERVABILITY.md @@ -139,7 +139,7 @@ Emitted as custom CloudWatch metrics and used in dashboards and alarms. ## Dashboard -A CloudWatch dashboard (`BackgroundAgent-Tasks-${stackName}-${region}`) is deployed via the `TaskDashboard` CDK construct. Dashboard names are account-global, so the Region suffix lets identically named stacks coexist in different Regions. Updating a deployment from the earlier name replaces its dashboard resource; metric and log data remain in their existing stores, but saved dashboard links need the new name. It provides Logs Insights widgets for: +A CloudWatch dashboard (`BackgroundAgent-Tasks-${stackName}`, i.e. the base name suffixed with the stack name) is deployed via the `TaskDashboard` CDK construct. It provides Logs Insights widgets for: - Task success rate and count by status - Cost per task and turns per task diff --git a/docs/src/content/docs/architecture/Observability.md b/docs/src/content/docs/architecture/Observability.md index 841c2695a..a0991020e 100644 --- a/docs/src/content/docs/architecture/Observability.md +++ b/docs/src/content/docs/architecture/Observability.md @@ -143,7 +143,7 @@ Emitted as custom CloudWatch metrics and used in dashboards and alarms. ## Dashboard -A CloudWatch dashboard (`BackgroundAgent-Tasks-${stackName}-${region}`) is deployed via the `TaskDashboard` CDK construct. Dashboard names are account-global, so the Region suffix lets identically named stacks coexist in different Regions. Updating a deployment from the earlier name replaces its dashboard resource; metric and log data remain in their existing stores, but saved dashboard links need the new name. It provides Logs Insights widgets for: +A CloudWatch dashboard (`BackgroundAgent-Tasks-${stackName}`, i.e. the base name suffixed with the stack name) is deployed via the `TaskDashboard` CDK construct. It provides Logs Insights widgets for: - Task success rate and count by status - Cost per task and turns per task From 36f09407783aa88d3b3dc329d976ef002a961819 Mon Sep 17 00:00:00 2001 From: t Date: Wed, 23 Sep 2026 16:14:35 +0000 Subject: [PATCH 148/149] chore(645): drop test-only migration helpers and diagnostic probes cdk/src/migration/{template,permissions}.ts were imported only by their own tests and had no script, mise task or documented entry point; the explicit microvm_nested_stack requirement means a migration command is a follow-up, not part of this PR. The three agent/scripts probes are one-off diagnostics that nothing in the repo, CI or docs references. verify_microvm_credentials.py stays because agent/tests/test_continuation_sdk_probe.py imports it. Refs #645 Co-Authored-By: Claude Fable 5.1 --- .../microvm_lifecycle_listener_probe.py | 214 ------------- agent/scripts/microvm_process_observer.py | 137 --------- agent/scripts/verify_approval_hook_timeout.py | 281 ------------------ cdk/AGENTS.md | 1 - cdk/src/migration/permissions.ts | 162 ---------- cdk/src/migration/template.ts | 139 --------- cdk/test/migration/permissions.test.ts | 117 -------- cdk/test/migration/template.test.ts | 115 ------- 8 files changed, 1166 deletions(-) delete mode 100644 agent/scripts/microvm_lifecycle_listener_probe.py delete mode 100644 agent/scripts/microvm_process_observer.py delete mode 100644 agent/scripts/verify_approval_hook_timeout.py delete mode 100644 cdk/src/migration/permissions.ts delete mode 100644 cdk/src/migration/template.ts delete mode 100644 cdk/test/migration/permissions.test.ts delete mode 100644 cdk/test/migration/template.test.ts diff --git a/agent/scripts/microvm_lifecycle_listener_probe.py b/agent/scripts/microvm_lifecycle_listener_probe.py deleted file mode 100644 index 5c1ac2358..000000000 --- a/agent/scripts/microvm_lifecycle_listener_probe.py +++ /dev/null @@ -1,214 +0,0 @@ -#!/usr/bin/env python3 -# Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. -# SPDX-License-Identifier: MIT-0 - -"""Isolated MicroVM transport probe; never import into the application runtime. - -Serve the six lifecycle hooks using only Python's standard library. Logs contain -request boundaries, listener health and a disposable marker's hash, without AWS -calls or credentials. The explicit ``close_listener`` mode is a failure control. -Use only in an owned diagnostic image with no ingress and a bounded VM lifetime. -""" - -from __future__ import annotations - -import argparse -import faulthandler -import hashlib -import json -import os -import signal -import socket -import tempfile -import threading -import time -from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer -from pathlib import Path - -PREFIX = "/aws/lambda-microvms/runtime/v1/" -HOOKS = {"ready", "validate", "run", "suspend", "resume", "terminate"} -MAX_BODY_BYTES = 16_384 -MAX_CASE_ID_LENGTH = 100 - - -def emit(event: str, **fields: object) -> None: - """One short stdout write, independent of AWS clients and logging threads.""" - record = { - "event": event, - "wall_s": time.time(), - "monotonic_s": time.monotonic(), - "pid": os.getpid(), - **fields, - } - os.write(1, (json.dumps(record, sort_keys=True) + "\n").encode()) - - -class ProbeServer(ThreadingHTTPServer): - daemon_threads = True - - def __init__(self, host: str, port: int) -> None: - self.lock = threading.RLock() - self.case_id = "" - self.mode = "normal" - self.phase = "image" - self.suspends = 0 - self.resumes = 0 - self.marker_hash = "" - self.marker = Path(tempfile.gettempdir()) / f"abca-listener-probe-{os.getpid()}" - super().__init__((host, port), ProbeHandler) - - def health(self) -> dict[str, object]: - try: - listening: object = bool( - self.socket.getsockopt(socket.SOL_SOCKET, socket.SO_ACCEPTCONN) - ) - except OSError as exc: - listening = type(exc).__name__ - return { - "listening": listening, - "threads": threading.active_count(), - "case_id": self.case_id, - "phase": self.phase, - "suspends": self.suspends, - "resumes": self.resumes, - "marker_hash": self.marker_hash, - } - - -class ProbeHandler(BaseHTTPRequestHandler): - server: ProbeServer - protocol_version = "HTTP/1.1" - - def log_message(self, format: str, *args: object) -> None: - # Request headers and service payloads are deliberately not logged. - return - - def do_POST(self) -> None: - hook = self.path.removeprefix(PREFIX) - if not self.path.startswith(PREFIX) or hook not in HOOKS: - self.respond(404, {"code": "UNKNOWN_HOOK"}) - return - emit("hook_enter", hook=hook, **self.server.health()) - try: - self.connection.settimeout(2) - size = int(self.headers.get("Content-Length", "0")) - if not 0 <= size <= MAX_BODY_BYTES: - raise ValueError("Invalid body length") - body = self.rfile.read(size) - if len(body) != size: - raise ValueError("Truncated body") - envelope = json.loads(body) if body else {} - if not isinstance(envelope, dict): - raise ValueError("Expected object") - self.transition(hook, envelope) - except (OSError, ValueError, TypeError, KeyError) as exc: - emit("hook_error", hook=hook, error_type=type(exc).__name__) - self.respond(409, {"code": "PROBE_REJECTED"}) - - def transition(self, hook: str, envelope: dict) -> None: - server = self.server - close_listener = False - with server.lock: - if hook == "run": - payload = json.loads(envelope["runHookPayload"]) - case_id, mode = payload["case_id"], payload.get("mode", "normal") - if ( - not isinstance(case_id, str) - or not 1 <= len(case_id) <= MAX_CASE_ID_LENGTH - or mode not in {"normal", "close_listener"} - ): - raise ValueError("Invalid probe configuration") - if server.case_id and (server.case_id != case_id or server.mode != mode): - raise ValueError("Conflicting run") - if not server.case_id: - server.case_id, server.mode = case_id, mode - marker = os.urandom(32) - server.marker.write_bytes(marker) - server.marker_hash = hashlib.sha256(marker).hexdigest() - server.phase = "running" - elif hook == "suspend": - if server.phase == "running": - server.phase = "suspended" - server.suspends += 1 - close_listener = server.mode == "close_listener" - elif server.phase != "suspended": - raise ValueError("No running probe") - elif hook == "resume": - if server.phase == "suspended": - if hashlib.sha256(server.marker.read_bytes()).hexdigest() != server.marker_hash: - raise ValueError("Marker changed") - server.resumes += 1 - server.phase = "running" - elif server.phase != "running": - raise ValueError("No suspended probe") - elif hook == "terminate": - server.phase = "terminating" - - if close_listener: - # Close only the listening socket; this accepted /suspend connection - # can still send its response. The process stays alive for diagnostics. - server.shutdown() - server.server_close() - emit("listener_closed_intentionally", **server.health()) - result = {"status": "acknowledged", "hook": hook, **server.health()} - emit("hook_ack", **result) - self.respond(200, result) - - def respond(self, status: int, payload: dict) -> None: - body = json.dumps(payload).encode() - self.send_response(status) - self.send_header("Content-Type", "application/json") - self.send_header("Content-Length", str(len(body))) - # Make every service hook establish a connection to the listener. - self.send_header("Connection", "close") - self.end_headers() - self.wfile.write(body) - self.wfile.flush() - self.close_connection = True - - -def observe(server: ProbeServer) -> None: - previous_wall, previous_monotonic = time.time(), time.monotonic() - tick = 0 - while True: - time.sleep(0.25) - wall, monotonic = time.time(), time.monotonic() - wall_gap, monotonic_gap = wall - previous_wall, monotonic - previous_monotonic - tick += 1 - if wall_gap > 1 or monotonic_gap > 1 or tick % 20 == 0: - emit( - "listener_observation", - wall_gap_s=wall_gap, - monotonic_gap_s=monotonic_gap, - **server.health(), - ) - previous_wall, previous_monotonic = wall, monotonic - - -def main() -> None: - parser = argparse.ArgumentParser(description=__doc__) - parser.add_argument("--host", default="127.0.0.1") - parser.add_argument("--port", type=int, default=8080) - args = parser.parse_args() - faulthandler.enable() - - def stop(signum: int, _frame: object) -> None: - emit("process_signal", signal=signum) - raise SystemExit(0) - - signal.signal(signal.SIGTERM, stop) - signal.signal(signal.SIGINT, stop) - server = ProbeServer(args.host, args.port) - threading.Thread(target=observe, args=(server,), daemon=True).start() - emit("listener_started", port=server.server_port, **server.health()) - try: - server.serve_forever(poll_interval=0.05) - # The negative control intentionally stops acceptance, not the process. - threading.Event().wait() - finally: - server.server_close() - server.marker.unlink(missing_ok=True) - - -if __name__ == "__main__": - main() diff --git a/agent/scripts/microvm_process_observer.py b/agent/scripts/microvm_process_observer.py deleted file mode 100644 index cd2c85010..000000000 --- a/agent/scripts/microvm_process_observer.py +++ /dev/null @@ -1,137 +0,0 @@ -"""Temporary image diagnostic: observe the agent server without opening sockets. - -Run as the parent of the normal image command, after ``--``. This changes the -process tree and is diagnostic instrumentation, not a production entry point. -Only process state, port-8080 listener metadata, and cgroup memory counters are -recorded. Arguments, environment, request bodies and credentials are not logged. -""" - -import argparse -import contextlib -import json -import os -import signal -import subprocess -import time -from pathlib import Path -from typing import Any - -_TCP_FIELDS_THROUGH_INODE = 10 -_REPORT_INTERVAL_S = 5 - - -def emit(event: str, **fields: Any) -> None: - print( - json.dumps( - { - "probe": "microvm-process-observer", - "event": event, - "wall_time_ns": time.time_ns(), - "monotonic_ns": time.monotonic_ns(), - "observer_pid": os.getpid(), - **fields, - } - ), - flush=True, - ) - - -def snapshot(pid: int) -> dict[str, Any]: - result: dict[str, Any] = {"child_pid": pid} - try: - status = Path(f"/proc/{pid}/status").read_text() - allowed = {"State", "Threads", "VmRSS", "SigPnd", "ShdPnd"} - result["child_status"] = { - key: value.strip() - for line in status.splitlines() - for key, _, value in [line.partition(":")] - if key in allowed - } - except OSError as error: - result["child_status_error"] = type(error).__name__ - listeners = [] - for family in ("tcp", "tcp6"): - try: - for line in Path(f"/proc/net/{family}").read_text().splitlines()[1:]: - fields = line.split() - if ( - len(fields) >= _TCP_FIELDS_THROUGH_INODE - and fields[1].endswith(":1F90") - and fields[3] == "0A" - ): - listeners.append({"family": family, "inode": fields[9]}) - except OSError as error: - result[f"{family}_error"] = type(error).__name__ - result["listeners_8080"] = listeners - try: - inodes = { - os.readlink(entry) - for entry in Path(f"/proc/{pid}/fd").iterdir() - if entry.name.isdigit() - } - result["child_owns_listener"] = any( - f"socket:[{listener['inode']}]" in inodes for listener in listeners - ) - except OSError as error: - result["child_fds_error"] = type(error).__name__ - try: - result["memory_events"] = Path("/sys/fs/cgroup/memory.events").read_text().strip() - except OSError as error: - result["memory_events_error"] = type(error).__name__ - return result - - -def main() -> int: - parser = argparse.ArgumentParser(description=__doc__) - parser.add_argument("command", nargs=argparse.REMAINDER) - args = parser.parse_args() - command = args.command - if command[:1] == ["--"]: - command = command[1:] - if not command: - parser.error("a child command is required") - child = subprocess.Popen(command, start_new_session=True) - shutdown_at: float | None = None - - def forward_signal(number: int, _frame: Any) -> None: - nonlocal shutdown_at - emit("observer_signal", signal=number, child_pid=child.pid) - shutdown_at = time.monotonic() + 10 - with contextlib.suppress(ProcessLookupError): - os.killpg(child.pid, number) - - signal.signal(signal.SIGTERM, forward_signal) - signal.signal(signal.SIGINT, forward_signal) - emit("child_started", child_pid=child.pid) - previous: dict[str, Any] | None = None - last_sample = time.monotonic() - last_emit = 0.0 - try: - while True: - now = time.monotonic() - state = snapshot(child.pid) - code = child.poll() - if state != previous or now - last_emit >= _REPORT_INTERVAL_S or now - last_sample > 1: - emit( - "process_observation", returncode=code, sample_gap_s=now - last_sample, **state - ) - previous = state - last_emit = now - last_sample = now - if code is not None: - emit("child_exited", returncode=code, **state) - return code if code >= 0 else 128 - code - if shutdown_at is not None and now >= shutdown_at: - emit("shutdown_deadline", child_pid=child.pid) - with contextlib.suppress(ProcessLookupError): - os.killpg(child.pid, signal.SIGKILL) - time.sleep(0.25) - finally: - if child.poll() is None: - with contextlib.suppress(ProcessLookupError): - os.killpg(child.pid, signal.SIGKILL) - child.wait() - - -if __name__ == "__main__": - raise SystemExit(main()) diff --git a/agent/scripts/verify_approval_hook_timeout.py b/agent/scripts/verify_approval_hook_timeout.py deleted file mode 100644 index 57823a866..000000000 --- a/agent/scripts/verify_approval_hook_timeout.py +++ /dev/null @@ -1,281 +0,0 @@ -#!/usr/bin/env python3 -# Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. -# SPDX-License-Identifier: MIT-0 - -"""Opt-in pinned CLI approval callback probe; loopback model and synthetic keys. - -Compare --configured microvm --delay 650 with --delay 650 for the CLI default. -An explicit --timeout 1 --delay 3 provides a short failure control. This is -outside the unit suite; it tests actual SDK/CLI transport, not VM freezing. -""" - -import argparse -import asyncio -import importlib.metadata -import json -import os -import subprocess -import sys -import tempfile -import threading -import time -from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer -from pathlib import Path - -from verify_microvm_credentials import MODEL, event_frame, model_response - -ROOT = Path(__file__).resolve().parents[2] -MAX_PROBE_DELAY_S = 900 - - -def require(condition: bool, detail: str) -> None: - if not condition: - raise RuntimeError(detail) - - -async def probe(timeout, delay, configured): - import claude_agent_sdk - from claude_agent_sdk import ClaudeAgentOptions, ClaudeSDKClient, ResultMessage - from claude_agent_sdk.types import HookMatcher - - cli = Path(claude_agent_sdk.__file__).parent / "_bundled/claude" - version = subprocess.run( - [str(cli), "--version"], capture_output=True, text=True, check=True, timeout=10 - ).stdout.strip() - require(version == "2.1.191 (Claude Code)", "Probe requires reviewed CLI 2.1.191") - require( - importlib.metadata.version("claude-agent-sdk") == "0.2.110", - "Probe requires reviewed SDK 0.2.110", - ) - audit = [] - lifecycle = None - pre_matcher = HookMatcher(timeout=timeout) - if configured: - sys.path.insert(0, str(ROOT / "agent/src")) - from hooks import build_hook_matchers - from microvm_lifecycle import register_task - from policy import PolicyEngine - - if configured == "microvm": - lifecycle = register_task("local-hook-probe", "local-probe-vm") - # PolicyEngine logs with os.write(1), bypassing sys.stdout. Redirect - # only synchronous setup, before the probe starts any server threads. - saved_stdout = os.dup(1) - try: - os.dup2(2, 1) - pre_matcher = build_hook_matchers( - engine=PolicyEngine(task_type="new_task", repo="probe/owned"), - task_id="local-hook-probe", - )["PreToolUse"][0] - finally: - os.dup2(saved_stdout, 1) - os.close(saved_stdout) - with tempfile.TemporaryDirectory(prefix="abca-645-hook-timeout-") as directory: - temp = Path(directory) - target = temp / "owned-marker.txt" - target.write_text("OWNED_READ_MARKER\n") - settings = temp / "settings.json" - settings.write_text("{}") - aws_config = temp / "aws-config" - aws_config.write_text("[default]\nregion = us-west-2\n") - aws_creds = temp / "aws-credentials" - aws_creds.write_text("") - config = temp / "claude-config" - config.mkdir() - started = time.monotonic() - - def record(kind, **data): - audit.append({"kind": kind, "elapsed_s": time.monotonic() - started, **data}) - - def tool_response(): - events = [ - { - "type": "message_start", - "message": { - "id": "msg_tool", - "type": "message", - "role": "assistant", - "model": MODEL, - "content": [], - "stop_reason": None, - "stop_sequence": None, - "usage": {"input_tokens": 1, "output_tokens": 0}, - }, - }, - { - "type": "content_block_start", - "index": 0, - "content_block": { - "type": "tool_use", - "id": "toolu_owned", - "name": "Read", - "input": {}, - }, - }, - { - "type": "content_block_delta", - "index": 0, - "delta": { - "type": "input_json_delta", - "partial_json": json.dumps({"file_path": str(target)}), - }, - }, - {"type": "content_block_stop", "index": 0}, - { - "type": "message_delta", - "delta": { - "stop_reason": "tool_use", - "stop_sequence": None, - }, - "usage": {"output_tokens": 1}, - }, - {"type": "message_stop"}, - ] - return b"".join(event_frame(event) for event in events) - - class Handler(BaseHTTPRequestHandler): - def log_message(self, format, *args): - del format, args - - def do_POST(self): - body = json.loads(self.rfile.read(int(self.headers["Content-Length"]))) - prior = [ - item - for message in body.get("messages", []) - for item in message.get("content", []) - if isinstance(item, dict) and item.get("type") == "tool_result" - ] - record("model-request", results=prior) - stream = model_response() if prior else tool_response() - self.send_response(200) - self.send_header("Content-Type", "application/vnd.amazon.eventstream") - self.send_header("Content-Length", str(len(stream))) - self.end_headers() - self.wfile.write(stream) - - server = ThreadingHTTPServer(("127.0.0.1", 0), Handler) - thread = threading.Thread(target=server.serve_forever, daemon=True) - thread.start() - endpoint = f"http://127.0.0.1:{server.server_port}" - - async def pre(data, tool_id, ctx): - require(data["tool_name"] == "Read", "Unexpected tool") - require(data["tool_input"] == {"file_path": str(target)}, "Unexpected target") - record("pre-start", tool_id=tool_id) - try: - await asyncio.sleep(delay) - record("pre-allow") - return { - "hookSpecificOutput": { - "hookEventName": "PreToolUse", - "permissionDecision": "allow", - "permissionDecisionReason": "Owned local test gate released", - } - } - except BaseException as error: - record("pre-cancelled", error_type=type(error).__name__) - raise - - async def post(data, tool_id, ctx): - record("post", tool_name=data["tool_name"]) - return {} - - env = { - "CLAUDE_CONFIG_DIR": str(config), - "CLAUDE_CODE_USE_BEDROCK": "1", - "CLAUDE_CODE_DISABLE_NONESSENTIAL_TRAFFIC": "1", - "CLAUDE_CODE_MAX_RETRIES": "0", - "DISABLE_TELEMETRY": "1", - "DISABLE_ERROR_REPORTING": "1", - "DISABLE_AUTOUPDATER": "1", - "ANTHROPIC_BEDROCK_BASE_URL": endpoint, - "AWS_ENDPOINT_URL": endpoint, - "AWS_REGION": "us-west-2", - "AWS_DEFAULT_REGION": "us-west-2", - "AWS_CONFIG_FILE": str(aws_config), - "AWS_SHARED_CREDENTIALS_FILE": str(aws_creds), - "AWS_ACCESS_KEY_ID": "SYNTHETIC_NOT_VALID", - "AWS_SECRET_ACCESS_KEY": "synthetic-not-valid-in-aws", - "AWS_SESSION_TOKEN": "synthetic-not-valid-in-aws", - "AWS_EC2_METADATA_DISABLED": "true", - } - errors = [] - # Retain production matcher settings, replacing only its callback with - # a controlled wait so the transport's cancellation is observable. - pre_matcher.hooks = [pre] - try: - options = ClaudeAgentOptions( - model=MODEL, - max_turns=2, - cwd=directory, - tools=["Read"], - permission_mode="bypassPermissions", - setting_sources=[], - settings=str(settings), - env=env, - stderr=errors.append, - hooks={ - "PreToolUse": [pre_matcher], - "PostToolUse": [HookMatcher(hooks=[post])], - }, - ) - async with ClaudeSDKClient(options=options) as client: - await client.query("Read the owned marker once, then stop.") - async for message in client.receive_response(): - if isinstance(message, ResultMessage): - record("result", is_error=message.is_error) - return { - "sdk_version": "0.2.110", - "cli_version": version, - "timeout_s": pre_matcher.timeout, - "delay_s": delay, - "configured": configured, - "audit": audit, - "stderr": errors, - } - finally: - server.shutdown() - server.server_close() - thread.join(timeout=2) - if lifecycle: - from microvm_lifecycle import unregister_task - - unregister_task(lifecycle) - - -if __name__ == "__main__": - parser = argparse.ArgumentParser() - parser.add_argument("--timeout", type=float) - parser.add_argument("--delay", type=float, default=3) - parser.add_argument("--configured", choices=["microvm", "standard"]) - args = parser.parse_args() - if args.configured and args.timeout is not None: - parser.error("--configured uses production settings; omit --timeout") - if args.delay <= 0 or args.delay > MAX_PROBE_DELAY_S: - parser.error("--delay must be within (0, 900] seconds") - for key in tuple(os.environ): - if key.startswith(("AWS_", "ANTHROPIC_", "CLAUDE_", "OTEL_", "BEDROCK_")): - del os.environ[key] - result = asyncio.run( - asyncio.wait_for(probe(args.timeout, args.delay, args.configured), max(args.delay + 30, 45)) - ) - calls = [event for event in result["audit"] if event["kind"] == "pre-start"] - posts = [event for event in result["audit"] if event["kind"] == "post"] - outcomes = [ - item - for event in result["audit"] - if event["kind"] == "model-request" - for item in event["results"] - ] - require(len(calls) == 1 and len(outcomes) == 1, "Expected one tool call and one result") - if args.configured: - require(len(posts) == 1 and not outcomes[0].get("is_error"), "Read did not complete") - require("OWNED_READ_MARKER" in str(outcomes[0]["content"]), "Owned marker not read") - elif args.timeout is not None and args.timeout < args.delay: - require(not posts and outcomes[0].get("is_error"), "Expired hook permitted the tool") - require( - any(event["kind"] == "pre-cancelled" for event in result["audit"]), - "Expected callback cancellation", - ) - result["verified"] = True - print(json.dumps(result, indent=2)) diff --git a/cdk/AGENTS.md b/cdk/AGENTS.md index a9d4cde9a..35f6c75d5 100644 --- a/cdk/AGENTS.md +++ b/cdk/AGENTS.md @@ -31,7 +31,6 @@ mise //cdk:destroy # destroy stack | Handler entrypoints | `cdk/test/handlers/orchestrate-task.test.ts`, `create-task.test.ts`, `webhook-create-task.test.ts` | | Constructs | `cdk/test/constructs/task-orchestrator.test.ts`, `task-api.test.ts` | | MicroVM lifecycle/capacity transactions | `test/handlers/shared/*-local.test.ts` and `test/handlers/request-approval-local.test.ts`; use a loopback DynamoDB Local endpoint in `ABCA_DDB_LOCAL_ENDPOINT` (mandatory in CI) | -| Staged flat-to-nested migration helpers | `src/migration/`, `test/migration/`; these are not a deployable migration command | | Live verification harnesses | `test/live/`; explicitly invoked against an owned fixture, excluded from normal Jest collection | Construct tests: synthesize each distinct stack config once in `beforeAll`, assert against cached `Template` — do not re-synth per test. Bundling is disabled globally via `test/setup/disable-bundling.ts` (see Common mistakes). diff --git a/cdk/src/migration/permissions.ts b/cdk/src/migration/permissions.ts deleted file mode 100644 index 4440e74b7..000000000 --- a/cdk/src/migration/permissions.ts +++ /dev/null @@ -1,162 +0,0 @@ -/** - * MIT No Attribution - * - * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. - * - * Permission is hereby granted, free of charge, to any person obtaining a copy of - * the Software without restriction, including without limitation the rights to - * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of - * the Software, and to permit persons to whom the Software is furnished to do so. - * - * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR - * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, - * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE - * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER - * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, - * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE - * SOFTWARE. - */ - -import { canonical, digest, MigrationTemplate, referencesAny } from './template'; - -interface Policy { - id: string; - roles: any[]; - statements: any[]; - save: () => void; -} - -const POLICY_ID_HASH_LENGTH = 12; -// Leave space below IAM's 6,144-character quota for resolved ARN values. -const POLICY_DOCUMENT_BUDGET = 5000; - -const list = (value: any): any[] => Array.isArray(value) ? value : [value]; -const unique = (values: any[]): any[] => [...new Map(values.map(value => [canonical(value), value])).values()]; -const roleKey = (roles: any[]): string => canonical(roles.map(canonical).sort()); - -/** Include CDK overflow policies attached through Role.ManagedPolicyArns. */ -function policies(template: MigrationTemplate): Policy[] { - const result: Policy[] = []; - const attached = new Map(); - for (const [id, resource] of Object.entries(template.Resources)) { - if (resource.Type !== 'AWS::IAM::Role') continue; - for (const arn of resource.Properties?.ManagedPolicyArns ?? []) { - if (arn?.Ref) attached.set(arn.Ref, [...(attached.get(arn.Ref) ?? []), { Ref: id }]); - } - } - const add = (id: string, roles: any[], holder: Record): void => { - if (!roles.length) return; - const encoded = typeof holder.PolicyDocument === 'string'; - const document = encoded ? JSON.parse(holder.PolicyDocument) : holder.PolicyDocument; - if (!document || !Array.isArray(document.Statement)) throw new Error(`Unsupported IAM policy document: ${id}`); - result.push({ - id, - roles, - statements: document.Statement, - save: () => { holder.PolicyDocument = encoded ? JSON.stringify(document) : document; }, - }); - }; - for (const [id, resource] of Object.entries(template.Resources)) { - const props = resource.Properties ?? {}; - if (['AWS::IAM::Policy', 'AWS::IAM::ManagedPolicy'].includes(resource.Type)) { - add(id, unique([...(props.Roles ?? []), ...(attached.get(id) ?? [])]), props); - } else if (resource.Type === 'AWS::IAM::Role') { - for (const policy of props.Policies ?? []) add(`${id}/${policy.PolicyName}`, [{ Ref: id }], policy); - } - } - return result; -} - -function shape(statement: Record): string { - const copy = structuredClone(statement); - delete copy.Resource; - delete copy.NotResource; - delete copy.Sid; - for (const field of ['Action', 'NotAction']) { - if (copy[field]) copy[field] = list(copy[field]).sort(); - } - return canonical(copy); -} - -function mergeScope(target: Record, source: Record): void { - for (const field of ['Resource', 'NotResource']) { - if (Object.hasOwn(target, field) !== Object.hasOwn(source, field)) { - throw new Error('IAM migration cannot change Resource into NotResource'); - } - if (target[field] !== undefined) target[field] = unique([...list(target[field]), ...list(source[field])]); - } -} - -/** - * Keep legacy access on its original roles while both resource sets exist. - * Separate managed policies avoid exceeding a role's aggregate inline-policy - * quota. They disappear from the final template after the retirement check. - */ -export function preserveLegacyPermissions( - baseline: MigrationTemplate, - target: MigrationTemplate, - legacyIds: ReadonlySet, -): { template: MigrationTemplate; bridgePolicyIds: string[] } { - const template = structuredClone(target); - const current = policies(template); - const bridges = new Map(); - for (const old of policies(baseline)) { - if (legacyIds.has(old.id)) continue; // The old build/operator policy is retained verbatim. - const affected = old.statements.filter(statement => referencesAny(statement, legacyIds)); - if (!affected.length) continue; - const matchingPolicies = current.filter(policy => roleKey(policy.roles) === roleKey(old.roles)); - if (!matchingPolicies.length) throw new Error(`Missing original IAM principal for migration policy ${old.id}`); - const statements = matchingPolicies.flatMap(policy => policy.statements); - for (const statement of affected) { - const matching = statements.filter(candidate => shape(candidate) === shape(statement)); - if (statement.Effect === 'Deny') { - if (matching.length !== 1) throw new Error(`Ambiguous legacy Deny in ${old.id}; refusing to append another deny`); - mergeScope(matching[0], statement); - continue; - } - if (statement.Effect !== 'Allow' || !statement.Resource || statement.NotResource || statement.Principal) { - throw new Error(`Unsupported legacy identity grant in ${old.id}`); - } - const key = roleKey(old.roles); - const bridge = bridges.get(key) ?? { roles: old.roles, statements: [] }; - const retained = structuredClone(statement); - delete retained.Sid; - bridge.statements.push(retained); - bridges.set(key, bridge); - - // P2 allowed direct reads of the old payload bucket and had no bootstrap - // deny. P3's new deny must also exempt those pre-existing read resources - // until old workers drain. This does not grant access to the NEW bucket. - const oldActions: string[] = list(statement.Action ?? []); - if (oldActions.some(action => /^s3:(GetObject\*?|\*)$/i.test(action))) { - const oldObjects = list(statement.Resource).filter(value => referencesAny(value, legacyIds)); - const denies = statements.filter(candidate => - candidate.Effect === 'Deny' && candidate.NotResource - && list(candidate.Action ?? []).some(action => /^s3:(GetObject\*?|\*)$/i.test(action)), - ); - if (denies.length > 1) throw new Error(`Multiple S3 bootstrap denies for ${old.id}; explicit review required`); - for (const deny of denies) deny.NotResource = unique([...list(deny.NotResource), ...oldObjects]); - } - } - } - for (const policy of current) policy.save(); - const bridgePolicyIds: string[] = []; - for (const bridge of bridges.values()) { - const id = `MicrovmMigrationAccess${digest(bridge.roles).slice(0, POLICY_ID_HASH_LENGTH)}`; - if (template.Resources[id]) throw new Error(`Migration policy identifier already exists: ${id}`); - const document = { Version: '2012-10-17', Statement: unique(bridge.statements) }; - if (Buffer.byteLength(JSON.stringify(document)) > POLICY_DOCUMENT_BUDGET) { - throw new Error(`Legacy grants for ${id} need more than one managed policy; explicit review required`); - } - template.Resources[id] = { - Type: 'AWS::IAM::ManagedPolicy', - Properties: { - Description: 'Temporary original-role access while the MicroVM migration drains old workers', - Roles: bridge.roles, - PolicyDocument: document, - }, - }; - bridgePolicyIds.push(id); - } - return { template, bridgePolicyIds }; -} diff --git a/cdk/src/migration/template.ts b/cdk/src/migration/template.ts deleted file mode 100644 index 4c28cd0cf..000000000 --- a/cdk/src/migration/template.ts +++ /dev/null @@ -1,139 +0,0 @@ -/** - * MIT No Attribution - * - * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. - * - * Permission is hereby granted, free of charge, to any person obtaining a copy of - * the Software without restriction, including without limitation the rights to - * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of - * the Software, and to permit persons to whom the Software is furnished to do so. - * - * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR - * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, - * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE - * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER - * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, - * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE - * SOFTWARE. - */ - -import { createHash } from 'node:crypto'; - -export interface TemplateResource { - Type: string; - Properties?: Record; - DependsOn?: string | string[]; - DeletionPolicy?: string; - UpdateReplacePolicy?: string; - [key: string]: any; -} - -export interface MigrationTemplate { - Resources: Record; - Parameters?: Record; - Conditions?: Record; - Outputs?: Record; - [key: string]: any; -} - -/** Template digests use code-unit key order; runtime receipts use localeCompare. Do not interchange persisted hashes. */ -export function canonical(value: unknown): string { - const sort = (item: any): any => { - if (Array.isArray(item)) return item.map(sort); - if (item !== null && typeof item === 'object') { - return Object.fromEntries(Object.keys(item).sort().map(key => [key, sort(item[key])])); - } - return item; - }; - return JSON.stringify(sort(value)) ?? 'undefined'; -} - -export function digest(value: unknown): string { - return createHash('sha256').update(canonical(value)).digest('hex'); -} - -/** References in prose, asset names and metadata are not CloudFormation dependencies. */ -export function references(value: unknown): Set { - const result = new Set(); - const visit = (item: any): void => { - if (!item || typeof item !== 'object') return; - if (Array.isArray(item)) { - item.forEach(visit); - return; - } - if (typeof item.Ref === 'string') result.add(item.Ref); - const attribute = item['Fn::GetAtt']; - if (Array.isArray(attribute) && typeof attribute[0] === 'string') result.add(attribute[0]); - if (typeof attribute === 'string') result.add(attribute.split('.')[0]!); - const substitution = item['Fn::Sub']; - if (substitution !== undefined) { - const expression = typeof substitution === 'string' ? substitution : substitution[0]; - const bindings = typeof substitution === 'string' ? {} : substitution[1] ?? {}; - if (typeof expression !== 'string') throw new Error('Invalid Fn::Sub in migration template'); - for (const match of expression.matchAll(/\$\{([^}]+)\}/g)) { - const token = match[1]!; - if (token.startsWith('!') || Object.hasOwn(bindings, token)) continue; - result.add(token.split('.')[0]!); - } - visit(bindings); - } - for (const [key, child] of Object.entries(item)) { - if (key !== 'Fn::Sub' && key !== 'Metadata') visit(child); - } - }; - visit(value); - return result; -} - -export function referencesAny(value: unknown, ids: ReadonlySet): boolean { - return [...references(value)].some(id => ids.has(id)); -} - -/** Refuse incomplete staged templates before publishing any asset or change set. */ -export function validateTemplate(template: MigrationTemplate, label: string): void { - if (!template.Resources || typeof template.Resources !== 'object' - || Array.isArray(template.Resources) || !Object.keys(template.Resources).length) { - throw new Error(`${label}: template must contain resources`); - } - const count = Object.keys(template.Resources).length; - if (count > 500) throw new Error(`${label}: ${count} resources exceed the CloudFormation limit of 500`); - if (Buffer.byteLength(JSON.stringify(template)) > 1024 * 1024) { - throw new Error(`${label}: template exceeds the CloudFormation 1 MiB limit`); - } - for (const section of ['Parameters', 'Outputs'] as const) { - if (Object.keys(template[section] ?? {}).length > 200) { - throw new Error(`${label}: ${section} exceeds the CloudFormation limit of 200`); - } - } - const known = new Set([...Object.keys(template.Resources), ...Object.keys(template.Parameters ?? {})]); - for (const ref of references(template)) { - if (!known.has(ref) && !ref.startsWith('AWS::')) { - throw new Error(`${label}: unresolved reference ${ref}`); - } - } - for (const [id, resource] of Object.entries(template.Resources)) { - if (!resource || typeof resource.Type !== 'string') throw new Error(`${label}: invalid resource ${id}`); - const dependencies = resource.DependsOn - ? (Array.isArray(resource.DependsOn) ? resource.DependsOn : [resource.DependsOn]) - : []; - for (const dependency of dependencies) { - if (!template.Resources[dependency]) throw new Error(`${label}: ${id} depends on missing ${dependency}`); - } - } -} - -export interface TemplateDelta { - added: string[]; - removed: string[]; - modified: string[]; -} - -export function templateDelta(before: MigrationTemplate, after: MigrationTemplate): TemplateDelta { - return { - added: Object.keys(after.Resources).filter(id => !before.Resources[id]).sort(), - removed: Object.keys(before.Resources).filter(id => !after.Resources[id]).sort(), - modified: Object.keys(after.Resources).filter(id => - before.Resources[id] && canonical(before.Resources[id]) !== canonical(after.Resources[id]), - ).sort(), - }; -} diff --git a/cdk/test/migration/permissions.test.ts b/cdk/test/migration/permissions.test.ts deleted file mode 100644 index 2530a79b0..000000000 --- a/cdk/test/migration/permissions.test.ts +++ /dev/null @@ -1,117 +0,0 @@ -/** - * MIT No Attribution - * - * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. - * - * Permission is hereby granted, free of charge, to any person obtaining a copy of - * the Software without restriction, including without limitation the rights to - * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of - * the Software, and to permit persons to whom the Software is furnished to do so. - * - * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR - * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, - * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE - * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER - * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, - * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE - * SOFTWARE. - */ - -import { preserveLegacyPermissions } from '../../src/migration/permissions'; -import { MigrationTemplate } from '../../src/migration/template'; - -const oldObject = { 'Fn::Sub': '${OldPayload.Arn}/*' }; -const newObject = { 'Fn::Sub': '${Child.Outputs.PayloadArn}/bootstrap/*' }; -const ids = new Set(['OldPayload', 'OldImage']); -const policy = (statements: any[], role = 'Worker') => ({ - Type: 'AWS::IAM::Policy', - Properties: { Roles: [{ Ref: role }], PolicyDocument: { Statement: statements } }, -}); -const allow = (resource: any) => ({ Effect: 'Allow', Action: ['s3:GetObject*'], Resource: resource }); -const deny = (resource: any) => ({ Effect: 'Deny', Action: ['s3:GetObject*'], NotResource: resource }); -const baseline = (): MigrationTemplate => ({ Resources: { WorkerPolicy: policy([allow(oldObject)]) } }); -const target = (): MigrationTemplate => ({ - Resources: { WorkerPolicy: policy([allow(newObject), deny(newObject)]) }, -}); - -describe('migration permission overlap', () => { - test('P2 direct reads remain possible on the original role without granting new-bucket payload reads', () => { - const original = baseline(); - const next = target(); - const { template, bridgePolicyIds } = preserveLegacyPermissions(original, next, ids); - const statements = template.Resources.WorkerPolicy!.Properties!.PolicyDocument.Statement; - expect(statements.filter((statement: any) => statement.Effect === 'Deny')).toEqual([ - { ...deny(newObject), NotResource: [newObject, oldObject] }, - ]); - expect(bridgePolicyIds).toHaveLength(1); - expect(template.Resources[bridgePolicyIds[0]!]!.Properties).toMatchObject({ - Roles: [{ Ref: 'Worker' }], - PolicyDocument: { Statement: [allow(oldObject)] }, - }); - expect(original).toEqual(baseline()); - expect(next).toEqual(target()); - }); - - test('P3 flat bootstrap deny merges exceptions into one deny', () => { - const original = baseline(); - original.Resources.WorkerPolicy = policy([allow(oldObject), deny(oldObject)]); - const { template } = preserveLegacyPermissions(original, target(), ids); - const denies = template.Resources.WorkerPolicy!.Properties!.PolicyDocument.Statement - .filter((statement: any) => statement.Effect === 'Deny'); - expect(denies).toEqual([{ ...deny(newObject), NotResource: [newObject, oldObject] }]); - }); - - test('rejects unmatched legacy NotResource deny rather than denying both buckets', () => { - const original = baseline(); - original.Resources.WorkerPolicy = policy([{ ...deny(oldObject), Condition: { Bool: { Example: true } } }]); - expect(() => preserveLegacyPermissions(original, target(), ids)).toThrow('Ambiguous legacy Deny'); - }); - - test('does not share a worker bridge with the coordinator or retain unrelated grants', () => { - const original = baseline(); - original.Resources.WorkerPolicy!.Properties!.PolicyDocument.Statement.push(allow({ 'Fn::Sub': '${Other.Arn}/*' })); - original.Resources.CoordinatorPolicy = policy([ - { Effect: 'Allow', Action: ['lambda:TerminateMicrovm'], Resource: { Ref: 'OldImage' } }, - ], 'Coordinator'); - const next = target(); - next.Resources.CoordinatorPolicy = policy([ - { Effect: 'Allow', Action: ['lambda:TerminateMicrovm'], Resource: { Ref: 'NewImage' } }, - ], 'Coordinator'); - const { template, bridgePolicyIds } = preserveLegacyPermissions(original, next, ids); - expect(bridgePolicyIds).toHaveLength(2); - const worker = bridgePolicyIds.map(id => template.Resources[id]!.Properties!) - .find(props => props.Roles[0].Ref === 'Worker'); - expect(worker.PolicyDocument.Statement).toEqual([allow(oldObject)]); - }); - - test('finds managed overflow policies attached from the role and preserves JSON-string documents', () => { - const next = target(); - next.Resources.Worker = { - Type: 'AWS::IAM::Role', Properties: { ManagedPolicyArns: [{ Ref: 'Overflow' }] }, - }; - next.Resources.Overflow = { - Type: 'AWS::IAM::ManagedPolicy', - Properties: { PolicyDocument: JSON.stringify({ Statement: [deny(newObject)] }) }, - }; - next.Resources.WorkerPolicy = policy([allow(newObject)]); - const { template } = preserveLegacyPermissions(baseline(), next, ids); - expect(typeof template.Resources.Overflow!.Properties!.PolicyDocument).toBe('string'); - expect(JSON.parse(template.Resources.Overflow!.Properties!.PolicyDocument).Statement[0].NotResource) - .toEqual([newObject, oldObject]); - }); - - test('keeps a removed role or unsupported grant from being silently reassigned', () => { - const next = target(); - next.Resources.WorkerPolicy!.Properties!.Roles = [{ Ref: 'OtherWorker' }]; - expect(() => preserveLegacyPermissions(baseline(), next, ids)).toThrow('Missing original IAM principal'); - const original = baseline(); - original.Resources.WorkerPolicy!.Properties!.PolicyDocument.Statement[0].Principal = '*'; - expect(() => preserveLegacyPermissions(original, target(), ids)).toThrow('Unsupported legacy identity grant'); - }); - - test('rejects duplicate bootstrap denies', () => { - const next = target(); - next.Resources.WorkerPolicy!.Properties!.PolicyDocument.Statement.push(deny({ Ref: 'Other' })); - expect(() => preserveLegacyPermissions(baseline(), next, ids)).toThrow('Multiple S3 bootstrap denies'); - }); -}); diff --git a/cdk/test/migration/template.test.ts b/cdk/test/migration/template.test.ts deleted file mode 100644 index b33cfdf95..000000000 --- a/cdk/test/migration/template.test.ts +++ /dev/null @@ -1,115 +0,0 @@ -/** - * MIT No Attribution - * - * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. - * - * Permission is hereby granted, free of charge, to any person obtaining a copy of - * the Software without restriction, including without limitation the rights to - * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of - * the Software, and to permit persons to whom the Software is furnished to do so. - * - * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR - * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, - * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE - * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER - * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, - * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE - * SOFTWARE. - */ - -import { - canonical, digest, MigrationTemplate, references, referencesAny, templateDelta, validateTemplate, -} from '../../src/migration/template'; - -const fixture = (): MigrationTemplate => ({ - Parameters: { Environment: { Type: 'String' } }, - Resources: { - Bucket: { Type: 'AWS::S3::Bucket' }, - Consumer: { - Type: 'AWS::IAM::Policy', - DependsOn: 'Bucket', - Properties: { - Resource: { 'Fn::Sub': '${Bucket.Arn}/${Environment}/${AWS::Region}/*' }, - }, - }, - }, - Outputs: { Name: { Value: { Ref: 'Bucket' } } }, -}); - -describe('migration template boundaries', () => { - test('recognizes Ref, both GetAtt forms, Sub bindings and escaped variables', () => { - expect(references({ - one: { Ref: 'Role' }, - two: { 'Fn::GetAtt': ['Bucket', 'Arn'] }, - three: { 'Fn::GetAtt': 'Image.ImageArn' }, - four: { 'Fn::Sub': ['${Alias}/${!Literal}/${Target.Arn}/${AWS::Region}', { Alias: { Ref: 'Bound' } }] }, - prose: 'OldBucket', - Metadata: { Ref: 'NotARealReference' }, - })).toEqual(new Set(['Role', 'Bucket', 'Image', 'Target', 'AWS::Region', 'Bound'])); - expect(referencesAny({ 'Fn::Sub': '${OldBucket.Arn}/*' }, new Set(['OldBucket']))).toBe(true); - expect(referencesAny('OldBucket', new Set(['OldBucket']))).toBe(false); - }); - - test('hashes object key order consistently without changing ordered arrays or strings', () => { - expect(digest({ a: 1, b: 2 })).toBe(digest({ b: 2, a: 1 })); - expect(digest([1, 2])).not.toBe(digest([2, 1])); - expect(canonical({ policy: '{"Version":"2012-10-17"}' })).not.toBe( - canonical({ policy: { Version: '2012-10-17' } }), - ); - }); - - test('accepts parameters and AWS pseudo-parameters in nested resource references', () => { - expect(() => validateTemplate(fixture(), 'prepare')).not.toThrow(); - }); - - test.each([ - { Ref: 'OldBucket' }, - { 'Fn::GetAtt': ['OldBucket', 'Arn'] }, - { 'Fn::GetAtt': 'OldBucket.Arn' }, - { 'Fn::Sub': 'arn:${AWS::Partition}:s3:::${OldBucket}/*' }, - ])('rejects stale resource references in retirement: %p', (value) => { - const template = fixture(); - template.Outputs!.Stale = { Value: value }; - expect(() => validateTemplate(template, 'retire')).toThrow('retire: unresolved reference OldBucket'); - }); - - test('rejects dangling DependsOn separately from property references', () => { - const template = fixture(); - template.Resources.Consumer!.DependsOn = ['Deleted']; - expect(() => validateTemplate(template, 'retire')).toThrow('depends on missing Deleted'); - }); - - test.each(['Parameters', 'Outputs'] as const)('rejects oversized %s', (section) => { - const template = fixture(); - template[section] = Object.fromEntries(Array.from({ length: 201 }, (_, index) => [`P${index}`, {}])); - expect(() => validateTemplate(template, 'cutover')).toThrow(`${section} exceeds`); - }); - - test('rejects overlap exceeding 500 resources instead of dropping legacy resources to fit', () => { - const template: MigrationTemplate = { - Resources: Object.fromEntries(Array.from({ length: 501 }, (_, index) => [`R${index}`, { Type: 'AWS::S3::Bucket' }])), - }; - expect(() => validateTemplate(template, 'cutover')).toThrow('501 resources'); - }); - - test('rejects oversized templates and malformed resource collections', () => { - const template = fixture(); - template.Description = 'x'.repeat(1024 * 1024); - expect(() => validateTemplate(template, 'prepare')).toThrow('1 MiB'); - expect(() => validateTemplate({ Resources: {} }, 'prepare')).toThrow('must contain resources'); - expect(() => references({ 'Fn::Sub': [null, {}] })).toThrow('Invalid Fn::Sub'); - }); - - test('reports additions, removals and changes without mutating the baseline', () => { - const before = fixture(); - const snapshot = structuredClone(before); - const after = structuredClone(before); - after.Resources.Bucket!.DeletionPolicy = 'Retain'; - delete after.Resources.Consumer; - after.Resources.Child = { Type: 'AWS::CloudFormation::Stack' }; - expect(templateDelta(before, after)).toEqual({ - added: ['Child'], removed: ['Consumer'], modified: ['Bucket'], - }); - expect(before).toEqual(snapshot); - }); -}); From 9c1255f38e8f9e87b7106aa86fcc5504cf1b0005 Mon Sep 17 00:00:00 2001 From: t Date: Wed, 23 Sep 2026 16:14:36 +0000 Subject: [PATCH 149/149] refactor(approvals): share the post-commit audit and wake tail approve-task.ts and deny-task.ts carried identical ~60-line blocks for the approval_decision_recorded audit write and the best-effort MicroVM wake that this PR added. Move them to recordDecisionPostCommit in a separate module so the handler tests' module mock of wakeMicrovmAfterApproval still applies. Behaviour, event shapes and error handling are unchanged. Refs #645 Co-Authored-By: Claude Fable 5.1 --- cdk/src/handlers/approve-task.ts | 84 +++----------- cdk/src/handlers/deny-task.ts | 81 +++----------- cdk/src/handlers/shared/approval-decision.ts | 111 +++++++++++++++++++ 3 files changed, 147 insertions(+), 129 deletions(-) create mode 100644 cdk/src/handlers/shared/approval-decision.ts diff --git a/cdk/src/handlers/approve-task.ts b/cdk/src/handlers/approve-task.ts index be32195ef..6331541c1 100644 --- a/cdk/src/handlers/approve-task.ts +++ b/cdk/src/handlers/approve-task.ts @@ -18,14 +18,13 @@ */ import { TransactionCanceledException } from '@aws-sdk/client-dynamodb'; -import { PutCommand, TransactWriteCommand, UpdateCommand } from '@aws-sdk/lib-dynamodb'; +import { TransactWriteCommand, UpdateCommand } from '@aws-sdk/lib-dynamodb'; import type { APIGatewayProxyEvent, APIGatewayProxyResult, Context } from 'aws-lambda'; import { ulid } from 'ulid'; +import { recordDecisionPostCommit } from './shared/approval-decision'; import { VALID_APPROVAL_SCOPE_PREFIXES, parseApprovalScope } from './shared/approval-scope'; import { extractUserId } from './shared/gateway'; import { logger } from './shared/logger'; -import { APPROVAL_AUDIT_TIMEOUT_MS, approvalPostCommitOptions, wakeMicrovmAfterApproval } from './shared/microvm-approval-wake'; -import { microvmErrorIdentity } from './shared/microvm-control'; import { formatMinuteBucket, RATE_LIMIT_ROW_TTL_SECONDS } from './shared/rate-limit'; import { ErrorCode, errorResponse, successResponse } from './shared/response'; import type { ApprovalRequest, ApprovalResponse, ApprovalScope } from './shared/types'; @@ -228,69 +227,22 @@ export async function recordApprovalForUser( throw err; } - const postCommit = approvalPostCommitOptions(invocationStartedMs, context); - // 5. Audit event (IMPL-6). Failure to write the audit is logged - // but does not fail the request — the decision is already - // committed on TaskApprovalsTable. A sleeping worker is also recovered by - // the durable supervisor if this request cannot finish its optional wake. - try { - const abortSignal = AbortSignal.any([postCommit.abortSignal!, AbortSignal.timeout(APPROVAL_AUDIT_TIMEOUT_MS)]); - abortSignal.throwIfAborted(); - await ddb.send(new PutCommand({ - TableName: EVENTS_TABLE_NAME, - Item: { - task_id: taskId, - event_id: ulid(), - event_type: 'approval_decision_recorded', - timestamp: nowIso, - ttl: nowEpoch + AUDIT_EVENT_RETENTION_DAYS * 86400, - metadata: { - request_id, - status: 'APPROVED', - scope, - decided_at: nowIso, - caller_user_id: callerUserId, - }, - }, - }), { abortSignal }); - } catch (auditErr) { - logger.warn('approval_decision_recorded audit write failed (decision already committed)', { - task_id: taskId, - request_id, - ...microvmErrorIdentity(auditErr), - }); - } - - // Wake is best-effort after commit. Even an unexpected helper failure must - // not turn an accepted human decision into an HTTP failure. - try { - await wakeMicrovmAfterApproval({ - taskId, - userId: callerUserId, - requestId: request_id, - decision: 'APPROVED', - options: postCommit, - emitEvent: async (eventType, metadata, options) => { - options.abortSignal?.throwIfAborted(); - await ddb.send(new PutCommand({ - TableName: EVENTS_TABLE_NAME, - Item: { - task_id: taskId, - user_id: callerUserId, - event_id: ulid(), - event_type: eventType, - timestamp: new Date().toISOString(), - ttl: nowEpoch + AUDIT_EVENT_RETENTION_DAYS * 86400, - metadata, - }, - }), options); - }, - }); - } catch (wakeError) { - logger.warn('MicroVM wake helper failed after decision commit', { - task_id: taskId, request_id, ...microvmErrorIdentity(wakeError), - }); - } + // Audit + optional MicroVM wake are best-effort after commit (shared with + // deny); neither can turn an accepted human decision into an HTTP failure. + await recordDecisionPostCommit({ + ddb, + eventsTableName: EVENTS_TABLE_NAME!, + taskId, + callerUserId, + requestId: request_id, + decision: 'APPROVED', + auditMetadata: { scope: scope }, + decidedAt: nowIso, + nowEpoch, + retentionDays: AUDIT_EVENT_RETENTION_DAYS, + invocationStartedMs, + context, + }); logger.info('Approval recorded', { task_id: taskId, diff --git a/cdk/src/handlers/deny-task.ts b/cdk/src/handlers/deny-task.ts index 4bac0452c..dcf248352 100644 --- a/cdk/src/handlers/deny-task.ts +++ b/cdk/src/handlers/deny-task.ts @@ -18,14 +18,13 @@ */ import { TransactionCanceledException } from '@aws-sdk/client-dynamodb'; -import { PutCommand, TransactWriteCommand, UpdateCommand } from '@aws-sdk/lib-dynamodb'; +import { TransactWriteCommand, UpdateCommand } from '@aws-sdk/lib-dynamodb'; import type { APIGatewayProxyEvent, APIGatewayProxyResult, Context } from 'aws-lambda'; import { ulid } from 'ulid'; +import { recordDecisionPostCommit } from './shared/approval-decision'; import { scanDenyReason } from './shared/deny-reason-scanner'; import { extractUserId } from './shared/gateway'; import { logger } from './shared/logger'; -import { APPROVAL_AUDIT_TIMEOUT_MS, approvalPostCommitOptions, wakeMicrovmAfterApproval } from './shared/microvm-approval-wake'; -import { microvmErrorIdentity } from './shared/microvm-control'; import { formatMinuteBucket, RATE_LIMIT_ROW_TTL_SECONDS } from './shared/rate-limit'; import { ErrorCode, errorResponse, successResponse } from './shared/response'; import { DENY_REASON_MAX_LENGTH, type DenyRequest, type DenyResponse } from './shared/types'; @@ -205,66 +204,22 @@ export async function recordDenialForUser( throw err; } - const postCommit = approvalPostCommitOptions(invocationStartedMs, context); - // 5. Audit event. - try { - const abortSignal = AbortSignal.any([postCommit.abortSignal!, AbortSignal.timeout(APPROVAL_AUDIT_TIMEOUT_MS)]); - abortSignal.throwIfAborted(); - await ddb.send(new PutCommand({ - TableName: EVENTS_TABLE_NAME, - Item: { - task_id: taskId, - event_id: ulid(), - event_type: 'approval_decision_recorded', - timestamp: nowIso, - ttl: nowEpoch + AUDIT_EVENT_RETENTION_DAYS * 86400, - metadata: { - request_id, - status: 'DENIED', - reason: sanitizedReason, - decided_at: nowIso, - caller_user_id: callerUserId, - }, - }, - }), { abortSignal }); - } catch (auditErr) { - logger.warn('approval_decision_recorded audit write failed (decision already committed)', { - task_id: taskId, - request_id, - ...microvmErrorIdentity(auditErr), - }); - } - - // Wake is best-effort after commit. Even an unexpected helper failure must - // not turn an accepted human decision into an HTTP failure. - try { - await wakeMicrovmAfterApproval({ - taskId, - userId: callerUserId, - requestId: request_id, - decision: 'DENIED', - options: postCommit, - emitEvent: async (eventType, metadata, options) => { - options.abortSignal?.throwIfAborted(); - await ddb.send(new PutCommand({ - TableName: EVENTS_TABLE_NAME, - Item: { - task_id: taskId, - user_id: callerUserId, - event_id: ulid(), - event_type: eventType, - timestamp: new Date().toISOString(), - ttl: nowEpoch + AUDIT_EVENT_RETENTION_DAYS * 86400, - metadata, - }, - }), options); - }, - }); - } catch (wakeError) { - logger.warn('MicroVM wake helper failed after decision commit', { - task_id: taskId, request_id, ...microvmErrorIdentity(wakeError), - }); - } + // Audit + optional MicroVM wake are best-effort after commit (shared with + // approve); neither can turn an accepted human decision into an HTTP failure. + await recordDecisionPostCommit({ + ddb, + eventsTableName: EVENTS_TABLE_NAME!, + taskId, + callerUserId, + requestId: request_id, + decision: 'DENIED', + auditMetadata: { reason: sanitizedReason }, + decidedAt: nowIso, + nowEpoch, + retentionDays: AUDIT_EVENT_RETENTION_DAYS, + invocationStartedMs, + context, + }); logger.info('Denial recorded', { task_id: taskId, diff --git a/cdk/src/handlers/shared/approval-decision.ts b/cdk/src/handlers/shared/approval-decision.ts new file mode 100644 index 000000000..c80ace88c --- /dev/null +++ b/cdk/src/handlers/shared/approval-decision.ts @@ -0,0 +1,111 @@ +/** + * MIT No Attribution + * + * Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of + * the Software, and to permit persons to whom the Software is furnished to do so. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + */ + +// SPDX-License-Identifier: MIT-0 + +import { PutCommand, type DynamoDBDocumentClient } from '@aws-sdk/lib-dynamodb'; +import type { Context } from 'aws-lambda'; +import { ulid } from 'ulid'; +import { logger } from './logger'; +import { APPROVAL_AUDIT_TIMEOUT_MS, approvalPostCommitOptions, wakeMicrovmAfterApproval } from './microvm-approval-wake'; +import { microvmErrorIdentity } from './microvm-control'; + +export interface DecisionPostCommitInput { + readonly ddb: DynamoDBDocumentClient; + readonly eventsTableName: string; + readonly taskId: string; + readonly callerUserId: string; + readonly requestId: string; + readonly decision: 'APPROVED' | 'DENIED'; + /** Decision-specific audit fields (approve: `scope`; deny: `reason`). */ + readonly auditMetadata: Record; + readonly decidedAt: string; + readonly nowEpoch: number; + readonly retentionDays: number; + readonly invocationStartedMs: number; + readonly context?: Pick; +} + +/** + * Shared approve/deny tail after the decision transaction commits: write the + * `approval_decision_recorded` audit event, then attempt the optional MicroVM + * wake. Neither step may fail the request — the human decision is already + * committed on TaskApprovalsTable, and the durable supervisor recovers a + * sleeping worker if the wake cannot finish here. + */ +export async function recordDecisionPostCommit(input: DecisionPostCommitInput): Promise { + const postCommit = approvalPostCommitOptions(input.invocationStartedMs, input.context); + const ttl = input.nowEpoch + input.retentionDays * 86400; + try { + const abortSignal = AbortSignal.any([postCommit.abortSignal!, AbortSignal.timeout(APPROVAL_AUDIT_TIMEOUT_MS)]); + abortSignal.throwIfAborted(); + await input.ddb.send(new PutCommand({ + TableName: input.eventsTableName, + Item: { + task_id: input.taskId, + event_id: ulid(), + event_type: 'approval_decision_recorded', + timestamp: input.decidedAt, + ttl, + metadata: { + request_id: input.requestId, + status: input.decision, + ...input.auditMetadata, + decided_at: input.decidedAt, + caller_user_id: input.callerUserId, + }, + }, + }), { abortSignal }); + } catch (auditErr) { + logger.warn('approval_decision_recorded audit write failed (decision already committed)', { + task_id: input.taskId, + request_id: input.requestId, + ...microvmErrorIdentity(auditErr), + }); + } + + try { + await wakeMicrovmAfterApproval({ + taskId: input.taskId, + userId: input.callerUserId, + requestId: input.requestId, + decision: input.decision, + options: postCommit, + emitEvent: async (eventType, metadata, options) => { + options.abortSignal?.throwIfAborted(); + await input.ddb.send(new PutCommand({ + TableName: input.eventsTableName, + Item: { + task_id: input.taskId, + user_id: input.callerUserId, + event_id: ulid(), + event_type: eventType, + timestamp: new Date().toISOString(), + ttl, + metadata, + }, + }), options); + }, + }); + } catch (wakeError) { + logger.warn('MicroVM wake helper failed after decision commit', { + task_id: input.taskId, request_id: input.requestId, ...microvmErrorIdentity(wakeError), + }); + } +}