diff --git a/backend/scripts/experiment_agent_build_arms.py b/backend/scripts/experiment_agent_build_arms.py deleted file mode 100644 index 001262361..000000000 --- a/backend/scripts/experiment_agent_build_arms.py +++ /dev/null @@ -1,299 +0,0 @@ -"""Agent-build A/B: does sharing boto3 clients help a first turn? - -Drives real first turns through the deployed AgentCore Runtime and compares the -two arms of ``agent_build_experiment_arm`` (``apis/shared/feature_flags.py``): - - control today's build - shared_clients the session manager, strategy-id discovery and Bedrock - model client share one process-wide boto3 session, built - at container warm-up - -**Precondition:** the Runtime must have ``AGENT_BUILD_EXPERIMENT=ab``. Arms are -assigned server-side by hashing the session id, so this script generates session -ids until each lands in the arm it wants, then runs the arms interleaved (one of -each per round) so network drift over the run hits every arm alike. Every -conversation runs in its own Runtime process, which is exactly the first-turn, -cold-process build under test (``processBuilds`` = 1 on ``turn_prelude``). - -Two sources per turn, joined on session id: - -- **Client side** (this script, via ``on_event``): when ``preparing``, - ``prepared``, ``session_title``, the first ``content_block_delta`` and - ``done`` arrived, measured from the request. -- **Server side** (``turn_prelude`` in the Runtime log group): the build's - sub-stages, ``buildArm`` and ``processBuilds``. - -What it establishes: per-arm medians for the build and its sub-stages, time to -first token, and whether the title beats ``prepared`` (it cannot while the build -is synchronous on the loop; the column is kept so a regression there is visible). -What it cannot: fleet magnitude, or behaviour under concurrent load. Report which -claim you make. - -Each turn is a real conversation owned by ``--user-id``: it draws on that user's -quota and appears in their sidebar. Use ``--cleanup`` to soft-delete the -experiment's sessions afterwards. - -Usage (an authenticated dev-ai profile and an active headless grant for ---user-id; see apis/shared/harness/grants.py): - - cd backend - AWS_PROFILE=dev-ai uv run python scripts/experiment_agent_build_arms.py \\ - --user-id --per-arm 15 --cleanup -""" - -from __future__ import annotations - -import argparse -import asyncio -import json -import logging -import os -import re -import statistics -import sys -import time -import uuid -from dataclasses import asdict, dataclass, field -from typing import Any, Dict, List, Optional - -logging.basicConfig(level=logging.INFO, format="%(levelname)s %(name)s: %(message)s") -logging.getLogger("httpx").setLevel(logging.WARNING) -logger = logging.getLogger("experiment") - -sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) -sys.path.insert(0, os.path.join(os.path.dirname(os.path.abspath(__file__)), "..", "src")) - -ARMS = ("control", "shared_clients") - -# Frames whose first arrival is recorded, keyed by how the table names them. -_TIMED = ("preparing", "prepared", "session_title", "first_token", "done") - - -@dataclass -class Turn: - arm: str - session_id: str - ok: bool = False - error: Optional[str] = None - client_ms: Dict[str, int] = field(default_factory=dict) - prelude: Dict[str, Any] = field(default_factory=dict) - - -def session_for(arm: str) -> str: - """A fresh session id that the server's ``ab`` hash puts in ``arm``.""" - from apis.shared.feature_flags import agent_build_experiment_arm - - previous = os.environ.get("AGENT_BUILD_EXPERIMENT") - os.environ["AGENT_BUILD_EXPERIMENT"] = "ab" - try: - while True: - candidate = str(uuid.uuid4()) - if agent_build_experiment_arm(candidate) == arm: - return candidate - finally: - if previous is None: - os.environ.pop("AGENT_BUILD_EXPERIMENT", None) - else: - os.environ["AGENT_BUILD_EXPERIMENT"] = previous - - -async def run_turn(*, arm: str, user_id: str, prompt: str, auth: Any, model_id: Optional[str]) -> Turn: - from apis.shared.harness import run_agent_headless - - turn = Turn(arm=arm, session_id=session_for(arm)) - started = time.monotonic() - - def stamp(key: str) -> None: - turn.client_ms.setdefault(key, int((time.monotonic() - started) * 1000)) - - async def on_event(name: str, data: Dict[str, Any]) -> None: - if name == "agent_status" and data.get("phase") in ("preparing", "prepared"): - stamp(data["phase"]) - elif name == "session_title": - stamp("session_title") - elif name == "content_block_delta": - stamp("first_token") - elif name == "done": - stamp("done") - - try: - run = await run_agent_headless( - user_id=user_id, - prompt=prompt, - auth=auth, - session_id=turn.session_id, - model_id=model_id, - trigger="experiment", - on_event=on_event, - ) - turn.ok = run.status == "completed" - turn.error = None if turn.ok else (run.error or run.status) - except Exception as exc: # noqa: BLE001 - one bad turn must not end the run - turn.error = str(exc)[:200] - logger.info( - " %-24s %s %s", arm, turn.session_id[:8], - "ok" if turn.ok else f"FAILED: {turn.error}", - ) - return turn - - -def attach_preludes(turns: List[Turn], log_group: str, region: str, since: float) -> None: - """Join each turn to its ``turn_prelude`` line by session id.""" - import boto3 - - logs = boto3.client("logs", region_name=region) - by_session = {t.session_id: t for t in turns} - query = logs.start_query( - logGroupName=log_group, - startTime=int(since) - 60, - endTime=int(time.time()) + 60, - queryString="fields body | filter body like /turn_prelude/ | limit 10000", - )["queryId"] - while True: - result = logs.get_query_results(queryId=query) - if result["status"] in ("Complete", "Failed", "Cancelled", "Timeout"): - break - time.sleep(2) - for row in result.get("results", []): - body = next((f["value"] for f in row if f["field"] == "body"), "") - match = re.search(r"turn_prelude (\{.*\})", body) - if not match: - continue - prelude = json.loads(match.group(1)) - turn = by_session.get(prelude.get("sessionId")) - if turn is not None: - turn.prelude = prelude - - -def _median(values: List[float]) -> str: - return f"{statistics.median(values):.0f}" if values else "-" - - -def _p75(values: List[float]) -> str: - if len(values) < 4: - return "-" - return f"{statistics.quantiles(values, n=4)[2]:.0f}" - - -def report(turns: List[Turn]) -> None: - def stage(t: Turn, name: str) -> Optional[float]: - return t.prelude.get("stages", {}).get(name) - - rows = [ - ("agent_build (group)", lambda t: t.prelude.get("groups", {}).get("agent_build")), - (" session_mgr_clients", lambda t: stage(t, "agent_build.session_mgr_clients")), - (" session_mgr (network)", lambda t: stage(t, "agent_build.session_mgr")), - (" strands_agent", lambda t: stage(t, "agent_build.strands_agent")), - (" finalize (restore)", lambda t: stage(t, "agent_build.finalize")), - ("prelude total", lambda t: t.prelude.get("totalMs")), - ("client: prepared", lambda t: t.client_ms.get("prepared")), - ("client: first token", lambda t: t.client_ms.get("first_token")), - ("client: session_title", lambda t: t.client_ms.get("session_title")), - ( - "title minus prepared", - lambda t: (t.client_ms["session_title"] - t.client_ms["prepared"]) - if "session_title" in t.client_ms and "prepared" in t.client_ms else None, - ), - ] - - print() - header = f"{'metric (ms)':26s}" + "".join(f"{arm:>28s}" for arm in ARMS) - print(header) - print(f"{'':26s}" + "".join(f"{'median / p75':>28s}" for _ in ARMS)) - for label, fn in rows: - cells = [] - for arm in ARMS: - values = [v for t in turns if t.arm == arm and t.ok and (v := fn(t)) is not None] - cells.append(f"{_median(values)} / {_p75(values)} (n={len(values)})") - print(f"{label:26s}" + "".join(f"{c:>28s}" for c in cells)) - - print() - for arm in ARMS: - mine = [t for t in turns if t.arm == arm] - matched = [t for t in mine if t.prelude] - mislabeled = [t for t in matched if t.prelude.get("buildArm") != arm] - warm = [t for t in matched if t.prelude.get("processBuilds") not in (1, None)] - early = [ - t for t in mine - if "session_title" in t.client_ms and "prepared" in t.client_ms - and t.client_ms["session_title"] < t.client_ms["prepared"] - ] - print( - f"{arm:24s} turns={len(mine)} ok={sum(t.ok for t in mine)} " - f"prelude-matched={len(matched)} arm-mismatch={len(mislabeled)} " - f"not-first-build={len(warm)} title-before-prepared={len(early)}" - ) - if any(t.prelude and t.prelude.get("buildArm") == "control" for t in turns if t.arm != "control"): - print("\n⚠️ Non-control turns reported buildArm=control: is AGENT_BUILD_EXPERIMENT=ab set on the Runtime?") - - -def cleanup(turns: List[Turn], user_id: str) -> None: - """Soft-delete the experiment's conversations from the user's sidebar.""" - import boto3 - - table = boto3.resource("dynamodb", region_name=os.environ["AWS_REGION"]).Table( - os.environ["DYNAMODB_SESSIONS_METADATA_TABLE_NAME"] - ) - removed = 0 - for turn in turns: - try: - table.update_item( - Key={"PK": f"USER#{user_id}", "SK": f"S#{turn.session_id}"}, - UpdateExpression="SET #s = :deleted REMOVE GSI4_PK, GSI4_SK", - ConditionExpression="attribute_exists(PK)", - ExpressionAttributeNames={"#s": "status"}, - ExpressionAttributeValues={":deleted": "deleted"}, - ) - removed += 1 - except Exception as exc: # noqa: BLE001 - best effort - logger.warning("cleanup %s: %s", turn.session_id, exc) - logger.info("soft-deleted %d/%d experiment sessions", removed, len(turns)) - - -async def main() -> None: - parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) - parser.add_argument("--user-id", required=True) - parser.add_argument("--per-arm", type=int, default=15) - parser.add_argument("--gap-seconds", type=float, default=3.0) - parser.add_argument("--model-id", default=None) - parser.add_argument("--prefix", default="dev-boisestateai-v2") - parser.add_argument("--region", default="us-west-2") - parser.add_argument("--out", default=None, help="write raw per-turn JSON here") - parser.add_argument("--cleanup", action="store_true", help="soft-delete the sessions afterwards") - args = parser.parse_args() - - from spike_headless_run import resolve_environment - from apis.shared.harness.auth import CognitoRefreshBearerAuth - - env = resolve_environment(args.prefix, args.region) - runtime_id = env["runtime_arn"].rsplit("/", 1)[1] - log_group = f"/aws/bedrock-agentcore/runtimes/{runtime_id}-DEFAULT" - auth = CognitoRefreshBearerAuth() - - since = time.time() - turns: List[Turn] = [] - for round_index in range(args.per_arm): - order = list(ARMS) - # Rotate the order each round so no arm always runs first. - order = order[round_index % len(order):] + order[: round_index % len(order)] - logger.info("── round %d/%d", round_index + 1, args.per_arm) - for arm in order: - prompt = f"In one short sentence, name a river in country number {round_index + 1} of Africa, alphabetically." - turns.append(await run_turn(arm=arm, user_id=args.user_id, prompt=prompt, auth=auth, model_id=args.model_id)) - await asyncio.sleep(args.gap_seconds) - - logger.info("waiting 45s for turn_prelude lines to reach CloudWatch…") - await asyncio.sleep(45) - attach_preludes(turns, log_group, args.region, since) - report(turns) - - if args.out: - with open(args.out, "w") as fh: - json.dump([asdict(t) for t in turns], fh, indent=2) - logger.info("raw results: %s", args.out) - if args.cleanup: - cleanup(turns, args.user_id) - - -if __name__ == "__main__": - asyncio.run(main()) diff --git a/backend/src/agents/main_agent/chat_agent.py b/backend/src/agents/main_agent/chat_agent.py index 08d651aae..4b71e045b 100644 --- a/backend/src/agents/main_agent/chat_agent.py +++ b/backend/src/agents/main_agent/chat_agent.py @@ -105,7 +105,6 @@ def _create_agent(self) -> None: hooks=hooks, plugins=plugins, memory_context=getattr(self, "memory_context", None), - session_id=getattr(self, "session_id", None), ) except Exception as e: diff --git a/backend/src/agents/main_agent/core/agent_factory.py b/backend/src/agents/main_agent/core/agent_factory.py index 39f8abf46..b2cbf2dea 100644 --- a/backend/src/agents/main_agent/core/agent_factory.py +++ b/backend/src/agents/main_agent/core/agent_factory.py @@ -24,20 +24,18 @@ class AgentFactory: """Factory for creating configured Strands Agent instances with multi-provider support""" @staticmethod - def _create_bedrock_model(model_config: ModelConfig, session_id: Optional[str] = None) -> BedrockModel: + def _create_bedrock_model(model_config: ModelConfig) -> BedrockModel: """ Create a BedrockModel instance Args: model_config: Model configuration - session_id: The conversation, which decides the agent-build A/B - arm (see ``ModelConfig.to_bedrock_config``). Returns: BedrockModel: Configured Bedrock model (a ``CountTokensBedrockModel`` so native CountTokens works for inference-profile model ids). """ - bedrock_config = model_config.to_bedrock_config(session_id=session_id) + bedrock_config = model_config.to_bedrock_config() # Strands awaits count_tokens before every model call; keep that local. # Native counts are taken off the critical path by the # context-attribution hook (native_count_tokens in a background task). @@ -192,7 +190,6 @@ def create_agent( hooks: Optional[List[Any]] = None, plugins: Optional[List[Any]] = None, memory_context: Optional[str] = None, - session_id: Optional[str] = None, ) -> Agent: """ Create a Strands Agent instance with the appropriate model provider @@ -209,8 +206,6 @@ def create_agent( memory_context: Optional rendered Memory-Space block. Sent after the system prompt, behind a cache point of its own when the model supports cache points (see below). - session_id: The conversation this agent serves. Only the Bedrock - provider reads it, to pick the agent-build A/B arm. Returns: Agent: Configured Strands Agent instance @@ -224,7 +219,7 @@ def create_agent( # Create appropriate model based on provider if provider == ModelProvider.BEDROCK: - model = AgentFactory._create_bedrock_model(model_config, session_id=session_id) + model = AgentFactory._create_bedrock_model(model_config) elif provider == ModelProvider.OPENAI: model = AgentFactory._create_openai_model(model_config) elif provider == ModelProvider.MANTLE: diff --git a/backend/src/agents/main_agent/core/model_config.py b/backend/src/agents/main_agent/core/model_config.py index 7770c52d7..2855bfa81 100644 --- a/backend/src/agents/main_agent/core/model_config.py +++ b/backend/src/agents/main_agent/core/model_config.py @@ -414,25 +414,24 @@ def bedrock_cache_points_supported(self) -> bool: and ("claude" in model_lower or "anthropic" in model_lower) ) - def to_bedrock_config(self, session_id: Optional[str] = None) -> Dict[str, Any]: + def to_bedrock_config(self) -> Dict[str, Any]: """Convert to BedrockModel kwargs, translating canonical inference params. - ``session_id`` decides the agent-build A/B arm - (``memory_shared_clients_enabled``): on the shared arm the model is + With ``agent_build_shared_session_enabled`` (default on) the model is built on the process-wide boto3 session instead of the fresh ``boto3.Session()`` Strands would otherwise construct (and re-parse the bedrock-runtime model on). Nothing here reaches the prompt. """ config: Dict[str, Any] = {"model_id": self.model_id} - from apis.shared.feature_flags import memory_shared_clients_enabled + from apis.shared.feature_flags import agent_build_shared_session_enabled - if memory_shared_clients_enabled(session_id): + if agent_build_shared_session_enabled(): from apis.shared.aws_clients import shared_boto_session # Never alongside `region_name`: BedrockModel.__init__ raises when # both are given (strands-agents 1.55.0). This config sets no - # region on either arm; the session resolves it from the + # region either way; the session resolves it from the # environment exactly as Strands' own fresh session would. config["boto_session"] = shared_boto_session() _apply_canonical_params( diff --git a/backend/src/agents/main_agent/session/session_factory.py b/backend/src/agents/main_agent/session/session_factory.py index 6176affc9..8ef60f265 100644 --- a/backend/src/agents/main_agent/session/session_factory.py +++ b/backend/src/agents/main_agent/session/session_factory.py @@ -80,7 +80,7 @@ def session_async_persistence_enabled() -> bool: # --------------------------------------------------------------------------- # # The SDK's session manager builds its clients from scratch on every -# construction, twice (see ``memory_shared_clients_enabled``). Everything +# construction, twice (see ``agent_build_shared_session_enabled``). Everything # below exists to hand it the process-wide session from # ``apis.shared.aws_clients`` instead, which warm-up builds at container # start (``apis/inference_api/warmup.py``). @@ -142,10 +142,10 @@ def _discover_strategy_ids( memory_id: AgentCore Memory ID region: AWS region shared_session: Build the ``MemoryClient`` on the process-wide session - (``memory_shared_clients_enabled``) instead of a fresh one. Part - of the cache key on purpose: warm-up primes the shared entry at - container start, and the control arm's first turn must still do - exactly what it did before the experiment. + (``agent_build_shared_session_enabled``) instead of a fresh one. + Part of the cache key on purpose: warm-up primes the shared entry + at container start, and a process with the kill switch set must + still do exactly what it did before. Returns: Tuple of (semantic_strategy_id, preference_strategy_id, summary_strategy_id) @@ -189,8 +189,7 @@ def warm_strategy_ids() -> Tuple[Optional[str], Optional[str], Optional[str]]: """Discover the memory's strategy ids on the shared session, once, at container start. Called from ``apis/inference_api/warmup.py`` on the startup daemon thread - so the shared arm's first turn finds the ids cached and its clients - built. Raises when no memory is configured (``load_memory_config``); the + so the first turn finds the ids cached and its clients built. Raises when no memory is configured (``load_memory_config``); the warm-up step logs that and moves on. """ if not AGENTCORE_MEMORY_AVAILABLE: @@ -282,12 +281,12 @@ def _create_cloud_session_manager( logger.info(f" • Memory ID: {memory_id}") logger.info(f" • Region: {aws_region}") - # Discover actual strategy IDs from the memory configuration. On the - # shared arm this is a cache hit: warm-up discovered them at + # Discover actual strategy IDs from the memory configuration. With the + # shared session on this is a cache hit: warm-up discovered them at # container start (`warm_strategy_ids`). - from apis.shared.feature_flags import memory_shared_clients_enabled + from apis.shared.feature_flags import agent_build_shared_session_enabled - shared_clients = memory_shared_clients_enabled(session_id) + shared_clients = agent_build_shared_session_enabled() semantic_id, preference_id, summary_id = _discover_strategy_ids( memory_id, aws_region, shared_session=shared_clients ) diff --git a/backend/src/apis/inference_api/chat/routes.py b/backend/src/apis/inference_api/chat/routes.py index 3630bd13b..5bf41e57d 100644 --- a/backend/src/apis/inference_api/chat/routes.py +++ b/backend/src/apis/inference_api/chat/routes.py @@ -27,7 +27,7 @@ ) from apis.inference_api.runtime_health import ping_payload from apis.shared.feature_flags import ( - agent_build_experiment_arm, + agent_build_shared_session_enabled, agent_preparing_phase_enabled, agents_enabled, attachment_turn_guard_enabled, @@ -4165,11 +4165,12 @@ async def _guarded_stream() -> AsyncGenerator[str, None]: "isResume": is_resume, "hasAssistant": bool(input_data.rag_assistant_id), "deferredBuild": deferred_build, - # The agent-build A/B (`agent_build_experiment_arm`), - # and how many builds this process has run: 1 on a - # conversation's first turn, which is always a fresh - # Runtime process. - "buildArm": agent_build_experiment_arm(input_data.session_id), + # Whether the build's SDK clients came from the + # process-wide session (`agent_build_shared_session_enabled`, + # a kill switch), and how many builds this process has + # run: 1 on a conversation's first turn, which is + # always a fresh Runtime process. + "sharedSession": agent_build_shared_session_enabled(), "processBuilds": process_build_count(), }, ) diff --git a/backend/src/apis/inference_api/chat/turn_timing.py b/backend/src/apis/inference_api/chat/turn_timing.py index daad56dc0..54bfdb039 100644 --- a/backend/src/apis/inference_api/chat/turn_timing.py +++ b/backend/src/apis/inference_api/chat/turn_timing.py @@ -280,9 +280,9 @@ def _emit_metrics( metrics.setdefault(_metric_name(prefix), total) properties: Dict[str, Any] = {"streamKind": stream_kind, "sessionId": session_id} - # Properties, never dimensions: `buildArm` is the agent-build A/B and - # `processBuilds` is unbounded. - for key in ("isResume", "deferredBuild", "hasAssistant", "buildArm", "processBuilds"): + # Properties, never dimensions: `sharedSession` is a kill-switch state + # and `processBuilds` is unbounded. + for key in ("isResume", "deferredBuild", "hasAssistant", "sharedSession", "processBuilds"): if extra and key in extra: properties[key] = extra[key] diff --git a/backend/src/apis/inference_api/warmup.py b/backend/src/apis/inference_api/warmup.py index 3b21bcdf2..b98b01deb 100644 --- a/backend/src/apis/inference_api/warmup.py +++ b/backend/src/apis/inference_api/warmup.py @@ -55,10 +55,9 @@ # The parse above is per *session*, and the two SDKs on the agent build (the # AgentCore Memory session manager, Strands' BedrockModel) build their clients -# on the process-wide session from `apis.shared.aws_clients` when the -# `shared_clients` arm is on. Build those clients on it here, so the arm's -# first turn finds them parsed; on the control arm the session sits unused, -# which costs nothing on the request path. +# on the process-wide session from `apis.shared.aws_clients` +# (`agent_build_shared_session_enabled`, default on). Build those clients on +# it here, so the first turn finds them parsed. WARM_SHARED_SESSION_SERVICES: tuple[str, ...] = ( "bedrock-agentcore", "bedrock-agentcore-control", @@ -113,6 +112,11 @@ def warm_shared_session(services: Iterable[str] = WARM_SHARED_SESSION_SERVICES) restore's first turn is the thing to watch for a pool holding a socket that did not survive. """ + from apis.shared.feature_flags import agent_build_shared_session_enabled + + if not agent_build_shared_session_enabled(): + logger.info("warmup step=shared outcome=skipped error=AGENT_BUILD_SHARED_SESSION_ENABLED=false") + return region = os.environ.get("AWS_REGION") or os.environ.get("AWS_DEFAULT_REGION") if not region: logger.info("warmup step=shared outcome=skipped error=no region configured") diff --git a/backend/src/apis/shared/aws_clients.py b/backend/src/apis/shared/aws_clients.py index 7e4f4f9ae..050b7c218 100644 --- a/backend/src/apis/shared/aws_clients.py +++ b/backend/src/apis/shared/aws_clients.py @@ -28,8 +28,8 @@ touches (the parse is per session, not per process), so a cold first turn paid that parse several times over — see `docs/specs/turn-path-ttft.md` §5 P2. Both SDKs accept a session, so `shared_boto_session()` is the one -process-wide session handed to them (behind `memory_shared_clients_enabled`, -an A/B arm) and to `apis/inference_api/warmup.py`, which builds its clients +process-wide session handed to them (`agent_build_shared_session_enabled`, +default on) and to `apis/inference_api/warmup.py`, which builds its clients at container start so a first turn finds them already parsed. Its `client()` returns one client per configuration, so every caller that asks for the same service with the same config shares one client and one connection pool. diff --git a/backend/src/apis/shared/feature_flags.py b/backend/src/apis/shared/feature_flags.py index 36c4d42aa..1d11e0ef7 100644 --- a/backend/src/apis/shared/feature_flags.py +++ b/backend/src/apis/shared/feature_flags.py @@ -15,9 +15,7 @@ module reload (import-time paths) without a process restart. """ -import hashlib import os -from typing import Optional def skills_enabled() -> bool: @@ -658,65 +656,29 @@ def compaction_summary_extract_enabled() -> bool: -AGENT_BUILD_ARMS = ("control", "shared_clients") +def agent_build_shared_session_enabled() -> bool: + """Whether the agent build's SDK clients are built on one process-wide boto3 session. + Three things on a first-turn build construct a fresh ``boto3.Session`` + and parse service models on it: the AgentCore Memory session manager + (which builds a ``MemoryClient``, then a second session and clients that + replace the first pair), ``_discover_strategy_ids``, and Strands' + ``BedrockModel``. A fresh session re-parses every model it touches, and + the parse is the cost. With this on, all three are handed the session + from ``apis.shared.aws_clients.shared_boto_session``, which warm-up + builds at container start together with its clients and the strategy + ids, so the first turn finds them ready. -def agent_build_experiment_arm(session_id: Optional[str]) -> str: - """Which agent-build variant this session runs (an A/B experiment, default OFF). - - One change to the first-turn agent build, measured before it ships: - - - ``shared_clients``: AgentCore Memory session managers, the strategy-id - discovery and the Bedrock model client share one process-wide boto3 - session (``memory_shared_clients_enabled``). - - A second arm, ``shared_clients_off_loop`` (the synchronous build on a - worker thread), was withdrawn before the A/B ran: a thread does not make - a synchronous build faster, and the overlap it would have enabled is - reachable inside the constructor without the MCP hardening it needed - (docs/specs/turn-path-ttft.md, sections 4 and 5 P3). - - ``AGENT_BUILD_EXPERIMENT`` selects the mode: - - - unset / empty / anything unrecognised: ``control`` for every session. - This is the default everywhere, so other deployments see no change. - - ``ab``: each session is hashed into one of the two arms. Every - conversation runs in its own Runtime process, so arms never share - process state, and they run interleaved in time, which cancels network - drift between arms. - - an arm name: every session runs that arm. - - The arm is stamped on ``turn_prelude`` (``buildArm``) so Logs Insights can - compare stage timings per arm. There is no CDK entry: the Runtime's - environment is capped at 50 variables and this is a temporary experiment, - so it is set out of band on the Runtime (``update-agent-runtime``), which - ``backend.yml`` deploys preserve and a ``platform.yml`` deploy resets. - """ - mode = os.environ.get("AGENT_BUILD_EXPERIMENT", "").strip().lower() - if mode in AGENT_BUILD_ARMS: - return mode - if mode != "ab" or not session_id: - return "control" - digest = hashlib.sha256(session_id.encode("utf-8")).digest() - return AGENT_BUILD_ARMS[digest[0] % len(AGENT_BUILD_ARMS)] - - -def memory_shared_clients_enabled(session_id: Optional[str]) -> bool: - """Whether this session's AgentCore Memory session manager uses shared clients. - - The SDK's ``AgentCoreMemorySessionManager.__init__`` builds a - ``MemoryClient`` (a fresh ``boto3.Session`` plus two clients) and then a - second fresh session plus two more clients that replace the first pair. - A fresh session re-loads botocore's service models, so every session - manager paid ~360ms of CPU (measured locally, before any network call), - and each one opened its own connection pool, so its first ``list_events`` - also paid a TLS handshake. With this on, the factory hands the SDK one - process-wide session whose ``client()`` returns the same client per - configuration. In a fresh process that halves the model loading; in a - warm one (a later cache-miss build in the same conversation) the clients - cost nothing and their connections are already open. - - Arm of ``agent_build_experiment_arm``; off by default. Nothing reaches - the prompt. + **Default ON with a kill switch** (house style, mirroring + ``agent_status_enabled``): unset or empty resolves to enabled; only the + literal ``"false"`` (case-insensitive) disables. Off, every SDK builds its + own session exactly as before, and warm-up skips the shared step. + + Measured on dev, 2026-09-30, 15 first turns per arm, interleaved, every + turn a cold Runtime process (docs/specs/turn-latency-preamble.md PR-6): + ``agent_build`` median 869ms → 372ms, time to first token at the client + 4211ms → 3564ms, and the two distributions did not overlap. Nothing + reaches the prompt; the stamp on ``turn_prelude`` (``sharedSession``) + is a property, never a dimension. """ - return agent_build_experiment_arm(session_id) == "shared_clients" + return os.environ.get("AGENT_BUILD_SHARED_SESSION_ENABLED", "").strip().lower() != "false" diff --git a/backend/tests/agents/main_agent/core/test_bedrock_model_shared_session.py b/backend/tests/agents/main_agent/core/test_bedrock_model_shared_session.py index 8f00b566b..f26f4132c 100644 --- a/backend/tests/agents/main_agent/core/test_bedrock_model_shared_session.py +++ b/backend/tests/agents/main_agent/core/test_bedrock_model_shared_session.py @@ -1,8 +1,9 @@ -"""On the `shared_clients` arm, Strands' BedrockModel is built on the process-wide -boto3 session instead of the fresh `boto3.Session()` it would construct itself. +"""With `AGENT_BUILD_SHARED_SESSION_ENABLED` on (the default), Strands' BedrockModel +is built on the process-wide boto3 session instead of the fresh `boto3.Session()` +it would construct itself. `BedrockModel.__init__` raises when `region_name` and `boto_session` are both -given (strands-agents 1.55.0), so the arm must never send both. Real objects, +given (strands-agents 1.55.0), so the config must never send both. Real objects, no network: constructing a client opens no socket. """ @@ -23,60 +24,57 @@ def _region_and_fresh_session(monkeypatch): aws_clients.reset_cached_clients() -class TestOnTheArm: +class TestDefaultOn: @pytest.fixture(autouse=True) - def _arm(self, monkeypatch): - monkeypatch.setenv("AGENT_BUILD_EXPERIMENT", "shared_clients") + def _default(self, monkeypatch): + monkeypatch.delenv("AGENT_BUILD_SHARED_SESSION_ENABLED", raising=False) def test_config_carries_the_shared_session_and_no_region(self): - config = ModelConfig(model_id=MODEL_ID).to_bedrock_config(session_id="s") + config = ModelConfig(model_id=MODEL_ID).to_bedrock_config() assert config["boto_session"] is aws_clients.shared_boto_session() assert "region_name" not in config def test_the_model_client_is_the_shared_sessions_client(self): - first = AgentFactory._create_bedrock_model(ModelConfig(model_id=MODEL_ID), session_id="s") - second = AgentFactory._create_bedrock_model(ModelConfig(model_id=MODEL_ID), session_id="s") + first = AgentFactory._create_bedrock_model(ModelConfig(model_id=MODEL_ID)) + second = AgentFactory._create_bedrock_model(ModelConfig(model_id=MODEL_ID)) assert first.client is second.client, "two models, one bedrock-runtime client" assert first.client.meta.config.max_pool_connections == aws_clients.SHARED_SESSION_MAX_POOL_CONNECTIONS - def test_the_arm_resolves_the_same_region_as_a_fresh_session(self, monkeypatch): + def test_the_shared_session_resolves_the_same_region_as_a_fresh_one(self, monkeypatch): """Strands resolves the region from the session it is given, and a fresh `boto3.Session()` reads the same environment the shared one - does, so the arm must land on the same region as control.""" - shared = AgentFactory._create_bedrock_model(ModelConfig(model_id=MODEL_ID), session_id="s") - monkeypatch.setenv("AGENT_BUILD_EXPERIMENT", "control") - control = AgentFactory._create_bedrock_model(ModelConfig(model_id=MODEL_ID), session_id="s") + does, so switching the sharing off must not move the region.""" + shared = AgentFactory._create_bedrock_model(ModelConfig(model_id=MODEL_ID)) + monkeypatch.setenv("AGENT_BUILD_SHARED_SESSION_ENABLED", "false") + own = AgentFactory._create_bedrock_model(ModelConfig(model_id=MODEL_ID)) - assert shared.client.meta.region_name == control.client.meta.region_name + assert shared.client.meta.region_name == own.client.meta.region_name def test_a_deliberate_region_next_to_the_session_is_still_refused(self): - """Pins the Strands contract the arm is written around.""" + """Pins the Strands contract the config is written around.""" from strands.models import BedrockModel with pytest.raises(ValueError, match="both"): BedrockModel(model_id=MODEL_ID, region_name="us-west-2", boto_session=aws_clients.shared_boto_session()) -class TestOffTheArm: - @pytest.mark.parametrize("value", [None, "", "control", "ab"]) +class TestKillSwitch: + @pytest.mark.parametrize("value", ["false", "FALSE", " False "]) def test_config_carries_no_session(self, monkeypatch, value): - if value is None: - monkeypatch.delenv("AGENT_BUILD_EXPERIMENT", raising=False) - else: - monkeypatch.setenv("AGENT_BUILD_EXPERIMENT", value) + monkeypatch.setenv("AGENT_BUILD_SHARED_SESSION_ENABLED", value) - config = ModelConfig(model_id=MODEL_ID).to_bedrock_config(session_id=None) + config = ModelConfig(model_id=MODEL_ID).to_bedrock_config() assert "boto_session" not in config assert "region_name" not in config def test_the_model_builds_its_own_client(self, monkeypatch): - monkeypatch.delenv("AGENT_BUILD_EXPERIMENT", raising=False) + monkeypatch.setenv("AGENT_BUILD_SHARED_SESSION_ENABLED", "false") first = AgentFactory._create_bedrock_model(ModelConfig(model_id=MODEL_ID)) second = AgentFactory._create_bedrock_model(ModelConfig(model_id=MODEL_ID)) assert first.client is not second.client - assert aws_clients._shared_session is None, "control never touches the shared session" + assert aws_clients._shared_session is None, "switched off, nothing touches the shared session" diff --git a/backend/tests/agents/main_agent/session/test_session_factory_shared_clients.py b/backend/tests/agents/main_agent/session/test_session_factory_shared_clients.py index 3d5ac910f..4b45140f9 100644 --- a/backend/tests/agents/main_agent/session/test_session_factory_shared_clients.py +++ b/backend/tests/agents/main_agent/session/test_session_factory_shared_clients.py @@ -28,7 +28,7 @@ def memory_env(monkeypatch): """Real construction, no network: the SDK's session read/create are stubbed.""" monkeypatch.setenv("AGENTCORE_MEMORY_ID", "mem-test") monkeypatch.setenv("AWS_REGION", "us-west-2") - monkeypatch.setenv("AGENT_BUILD_EXPERIMENT", "shared_clients") + monkeypatch.delenv("AGENT_BUILD_SHARED_SESSION_ENABLED", raising=False) # default on monkeypatch.setattr(factory, "_discover_strategy_ids", lambda memory_id, region, **kwargs: (None, None, None)) monkeypatch.setattr(sdk.AgentCoreMemorySessionManager, "read_session", lambda self, session_id, **k: None) monkeypatch.setattr(sdk.AgentCoreMemorySessionManager, "create_session", lambda self, session, **k: session) @@ -101,10 +101,10 @@ def build(i: int) -> Any: assert len({id(m.memory_client.gmdp_client) for m in managers}) == 1 -class TestControlArm: - def test_control_keeps_the_sdks_per_manager_clients(self, memory_env, client_builds, monkeypatch): - """The default: no experiment set means every session is control.""" - monkeypatch.delenv("AGENT_BUILD_EXPERIMENT", raising=False) +class TestKillSwitch: + def test_off_keeps_the_sdks_per_manager_clients(self, memory_env, client_builds, monkeypatch): + """`AGENT_BUILD_SHARED_SESSION_ENABLED=false` restores the SDK's own sessions.""" + monkeypatch.setenv("AGENT_BUILD_SHARED_SESSION_ENABLED", "false") first = _build("session-a") built_by_first = len(client_builds) @@ -141,11 +141,11 @@ def test_different_configs_get_different_clients(self): class TestWhatTheFactoryHandsTheSdk: - """The seam the arm rides on: the SDK's constructor takes ``boto_session``, - and ``MemoryClient`` takes ``boto3_session``. On the arm both get the - shared session; off it, nothing — the SDKs build their own, as before.""" + """The seam the shared session rides on: the SDK's constructor takes ``boto_session``, + and ``MemoryClient`` takes ``boto3_session``. On (the default) both get the + shared session; off, nothing — the SDKs build their own, as before.""" - def test_the_sdk_constructor_gets_the_shared_session_on_the_arm(self, memory_env, monkeypatch): + def test_the_sdk_constructor_gets_the_shared_session_by_default(self, memory_env, monkeypatch): captured = {} original = sdk.AgentCoreMemorySessionManager.__init__ @@ -159,8 +159,8 @@ def spy(self, *args, **kwargs): assert captured["boto_session"] is aws_clients.shared_boto_session() - def test_the_sdk_constructor_gets_nothing_off_the_arm(self, memory_env, monkeypatch): - monkeypatch.delenv("AGENT_BUILD_EXPERIMENT", raising=False) + def test_the_sdk_constructor_gets_nothing_with_the_kill_switch(self, memory_env, monkeypatch): + monkeypatch.setenv("AGENT_BUILD_SHARED_SESSION_ENABLED", "false") captured = {} original = sdk.AgentCoreMemorySessionManager.__init__ @@ -174,7 +174,7 @@ def spy(self, *args, **kwargs): assert captured["boto_session"] is None - def test_strategy_discovery_builds_its_client_on_the_shared_session_on_the_arm(self): + def test_strategy_discovery_builds_its_client_on_the_shared_session(self): fetch = factory._discover_strategy_ids.__wrapped__ with patch.object(factory, "MemoryClient") as memory_client: memory_client.return_value.get_memory_strategies.return_value = [] @@ -184,7 +184,7 @@ def test_strategy_discovery_builds_its_client_on_the_shared_session_on_the_arm(s region_name="us-west-2", boto3_session=aws_clients.shared_boto_session() ) - def test_strategy_discovery_builds_a_fresh_client_off_the_arm(self): + def test_strategy_discovery_builds_a_fresh_client_when_asked_not_to_share(self): fetch = factory._discover_strategy_ids.__wrapped__ with patch.object(factory, "MemoryClient") as memory_client: memory_client.return_value.get_memory_strategies.return_value = [] @@ -192,7 +192,7 @@ def test_strategy_discovery_builds_a_fresh_client_off_the_arm(self): memory_client.assert_called_once_with(region_name="us-west-2", boto3_session=None) - def test_the_factory_asks_for_the_shared_entry_only_on_the_arm(self, memory_env, monkeypatch): + def test_the_factory_asks_for_the_shared_entry_unless_switched_off(self, memory_env, monkeypatch): asked: List[bool] = [] monkeypatch.setattr( factory, @@ -201,13 +201,13 @@ def test_the_factory_asks_for_the_shared_entry_only_on_the_arm(self, memory_env, ) _build("session-a") - monkeypatch.delenv("AGENT_BUILD_EXPERIMENT", raising=False) + monkeypatch.setenv("AGENT_BUILD_SHARED_SESSION_ENABLED", "false") _build("session-b") assert asked == [True, False] def test_warm_strategy_ids_primes_the_shared_entry(self, memory_env, monkeypatch): - """Warm-up's call and the arm's first turn must hit the same cache key, + """Warm-up's call and the first turn must hit the same cache key, or the first turn pays the call warm-up already made.""" with patch.object(factory, "_discover_strategy_ids", return_value=(None, None, None)) as discover: factory.warm_strategy_ids() diff --git a/backend/tests/apis/inference_api/test_warmup.py b/backend/tests/apis/inference_api/test_warmup.py index f6b8c3d87..b4f4b1f35 100644 --- a/backend/tests/apis/inference_api/test_warmup.py +++ b/backend/tests/apis/inference_api/test_warmup.py @@ -148,6 +148,21 @@ def test_skips_without_a_region(self, monkeypatch): warmup.warm_shared_session() session.assert_not_called() + def test_the_kill_switch_skips_the_whole_step(self, monkeypatch): + """Off means the SDKs build their own sessions, so nothing here would + be used — including the one connection the strategy read opens.""" + monkeypatch.setenv("AWS_REGION", "us-west-2") + monkeypatch.setenv("AGENT_BUILD_SHARED_SESSION_ENABLED", "false") + from agents.main_agent.session import session_factory + + with patch("apis.shared.aws_clients.shared_boto_session") as session, patch.object( + session_factory, "_discover_strategy_ids" + ) as discover: + warmup.warm_shared_session() + + session.assert_not_called() + discover.assert_not_called() + def test_a_failing_client_does_not_stop_the_rest(self, monkeypatch): monkeypatch.setenv("AWS_REGION", "us-west-2") session = MagicMock() diff --git a/backend/tests/shared/test_agent_build_experiment_arm.py b/backend/tests/shared/test_agent_build_experiment_arm.py deleted file mode 100644 index 4fa382bca..000000000 --- a/backend/tests/shared/test_agent_build_experiment_arm.py +++ /dev/null @@ -1,60 +0,0 @@ -"""`agent_build_experiment_arm`: default OFF, and a stable per-session split.""" - -import uuid - -import pytest - -from apis.shared.feature_flags import ( - AGENT_BUILD_ARMS, - agent_build_experiment_arm, - memory_shared_clients_enabled, -) - - -@pytest.mark.parametrize("value", [None, "", "off", "false", "true", "nonsense"]) -def test_everything_but_ab_or_an_arm_name_is_control(monkeypatch, value): - if value is None: - monkeypatch.delenv("AGENT_BUILD_EXPERIMENT", raising=False) - else: - monkeypatch.setenv("AGENT_BUILD_EXPERIMENT", value) - - assert agent_build_experiment_arm("s1") == "control" - assert not memory_shared_clients_enabled("s1") - - -@pytest.mark.parametrize("arm", AGENT_BUILD_ARMS) -def test_an_arm_name_forces_that_arm(monkeypatch, arm): - monkeypatch.setenv("AGENT_BUILD_EXPERIMENT", f" {arm.upper()} ") - assert agent_build_experiment_arm("s1") == arm - - -def test_the_shared_arm_implies_its_change(monkeypatch): - monkeypatch.setenv("AGENT_BUILD_EXPERIMENT", "shared_clients") - assert memory_shared_clients_enabled("s") - - monkeypatch.setenv("AGENT_BUILD_EXPERIMENT", "control") - assert not memory_shared_clients_enabled("s") - - -def test_the_withdrawn_off_loop_arm_is_not_an_arm(monkeypatch): - """`shared_clients_off_loop` was withdrawn before the A/B; naming it now - means control, like any other unrecognised value.""" - assert AGENT_BUILD_ARMS == ("control", "shared_clients") - monkeypatch.setenv("AGENT_BUILD_EXPERIMENT", "shared_clients_off_loop") - assert agent_build_experiment_arm("s") == "control" - - -def test_ab_is_stable_per_session_and_uses_every_arm(monkeypatch): - monkeypatch.setenv("AGENT_BUILD_EXPERIMENT", "ab") - sessions = [str(uuid.UUID(int=i)) for i in range(300)] - - arms = [agent_build_experiment_arm(s) for s in sessions] - - assert arms == [agent_build_experiment_arm(s) for s in sessions] - counts = {arm: arms.count(arm) for arm in AGENT_BUILD_ARMS} - assert all(110 <= n <= 190 for n in counts.values()), counts - - -def test_ab_without_a_session_is_control(monkeypatch): - monkeypatch.setenv("AGENT_BUILD_EXPERIMENT", "ab") - assert agent_build_experiment_arm(None) == "control" diff --git a/backend/tests/shared/test_agent_build_shared_session_flag.py b/backend/tests/shared/test_agent_build_shared_session_flag.py new file mode 100644 index 000000000..3dff4dfa2 --- /dev/null +++ b/backend/tests/shared/test_agent_build_shared_session_flag.py @@ -0,0 +1,29 @@ +"""`agent_build_shared_session_enabled`: default ON, `false` is the kill switch.""" + +import pytest + +from apis.shared.feature_flags import agent_build_shared_session_enabled + + +@pytest.mark.parametrize("value", [None, "", "true", "on", "ab", "shared_clients", "nonsense"]) +def test_everything_but_false_is_on(monkeypatch, value): + if value is None: + monkeypatch.delenv("AGENT_BUILD_SHARED_SESSION_ENABLED", raising=False) + else: + monkeypatch.setenv("AGENT_BUILD_SHARED_SESSION_ENABLED", value) + + assert agent_build_shared_session_enabled() + + +@pytest.mark.parametrize("value", ["false", "FALSE", " False "]) +def test_false_is_off(monkeypatch, value): + monkeypatch.setenv("AGENT_BUILD_SHARED_SESSION_ENABLED", value) + assert not agent_build_shared_session_enabled() + + +def test_the_retired_experiment_variable_is_ignored(monkeypatch): + """`AGENT_BUILD_EXPERIMENT` drove the A/B that decided this; a Runtime + still carrying it must not read as anything.""" + monkeypatch.setenv("AGENT_BUILD_EXPERIMENT", "ab") + monkeypatch.delenv("AGENT_BUILD_SHARED_SESSION_ENABLED", raising=False) + assert agent_build_shared_session_enabled() diff --git a/docs-site/src/content/docs/configuration/feature-flags.md b/docs-site/src/content/docs/configuration/feature-flags.md index 5dff75c2c..075246803 100644 --- a/docs-site/src/content/docs/configuration/feature-flags.md +++ b/docs-site/src/content/docs/configuration/feature-flags.md @@ -109,6 +109,7 @@ variables. None of them costs extra against the model **except** tool summaries. | `ADMIN_ALWAYS_ON_TOOLS_ENABLED` | ON | none | Unions admin-flagged `alwaysOn` tools into every turn (inert with no data) | | `COST_DIAGNOSTICS_ENABLED` | ON | none | Content-free behavioral counters for the admin session profile | | `CONFIG_CACHE_ENABLED` | ON | **saves** money | In-process cache of tenant-global catalogs (fewer DynamoDB reads) | +| `AGENT_BUILD_SHARED_SESSION_ENABLED` | ON | none (**saves** ~0.5s on a cold first turn) | One process-wide boto3 session for the agent build's SDK clients (Memory session manager, strategy-id discovery, Bedrock model), built at container warm-up. Off ⇒ each SDK builds its own session, as before | | `DOCUMENT_OFFLOAD_ENABLED` | ON | none | Document-context-offload pipeline (see spec) | | `DOCUMENT_REHYDRATE_ENABLED` | ON | none | Re-injects offloaded document context on demand | | `DOCUMENT_DIGEST_ENABLED` | ON | possible side-channel | Document digest step of the offload pipeline | diff --git a/docs/specs/turn-latency-preamble.md b/docs/specs/turn-latency-preamble.md index b0c18f1aa..120c81174 100644 --- a/docs/specs/turn-latency-preamble.md +++ b/docs/specs/turn-latency-preamble.md @@ -748,7 +748,7 @@ at. Second target after that: `agent_build.session_mgr` at 830ms (AgentCore Memory restore, never timed). Now PR-6. -## PR-6 — agent-build A/B: shared boto3 clients, built at warm-up (IN PROGRESS) +## PR-6 — one shared boto3 session for the agent build, built at warm-up (SHIPPED) The second target named above, `agent_build.session_mgr`, measured **616-738ms** on dev first turns (2026-09-28) — more than half the 1.0-1.2s build. @@ -758,37 +758,67 @@ builds a `MemoryClient` (a fresh `boto3.Session` plus two clients), then a secon fresh session plus two more clients that **replace** the first pair, then calls `read_session`. For a new session that is two sequential `list_events` (the second is a legacy-format fallback) and a `create_event`, each on a cold -connection pool. Laptop timing: ~360ms of client construction per session -manager. PR-3 above is the warning about reading that number: construction was -~30x slower on the container than on a laptop. Two more fresh sessions sit next -to it on the same first turn: `_discover_strategy_ids` builds its own -`MemoryClient`, and Strands' `BedrockModel` builds its own `boto3.Session`. +connection pool. A fresh session re-parses every service model it touches, and +the parse is the cost. Two more fresh sessions sat next to it on the same first +turn: `_discover_strategy_ids` built its own `MemoryClient`, and Strands' +`BedrockModel` built its own `boto3.Session`. **Every first turn is a fresh process.** Each conversation's turns run in their own Runtime process (`service.instance.id` differs per session in the runtime -logs). So a first turn always pays cold construction, and "a build freezes other -users' streams" does not happen: a process serves one conversation. Process-wide -caching helps a first turn only if the shared work is done **before** the turn -arrives, which is why the shared session is built at container warm-up -(`apis/inference_api/warmup.py`, on the startup daemon thread): its -`bedrock-agentcore`, `bedrock-agentcore-control` and `bedrock-runtime` clients -are constructed there, and `_discover_strategy_ids` is called once so the -strategy ids are cached before any turn. That discovery is a control-plane read -of static configuration, and it opens the one connection warm-up otherwise -avoids; `docs/specs/turn-path-ttft.md` §5 P2 accepts that (botocore retries a -connection error on this idempotent call) and names it as the thing to watch on -the Runtime V2 restore. - -**One change, behind one per-session experiment flag, default off** -(`agent_build_experiment_arm` in `apis/shared/feature_flags.py`, -`AGENT_BUILD_EXPERIMENT`): - -| Arm | Change | -|---|---| -| `control` | today's build | -| `shared_clients` | the factory hands the SDK the process-wide boto3 session (whose `client()` returns one client per configuration) and rebinds the SDK module's `MemoryClient` so the discarded pair is built from it too; `_discover_strategy_ids` builds its `MemoryClient` on it; `ModelConfig.to_bedrock_config` passes it to `BedrockModel` as `boto_session` (and then no `region_name`, which Strands rejects alongside a session) | - -**The off-loop arm was withdrawn before the A/B.** The PR as opened had a third +logs). So a first turn always pays cold construction, and process-wide caching +helps it only if the shared work is done **before** the turn arrives. That is why +the session lives in `apis.shared.aws_clients` (`shared_boto_session`, a +`ClientReusingSession` whose `client()` returns one client per configuration) +and is built at container warm-up (`apis/inference_api/warmup.py`, on the +startup daemon thread): its `bedrock-agentcore`, `bedrock-agentcore-control` +and `bedrock-runtime` clients are constructed there, and `_discover_strategy_ids` +is called once so the strategy ids are cached before any turn. That discovery is +a control-plane read of static configuration, and it opens the one connection +warm-up otherwise avoids; `docs/specs/turn-path-ttft.md` §5 P2 accepts that +(botocore retries a connection error on this idempotent call) and names it as +the thing to watch on the Runtime V2 restore. + +**Who gets the session.** The factory hands it to the SDK session manager +(`boto_session=`) and rebinds the SDK module's `MemoryClient` so the discarded +pair is built from it too (the only seam the pinned bedrock-agentcore offers for +a one-line upstream bug; a test fails if an upgrade stops going through it); +`_discover_strategy_ids` builds its `MemoryClient` on it; and +`ModelConfig.to_bedrock_config` passes it to `BedrockModel` as `boto_session` +(and then no `region_name`, which Strands rejects alongside a session). + +**The A/B that decided it** (dev, 2026-09-30, `AGENT_BUILD_EXPERIMENT=ab` set on +the Runtime out of band, 15 first turns per arm interleaved in rotating order, +every turn a cold process, all 30 ok, arms verified server-side, sessions +soft-deleted afterwards). Median / p75 in ms: + +| Stage | control | shared session | +|---|---|---| +| `agent_build.session_mgr_clients` | 446 / 497 | 8 / 9 | +| `agent_build.session_mgr` (network) | 197 / 217 | 207 / 237 | +| `agent_build.strands_agent` | 55 / 72 | 5 / 5 | +| `agent_build.finalize` (restore) | 100 / 109 | 99 / 103 | +| `agent_build` (group) | 869 / 955 | 372 / 408 | +| prelude total | 1497 / 1655 | 1003 / 1049 | +| client: first token | 4211 / 4632 | 3564 / 3902 | + +Decision metric (`session_mgr_clients` + `session_mgr`): control ranged 601 to +813, the shared arm 171 to 289 — the distributions do not overlap, and no stage +regressed (`session_mgr`'s +10 is inside its own spread). The `strands_agent` +drop is `BedrockModel`'s fresh session going away. Warm-up did its part on the +Runtime: logs show every fresh container building the three shared clients in +15-53ms and reading the strategy ids at warm-up (median 193ms, one outlier at +3.2s, all ok). The rule written before the data — ship if the decision metric +beats control by more than the run-to-run spread with no regression — was met, +so the experiment flag and its arms were replaced by one kill switch: + +- **`AGENT_BUILD_SHARED_SESSION_ENABLED`** — default ON; only the literal + `false` disables (house style). Off, every SDK builds its own session exactly + as before and warm-up skips the shared step. No CDK entry (the Runtime is at + 48 of its 50 environment variables and unset means on); a deployment that + needs it off sets the variable on the Runtime out of band. `turn_prelude` + carries `sharedSession` (a property) and `processBuilds` (1 on a first turn). + +**Withdrawn before the A/B: the off-loop arm.** The PR as opened had a third arm, `shared_clients_off_loop`, which ran `create_agent` under `asyncio.to_thread` so the route could emit a first-turn title mid-build. It came with hardening the frozen loop used to provide for free: process-wide @@ -804,29 +834,16 @@ constructor itself (§5 P3a, two threads of one executor for `session_mgr` and `tools`) without any of the hardening. A title that lands before `prepared` is not worth shipping locks for. -**New sub-stages.** `agent_build.session_mgr_clients` (everything before the SDK's -`read_session`, i.e. client setup) now precedes `agent_build.session_mgr` (the -session read/create network). `agent_build.strands_agent` (Strands' own `Agent` -construction, including MCP `load_tools`) now precedes `agent_build.finalize` -(the session restore). `turn_prelude` carries `buildArm` and `processBuilds` -(1 on a first turn). - -**How to run it.** Set `AGENT_BUILD_EXPERIMENT=ab` on the dev Runtime out of band -(`update-agent-runtime`; the Runtime is at 48 of its 50 environment variables, -and a temporary experiment should not take a CDK slot). `backend.yml` deploys -preserve it; a `platform.yml` deploy resets it. Then: - - cd backend - AWS_PROFILE=dev-ai uv run python scripts/experiment_agent_build_arms.py \ - --user-id --per-arm 15 --cleanup - -Arms are assigned by hashing the session id, and the script runs one turn per -arm per round in rotating order, so time-of-day drift hits every arm alike. - -**Decision rule, fixed before the data.** Ship `shared_clients` (default on, with -a kill switch) if its median `agent_build.session_mgr_clients` + -`agent_build.session_mgr` beats control's by more than the run-to-run spread -and no stage regresses. Otherwise remove the arm and keep the instrumentation. +**Sub-stages that stay.** `agent_build.session_mgr_clients` (everything before +the SDK's `read_session`, i.e. client setup) precedes `agent_build.session_mgr` +(the session read/create network). `agent_build.strands_agent` (Strands' own +`Agent` construction, including MCP `load_tools`) precedes `agent_build.finalize` +(the session restore). + +**Next target after this:** the two `list_events` in `session_mgr` (~200ms) and +the restore's `ListEvents` in `finalize` (~100ms), which are now the whole of +the build's network — see `turn-path-ttft.md` §5 P3 for overlapping them with +`tools`. ## Declined / overtaken — `asyncio.to_thread` for the DynamoDB calls