@aws/agentcore 1.0.0-preview.17 → 1.0.0-preview.19

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (77) hide show
  1. package/README.md +17 -17
  2. package/dist/assets/__tests__/__snapshots__/assets.snapshot.test.ts.snap +531 -90
  3. package/dist/assets/__tests__/googleadk-session-eviction.test.ts +87 -0
  4. package/dist/assets/__tests__/summarization-namespace.test.ts +43 -0
  5. package/dist/assets/python/a2a/strands/capabilities/memory/session.py +1 -1
  6. package/dist/assets/python/agui/strands/capabilities/memory/session.py +1 -1
  7. package/dist/assets/python/http/autogen/base/main.py +33 -10
  8. package/dist/assets/python/http/googleadk/base/main.py +45 -8
  9. package/dist/assets/python/http/langchain_langgraph/base/main.py +46 -7
  10. package/dist/assets/python/http/openaiagents/base/main.py +19 -8
  11. package/dist/assets/python/http/strands/base/main.py +36 -29
  12. package/dist/assets/python/http/strands/base/mcp_client/client.py +4 -1
  13. package/dist/assets/python/http/strands/base/model/load.py +116 -0
  14. package/dist/assets/python/http/strands/base/model/mantle_compat.py +21 -0
  15. package/dist/assets/python/http/strands/base/pyproject.toml +3 -0
  16. package/dist/assets/python/http/strands/capabilities/memory/session.py +1 -1
  17. package/dist/assets/typescript/http/strands/base/main.ts +96 -18
  18. package/dist/assets/typescript/http/strands/base/package.json +3 -2
  19. package/dist/assets/typescript/http/strands/capabilities/memory/memory.ts +52 -0
  20. package/dist/assets/typescript/http/vercelai/base/main.ts +42 -2
  21. package/dist/cli/index.mjs +574 -608
  22. package/dist/lib/errors/types.d.ts +25 -0
  23. package/dist/lib/errors/types.d.ts.map +1 -1
  24. package/dist/lib/errors/types.js +40 -1
  25. package/dist/lib/errors/types.js.map +1 -1
  26. package/dist/lib/secrets/cipher.d.ts +12 -0
  27. package/dist/lib/secrets/cipher.d.ts.map +1 -0
  28. package/dist/lib/secrets/cipher.js +54 -0
  29. package/dist/lib/secrets/cipher.js.map +1 -0
  30. package/dist/lib/secrets/index.d.ts +4 -0
  31. package/dist/lib/secrets/index.d.ts.map +1 -0
  32. package/dist/lib/secrets/index.js +15 -0
  33. package/dist/lib/secrets/index.js.map +1 -0
  34. package/dist/lib/secrets/key-provider.d.ts +16 -0
  35. package/dist/lib/secrets/key-provider.d.ts.map +1 -0
  36. package/dist/lib/secrets/key-provider.js +191 -0
  37. package/dist/lib/secrets/key-provider.js.map +1 -0
  38. package/dist/lib/secrets/sensitive-keys.d.ts +21 -0
  39. package/dist/lib/secrets/sensitive-keys.d.ts.map +1 -0
  40. package/dist/lib/secrets/sensitive-keys.js +67 -0
  41. package/dist/lib/secrets/sensitive-keys.js.map +1 -0
  42. package/dist/lib/utils/env.d.ts +4 -2
  43. package/dist/lib/utils/env.d.ts.map +1 -1
  44. package/dist/lib/utils/env.js +57 -27
  45. package/dist/lib/utils/env.js.map +1 -1
  46. package/dist/schema/constants.d.ts +29 -2
  47. package/dist/schema/constants.d.ts.map +1 -1
  48. package/dist/schema/constants.js +41 -5
  49. package/dist/schema/constants.js.map +1 -1
  50. package/dist/schema/schemas/agent-env.d.ts +47 -2
  51. package/dist/schema/schemas/agent-env.d.ts.map +1 -1
  52. package/dist/schema/schemas/agent-env.js +34 -5
  53. package/dist/schema/schemas/agent-env.js.map +1 -1
  54. package/dist/schema/schemas/agentcore-project.d.ts +46 -1
  55. package/dist/schema/schemas/agentcore-project.d.ts.map +1 -1
  56. package/dist/schema/schemas/auth.d.ts +2 -3
  57. package/dist/schema/schemas/auth.d.ts.map +1 -1
  58. package/dist/schema/schemas/auth.js +8 -7
  59. package/dist/schema/schemas/auth.js.map +1 -1
  60. package/dist/schema/schemas/connections.d.ts +185 -0
  61. package/dist/schema/schemas/connections.d.ts.map +1 -0
  62. package/dist/schema/schemas/connections.js +176 -0
  63. package/dist/schema/schemas/connections.js.map +1 -0
  64. package/dist/schema/schemas/index.d.ts +1 -0
  65. package/dist/schema/schemas/index.d.ts.map +1 -1
  66. package/dist/schema/schemas/index.js +1 -0
  67. package/dist/schema/schemas/index.js.map +1 -1
  68. package/dist/schema/schemas/primitives/config-bundle.d.ts +1 -0
  69. package/dist/schema/schemas/primitives/config-bundle.d.ts.map +1 -1
  70. package/dist/schema/schemas/primitives/config-bundle.js +3 -0
  71. package/dist/schema/schemas/primitives/config-bundle.js.map +1 -1
  72. package/dist/schema/schemas/primitives/harness.d.ts +43 -0
  73. package/dist/schema/schemas/primitives/harness.d.ts.map +1 -1
  74. package/dist/schema/schemas/primitives/harness.js +20 -0
  75. package/dist/schema/schemas/primitives/harness.js.map +1 -1
  76. package/npm-shrinkwrap.json +223 -0
  77. package/package.json +4 -1
@@ -831,6 +831,7 @@ exports[`Assets Directory Snapshots > File listing > should match the expected f
831
831
  "python/http/strands/base/mcp_client/client.py",
832
832
  "python/http/strands/base/model/__init__.py",
833
833
  "python/http/strands/base/model/load.py",
834
+ "python/http/strands/base/model/mantle_compat.py",
834
835
  "python/http/strands/base/pyproject.toml",
835
836
  "python/http/strands/base/skills/fetcher.py",
836
837
  "python/http/strands/capabilities/execution-limits/hooks/execution_limits.py",
@@ -850,6 +851,7 @@ exports[`Assets Directory Snapshots > File listing > should match the expected f
850
851
  "typescript/http/strands/base/model/load.ts",
851
852
  "typescript/http/strands/base/package.json",
852
853
  "typescript/http/strands/base/tsconfig.json",
854
+ "typescript/http/strands/capabilities/memory/memory.ts",
853
855
  "typescript/http/vercelai/base/README.md",
854
856
  "typescript/http/vercelai/base/gitignore.template",
855
857
  "typescript/http/vercelai/base/main.ts",
@@ -2255,7 +2257,7 @@ def get_memory_session_manager(session_id: Optional[str], actor_id: str) -> Opti
2255
2257
  f"/episodes/{actor_id}/{session_id}": RetrievalConfig(top_k=5, relevance_score=0.5),
2256
2258
  {{/if}}
2257
2259
  {{#if (includes memoryProviders.[0].strategies "SUMMARIZATION")}}
2258
- f"/summaries/{actor_id}/{session_id}": RetrievalConfig(top_k=3, relevance_score=0.5),
2260
+ f"/summaries/{actor_id}": RetrievalConfig(top_k=3, relevance_score=0.5),
2259
2261
  {{/if}}
2260
2262
  }
2261
2263
  {{/if}}
@@ -3277,7 +3279,7 @@ def get_memory_session_manager(session_id: Optional[str], actor_id: str) -> Opti
3277
3279
  f"/episodes/{actor_id}/{session_id}": RetrievalConfig(top_k=5, relevance_score=0.5),
3278
3280
  {{/if}}
3279
3281
  {{#if (includes memoryProviders.[0].strategies "SUMMARIZATION")}}
3280
- f"/summaries/{actor_id}/{session_id}": RetrievalConfig(top_k=3, relevance_score=0.5),
3282
+ f"/summaries/{actor_id}": RetrievalConfig(top_k=3, relevance_score=0.5),
3281
3283
  {{/if}}
3282
3284
  }
3283
3285
  {{/if}}
@@ -3388,6 +3390,7 @@ exports[`Assets Directory Snapshots > Python framework assets > python/python/ht
3388
3390
  "{{#if needsOs}}
3389
3391
  import os
3390
3392
  {{/if}}
3393
+ from collections import OrderedDict
3391
3394
  from autogen_agentchat.agents import AssistantAgent
3392
3395
  from autogen_core.tools import FunctionTool
3393
3396
  from bedrock_agentcore.runtime import BedrockAgentCoreApp
@@ -3478,23 +3481,45 @@ You have access to the following mounted filesystems. Use file_read, file_write,
3478
3481
  {{/each}}{{/if}}
3479
3482
  """
3480
3483
 
3481
- @app.entrypoint
3482
- async def invoke(payload, context):
3483
- log.info("Invoking Agent.....")
3484
+ # Reuses one AssistantAgent per session_id so each session keeps its own
3485
+ # in-process conversation history (best-effort; resets on cold start). Caches up
3486
+ # to 128 active sessions with LRU eviction (least-recently-used is dropped and
3487
+ # its history reset).
3488
+ _agents = OrderedDict()
3484
3489
 
3490
+
3491
+ async def get_or_create_agent(session_id):
3492
+ if session_id in _agents:
3493
+ _agents.move_to_end(session_id)
3494
+ return _agents[session_id]
3495
+ if len(_agents) >= 128:
3496
+ _agents.popitem(last=False)
3485
3497
  # Get MCP Tools
3486
3498
  mcp_tools = await get_streamable_http_mcp_tools()
3499
+ # Re-check after the await: a concurrent first-invocation for the same
3500
+ # session_id may have built and stored the agent while we were awaiting.
3501
+ # Don't overwrite it (that would orphan the agent the other request is using).
3502
+ if session_id not in _agents:
3503
+ _agents[session_id] = AssistantAgent(
3504
+ name="{{ name }}",
3505
+ model_client=load_model(),
3506
+ tools=tools + mcp_tools,
3507
+ system_message=SYSTEM_MESSAGE,
3508
+ )
3509
+ _agents.move_to_end(session_id)
3510
+ return _agents[session_id]
3487
3511
 
3488
- # Define an AssistantAgent with the model and tools
3489
- agent = AssistantAgent(
3490
- name="{{ name }}",
3491
- model_client=load_model(),
3492
- tools=tools + mcp_tools,
3493
- system_message=SYSTEM_MESSAGE,
3494
- )
3512
+
3513
+ @app.entrypoint
3514
+ async def invoke(payload, context):
3515
+ log.info("Invoking Agent.....")
3495
3516
 
3496
3517
  # Process the user prompt
3497
3518
  prompt = payload.get("prompt", "What can you help me with?")
3519
+ session_id = getattr(context, "session_id", "default-session")
3520
+
3521
+ # Reuse the per-session agent (preserves conversation history)
3522
+ agent = await get_or_create_agent(session_id)
3498
3523
 
3499
3524
  # Run the agent
3500
3525
  result = await agent.run(task=prompt)
@@ -3820,6 +3845,7 @@ exports[`Assets Directory Snapshots > Python framework assets > python/python/ht
3820
3845
  "{{#if needsOs}}
3821
3846
  import os
3822
3847
  {{/if}}
3848
+ from collections import OrderedDict
3823
3849
  from google.adk.agents import Agent
3824
3850
  from google.adk.runners import Runner
3825
3851
  from google.adk.sessions import InMemorySessionService
@@ -3940,21 +3966,57 @@ agent = Agent(
3940
3966
  )
3941
3967
 
3942
3968
 
3943
- # Session and Runner
3944
- async def setup_session_and_runner(user_id, session_id):
3945
- ensure_credentials_loaded()
3946
- session_service = InMemorySessionService()
3947
- session = await session_service.create_session(
3969
+ # Module-level session service and runner preserve history across invocations.
3970
+ # InMemorySessionService retains every (app_name, user_id, session_id) triple
3971
+ # forever, so we bound it to 128 active sessions with LRU eviction (the
3972
+ # least-recently-used session is deleted and its history reset) to keep a
3973
+ # long-running process from growing without limit. For durable history, swap in
3974
+ # a persistent session service (e.g. DatabaseSessionService).
3975
+ _SESSION_LIMIT = 128
3976
+ _session_service = InMemorySessionService()
3977
+ _session_keys = OrderedDict()
3978
+ _runner = None
3979
+
3980
+
3981
+ def get_or_create_runner():
3982
+ global _runner
3983
+ if _runner is None:
3984
+ ensure_credentials_loaded()
3985
+ _runner = Runner(
3986
+ agent=agent,
3987
+ app_name=APP_NAME,
3988
+ session_service=_session_service,
3989
+ )
3990
+ return _runner
3991
+
3992
+
3993
+ async def get_or_create_session(user_id, session_id):
3994
+ key = (user_id, session_id)
3995
+ if key in _session_keys:
3996
+ _session_keys.move_to_end(key)
3997
+ else:
3998
+ while len(_session_keys) >= _SESSION_LIMIT:
3999
+ (old_user_id, old_session_id), _ = _session_keys.popitem(last=False)
4000
+ await _session_service.delete_session(
4001
+ app_name=APP_NAME, user_id=old_user_id, session_id=old_session_id
4002
+ )
4003
+ _session_keys[key] = True
4004
+
4005
+ session = await _session_service.get_session(
3948
4006
  app_name=APP_NAME, user_id=user_id, session_id=session_id
3949
4007
  )
3950
- runner = Runner(agent=agent, app_name=APP_NAME, session_service=session_service)
3951
- return session, runner
4008
+ if session is None:
4009
+ session = await _session_service.create_session(
4010
+ app_name=APP_NAME, user_id=user_id, session_id=session_id
4011
+ )
4012
+ return session
3952
4013
 
3953
4014
 
3954
4015
  # Agent Interaction
3955
4016
  async def call_agent_async(query, user_id, session_id):
3956
4017
  content = types.Content(role="user", parts=[types.Part(text=query)])
3957
- session, runner = await setup_session_and_runner(user_id, session_id)
4018
+ runner = get_or_create_runner()
4019
+ session = await get_or_create_session(user_id, session_id)
3958
4020
  events = runner.run_async(
3959
4021
  user_id=user_id, session_id=session.id, new_message=content
3960
4022
  )
@@ -4241,9 +4303,11 @@ exports[`Assets Directory Snapshots > Python framework assets > python/python/ht
4241
4303
  "{{#if needsOs}}
4242
4304
  import os
4243
4305
  {{/if}}
4306
+ from collections import OrderedDict
4244
4307
  from typing import Any
4245
4308
 
4246
4309
  from langchain_core.messages import HumanMessage{{#if hasConfigBundle}}, SystemMessage{{/if}}
4310
+ from langgraph.checkpoint.memory import InMemorySaver
4247
4311
  from langgraph.prebuilt import create_react_agent
4248
4312
  from langchain.tools import tool
4249
4313
  {{#if hasConfigBundle}}
@@ -4294,6 +4358,26 @@ def add_numbers(a: int, b: int) -> int:
4294
4358
  # Define a collection of tools used by the model
4295
4359
  tools = [add_numbers]
4296
4360
 
4361
+ # Module-level checkpointer preserves conversation history across invocations.
4362
+ # InMemorySaver keeps every thread_id (= session_id) checkpoint in memory
4363
+ # forever, so we bound it to 128 active threads with LRU eviction (the
4364
+ # least-recently-used thread is deleted and its history reset) to keep a
4365
+ # long-running process from growing without limit. For durable history, swap in
4366
+ # a persistent checkpointer (e.g. SqliteSaver/AsyncSqliteSaver with a file path).
4367
+ _CHECKPOINT_LIMIT = 128
4368
+ _checkpointer = InMemorySaver()
4369
+ _thread_ids = OrderedDict()
4370
+
4371
+
4372
+ def touch_thread(thread_id):
4373
+ if thread_id in _thread_ids:
4374
+ _thread_ids.move_to_end(thread_id)
4375
+ return
4376
+ while len(_thread_ids) >= _CHECKPOINT_LIMIT:
4377
+ evicted, _ = _thread_ids.popitem(last=False)
4378
+ _checkpointer.delete_thread(evicted)
4379
+ _thread_ids[thread_id] = True
4380
+
4297
4381
  {{#if needsOs}}
4298
4382
  _MOUNT_PATHS = [
4299
4383
  {{#if sessionStorageMountPath}}"{{sessionStorageMountPath}}",{{/if}}
@@ -4389,29 +4473,46 @@ async def invoke(payload, context):
4389
4473
  if mcp_client:
4390
4474
  mcp_tools = await mcp_client.get_tools()
4391
4475
 
4392
- # Define the agent using create_react_agent
4476
+ # Define the agent using create_react_agent (checkpointer is shared across invocations)
4393
4477
  {{#if hasConfigBundle}}
4394
- graph = create_react_agent(get_or_create_model(), tools=mcp_tools + tools, prompt=DEFAULT_SYSTEM_PROMPT)
4478
+ graph = create_react_agent(
4479
+ get_or_create_model(),
4480
+ tools=mcp_tools + tools,
4481
+ prompt=DEFAULT_SYSTEM_PROMPT,
4482
+ checkpointer=_checkpointer,
4483
+ )
4395
4484
  callback = ConfigBundleCallback()
4396
4485
 
4397
4486
  # Process the user prompt
4398
4487
  prompt = payload.get("prompt", "What can you help me with?")
4488
+ session_id = getattr(context, "session_id", "default-session")
4489
+ touch_thread(session_id)
4399
4490
  log.info(f"Agent input: {prompt}")
4400
4491
 
4401
- # Run the agent with config bundle callback
4492
+ # Run the agent with config bundle callback (checkpointer auto-loads/saves history per session)
4402
4493
  result = await graph.ainvoke(
4403
4494
  {"messages": [HumanMessage(content=prompt)]},
4404
- config={"callbacks": [callback]},
4495
+ config={"callbacks": [callback], "configurable": {"thread_id": session_id}},
4405
4496
  )
4406
4497
  {{else}}
4407
- graph = create_react_agent(get_or_create_model(), tools=mcp_tools + tools, prompt=DEFAULT_SYSTEM_PROMPT)
4498
+ graph = create_react_agent(
4499
+ get_or_create_model(),
4500
+ tools=mcp_tools + tools,
4501
+ prompt=DEFAULT_SYSTEM_PROMPT,
4502
+ checkpointer=_checkpointer,
4503
+ )
4408
4504
 
4409
4505
  # Process the user prompt
4410
4506
  prompt = payload.get("prompt", "What can you help me with?")
4507
+ session_id = getattr(context, "session_id", "default-session")
4508
+ touch_thread(session_id)
4411
4509
  log.info(f"Agent input: {prompt}")
4412
4510
 
4413
- # Run the agent
4414
- result = await graph.ainvoke({"messages": [HumanMessage(content=prompt)]})
4511
+ # Run the agent (checkpointer auto-loads/saves history per session)
4512
+ result = await graph.ainvoke(
4513
+ {"messages": [HumanMessage(content=prompt)]},
4514
+ config={"configurable": {"thread_id": session_id}},
4515
+ )
4415
4516
  {{/if}}
4416
4517
 
4417
4518
  # Return result
@@ -4782,7 +4883,8 @@ import os
4782
4883
  {{#if hasGateway}}
4783
4884
  from contextlib import AsyncExitStack
4784
4885
  {{/if}}
4785
- from agents import Agent, Runner, function_tool
4886
+ from functools import lru_cache
4887
+ from agents import Agent, Runner, SQLiteSession, function_tool
4786
4888
  from bedrock_agentcore.runtime import BedrockAgentCoreApp
4787
4889
  from model.load import load_model
4788
4890
  {{#if hasGateway}}
@@ -4886,8 +4988,16 @@ You have access to the following mounted filesystems. Use file_read, file_write,
4886
4988
  {{/each}}{{/if}}
4887
4989
  """
4888
4990
 
4991
+ # Caches up to 128 active sessions; LRU eviction silently resets history for
4992
+ # the oldest session. For production use, replace with a durable session store
4993
+ # (e.g. SQLiteSession with a file path).
4994
+ @lru_cache(maxsize=128)
4995
+ def get_session(session_id):
4996
+ return SQLiteSession(session_id)
4997
+
4998
+
4889
4999
  # Define the agent execution
4890
- async def main(query):
5000
+ async def main(query, session):
4891
5001
  ensure_credentials_loaded()
4892
5002
  try:
4893
5003
  {{#if hasGateway}}
@@ -4906,7 +5016,7 @@ async def main(query):
4906
5016
  tools=tools,
4907
5017
  mcp_config={"include_server_in_tool_names": True},
4908
5018
  )
4909
- result = await Runner.run(agent, query)
5019
+ result = await Runner.run(agent, query, session=session)
4910
5020
  return result
4911
5021
  else:
4912
5022
  agent = Agent(
@@ -4916,7 +5026,7 @@ async def main(query):
4916
5026
  mcp_servers=[],
4917
5027
  tools=tools
4918
5028
  )
4919
- result = await Runner.run(agent, query)
5029
+ result = await Runner.run(agent, query, session=session)
4920
5030
  return result
4921
5031
  {{else}}
4922
5032
  if mcp_servers:
@@ -4929,7 +5039,7 @@ async def main(query):
4929
5039
  mcp_servers=active_servers,
4930
5040
  tools=tools
4931
5041
  )
4932
- result = await Runner.run(agent, query)
5042
+ result = await Runner.run(agent, query, session=session)
4933
5043
  return result
4934
5044
  else:
4935
5045
  agent = Agent(
@@ -4939,7 +5049,7 @@ async def main(query):
4939
5049
  mcp_servers=[],
4940
5050
  tools=tools
4941
5051
  )
4942
- result = await Runner.run(agent, query)
5052
+ result = await Runner.run(agent, query, session=session)
4943
5053
  return result
4944
5054
  {{/if}}
4945
5055
  except Exception as e:
@@ -4953,9 +5063,11 @@ async def invoke(payload, context):
4953
5063
 
4954
5064
  # Process the user prompt
4955
5065
  prompt = payload.get("prompt", "What can you help me with?")
5066
+ session_id = getattr(context, "session_id", "default-session")
5067
+ session = get_session(session_id)
4956
5068
 
4957
- # Run the agent
4958
- result = await main(prompt)
5069
+ # Run the agent (session automatically loads/saves conversation history)
5070
+ result = await main(prompt, session)
4959
5071
 
4960
5072
  # Return result
4961
5073
  return {"result": result.final_output}
@@ -5209,6 +5321,7 @@ Thumbs.db"
5209
5321
 
5210
5322
  exports[`Assets Directory Snapshots > Python framework assets > python/python/http/strands/base/main.py should match snapshot 1`] = `
5211
5323
  "from typing import Any
5324
+ from collections import OrderedDict
5212
5325
  {{#if inlineFunctionTools}}
5213
5326
  import json
5214
5327
 
@@ -5276,7 +5389,7 @@ from mcp_client.client import get_streamable_http_mcp_client
5276
5389
  from memory.session import get_memory_session_manager
5277
5390
  {{/if}}
5278
5391
  {{#unless hasFileOperations}}
5279
- {{#if (or needsOs (some gitSkills "credentialArn"))}}
5392
+ {{#if (or needsOs browserIdentifierEnvVar codeInterpreterIdentifierEnvVar (some gitSkills "credentialArn"))}}
5280
5393
  import os
5281
5394
  {{/if}}
5282
5395
  {{/unless}}
@@ -5362,10 +5475,20 @@ tools.append(add_numbers)
5362
5475
  {{/unless}}
5363
5476
  {{/if}}
5364
5477
  {{#if hasBrowser}}
5365
- tools.append(AgentCoreBrowser({{#if browserIdentifier}}identifier="{{browserIdentifier}}"{{/if}}).browser)
5478
+ {{#if browserIdentifierEnvVar}}
5479
+ _browser_id = os.getenv("{{browserIdentifierEnvVar}}")
5480
+ tools.append(AgentCoreBrowser(**({"identifier": _browser_id} if _browser_id else {})).browser)
5481
+ {{else}}
5482
+ tools.append(AgentCoreBrowser().browser)
5483
+ {{/if}}
5366
5484
  {{/if}}
5367
5485
  {{#if hasCodeInterpreter}}
5368
- tools.append(AgentCoreCodeInterpreter({{#if codeInterpreterIdentifier}}identifier="{{codeInterpreterIdentifier}}"{{/if}}).code_interpreter)
5486
+ {{#if codeInterpreterIdentifierEnvVar}}
5487
+ _code_interpreter_id = os.getenv("{{codeInterpreterIdentifierEnvVar}}")
5488
+ tools.append(AgentCoreCodeInterpreter(**({"identifier": _code_interpreter_id} if _code_interpreter_id else {})).code_interpreter)
5489
+ {{else}}
5490
+ tools.append(AgentCoreCodeInterpreter().code_interpreter)
5491
+ {{/if}}
5369
5492
  {{/if}}
5370
5493
  {{#if hasShell}}
5371
5494
  @tool
@@ -5634,26 +5757,21 @@ def agent_factory():
5634
5757
  get_or_create_agent = agent_factory()
5635
5758
  {{/unless}}
5636
5759
  {{else}}
5637
- {{#if hasConfigBundle}}
5638
- def create_agent({{#if hasSkillsFetcher}}skill_plugins=None{{/if}}):
5639
- return Agent(
5640
- model=load_model(),
5641
- system_prompt=DEFAULT_SYSTEM_PROMPT,
5642
- tools=tools,
5643
- conversation_manager=_make_conversation_manager(),
5644
- {{#if hasSkillsFetcher}}
5645
- plugins=skill_plugins or None,
5646
- {{/if}}
5647
- hooks=[ConfigBundleHook()],
5648
- )
5649
- {{else}}
5650
5760
  {{#unless hasPayment}}
5651
- _agent = None
5652
-
5653
- def get_or_create_agent({{#if hasSkillsFetcher}}skill_plugins=None{{/if}}):
5654
- global _agent
5655
- if _agent is None:
5656
- _agent = Agent(
5761
+ # Reuses one Agent per session_id so each session keeps its own in-process
5762
+ # conversation history (best-effort; resets on cold start). The cache is bounded
5763
+ # to 128 sessions with LRU eviction (least-recently-used is dropped and its
5764
+ # history reset) so a single process serving many sessions cannot leak history
5765
+ # between them or grow without limit. For durable history, attach a session manager.
5766
+ def agent_factory():
5767
+ cache = OrderedDict()
5768
+ def get_or_create_agent(session_id{{#if hasSkillsFetcher}}, skill_plugins=None{{/if}}):
5769
+ if session_id in cache:
5770
+ cache.move_to_end(session_id)
5771
+ return cache[session_id]
5772
+ if len(cache) >= 128:
5773
+ cache.popitem(last=False)
5774
+ cache[session_id] = Agent(
5657
5775
  model=load_model(),
5658
5776
  system_prompt=DEFAULT_SYSTEM_PROMPT,
5659
5777
  tools=tools,
@@ -5673,12 +5791,16 @@ def get_or_create_agent({{#if hasSkillsFetcher}}skill_plugins=None{{/if}}):
5673
5791
  {{#if timeoutSeconds}}timeout_seconds={{timeoutSeconds}},{{/if}}
5674
5792
  ),
5675
5793
  {{/if}}
5794
+ {{#if hasConfigBundle}}
5795
+ ConfigBundleHook(),
5796
+ {{/if}}
5676
5797
  ],
5677
5798
  )
5678
- return _agent
5799
+ return cache[session_id]
5800
+ return get_or_create_agent
5801
+ get_or_create_agent = agent_factory()
5679
5802
  {{/unless}}
5680
5803
  {{/if}}
5681
- {{/if}}
5682
5804
 
5683
5805
 
5684
5806
  def _extract_prompt(payload: dict):
@@ -5785,11 +5907,8 @@ async def invoke(payload, context):
5785
5907
  hooks=[ConfigBundleHook()],{{/if}}
5786
5908
  )
5787
5909
  {{else}}
5788
- {{#if hasConfigBundle}}
5789
- agent = create_agent({{#if hasSkillsFetcher}}_skill_plugins{{/if}})
5790
- {{else}}
5791
- agent = get_or_create_agent({{#if hasSkillsFetcher}}_skill_plugins{{/if}})
5792
- {{/if}}
5910
+ session_id = getattr(context, 'session_id', 'default-session')
5911
+ agent = get_or_create_agent(session_id{{#if hasSkillsFetcher}}, _skill_plugins{{/if}})
5793
5912
  {{/if}}
5794
5913
  {{/if}}
5795
5914
 
@@ -5908,7 +6027,10 @@ from bedrock_agentcore.identity import requires_access_token
5908
6027
  @requires_access_token(
5909
6028
  provider_name="{{credentialProviderName}}",
5910
6029
  scopes=[{{#if scopes}}"{{scopes}}"{{/if}}],
5911
- auth_flow="M2M",
6030
+ auth_flow="{{#if authFlow}}{{authFlow}}{{else}}M2M{{/if}}",
6031
+ {{#if customParameters}}
6032
+ custom_parameters={{safeJson customParameters}},
6033
+ {{/if}}
5912
6034
  )
5913
6035
  def _get_bearer_token_{{snakeCase name}}(*, access_token: str):
5914
6036
  """Obtain OAuth access token via AgentCore Identity for {{name}}."""
@@ -6011,6 +6133,67 @@ exports[`Assets Directory Snapshots > Python framework assets > python/python/ht
6011
6133
 
6012
6134
  exports[`Assets Directory Snapshots > Python framework assets > python/python/http/strands/base/model/load.py should match snapshot 1`] = `
6013
6135
  "{{#if (eq modelProvider "Bedrock")}}
6136
+ {{#if bedrockMantle}}
6137
+ import os
6138
+
6139
+ from aws_bedrock_token_generator import provide_token
6140
+ {{#if (eq mantleApiFormat "chat_completions")}}
6141
+ from strands.models.openai import OpenAIModel
6142
+ {{else}}
6143
+ {{#if mantleProprietary}}
6144
+ from strands.models.openai_responses import OpenAIResponsesModel
6145
+ {{else}}
6146
+ from model.mantle_compat import MantleCompatResponsesModel
6147
+ {{/if}}
6148
+ {{/if}}
6149
+
6150
+ MODEL_ID = "{{modelId}}"
6151
+
6152
+
6153
+ def load_model():
6154
+ """
6155
+ Get a Bedrock Mantle model client. These OpenAI-compatible models (e.g. openai.gpt-5.5,
6156
+ openai.gpt-oss-120b) are served via the Bedrock Mantle endpoint, NOT the Converse API — so they
6157
+ are invoked through an OpenAI-style client authenticated with a short-lived Bedrock bearer token.
6158
+ Region is read from AWS_REGION (set by the AgentCore runtime).
6159
+ """
6160
+ region = os.environ.get("AWS_REGION", os.environ.get("AWS_DEFAULT_REGION", "us-east-1"))
6161
+ token = provide_token(region=region)
6162
+ {{#if mantleProprietary}}
6163
+ # Proprietary OpenAI models only work on the /openai/v1 Mantle path.
6164
+ base_url = f"https://bedrock-mantle.{region}.api.aws/openai/v1"
6165
+ {{else}}
6166
+ # Open-source OpenAI models (gpt-oss-*) only work on the /v1 Mantle path.
6167
+ base_url = f"https://bedrock-mantle.{region}.api.aws/v1"
6168
+ {{/if}}
6169
+ client_args = {"api_key": token, "base_url": base_url}
6170
+
6171
+ params = {}
6172
+ {{#if modelMaxTokens}}
6173
+ {{#if (eq mantleApiFormat "chat_completions")}}
6174
+ params["max_completion_tokens"] = {{modelMaxTokens}}
6175
+ {{else}}
6176
+ params["max_output_tokens"] = {{modelMaxTokens}}
6177
+ {{/if}}
6178
+ {{/if}}
6179
+ {{#if modelTemperature}}
6180
+ params["temperature"] = {{modelTemperature}}
6181
+ {{/if}}
6182
+ {{#if modelTopP}}
6183
+ params["top_p"] = {{modelTopP}}
6184
+ {{/if}}
6185
+ {{#if (eq mantleApiFormat "chat_completions")}}
6186
+ return OpenAIModel(client_args=client_args, model_id=MODEL_ID, params=params)
6187
+ {{else}}
6188
+ # Responses API: Mantle does not persist responses, so disable server-side storage.
6189
+ params["store"] = False
6190
+ {{#if mantleProprietary}}
6191
+ return OpenAIResponsesModel(client_args=client_args, model_id=MODEL_ID, params=params)
6192
+ {{else}}
6193
+ return MantleCompatResponsesModel(client_args=client_args, model_id=MODEL_ID, params=params)
6194
+ {{/if}}
6195
+ {{/if}}
6196
+ {{else}}
6014
6197
  from strands.models.bedrock import BedrockModel
6015
6198
 
6016
6199
 
@@ -6018,6 +6201,7 @@ def load_model() -> BedrockModel:
6018
6201
  """Get Bedrock model client using IAM credentials."""
6019
6202
  return BedrockModel(model_id="{{#if modelId}}{{modelId}}{{else}}global.anthropic.claude-sonnet-4-5-20250929-v1:0{{/if}}")
6020
6203
  {{/if}}
6204
+ {{/if}}
6021
6205
  {{#if (eq modelProvider "Anthropic")}}
6022
6206
  import os
6023
6207
 
@@ -6133,6 +6317,85 @@ def load_model() -> GeminiModel:
6133
6317
  model_id="{{#if modelId}}{{modelId}}{{else}}gemini-2.5-flash{{/if}}",
6134
6318
  )
6135
6319
  {{/if}}
6320
+ {{#if (eq modelProvider "LiteLLM")}}
6321
+ import os
6322
+ {{#if litellmAdditionalParams}}
6323
+ import json
6324
+ {{/if}}
6325
+
6326
+ from strands.models.litellm import LiteLLMModel
6327
+ {{#if identityProviders.[0].name}}
6328
+ from bedrock_agentcore.identity.auth import requires_api_key
6329
+
6330
+ IDENTITY_PROVIDER_NAME = "{{identityProviders.[0].name}}"
6331
+ IDENTITY_ENV_VAR = "{{identityProviders.[0].envVarName}}"
6332
+
6333
+
6334
+ @requires_api_key(provider_name=IDENTITY_PROVIDER_NAME)
6335
+ def _agentcore_identity_api_key_provider(api_key: str) -> str:
6336
+ """Fetch API key from AgentCore Identity."""
6337
+ return api_key
6338
+
6339
+
6340
+ def _get_api_key() -> str:
6341
+ """
6342
+ Uses AgentCore Identity for API key management in deployed environments.
6343
+ For local development, run via 'agentcore dev' which loads agentcore/.env.
6344
+ """
6345
+ if os.getenv("LOCAL_DEV") == "1":
6346
+ api_key = os.getenv(IDENTITY_ENV_VAR)
6347
+ if not api_key:
6348
+ raise RuntimeError(
6349
+ f"{IDENTITY_ENV_VAR} not found. Add {IDENTITY_ENV_VAR}=your-key to .env.local"
6350
+ )
6351
+ return api_key
6352
+ return _agentcore_identity_api_key_provider()
6353
+ {{/if}}
6354
+
6355
+
6356
+
6357
+
6358
+ def load_model() -> LiteLLMModel:
6359
+ """Get a LiteLLM model client (proxies to the provider encoded in model_id)."""
6360
+ client_args = {}
6361
+ {{#if identityProviders.[0].name}}
6362
+ client_args["api_key"] = _get_api_key()
6363
+ {{/if}}
6364
+ {{#if litellmApiBase}}
6365
+ client_args["api_base"] = {{safeJson litellmApiBase}}
6366
+ {{/if}}
6367
+ params = {{#if litellmAdditionalParams}}json.loads({{pyJsonStr litellmAdditionalParams}}){{else}}{}{{/if}}
6368
+ return LiteLLMModel(
6369
+ client_args=client_args,
6370
+ model_id="{{#if modelId}}{{modelId}}{{else}}bedrock/us.anthropic.claude-sonnet-4-5-20250514-v1:0{{/if}}",
6371
+ params=params,
6372
+ )
6373
+ {{/if}}
6374
+ "
6375
+ `;
6376
+
6377
+ exports[`Assets Directory Snapshots > Python framework assets > python/python/http/strands/base/model/mantle_compat.py should match snapshot 1`] = `
6378
+ "from strands.models.openai_responses import OpenAIResponsesModel
6379
+
6380
+
6381
+ class MantleCompatResponsesModel(OpenAIResponsesModel):
6382
+ """Workaround for Bedrock Mantle rejecting output_text in EasyInputMessage content arrays.
6383
+
6384
+ Mantle's Pydantic validation only accepts content as a plain string for assistant messages, while
6385
+ real OpenAI accepts both formats. Flatten assistant content arrays to strings so multi-turn works.
6386
+ Used for open-source OpenAI models (gpt-oss-*) on the /v1 Mantle path; proprietary models use the
6387
+ plain OpenAIResponsesModel on /openai/v1.
6388
+ """
6389
+
6390
+ @classmethod
6391
+ def _format_request_messages(cls, messages):
6392
+ formatted = super()._format_request_messages(messages)
6393
+ for msg in formatted:
6394
+ if msg.get("role") == "assistant" and isinstance(msg.get("content"), list):
6395
+ msg["content"] = "".join(
6396
+ part.get("text", "") for part in msg["content"] if part.get("type") == "output_text"
6397
+ )
6398
+ return formatted
6136
6399
  "
6137
6400
  `;
6138
6401
 
@@ -6155,6 +6418,9 @@ dependencies = [
6155
6418
  {{#if (eq modelProvider "Gemini")}}"google-genai >= 1.0.0",
6156
6419
  {{/if}}"mcp >= 1.19.0",
6157
6420
  {{#if (eq modelProvider "OpenAI")}}"openai >= 1.0.0",
6421
+ {{/if}}{{#if (eq modelProvider "LiteLLM")}}"litellm >= 1.0.0",
6422
+ {{/if}}{{#if bedrockMantle}}"openai >= 1.0.0",
6423
+ "aws-bedrock-token-generator >= 1.0.0",
6158
6424
  {{/if}}"strands-agents >= 1.15.0",
6159
6425
  {{#if (or hasBrowser hasCodeInterpreter)}}"strands-agents-tools >= 0.1.0",
6160
6426
  {{/if}}{{#if hasBrowser}}"nest-asyncio >= 1.5.0",
@@ -6545,7 +6811,7 @@ def get_memory_session_manager(session_id: Optional[str], actor_id: str) -> Opti
6545
6811
  f"/episodes/{actor_id}/{session_id}": RetrievalConfig(top_k=5, relevance_score=0.5),
6546
6812
  {{/if}}
6547
6813
  {{#if (includes memoryProviders.[0].strategies "SUMMARIZATION")}}
6548
- f"/summaries/{actor_id}/{session_id}": RetrievalConfig(top_k=3, relevance_score=0.5),
6814
+ f"/summaries/{actor_id}": RetrievalConfig(top_k=3, relevance_score=0.5),
6549
6815
  {{/if}}
6550
6816
  }
6551
6817
  {{/if}}
@@ -7379,6 +7645,9 @@ import { Agent, McpClient, tool, type ToolList } from '@strands-agents/sdk';
7379
7645
  import { z } from 'zod';
7380
7646
  import { loadModel } from './model/load.js';
7381
7647
  import { getStreamableHttpMcpClient } from './mcp_client/client.js';
7648
+ {{#if hasMemory}}
7649
+ import { getActorId, getOrCreateMemoryManager } from './memory/memory.js';
7650
+ {{/if}}
7382
7651
 
7383
7652
  // Define a collection of MCP clients (filter out anything that failed to initialize)
7384
7653
  const mcpClients: McpClient[] = [getStreamableHttpMcpClient()].filter(
@@ -7407,34 +7676,109 @@ const SYSTEM_PROMPT = \`
7407
7676
  You are a helpful assistant. Use tools when appropriate.
7408
7677
  \`;
7409
7678
 
7410
- let cachedAgent: Agent | null = null;
7411
-
7412
- async function getOrCreateAgent(): Promise<Agent> {
7413
- if (!cachedAgent) {
7414
- const model = await loadModel();
7415
- cachedAgent = new Agent({
7416
- model,
7417
- systemPrompt: SYSTEM_PROMPT,
7418
- tools,
7419
- });
7679
+ {{#if hasMemory}}
7680
+ const agentCache = new Map<string, Agent>();
7681
+
7682
+ async function getOrCreateAgent(sessionId: string, actorId: string): Promise<Agent> {
7683
+ const key = \`\${actorId}:\${sessionId}\`;
7684
+ let agent = agentCache.get(key);
7685
+ if (agent) return agent;
7686
+
7687
+ const model = await loadModel();
7688
+ agent = new Agent({
7689
+ model,
7690
+ systemPrompt: SYSTEM_PROMPT,
7691
+ tools,
7692
+ memoryManager: getOrCreateMemoryManager(sessionId, actorId) ?? undefined,
7693
+ });
7694
+ agentCache.set(key, agent);
7695
+ return agent;
7696
+ }
7697
+ {{else}}
7698
+ const AGENT_CACHE_LIMIT = 128;
7699
+
7700
+ // Reuses one Agent per sessionId so each session keeps its own in-process
7701
+ // conversation history (best-effort; resets on cold start). A Map preserves
7702
+ // insertion order, so it doubles as an LRU bounded to 128 sessions — a local
7703
+ // dev process serving many sessions cannot leak history between them or grow
7704
+ // without bound. On AgentCore Runtime each microVM serves a single session, so
7705
+ // this holds one entry. For durable history, attach memory.
7706
+ const agentCache = new Map<string, Agent>();
7707
+
7708
+ async function getOrCreateAgent(sessionId: string): Promise<Agent> {
7709
+ const existing = agentCache.get(sessionId);
7710
+ if (existing) {
7711
+ agentCache.delete(sessionId);
7712
+ agentCache.set(sessionId, existing);
7713
+ return existing;
7714
+ }
7715
+ if (agentCache.size >= AGENT_CACHE_LIMIT) {
7716
+ const oldest = agentCache.keys().next().value;
7717
+ if (oldest !== undefined) agentCache.delete(oldest);
7420
7718
  }
7421
- return cachedAgent;
7719
+ const model = await loadModel();
7720
+ const agent = new Agent({
7721
+ model,
7722
+ systemPrompt: SYSTEM_PROMPT,
7723
+ tools,
7724
+ });
7725
+ agentCache.set(sessionId, agent);
7726
+ return agent;
7422
7727
  }
7728
+ {{/if}}
7423
7729
 
7424
7730
  const app = new BedrockAgentCoreApp({
7425
7731
  invocationHandler: {
7426
7732
  async *process(payload: any, context: any) {
7427
- const agent = await getOrCreateAgent();
7428
-
7429
- for await (const event of agent.stream(payload.prompt ?? '')) {
7430
- if (
7431
- event.type === 'modelStreamUpdateEvent' &&
7432
- event.event?.type === 'modelContentBlockDeltaEvent' &&
7433
- event.event.delta?.type === 'textDelta'
7434
- ) {
7435
- yield { data: event.event.delta.text };
7733
+ {{#if hasMemory}}
7734
+ const sessionId = context?.sessionId ?? 'default-session';
7735
+ const actorId = getActorId(payload, context);
7736
+ const agent = await getOrCreateAgent(sessionId, actorId);
7737
+ {{else}}
7738
+ const sessionId = context?.sessionId ?? 'default-session';
7739
+ const agent = await getOrCreateAgent(sessionId);
7740
+ {{/if}}
7741
+
7742
+ {{#if hasMemory}}
7743
+ try {
7744
+ for await (const event of agent.stream(payload.prompt ?? '')) {
7745
+ if (
7746
+ event.type === 'modelStreamUpdateEvent' &&
7747
+ event.event?.type === 'modelContentBlockDeltaEvent' &&
7748
+ event.event.delta?.type === 'textDelta'
7749
+ ) {
7750
+ yield { data: event.event.delta.text };
7751
+ }
7752
+ }
7753
+ } finally {
7754
+ // Drain in-flight createEvent calls before the runtime can reclaim
7755
+ // the session microVM. flush() is the durability mechanism — without
7756
+ // it, an idle reclamation can lose the tail of the conversation.
7757
+ await agent.memoryManager?.flush();
7758
+ }
7759
+ {{else}}
7760
+ // Snapshot history before streaming so a failed turn can be rolled back.
7761
+ // Agent.stream() appends the user message before invoking the model; on a
7762
+ // mid-stream error that user turn would otherwise linger in the cached
7763
+ // agent, and the next turn for this session would send consecutive user
7764
+ // messages (rejected by providers that require strict role alternation,
7765
+ // e.g. Anthropic). Restoring on error keeps the session reusable.
7766
+ const snapshot = agent.takeSnapshot({ include: ['messages'] });
7767
+ try {
7768
+ for await (const event of agent.stream(payload.prompt ?? '')) {
7769
+ if (
7770
+ event.type === 'modelStreamUpdateEvent' &&
7771
+ event.event?.type === 'modelContentBlockDeltaEvent' &&
7772
+ event.event.delta?.type === 'textDelta'
7773
+ ) {
7774
+ yield { data: event.event.delta.text };
7775
+ }
7436
7776
  }
7777
+ } catch (error) {
7778
+ agent.loadSnapshot(snapshot);
7779
+ throw error;
7437
7780
  }
7781
+ {{/if}}
7438
7782
  },
7439
7783
  },
7440
7784
  });
@@ -7587,8 +7931,9 @@ exports[`Assets Directory Snapshots > TypeScript assets > typescript/typescript/
7587
7931
  "@google/genai": "^1.40.0",
7588
7932
  {{/if}}
7589
7933
  "@modelcontextprotocol/sdk": "^1.25.2",
7590
- "@strands-agents/sdk": "1.0.0-rc.4",
7591
- "bedrock-agentcore": "^0.2.4",
7934
+ "@opentelemetry/api": "^1.9.0",
7935
+ "@strands-agents/sdk": "^1.5.0",
7936
+ "bedrock-agentcore": "^0.3.0",
7592
7937
  "tsx": "^4.19.0",
7593
7938
  "zod": "^4.4.3"
7594
7939
  },
@@ -7628,6 +7973,62 @@ exports[`Assets Directory Snapshots > TypeScript assets > typescript/typescript/
7628
7973
  "
7629
7974
  `;
7630
7975
 
7976
+ exports[`Assets Directory Snapshots > TypeScript assets > typescript/typescript/http/strands/capabilities/memory/memory.ts should match snapshot 1`] = `
7977
+ "import { randomUUID } from 'node:crypto';
7978
+ import { MemoryManager } from '@strands-agents/sdk';
7979
+ import { createAgentCoreMemoryStores } from 'bedrock-agentcore/experimental/memory/strands';
7980
+
7981
+ const MEMORY_ID = process.env.{{memoryProviders.[0].envVarName}};
7982
+
7983
+ const CUSTOM_ACTOR_ID_HEADER = 'x-amzn-bedrock-agentcore-runtime-custom-actor-id';
7984
+
7985
+ export function getActorId(payload: any, context: any): string {
7986
+ const raw =
7987
+ context?.headers?.[CUSTOM_ACTOR_ID_HEADER] ||
7988
+ payload?.userId ||
7989
+ context?.sessionId;
7990
+ return typeof raw === 'string' && raw.trim().length > 0 ? raw.trim() : randomUUID();
7991
+ }
7992
+
7993
+ const memoryManagerCache = new Map<string, MemoryManager>();
7994
+
7995
+ export function getOrCreateMemoryManager(sessionId: string, actorId: string): MemoryManager | null {
7996
+ if (!MEMORY_ID) return null;
7997
+
7998
+ const key = \`\${actorId}:\${sessionId}\`;
7999
+ let manager = memoryManagerCache.get(key);
8000
+ if (manager) return manager;
8001
+
8002
+ const stores = createAgentCoreMemoryStores({
8003
+ memoryId: MEMORY_ID,
8004
+ actorId,
8005
+ sessionId,
8006
+ namespaces: [
8007
+ {{#if (includes memoryProviders.[0].strategies "SEMANTIC")}}
8008
+ { namespace: '/users/{actorId}/facts' },
8009
+ {{/if}}
8010
+ {{#if (includes memoryProviders.[0].strategies "USER_PREFERENCE")}}
8011
+ { namespace: '/users/{actorId}/preferences' },
8012
+ {{/if}}
8013
+ {{#if (includes memoryProviders.[0].strategies "EPISODIC")}}
8014
+ { namespace: '/episodes/{actorId}/{sessionId}' },
8015
+ {{/if}}
8016
+ {{#if (includes memoryProviders.[0].strategies "SUMMARIZATION")}}
8017
+ { namespace: '/summaries/{actorId}/{sessionId}' },
8018
+ {{/if}}
8019
+ ],
8020
+ // readMode defaults to 'per-namespace' (one retrieve call per namespace).
8021
+ // Switch to 'subtree' to consolidate to a single hierarchical recall call.
8022
+ extraction: true,
8023
+ });
8024
+
8025
+ manager = new MemoryManager({ stores });
8026
+ memoryManagerCache.set(key, manager);
8027
+ return manager;
8028
+ }
8029
+ "
8030
+ `;
8031
+
7631
8032
  exports[`Assets Directory Snapshots > TypeScript assets > typescript/typescript/http/vercelai/base/README.md should match snapshot 1`] = `
7632
8033
  "This is a project generated by the AgentCore CLI!
7633
8034
 
@@ -7697,24 +8098,64 @@ Thumbs.db
7697
8098
 
7698
8099
  exports[`Assets Directory Snapshots > TypeScript assets > typescript/typescript/http/vercelai/base/main.ts should match snapshot 1`] = `
7699
8100
  "import { BedrockAgentCoreApp } from 'bedrock-agentcore/runtime';
7700
- import { streamText } from 'ai';
8101
+ import { streamText, type ModelMessage } from 'ai';
7701
8102
  import { loadModel } from './model/load.js';
7702
8103
 
7703
8104
  const SYSTEM_PROMPT = \`You are a helpful assistant.\`;
7704
8105
 
8106
+ const HISTORY_LIMIT = 128;
8107
+
8108
+ // Keeps one message history per sessionId so each session remembers its own
8109
+ // turns (best-effort; resets on cold start). A Map preserves insertion order,
8110
+ // so it doubles as an LRU bounded to 128 sessions — a local dev process serving
8111
+ // many sessions cannot leak history between them or grow without bound. On
8112
+ // AgentCore Runtime each microVM serves a single session, so this holds one
8113
+ // entry. For durable history, persist messages to an external store.
8114
+ const histories = new Map<string, ModelMessage[]>();
8115
+
8116
+ function getHistory(sessionId: string): ModelMessage[] {
8117
+ const existing = histories.get(sessionId);
8118
+ if (existing) {
8119
+ histories.delete(sessionId);
8120
+ histories.set(sessionId, existing);
8121
+ return existing;
8122
+ }
8123
+ if (histories.size >= HISTORY_LIMIT) {
8124
+ const oldest = histories.keys().next().value;
8125
+ if (oldest !== undefined) histories.delete(oldest);
8126
+ }
8127
+ const fresh: ModelMessage[] = [];
8128
+ histories.set(sessionId, fresh);
8129
+ return fresh;
8130
+ }
8131
+
7705
8132
  const app = new BedrockAgentCoreApp({
7706
8133
  invocationHandler: {
7707
8134
  async *process(payload: any, context: any) {
8135
+ const sessionId = context?.sessionId ?? 'default-session';
8136
+ const history = getHistory(sessionId);
8137
+ const userMessage: ModelMessage = { role: 'user', content: payload.prompt ?? '' };
8138
+
7708
8139
  const model = await loadModel();
7709
8140
  const result = streamText({
7710
8141
  model,
7711
8142
  system: SYSTEM_PROMPT,
7712
- prompt: payload.prompt ?? '',
8143
+ messages: [...history, userMessage],
7713
8144
  });
7714
8145
 
8146
+ let assistant = '';
7715
8147
  for await (const chunk of result.textStream) {
8148
+ assistant += chunk;
7716
8149
  yield { data: chunk };
7717
8150
  }
8151
+
8152
+ // Commit the exchange to history only after a non-empty reply. On a failed
8153
+ // or empty stream the turn is dropped instead of leaving a dangling user
8154
+ // (or empty assistant) message — consecutive same-role or empty-content
8155
+ // messages would otherwise be rejected on the next turn for this session.
8156
+ if (assistant.length > 0) {
8157
+ history.push(userMessage, { role: 'assistant', content: assistant });
8158
+ }
7718
8159
  },
7719
8160
  },
7720
8161
  });