@aws/agentcore 1.0.0-preview.18 → 1.0.0-preview.19

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (63) hide show
  1. package/README.md +17 -17
  2. package/dist/assets/__tests__/__snapshots__/assets.snapshot.test.ts.snap +260 -83
  3. package/dist/assets/__tests__/googleadk-session-eviction.test.ts +87 -0
  4. package/dist/assets/__tests__/summarization-namespace.test.ts +43 -0
  5. package/dist/assets/python/a2a/strands/capabilities/memory/session.py +1 -1
  6. package/dist/assets/python/agui/strands/capabilities/memory/session.py +1 -1
  7. package/dist/assets/python/http/autogen/base/main.py +33 -10
  8. package/dist/assets/python/http/googleadk/base/main.py +45 -8
  9. package/dist/assets/python/http/langchain_langgraph/base/main.py +46 -7
  10. package/dist/assets/python/http/openaiagents/base/main.py +19 -8
  11. package/dist/assets/python/http/strands/base/main.py +23 -26
  12. package/dist/assets/python/http/strands/capabilities/memory/session.py +1 -1
  13. package/dist/assets/typescript/http/strands/base/main.ts +48 -18
  14. package/dist/assets/typescript/http/vercelai/base/main.ts +42 -2
  15. package/dist/cli/index.mjs +516 -516
  16. package/dist/lib/errors/types.d.ts +25 -0
  17. package/dist/lib/errors/types.d.ts.map +1 -1
  18. package/dist/lib/errors/types.js +40 -1
  19. package/dist/lib/errors/types.js.map +1 -1
  20. package/dist/lib/secrets/cipher.d.ts +12 -0
  21. package/dist/lib/secrets/cipher.d.ts.map +1 -0
  22. package/dist/lib/secrets/cipher.js +54 -0
  23. package/dist/lib/secrets/cipher.js.map +1 -0
  24. package/dist/lib/secrets/index.d.ts +4 -0
  25. package/dist/lib/secrets/index.d.ts.map +1 -0
  26. package/dist/lib/secrets/index.js +15 -0
  27. package/dist/lib/secrets/index.js.map +1 -0
  28. package/dist/lib/secrets/key-provider.d.ts +16 -0
  29. package/dist/lib/secrets/key-provider.d.ts.map +1 -0
  30. package/dist/lib/secrets/key-provider.js +191 -0
  31. package/dist/lib/secrets/key-provider.js.map +1 -0
  32. package/dist/lib/secrets/sensitive-keys.d.ts +21 -0
  33. package/dist/lib/secrets/sensitive-keys.d.ts.map +1 -0
  34. package/dist/lib/secrets/sensitive-keys.js +67 -0
  35. package/dist/lib/secrets/sensitive-keys.js.map +1 -0
  36. package/dist/lib/utils/env.d.ts +4 -2
  37. package/dist/lib/utils/env.d.ts.map +1 -1
  38. package/dist/lib/utils/env.js +57 -27
  39. package/dist/lib/utils/env.js.map +1 -1
  40. package/dist/schema/constants.d.ts +26 -0
  41. package/dist/schema/constants.d.ts.map +1 -1
  42. package/dist/schema/constants.js +34 -1
  43. package/dist/schema/constants.js.map +1 -1
  44. package/dist/schema/schemas/agent-env.d.ts +4 -2
  45. package/dist/schema/schemas/agent-env.d.ts.map +1 -1
  46. package/dist/schema/schemas/agent-env.js +23 -5
  47. package/dist/schema/schemas/agent-env.js.map +1 -1
  48. package/dist/schema/schemas/agentcore-project.d.ts +3 -1
  49. package/dist/schema/schemas/agentcore-project.d.ts.map +1 -1
  50. package/dist/schema/schemas/auth.d.ts +2 -3
  51. package/dist/schema/schemas/auth.d.ts.map +1 -1
  52. package/dist/schema/schemas/auth.js +8 -7
  53. package/dist/schema/schemas/auth.js.map +1 -1
  54. package/dist/schema/schemas/primitives/config-bundle.d.ts +1 -0
  55. package/dist/schema/schemas/primitives/config-bundle.d.ts.map +1 -1
  56. package/dist/schema/schemas/primitives/config-bundle.js +3 -0
  57. package/dist/schema/schemas/primitives/config-bundle.js.map +1 -1
  58. package/dist/schema/schemas/primitives/harness.d.ts +1 -0
  59. package/dist/schema/schemas/primitives/harness.d.ts.map +1 -1
  60. package/dist/schema/schemas/primitives/harness.js +16 -0
  61. package/dist/schema/schemas/primitives/harness.js.map +1 -1
  62. package/npm-shrinkwrap.json +223 -0
  63. package/package.json +4 -1
package/README.md CHANGED
@@ -93,11 +93,26 @@ agentcore invoke
93
93
 
94
94
  | Command | Description |
95
95
  | -------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
96
- | `add` | Add agents, memory, credentials, gateways and gateway-targets, evaluators, online evals, online insights, knowledge bases, config bundles, datasets, harnesses, policy engines and policies, payment managers and payment connectors, runtime endpoints |
96
+ | `add` | Add harnesses, agents, memory, credentials, gateways and gateway-targets, evaluators, online evals, online insights, knowledge bases, config bundles, datasets, policy engines and policies, payment managers and payment connectors, runtime endpoints |
97
97
  | `remove` | Remove any of the above resources from the project |
98
98
 
99
99
  > **Note**: Run `agentcore deploy` after `add` or `remove` to update resources in AWS.
100
100
 
101
+ ### Harness
102
+
103
+ A harness bundles a runtime, model, tools, skills, memory, and observability into one declarative config. Use it when
104
+ you want infra without writing agent code.
105
+
106
+ | Command | Description |
107
+ | ---------------- | --------------------------------------------------------------------------- |
108
+ | `add harness` | Add a harness resource (runtime + model + memory) |
109
+ | `add tool` | Add a tool to a harness (`--harness <name> --type <type> --name <name>`) |
110
+ | `add skill` | Add a skill to a harness (`--harness <name>` + `--path` / `--s3` / `--git`) |
111
+ | `export harness` | Export a harness config to a deployable Strands Python agent under `app/` |
112
+
113
+ > After `export harness`, **read `app/<agentName>/EXPORT_NOTES.md`** before running `deploy` — it lists any manual
114
+ > follow-up the exporter could not automate.
115
+
101
116
  ### Observability
102
117
 
103
118
  | Command | Description |
@@ -167,21 +182,6 @@ clusters of bad outcomes.
167
182
  | `resume online-insights` | Resume a paused online insights config |
168
183
  | `archive insights` | Delete an insights job on the service + clear local history |
169
184
 
170
- ### Harness
171
-
172
- A harness bundles a runtime, model, tools, skills, memory, and observability into one declarative config. Use it when
173
- you want infra without writing agent code.
174
-
175
- | Command | Description |
176
- | ---------------- | --------------------------------------------------------------------------- |
177
- | `add harness` | Add a harness resource (runtime + model + memory) |
178
- | `add tool` | Add a tool to a harness (`--harness <name> --type <type> --name <name>`) |
179
- | `add skill` | Add a skill to a harness (`--harness <name>` + `--path` / `--s3` / `--git`) |
180
- | `export harness` | Export a harness config to a deployable Strands Python agent under `app/` |
181
-
182
- > After `export harness`, **read `app/<agentName>/EXPORT_NOTES.md`** before running `deploy` — it lists any manual
183
- > follow-up the exporter could not automate.
184
-
185
185
  ### Policies & Guardrails
186
186
 
187
187
  Policy engines apply Cedar-based pre/post-call policies to agent invocations — including Bedrock content filters
@@ -271,6 +271,7 @@ Projects use JSON schema files in the `agentcore/` directory:
271
271
 
272
272
  ## Capabilities
273
273
 
274
+ - **Harness** - Declarative agent: bundle runtime + tools + skills + memory + observability without writing agent code
274
275
  - **Runtime** - Managed execution environment for deployed agents
275
276
  - **Memory** - Semantic, summarization, user-preference, and episodic strategies
276
277
  - **Credentials** - Secure API key + OAuth credential management via Secrets Manager
@@ -281,7 +282,6 @@ Projects use JSON schema files in the `agentcore/` directory:
281
282
  - **A/B Tests** - Traffic-split between config-bundle or target-based variants and promote the winner
282
283
  - **Insights** _[preview]_ - Failure-pattern analysis and clustering across agent sessions
283
284
  - **Knowledge Bases** - Managed Bedrock Knowledge Bases auto-wired to gateways
284
- - **Harness** - Declarative agent: bundle runtime + tools + skills + memory + observability without writing agent code
285
285
  - **Policies & Guardrails** - Cedar pre/post-call policies including Bedrock content filters, prompt-attack detection,
286
286
  and sensitive-information redaction
287
287
  - **Payments** - x402-protocol microtransactions for pay-per-call tools and APIs
@@ -2257,7 +2257,7 @@ def get_memory_session_manager(session_id: Optional[str], actor_id: str) -> Opti
2257
2257
  f"/episodes/{actor_id}/{session_id}": RetrievalConfig(top_k=5, relevance_score=0.5),
2258
2258
  {{/if}}
2259
2259
  {{#if (includes memoryProviders.[0].strategies "SUMMARIZATION")}}
2260
- f"/summaries/{actor_id}/{session_id}": RetrievalConfig(top_k=3, relevance_score=0.5),
2260
+ f"/summaries/{actor_id}": RetrievalConfig(top_k=3, relevance_score=0.5),
2261
2261
  {{/if}}
2262
2262
  }
2263
2263
  {{/if}}
@@ -3279,7 +3279,7 @@ def get_memory_session_manager(session_id: Optional[str], actor_id: str) -> Opti
3279
3279
  f"/episodes/{actor_id}/{session_id}": RetrievalConfig(top_k=5, relevance_score=0.5),
3280
3280
  {{/if}}
3281
3281
  {{#if (includes memoryProviders.[0].strategies "SUMMARIZATION")}}
3282
- f"/summaries/{actor_id}/{session_id}": RetrievalConfig(top_k=3, relevance_score=0.5),
3282
+ f"/summaries/{actor_id}": RetrievalConfig(top_k=3, relevance_score=0.5),
3283
3283
  {{/if}}
3284
3284
  }
3285
3285
  {{/if}}
@@ -3390,6 +3390,7 @@ exports[`Assets Directory Snapshots > Python framework assets > python/python/ht
3390
3390
  "{{#if needsOs}}
3391
3391
  import os
3392
3392
  {{/if}}
3393
+ from collections import OrderedDict
3393
3394
  from autogen_agentchat.agents import AssistantAgent
3394
3395
  from autogen_core.tools import FunctionTool
3395
3396
  from bedrock_agentcore.runtime import BedrockAgentCoreApp
@@ -3480,23 +3481,45 @@ You have access to the following mounted filesystems. Use file_read, file_write,
3480
3481
  {{/each}}{{/if}}
3481
3482
  """
3482
3483
 
3483
- @app.entrypoint
3484
- async def invoke(payload, context):
3485
- log.info("Invoking Agent.....")
3484
+ # Reuses one AssistantAgent per session_id so each session keeps its own
3485
+ # in-process conversation history (best-effort; resets on cold start). Caches up
3486
+ # to 128 active sessions with LRU eviction (least-recently-used is dropped and
3487
+ # its history reset).
3488
+ _agents = OrderedDict()
3486
3489
 
3490
+
3491
+ async def get_or_create_agent(session_id):
3492
+ if session_id in _agents:
3493
+ _agents.move_to_end(session_id)
3494
+ return _agents[session_id]
3495
+ if len(_agents) >= 128:
3496
+ _agents.popitem(last=False)
3487
3497
  # Get MCP Tools
3488
3498
  mcp_tools = await get_streamable_http_mcp_tools()
3499
+ # Re-check after the await: a concurrent first-invocation for the same
3500
+ # session_id may have built and stored the agent while we were awaiting.
3501
+ # Don't overwrite it (that would orphan the agent the other request is using).
3502
+ if session_id not in _agents:
3503
+ _agents[session_id] = AssistantAgent(
3504
+ name="{{ name }}",
3505
+ model_client=load_model(),
3506
+ tools=tools + mcp_tools,
3507
+ system_message=SYSTEM_MESSAGE,
3508
+ )
3509
+ _agents.move_to_end(session_id)
3510
+ return _agents[session_id]
3489
3511
 
3490
- # Define an AssistantAgent with the model and tools
3491
- agent = AssistantAgent(
3492
- name="{{ name }}",
3493
- model_client=load_model(),
3494
- tools=tools + mcp_tools,
3495
- system_message=SYSTEM_MESSAGE,
3496
- )
3512
+
3513
+ @app.entrypoint
3514
+ async def invoke(payload, context):
3515
+ log.info("Invoking Agent.....")
3497
3516
 
3498
3517
  # Process the user prompt
3499
3518
  prompt = payload.get("prompt", "What can you help me with?")
3519
+ session_id = getattr(context, "session_id", "default-session")
3520
+
3521
+ # Reuse the per-session agent (preserves conversation history)
3522
+ agent = await get_or_create_agent(session_id)
3500
3523
 
3501
3524
  # Run the agent
3502
3525
  result = await agent.run(task=prompt)
@@ -3822,6 +3845,7 @@ exports[`Assets Directory Snapshots > Python framework assets > python/python/ht
3822
3845
  "{{#if needsOs}}
3823
3846
  import os
3824
3847
  {{/if}}
3848
+ from collections import OrderedDict
3825
3849
  from google.adk.agents import Agent
3826
3850
  from google.adk.runners import Runner
3827
3851
  from google.adk.sessions import InMemorySessionService
@@ -3942,21 +3966,57 @@ agent = Agent(
3942
3966
  )
3943
3967
 
3944
3968
 
3945
- # Session and Runner
3946
- async def setup_session_and_runner(user_id, session_id):
3947
- ensure_credentials_loaded()
3948
- session_service = InMemorySessionService()
3949
- session = await session_service.create_session(
3969
+ # Module-level session service and runner preserve history across invocations.
3970
+ # InMemorySessionService retains every (app_name, user_id, session_id) triple
3971
+ # forever, so we bound it to 128 active sessions with LRU eviction (the
3972
+ # least-recently-used session is deleted and its history reset) to keep a
3973
+ # long-running process from growing without limit. For durable history, swap in
3974
+ # a persistent session service (e.g. DatabaseSessionService).
3975
+ _SESSION_LIMIT = 128
3976
+ _session_service = InMemorySessionService()
3977
+ _session_keys = OrderedDict()
3978
+ _runner = None
3979
+
3980
+
3981
+ def get_or_create_runner():
3982
+ global _runner
3983
+ if _runner is None:
3984
+ ensure_credentials_loaded()
3985
+ _runner = Runner(
3986
+ agent=agent,
3987
+ app_name=APP_NAME,
3988
+ session_service=_session_service,
3989
+ )
3990
+ return _runner
3991
+
3992
+
3993
+ async def get_or_create_session(user_id, session_id):
3994
+ key = (user_id, session_id)
3995
+ if key in _session_keys:
3996
+ _session_keys.move_to_end(key)
3997
+ else:
3998
+ while len(_session_keys) >= _SESSION_LIMIT:
3999
+ (old_user_id, old_session_id), _ = _session_keys.popitem(last=False)
4000
+ await _session_service.delete_session(
4001
+ app_name=APP_NAME, user_id=old_user_id, session_id=old_session_id
4002
+ )
4003
+ _session_keys[key] = True
4004
+
4005
+ session = await _session_service.get_session(
3950
4006
  app_name=APP_NAME, user_id=user_id, session_id=session_id
3951
4007
  )
3952
- runner = Runner(agent=agent, app_name=APP_NAME, session_service=session_service)
3953
- return session, runner
4008
+ if session is None:
4009
+ session = await _session_service.create_session(
4010
+ app_name=APP_NAME, user_id=user_id, session_id=session_id
4011
+ )
4012
+ return session
3954
4013
 
3955
4014
 
3956
4015
  # Agent Interaction
3957
4016
  async def call_agent_async(query, user_id, session_id):
3958
4017
  content = types.Content(role="user", parts=[types.Part(text=query)])
3959
- session, runner = await setup_session_and_runner(user_id, session_id)
4018
+ runner = get_or_create_runner()
4019
+ session = await get_or_create_session(user_id, session_id)
3960
4020
  events = runner.run_async(
3961
4021
  user_id=user_id, session_id=session.id, new_message=content
3962
4022
  )
@@ -4243,9 +4303,11 @@ exports[`Assets Directory Snapshots > Python framework assets > python/python/ht
4243
4303
  "{{#if needsOs}}
4244
4304
  import os
4245
4305
  {{/if}}
4306
+ from collections import OrderedDict
4246
4307
  from typing import Any
4247
4308
 
4248
4309
  from langchain_core.messages import HumanMessage{{#if hasConfigBundle}}, SystemMessage{{/if}}
4310
+ from langgraph.checkpoint.memory import InMemorySaver
4249
4311
  from langgraph.prebuilt import create_react_agent
4250
4312
  from langchain.tools import tool
4251
4313
  {{#if hasConfigBundle}}
@@ -4296,6 +4358,26 @@ def add_numbers(a: int, b: int) -> int:
4296
4358
  # Define a collection of tools used by the model
4297
4359
  tools = [add_numbers]
4298
4360
 
4361
+ # Module-level checkpointer preserves conversation history across invocations.
4362
+ # InMemorySaver keeps every thread_id (= session_id) checkpoint in memory
4363
+ # forever, so we bound it to 128 active threads with LRU eviction (the
4364
+ # least-recently-used thread is deleted and its history reset) to keep a
4365
+ # long-running process from growing without limit. For durable history, swap in
4366
+ # a persistent checkpointer (e.g. SqliteSaver/AsyncSqliteSaver with a file path).
4367
+ _CHECKPOINT_LIMIT = 128
4368
+ _checkpointer = InMemorySaver()
4369
+ _thread_ids = OrderedDict()
4370
+
4371
+
4372
+ def touch_thread(thread_id):
4373
+ if thread_id in _thread_ids:
4374
+ _thread_ids.move_to_end(thread_id)
4375
+ return
4376
+ while len(_thread_ids) >= _CHECKPOINT_LIMIT:
4377
+ evicted, _ = _thread_ids.popitem(last=False)
4378
+ _checkpointer.delete_thread(evicted)
4379
+ _thread_ids[thread_id] = True
4380
+
4299
4381
  {{#if needsOs}}
4300
4382
  _MOUNT_PATHS = [
4301
4383
  {{#if sessionStorageMountPath}}"{{sessionStorageMountPath}}",{{/if}}
@@ -4391,29 +4473,46 @@ async def invoke(payload, context):
4391
4473
  if mcp_client:
4392
4474
  mcp_tools = await mcp_client.get_tools()
4393
4475
 
4394
- # Define the agent using create_react_agent
4476
+ # Define the agent using create_react_agent (checkpointer is shared across invocations)
4395
4477
  {{#if hasConfigBundle}}
4396
- graph = create_react_agent(get_or_create_model(), tools=mcp_tools + tools, prompt=DEFAULT_SYSTEM_PROMPT)
4478
+ graph = create_react_agent(
4479
+ get_or_create_model(),
4480
+ tools=mcp_tools + tools,
4481
+ prompt=DEFAULT_SYSTEM_PROMPT,
4482
+ checkpointer=_checkpointer,
4483
+ )
4397
4484
  callback = ConfigBundleCallback()
4398
4485
 
4399
4486
  # Process the user prompt
4400
4487
  prompt = payload.get("prompt", "What can you help me with?")
4488
+ session_id = getattr(context, "session_id", "default-session")
4489
+ touch_thread(session_id)
4401
4490
  log.info(f"Agent input: {prompt}")
4402
4491
 
4403
- # Run the agent with config bundle callback
4492
+ # Run the agent with config bundle callback (checkpointer auto-loads/saves history per session)
4404
4493
  result = await graph.ainvoke(
4405
4494
  {"messages": [HumanMessage(content=prompt)]},
4406
- config={"callbacks": [callback]},
4495
+ config={"callbacks": [callback], "configurable": {"thread_id": session_id}},
4407
4496
  )
4408
4497
  {{else}}
4409
- graph = create_react_agent(get_or_create_model(), tools=mcp_tools + tools, prompt=DEFAULT_SYSTEM_PROMPT)
4498
+ graph = create_react_agent(
4499
+ get_or_create_model(),
4500
+ tools=mcp_tools + tools,
4501
+ prompt=DEFAULT_SYSTEM_PROMPT,
4502
+ checkpointer=_checkpointer,
4503
+ )
4410
4504
 
4411
4505
  # Process the user prompt
4412
4506
  prompt = payload.get("prompt", "What can you help me with?")
4507
+ session_id = getattr(context, "session_id", "default-session")
4508
+ touch_thread(session_id)
4413
4509
  log.info(f"Agent input: {prompt}")
4414
4510
 
4415
- # Run the agent
4416
- result = await graph.ainvoke({"messages": [HumanMessage(content=prompt)]})
4511
+ # Run the agent (checkpointer auto-loads/saves history per session)
4512
+ result = await graph.ainvoke(
4513
+ {"messages": [HumanMessage(content=prompt)]},
4514
+ config={"configurable": {"thread_id": session_id}},
4515
+ )
4417
4516
  {{/if}}
4418
4517
 
4419
4518
  # Return result
@@ -4784,7 +4883,8 @@ import os
4784
4883
  {{#if hasGateway}}
4785
4884
  from contextlib import AsyncExitStack
4786
4885
  {{/if}}
4787
- from agents import Agent, Runner, function_tool
4886
+ from functools import lru_cache
4887
+ from agents import Agent, Runner, SQLiteSession, function_tool
4788
4888
  from bedrock_agentcore.runtime import BedrockAgentCoreApp
4789
4889
  from model.load import load_model
4790
4890
  {{#if hasGateway}}
@@ -4888,8 +4988,16 @@ You have access to the following mounted filesystems. Use file_read, file_write,
4888
4988
  {{/each}}{{/if}}
4889
4989
  """
4890
4990
 
4991
+ # Caches up to 128 active sessions; LRU eviction silently resets history for
4992
+ # the oldest session. For production use, replace with a durable session store
4993
+ # (e.g. SQLiteSession with a file path).
4994
+ @lru_cache(maxsize=128)
4995
+ def get_session(session_id):
4996
+ return SQLiteSession(session_id)
4997
+
4998
+
4891
4999
  # Define the agent execution
4892
- async def main(query):
5000
+ async def main(query, session):
4893
5001
  ensure_credentials_loaded()
4894
5002
  try:
4895
5003
  {{#if hasGateway}}
@@ -4908,7 +5016,7 @@ async def main(query):
4908
5016
  tools=tools,
4909
5017
  mcp_config={"include_server_in_tool_names": True},
4910
5018
  )
4911
- result = await Runner.run(agent, query)
5019
+ result = await Runner.run(agent, query, session=session)
4912
5020
  return result
4913
5021
  else:
4914
5022
  agent = Agent(
@@ -4918,7 +5026,7 @@ async def main(query):
4918
5026
  mcp_servers=[],
4919
5027
  tools=tools
4920
5028
  )
4921
- result = await Runner.run(agent, query)
5029
+ result = await Runner.run(agent, query, session=session)
4922
5030
  return result
4923
5031
  {{else}}
4924
5032
  if mcp_servers:
@@ -4931,7 +5039,7 @@ async def main(query):
4931
5039
  mcp_servers=active_servers,
4932
5040
  tools=tools
4933
5041
  )
4934
- result = await Runner.run(agent, query)
5042
+ result = await Runner.run(agent, query, session=session)
4935
5043
  return result
4936
5044
  else:
4937
5045
  agent = Agent(
@@ -4941,7 +5049,7 @@ async def main(query):
4941
5049
  mcp_servers=[],
4942
5050
  tools=tools
4943
5051
  )
4944
- result = await Runner.run(agent, query)
5052
+ result = await Runner.run(agent, query, session=session)
4945
5053
  return result
4946
5054
  {{/if}}
4947
5055
  except Exception as e:
@@ -4955,9 +5063,11 @@ async def invoke(payload, context):
4955
5063
 
4956
5064
  # Process the user prompt
4957
5065
  prompt = payload.get("prompt", "What can you help me with?")
5066
+ session_id = getattr(context, "session_id", "default-session")
5067
+ session = get_session(session_id)
4958
5068
 
4959
- # Run the agent
4960
- result = await main(prompt)
5069
+ # Run the agent (session automatically loads/saves conversation history)
5070
+ result = await main(prompt, session)
4961
5071
 
4962
5072
  # Return result
4963
5073
  return {"result": result.final_output}
@@ -5211,6 +5321,7 @@ Thumbs.db"
5211
5321
 
5212
5322
  exports[`Assets Directory Snapshots > Python framework assets > python/python/http/strands/base/main.py should match snapshot 1`] = `
5213
5323
  "from typing import Any
5324
+ from collections import OrderedDict
5214
5325
  {{#if inlineFunctionTools}}
5215
5326
  import json
5216
5327
 
@@ -5646,26 +5757,21 @@ def agent_factory():
5646
5757
  get_or_create_agent = agent_factory()
5647
5758
  {{/unless}}
5648
5759
  {{else}}
5649
- {{#if hasConfigBundle}}
5650
- def create_agent({{#if hasSkillsFetcher}}skill_plugins=None{{/if}}):
5651
- return Agent(
5652
- model=load_model(),
5653
- system_prompt=DEFAULT_SYSTEM_PROMPT,
5654
- tools=tools,
5655
- conversation_manager=_make_conversation_manager(),
5656
- {{#if hasSkillsFetcher}}
5657
- plugins=skill_plugins or None,
5658
- {{/if}}
5659
- hooks=[ConfigBundleHook()],
5660
- )
5661
- {{else}}
5662
5760
  {{#unless hasPayment}}
5663
- _agent = None
5664
-
5665
- def get_or_create_agent({{#if hasSkillsFetcher}}skill_plugins=None{{/if}}):
5666
- global _agent
5667
- if _agent is None:
5668
- _agent = Agent(
5761
+ # Reuses one Agent per session_id so each session keeps its own in-process
5762
+ # conversation history (best-effort; resets on cold start). The cache is bounded
5763
+ # to 128 sessions with LRU eviction (least-recently-used is dropped and its
5764
+ # history reset) so a single process serving many sessions cannot leak history
5765
+ # between them or grow without limit. For durable history, attach a session manager.
5766
+ def agent_factory():
5767
+ cache = OrderedDict()
5768
+ def get_or_create_agent(session_id{{#if hasSkillsFetcher}}, skill_plugins=None{{/if}}):
5769
+ if session_id in cache:
5770
+ cache.move_to_end(session_id)
5771
+ return cache[session_id]
5772
+ if len(cache) >= 128:
5773
+ cache.popitem(last=False)
5774
+ cache[session_id] = Agent(
5669
5775
  model=load_model(),
5670
5776
  system_prompt=DEFAULT_SYSTEM_PROMPT,
5671
5777
  tools=tools,
@@ -5685,12 +5791,16 @@ def get_or_create_agent({{#if hasSkillsFetcher}}skill_plugins=None{{/if}}):
5685
5791
  {{#if timeoutSeconds}}timeout_seconds={{timeoutSeconds}},{{/if}}
5686
5792
  ),
5687
5793
  {{/if}}
5794
+ {{#if hasConfigBundle}}
5795
+ ConfigBundleHook(),
5796
+ {{/if}}
5688
5797
  ],
5689
5798
  )
5690
- return _agent
5799
+ return cache[session_id]
5800
+ return get_or_create_agent
5801
+ get_or_create_agent = agent_factory()
5691
5802
  {{/unless}}
5692
5803
  {{/if}}
5693
- {{/if}}
5694
5804
 
5695
5805
 
5696
5806
  def _extract_prompt(payload: dict):
@@ -5797,11 +5907,8 @@ async def invoke(payload, context):
5797
5907
  hooks=[ConfigBundleHook()],{{/if}}
5798
5908
  )
5799
5909
  {{else}}
5800
- {{#if hasConfigBundle}}
5801
- agent = create_agent({{#if hasSkillsFetcher}}_skill_plugins{{/if}})
5802
- {{else}}
5803
- agent = get_or_create_agent({{#if hasSkillsFetcher}}_skill_plugins{{/if}})
5804
- {{/if}}
5910
+ session_id = getattr(context, 'session_id', 'default-session')
5911
+ agent = get_or_create_agent(session_id{{#if hasSkillsFetcher}}, _skill_plugins{{/if}})
5805
5912
  {{/if}}
5806
5913
  {{/if}}
5807
5914
 
@@ -6704,7 +6811,7 @@ def get_memory_session_manager(session_id: Optional[str], actor_id: str) -> Opti
6704
6811
  f"/episodes/{actor_id}/{session_id}": RetrievalConfig(top_k=5, relevance_score=0.5),
6705
6812
  {{/if}}
6706
6813
  {{#if (includes memoryProviders.[0].strategies "SUMMARIZATION")}}
6707
- f"/summaries/{actor_id}/{session_id}": RetrievalConfig(top_k=3, relevance_score=0.5),
6814
+ f"/summaries/{actor_id}": RetrievalConfig(top_k=3, relevance_score=0.5),
6708
6815
  {{/if}}
6709
6816
  }
6710
6817
  {{/if}}
@@ -7588,18 +7695,35 @@ async function getOrCreateAgent(sessionId: string, actorId: string): Promise<Age
7588
7695
  return agent;
7589
7696
  }
7590
7697
  {{else}}
7591
- let cachedAgent: Agent | null = null;
7592
-
7593
- async function getOrCreateAgent(): Promise<Agent> {
7594
- if (!cachedAgent) {
7595
- const model = await loadModel();
7596
- cachedAgent = new Agent({
7597
- model,
7598
- systemPrompt: SYSTEM_PROMPT,
7599
- tools,
7600
- });
7698
+ const AGENT_CACHE_LIMIT = 128;
7699
+
7700
+ // Reuses one Agent per sessionId so each session keeps its own in-process
7701
+ // conversation history (best-effort; resets on cold start). A Map preserves
7702
+ // insertion order, so it doubles as an LRU bounded to 128 sessions — a local
7703
+ // dev process serving many sessions cannot leak history between them or grow
7704
+ // without bound. On AgentCore Runtime each microVM serves a single session, so
7705
+ // this holds one entry. For durable history, attach memory.
7706
+ const agentCache = new Map<string, Agent>();
7707
+
7708
+ async function getOrCreateAgent(sessionId: string): Promise<Agent> {
7709
+ const existing = agentCache.get(sessionId);
7710
+ if (existing) {
7711
+ agentCache.delete(sessionId);
7712
+ agentCache.set(sessionId, existing);
7713
+ return existing;
7714
+ }
7715
+ if (agentCache.size >= AGENT_CACHE_LIMIT) {
7716
+ const oldest = agentCache.keys().next().value;
7717
+ if (oldest !== undefined) agentCache.delete(oldest);
7601
7718
  }
7602
- return cachedAgent;
7719
+ const model = await loadModel();
7720
+ const agent = new Agent({
7721
+ model,
7722
+ systemPrompt: SYSTEM_PROMPT,
7723
+ tools,
7724
+ });
7725
+ agentCache.set(sessionId, agent);
7726
+ return agent;
7603
7727
  }
7604
7728
  {{/if}}
7605
7729
 
@@ -7611,7 +7735,8 @@ const app = new BedrockAgentCoreApp({
7611
7735
  const actorId = getActorId(payload, context);
7612
7736
  const agent = await getOrCreateAgent(sessionId, actorId);
7613
7737
  {{else}}
7614
- const agent = await getOrCreateAgent();
7738
+ const sessionId = context?.sessionId ?? 'default-session';
7739
+ const agent = await getOrCreateAgent(sessionId);
7615
7740
  {{/if}}
7616
7741
 
7617
7742
  {{#if hasMemory}}
@@ -7632,14 +7757,26 @@ const app = new BedrockAgentCoreApp({
7632
7757
  await agent.memoryManager?.flush();
7633
7758
  }
7634
7759
  {{else}}
7635
- for await (const event of agent.stream(payload.prompt ?? '')) {
7636
- if (
7637
- event.type === 'modelStreamUpdateEvent' &&
7638
- event.event?.type === 'modelContentBlockDeltaEvent' &&
7639
- event.event.delta?.type === 'textDelta'
7640
- ) {
7641
- yield { data: event.event.delta.text };
7760
+ // Snapshot history before streaming so a failed turn can be rolled back.
7761
+ // Agent.stream() appends the user message before invoking the model; on a
7762
+ // mid-stream error that user turn would otherwise linger in the cached
7763
+ // agent, and the next turn for this session would send consecutive user
7764
+ // messages (rejected by providers that require strict role alternation,
7765
+ // e.g. Anthropic). Restoring on error keeps the session reusable.
7766
+ const snapshot = agent.takeSnapshot({ include: ['messages'] });
7767
+ try {
7768
+ for await (const event of agent.stream(payload.prompt ?? '')) {
7769
+ if (
7770
+ event.type === 'modelStreamUpdateEvent' &&
7771
+ event.event?.type === 'modelContentBlockDeltaEvent' &&
7772
+ event.event.delta?.type === 'textDelta'
7773
+ ) {
7774
+ yield { data: event.event.delta.text };
7775
+ }
7642
7776
  }
7777
+ } catch (error) {
7778
+ agent.loadSnapshot(snapshot);
7779
+ throw error;
7643
7780
  }
7644
7781
  {{/if}}
7645
7782
  },
@@ -7961,24 +8098,64 @@ Thumbs.db
7961
8098
 
7962
8099
  exports[`Assets Directory Snapshots > TypeScript assets > typescript/typescript/http/vercelai/base/main.ts should match snapshot 1`] = `
7963
8100
  "import { BedrockAgentCoreApp } from 'bedrock-agentcore/runtime';
7964
- import { streamText } from 'ai';
8101
+ import { streamText, type ModelMessage } from 'ai';
7965
8102
  import { loadModel } from './model/load.js';
7966
8103
 
7967
8104
  const SYSTEM_PROMPT = \`You are a helpful assistant.\`;
7968
8105
 
8106
+ const HISTORY_LIMIT = 128;
8107
+
8108
+ // Keeps one message history per sessionId so each session remembers its own
8109
+ // turns (best-effort; resets on cold start). A Map preserves insertion order,
8110
+ // so it doubles as an LRU bounded to 128 sessions — a local dev process serving
8111
+ // many sessions cannot leak history between them or grow without bound. On
8112
+ // AgentCore Runtime each microVM serves a single session, so this holds one
8113
+ // entry. For durable history, persist messages to an external store.
8114
+ const histories = new Map<string, ModelMessage[]>();
8115
+
8116
+ function getHistory(sessionId: string): ModelMessage[] {
8117
+ const existing = histories.get(sessionId);
8118
+ if (existing) {
8119
+ histories.delete(sessionId);
8120
+ histories.set(sessionId, existing);
8121
+ return existing;
8122
+ }
8123
+ if (histories.size >= HISTORY_LIMIT) {
8124
+ const oldest = histories.keys().next().value;
8125
+ if (oldest !== undefined) histories.delete(oldest);
8126
+ }
8127
+ const fresh: ModelMessage[] = [];
8128
+ histories.set(sessionId, fresh);
8129
+ return fresh;
8130
+ }
8131
+
7969
8132
  const app = new BedrockAgentCoreApp({
7970
8133
  invocationHandler: {
7971
8134
  async *process(payload: any, context: any) {
8135
+ const sessionId = context?.sessionId ?? 'default-session';
8136
+ const history = getHistory(sessionId);
8137
+ const userMessage: ModelMessage = { role: 'user', content: payload.prompt ?? '' };
8138
+
7972
8139
  const model = await loadModel();
7973
8140
  const result = streamText({
7974
8141
  model,
7975
8142
  system: SYSTEM_PROMPT,
7976
- prompt: payload.prompt ?? '',
8143
+ messages: [...history, userMessage],
7977
8144
  });
7978
8145
 
8146
+ let assistant = '';
7979
8147
  for await (const chunk of result.textStream) {
8148
+ assistant += chunk;
7980
8149
  yield { data: chunk };
7981
8150
  }
8151
+
8152
+ // Commit the exchange to history only after a non-empty reply. On a failed
8153
+ // or empty stream the turn is dropped instead of leaving a dangling user
8154
+ // (or empty assistant) message — consecutive same-role or empty-content
8155
+ // messages would otherwise be rejected on the next turn for this session.
8156
+ if (assistant.length > 0) {
8157
+ history.push(userMessage, { role: 'assistant', content: assistant });
8158
+ }
7982
8159
  },
7983
8160
  },
7984
8161
  });