@aws/agentcore 1.0.0-preview.18 → 1.0.0-preview.19

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (63) hide show
  1. package/README.md +17 -17
  2. package/dist/assets/__tests__/__snapshots__/assets.snapshot.test.ts.snap +260 -83
  3. package/dist/assets/__tests__/googleadk-session-eviction.test.ts +87 -0
  4. package/dist/assets/__tests__/summarization-namespace.test.ts +43 -0
  5. package/dist/assets/python/a2a/strands/capabilities/memory/session.py +1 -1
  6. package/dist/assets/python/agui/strands/capabilities/memory/session.py +1 -1
  7. package/dist/assets/python/http/autogen/base/main.py +33 -10
  8. package/dist/assets/python/http/googleadk/base/main.py +45 -8
  9. package/dist/assets/python/http/langchain_langgraph/base/main.py +46 -7
  10. package/dist/assets/python/http/openaiagents/base/main.py +19 -8
  11. package/dist/assets/python/http/strands/base/main.py +23 -26
  12. package/dist/assets/python/http/strands/capabilities/memory/session.py +1 -1
  13. package/dist/assets/typescript/http/strands/base/main.ts +48 -18
  14. package/dist/assets/typescript/http/vercelai/base/main.ts +42 -2
  15. package/dist/cli/index.mjs +516 -516
  16. package/dist/lib/errors/types.d.ts +25 -0
  17. package/dist/lib/errors/types.d.ts.map +1 -1
  18. package/dist/lib/errors/types.js +40 -1
  19. package/dist/lib/errors/types.js.map +1 -1
  20. package/dist/lib/secrets/cipher.d.ts +12 -0
  21. package/dist/lib/secrets/cipher.d.ts.map +1 -0
  22. package/dist/lib/secrets/cipher.js +54 -0
  23. package/dist/lib/secrets/cipher.js.map +1 -0
  24. package/dist/lib/secrets/index.d.ts +4 -0
  25. package/dist/lib/secrets/index.d.ts.map +1 -0
  26. package/dist/lib/secrets/index.js +15 -0
  27. package/dist/lib/secrets/index.js.map +1 -0
  28. package/dist/lib/secrets/key-provider.d.ts +16 -0
  29. package/dist/lib/secrets/key-provider.d.ts.map +1 -0
  30. package/dist/lib/secrets/key-provider.js +191 -0
  31. package/dist/lib/secrets/key-provider.js.map +1 -0
  32. package/dist/lib/secrets/sensitive-keys.d.ts +21 -0
  33. package/dist/lib/secrets/sensitive-keys.d.ts.map +1 -0
  34. package/dist/lib/secrets/sensitive-keys.js +67 -0
  35. package/dist/lib/secrets/sensitive-keys.js.map +1 -0
  36. package/dist/lib/utils/env.d.ts +4 -2
  37. package/dist/lib/utils/env.d.ts.map +1 -1
  38. package/dist/lib/utils/env.js +57 -27
  39. package/dist/lib/utils/env.js.map +1 -1
  40. package/dist/schema/constants.d.ts +26 -0
  41. package/dist/schema/constants.d.ts.map +1 -1
  42. package/dist/schema/constants.js +34 -1
  43. package/dist/schema/constants.js.map +1 -1
  44. package/dist/schema/schemas/agent-env.d.ts +4 -2
  45. package/dist/schema/schemas/agent-env.d.ts.map +1 -1
  46. package/dist/schema/schemas/agent-env.js +23 -5
  47. package/dist/schema/schemas/agent-env.js.map +1 -1
  48. package/dist/schema/schemas/agentcore-project.d.ts +3 -1
  49. package/dist/schema/schemas/agentcore-project.d.ts.map +1 -1
  50. package/dist/schema/schemas/auth.d.ts +2 -3
  51. package/dist/schema/schemas/auth.d.ts.map +1 -1
  52. package/dist/schema/schemas/auth.js +8 -7
  53. package/dist/schema/schemas/auth.js.map +1 -1
  54. package/dist/schema/schemas/primitives/config-bundle.d.ts +1 -0
  55. package/dist/schema/schemas/primitives/config-bundle.d.ts.map +1 -1
  56. package/dist/schema/schemas/primitives/config-bundle.js +3 -0
  57. package/dist/schema/schemas/primitives/config-bundle.js.map +1 -1
  58. package/dist/schema/schemas/primitives/harness.d.ts +1 -0
  59. package/dist/schema/schemas/primitives/harness.d.ts.map +1 -1
  60. package/dist/schema/schemas/primitives/harness.js +16 -0
  61. package/dist/schema/schemas/primitives/harness.js.map +1 -1
  62. package/npm-shrinkwrap.json +223 -0
  63. package/package.json +4 -1
@@ -0,0 +1,87 @@
1
+ /**
2
+ * Regression test for the GoogleADK session-store LRU eviction (#808, #1639).
3
+ *
4
+ * The rendered template bounds InMemorySessionService to 128 sessions via an
5
+ * OrderedDict whose KEY is the composite (user_id, session_id) tuple. The
6
+ * eviction must unpack the key tuple from popitem() -- popitem() returns
7
+ * (key, value), so the value (always True) must be discarded. A previous
8
+ * bug unpacked `old_user_id, old_session_id = popitem(...)`, which assigned the
9
+ * tuple key to old_user_id and True to old_session_id, so delete_session() was
10
+ * called with garbage and the real session was never freed -- growing unbounded.
11
+ */
12
+ import * as fs from 'fs';
13
+ import Handlebars from 'handlebars';
14
+ import * as path from 'path';
15
+ import { describe, expect, it } from 'vitest';
16
+
17
+ const MAIN_PATH = path.resolve(__dirname, '..', 'python', 'http', 'googleadk', 'base', 'main.py');
18
+
19
+ const SESSION_LIMIT = 128;
20
+ const SEP = '::';
21
+
22
+ /**
23
+ * Mirror of the rendered template's `get_or_create_session` eviction loop so we
24
+ * can assert the observable behavior (store cap + correct delete ids) without a
25
+ * Python runtime. Returns the keys still tracked and the (user_id, session_id)
26
+ * pairs that delete_session was invoked with, in eviction order.
27
+ */
28
+ function simulateEviction(sessions: Array<[string, string]>): {
29
+ tracked: Set<string>;
30
+ deleted: Array<[string, string]>;
31
+ } {
32
+ const keys = new Map<string, true>();
33
+ const deleted: Array<[string, string]> = [];
34
+
35
+ for (const [userId, sessionId] of sessions) {
36
+ const key = `${userId}${SEP}${sessionId}`;
37
+ if (keys.has(key)) {
38
+ keys.delete(key);
39
+ keys.set(key, true);
40
+ continue;
41
+ }
42
+ while (keys.size >= SESSION_LIMIT) {
43
+ const oldest = keys.keys().next().value as string;
44
+ keys.delete(oldest);
45
+ const [oldUser, oldSession] = oldest.split(SEP);
46
+ deleted.push([oldUser, oldSession]);
47
+ }
48
+ keys.set(key, true);
49
+ }
50
+
51
+ return { tracked: new Set(keys.keys()), deleted };
52
+ }
53
+
54
+ describe('GoogleADK session store LRU eviction', () => {
55
+ const template = Handlebars.compile(fs.readFileSync(MAIN_PATH, 'utf-8'));
56
+ const rendered = template({ name: 'evictionagent', needsOs: false, hasGateway: false });
57
+
58
+ it('unpacks the composite key tuple from popitem (not key->user_id, value->session_id)', () => {
59
+ // The load-bearing fix: the key tuple must be destructured, value discarded.
60
+ expect(rendered).toContain('(old_user_id, old_session_id), _ = _session_keys.popitem(last=False)');
61
+ // Guard against the buggy form regressing back in.
62
+ expect(rendered).not.toMatch(/^\s*old_user_id, old_session_id = _session_keys\.popitem/m);
63
+ });
64
+
65
+ it('passes the unpacked ids straight to delete_session', () => {
66
+ expect(rendered).toContain('user_id=old_user_id, session_id=old_session_id');
67
+ });
68
+
69
+ it('caps the store at 128 and evicts the oldest sessions in order', () => {
70
+ const sessions: Array<[string, string]> = [];
71
+ for (let i = 0; i < 200; i++) {
72
+ sessions.push(['u', `s${i}`]);
73
+ }
74
+
75
+ const { tracked, deleted } = simulateEviction(sessions);
76
+
77
+ // Store stays bounded at the limit regardless of how many distinct sessions arrive.
78
+ expect(tracked.size).toBe(SESSION_LIMIT);
79
+ // 200 inserts - 128 retained = 72 evictions, oldest first.
80
+ expect(deleted.length).toBe(200 - SESSION_LIMIT);
81
+ expect(deleted[0]).toEqual(['u', 's0']);
82
+ expect(deleted[deleted.length - 1]).toEqual(['u', `s${200 - SESSION_LIMIT - 1}`]);
83
+ // The most recent session is retained; the first evicted one is gone.
84
+ expect(tracked.has(`u${SEP}s199`)).toBe(true);
85
+ expect(tracked.has(`u${SEP}s0`)).toBe(false);
86
+ });
87
+ });
@@ -0,0 +1,43 @@
1
+ /**
2
+ * Regression test for cross-session SUMMARIZATION recall (issue #665).
3
+ *
4
+ * The SUMMARIZATION retrieval namespace must be actor-scoped (`/summaries/{actor_id}`),
5
+ * not session-scoped. The SDK's namespace_path is a hierarchical prefix match, so a
6
+ * per-session prefix (`/summaries/{actor_id}/{session_id}`) only ever matches the current
7
+ * session's own summaries and never surfaces summaries written by prior sessions —
8
+ * silently breaking cross-session recall. This guards against another silent revert
9
+ * (see PR #1299, reverted by squash release #1547).
10
+ */
11
+ import Handlebars from 'handlebars';
12
+ import * as fs from 'fs';
13
+ import * as path from 'path';
14
+ import { describe, expect, it } from 'vitest';
15
+ // Importing render registers the `includes` Handlebars helper used by the template.
16
+ import '../../cli/templates/render.js';
17
+
18
+ const FLAVORS = ['http', 'agui', 'a2a'] as const;
19
+
20
+ function renderSessionTemplate(flavor: string): string {
21
+ const templatePath = path.resolve(
22
+ __dirname,
23
+ '..',
24
+ 'python',
25
+ flavor,
26
+ 'strands',
27
+ 'capabilities',
28
+ 'memory',
29
+ 'session.py'
30
+ );
31
+ const content = fs.readFileSync(templatePath, 'utf-8');
32
+ return Handlebars.compile(content)({
33
+ memoryProviders: [{ envVarName: 'MEMORY_TEST_ID', strategies: ['SUMMARIZATION'] }],
34
+ });
35
+ }
36
+
37
+ describe('SUMMARIZATION retrieval namespace', () => {
38
+ it.each(FLAVORS)('%s session.py uses an actor-scoped summary namespace', flavor => {
39
+ const rendered = renderSessionTemplate(flavor);
40
+ expect(rendered).toContain('f"/summaries/{actor_id}": RetrievalConfig');
41
+ expect(rendered).not.toContain('/summaries/{actor_id}/{session_id}');
42
+ });
43
+ });
@@ -28,7 +28,7 @@ def get_memory_session_manager(session_id: Optional[str], actor_id: str) -> Opti
28
28
  f"/episodes/{actor_id}/{session_id}": RetrievalConfig(top_k=5, relevance_score=0.5),
29
29
  {{/if}}
30
30
  {{#if (includes memoryProviders.[0].strategies "SUMMARIZATION")}}
31
- f"/summaries/{actor_id}/{session_id}": RetrievalConfig(top_k=3, relevance_score=0.5),
31
+ f"/summaries/{actor_id}": RetrievalConfig(top_k=3, relevance_score=0.5),
32
32
  {{/if}}
33
33
  }
34
34
  {{/if}}
@@ -28,7 +28,7 @@ def get_memory_session_manager(session_id: Optional[str], actor_id: str) -> Opti
28
28
  f"/episodes/{actor_id}/{session_id}": RetrievalConfig(top_k=5, relevance_score=0.5),
29
29
  {{/if}}
30
30
  {{#if (includes memoryProviders.[0].strategies "SUMMARIZATION")}}
31
- f"/summaries/{actor_id}/{session_id}": RetrievalConfig(top_k=3, relevance_score=0.5),
31
+ f"/summaries/{actor_id}": RetrievalConfig(top_k=3, relevance_score=0.5),
32
32
  {{/if}}
33
33
  }
34
34
  {{/if}}
@@ -1,6 +1,7 @@
1
1
  {{#if needsOs}}
2
2
  import os
3
3
  {{/if}}
4
+ from collections import OrderedDict
4
5
  from autogen_agentchat.agents import AssistantAgent
5
6
  from autogen_core.tools import FunctionTool
6
7
  from bedrock_agentcore.runtime import BedrockAgentCoreApp
@@ -91,23 +92,45 @@ You have access to the following mounted filesystems. Use file_read, file_write,
91
92
  {{/each}}{{/if}}
92
93
  """
93
94
 
94
- @app.entrypoint
95
- async def invoke(payload, context):
96
- log.info("Invoking Agent.....")
95
+ # Reuses one AssistantAgent per session_id so each session keeps its own
96
+ # in-process conversation history (best-effort; resets on cold start). Caches up
97
+ # to 128 active sessions with LRU eviction (least-recently-used is dropped and
98
+ # its history reset).
99
+ _agents = OrderedDict()
97
100
 
101
+
102
+ async def get_or_create_agent(session_id):
103
+ if session_id in _agents:
104
+ _agents.move_to_end(session_id)
105
+ return _agents[session_id]
106
+ if len(_agents) >= 128:
107
+ _agents.popitem(last=False)
98
108
  # Get MCP Tools
99
109
  mcp_tools = await get_streamable_http_mcp_tools()
110
+ # Re-check after the await: a concurrent first-invocation for the same
111
+ # session_id may have built and stored the agent while we were awaiting.
112
+ # Don't overwrite it (that would orphan the agent the other request is using).
113
+ if session_id not in _agents:
114
+ _agents[session_id] = AssistantAgent(
115
+ name="{{ name }}",
116
+ model_client=load_model(),
117
+ tools=tools + mcp_tools,
118
+ system_message=SYSTEM_MESSAGE,
119
+ )
120
+ _agents.move_to_end(session_id)
121
+ return _agents[session_id]
100
122
 
101
- # Define an AssistantAgent with the model and tools
102
- agent = AssistantAgent(
103
- name="{{ name }}",
104
- model_client=load_model(),
105
- tools=tools + mcp_tools,
106
- system_message=SYSTEM_MESSAGE,
107
- )
123
+
124
+ @app.entrypoint
125
+ async def invoke(payload, context):
126
+ log.info("Invoking Agent.....")
108
127
 
109
128
  # Process the user prompt
110
129
  prompt = payload.get("prompt", "What can you help me with?")
130
+ session_id = getattr(context, "session_id", "default-session")
131
+
132
+ # Reuse the per-session agent (preserves conversation history)
133
+ agent = await get_or_create_agent(session_id)
111
134
 
112
135
  # Run the agent
113
136
  result = await agent.run(task=prompt)
@@ -1,6 +1,7 @@
1
1
  {{#if needsOs}}
2
2
  import os
3
3
  {{/if}}
4
+ from collections import OrderedDict
4
5
  from google.adk.agents import Agent
5
6
  from google.adk.runners import Runner
6
7
  from google.adk.sessions import InMemorySessionService
@@ -121,21 +122,57 @@ agent = Agent(
121
122
  )
122
123
 
123
124
 
124
- # Session and Runner
125
- async def setup_session_and_runner(user_id, session_id):
126
- ensure_credentials_loaded()
127
- session_service = InMemorySessionService()
128
- session = await session_service.create_session(
125
+ # Module-level session service and runner preserve history across invocations.
126
+ # InMemorySessionService retains every (app_name, user_id, session_id) triple
127
+ # forever, so we bound it to 128 active sessions with LRU eviction (the
128
+ # least-recently-used session is deleted and its history reset) to keep a
129
+ # long-running process from growing without limit. For durable history, swap in
130
+ # a persistent session service (e.g. DatabaseSessionService).
131
+ _SESSION_LIMIT = 128
132
+ _session_service = InMemorySessionService()
133
+ _session_keys = OrderedDict()
134
+ _runner = None
135
+
136
+
137
+ def get_or_create_runner():
138
+ global _runner
139
+ if _runner is None:
140
+ ensure_credentials_loaded()
141
+ _runner = Runner(
142
+ agent=agent,
143
+ app_name=APP_NAME,
144
+ session_service=_session_service,
145
+ )
146
+ return _runner
147
+
148
+
149
+ async def get_or_create_session(user_id, session_id):
150
+ key = (user_id, session_id)
151
+ if key in _session_keys:
152
+ _session_keys.move_to_end(key)
153
+ else:
154
+ while len(_session_keys) >= _SESSION_LIMIT:
155
+ (old_user_id, old_session_id), _ = _session_keys.popitem(last=False)
156
+ await _session_service.delete_session(
157
+ app_name=APP_NAME, user_id=old_user_id, session_id=old_session_id
158
+ )
159
+ _session_keys[key] = True
160
+
161
+ session = await _session_service.get_session(
129
162
  app_name=APP_NAME, user_id=user_id, session_id=session_id
130
163
  )
131
- runner = Runner(agent=agent, app_name=APP_NAME, session_service=session_service)
132
- return session, runner
164
+ if session is None:
165
+ session = await _session_service.create_session(
166
+ app_name=APP_NAME, user_id=user_id, session_id=session_id
167
+ )
168
+ return session
133
169
 
134
170
 
135
171
  # Agent Interaction
136
172
  async def call_agent_async(query, user_id, session_id):
137
173
  content = types.Content(role="user", parts=[types.Part(text=query)])
138
- session, runner = await setup_session_and_runner(user_id, session_id)
174
+ runner = get_or_create_runner()
175
+ session = await get_or_create_session(user_id, session_id)
139
176
  events = runner.run_async(
140
177
  user_id=user_id, session_id=session.id, new_message=content
141
178
  )
@@ -1,9 +1,11 @@
1
1
  {{#if needsOs}}
2
2
  import os
3
3
  {{/if}}
4
+ from collections import OrderedDict
4
5
  from typing import Any
5
6
 
6
7
  from langchain_core.messages import HumanMessage{{#if hasConfigBundle}}, SystemMessage{{/if}}
8
+ from langgraph.checkpoint.memory import InMemorySaver
7
9
  from langgraph.prebuilt import create_react_agent
8
10
  from langchain.tools import tool
9
11
  {{#if hasConfigBundle}}
@@ -54,6 +56,26 @@ def add_numbers(a: int, b: int) -> int:
54
56
  # Define a collection of tools used by the model
55
57
  tools = [add_numbers]
56
58
 
59
+ # Module-level checkpointer preserves conversation history across invocations.
60
+ # InMemorySaver keeps every thread_id (= session_id) checkpoint in memory
61
+ # forever, so we bound it to 128 active threads with LRU eviction (the
62
+ # least-recently-used thread is deleted and its history reset) to keep a
63
+ # long-running process from growing without limit. For durable history, swap in
64
+ # a persistent checkpointer (e.g. SqliteSaver/AsyncSqliteSaver with a file path).
65
+ _CHECKPOINT_LIMIT = 128
66
+ _checkpointer = InMemorySaver()
67
+ _thread_ids = OrderedDict()
68
+
69
+
70
+ def touch_thread(thread_id):
71
+ if thread_id in _thread_ids:
72
+ _thread_ids.move_to_end(thread_id)
73
+ return
74
+ while len(_thread_ids) >= _CHECKPOINT_LIMIT:
75
+ evicted, _ = _thread_ids.popitem(last=False)
76
+ _checkpointer.delete_thread(evicted)
77
+ _thread_ids[thread_id] = True
78
+
57
79
  {{#if needsOs}}
58
80
  _MOUNT_PATHS = [
59
81
  {{#if sessionStorageMountPath}}"{{sessionStorageMountPath}}",{{/if}}
@@ -149,29 +171,46 @@ async def invoke(payload, context):
149
171
  if mcp_client:
150
172
  mcp_tools = await mcp_client.get_tools()
151
173
 
152
- # Define the agent using create_react_agent
174
+ # Define the agent using create_react_agent (checkpointer is shared across invocations)
153
175
  {{#if hasConfigBundle}}
154
- graph = create_react_agent(get_or_create_model(), tools=mcp_tools + tools, prompt=DEFAULT_SYSTEM_PROMPT)
176
+ graph = create_react_agent(
177
+ get_or_create_model(),
178
+ tools=mcp_tools + tools,
179
+ prompt=DEFAULT_SYSTEM_PROMPT,
180
+ checkpointer=_checkpointer,
181
+ )
155
182
  callback = ConfigBundleCallback()
156
183
 
157
184
  # Process the user prompt
158
185
  prompt = payload.get("prompt", "What can you help me with?")
186
+ session_id = getattr(context, "session_id", "default-session")
187
+ touch_thread(session_id)
159
188
  log.info(f"Agent input: {prompt}")
160
189
 
161
- # Run the agent with config bundle callback
190
+ # Run the agent with config bundle callback (checkpointer auto-loads/saves history per session)
162
191
  result = await graph.ainvoke(
163
192
  {"messages": [HumanMessage(content=prompt)]},
164
- config={"callbacks": [callback]},
193
+ config={"callbacks": [callback], "configurable": {"thread_id": session_id}},
165
194
  )
166
195
  {{else}}
167
- graph = create_react_agent(get_or_create_model(), tools=mcp_tools + tools, prompt=DEFAULT_SYSTEM_PROMPT)
196
+ graph = create_react_agent(
197
+ get_or_create_model(),
198
+ tools=mcp_tools + tools,
199
+ prompt=DEFAULT_SYSTEM_PROMPT,
200
+ checkpointer=_checkpointer,
201
+ )
168
202
 
169
203
  # Process the user prompt
170
204
  prompt = payload.get("prompt", "What can you help me with?")
205
+ session_id = getattr(context, "session_id", "default-session")
206
+ touch_thread(session_id)
171
207
  log.info(f"Agent input: {prompt}")
172
208
 
173
- # Run the agent
174
- result = await graph.ainvoke({"messages": [HumanMessage(content=prompt)]})
209
+ # Run the agent (checkpointer auto-loads/saves history per session)
210
+ result = await graph.ainvoke(
211
+ {"messages": [HumanMessage(content=prompt)]},
212
+ config={"configurable": {"thread_id": session_id}},
213
+ )
175
214
  {{/if}}
176
215
 
177
216
  # Return result
@@ -4,7 +4,8 @@ import os
4
4
  {{#if hasGateway}}
5
5
  from contextlib import AsyncExitStack
6
6
  {{/if}}
7
- from agents import Agent, Runner, function_tool
7
+ from functools import lru_cache
8
+ from agents import Agent, Runner, SQLiteSession, function_tool
8
9
  from bedrock_agentcore.runtime import BedrockAgentCoreApp
9
10
  from model.load import load_model
10
11
  {{#if hasGateway}}
@@ -108,8 +109,16 @@ You have access to the following mounted filesystems. Use file_read, file_write,
108
109
  {{/each}}{{/if}}
109
110
  """
110
111
 
112
+ # Caches up to 128 active sessions; LRU eviction silently resets history for
113
+ # the oldest session. For production use, replace with a durable session store
114
+ # (e.g. SQLiteSession with a file path).
115
+ @lru_cache(maxsize=128)
116
+ def get_session(session_id):
117
+ return SQLiteSession(session_id)
118
+
119
+
111
120
  # Define the agent execution
112
- async def main(query):
121
+ async def main(query, session):
113
122
  ensure_credentials_loaded()
114
123
  try:
115
124
  {{#if hasGateway}}
@@ -128,7 +137,7 @@ async def main(query):
128
137
  tools=tools,
129
138
  mcp_config={"include_server_in_tool_names": True},
130
139
  )
131
- result = await Runner.run(agent, query)
140
+ result = await Runner.run(agent, query, session=session)
132
141
  return result
133
142
  else:
134
143
  agent = Agent(
@@ -138,7 +147,7 @@ async def main(query):
138
147
  mcp_servers=[],
139
148
  tools=tools
140
149
  )
141
- result = await Runner.run(agent, query)
150
+ result = await Runner.run(agent, query, session=session)
142
151
  return result
143
152
  {{else}}
144
153
  if mcp_servers:
@@ -151,7 +160,7 @@ async def main(query):
151
160
  mcp_servers=active_servers,
152
161
  tools=tools
153
162
  )
154
- result = await Runner.run(agent, query)
163
+ result = await Runner.run(agent, query, session=session)
155
164
  return result
156
165
  else:
157
166
  agent = Agent(
@@ -161,7 +170,7 @@ async def main(query):
161
170
  mcp_servers=[],
162
171
  tools=tools
163
172
  )
164
- result = await Runner.run(agent, query)
173
+ result = await Runner.run(agent, query, session=session)
165
174
  return result
166
175
  {{/if}}
167
176
  except Exception as e:
@@ -175,9 +184,11 @@ async def invoke(payload, context):
175
184
 
176
185
  # Process the user prompt
177
186
  prompt = payload.get("prompt", "What can you help me with?")
187
+ session_id = getattr(context, "session_id", "default-session")
188
+ session = get_session(session_id)
178
189
 
179
- # Run the agent
180
- result = await main(prompt)
190
+ # Run the agent (session automatically loads/saves conversation history)
191
+ result = await main(prompt, session)
181
192
 
182
193
  # Return result
183
194
  return {"result": result.final_output}
@@ -1,4 +1,5 @@
1
1
  from typing import Any
2
+ from collections import OrderedDict
2
3
  {{#if inlineFunctionTools}}
3
4
  import json
4
5
 
@@ -434,26 +435,21 @@ def agent_factory():
434
435
  get_or_create_agent = agent_factory()
435
436
  {{/unless}}
436
437
  {{else}}
437
- {{#if hasConfigBundle}}
438
- def create_agent({{#if hasSkillsFetcher}}skill_plugins=None{{/if}}):
439
- return Agent(
440
- model=load_model(),
441
- system_prompt=DEFAULT_SYSTEM_PROMPT,
442
- tools=tools,
443
- conversation_manager=_make_conversation_manager(),
444
- {{#if hasSkillsFetcher}}
445
- plugins=skill_plugins or None,
446
- {{/if}}
447
- hooks=[ConfigBundleHook()],
448
- )
449
- {{else}}
450
438
  {{#unless hasPayment}}
451
- _agent = None
452
-
453
- def get_or_create_agent({{#if hasSkillsFetcher}}skill_plugins=None{{/if}}):
454
- global _agent
455
- if _agent is None:
456
- _agent = Agent(
439
+ # Reuses one Agent per session_id so each session keeps its own in-process
440
+ # conversation history (best-effort; resets on cold start). The cache is bounded
441
+ # to 128 sessions with LRU eviction (least-recently-used is dropped and its
442
+ # history reset) so a single process serving many sessions cannot leak history
443
+ # between them or grow without limit. For durable history, attach a session manager.
444
+ def agent_factory():
445
+ cache = OrderedDict()
446
+ def get_or_create_agent(session_id{{#if hasSkillsFetcher}}, skill_plugins=None{{/if}}):
447
+ if session_id in cache:
448
+ cache.move_to_end(session_id)
449
+ return cache[session_id]
450
+ if len(cache) >= 128:
451
+ cache.popitem(last=False)
452
+ cache[session_id] = Agent(
457
453
  model=load_model(),
458
454
  system_prompt=DEFAULT_SYSTEM_PROMPT,
459
455
  tools=tools,
@@ -473,12 +469,16 @@ def get_or_create_agent({{#if hasSkillsFetcher}}skill_plugins=None{{/if}}):
473
469
  {{#if timeoutSeconds}}timeout_seconds={{timeoutSeconds}},{{/if}}
474
470
  ),
475
471
  {{/if}}
472
+ {{#if hasConfigBundle}}
473
+ ConfigBundleHook(),
474
+ {{/if}}
476
475
  ],
477
476
  )
478
- return _agent
477
+ return cache[session_id]
478
+ return get_or_create_agent
479
+ get_or_create_agent = agent_factory()
479
480
  {{/unless}}
480
481
  {{/if}}
481
- {{/if}}
482
482
 
483
483
 
484
484
  def _extract_prompt(payload: dict):
@@ -585,11 +585,8 @@ async def invoke(payload, context):
585
585
  hooks=[ConfigBundleHook()],{{/if}}
586
586
  )
587
587
  {{else}}
588
- {{#if hasConfigBundle}}
589
- agent = create_agent({{#if hasSkillsFetcher}}_skill_plugins{{/if}})
590
- {{else}}
591
- agent = get_or_create_agent({{#if hasSkillsFetcher}}_skill_plugins{{/if}})
592
- {{/if}}
588
+ session_id = getattr(context, 'session_id', 'default-session')
589
+ agent = get_or_create_agent(session_id{{#if hasSkillsFetcher}}, _skill_plugins{{/if}})
593
590
  {{/if}}
594
591
  {{/if}}
595
592
 
@@ -28,7 +28,7 @@ def get_memory_session_manager(session_id: Optional[str], actor_id: str) -> Opti
28
28
  f"/episodes/{actor_id}/{session_id}": RetrievalConfig(top_k=5, relevance_score=0.5),
29
29
  {{/if}}
30
30
  {{#if (includes memoryProviders.[0].strategies "SUMMARIZATION")}}
31
- f"/summaries/{actor_id}/{session_id}": RetrievalConfig(top_k=3, relevance_score=0.5),
31
+ f"/summaries/{actor_id}": RetrievalConfig(top_k=3, relevance_score=0.5),
32
32
  {{/if}}
33
33
  }
34
34
  {{/if}}
@@ -53,18 +53,35 @@ async function getOrCreateAgent(sessionId: string, actorId: string): Promise<Age
53
53
  return agent;
54
54
  }
55
55
  {{else}}
56
- let cachedAgent: Agent | null = null;
56
+ const AGENT_CACHE_LIMIT = 128;
57
57
 
58
- async function getOrCreateAgent(): Promise<Agent> {
59
- if (!cachedAgent) {
60
- const model = await loadModel();
61
- cachedAgent = new Agent({
62
- model,
63
- systemPrompt: SYSTEM_PROMPT,
64
- tools,
65
- });
58
+ // Reuses one Agent per sessionId so each session keeps its own in-process
59
+ // conversation history (best-effort; resets on cold start). A Map preserves
60
+ // insertion order, so it doubles as an LRU bounded to 128 sessions — a local
61
+ // dev process serving many sessions cannot leak history between them or grow
62
+ // without bound. On AgentCore Runtime each microVM serves a single session, so
63
+ // this holds one entry. For durable history, attach memory.
64
+ const agentCache = new Map<string, Agent>();
65
+
66
+ async function getOrCreateAgent(sessionId: string): Promise<Agent> {
67
+ const existing = agentCache.get(sessionId);
68
+ if (existing) {
69
+ agentCache.delete(sessionId);
70
+ agentCache.set(sessionId, existing);
71
+ return existing;
72
+ }
73
+ if (agentCache.size >= AGENT_CACHE_LIMIT) {
74
+ const oldest = agentCache.keys().next().value;
75
+ if (oldest !== undefined) agentCache.delete(oldest);
66
76
  }
67
- return cachedAgent;
77
+ const model = await loadModel();
78
+ const agent = new Agent({
79
+ model,
80
+ systemPrompt: SYSTEM_PROMPT,
81
+ tools,
82
+ });
83
+ agentCache.set(sessionId, agent);
84
+ return agent;
68
85
  }
69
86
  {{/if}}
70
87
 
@@ -76,7 +93,8 @@ const app = new BedrockAgentCoreApp({
76
93
  const actorId = getActorId(payload, context);
77
94
  const agent = await getOrCreateAgent(sessionId, actorId);
78
95
  {{else}}
79
- const agent = await getOrCreateAgent();
96
+ const sessionId = context?.sessionId ?? 'default-session';
97
+ const agent = await getOrCreateAgent(sessionId);
80
98
  {{/if}}
81
99
 
82
100
  {{#if hasMemory}}
@@ -97,14 +115,26 @@ const app = new BedrockAgentCoreApp({
97
115
  await agent.memoryManager?.flush();
98
116
  }
99
117
  {{else}}
100
- for await (const event of agent.stream(payload.prompt ?? '')) {
101
- if (
102
- event.type === 'modelStreamUpdateEvent' &&
103
- event.event?.type === 'modelContentBlockDeltaEvent' &&
104
- event.event.delta?.type === 'textDelta'
105
- ) {
106
- yield { data: event.event.delta.text };
118
+ // Snapshot history before streaming so a failed turn can be rolled back.
119
+ // Agent.stream() appends the user message before invoking the model; on a
120
+ // mid-stream error that user turn would otherwise linger in the cached
121
+ // agent, and the next turn for this session would send consecutive user
122
+ // messages (rejected by providers that require strict role alternation,
123
+ // e.g. Anthropic). Restoring on error keeps the session reusable.
124
+ const snapshot = agent.takeSnapshot({ include: ['messages'] });
125
+ try {
126
+ for await (const event of agent.stream(payload.prompt ?? '')) {
127
+ if (
128
+ event.type === 'modelStreamUpdateEvent' &&
129
+ event.event?.type === 'modelContentBlockDeltaEvent' &&
130
+ event.event.delta?.type === 'textDelta'
131
+ ) {
132
+ yield { data: event.event.delta.text };
133
+ }
107
134
  }
135
+ } catch (error) {
136
+ agent.loadSnapshot(snapshot);
137
+ throw error;
108
138
  }
109
139
  {{/if}}
110
140
  },