@mastra/memory 0.0.0-error-handler-fix-20251020202607 → 0.0.0-execa-dynamic-import-20260304221256

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (119) hide show
  1. package/CHANGELOG.md +2181 -3
  2. package/LICENSE.md +15 -0
  3. package/dist/_types/@internal_ai-sdk-v4/dist/index.d.ts +7562 -0
  4. package/dist/chunk-23EXJLET.cjs +84 -0
  5. package/dist/chunk-23EXJLET.cjs.map +1 -0
  6. package/dist/chunk-33ZIVK5L.js +5220 -0
  7. package/dist/chunk-33ZIVK5L.js.map +1 -0
  8. package/dist/chunk-BSDWQEU3.js +79 -0
  9. package/dist/chunk-BSDWQEU3.js.map +1 -0
  10. package/dist/chunk-DGUM43GV.js +10 -0
  11. package/dist/chunk-DGUM43GV.js.map +1 -0
  12. package/dist/chunk-EQ4M72KU.js +439 -0
  13. package/dist/chunk-EQ4M72KU.js.map +1 -0
  14. package/dist/chunk-HJYHDIOC.js +250 -0
  15. package/dist/chunk-HJYHDIOC.js.map +1 -0
  16. package/dist/chunk-IDRQZVB4.cjs +84 -0
  17. package/dist/chunk-IDRQZVB4.cjs.map +1 -0
  18. package/dist/chunk-JEQ2X3Z6.cjs +12 -0
  19. package/dist/chunk-JEQ2X3Z6.cjs.map +1 -0
  20. package/dist/chunk-LIBOSOHM.cjs +252 -0
  21. package/dist/chunk-LIBOSOHM.cjs.map +1 -0
  22. package/dist/chunk-Q7VDGOMJ.cjs +5240 -0
  23. package/dist/chunk-Q7VDGOMJ.cjs.map +1 -0
  24. package/dist/chunk-RC6RZVYE.js +79 -0
  25. package/dist/chunk-RC6RZVYE.js.map +1 -0
  26. package/dist/chunk-ZD3BKU5O.cjs +441 -0
  27. package/dist/chunk-ZD3BKU5O.cjs.map +1 -0
  28. package/dist/docs/SKILL.md +55 -0
  29. package/dist/docs/assets/SOURCE_MAP.json +103 -0
  30. package/dist/docs/references/docs-agents-agent-approval.md +588 -0
  31. package/dist/docs/references/docs-agents-agent-memory.md +209 -0
  32. package/dist/docs/references/docs-agents-network-approval.md +275 -0
  33. package/dist/docs/references/docs-agents-networks.md +299 -0
  34. package/dist/docs/references/docs-agents-supervisor-agents.md +304 -0
  35. package/dist/docs/references/docs-memory-memory-processors.md +314 -0
  36. package/dist/docs/references/docs-memory-message-history.md +260 -0
  37. package/dist/docs/references/docs-memory-observational-memory.md +255 -0
  38. package/dist/docs/references/docs-memory-overview.md +45 -0
  39. package/dist/docs/references/docs-memory-semantic-recall.md +288 -0
  40. package/dist/docs/references/docs-memory-storage.md +261 -0
  41. package/dist/docs/references/docs-memory-working-memory.md +400 -0
  42. package/dist/docs/references/reference-core-getMemory.md +50 -0
  43. package/dist/docs/references/reference-core-listMemory.md +56 -0
  44. package/dist/docs/references/reference-memory-clone-utilities.md +199 -0
  45. package/dist/docs/references/reference-memory-cloneThread.md +142 -0
  46. package/dist/docs/references/reference-memory-createThread.md +68 -0
  47. package/dist/docs/references/reference-memory-getThreadById.md +24 -0
  48. package/dist/docs/references/reference-memory-listThreads.md +145 -0
  49. package/dist/docs/references/reference-memory-memory-class.md +147 -0
  50. package/dist/docs/references/reference-memory-observational-memory.md +577 -0
  51. package/dist/docs/references/reference-processors-token-limiter-processor.md +115 -0
  52. package/dist/docs/references/reference-storage-dynamodb.md +282 -0
  53. package/dist/docs/references/reference-storage-libsql.md +135 -0
  54. package/dist/docs/references/reference-storage-mongodb.md +262 -0
  55. package/dist/docs/references/reference-storage-postgresql.md +526 -0
  56. package/dist/docs/references/reference-storage-upstash.md +160 -0
  57. package/dist/docs/references/reference-vectors-libsql.md +305 -0
  58. package/dist/docs/references/reference-vectors-mongodb.md +295 -0
  59. package/dist/docs/references/reference-vectors-pg.md +408 -0
  60. package/dist/docs/references/reference-vectors-upstash.md +294 -0
  61. package/dist/index.cjs +16008 -291
  62. package/dist/index.cjs.map +1 -1
  63. package/dist/index.d.ts +252 -49
  64. package/dist/index.d.ts.map +1 -1
  65. package/dist/index.js +15970 -295
  66. package/dist/index.js.map +1 -1
  67. package/dist/observational-memory-3NJCJNCQ.cjs +64 -0
  68. package/dist/observational-memory-3NJCJNCQ.cjs.map +1 -0
  69. package/dist/observational-memory-SKDDNPFW.js +3 -0
  70. package/dist/observational-memory-SKDDNPFW.js.map +1 -0
  71. package/dist/processors/index.cjs +57 -158
  72. package/dist/processors/index.cjs.map +1 -1
  73. package/dist/processors/index.d.ts +1 -2
  74. package/dist/processors/index.d.ts.map +1 -1
  75. package/dist/processors/index.js +1 -156
  76. package/dist/processors/index.js.map +1 -1
  77. package/dist/processors/observational-memory/date-utils.d.ts +35 -0
  78. package/dist/processors/observational-memory/date-utils.d.ts.map +1 -0
  79. package/dist/processors/observational-memory/index.d.ts +18 -0
  80. package/dist/processors/observational-memory/index.d.ts.map +1 -0
  81. package/dist/processors/observational-memory/markers.d.ts +94 -0
  82. package/dist/processors/observational-memory/markers.d.ts.map +1 -0
  83. package/dist/processors/observational-memory/observational-memory.d.ts +842 -0
  84. package/dist/processors/observational-memory/observational-memory.d.ts.map +1 -0
  85. package/dist/processors/observational-memory/observer-agent.d.ts +140 -0
  86. package/dist/processors/observational-memory/observer-agent.d.ts.map +1 -0
  87. package/dist/processors/observational-memory/operation-registry.d.ts +14 -0
  88. package/dist/processors/observational-memory/operation-registry.d.ts.map +1 -0
  89. package/dist/processors/observational-memory/reflector-agent.d.ts +55 -0
  90. package/dist/processors/observational-memory/reflector-agent.d.ts.map +1 -0
  91. package/dist/processors/observational-memory/thresholds.d.ts +52 -0
  92. package/dist/processors/observational-memory/thresholds.d.ts.map +1 -0
  93. package/dist/processors/observational-memory/token-counter.d.ts +33 -0
  94. package/dist/processors/observational-memory/token-counter.d.ts.map +1 -0
  95. package/dist/processors/observational-memory/types.d.ts +539 -0
  96. package/dist/processors/observational-memory/types.d.ts.map +1 -0
  97. package/dist/token-6GSAFR2W-ABXTQD64.js +61 -0
  98. package/dist/token-6GSAFR2W-ABXTQD64.js.map +1 -0
  99. package/dist/token-6GSAFR2W-TW2P7HCS.cjs +63 -0
  100. package/dist/token-6GSAFR2W-TW2P7HCS.cjs.map +1 -0
  101. package/dist/token-APYSY3BW-2DN6RAUY.js +61 -0
  102. package/dist/token-APYSY3BW-2DN6RAUY.js.map +1 -0
  103. package/dist/token-APYSY3BW-ZQ7TMBY7.cjs +63 -0
  104. package/dist/token-APYSY3BW-ZQ7TMBY7.cjs.map +1 -0
  105. package/dist/token-util-NEHG7TUY-GYFEVMWP.cjs +10 -0
  106. package/dist/token-util-NEHG7TUY-GYFEVMWP.cjs.map +1 -0
  107. package/dist/token-util-NEHG7TUY-XQP3QSPX.js +8 -0
  108. package/dist/token-util-NEHG7TUY-XQP3QSPX.js.map +1 -0
  109. package/dist/token-util-RMHT2CPJ-6TGPE335.cjs +10 -0
  110. package/dist/token-util-RMHT2CPJ-6TGPE335.cjs.map +1 -0
  111. package/dist/token-util-RMHT2CPJ-RJEA3FAN.js +8 -0
  112. package/dist/token-util-RMHT2CPJ-RJEA3FAN.js.map +1 -0
  113. package/dist/tools/working-memory.d.ts +17 -24
  114. package/dist/tools/working-memory.d.ts.map +1 -1
  115. package/package.json +26 -25
  116. package/dist/processors/token-limiter.d.ts +0 -32
  117. package/dist/processors/token-limiter.d.ts.map +0 -1
  118. package/dist/processors/tool-call-filter.d.ts +0 -20
  119. package/dist/processors/tool-call-filter.d.ts.map +0 -1
@@ -0,0 +1,577 @@
1
+ # Observational Memory
2
+
3
+ **Added in:** `@mastra/memory@1.1.0`
4
+
5
+ Observational Memory (OM) is Mastra's memory system for long-context agentic memory. Two background agents — an **Observer** that watches conversations and creates observations, and a **Reflector** that restructures observations by combining related items, reflecting on overarching patterns, and condensing where possible — maintain an observation log that replaces raw message history as it grows.
6
+
7
+ ## Usage
8
+
9
+ ```typescript
10
+ import { Memory } from '@mastra/memory'
11
+ import { Agent } from '@mastra/core/agent'
12
+
13
+ export const agent = new Agent({
14
+ name: 'my-agent',
15
+ instructions: 'You are a helpful assistant.',
16
+ model: 'openai/gpt-5-mini',
17
+ memory: new Memory({
18
+ options: {
19
+ observationalMemory: true,
20
+ },
21
+ }),
22
+ })
23
+ ```
24
+
25
+ ## Configuration
26
+
27
+ The `observationalMemory` option accepts `true`, a configuration object, or `false`. Setting `true` enables OM with `google/gemini-2.5-flash` as the default model. When passing a config object, a `model` must be explicitly set — either at the top level, or on `observation.model` and/or `reflection.model`.
28
+
29
+ **enabled?:** (`boolean`): Enable or disable Observational Memory. When omitted from a config object, defaults to \`true\`. Only \`enabled: false\` explicitly disables it. (Default: `true`)
30
+
31
+ **model?:** (`string | LanguageModel | DynamicModel | ModelWithRetries[]`): Model for both the Observer and Reflector agents. Sets the model for both at once. Cannot be used together with \`observation.model\` or \`reflection.model\` — an error will be thrown if both are set. When using \`observationalMemory: true\`, defaults to \`google/gemini-2.5-flash\`. When passing a config object, this or \`observation.model\`/\`reflection.model\` must be set. Use \`"default"\` to explicitly use the default model (\`google/gemini-2.5-flash\`). (Default: `'google/gemini-2.5-flash' (when using observationalMemory: true)`)
32
+
33
+ **scope?:** (`'resource' | 'thread'`): Memory scope for observations. \`'thread'\` keeps observations per-thread. \`'resource'\` (experimental) shares observations across all threads for a resource, enabling cross-conversation memory. (Default: `'thread'`)
34
+
35
+ **shareTokenBudget?:** (`boolean`): Share the token budget between messages and observations. When enabled, the total budget is \`observation.messageTokens + reflection.observationTokens\`. Messages can use more space when observations are small, and vice versa. This maximizes context usage through flexible allocation. \*\*Note:\*\* \`shareTokenBudget\` is not yet compatible with async buffering. You must set \`observation: { bufferTokens: false }\` when using this option (this is a temporary limitation). (Default: `false`)
36
+
37
+ **observation?:** (`ObservationalMemoryObservationConfig`): Configuration for the observation step. Controls when the Observer agent runs and how it behaves.
38
+
39
+ **reflection?:** (`ObservationalMemoryReflectionConfig`): Configuration for the reflection step. Controls when the Reflector agent runs and how it behaves.
40
+
41
+ ### Observation config
42
+
43
+ **model?:** (`string | LanguageModel | DynamicModel | ModelWithRetries[]`): Model for the Observer agent. Cannot be set if a top-level \`model\` is also provided. If neither this nor the top-level \`model\` is set, falls back to \`reflection.model\`.
44
+
45
+ **instruction?:** (`string`): Custom instruction appended to the Observer's system prompt. Use this to customize what the Observer focuses on, such as domain-specific preferences or priorities.
46
+
47
+ **messageTokens?:** (`number`): Token count of unobserved messages that triggers observation. When unobserved message tokens exceed this threshold, the Observer agent is called. (Default: `30000`)
48
+
49
+ **maxTokensPerBatch?:** (`number`): Maximum tokens per batch when observing multiple threads in resource scope. Threads are chunked into batches of this size and processed in parallel. Lower values mean more parallelism but more API calls. (Default: `10000`)
50
+
51
+ **modelSettings?:** (`ObservationalMemoryModelSettings`): Model settings for the Observer agent. (Default: `{ temperature: 0.3, maxOutputTokens: 100_000 }`)
52
+
53
+ **bufferTokens?:** (`number | false`): Token interval for async background observation buffering. Can be an absolute token count (e.g. \`5000\`) or a fraction of \`messageTokens\` (e.g. \`0.25\` = buffer every 25% of threshold). When set, observations run in the background at this interval, storing results in a buffer. When the main \`messageTokens\` threshold is reached, buffered observations activate instantly without a blocking LLM call. Must resolve to less than \`messageTokens\`. Set to \`false\` to explicitly disable all async buffering (both observation and reflection). (Default: `0.2`)
54
+
55
+ **bufferActivation?:** (`number`): Controls how much of the message window to retain after activation. Accepts a ratio (0-1) or an absolute token count (≥ 1000). For example, \`0.8\` means: activate enough buffers to remove 80% of \`messageTokens\` and leave 20% as active message history. An absolute token count like \`4000\` targets a goal of keeping \~4k message tokens remaining after activation. Higher values remove more message history per activation when using a ratio. Higher values keep more message history when using a token count. (Default: `0.8`)
56
+
57
+ **blockAfter?:** (`number`): Token threshold above which synchronous (blocking) observation is forced. Between \`messageTokens\` and \`blockAfter\`, only async buffering/activation is used. Above \`blockAfter\`, a synchronous observation runs as a last resort, while buffered activation still preserves a minimum remaining context (min(1000, retention floor)). Accepts a multiplier (1 < value < 2, multiplied by \`messageTokens\`) or an absolute token count (≥ 2, must be greater than \`messageTokens\`). Only relevant when \`bufferTokens\` is set. Defaults to \`1.2\` when async buffering is enabled. (Default: `1.2 (when bufferTokens is set)`)
58
+
59
+ ### Token estimate metadata cache
60
+
61
+ OM persists token payload estimates so repeated counting can reuse prior tiktoken work.
62
+
63
+ - Part-level cache: `part.providerMetadata.mastra`.
64
+ - String-content fallback cache: message-level metadata when no parts exist.
65
+ - Cache entries are ignored and recomputed if cache version/tokenizer source does not match.
66
+ - Per-message and per-conversation overhead are always recomputed at runtime and are not cached.
67
+ - `data-*` and `reasoning` parts are skipped and do not receive cache entries.
68
+
69
+ ### Reflection config
70
+
71
+ **model?:** (`string | LanguageModel | DynamicModel | ModelWithRetries[]`): Model for the Reflector agent. Cannot be set if a top-level \`model\` is also provided. If neither this nor the top-level \`model\` is set, falls back to \`observation.model\`.
72
+
73
+ **instruction?:** (`string`): Custom instruction appended to the Reflector's system prompt. Use this to customize how the Reflector consolidates observations, such as prioritizing certain types of information.
74
+
75
+ **observationTokens?:** (`number`): Token count of observations that triggers reflection. When observation tokens exceed this threshold, the Reflector agent is called to condense them. (Default: `40000`)
76
+
77
+ **modelSettings?:** (`ObservationalMemoryModelSettings`): Model settings for the Reflector agent. (Default: `{ temperature: 0, maxOutputTokens: 100_000 }`)
78
+
79
+ **bufferActivation?:** (`number`): Ratio (0-1) controlling when async reflection buffering starts. When observation tokens reach \`observationTokens \* bufferActivation\`, reflection runs in the background. On activation at the full threshold, the buffered reflection replaces the observations it covers, preserving any new observations appended after that range. (Default: `0.5`)
80
+
81
+ **blockAfter?:** (`number`): Token threshold above which synchronous (blocking) reflection is forced. Between \`observationTokens\` and \`blockAfter\`, only async buffering/activation is used. Above \`blockAfter\`, a synchronous reflection runs as a last resort. Accepts a multiplier (1 < value < 2, multiplied by \`observationTokens\`) or an absolute token count (≥ 2, must be greater than \`observationTokens\`). Only relevant when \`bufferActivation\` is set. Defaults to \`1.2\` when async reflection is enabled. (Default: `1.2 (when bufferActivation is set)`)
82
+
83
+ ### Model settings
84
+
85
+ **temperature?:** (`number`): Temperature for generation. Lower values produce more consistent output. (Default: `0.3`)
86
+
87
+ **maxOutputTokens?:** (`number`): Maximum output tokens. Set high to prevent truncation of observations. (Default: `100000`)
88
+
89
+ ## Examples
90
+
91
+ ### Resource scope with custom thresholds (experimental)
92
+
93
+ ```typescript
94
+ import { Memory } from '@mastra/memory'
95
+ import { Agent } from '@mastra/core/agent'
96
+
97
+ export const agent = new Agent({
98
+ name: 'my-agent',
99
+ instructions: 'You are a helpful assistant.',
100
+ model: 'openai/gpt-5-mini',
101
+ memory: new Memory({
102
+ options: {
103
+ observationalMemory: {
104
+ model: 'google/gemini-2.5-flash',
105
+ scope: 'resource',
106
+ observation: {
107
+ messageTokens: 20_000,
108
+ },
109
+ reflection: {
110
+ observationTokens: 60_000,
111
+ },
112
+ },
113
+ },
114
+ }),
115
+ })
116
+ ```
117
+
118
+ ### Shared token budget
119
+
120
+ When `shareTokenBudget` is enabled, the total budget is `observation.messageTokens + reflection.observationTokens` (100k in this example). If observations only use 30k tokens, messages can expand to use up to 70k. If messages are short, observations have more room before triggering reflection.
121
+
122
+ ```typescript
123
+ import { Memory } from '@mastra/memory'
124
+ import { Agent } from '@mastra/core/agent'
125
+
126
+ export const agent = new Agent({
127
+ name: 'my-agent',
128
+ instructions: 'You are a helpful assistant.',
129
+ model: 'openai/gpt-5-mini',
130
+ memory: new Memory({
131
+ options: {
132
+ observationalMemory: {
133
+ shareTokenBudget: true,
134
+ observation: {
135
+ messageTokens: 20_000,
136
+ bufferTokens: false, // required when using shareTokenBudget (temporary limitation)
137
+ },
138
+ reflection: {
139
+ observationTokens: 80_000,
140
+ },
141
+ },
142
+ },
143
+ }),
144
+ })
145
+ ```
146
+
147
+ ### Custom model
148
+
149
+ By passing a `model` in the config, you can use any model from Mastra's model router.
150
+
151
+ ```typescript
152
+ import { Memory } from '@mastra/memory'
153
+ import { Agent } from '@mastra/core/agent'
154
+
155
+ export const agent = new Agent({
156
+ name: 'my-agent',
157
+ instructions: 'You are a helpful assistant.',
158
+ model: 'openai/gpt-5-mini',
159
+ memory: new Memory({
160
+ options: {
161
+ observationalMemory: {
162
+ model: 'openai/gpt-4o-mini',
163
+ },
164
+ },
165
+ }),
166
+ })
167
+ ```
168
+
169
+ ### Different models per agent
170
+
171
+ ```typescript
172
+ import { Memory } from '@mastra/memory'
173
+ import { Agent } from '@mastra/core/agent'
174
+
175
+ export const agent = new Agent({
176
+ name: 'my-agent',
177
+ instructions: 'You are a helpful assistant.',
178
+ model: 'openai/gpt-5-mini',
179
+ memory: new Memory({
180
+ options: {
181
+ observationalMemory: {
182
+ observation: {
183
+ model: 'google/gemini-2.5-flash',
184
+ },
185
+ reflection: {
186
+ model: 'openai/gpt-4o-mini',
187
+ },
188
+ },
189
+ },
190
+ }),
191
+ })
192
+ ```
193
+
194
+ ### Custom instructions
195
+
196
+ Customize what the Observer and Reflector focus on by providing custom instructions:
197
+
198
+ ```typescript
199
+ import { Memory } from '@mastra/memory'
200
+ import { Agent } from '@mastra/core/agent'
201
+
202
+ export const agent = new Agent({
203
+ name: 'health-assistant',
204
+ instructions: 'You are a health and wellness assistant.',
205
+ model: 'openai/gpt-5-mini',
206
+ memory: new Memory({
207
+ options: {
208
+ observationalMemory: {
209
+ model: 'google/gemini-2.5-flash',
210
+ observation: {
211
+ // Focus observations on health-related preferences and goals
212
+ instruction:
213
+ 'Prioritize capturing user health goals, dietary restrictions, exercise preferences, and medical considerations. Avoid capturing general chit-chat.',
214
+ },
215
+ reflection: {
216
+ // Guide reflection to consolidate health patterns
217
+ instruction:
218
+ 'When consolidating, group related health information together. Preserve specific metrics, dates, and medical details.',
219
+ },
220
+ },
221
+ },
222
+ }),
223
+ })
224
+ ```
225
+
226
+ ### Async buffering
227
+
228
+ Async buffering is **enabled by default**. It pre-computes observations in the background as the conversation grows — when the `messageTokens` threshold is reached, buffered observations activate instantly with no blocking LLM call.
229
+
230
+ The lifecycle is: **buffer → activate → remove messages → repeat**. Background Observer calls run at `bufferTokens` intervals, each producing a chunk of observations. At threshold, chunks activate: observations move into the log, raw messages are removed from context. The `blockAfter` threshold forces a synchronous fallback if buffering can't keep up.
231
+
232
+ Default settings:
233
+
234
+ - `observation.bufferTokens: 0.2` — buffer every 20% of `messageTokens` (e.g. every \~6k tokens with a 30k threshold)
235
+ - `observation.bufferActivation: 0.8` — on activation, remove enough messages to keep only 20% of the threshold remaining
236
+ - Buffered observations include continuation hints (`suggestedResponse`, `currentTask`) that survive activation to maintain conversational continuity
237
+ - `reflection.bufferActivation: 0.5` — start background reflection at 50% of observation threshold
238
+
239
+ To customize:
240
+
241
+ ```typescript
242
+ import { Memory } from '@mastra/memory'
243
+ import { Agent } from '@mastra/core/agent'
244
+
245
+ export const agent = new Agent({
246
+ name: 'my-agent',
247
+ instructions: 'You are a helpful assistant.',
248
+ model: 'openai/gpt-5-mini',
249
+ memory: new Memory({
250
+ options: {
251
+ observationalMemory: {
252
+ model: 'google/gemini-2.5-flash',
253
+ observation: {
254
+ messageTokens: 30_000,
255
+ // Buffer every 5k tokens (runs in background)
256
+ bufferTokens: 5_000,
257
+ // Activate to retain 30% of threshold
258
+ bufferActivation: 0.7,
259
+ // Force synchronous observation at 1.5x threshold
260
+ blockAfter: 1.5,
261
+ },
262
+ reflection: {
263
+ observationTokens: 60_000,
264
+ // Start background reflection at 50% of threshold
265
+ bufferActivation: 0.5,
266
+ // Force synchronous reflection at 1.2x threshold
267
+ blockAfter: 1.2,
268
+ },
269
+ },
270
+ },
271
+ }),
272
+ })
273
+ ```
274
+
275
+ To disable async buffering entirely:
276
+
277
+ ```typescript
278
+ observationalMemory: {
279
+ model: "google/gemini-2.5-flash",
280
+ observation: {
281
+ bufferTokens: false,
282
+ },
283
+ }
284
+ ```
285
+
286
+ Setting `bufferTokens: false` disables both observation and reflection async buffering. Observations and reflections will run synchronously when their thresholds are reached.
287
+
288
+ > **Note:** Async buffering is not supported with `scope: 'resource'` and is automatically disabled in resource scope.
289
+
290
+ ## Streaming data parts
291
+
292
+ Observational Memory emits typed data parts during agent execution that clients can use for real-time UI feedback. These are streamed alongside the agent's response.
293
+
294
+ ### `data-om-status`
295
+
296
+ Emitted once per agent loop step, before model generation. Provides a snapshot of the current memory state, including token usage for both context windows and the state of any async buffered content.
297
+
298
+ ```typescript
299
+ interface DataOmStatusPart {
300
+ type: 'data-om-status'
301
+ data: {
302
+ windows: {
303
+ active: {
304
+ /** Unobserved message tokens and the threshold that triggers observation */
305
+ messages: { tokens: number; threshold: number }
306
+ /** Observation tokens and the threshold that triggers reflection */
307
+ observations: { tokens: number; threshold: number }
308
+ }
309
+ buffered: {
310
+ observations: {
311
+ /** Number of buffered chunks staged for activation */
312
+ chunks: number
313
+ /** Total message tokens across all buffered chunks */
314
+ messageTokens: number
315
+ /** Projected message tokens that would be removed if activation happened now (based on bufferActivation ratio and chunk boundaries) */
316
+ projectedMessageRemoval: number
317
+ /** Observation tokens that will be added on activation */
318
+ observationTokens: number
319
+ /** idle: no buffering in progress. running: background observer is working. complete: chunks are ready for activation. */
320
+ status: 'idle' | 'running' | 'complete'
321
+ }
322
+ reflection: {
323
+ /** Observation tokens that were fed into the reflector (pre-compression size) */
324
+ inputObservationTokens: number
325
+ /** Observation tokens the reflection will produce on activation (post-compression size) */
326
+ observationTokens: number
327
+ /** idle: no reflection buffered. running: background reflector is working. complete: reflection is ready for activation. */
328
+ status: 'idle' | 'running' | 'complete'
329
+ }
330
+ }
331
+ }
332
+ recordId: string
333
+ threadId: string
334
+ stepNumber: number
335
+ /** Increments each time the Reflector creates a new generation */
336
+ generationCount: number
337
+ }
338
+ }
339
+ ```
340
+
341
+ `buffered.reflection.inputObservationTokens` is the size of the observations that were sent to the Reflector. `buffered.reflection.observationTokens` is the compressed result — the size of what will replace those observations when the reflection activates. A client can use these two values to show a compression ratio.
342
+
343
+ Clients can derive percentages and post-activation estimates from the raw values:
344
+
345
+ ```typescript
346
+ // Message window usage %
347
+ const msgPercent = status.windows.active.messages.tokens / status.windows.active.messages.threshold
348
+
349
+ // Observation window usage %
350
+ const obsPercent =
351
+ status.windows.active.observations.tokens / status.windows.active.observations.threshold
352
+
353
+ // Projected message tokens after buffered observations activate
354
+ // Uses projectedMessageRemoval which accounts for bufferActivation ratio and chunk boundaries
355
+ const postActivation =
356
+ status.windows.active.messages.tokens -
357
+ status.windows.buffered.observations.projectedMessageRemoval
358
+
359
+ // Reflection compression ratio (when buffered reflection exists)
360
+ const { inputObservationTokens, observationTokens } = status.windows.buffered.reflection
361
+ if (inputObservationTokens > 0) {
362
+ const compressionRatio = observationTokens / inputObservationTokens
363
+ }
364
+ ```
365
+
366
+ ### `data-om-observation-start`
367
+
368
+ Emitted when the Observer or Reflector agent begins processing.
369
+
370
+ **cycleId:** (`string`): Unique ID for this cycle — shared between start/end/failed markers.
371
+
372
+ **operationType:** (`'observation' | 'reflection'`): Whether this is an observation or reflection operation.
373
+
374
+ **startedAt:** (`string`): ISO timestamp when processing started.
375
+
376
+ **tokensToObserve:** (`number`): Message tokens (input) being processed in this batch.
377
+
378
+ **recordId:** (`string`): The OM record ID.
379
+
380
+ **threadId:** (`string`): This thread's ID.
381
+
382
+ **threadIds:** (`string[]`): All thread IDs in this batch (for resource-scoped).
383
+
384
+ **config:** (`ObservationMarkerConfig`): Snapshot of \`messageTokens\`, \`observationTokens\`, and \`scope\` at observation time.
385
+
386
+ ### `data-om-observation-end`
387
+
388
+ Emitted when observation or reflection completes successfully.
389
+
390
+ **cycleId:** (`string`): Matches the corresponding \`start\` marker.
391
+
392
+ **operationType:** (`'observation' | 'reflection'`): Type of operation that completed.
393
+
394
+ **completedAt:** (`string`): ISO timestamp when processing completed.
395
+
396
+ **durationMs:** (`number`): Duration in milliseconds.
397
+
398
+ **tokensObserved:** (`number`): Message tokens (input) that were processed.
399
+
400
+ **observationTokens:** (`number`): Resulting observation tokens (output) after the Observer compressed them.
401
+
402
+ **observations?:** (`string`): The generated observations text.
403
+
404
+ **currentTask?:** (`string`): Current task extracted by the Observer.
405
+
406
+ **suggestedResponse?:** (`string`): Suggested response extracted by the Observer.
407
+
408
+ **recordId:** (`string`): The OM record ID.
409
+
410
+ **threadId:** (`string`): This thread's ID.
411
+
412
+ ### `data-om-observation-failed`
413
+
414
+ Emitted when observation or reflection fails. The system falls back to synchronous processing.
415
+
416
+ **cycleId:** (`string`): Matches the corresponding \`start\` marker.
417
+
418
+ **operationType:** (`'observation' | 'reflection'`): Type of operation that failed.
419
+
420
+ **failedAt:** (`string`): ISO timestamp when the failure occurred.
421
+
422
+ **durationMs:** (`number`): Duration until failure in milliseconds.
423
+
424
+ **tokensAttempted:** (`number`): Message tokens (input) that were attempted.
425
+
426
+ **error:** (`string`): Error message.
427
+
428
+ **observations?:** (`string`): Any partial content available for display.
429
+
430
+ **recordId:** (`string`): The OM record ID.
431
+
432
+ **threadId:** (`string`): This thread's ID.
433
+
434
+ ### `data-om-buffering-start`
435
+
436
+ Emitted when async buffering begins in the background. Buffering pre-computes observations or reflections before the main threshold is reached.
437
+
438
+ **cycleId:** (`string`): Unique ID for this buffering cycle.
439
+
440
+ **operationType:** (`'observation' | 'reflection'`): Type of operation being buffered.
441
+
442
+ **startedAt:** (`string`): ISO timestamp when buffering started.
443
+
444
+ **tokensToBuffer:** (`number`): Message tokens (input) being buffered in this cycle.
445
+
446
+ **recordId:** (`string`): The OM record ID.
447
+
448
+ **threadId:** (`string`): This thread's ID.
449
+
450
+ **threadIds:** (`string[]`): All thread IDs being buffered (for resource-scoped).
451
+
452
+ **config:** (`ObservationMarkerConfig`): Snapshot of config at buffering time.
453
+
454
+ ### `data-om-buffering-end`
455
+
456
+ Emitted when async buffering completes. The content is stored but not yet activated in the main context.
457
+
458
+ **cycleId:** (`string`): Matches the corresponding \`buffering-start\` marker.
459
+
460
+ **operationType:** (`'observation' | 'reflection'`): Type of operation that was buffered.
461
+
462
+ **completedAt:** (`string`): ISO timestamp when buffering completed.
463
+
464
+ **durationMs:** (`number`): Duration in milliseconds.
465
+
466
+ **tokensBuffered:** (`number`): Message tokens (input) that were buffered.
467
+
468
+ **bufferedTokens:** (`number`): Observation tokens (output) after the Observer compressed them.
469
+
470
+ **observations?:** (`string`): The buffered content.
471
+
472
+ **recordId:** (`string`): The OM record ID.
473
+
474
+ **threadId:** (`string`): This thread's ID.
475
+
476
+ ### `data-om-buffering-failed`
477
+
478
+ Emitted when async buffering fails. The system falls back to synchronous processing when the threshold is reached.
479
+
480
+ **cycleId:** (`string`): Matches the corresponding \`buffering-start\` marker.
481
+
482
+ **operationType:** (`'observation' | 'reflection'`): Type of operation that failed.
483
+
484
+ **failedAt:** (`string`): ISO timestamp when the failure occurred.
485
+
486
+ **durationMs:** (`number`): Duration until failure in milliseconds.
487
+
488
+ **tokensAttempted:** (`number`): Message tokens (input) that were attempted to buffer.
489
+
490
+ **error:** (`string`): Error message.
491
+
492
+ **observations?:** (`string`): Any partial content.
493
+
494
+ **recordId:** (`string`): The OM record ID.
495
+
496
+ **threadId:** (`string`): This thread's ID.
497
+
498
+ ### `data-om-activation`
499
+
500
+ Emitted when buffered observations or reflections are activated (moved into the active context window). This is an instant operation — no LLM call is involved.
501
+
502
+ **cycleId:** (`string`): Unique ID for this activation event.
503
+
504
+ **operationType:** (`'observation' | 'reflection'`): Type of content activated.
505
+
506
+ **activatedAt:** (`string`): ISO timestamp when activation occurred.
507
+
508
+ **chunksActivated:** (`number`): Number of buffered chunks activated.
509
+
510
+ **tokensActivated:** (`number`): Message tokens (input) from activated chunks. For observation activation, these are removed from the message window. For reflection activation, this is the observation tokens that were compressed.
511
+
512
+ **observationTokens:** (`number`): Resulting observation tokens after activation.
513
+
514
+ **messagesActivated:** (`number`): Number of messages that were observed via activation.
515
+
516
+ **generationCount:** (`number`): Current reflection generation count.
517
+
518
+ **observations?:** (`string`): The activated observations text.
519
+
520
+ **recordId:** (`string`): The OM record ID.
521
+
522
+ **threadId:** (`string`): This thread's ID.
523
+
524
+ **config:** (`ObservationMarkerConfig`): Snapshot of config at activation time.
525
+
526
+ ## Standalone usage
527
+
528
+ Most users should use the `Memory` class above. Using `ObservationalMemory` directly is mainly useful for benchmarking, experimentation, or when you need to control processor ordering with other processors (like [guardrails](https://mastra.ai/docs/agents/guardrails)).
529
+
530
+ ```typescript
531
+ import { ObservationalMemory } from '@mastra/memory/processors'
532
+ import { Agent } from '@mastra/core/agent'
533
+ import { LibSQLStore } from '@mastra/libsql'
534
+
535
+ const storage = new LibSQLStore({
536
+ id: 'my-storage',
537
+ url: 'file:./memory.db',
538
+ })
539
+
540
+ const om = new ObservationalMemory({
541
+ storage: storage.stores.memory,
542
+ model: 'google/gemini-2.5-flash',
543
+ scope: 'resource',
544
+ observation: {
545
+ messageTokens: 20_000,
546
+ },
547
+ reflection: {
548
+ observationTokens: 60_000,
549
+ },
550
+ })
551
+
552
+ export const agent = new Agent({
553
+ name: 'my-agent',
554
+ instructions: 'You are a helpful assistant.',
555
+ model: 'openai/gpt-5-mini',
556
+ inputProcessors: [om],
557
+ outputProcessors: [om],
558
+ })
559
+ ```
560
+
561
+ ### Standalone config
562
+
563
+ The standalone `ObservationalMemory` class accepts all the same options as the `observationalMemory` config object above, plus the following:
564
+
565
+ **storage:** (`MemoryStorage`): Storage adapter for persisting observations. Must be a MemoryStorage instance (from \`MastraStorage.stores.memory\`).
566
+
567
+ **onDebugEvent?:** (`(event: ObservationDebugEvent) => void`): Debug callback for observation events. Called whenever observation-related events occur. Useful for debugging and understanding the observation flow.
568
+
569
+ **obscureThreadIds?:** (`boolean`): When enabled, thread IDs are hashed before being included in observation context. This prevents the LLM from recognizing patterns in thread identifiers. Automatically enabled when using resource scope through the Memory class. (Default: `false`)
570
+
571
+ ### Related
572
+
573
+ - [Observational Memory](https://mastra.ai/docs/memory/observational-memory)
574
+ - [Memory Overview](https://mastra.ai/docs/memory/overview)
575
+ - [Memory Class](https://mastra.ai/reference/memory/memory-class)
576
+ - [Memory Processors](https://mastra.ai/docs/memory/memory-processors)
577
+ - [Processors](https://mastra.ai/docs/agents/processors)
@@ -0,0 +1,115 @@
1
+ # TokenLimiterProcessor
2
+
3
+ The `TokenLimiterProcessor` limits the number of tokens in messages. It can be used as both an input and output processor:
4
+
5
+ - **Input processor**: Filters historical messages to fit within the context window, prioritizing recent messages
6
+ - **Output processor**: Limits generated response tokens via streaming or non-streaming with configurable strategies for handling exceeded limits
7
+
8
+ ## Usage example
9
+
10
+ ```typescript
11
+ import { TokenLimiterProcessor } from '@mastra/core/processors'
12
+
13
+ const processor = new TokenLimiterProcessor({
14
+ limit: 1000,
15
+ strategy: 'truncate',
16
+ countMode: 'cumulative',
17
+ })
18
+ ```
19
+
20
+ ## Constructor parameters
21
+
22
+ **options:** (`number | Options`): Either a simple number for token limit, or configuration options object
23
+
24
+ ### Options
25
+
26
+ **limit:** (`number`): Maximum number of tokens to allow in the response
27
+
28
+ **encoding?:** (`TiktokenBPE`): Optional encoding to use. Defaults to o200k\_base which is used by gpt-5.1
29
+
30
+ **strategy?:** (`'truncate' | 'abort'`): Strategy when token limit is reached: 'truncate' stops emitting chunks, 'abort' calls abort() to stop the stream
31
+
32
+ **countMode?:** (`'cumulative' | 'part'`): Whether to count tokens from the beginning of the stream or just the current part: 'cumulative' counts all tokens from start, 'part' only counts tokens in current part
33
+
34
+ ## Returns
35
+
36
+ **id:** (`string`): Processor identifier set to 'token-limiter'
37
+
38
+ **name?:** (`string`): Optional processor display name
39
+
40
+ **processInput:** (`(args: { messages: MastraDBMessage[]; abort: (reason?: string) => never }) => Promise<MastraDBMessage[]>`): Filters input messages to fit within token limit, prioritizing recent messages while preserving system messages
41
+
42
+ **processOutputStream:** (`(args: { part: ChunkType; streamParts: ChunkType[]; state: Record<string, any>; abort: (reason?: string) => never }) => Promise<ChunkType | null>`): Processes streaming output parts to limit token count during streaming
43
+
44
+ **processOutputResult:** (`(args: { messages: MastraDBMessage[]; abort: (reason?: string) => never }) => Promise<MastraDBMessage[]>`): Processes final output results to limit token count in non-streaming scenarios
45
+
46
+ **getMaxTokens:** (`() => number`): Get the maximum token limit
47
+
48
+ ## Error behavior
49
+
50
+ When used as an input processor, `TokenLimiterProcessor` throws a `TripWire` error in the following cases:
51
+
52
+ - **Empty messages**: If there are no messages to process, a TripWire is thrown because you cannot send an LLM request with no messages.
53
+ - **System messages exceed limit**: If system messages alone exceed the token limit, a TripWire is thrown because you cannot send an LLM request with only system messages and no user/assistant messages.
54
+
55
+ ```typescript
56
+ import { TripWire } from '@mastra/core/agent'
57
+
58
+ try {
59
+ await agent.generate('Hello')
60
+ } catch (error) {
61
+ if (error instanceof TripWire) {
62
+ console.log('Token limit error:', error.message)
63
+ }
64
+ }
65
+ ```
66
+
67
+ ## Extended usage example
68
+
69
+ ### As an input processor (limit context window)
70
+
71
+ Use `inputProcessors` to limit historical messages sent to the model, which helps stay within context window limits:
72
+
73
+ ```typescript
74
+ import { Agent } from '@mastra/core/agent'
75
+ import { Memory } from '@mastra/memory'
76
+ import { TokenLimiterProcessor } from '@mastra/core/processors'
77
+
78
+ export const agent = new Agent({
79
+ name: 'context-limited-agent',
80
+ instructions: 'You are a helpful assistant',
81
+ model: 'openai/gpt-4o',
82
+ memory: new Memory({
83
+ /* ... */
84
+ }),
85
+ inputProcessors: [
86
+ new TokenLimiterProcessor({ limit: 4000 }), // Limits historical messages to ~4000 tokens
87
+ ],
88
+ })
89
+ ```
90
+
91
+ ### As an output processor (limit response length)
92
+
93
+ Use `outputProcessors` to limit the length of generated responses:
94
+
95
+ ```typescript
96
+ import { Agent } from '@mastra/core/agent'
97
+ import { TokenLimiterProcessor } from '@mastra/core/processors'
98
+
99
+ export const agent = new Agent({
100
+ name: 'response-limited-agent',
101
+ instructions: 'You are a helpful assistant',
102
+ model: 'openai/gpt-4o',
103
+ outputProcessors: [
104
+ new TokenLimiterProcessor({
105
+ limit: 1000,
106
+ strategy: 'truncate',
107
+ countMode: 'cumulative',
108
+ }),
109
+ ],
110
+ })
111
+ ```
112
+
113
+ ## Related
114
+
115
+ - [Guardrails](https://mastra.ai/docs/agents/guardrails)