@mastra/mongodb 1.21.0 → 1.22.0-alpha.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -3,7 +3,7 @@ name: mastra-mongodb
3
3
  description: Documentation for @mastra/mongodb. Use when working with @mastra/mongodb APIs, configuration, or implementation.
4
4
  metadata:
5
5
  package: "@mastra/mongodb"
6
- version: "1.21.0"
6
+ version: "1.22.0-alpha.1"
7
7
  ---
8
8
 
9
9
  ## When to use
@@ -1,5 +1,5 @@
1
1
  {
2
- "version": "1.21.0",
2
+ "version": "1.22.0-alpha.1",
3
3
  "package": "@mastra/mongodb",
4
4
  "exports": {},
5
5
  "modules": {}
@@ -20,7 +20,7 @@ After getting a response from the LLM, all new messages (user, assistant, and to
20
20
 
21
21
  ## Quickstart
22
22
 
23
- Semantic recall is disabled by default. To enable it, set `semanticRecall: true` in `options` and provide a `vector` store and `embedder`:
23
+ Semantic recall is disabled by default. To enable it, set `semanticRecall: true` in `options` and provide a `vector` store and an `embedder`. A store that generates embeddings itself, such as MongoDB with Automated Embedding, works without one. See [Server-side embeddings](#server-side-embeddings).
24
24
 
25
25
  **LibSQL**:
26
26
 
@@ -242,6 +242,40 @@ const options = {
242
242
  }
243
243
  ```
244
244
 
245
+ ## Server-side embeddings
246
+
247
+ Some vector stores generate the embeddings themselves. With one of those, semantic recall runs without an `embedder`: the database embeds each message on write and the search string on arrival. The embedding provider stays out of your application entirely.
248
+
249
+ MongoDB does this through [Automated Embedding](https://mastra.ai/reference/vectors/mongodb): name a Voyage AI model on the vector store and leave the `embedder` out.
250
+
251
+ ```typescript
252
+ import { Memory } from '@mastra/memory'
253
+ import { MongoDBStore, MongoDBVector } from '@mastra/mongodb'
254
+
255
+ const memory = new Memory({
256
+ storage: new MongoDBStore({
257
+ id: 'agent-storage',
258
+ uri: process.env.MONGODB_URI,
259
+ dbName: process.env.MONGODB_DB_NAME,
260
+ }),
261
+ vector: new MongoDBVector({
262
+ id: 'agent-vector',
263
+ uri: process.env.MONGODB_URI,
264
+ dbName: process.env.MONGODB_DB_NAME,
265
+ autoEmbed: { model: 'voyage-4' },
266
+ }),
267
+ options: {
268
+ semanticRecall: { topK: 3, messageRange: 2 },
269
+ },
270
+ })
271
+ ```
272
+
273
+ Messages are stored as text and MongoDB embeds them, so the model choice and the vector size belong to the index rather than to your application. Recall works the same way through an agent and through [`recall()`](#using-the-recall-method).
274
+
275
+ MongoDB embeds newly written messages in the background. A message saved moments ago may not be recallable immediately.
276
+
277
+ Every other store expects your application to supply the vectors, and keeps requiring an `embedder`. Enabling semantic recall with neither an `embedder` nor a store that embeds server-side raises an error naming both options.
278
+
245
279
  ## Embedder configuration
246
280
 
247
281
  Semantic recall relies on an [embedding model](https://mastra.ai/reference/memory/memory-class) to convert messages into embeddings. Mastra supports embedding models through the model router using `provider/model` strings, or you can use any [embedding model](https://sdk.vercel.ai/docs/ai-sdk-core/embeddings) compatible with the AI SDK.
@@ -194,6 +194,7 @@ Each provider page includes installation instructions, configuration parameters,
194
194
 
195
195
  - [Aurora DSQL](https://mastra.ai/integrations/databases/aurora-dsql)
196
196
  - [ClickHouse](https://mastra.ai/integrations/databases/clickhouse)
197
+ - [ClickHouse Managed Postgres](https://mastra.ai/integrations/databases/clickhouse-managed-postgres)
197
198
  - [Cloudflare D1](https://mastra.ai/integrations/databases/cloudflare-d1)
198
199
  - [Cloudflare KV](https://mastra.ai/integrations/databases/cloudflare-kv)
199
200
  - [Convex](https://mastra.ai/integrations/databases/convex)
package/dist/index.cjs CHANGED
@@ -9,7 +9,7 @@ let _mastra_core_agent = require("@mastra/core/agent");
9
9
  let _mastra_core_evals = require("@mastra/core/evals");
10
10
  let _mastra_core_storage_domains_skills = require("@mastra/core/storage/domains/skills");
11
11
  //#region package.json
12
- var version = "1.21.0";
12
+ var version = "1.22.0-alpha.1";
13
13
  //#endregion
14
14
  //#region src/vector/filter.ts
15
15
  /**
@@ -85,11 +85,22 @@ var MongoDBFilterTranslator = class extends _mastra_core_vector_filter.BaseFilte
85
85
  function describeEmbedding(config) {
86
86
  return config ? `autoEmbed (path "${config.path}", model "${config.model}")` : "client-side vectors";
87
87
  }
88
+ /**
89
+ * Whether Atlas refused a `$vectorSearch` because the index is still in its initial sync. The
90
+ * server reports it as a generic `UnknownError`, so the state named in the message is the only
91
+ * thing that tells it apart from other failures with the same code.
92
+ */
93
+ function isInitialSyncError(error) {
94
+ const message = error?.message;
95
+ return typeof message === "string" && message.includes("while in state INITIAL_SYNC");
96
+ }
88
97
  var MongoDBVector = class MongoDBVector extends _mastra_core_vector.MastraVector {
89
98
  client;
90
99
  db;
91
100
  collections;
92
101
  embeddingFieldName;
102
+ /** Automated Embedding defaults applied to indexes created without their own config. */
103
+ defaultAutoEmbed;
93
104
  metadataFieldName = "metadata";
94
105
  documentFieldName = "document";
95
106
  /**
@@ -125,6 +136,19 @@ var MongoDBVector = class MongoDBVector extends _mastra_core_vector.MastraVector
125
136
  */
126
137
  static REGISTRY_COLLECTION = "__mastra_vector_indexes__";
127
138
  /**
139
+ * Waits between attempts while a queried index is in INITIAL_SYNC, about six seconds in all.
140
+ * On a live Atlas cluster the window lasted around two seconds for both a regular and an
141
+ * autoEmbed index; the budget leaves room for a slower build without holding a failing query
142
+ * for long.
143
+ */
144
+ static INITIAL_SYNC_RETRY_DELAYS_MS = [
145
+ 250,
146
+ 500,
147
+ 1e3,
148
+ 2e3,
149
+ 2e3
150
+ ];
151
+ /**
128
152
  * MongoDB query operators supported inside `$vectorSearch.filter`. Intentionally
129
153
  * conservative: filters using any operator outside this set fall back to the
130
154
  * `$match` pre-filter, which supports the full query language. Widening this set
@@ -147,7 +171,7 @@ var MongoDBVector = class MongoDBVector extends _mastra_core_vector.MastraVector
147
171
  euclidean: "euclidean",
148
172
  dotproduct: "dotProduct"
149
173
  };
150
- constructor({ id, uri, dbName, options, embeddingFieldPath }) {
174
+ constructor({ id, uri, dbName, options, embeddingFieldPath, autoEmbed }) {
151
175
  super({ id });
152
176
  if (!uri) throw new Error("MongoDBVector requires a connection string. Provide \"uri\" in the constructor options.");
153
177
  const client = new mongodb.MongoClient(uri, {
@@ -161,6 +185,11 @@ var MongoDBVector = class MongoDBVector extends _mastra_core_vector.MastraVector
161
185
  this.db = this.client.db(dbName);
162
186
  this.collections = /* @__PURE__ */ new Map();
163
187
  this.embeddingFieldName = embeddingFieldPath ?? "embedding";
188
+ this.defaultAutoEmbed = autoEmbed;
189
+ }
190
+ /** True once the store is configured with Automated Embedding defaults. */
191
+ get isSelfEmbedding() {
192
+ return this.defaultAutoEmbed !== void 0;
164
193
  }
165
194
  /**
166
195
  * Resolve the collection + vector-index names for an index, honoring BYO overrides.
@@ -442,7 +471,8 @@ var MongoDBVector = class MongoDBVector extends _mastra_core_vector.MastraVector
442
471
  * "index not found" or "index not ready" errors on subsequent operations.
443
472
  */
444
473
  async createIndex(params) {
445
- const { indexName, dimension, metric = "cosine", filterFields, collectionName, searchIndexName, allowWrites, autoEmbed } = params;
474
+ const { indexName, dimension, metric = "cosine", filterFields, collectionName, searchIndexName, allowWrites, autoEmbed: autoEmbedParam } = params;
475
+ const autoEmbed = autoEmbedParam ?? (dimension === void 0 ? this.defaultAutoEmbed : void 0);
446
476
  let mongoMetric;
447
477
  try {
448
478
  if (autoEmbed) {
@@ -1030,7 +1060,7 @@ var MongoDBVector = class MongoDBVector extends _mastra_core_vector.MastraVector
1030
1060
  vectorSearch.filter = pushed.length > 1 ? { $and: pushed } : pushed[0];
1031
1061
  }
1032
1062
  const pipeline = [{ $vectorSearch: vectorSearch }, ...this.buildProjection(metadataMode, includeVector, "vectorSearchScore", autoEmbed?.path)];
1033
- return (await collection.aggregate(pipeline).toArray()).map((result) => ({
1063
+ return (await this.aggregateVectorSearch(collection, pipeline)).map((result) => ({
1034
1064
  id: this.idToString(result._id),
1035
1065
  score: result.score,
1036
1066
  metadata: result.metadata,
@@ -1047,6 +1077,24 @@ var MongoDBVector = class MongoDBVector extends _mastra_core_vector.MastraVector
1047
1077
  }
1048
1078
  }
1049
1079
  /**
1080
+ * Runs a `$vectorSearch` pipeline, retrying while the index is in INITIAL_SYNC.
1081
+ *
1082
+ * While Atlas first builds a vector search index it usually answers with an empty result, but
1083
+ * for a few seconds it fails with "cannot query vector index ... while in state INITIAL_SYNC"
1084
+ * instead. A caller that creates an index and queries it straight away, as `Memory` does on a
1085
+ * fresh database, can land in that window. Any other error, and an INITIAL_SYNC that outlasts
1086
+ * the retry budget, is thrown unchanged.
1087
+ */
1088
+ async aggregateVectorSearch(collection, pipeline) {
1089
+ for (const delayMs of MongoDBVector.INITIAL_SYNC_RETRY_DELAYS_MS) try {
1090
+ return await collection.aggregate(pipeline).toArray();
1091
+ } catch (error) {
1092
+ if (!isInitialSyncError(error)) throw error;
1093
+ await new Promise((resolve) => setTimeout(resolve, delayMs));
1094
+ }
1095
+ return collection.aggregate(pipeline).toArray();
1096
+ }
1097
+ /**
1050
1098
  * Lists LOGICAL Mastra index names, not physical collection names.
1051
1099
  *
1052
1100
  * Returns the union of:
@@ -6701,8 +6749,11 @@ var MemoryStorageMongoDB = class MemoryStorageMongoDB extends _mastra_core_stora
6701
6749
  if (!target) continue;
6702
6750
  const prevMessages = await collection.find({
6703
6751
  thread_id: target.threadId,
6704
- createdAt: { $lte: target.createdAt },
6705
- ...resourceFilter
6752
+ ...resourceFilter,
6753
+ $or: [{ createdAt: { $lt: target.createdAt } }, {
6754
+ createdAt: target.createdAt,
6755
+ id: { $lte: id }
6756
+ }]
6706
6757
  }).sort({
6707
6758
  createdAt: -1,
6708
6759
  id: -1
@@ -6711,8 +6762,11 @@ var MemoryStorageMongoDB = class MemoryStorageMongoDB extends _mastra_core_stora
6711
6762
  if (withNextMessages > 0) {
6712
6763
  const nextMessages = await collection.find({
6713
6764
  thread_id: target.threadId,
6714
- createdAt: { $gt: target.createdAt },
6715
- ...resourceFilter
6765
+ ...resourceFilter,
6766
+ $or: [{ createdAt: { $gt: target.createdAt } }, {
6767
+ createdAt: target.createdAt,
6768
+ id: { $gt: id }
6769
+ }]
6716
6770
  }).sort({
6717
6771
  createdAt: 1,
6718
6772
  id: 1
@@ -9929,7 +9983,7 @@ var SchedulesMongoDB = class SchedulesMongoDB extends _mastra_core_storage.Sched
9929
9983
  if (!result) throw new Error(`Schedule ${id} not found`);
9930
9984
  return docToSchedule(result);
9931
9985
  }
9932
- async updateScheduleNextFire(id, expectedNextFireAt, newNextFireAt, lastFireAt, lastRunId) {
9986
+ async updateScheduleNextFire(id, expectedNextFireAt, newNextFireAt, lastFireAt, lastRunId, newStatus) {
9933
9987
  return (await (await this.getSchedulesCollection()).updateOne({
9934
9988
  id,
9935
9989
  next_fire_at: expectedNextFireAt,
@@ -9938,7 +9992,8 @@ var SchedulesMongoDB = class SchedulesMongoDB extends _mastra_core_storage.Sched
9938
9992
  next_fire_at: newNextFireAt,
9939
9993
  last_fire_at: lastFireAt,
9940
9994
  last_run_id: lastRunId,
9941
- updated_at: Date.now()
9995
+ updated_at: Date.now(),
9996
+ ...newStatus ? { status: newStatus } : {}
9942
9997
  } })).matchedCount > 0;
9943
9998
  }
9944
9999
  async deleteSchedule(id) {