@sentry/junior-memory 0.133.0 → 0.135.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -65,7 +65,17 @@ exported types, tools, and tests are authoritative.
65
65
  - Search combines independently ranked vector and PostgreSQL full-text matches
66
66
  with reciprocal rank fusion; provider-specific raw scores are never added
67
67
  together.
68
- - Automatic recall retrieves a broad candidate window, then uses the
68
+ - Both retrieval legs always run in parallel as bounded top-k probes. Each leg
69
+ fetches at least the caller's requested limit (and never more than the store
70
+ limit ceiling). Recall keeps a smaller overfetch window than explicit search
71
+ and slightly prefers lexical ranks so exact tokens survive soft semantic
72
+ neighbors. Vector recall also applies the cosine distance cutoff in SQL, and
73
+ embeddings use an HNSW cosine index (`vector_cosine_ops`).
74
+ - Automatic recall also runs personal-scope-only vector and lexical probes so
75
+ older actor preferences are not buried when newer workspace conversation
76
+ memories fill the shared lexical recency window with common tokens. On RRF
77
+ score ties, personal-scope matches rank ahead of conversation matches.
78
+ - Automatic recall retrieves a bounded candidate window, then uses the
69
79
  memory-owned relevance model to admit at most five directly useful memories.
70
80
  An empty result contributes no filler prompt text.
71
81
  - Every completed automatic recall attempt emits an invisible, namespaced
@@ -80,8 +90,12 @@ exported types, tools, and tests are authoritative.
80
90
 
81
91
  - `AI_MEMORY_MODEL` or `memoryPlugin({ modelId })` selects the structured
82
92
  review model.
83
- - `MEMORY_RECALL_MAX_VECTOR_DISTANCE` or
84
- `recallMaxVectorDistance` configures the vector candidate threshold.
93
+ - `memoryPlugin({ disableRecall: true })` disables automatic prompt recall.
94
+ - `memoryPlugin({ disableExtraction: true })` disables passive session
95
+ extraction. The two flags are independent and do not disable explicit memory
96
+ tools.
97
+ - Automatic recall uses a fixed cosine distance cutoff of `0.45` (for
98
+ `text-embedding-3-small`). Explicit search does not apply that cutoff.
85
99
  - Generate schema changes with `pnpm --filter @sentry/junior-memory db:generate`.
86
100
 
87
101
  Follow `../../policies/data-redaction.md`, `../../policies/security.md`, and the
package/dist/index.js CHANGED
@@ -167,6 +167,9 @@ var juniorMemoryEmbeddings = pgTable(
167
167
  table.dimensions,
168
168
  table.metric
169
169
  ),
170
+ // Cosine ANN for vector recall/search. Ops must match cosineDistance (<=>).
171
+ // Keep this unfiltered so planners can use HNSW before scope/status joins.
172
+ index("junior_memory_embeddings_embedding_hnsw_idx").using("hnsw", table.embedding.op("vector_cosine_ops")).with({ m: 16, ef_construction: 64 }),
170
173
  check(
171
174
  "junior_memory_embeddings_metric_check",
172
175
  sql`${table.metric} IN ('cosine')`
@@ -181,11 +184,12 @@ var juniorMemoryEmbeddings = pgTable(
181
184
  // src/ranking.ts
182
185
  var RECIPROCAL_RANK_FUSION_K = 60;
183
186
  var ONE_DAY_MS = 24 * 60 * 60 * 1e3;
184
- function reciprocalRank(rank) {
185
- return 1 / (RECIPROCAL_RANK_FUSION_K + rank);
187
+ var DEFAULT_RRF_WEIGHT = 1;
188
+ function reciprocalRank(rank, weight) {
189
+ return weight / (RECIPROCAL_RANK_FUSION_K + rank);
186
190
  }
187
- function matchScore(match) {
188
- return (match.vector ? reciprocalRank(match.vector.rank) : 0) + (match.lexical ? reciprocalRank(match.lexical.rank) : 0);
191
+ function matchScore(match, weights) {
192
+ return (match.vector ? reciprocalRank(match.vector.rank, weights.vectorWeight) : 0) + (match.lexical ? reciprocalRank(match.lexical.rank, weights.lexicalWeight) : 0);
189
193
  }
190
194
  function currentChannel(match, channelPrefix) {
191
195
  return channelPrefix ? match.sourceKey.startsWith(channelPrefix) : false;
@@ -203,7 +207,14 @@ function observedAgeRank(memory, nowMs) {
203
207
  }
204
208
  return 0;
205
209
  }
210
+ function positiveWeight(value, fallback) {
211
+ return value !== void 0 && Number.isFinite(value) && value > 0 ? value : fallback;
212
+ }
206
213
  function rankMemoryMatches(matches, options) {
214
+ const weights = {
215
+ lexicalWeight: positiveWeight(options.lexicalWeight, DEFAULT_RRF_WEIGHT),
216
+ vectorWeight: positiveWeight(options.vectorWeight, DEFAULT_RRF_WEIGHT)
217
+ };
207
218
  const byId = /* @__PURE__ */ new Map();
208
219
  for (const match of matches) {
209
220
  const existing = byId.get(match.memory.id);
@@ -213,15 +224,19 @@ function rankMemoryMatches(matches, options) {
213
224
  }
214
225
  byId.set(match.memory.id, {
215
226
  ...existing,
216
- ...match.lexical ? { lexical: match.lexical } : {},
217
- ...match.vector ? { vector: match.vector } : {}
227
+ ...!existing.lexical && match.lexical ? { lexical: match.lexical } : {},
228
+ ...!existing.vector && match.vector ? { vector: match.vector } : {}
218
229
  });
219
230
  }
220
231
  return [...byId.values()].sort((left, right) => {
221
- const scoreDelta = matchScore(right) - matchScore(left);
232
+ const scoreDelta = matchScore(right, weights) - matchScore(left, weights);
222
233
  if (scoreDelta !== 0) {
223
234
  return scoreDelta;
224
235
  }
236
+ const personalDelta = Number(right.memory.scope === "personal") - Number(left.memory.scope === "personal");
237
+ if (personalDelta !== 0) {
238
+ return personalDelta;
239
+ }
225
240
  const channelDelta = Number(currentChannel(right, options.channelPrefix)) - Number(currentChannel(left, options.channelPrefix));
226
241
  if (channelDelta !== 0) {
227
242
  return channelDelta;
@@ -344,11 +359,15 @@ var DEFAULT_SEARCH_LIMIT = 10;
344
359
  var DEFAULT_EXPIRED_ARCHIVE_LIMIT = 100;
345
360
  var PREFERENCE_ADJUDICATION_CANDIDATE_LIMIT = 10;
346
361
  var PREFERENCE_ADJUDICATION_VECTOR_LIMIT = 5;
347
- var VECTOR_SEARCH_OVERFETCH = 4;
348
- var LEXICAL_RANK_OVERFETCH = 4;
349
- var MAX_LEXICAL_RANK_CANDIDATES = 1e3;
362
+ var SEARCH_RETRIEVAL_OVERFETCH = 4;
363
+ var RECALL_RETRIEVAL_OVERFETCH = 2;
364
+ var MAX_RETRIEVAL_LEG_CANDIDATES = 200;
365
+ var MAX_LEXICAL_RANK_CANDIDATES = 200;
366
+ var LEXICAL_RANK_WINDOW_MULTIPLIER = 4;
367
+ var MAX_RETRIEVAL_QUERY_CHARS = 1500;
350
368
  var MAX_MEMORY_CONTENT_CHARS = 4e3;
351
369
  var EMBEDDING_METRIC = "cosine";
370
+ var RECALL_MAX_VECTOR_DISTANCE = 0.45;
352
371
  var nonEmptyStringSchema2 = z2.string().min(1);
353
372
  var memoryContentSchema = z2.string().refine((content) => content.trim().length > 0, {
354
373
  message: "Memory content is required."
@@ -892,12 +911,31 @@ async function listVisibleMemories(args) {
892
911
  ).limit(limit);
893
912
  return rows.map(parseMemoryRow);
894
913
  }
914
+ function normalizeRetrievalQuery(query) {
915
+ const normalized = query.replace(/\s+/g, " ").trim();
916
+ if (normalized.length <= MAX_RETRIEVAL_QUERY_CHARS) {
917
+ return normalized;
918
+ }
919
+ return normalized.slice(0, MAX_RETRIEVAL_QUERY_CHARS).trimEnd();
920
+ }
921
+ function retrievalLegLimit(limit, overfetch) {
922
+ const requested = Math.max(1, limit);
923
+ const withOverfetch = requested * Math.max(1, overfetch);
924
+ return Math.min(
925
+ MAX_RETRIEVAL_LEG_CANDIDATES,
926
+ Math.max(requested, withOverfetch)
927
+ );
928
+ }
895
929
  async function searchVisibleLexicalMemories(args) {
896
930
  const predicate = activeVisiblePredicate(args);
897
931
  if (!predicate) {
898
932
  return [];
899
933
  }
900
- const queryVector = sql2`to_tsvector('english', ${args.query})`;
934
+ const query = normalizeRetrievalQuery(args.query);
935
+ if (!query) {
936
+ return [];
937
+ }
938
+ const queryVector = sql2`to_tsvector('english', ${query})`;
901
939
  const tsquery = sql2`(
902
940
  SELECT COALESCE(
903
941
  string_agg(quote_literal(term), ' | ')::tsquery,
@@ -907,7 +945,7 @@ async function searchVisibleLexicalMemories(args) {
907
945
  )`;
908
946
  const candidateLimit = Math.min(
909
947
  MAX_LEXICAL_RANK_CANDIDATES,
910
- args.limit * LEXICAL_RANK_OVERFETCH
948
+ args.limit * LEXICAL_RANK_WINDOW_MULTIPLIER
911
949
  );
912
950
  const candidates = args.db.select().from(juniorMemoryMemories).where(
913
951
  and(predicate, sql2`${juniorMemoryMemories.searchVector} @@ ${tsquery}`)
@@ -947,23 +985,16 @@ async function searchVisibleLexicalMemories(args) {
947
985
  }));
948
986
  }
949
987
  async function searchVisibleVectorMemories(args) {
950
- if (!args.embedder) {
951
- return [];
952
- }
953
988
  const predicate = activeVisiblePredicate(args);
954
989
  if (!predicate) {
955
990
  return [];
956
991
  }
957
- let embedding;
958
- try {
959
- embedding = await embedOne(args.embedder, args.query);
960
- } catch {
961
- return [];
962
- }
992
+ const embedding = args.embedding;
963
993
  const distance = cosineDistance(
964
994
  juniorMemoryEmbeddings.embedding,
965
995
  embedding.vector
966
996
  );
997
+ const distancePredicate = args.maxDistance === void 0 ? void 0 : sql2`${distance} <= ${args.maxDistance}`;
967
998
  const rows = await args.db.select({
968
999
  contentHash: juniorMemoryEmbeddings.contentHash,
969
1000
  distance,
@@ -977,7 +1008,8 @@ async function searchVisibleVectorMemories(args) {
977
1008
  eq(juniorMemoryEmbeddings.provider, embedding.provider),
978
1009
  eq(juniorMemoryEmbeddings.model, embedding.model),
979
1010
  eq(juniorMemoryEmbeddings.dimensions, MEMORY_EMBEDDING_DIMENSIONS),
980
- eq(juniorMemoryEmbeddings.metric, EMBEDDING_METRIC)
1011
+ eq(juniorMemoryEmbeddings.metric, EMBEDDING_METRIC),
1012
+ ...distancePredicate ? [distancePredicate] : []
981
1013
  )
982
1014
  ).orderBy(
983
1015
  distance,
@@ -990,9 +1022,6 @@ async function searchVisibleVectorMemories(args) {
990
1022
  if (row.distance === null || !Number.isFinite(distanceValue) || hashEmbeddedContent(row.memory.content) !== row.contentHash) {
991
1023
  return [];
992
1024
  }
993
- if (args.maxDistance !== void 0 && distanceValue > args.maxDistance) {
994
- return [];
995
- }
996
1025
  return [
997
1026
  {
998
1027
  memory: parseMemoryRow(row.memory),
@@ -1008,7 +1037,6 @@ function createMemoryStore(db, context, options = {}) {
1008
1037
  const runtimeContext = memoryRuntimeContextSchema.parse(context);
1009
1038
  const parsedOptions = memoryStoreOptionsSchema.parse({ now: options.now });
1010
1039
  const embedder = options.embedder;
1011
- const maxVectorDistance = options.maxVectorDistance;
1012
1040
  const supersessionDecider = options.supersessionDecider;
1013
1041
  const getNowMs = parsedOptions.now ?? Date.now;
1014
1042
  async function archiveExpiredVisibleMemories(input, nowMs) {
@@ -1221,28 +1249,57 @@ function createMemoryStore(db, context, options = {}) {
1221
1249
  scopes
1222
1250
  });
1223
1251
  const limit = boundedLimit(input.limit, DEFAULT_SEARCH_LIMIT);
1224
- const candidateLimit = limit * VECTOR_SEARCH_OVERFETCH;
1225
- const [vectorMatches, lexicalMatches] = await Promise.all([
1226
- searchVisibleVectorMemories({
1252
+ const overfetch = vectorMaxDistance === void 0 ? SEARCH_RETRIEVAL_OVERFETCH : RECALL_RETRIEVAL_OVERFETCH;
1253
+ const candidateLimit = retrievalLegLimit(limit, overfetch);
1254
+ const personalScopes = scopes.filter((scope) => scope.scope === "personal");
1255
+ const probePersonal = vectorMaxDistance !== void 0 && personalScopes.length > 0;
1256
+ const query = normalizeRetrievalQuery(input.query);
1257
+ let queryEmbedding;
1258
+ if (embedder && query) {
1259
+ try {
1260
+ queryEmbedding = await embedOne(embedder, query);
1261
+ } catch {
1262
+ queryEmbedding = void 0;
1263
+ }
1264
+ }
1265
+ const emptyMatches = Promise.resolve([]);
1266
+ const lexicalArgs = {
1267
+ db,
1268
+ limit: candidateLimit,
1269
+ nowMs,
1270
+ query: input.query
1271
+ };
1272
+ const matches = await Promise.all([
1273
+ queryEmbedding ? searchVisibleVectorMemories({
1227
1274
  db,
1228
- embedder,
1275
+ embedding: queryEmbedding,
1229
1276
  limit: candidateLimit,
1230
1277
  ...vectorMaxDistance !== void 0 ? { maxDistance: vectorMaxDistance } : {},
1231
1278
  nowMs,
1232
- query: input.query,
1233
1279
  scopes
1234
- }),
1280
+ }) : emptyMatches,
1235
1281
  searchVisibleLexicalMemories({
1282
+ ...lexicalArgs,
1283
+ scopes
1284
+ }),
1285
+ queryEmbedding && probePersonal ? searchVisibleVectorMemories({
1236
1286
  db,
1287
+ embedding: queryEmbedding,
1237
1288
  limit: candidateLimit,
1289
+ maxDistance: vectorMaxDistance,
1238
1290
  nowMs,
1239
- query: input.query,
1240
- scopes
1241
- })
1291
+ scopes: personalScopes
1292
+ }) : emptyMatches,
1293
+ probePersonal ? searchVisibleLexicalMemories({
1294
+ ...lexicalArgs,
1295
+ scopes: personalScopes
1296
+ }) : emptyMatches
1242
1297
  ]);
1243
1298
  const channelPrefix = sourceChannelPrefix(runtimeContext);
1244
- return rankMemoryMatches([...vectorMatches, ...lexicalMatches], {
1299
+ return rankMemoryMatches(matches.flat(), {
1245
1300
  nowMs,
1301
+ // Slight lexical preference protects exact ids/names/timezones on ties.
1302
+ ...vectorMaxDistance === void 0 ? {} : { lexicalWeight: 1, vectorWeight: 0.85 },
1246
1303
  ...channelPrefix ? { channelPrefix } : {}
1247
1304
  }).slice(0, limit).map(({ memory }) => memory);
1248
1305
  }
@@ -1289,7 +1346,7 @@ function createMemoryStore(db, context, options = {}) {
1289
1346
  });
1290
1347
  },
1291
1348
  async recallMemories(input) {
1292
- return await retrieveVisibleMemories(input, maxVectorDistance);
1349
+ return await retrieveVisibleMemories(input, RECALL_MAX_VECTOR_DISTANCE);
1293
1350
  },
1294
1351
  async searchMemories(input) {
1295
1352
  return await retrieveVisibleMemories(input, void 0);
@@ -3169,8 +3226,7 @@ async function createMemoryPromptContributions(context) {
3169
3226
  }
3170
3227
  } : void 0;
3171
3228
  const candidates = await createMemoryStore(context.db, runtimeContext, {
3172
- ...embedder ? { embedder } : {},
3173
- ...context.maxVectorDistance !== void 0 ? { maxVectorDistance: context.maxVectorDistance } : {}
3229
+ embedder
3174
3230
  }).recallMemories({
3175
3231
  query: context.text,
3176
3232
  limit: RECALL_CANDIDATE_LIMIT
@@ -3482,8 +3538,6 @@ function createMemoryUserPage() {
3482
3538
 
3483
3539
  // src/plugin.ts
3484
3540
  var MEMORY_MODEL_ENV = "AI_MEMORY_MODEL";
3485
- var MEMORY_RECALL_MAX_VECTOR_DISTANCE_ENV = "MEMORY_RECALL_MAX_VECTOR_DISTANCE";
3486
- var DEFAULT_RECALL_MAX_VECTOR_DISTANCE = 0.45;
3487
3541
  function memoryModelId(options) {
3488
3542
  const explicitModelId = options.modelId?.trim();
3489
3543
  if (explicitModelId) {
@@ -3492,19 +3546,6 @@ function memoryModelId(options) {
3492
3546
  const envModelId = process.env[MEMORY_MODEL_ENV]?.trim();
3493
3547
  return envModelId || void 0;
3494
3548
  }
3495
- function recallMaxVectorDistance(options) {
3496
- if (options.recallMaxVectorDistance !== void 0) {
3497
- return options.recallMaxVectorDistance;
3498
- }
3499
- const raw = process.env[MEMORY_RECALL_MAX_VECTOR_DISTANCE_ENV]?.trim();
3500
- if (raw) {
3501
- const parsed = Number(raw);
3502
- if (Number.isFinite(parsed) && parsed > 0) {
3503
- return Math.min(parsed, 1);
3504
- }
3505
- }
3506
- return DEFAULT_RECALL_MAX_VECTOR_DISTANCE;
3507
- }
3508
3549
  function memoryToolContext(ctx) {
3509
3550
  return {
3510
3551
  agent: ctx.agent,
@@ -3540,7 +3581,7 @@ function memoryPlugin(options = {}) {
3540
3581
  cli: {
3541
3582
  commands: [createMemoryCliCommand()]
3542
3583
  },
3543
- tasks: {
3584
+ tasks: options.disableExtraction ? {} : {
3544
3585
  processSession: {
3545
3586
  async run(ctx) {
3546
3587
  await processMemorySession(ctx);
@@ -3590,20 +3631,21 @@ function memoryPlugin(options = {}) {
3590
3631
  searchMemories: createMemorySearchTool(context)
3591
3632
  };
3592
3633
  },
3593
- async userPrompt(ctx) {
3594
- return await createMemoryPromptContributions({
3595
- agent: createMemoryAgent(ctx.model),
3596
- ...ctx.conversationId ? { conversationId: ctx.conversationId } : {},
3597
- ...ctx.actor ? { actor: ctx.actor } : {},
3598
- db: ctx.db,
3599
- embedder: ctx.embedder,
3600
- events: ctx.events,
3601
- log: ctx.log,
3602
- maxVectorDistance: recallMaxVectorDistance(options),
3603
- source: ctx.source,
3604
- text: ctx.text
3605
- });
3606
- }
3634
+ ...!options.disableRecall ? {
3635
+ async userPrompt(ctx) {
3636
+ return await createMemoryPromptContributions({
3637
+ agent: createMemoryAgent(ctx.model),
3638
+ ...ctx.conversationId ? { conversationId: ctx.conversationId } : {},
3639
+ ...ctx.actor ? { actor: ctx.actor } : {},
3640
+ db: ctx.db,
3641
+ embedder: ctx.embedder,
3642
+ events: ctx.events,
3643
+ log: ctx.log,
3644
+ source: ctx.source,
3645
+ text: ctx.text
3646
+ });
3647
+ }
3648
+ } : {}
3607
3649
  }
3608
3650
  });
3609
3651
  }