@sentry/junior-memory 0.133.0 → 0.134.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -65,7 +65,13 @@ exported types, tools, and tests are authoritative.
65
65
  - Search combines independently ranked vector and PostgreSQL full-text matches
66
66
  with reciprocal rank fusion; provider-specific raw scores are never added
67
67
  together.
68
- - Automatic recall retrieves a broad candidate window, then uses the
68
+ - Both retrieval legs always run in parallel as bounded top-k probes. Each leg
69
+ fetches at least the caller's requested limit (and never more than the store
70
+ limit ceiling). Recall keeps a smaller overfetch window than explicit search
71
+ and slightly prefers lexical ranks so exact tokens survive soft semantic
72
+ neighbors. Vector recall also applies the cosine distance cutoff in SQL, and
73
+ embeddings use an HNSW cosine index (`vector_cosine_ops`).
74
+ - Automatic recall retrieves a bounded candidate window, then uses the
69
75
  memory-owned relevance model to admit at most five directly useful memories.
70
76
  An empty result contributes no filler prompt text.
71
77
  - Every completed automatic recall attempt emits an invisible, namespaced
@@ -80,8 +86,12 @@ exported types, tools, and tests are authoritative.
80
86
 
81
87
  - `AI_MEMORY_MODEL` or `memoryPlugin({ modelId })` selects the structured
82
88
  review model.
83
- - `MEMORY_RECALL_MAX_VECTOR_DISTANCE` or
84
- `recallMaxVectorDistance` configures the vector candidate threshold.
89
+ - `memoryPlugin({ disableRecall: true })` disables automatic prompt recall.
90
+ - `memoryPlugin({ disableExtraction: true })` disables passive session
91
+ extraction. The two flags are independent and do not disable explicit memory
92
+ tools.
93
+ - Automatic recall uses a fixed cosine distance cutoff of `0.45` (for
94
+ `text-embedding-3-small`). Explicit search does not apply that cutoff.
85
95
  - Generate schema changes with `pnpm --filter @sentry/junior-memory db:generate`.
86
96
 
87
97
  Follow `../../policies/data-redaction.md`, `../../policies/security.md`, and the
package/dist/index.js CHANGED
@@ -167,6 +167,9 @@ var juniorMemoryEmbeddings = pgTable(
167
167
  table.dimensions,
168
168
  table.metric
169
169
  ),
170
+ // Cosine ANN for vector recall/search. Ops must match cosineDistance (<=>).
171
+ // Keep this unfiltered so planners can use HNSW before scope/status joins.
172
+ index("junior_memory_embeddings_embedding_hnsw_idx").using("hnsw", table.embedding.op("vector_cosine_ops")).with({ m: 16, ef_construction: 64 }),
170
173
  check(
171
174
  "junior_memory_embeddings_metric_check",
172
175
  sql`${table.metric} IN ('cosine')`
@@ -181,11 +184,12 @@ var juniorMemoryEmbeddings = pgTable(
181
184
  // src/ranking.ts
182
185
  var RECIPROCAL_RANK_FUSION_K = 60;
183
186
  var ONE_DAY_MS = 24 * 60 * 60 * 1e3;
184
- function reciprocalRank(rank) {
185
- return 1 / (RECIPROCAL_RANK_FUSION_K + rank);
187
+ var DEFAULT_RRF_WEIGHT = 1;
188
+ function reciprocalRank(rank, weight) {
189
+ return weight / (RECIPROCAL_RANK_FUSION_K + rank);
186
190
  }
187
- function matchScore(match) {
188
- return (match.vector ? reciprocalRank(match.vector.rank) : 0) + (match.lexical ? reciprocalRank(match.lexical.rank) : 0);
191
+ function matchScore(match, weights) {
192
+ return (match.vector ? reciprocalRank(match.vector.rank, weights.vectorWeight) : 0) + (match.lexical ? reciprocalRank(match.lexical.rank, weights.lexicalWeight) : 0);
189
193
  }
190
194
  function currentChannel(match, channelPrefix) {
191
195
  return channelPrefix ? match.sourceKey.startsWith(channelPrefix) : false;
@@ -203,7 +207,14 @@ function observedAgeRank(memory, nowMs) {
203
207
  }
204
208
  return 0;
205
209
  }
210
+ function positiveWeight(value, fallback) {
211
+ return value !== void 0 && Number.isFinite(value) && value > 0 ? value : fallback;
212
+ }
206
213
  function rankMemoryMatches(matches, options) {
214
+ const weights = {
215
+ lexicalWeight: positiveWeight(options.lexicalWeight, DEFAULT_RRF_WEIGHT),
216
+ vectorWeight: positiveWeight(options.vectorWeight, DEFAULT_RRF_WEIGHT)
217
+ };
207
218
  const byId = /* @__PURE__ */ new Map();
208
219
  for (const match of matches) {
209
220
  const existing = byId.get(match.memory.id);
@@ -218,7 +229,7 @@ function rankMemoryMatches(matches, options) {
218
229
  });
219
230
  }
220
231
  return [...byId.values()].sort((left, right) => {
221
- const scoreDelta = matchScore(right) - matchScore(left);
232
+ const scoreDelta = matchScore(right, weights) - matchScore(left, weights);
222
233
  if (scoreDelta !== 0) {
223
234
  return scoreDelta;
224
235
  }
@@ -344,11 +355,15 @@ var DEFAULT_SEARCH_LIMIT = 10;
344
355
  var DEFAULT_EXPIRED_ARCHIVE_LIMIT = 100;
345
356
  var PREFERENCE_ADJUDICATION_CANDIDATE_LIMIT = 10;
346
357
  var PREFERENCE_ADJUDICATION_VECTOR_LIMIT = 5;
347
- var VECTOR_SEARCH_OVERFETCH = 4;
348
- var LEXICAL_RANK_OVERFETCH = 4;
349
- var MAX_LEXICAL_RANK_CANDIDATES = 1e3;
358
+ var SEARCH_RETRIEVAL_OVERFETCH = 4;
359
+ var RECALL_RETRIEVAL_OVERFETCH = 2;
360
+ var MAX_RETRIEVAL_LEG_CANDIDATES = 200;
361
+ var MAX_LEXICAL_RANK_CANDIDATES = 200;
362
+ var LEXICAL_RANK_WINDOW_MULTIPLIER = 4;
363
+ var MAX_RETRIEVAL_QUERY_CHARS = 1500;
350
364
  var MAX_MEMORY_CONTENT_CHARS = 4e3;
351
365
  var EMBEDDING_METRIC = "cosine";
366
+ var RECALL_MAX_VECTOR_DISTANCE = 0.45;
352
367
  var nonEmptyStringSchema2 = z2.string().min(1);
353
368
  var memoryContentSchema = z2.string().refine((content) => content.trim().length > 0, {
354
369
  message: "Memory content is required."
@@ -892,12 +907,31 @@ async function listVisibleMemories(args) {
892
907
  ).limit(limit);
893
908
  return rows.map(parseMemoryRow);
894
909
  }
910
+ function normalizeRetrievalQuery(query) {
911
+ const normalized = query.replace(/\s+/g, " ").trim();
912
+ if (normalized.length <= MAX_RETRIEVAL_QUERY_CHARS) {
913
+ return normalized;
914
+ }
915
+ return normalized.slice(0, MAX_RETRIEVAL_QUERY_CHARS).trimEnd();
916
+ }
917
+ function retrievalLegLimit(limit, overfetch) {
918
+ const requested = Math.max(1, limit);
919
+ const withOverfetch = requested * Math.max(1, overfetch);
920
+ return Math.min(
921
+ MAX_RETRIEVAL_LEG_CANDIDATES,
922
+ Math.max(requested, withOverfetch)
923
+ );
924
+ }
895
925
  async function searchVisibleLexicalMemories(args) {
896
926
  const predicate = activeVisiblePredicate(args);
897
927
  if (!predicate) {
898
928
  return [];
899
929
  }
900
- const queryVector = sql2`to_tsvector('english', ${args.query})`;
930
+ const query = normalizeRetrievalQuery(args.query);
931
+ if (!query) {
932
+ return [];
933
+ }
934
+ const queryVector = sql2`to_tsvector('english', ${query})`;
901
935
  const tsquery = sql2`(
902
936
  SELECT COALESCE(
903
937
  string_agg(quote_literal(term), ' | ')::tsquery,
@@ -907,7 +941,7 @@ async function searchVisibleLexicalMemories(args) {
907
941
  )`;
908
942
  const candidateLimit = Math.min(
909
943
  MAX_LEXICAL_RANK_CANDIDATES,
910
- args.limit * LEXICAL_RANK_OVERFETCH
944
+ args.limit * LEXICAL_RANK_WINDOW_MULTIPLIER
911
945
  );
912
946
  const candidates = args.db.select().from(juniorMemoryMemories).where(
913
947
  and(predicate, sql2`${juniorMemoryMemories.searchVector} @@ ${tsquery}`)
@@ -954,9 +988,13 @@ async function searchVisibleVectorMemories(args) {
954
988
  if (!predicate) {
955
989
  return [];
956
990
  }
991
+ const query = normalizeRetrievalQuery(args.query);
992
+ if (!query) {
993
+ return [];
994
+ }
957
995
  let embedding;
958
996
  try {
959
- embedding = await embedOne(args.embedder, args.query);
997
+ embedding = await embedOne(args.embedder, query);
960
998
  } catch {
961
999
  return [];
962
1000
  }
@@ -964,6 +1002,7 @@ async function searchVisibleVectorMemories(args) {
964
1002
  juniorMemoryEmbeddings.embedding,
965
1003
  embedding.vector
966
1004
  );
1005
+ const distancePredicate = args.maxDistance === void 0 ? void 0 : sql2`${distance} <= ${args.maxDistance}`;
967
1006
  const rows = await args.db.select({
968
1007
  contentHash: juniorMemoryEmbeddings.contentHash,
969
1008
  distance,
@@ -977,7 +1016,8 @@ async function searchVisibleVectorMemories(args) {
977
1016
  eq(juniorMemoryEmbeddings.provider, embedding.provider),
978
1017
  eq(juniorMemoryEmbeddings.model, embedding.model),
979
1018
  eq(juniorMemoryEmbeddings.dimensions, MEMORY_EMBEDDING_DIMENSIONS),
980
- eq(juniorMemoryEmbeddings.metric, EMBEDDING_METRIC)
1019
+ eq(juniorMemoryEmbeddings.metric, EMBEDDING_METRIC),
1020
+ ...distancePredicate ? [distancePredicate] : []
981
1021
  )
982
1022
  ).orderBy(
983
1023
  distance,
@@ -990,9 +1030,6 @@ async function searchVisibleVectorMemories(args) {
990
1030
  if (row.distance === null || !Number.isFinite(distanceValue) || hashEmbeddedContent(row.memory.content) !== row.contentHash) {
991
1031
  return [];
992
1032
  }
993
- if (args.maxDistance !== void 0 && distanceValue > args.maxDistance) {
994
- return [];
995
- }
996
1033
  return [
997
1034
  {
998
1035
  memory: parseMemoryRow(row.memory),
@@ -1008,7 +1045,6 @@ function createMemoryStore(db, context, options = {}) {
1008
1045
  const runtimeContext = memoryRuntimeContextSchema.parse(context);
1009
1046
  const parsedOptions = memoryStoreOptionsSchema.parse({ now: options.now });
1010
1047
  const embedder = options.embedder;
1011
- const maxVectorDistance = options.maxVectorDistance;
1012
1048
  const supersessionDecider = options.supersessionDecider;
1013
1049
  const getNowMs = parsedOptions.now ?? Date.now;
1014
1050
  async function archiveExpiredVisibleMemories(input, nowMs) {
@@ -1221,7 +1257,8 @@ function createMemoryStore(db, context, options = {}) {
1221
1257
  scopes
1222
1258
  });
1223
1259
  const limit = boundedLimit(input.limit, DEFAULT_SEARCH_LIMIT);
1224
- const candidateLimit = limit * VECTOR_SEARCH_OVERFETCH;
1260
+ const overfetch = vectorMaxDistance === void 0 ? SEARCH_RETRIEVAL_OVERFETCH : RECALL_RETRIEVAL_OVERFETCH;
1261
+ const candidateLimit = retrievalLegLimit(limit, overfetch);
1225
1262
  const [vectorMatches, lexicalMatches] = await Promise.all([
1226
1263
  searchVisibleVectorMemories({
1227
1264
  db,
@@ -1243,6 +1280,8 @@ function createMemoryStore(db, context, options = {}) {
1243
1280
  const channelPrefix = sourceChannelPrefix(runtimeContext);
1244
1281
  return rankMemoryMatches([...vectorMatches, ...lexicalMatches], {
1245
1282
  nowMs,
1283
+ // Slight lexical preference protects exact ids/names/timezones on ties.
1284
+ ...vectorMaxDistance === void 0 ? {} : { lexicalWeight: 1, vectorWeight: 0.85 },
1246
1285
  ...channelPrefix ? { channelPrefix } : {}
1247
1286
  }).slice(0, limit).map(({ memory }) => memory);
1248
1287
  }
@@ -1289,7 +1328,7 @@ function createMemoryStore(db, context, options = {}) {
1289
1328
  });
1290
1329
  },
1291
1330
  async recallMemories(input) {
1292
- return await retrieveVisibleMemories(input, maxVectorDistance);
1331
+ return await retrieveVisibleMemories(input, RECALL_MAX_VECTOR_DISTANCE);
1293
1332
  },
1294
1333
  async searchMemories(input) {
1295
1334
  return await retrieveVisibleMemories(input, void 0);
@@ -3169,8 +3208,7 @@ async function createMemoryPromptContributions(context) {
3169
3208
  }
3170
3209
  } : void 0;
3171
3210
  const candidates = await createMemoryStore(context.db, runtimeContext, {
3172
- ...embedder ? { embedder } : {},
3173
- ...context.maxVectorDistance !== void 0 ? { maxVectorDistance: context.maxVectorDistance } : {}
3211
+ embedder
3174
3212
  }).recallMemories({
3175
3213
  query: context.text,
3176
3214
  limit: RECALL_CANDIDATE_LIMIT
@@ -3482,8 +3520,6 @@ function createMemoryUserPage() {
3482
3520
 
3483
3521
  // src/plugin.ts
3484
3522
  var MEMORY_MODEL_ENV = "AI_MEMORY_MODEL";
3485
- var MEMORY_RECALL_MAX_VECTOR_DISTANCE_ENV = "MEMORY_RECALL_MAX_VECTOR_DISTANCE";
3486
- var DEFAULT_RECALL_MAX_VECTOR_DISTANCE = 0.45;
3487
3523
  function memoryModelId(options) {
3488
3524
  const explicitModelId = options.modelId?.trim();
3489
3525
  if (explicitModelId) {
@@ -3492,19 +3528,6 @@ function memoryModelId(options) {
3492
3528
  const envModelId = process.env[MEMORY_MODEL_ENV]?.trim();
3493
3529
  return envModelId || void 0;
3494
3530
  }
3495
- function recallMaxVectorDistance(options) {
3496
- if (options.recallMaxVectorDistance !== void 0) {
3497
- return options.recallMaxVectorDistance;
3498
- }
3499
- const raw = process.env[MEMORY_RECALL_MAX_VECTOR_DISTANCE_ENV]?.trim();
3500
- if (raw) {
3501
- const parsed = Number(raw);
3502
- if (Number.isFinite(parsed) && parsed > 0) {
3503
- return Math.min(parsed, 1);
3504
- }
3505
- }
3506
- return DEFAULT_RECALL_MAX_VECTOR_DISTANCE;
3507
- }
3508
3531
  function memoryToolContext(ctx) {
3509
3532
  return {
3510
3533
  agent: ctx.agent,
@@ -3540,7 +3563,7 @@ function memoryPlugin(options = {}) {
3540
3563
  cli: {
3541
3564
  commands: [createMemoryCliCommand()]
3542
3565
  },
3543
- tasks: {
3566
+ tasks: options.disableExtraction ? {} : {
3544
3567
  processSession: {
3545
3568
  async run(ctx) {
3546
3569
  await processMemorySession(ctx);
@@ -3590,20 +3613,21 @@ function memoryPlugin(options = {}) {
3590
3613
  searchMemories: createMemorySearchTool(context)
3591
3614
  };
3592
3615
  },
3593
- async userPrompt(ctx) {
3594
- return await createMemoryPromptContributions({
3595
- agent: createMemoryAgent(ctx.model),
3596
- ...ctx.conversationId ? { conversationId: ctx.conversationId } : {},
3597
- ...ctx.actor ? { actor: ctx.actor } : {},
3598
- db: ctx.db,
3599
- embedder: ctx.embedder,
3600
- events: ctx.events,
3601
- log: ctx.log,
3602
- maxVectorDistance: recallMaxVectorDistance(options),
3603
- source: ctx.source,
3604
- text: ctx.text
3605
- });
3606
- }
3616
+ ...!options.disableRecall ? {
3617
+ async userPrompt(ctx) {
3618
+ return await createMemoryPromptContributions({
3619
+ agent: createMemoryAgent(ctx.model),
3620
+ ...ctx.conversationId ? { conversationId: ctx.conversationId } : {},
3621
+ ...ctx.actor ? { actor: ctx.actor } : {},
3622
+ db: ctx.db,
3623
+ embedder: ctx.embedder,
3624
+ events: ctx.events,
3625
+ log: ctx.log,
3626
+ source: ctx.source,
3627
+ text: ctx.text
3628
+ });
3629
+ }
3630
+ } : {}
3607
3631
  }
3608
3632
  });
3609
3633
  }