@sentry/junior-memory 0.132.0 → 0.134.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -65,7 +65,13 @@ exported types, tools, and tests are authoritative.
65
65
  - Search combines independently ranked vector and PostgreSQL full-text matches
66
66
  with reciprocal rank fusion; provider-specific raw scores are never added
67
67
  together.
68
- - Automatic recall retrieves a broad candidate window, then uses the
68
+ - Both retrieval legs always run in parallel as bounded top-k probes. Each leg
69
+ fetches at least the caller's requested limit (and never more than the store
70
+ limit ceiling). Recall keeps a smaller overfetch window than explicit search
71
+ and slightly prefers lexical ranks so exact tokens survive soft semantic
72
+ neighbors. Vector recall also applies the cosine distance cutoff in SQL, and
73
+ embeddings use an HNSW cosine index (`vector_cosine_ops`).
74
+ - Automatic recall retrieves a bounded candidate window, then uses the
69
75
  memory-owned relevance model to admit at most five directly useful memories.
70
76
  An empty result contributes no filler prompt text.
71
77
  - Every completed automatic recall attempt emits an invisible, namespaced
@@ -80,8 +86,12 @@ exported types, tools, and tests are authoritative.
80
86
 
81
87
  - `AI_MEMORY_MODEL` or `memoryPlugin({ modelId })` selects the structured
82
88
  review model.
83
- - `MEMORY_RECALL_MAX_VECTOR_DISTANCE` or
84
- `recallMaxVectorDistance` configures the vector candidate threshold.
89
+ - `memoryPlugin({ disableRecall: true })` disables automatic prompt recall.
90
+ - `memoryPlugin({ disableExtraction: true })` disables passive session
91
+ extraction. The two flags are independent and do not disable explicit memory
92
+ tools.
93
+ - Automatic recall uses a fixed cosine distance cutoff of `0.45` (for
94
+ `text-embedding-3-small`). Explicit search does not apply that cutoff.
85
95
  - Generate schema changes with `pnpm --filter @sentry/junior-memory db:generate`.
86
96
 
87
97
  Follow `../../policies/data-redaction.md`, `../../policies/security.md`, and the
package/dist/index.js CHANGED
@@ -167,6 +167,9 @@ var juniorMemoryEmbeddings = pgTable(
167
167
  table.dimensions,
168
168
  table.metric
169
169
  ),
170
+ // Cosine ANN for vector recall/search. Ops must match cosineDistance (<=>).
171
+ // Keep this unfiltered so planners can use HNSW before scope/status joins.
172
+ index("junior_memory_embeddings_embedding_hnsw_idx").using("hnsw", table.embedding.op("vector_cosine_ops")).with({ m: 16, ef_construction: 64 }),
170
173
  check(
171
174
  "junior_memory_embeddings_metric_check",
172
175
  sql`${table.metric} IN ('cosine')`
@@ -181,11 +184,12 @@ var juniorMemoryEmbeddings = pgTable(
181
184
  // src/ranking.ts
182
185
  var RECIPROCAL_RANK_FUSION_K = 60;
183
186
  var ONE_DAY_MS = 24 * 60 * 60 * 1e3;
184
- function reciprocalRank(rank) {
185
- return 1 / (RECIPROCAL_RANK_FUSION_K + rank);
187
+ var DEFAULT_RRF_WEIGHT = 1;
188
+ function reciprocalRank(rank, weight) {
189
+ return weight / (RECIPROCAL_RANK_FUSION_K + rank);
186
190
  }
187
- function matchScore(match) {
188
- return (match.vector ? reciprocalRank(match.vector.rank) : 0) + (match.lexical ? reciprocalRank(match.lexical.rank) : 0);
191
+ function matchScore(match, weights) {
192
+ return (match.vector ? reciprocalRank(match.vector.rank, weights.vectorWeight) : 0) + (match.lexical ? reciprocalRank(match.lexical.rank, weights.lexicalWeight) : 0);
189
193
  }
190
194
  function currentChannel(match, channelPrefix) {
191
195
  return channelPrefix ? match.sourceKey.startsWith(channelPrefix) : false;
@@ -203,7 +207,14 @@ function observedAgeRank(memory, nowMs) {
203
207
  }
204
208
  return 0;
205
209
  }
210
+ function positiveWeight(value, fallback) {
211
+ return value !== void 0 && Number.isFinite(value) && value > 0 ? value : fallback;
212
+ }
206
213
  function rankMemoryMatches(matches, options) {
214
+ const weights = {
215
+ lexicalWeight: positiveWeight(options.lexicalWeight, DEFAULT_RRF_WEIGHT),
216
+ vectorWeight: positiveWeight(options.vectorWeight, DEFAULT_RRF_WEIGHT)
217
+ };
207
218
  const byId = /* @__PURE__ */ new Map();
208
219
  for (const match of matches) {
209
220
  const existing = byId.get(match.memory.id);
@@ -218,7 +229,7 @@ function rankMemoryMatches(matches, options) {
218
229
  });
219
230
  }
220
231
  return [...byId.values()].sort((left, right) => {
221
- const scoreDelta = matchScore(right) - matchScore(left);
232
+ const scoreDelta = matchScore(right, weights) - matchScore(left, weights);
222
233
  if (scoreDelta !== 0) {
223
234
  return scoreDelta;
224
235
  }
@@ -344,9 +355,15 @@ var DEFAULT_SEARCH_LIMIT = 10;
344
355
  var DEFAULT_EXPIRED_ARCHIVE_LIMIT = 100;
345
356
  var PREFERENCE_ADJUDICATION_CANDIDATE_LIMIT = 10;
346
357
  var PREFERENCE_ADJUDICATION_VECTOR_LIMIT = 5;
347
- var VECTOR_SEARCH_OVERFETCH = 4;
358
+ var SEARCH_RETRIEVAL_OVERFETCH = 4;
359
+ var RECALL_RETRIEVAL_OVERFETCH = 2;
360
+ var MAX_RETRIEVAL_LEG_CANDIDATES = 200;
361
+ var MAX_LEXICAL_RANK_CANDIDATES = 200;
362
+ var LEXICAL_RANK_WINDOW_MULTIPLIER = 4;
363
+ var MAX_RETRIEVAL_QUERY_CHARS = 1500;
348
364
  var MAX_MEMORY_CONTENT_CHARS = 4e3;
349
365
  var EMBEDDING_METRIC = "cosine";
366
+ var RECALL_MAX_VECTOR_DISTANCE = 0.45;
350
367
  var nonEmptyStringSchema2 = z2.string().min(1);
351
368
  var memoryContentSchema = z2.string().refine((content) => content.trim().length > 0, {
352
369
  message: "Memory content is required."
@@ -890,12 +907,31 @@ async function listVisibleMemories(args) {
890
907
  ).limit(limit);
891
908
  return rows.map(parseMemoryRow);
892
909
  }
910
+ function normalizeRetrievalQuery(query) {
911
+ const normalized = query.replace(/\s+/g, " ").trim();
912
+ if (normalized.length <= MAX_RETRIEVAL_QUERY_CHARS) {
913
+ return normalized;
914
+ }
915
+ return normalized.slice(0, MAX_RETRIEVAL_QUERY_CHARS).trimEnd();
916
+ }
917
+ function retrievalLegLimit(limit, overfetch) {
918
+ const requested = Math.max(1, limit);
919
+ const withOverfetch = requested * Math.max(1, overfetch);
920
+ return Math.min(
921
+ MAX_RETRIEVAL_LEG_CANDIDATES,
922
+ Math.max(requested, withOverfetch)
923
+ );
924
+ }
893
925
  async function searchVisibleLexicalMemories(args) {
894
926
  const predicate = activeVisiblePredicate(args);
895
927
  if (!predicate) {
896
928
  return [];
897
929
  }
898
- const queryVector = sql2`to_tsvector('english', ${args.query})`;
930
+ const query = normalizeRetrievalQuery(args.query);
931
+ if (!query) {
932
+ return [];
933
+ }
934
+ const queryVector = sql2`to_tsvector('english', ${query})`;
899
935
  const tsquery = sql2`(
900
936
  SELECT COALESCE(
901
937
  string_agg(quote_literal(term), ' | ')::tsquery,
@@ -903,16 +939,40 @@ async function searchVisibleLexicalMemories(args) {
903
939
  )
904
940
  FROM unnest(tsvector_to_array(${queryVector})) AS query_terms(term)
905
941
  )`;
906
- const textsearch = juniorMemoryMemories.searchVector;
907
- const textRank = sql2`ts_rank_cd(${textsearch}, ${tsquery})`;
908
- const rows = await args.db.select({
909
- memory: juniorMemoryMemories,
910
- textRank
911
- }).from(juniorMemoryMemories).where(and(predicate, sql2`${textsearch} @@ ${tsquery}`)).orderBy(
912
- desc(textRank),
942
+ const candidateLimit = Math.min(
943
+ MAX_LEXICAL_RANK_CANDIDATES,
944
+ args.limit * LEXICAL_RANK_WINDOW_MULTIPLIER
945
+ );
946
+ const candidates = args.db.select().from(juniorMemoryMemories).where(
947
+ and(predicate, sql2`${juniorMemoryMemories.searchVector} @@ ${tsquery}`)
948
+ ).orderBy(
913
949
  desc(juniorMemoryMemories.observedAtMs),
914
950
  asc(juniorMemoryMemories.id)
915
- ).limit(args.limit);
951
+ ).limit(candidateLimit).as("lexical_candidates");
952
+ const textRank = sql2`ts_rank_cd(${candidates.searchVector}, ${tsquery})`;
953
+ const rows = await args.db.select({
954
+ memory: {
955
+ archiveReason: candidates.archiveReason,
956
+ archivedAtMs: candidates.archivedAtMs,
957
+ content: candidates.content,
958
+ createdAtMs: candidates.createdAtMs,
959
+ expiresAtMs: candidates.expiresAtMs,
960
+ id: candidates.id,
961
+ idempotencyKey: candidates.idempotencyKey,
962
+ kind: candidates.kind,
963
+ observedAtMs: candidates.observedAtMs,
964
+ scope: candidates.scope,
965
+ scopeKey: candidates.scopeKey,
966
+ searchVector: candidates.searchVector,
967
+ sourceKey: candidates.sourceKey,
968
+ sourcePlatform: candidates.sourcePlatform,
969
+ subjectKey: candidates.subjectKey,
970
+ subjectType: candidates.subjectType,
971
+ supersededAtMs: candidates.supersededAtMs,
972
+ supersededById: candidates.supersededById
973
+ },
974
+ textRank
975
+ }).from(candidates).orderBy(desc(textRank), desc(candidates.observedAtMs), asc(candidates.id)).limit(args.limit);
916
976
  const ranks = denseRanks(rows, (row) => Number(row.textRank));
917
977
  return rows.map((row, index2) => ({
918
978
  lexical: { rank: ranks[index2] },
@@ -928,9 +988,13 @@ async function searchVisibleVectorMemories(args) {
928
988
  if (!predicate) {
929
989
  return [];
930
990
  }
991
+ const query = normalizeRetrievalQuery(args.query);
992
+ if (!query) {
993
+ return [];
994
+ }
931
995
  let embedding;
932
996
  try {
933
- embedding = await embedOne(args.embedder, args.query);
997
+ embedding = await embedOne(args.embedder, query);
934
998
  } catch {
935
999
  return [];
936
1000
  }
@@ -938,6 +1002,7 @@ async function searchVisibleVectorMemories(args) {
938
1002
  juniorMemoryEmbeddings.embedding,
939
1003
  embedding.vector
940
1004
  );
1005
+ const distancePredicate = args.maxDistance === void 0 ? void 0 : sql2`${distance} <= ${args.maxDistance}`;
941
1006
  const rows = await args.db.select({
942
1007
  contentHash: juniorMemoryEmbeddings.contentHash,
943
1008
  distance,
@@ -951,7 +1016,8 @@ async function searchVisibleVectorMemories(args) {
951
1016
  eq(juniorMemoryEmbeddings.provider, embedding.provider),
952
1017
  eq(juniorMemoryEmbeddings.model, embedding.model),
953
1018
  eq(juniorMemoryEmbeddings.dimensions, MEMORY_EMBEDDING_DIMENSIONS),
954
- eq(juniorMemoryEmbeddings.metric, EMBEDDING_METRIC)
1019
+ eq(juniorMemoryEmbeddings.metric, EMBEDDING_METRIC),
1020
+ ...distancePredicate ? [distancePredicate] : []
955
1021
  )
956
1022
  ).orderBy(
957
1023
  distance,
@@ -964,9 +1030,6 @@ async function searchVisibleVectorMemories(args) {
964
1030
  if (row.distance === null || !Number.isFinite(distanceValue) || hashEmbeddedContent(row.memory.content) !== row.contentHash) {
965
1031
  return [];
966
1032
  }
967
- if (args.maxDistance !== void 0 && distanceValue > args.maxDistance) {
968
- return [];
969
- }
970
1033
  return [
971
1034
  {
972
1035
  memory: parseMemoryRow(row.memory),
@@ -982,7 +1045,6 @@ function createMemoryStore(db, context, options = {}) {
982
1045
  const runtimeContext = memoryRuntimeContextSchema.parse(context);
983
1046
  const parsedOptions = memoryStoreOptionsSchema.parse({ now: options.now });
984
1047
  const embedder = options.embedder;
985
- const maxVectorDistance = options.maxVectorDistance;
986
1048
  const supersessionDecider = options.supersessionDecider;
987
1049
  const getNowMs = parsedOptions.now ?? Date.now;
988
1050
  async function archiveExpiredVisibleMemories(input, nowMs) {
@@ -1195,7 +1257,8 @@ function createMemoryStore(db, context, options = {}) {
1195
1257
  scopes
1196
1258
  });
1197
1259
  const limit = boundedLimit(input.limit, DEFAULT_SEARCH_LIMIT);
1198
- const candidateLimit = limit * VECTOR_SEARCH_OVERFETCH;
1260
+ const overfetch = vectorMaxDistance === void 0 ? SEARCH_RETRIEVAL_OVERFETCH : RECALL_RETRIEVAL_OVERFETCH;
1261
+ const candidateLimit = retrievalLegLimit(limit, overfetch);
1199
1262
  const [vectorMatches, lexicalMatches] = await Promise.all([
1200
1263
  searchVisibleVectorMemories({
1201
1264
  db,
@@ -1217,6 +1280,8 @@ function createMemoryStore(db, context, options = {}) {
1217
1280
  const channelPrefix = sourceChannelPrefix(runtimeContext);
1218
1281
  return rankMemoryMatches([...vectorMatches, ...lexicalMatches], {
1219
1282
  nowMs,
1283
+ // Slight lexical preference protects exact ids/names/timezones on ties.
1284
+ ...vectorMaxDistance === void 0 ? {} : { lexicalWeight: 1, vectorWeight: 0.85 },
1220
1285
  ...channelPrefix ? { channelPrefix } : {}
1221
1286
  }).slice(0, limit).map(({ memory }) => memory);
1222
1287
  }
@@ -1263,7 +1328,7 @@ function createMemoryStore(db, context, options = {}) {
1263
1328
  });
1264
1329
  },
1265
1330
  async recallMemories(input) {
1266
- return await retrieveVisibleMemories(input, maxVectorDistance);
1331
+ return await retrieveVisibleMemories(input, RECALL_MAX_VECTOR_DISTANCE);
1267
1332
  },
1268
1333
  async searchMemories(input) {
1269
1334
  return await retrieveVisibleMemories(input, void 0);
@@ -3143,8 +3208,7 @@ async function createMemoryPromptContributions(context) {
3143
3208
  }
3144
3209
  } : void 0;
3145
3210
  const candidates = await createMemoryStore(context.db, runtimeContext, {
3146
- ...embedder ? { embedder } : {},
3147
- ...context.maxVectorDistance !== void 0 ? { maxVectorDistance: context.maxVectorDistance } : {}
3211
+ embedder
3148
3212
  }).recallMemories({
3149
3213
  query: context.text,
3150
3214
  limit: RECALL_CANDIDATE_LIMIT
@@ -3456,8 +3520,6 @@ function createMemoryUserPage() {
3456
3520
 
3457
3521
  // src/plugin.ts
3458
3522
  var MEMORY_MODEL_ENV = "AI_MEMORY_MODEL";
3459
- var MEMORY_RECALL_MAX_VECTOR_DISTANCE_ENV = "MEMORY_RECALL_MAX_VECTOR_DISTANCE";
3460
- var DEFAULT_RECALL_MAX_VECTOR_DISTANCE = 0.45;
3461
3523
  function memoryModelId(options) {
3462
3524
  const explicitModelId = options.modelId?.trim();
3463
3525
  if (explicitModelId) {
@@ -3466,19 +3528,6 @@ function memoryModelId(options) {
3466
3528
  const envModelId = process.env[MEMORY_MODEL_ENV]?.trim();
3467
3529
  return envModelId || void 0;
3468
3530
  }
3469
- function recallMaxVectorDistance(options) {
3470
- if (options.recallMaxVectorDistance !== void 0) {
3471
- return options.recallMaxVectorDistance;
3472
- }
3473
- const raw = process.env[MEMORY_RECALL_MAX_VECTOR_DISTANCE_ENV]?.trim();
3474
- if (raw) {
3475
- const parsed = Number(raw);
3476
- if (Number.isFinite(parsed) && parsed > 0) {
3477
- return Math.min(parsed, 1);
3478
- }
3479
- }
3480
- return DEFAULT_RECALL_MAX_VECTOR_DISTANCE;
3481
- }
3482
3531
  function memoryToolContext(ctx) {
3483
3532
  return {
3484
3533
  agent: ctx.agent,
@@ -3514,7 +3563,7 @@ function memoryPlugin(options = {}) {
3514
3563
  cli: {
3515
3564
  commands: [createMemoryCliCommand()]
3516
3565
  },
3517
- tasks: {
3566
+ tasks: options.disableExtraction ? {} : {
3518
3567
  processSession: {
3519
3568
  async run(ctx) {
3520
3569
  await processMemorySession(ctx);
@@ -3564,20 +3613,21 @@ function memoryPlugin(options = {}) {
3564
3613
  searchMemories: createMemorySearchTool(context)
3565
3614
  };
3566
3615
  },
3567
- async userPrompt(ctx) {
3568
- return await createMemoryPromptContributions({
3569
- agent: createMemoryAgent(ctx.model),
3570
- ...ctx.conversationId ? { conversationId: ctx.conversationId } : {},
3571
- ...ctx.actor ? { actor: ctx.actor } : {},
3572
- db: ctx.db,
3573
- embedder: ctx.embedder,
3574
- events: ctx.events,
3575
- log: ctx.log,
3576
- maxVectorDistance: recallMaxVectorDistance(options),
3577
- source: ctx.source,
3578
- text: ctx.text
3579
- });
3580
- }
3616
+ ...!options.disableRecall ? {
3617
+ async userPrompt(ctx) {
3618
+ return await createMemoryPromptContributions({
3619
+ agent: createMemoryAgent(ctx.model),
3620
+ ...ctx.conversationId ? { conversationId: ctx.conversationId } : {},
3621
+ ...ctx.actor ? { actor: ctx.actor } : {},
3622
+ db: ctx.db,
3623
+ embedder: ctx.embedder,
3624
+ events: ctx.events,
3625
+ log: ctx.log,
3626
+ source: ctx.source,
3627
+ text: ctx.text
3628
+ });
3629
+ }
3630
+ } : {}
3581
3631
  }
3582
3632
  });
3583
3633
  }