@sentry/junior-memory 0.132.0 → 0.134.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/recall.ts CHANGED
@@ -29,8 +29,6 @@ export interface MemoryRecallContext {
29
29
  embedder?: MemoryEmbeddingProvider;
30
30
  events?: PluginConversationEvents;
31
31
  log: PluginLogger;
32
- /** Maximum cosine distance for vector recall. Passed through to the memory store. */
33
- maxVectorDistance?: number;
34
32
  actor?: Actor;
35
33
  source: Source;
36
34
  text: string;
@@ -159,10 +157,7 @@ export async function createMemoryPromptContributions(
159
157
  }
160
158
  : undefined;
161
159
  const candidates = await createMemoryStore(context.db, runtimeContext, {
162
- ...(embedder ? { embedder } : {}),
163
- ...(context.maxVectorDistance !== undefined
164
- ? { maxVectorDistance: context.maxVectorDistance }
165
- : {}),
160
+ embedder,
166
161
  }).recallMemories({
167
162
  query: context.text,
168
163
  limit: RECALL_CANDIDATE_LIMIT,
package/src/store.ts CHANGED
@@ -51,9 +51,31 @@ const DEFAULT_SEARCH_LIMIT = 10;
51
51
  const DEFAULT_EXPIRED_ARCHIVE_LIMIT = 100;
52
52
  const PREFERENCE_ADJUDICATION_CANDIDATE_LIMIT = 10;
53
53
  const PREFERENCE_ADJUDICATION_VECTOR_LIMIT = 5;
54
- const VECTOR_SEARCH_OVERFETCH = 4;
54
+ /** Explicit search overfetch: keep a wider fusion window for tool/CLI search. */
55
+ const SEARCH_RETRIEVAL_OVERFETCH = 4;
56
+ /**
57
+ * Automatic recall overfetch. Recall already asks for ~20 candidates before the
58
+ * relevance gate, so each hybrid leg only needs a small top-k probe.
59
+ */
60
+ const RECALL_RETRIEVAL_OVERFETCH = 2;
61
+ /**
62
+ * Absolute ceiling per retrieval leg. Matches the store limit ceiling so a
63
+ * single healthy leg can still fill the caller's requested result window.
64
+ */
65
+ const MAX_RETRIEVAL_LEG_CANDIDATES = 200;
66
+ /** Cap ts_rank_cd work after GIN filtering; ranking is not indexable. */
67
+ const MAX_LEXICAL_RANK_CANDIDATES = 200;
68
+ /** Expand the GIN match window before ts_rank_cd, still under the hard cap. */
69
+ const LEXICAL_RANK_WINDOW_MULTIPLIER = 4;
70
+ /** Bound query text before embedding / FTS construction. */
71
+ const MAX_RETRIEVAL_QUERY_CHARS = 1_500;
55
72
  const MAX_MEMORY_CONTENT_CHARS = 4_000;
56
73
  const EMBEDDING_METRIC = "cosine";
74
+ /**
75
+ * Cosine distance cutoff for automatic recall only (not explicit search).
76
+ * Tuned for text-embedding-3-small; retune if the embedding model changes.
77
+ */
78
+ const RECALL_MAX_VECTOR_DISTANCE = 0.45;
57
79
 
58
80
  export type MemoryDb = PgDatabase<PgQueryResultHKT, typeof memorySqlSchema>;
59
81
 
@@ -299,8 +321,6 @@ export interface MemorySupersessionDecider {
299
321
 
300
322
  export interface MemoryStoreOptions {
301
323
  embedder?: MemoryEmbeddingProvider;
302
- /** Maximum cosine distance for vector recall candidates. Model-dependent; tune when changing AI_EMBEDDING_MODEL. */
303
- maxVectorDistance?: number;
304
324
  now?: () => number;
305
325
  supersessionDecider?: MemorySupersessionDecider;
306
326
  }
@@ -947,7 +967,26 @@ async function listVisibleMemories(args: {
947
967
  return rows.map(parseMemoryRow);
948
968
  }
949
969
 
950
- /** Search active visible records with PostgreSQL full-text ranking. */
970
+ function normalizeRetrievalQuery(query: string): string {
971
+ const normalized = query.replace(/\s+/g, " ").trim();
972
+ if (normalized.length <= MAX_RETRIEVAL_QUERY_CHARS) {
973
+ return normalized;
974
+ }
975
+ return normalized.slice(0, MAX_RETRIEVAL_QUERY_CHARS).trimEnd();
976
+ }
977
+
978
+ function retrievalLegLimit(limit: number, overfetch: number): number {
979
+ const requested = Math.max(1, limit);
980
+ const withOverfetch = requested * Math.max(1, overfetch);
981
+ // Never return fewer candidates than the caller asked for. A hard overfetch
982
+ // cap below `limit` under-fills when one modality is empty or both overlap.
983
+ return Math.min(
984
+ MAX_RETRIEVAL_LEG_CANDIDATES,
985
+ Math.max(requested, withOverfetch),
986
+ );
987
+ }
988
+
989
+ /** Search a bounded active candidate set with PostgreSQL full-text ranking. */
951
990
  async function searchVisibleLexicalMemories(args: {
952
991
  db: MemoryDb;
953
992
  limit: number;
@@ -959,7 +998,11 @@ async function searchVisibleLexicalMemories(args: {
959
998
  if (!predicate) {
960
999
  return [];
961
1000
  }
962
- const queryVector = sql`to_tsvector('english', ${args.query})`;
1001
+ const query = normalizeRetrievalQuery(args.query);
1002
+ if (!query) {
1003
+ return [];
1004
+ }
1005
+ const queryVector = sql`to_tsvector('english', ${query})`;
963
1006
  const tsquery = sql`(
964
1007
  SELECT COALESCE(
965
1008
  string_agg(quote_literal(term), ' | ')::tsquery,
@@ -967,20 +1010,50 @@ async function searchVisibleLexicalMemories(args: {
967
1010
  )
968
1011
  FROM unnest(tsvector_to_array(${queryVector})) AS query_terms(term)
969
1012
  )`;
970
- const textsearch = juniorMemoryMemories.searchVector;
971
- const textRank = sql<number>`ts_rank_cd(${textsearch}, ${tsquery})`;
972
- const rows = await args.db
973
- .select({
974
- memory: juniorMemoryMemories,
975
- textRank,
976
- })
1013
+ // GIN filter first, then rank only a bounded recent match window.
1014
+ const candidateLimit = Math.min(
1015
+ MAX_LEXICAL_RANK_CANDIDATES,
1016
+ args.limit * LEXICAL_RANK_WINDOW_MULTIPLIER,
1017
+ );
1018
+ const candidates = args.db
1019
+ .select()
977
1020
  .from(juniorMemoryMemories)
978
- .where(and(predicate, sql`${textsearch} @@ ${tsquery}`))
1021
+ .where(
1022
+ and(predicate, sql`${juniorMemoryMemories.searchVector} @@ ${tsquery}`),
1023
+ )
979
1024
  .orderBy(
980
- desc(textRank),
981
1025
  desc(juniorMemoryMemories.observedAtMs),
982
1026
  asc(juniorMemoryMemories.id),
983
1027
  )
1028
+ .limit(candidateLimit)
1029
+ .as("lexical_candidates");
1030
+ const textRank = sql<number>`ts_rank_cd(${candidates.searchVector}, ${tsquery})`;
1031
+ const rows = await args.db
1032
+ .select({
1033
+ memory: {
1034
+ archiveReason: candidates.archiveReason,
1035
+ archivedAtMs: candidates.archivedAtMs,
1036
+ content: candidates.content,
1037
+ createdAtMs: candidates.createdAtMs,
1038
+ expiresAtMs: candidates.expiresAtMs,
1039
+ id: candidates.id,
1040
+ idempotencyKey: candidates.idempotencyKey,
1041
+ kind: candidates.kind,
1042
+ observedAtMs: candidates.observedAtMs,
1043
+ scope: candidates.scope,
1044
+ scopeKey: candidates.scopeKey,
1045
+ searchVector: candidates.searchVector,
1046
+ sourceKey: candidates.sourceKey,
1047
+ sourcePlatform: candidates.sourcePlatform,
1048
+ subjectKey: candidates.subjectKey,
1049
+ subjectType: candidates.subjectType,
1050
+ supersededAtMs: candidates.supersededAtMs,
1051
+ supersededById: candidates.supersededById,
1052
+ },
1053
+ textRank,
1054
+ })
1055
+ .from(candidates)
1056
+ .orderBy(desc(textRank), desc(candidates.observedAtMs), asc(candidates.id))
984
1057
  .limit(args.limit);
985
1058
  const ranks = denseRanks(rows, (row) => Number(row.textRank));
986
1059
  return rows.map((row, index) => ({
@@ -990,7 +1063,7 @@ async function searchVisibleLexicalMemories(args: {
990
1063
  }));
991
1064
  }
992
1065
 
993
- /** Search active visible records with exact pgvector cosine distance. */
1066
+ /** Search active visible records with pgvector cosine distance. */
994
1067
  async function searchVisibleVectorMemories(args: {
995
1068
  db: MemoryDb;
996
1069
  embedder: MemoryEmbeddingProvider | undefined;
@@ -1007,9 +1080,13 @@ async function searchVisibleVectorMemories(args: {
1007
1080
  if (!predicate) {
1008
1081
  return [];
1009
1082
  }
1083
+ const query = normalizeRetrievalQuery(args.query);
1084
+ if (!query) {
1085
+ return [];
1086
+ }
1010
1087
  let embedding: Awaited<ReturnType<typeof embedOne>>;
1011
1088
  try {
1012
- embedding = await embedOne(args.embedder, args.query);
1089
+ embedding = await embedOne(args.embedder, query);
1013
1090
  } catch {
1014
1091
  return [];
1015
1092
  }
@@ -1017,6 +1094,11 @@ async function searchVisibleVectorMemories(args: {
1017
1094
  juniorMemoryEmbeddings.embedding,
1018
1095
  embedding.vector,
1019
1096
  );
1097
+ // Push distance cutoff into SQL so recall does not overfetch weak neighbors.
1098
+ const distancePredicate =
1099
+ args.maxDistance === undefined
1100
+ ? undefined
1101
+ : sql`${distance} <= ${args.maxDistance}`;
1020
1102
  const rows = await args.db
1021
1103
  .select({
1022
1104
  contentHash: juniorMemoryEmbeddings.contentHash,
@@ -1035,6 +1117,7 @@ async function searchVisibleVectorMemories(args: {
1035
1117
  eq(juniorMemoryEmbeddings.model, embedding.model),
1036
1118
  eq(juniorMemoryEmbeddings.dimensions, MEMORY_EMBEDDING_DIMENSIONS),
1037
1119
  eq(juniorMemoryEmbeddings.metric, EMBEDDING_METRIC),
1120
+ ...(distancePredicate ? [distancePredicate] : []),
1038
1121
  ),
1039
1122
  )
1040
1123
  .orderBy(
@@ -1053,9 +1136,6 @@ async function searchVisibleVectorMemories(args: {
1053
1136
  ) {
1054
1137
  return [];
1055
1138
  }
1056
- if (args.maxDistance !== undefined && distanceValue > args.maxDistance) {
1057
- return [];
1058
- }
1059
1139
  return [
1060
1140
  {
1061
1141
  memory: parseMemoryRow(row.memory),
@@ -1077,7 +1157,6 @@ export function createMemoryStore(
1077
1157
  const runtimeContext = memoryRuntimeContextSchema.parse(context);
1078
1158
  const parsedOptions = memoryStoreOptionsSchema.parse({ now: options.now });
1079
1159
  const embedder = options.embedder;
1080
- const maxVectorDistance = options.maxVectorDistance;
1081
1160
  const supersessionDecider = options.supersessionDecider;
1082
1161
  const getNowMs = parsedOptions.now ?? Date.now;
1083
1162
 
@@ -1323,6 +1402,14 @@ export function createMemoryStore(
1323
1402
  : { created: false, memory: idempotent.memory };
1324
1403
  }
1325
1404
 
1405
+ /**
1406
+ * Hybrid retrieval for both automatic recall and explicit search.
1407
+ *
1408
+ * Keep both legs parallel and fuse ranks with RRF. Never skip lexical when
1409
+ * vectors already hit: that drops exact/token memories and serializes the
1410
+ * miss path. Each leg is a hard-capped top-k probe so Postgres work stays
1411
+ * bounded even on broad queries.
1412
+ */
1326
1413
  async function retrieveVisibleMemories(
1327
1414
  rawInput: SearchMemoriesInput,
1328
1415
  vectorMaxDistance: number | undefined,
@@ -1336,7 +1423,13 @@ export function createMemoryStore(
1336
1423
  scopes,
1337
1424
  });
1338
1425
  const limit = boundedLimit(input.limit, DEFAULT_SEARCH_LIMIT);
1339
- const candidateLimit = limit * VECTOR_SEARCH_OVERFETCH;
1426
+ const overfetch =
1427
+ vectorMaxDistance === undefined
1428
+ ? SEARCH_RETRIEVAL_OVERFETCH
1429
+ : RECALL_RETRIEVAL_OVERFETCH;
1430
+ const candidateLimit = retrievalLegLimit(limit, overfetch);
1431
+ // Always run both legs in parallel. Conditional lexical skip is unsafe:
1432
+ // one in-threshold vector distractor can hide a stronger lexical hit.
1340
1433
  const [vectorMatches, lexicalMatches] = await Promise.all([
1341
1434
  searchVisibleVectorMemories({
1342
1435
  db,
@@ -1360,6 +1453,10 @@ export function createMemoryStore(
1360
1453
  const channelPrefix = sourceChannelPrefix(runtimeContext);
1361
1454
  return rankMemoryMatches([...vectorMatches, ...lexicalMatches], {
1362
1455
  nowMs,
1456
+ // Slight lexical preference protects exact ids/names/timezones on ties.
1457
+ ...(vectorMaxDistance === undefined
1458
+ ? {}
1459
+ : { lexicalWeight: 1, vectorWeight: 0.85 }),
1363
1460
  ...(channelPrefix ? { channelPrefix } : {}),
1364
1461
  })
1365
1462
  .slice(0, limit)
@@ -1414,7 +1511,7 @@ export function createMemoryStore(
1414
1511
  },
1415
1512
 
1416
1513
  async recallMemories(input) {
1417
- return await retrieveVisibleMemories(input, maxVectorDistance);
1514
+ return await retrieveVisibleMemories(input, RECALL_MAX_VECTOR_DISTANCE);
1418
1515
  },
1419
1516
 
1420
1517
  async searchMemories(input) {