@sentry/junior-memory 0.133.0 → 0.135.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/ranking.ts CHANGED
@@ -2,6 +2,7 @@ import type { MemoryRecord } from "./store";
2
2
 
3
3
  const RECIPROCAL_RANK_FUSION_K = 60;
4
4
  const ONE_DAY_MS = 24 * 60 * 60 * 1000;
5
+ const DEFAULT_RRF_WEIGHT = 1;
5
6
 
6
7
  export interface MemoryMatch {
7
8
  lexical?: {
@@ -14,14 +15,21 @@ export interface MemoryMatch {
14
15
  };
15
16
  }
16
17
 
17
- function reciprocalRank(rank: number): number {
18
- return 1 / (RECIPROCAL_RANK_FUSION_K + rank);
18
+ function reciprocalRank(rank: number, weight: number): number {
19
+ return weight / (RECIPROCAL_RANK_FUSION_K + rank);
19
20
  }
20
21
 
21
- function matchScore(match: MemoryMatch): number {
22
+ function matchScore(
23
+ match: MemoryMatch,
24
+ weights: { lexicalWeight: number; vectorWeight: number },
25
+ ): number {
22
26
  return (
23
- (match.vector ? reciprocalRank(match.vector.rank) : 0) +
24
- (match.lexical ? reciprocalRank(match.lexical.rank) : 0)
27
+ (match.vector
28
+ ? reciprocalRank(match.vector.rank, weights.vectorWeight)
29
+ : 0) +
30
+ (match.lexical
31
+ ? reciprocalRank(match.lexical.rank, weights.lexicalWeight)
32
+ : 0)
25
33
  );
26
34
  }
27
35
 
@@ -46,14 +54,28 @@ function observedAgeRank(memory: MemoryRecord, nowMs: number): number {
46
54
  return 0;
47
55
  }
48
56
 
57
+ function positiveWeight(value: number | undefined, fallback: number): number {
58
+ return value !== undefined && Number.isFinite(value) && value > 0
59
+ ? value
60
+ : fallback;
61
+ }
62
+
49
63
  /** Fuse lexical and vector ranks without comparing provider raw scores. */
50
64
  export function rankMemoryMatches(
51
65
  matches: MemoryMatch[],
52
66
  options: {
53
67
  channelPrefix?: string;
68
+ /** Optional RRF weight for the lexical leg. Defaults to 1. */
69
+ lexicalWeight?: number;
54
70
  nowMs: number;
71
+ /** Optional RRF weight for the vector leg. Defaults to 1. */
72
+ vectorWeight?: number;
55
73
  },
56
74
  ): MemoryMatch[] {
75
+ const weights = {
76
+ lexicalWeight: positiveWeight(options.lexicalWeight, DEFAULT_RRF_WEIGHT),
77
+ vectorWeight: positiveWeight(options.vectorWeight, DEFAULT_RRF_WEIGHT),
78
+ };
57
79
  const byId = new Map<string, MemoryMatch>();
58
80
  for (const match of matches) {
59
81
  const existing = byId.get(match.memory.id);
@@ -61,17 +83,31 @@ export function rankMemoryMatches(
61
83
  byId.set(match.memory.id, match);
62
84
  continue;
63
85
  }
86
+ // Keep the first rank per modality. Shared legs are fused before personal
87
+ // probes, so a smaller personal top-k cannot overwrite a shared dense rank
88
+ // with an inflated top rank for the same memory.
64
89
  byId.set(match.memory.id, {
65
90
  ...existing,
66
- ...(match.lexical ? { lexical: match.lexical } : {}),
67
- ...(match.vector ? { vector: match.vector } : {}),
91
+ ...(!existing.lexical && match.lexical
92
+ ? { lexical: match.lexical }
93
+ : {}),
94
+ ...(!existing.vector && match.vector ? { vector: match.vector } : {}),
68
95
  });
69
96
  }
70
97
  return [...byId.values()].sort((left, right) => {
71
- const scoreDelta = matchScore(right) - matchScore(left);
98
+ const scoreDelta = matchScore(right, weights) - matchScore(left, weights);
72
99
  if (scoreDelta !== 0) {
73
100
  return scoreDelta;
74
101
  }
102
+ // Prefer actor preferences over workspace knowledge when RRF ties. Shared
103
+ // lexical legs often assign the same top rank to recent conversation noise
104
+ // and a personal-scope probe hit for the same common token.
105
+ const personalDelta =
106
+ Number(right.memory.scope === "personal") -
107
+ Number(left.memory.scope === "personal");
108
+ if (personalDelta !== 0) {
109
+ return personalDelta;
110
+ }
75
111
  const channelDelta =
76
112
  Number(currentChannel(right, options.channelPrefix)) -
77
113
  Number(currentChannel(left, options.channelPrefix));
package/src/recall.ts CHANGED
@@ -29,8 +29,6 @@ export interface MemoryRecallContext {
29
29
  embedder?: MemoryEmbeddingProvider;
30
30
  events?: PluginConversationEvents;
31
31
  log: PluginLogger;
32
- /** Maximum cosine distance for vector recall. Passed through to the memory store. */
33
- maxVectorDistance?: number;
34
32
  actor?: Actor;
35
33
  source: Source;
36
34
  text: string;
@@ -159,10 +157,7 @@ export async function createMemoryPromptContributions(
159
157
  }
160
158
  : undefined;
161
159
  const candidates = await createMemoryStore(context.db, runtimeContext, {
162
- ...(embedder ? { embedder } : {}),
163
- ...(context.maxVectorDistance !== undefined
164
- ? { maxVectorDistance: context.maxVectorDistance }
165
- : {}),
160
+ embedder,
166
161
  }).recallMemories({
167
162
  query: context.text,
168
163
  limit: RECALL_CANDIDATE_LIMIT,
package/src/store.ts CHANGED
@@ -51,11 +51,31 @@ const DEFAULT_SEARCH_LIMIT = 10;
51
51
  const DEFAULT_EXPIRED_ARCHIVE_LIMIT = 100;
52
52
  const PREFERENCE_ADJUDICATION_CANDIDATE_LIMIT = 10;
53
53
  const PREFERENCE_ADJUDICATION_VECTOR_LIMIT = 5;
54
- const VECTOR_SEARCH_OVERFETCH = 4;
55
- const LEXICAL_RANK_OVERFETCH = 4;
56
- const MAX_LEXICAL_RANK_CANDIDATES = 1_000;
54
+ /** Explicit search overfetch: keep a wider fusion window for tool/CLI search. */
55
+ const SEARCH_RETRIEVAL_OVERFETCH = 4;
56
+ /**
57
+ * Automatic recall overfetch. Recall already asks for ~20 candidates before the
58
+ * relevance gate, so each hybrid leg only needs a small top-k probe.
59
+ */
60
+ const RECALL_RETRIEVAL_OVERFETCH = 2;
61
+ /**
62
+ * Absolute ceiling per retrieval leg. Matches the store limit ceiling so a
63
+ * single healthy leg can still fill the caller's requested result window.
64
+ */
65
+ const MAX_RETRIEVAL_LEG_CANDIDATES = 200;
66
+ /** Cap ts_rank_cd work after GIN filtering; ranking is not indexable. */
67
+ const MAX_LEXICAL_RANK_CANDIDATES = 200;
68
+ /** Expand the GIN match window before ts_rank_cd, still under the hard cap. */
69
+ const LEXICAL_RANK_WINDOW_MULTIPLIER = 4;
70
+ /** Bound query text before embedding / FTS construction. */
71
+ const MAX_RETRIEVAL_QUERY_CHARS = 1_500;
57
72
  const MAX_MEMORY_CONTENT_CHARS = 4_000;
58
73
  const EMBEDDING_METRIC = "cosine";
74
+ /**
75
+ * Cosine distance cutoff for automatic recall only (not explicit search).
76
+ * Tuned for text-embedding-3-small; retune if the embedding model changes.
77
+ */
78
+ const RECALL_MAX_VECTOR_DISTANCE = 0.45;
59
79
 
60
80
  export type MemoryDb = PgDatabase<PgQueryResultHKT, typeof memorySqlSchema>;
61
81
 
@@ -301,8 +321,6 @@ export interface MemorySupersessionDecider {
301
321
 
302
322
  export interface MemoryStoreOptions {
303
323
  embedder?: MemoryEmbeddingProvider;
304
- /** Maximum cosine distance for vector recall candidates. Model-dependent; tune when changing AI_EMBEDDING_MODEL. */
305
- maxVectorDistance?: number;
306
324
  now?: () => number;
307
325
  supersessionDecider?: MemorySupersessionDecider;
308
326
  }
@@ -949,6 +967,25 @@ async function listVisibleMemories(args: {
949
967
  return rows.map(parseMemoryRow);
950
968
  }
951
969
 
970
+ function normalizeRetrievalQuery(query: string): string {
971
+ const normalized = query.replace(/\s+/g, " ").trim();
972
+ if (normalized.length <= MAX_RETRIEVAL_QUERY_CHARS) {
973
+ return normalized;
974
+ }
975
+ return normalized.slice(0, MAX_RETRIEVAL_QUERY_CHARS).trimEnd();
976
+ }
977
+
978
+ function retrievalLegLimit(limit: number, overfetch: number): number {
979
+ const requested = Math.max(1, limit);
980
+ const withOverfetch = requested * Math.max(1, overfetch);
981
+ // Never return fewer candidates than the caller asked for. A hard overfetch
982
+ // cap below `limit` under-fills when one modality is empty or both overlap.
983
+ return Math.min(
984
+ MAX_RETRIEVAL_LEG_CANDIDATES,
985
+ Math.max(requested, withOverfetch),
986
+ );
987
+ }
988
+
952
989
  /** Search a bounded active candidate set with PostgreSQL full-text ranking. */
953
990
  async function searchVisibleLexicalMemories(args: {
954
991
  db: MemoryDb;
@@ -961,7 +998,11 @@ async function searchVisibleLexicalMemories(args: {
961
998
  if (!predicate) {
962
999
  return [];
963
1000
  }
964
- const queryVector = sql`to_tsvector('english', ${args.query})`;
1001
+ const query = normalizeRetrievalQuery(args.query);
1002
+ if (!query) {
1003
+ return [];
1004
+ }
1005
+ const queryVector = sql`to_tsvector('english', ${query})`;
965
1006
  const tsquery = sql`(
966
1007
  SELECT COALESCE(
967
1008
  string_agg(quote_literal(term), ' | ')::tsquery,
@@ -969,9 +1010,10 @@ async function searchVisibleLexicalMemories(args: {
969
1010
  )
970
1011
  FROM unnest(tsvector_to_array(${queryVector})) AS query_terms(term)
971
1012
  )`;
1013
+ // GIN filter first, then rank only a bounded recent match window.
972
1014
  const candidateLimit = Math.min(
973
1015
  MAX_LEXICAL_RANK_CANDIDATES,
974
- args.limit * LEXICAL_RANK_OVERFETCH,
1016
+ args.limit * LEXICAL_RANK_WINDOW_MULTIPLIER,
975
1017
  );
976
1018
  const candidates = args.db
977
1019
  .select()
@@ -1021,33 +1063,29 @@ async function searchVisibleLexicalMemories(args: {
1021
1063
  }));
1022
1064
  }
1023
1065
 
1024
- /** Search active visible records with exact pgvector cosine distance. */
1066
+ /** Search active visible records with pgvector cosine distance. */
1025
1067
  async function searchVisibleVectorMemories(args: {
1026
1068
  db: MemoryDb;
1027
- embedder: MemoryEmbeddingProvider | undefined;
1069
+ embedding: MemoryEmbedding;
1028
1070
  limit: number;
1029
1071
  maxDistance?: number;
1030
1072
  nowMs: number;
1031
- query: string;
1032
1073
  scopes: ResolvedMemoryScope[];
1033
1074
  }): Promise<MemoryMatch[]> {
1034
- if (!args.embedder) {
1035
- return [];
1036
- }
1037
1075
  const predicate = activeVisiblePredicate(args);
1038
1076
  if (!predicate) {
1039
1077
  return [];
1040
1078
  }
1041
- let embedding: Awaited<ReturnType<typeof embedOne>>;
1042
- try {
1043
- embedding = await embedOne(args.embedder, args.query);
1044
- } catch {
1045
- return [];
1046
- }
1079
+ const embedding = args.embedding;
1047
1080
  const distance = cosineDistance(
1048
1081
  juniorMemoryEmbeddings.embedding,
1049
1082
  embedding.vector,
1050
1083
  );
1084
+ // Push distance cutoff into SQL so recall does not overfetch weak neighbors.
1085
+ const distancePredicate =
1086
+ args.maxDistance === undefined
1087
+ ? undefined
1088
+ : sql`${distance} <= ${args.maxDistance}`;
1051
1089
  const rows = await args.db
1052
1090
  .select({
1053
1091
  contentHash: juniorMemoryEmbeddings.contentHash,
@@ -1066,6 +1104,7 @@ async function searchVisibleVectorMemories(args: {
1066
1104
  eq(juniorMemoryEmbeddings.model, embedding.model),
1067
1105
  eq(juniorMemoryEmbeddings.dimensions, MEMORY_EMBEDDING_DIMENSIONS),
1068
1106
  eq(juniorMemoryEmbeddings.metric, EMBEDDING_METRIC),
1107
+ ...(distancePredicate ? [distancePredicate] : []),
1069
1108
  ),
1070
1109
  )
1071
1110
  .orderBy(
@@ -1084,9 +1123,6 @@ async function searchVisibleVectorMemories(args: {
1084
1123
  ) {
1085
1124
  return [];
1086
1125
  }
1087
- if (args.maxDistance !== undefined && distanceValue > args.maxDistance) {
1088
- return [];
1089
- }
1090
1126
  return [
1091
1127
  {
1092
1128
  memory: parseMemoryRow(row.memory),
@@ -1108,7 +1144,6 @@ export function createMemoryStore(
1108
1144
  const runtimeContext = memoryRuntimeContextSchema.parse(context);
1109
1145
  const parsedOptions = memoryStoreOptionsSchema.parse({ now: options.now });
1110
1146
  const embedder = options.embedder;
1111
- const maxVectorDistance = options.maxVectorDistance;
1112
1147
  const supersessionDecider = options.supersessionDecider;
1113
1148
  const getNowMs = parsedOptions.now ?? Date.now;
1114
1149
 
@@ -1354,6 +1389,19 @@ export function createMemoryStore(
1354
1389
  : { created: false, memory: idempotent.memory };
1355
1390
  }
1356
1391
 
1392
+ /**
1393
+ * Hybrid retrieval for both automatic recall and explicit search.
1394
+ *
1395
+ * Keep both legs parallel and fuse ranks with RRF. Never skip lexical when
1396
+ * vectors already hit: that drops exact/token memories and serializes the
1397
+ * miss path. Each leg is a hard-capped top-k probe so Postgres work stays
1398
+ * bounded even on broad queries.
1399
+ *
1400
+ * Automatic recall also runs personal-scope-only probes. Workspace
1401
+ * conversation memories sharing common tokens (for example "time") can fill
1402
+ * the shared lexical recency window before ranking, which buries older actor
1403
+ * preferences that explicit search still finds.
1404
+ */
1357
1405
  async function retrieveVisibleMemories(
1358
1406
  rawInput: SearchMemoriesInput,
1359
1407
  vectorMaxDistance: number | undefined,
@@ -1367,30 +1415,76 @@ export function createMemoryStore(
1367
1415
  scopes,
1368
1416
  });
1369
1417
  const limit = boundedLimit(input.limit, DEFAULT_SEARCH_LIMIT);
1370
- const candidateLimit = limit * VECTOR_SEARCH_OVERFETCH;
1371
- const [vectorMatches, lexicalMatches] = await Promise.all([
1372
- searchVisibleVectorMemories({
1373
- db,
1374
- embedder,
1375
- limit: candidateLimit,
1376
- ...(vectorMaxDistance !== undefined
1377
- ? { maxDistance: vectorMaxDistance }
1378
- : {}),
1379
- nowMs,
1380
- query: input.query,
1381
- scopes,
1382
- }),
1418
+ const overfetch =
1419
+ vectorMaxDistance === undefined
1420
+ ? SEARCH_RETRIEVAL_OVERFETCH
1421
+ : RECALL_RETRIEVAL_OVERFETCH;
1422
+ const candidateLimit = retrievalLegLimit(limit, overfetch);
1423
+ const personalScopes = scopes.filter((scope) => scope.scope === "personal");
1424
+ // Automatic recall only: keep a personal-scope probe so workspace noise
1425
+ // cannot monopolize the shared lexical recency window.
1426
+ const probePersonal =
1427
+ vectorMaxDistance !== undefined && personalScopes.length > 0;
1428
+ const query = normalizeRetrievalQuery(input.query);
1429
+ let queryEmbedding: MemoryEmbedding | undefined;
1430
+ if (embedder && query) {
1431
+ try {
1432
+ queryEmbedding = await embedOne(embedder, query);
1433
+ } catch {
1434
+ queryEmbedding = undefined;
1435
+ }
1436
+ }
1437
+ const emptyMatches = Promise.resolve([] as MemoryMatch[]);
1438
+ const lexicalArgs = {
1439
+ db,
1440
+ limit: candidateLimit,
1441
+ nowMs,
1442
+ query: input.query,
1443
+ };
1444
+ // Always run both legs in parallel. Conditional lexical skip is unsafe:
1445
+ // one in-threshold vector distractor can hide a stronger lexical hit.
1446
+ // Embed once up front; vector probes only run when that embedding exists.
1447
+ const matches = await Promise.all([
1448
+ queryEmbedding
1449
+ ? searchVisibleVectorMemories({
1450
+ db,
1451
+ embedding: queryEmbedding,
1452
+ limit: candidateLimit,
1453
+ ...(vectorMaxDistance !== undefined
1454
+ ? { maxDistance: vectorMaxDistance }
1455
+ : {}),
1456
+ nowMs,
1457
+ scopes,
1458
+ })
1459
+ : emptyMatches,
1383
1460
  searchVisibleLexicalMemories({
1384
- db,
1385
- limit: candidateLimit,
1386
- nowMs,
1387
- query: input.query,
1461
+ ...lexicalArgs,
1388
1462
  scopes,
1389
1463
  }),
1464
+ queryEmbedding && probePersonal
1465
+ ? searchVisibleVectorMemories({
1466
+ db,
1467
+ embedding: queryEmbedding,
1468
+ limit: candidateLimit,
1469
+ maxDistance: vectorMaxDistance,
1470
+ nowMs,
1471
+ scopes: personalScopes,
1472
+ })
1473
+ : emptyMatches,
1474
+ probePersonal
1475
+ ? searchVisibleLexicalMemories({
1476
+ ...lexicalArgs,
1477
+ scopes: personalScopes,
1478
+ })
1479
+ : emptyMatches,
1390
1480
  ]);
1391
1481
  const channelPrefix = sourceChannelPrefix(runtimeContext);
1392
- return rankMemoryMatches([...vectorMatches, ...lexicalMatches], {
1482
+ return rankMemoryMatches(matches.flat(), {
1393
1483
  nowMs,
1484
+ // Slight lexical preference protects exact ids/names/timezones on ties.
1485
+ ...(vectorMaxDistance === undefined
1486
+ ? {}
1487
+ : { lexicalWeight: 1, vectorWeight: 0.85 }),
1394
1488
  ...(channelPrefix ? { channelPrefix } : {}),
1395
1489
  })
1396
1490
  .slice(0, limit)
@@ -1445,7 +1539,7 @@ export function createMemoryStore(
1445
1539
  },
1446
1540
 
1447
1541
  async recallMemories(input) {
1448
- return await retrieveVisibleMemories(input, maxVectorDistance);
1542
+ return await retrieveVisibleMemories(input, RECALL_MAX_VECTOR_DISTANCE);
1449
1543
  },
1450
1544
 
1451
1545
  async searchMemories(input) {