@open-mercato/shared 0.6.7-develop.6788.1.c2a1520a30 → 0.6.7-develop.6814.1.0627c7e9f1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (54) hide show
  1. package/.turbo/turbo-build.log +1 -1
  2. package/AGENTS.md +1 -0
  3. package/dist/lib/auth/organizationScope.js +34 -0
  4. package/dist/lib/auth/organizationScope.js.map +7 -0
  5. package/dist/lib/di/container.js +33 -3
  6. package/dist/lib/di/container.js.map +2 -2
  7. package/dist/lib/email/send.js +56 -0
  8. package/dist/lib/email/send.js.map +2 -2
  9. package/dist/lib/query/engine.js +25 -42
  10. package/dist/lib/query/engine.js.map +2 -2
  11. package/dist/lib/query/types.js.map +1 -1
  12. package/dist/lib/search/availability.js +107 -0
  13. package/dist/lib/search/availability.js.map +7 -0
  14. package/dist/lib/search/config.js +24 -2
  15. package/dist/lib/search/config.js.map +2 -2
  16. package/dist/lib/search/tokenize.js +23 -11
  17. package/dist/lib/search/tokenize.js.map +2 -2
  18. package/dist/lib/version.js +1 -1
  19. package/dist/lib/version.js.map +1 -1
  20. package/dist/lib/webhooks/body.js +57 -0
  21. package/dist/lib/webhooks/body.js.map +7 -0
  22. package/dist/lib/webhooks/index.js +13 -1
  23. package/dist/lib/webhooks/index.js.map +2 -2
  24. package/dist/lib/webhooks/verify.js +12 -7
  25. package/dist/lib/webhooks/verify.js.map +2 -2
  26. package/dist/modules/payment_gateways/types.js +8 -1
  27. package/dist/modules/payment_gateways/types.js.map +2 -2
  28. package/dist/modules/widgets/extension-points.js.map +1 -1
  29. package/package.json +8 -8
  30. package/src/lib/auth/__tests__/organizationScope.test.ts +107 -0
  31. package/src/lib/auth/organizationScope.ts +65 -0
  32. package/src/lib/di/__tests__/container-app-di-absent.test.ts +139 -0
  33. package/src/lib/di/__tests__/container-app-di.test.ts +173 -0
  34. package/src/lib/di/container.ts +45 -4
  35. package/src/lib/email/__tests__/send.test.ts +37 -1
  36. package/src/lib/email/send.ts +79 -0
  37. package/src/lib/i18n/__tests__/server-dictionary-cache.test.ts +30 -0
  38. package/src/lib/query/__tests__/engine.test.ts +65 -13
  39. package/src/lib/query/engine.ts +36 -60
  40. package/src/lib/query/types.ts +4 -5
  41. package/src/lib/search/__tests__/availability.test.ts +211 -0
  42. package/src/lib/search/__tests__/config.test.ts +59 -0
  43. package/src/lib/search/__tests__/tokenize.test.ts +49 -0
  44. package/src/lib/search/availability.ts +232 -0
  45. package/src/lib/search/config.ts +28 -0
  46. package/src/lib/search/tokenize.ts +33 -11
  47. package/src/lib/webhooks/__tests__/body.test.ts +95 -0
  48. package/src/lib/webhooks/__tests__/verify.test.ts +20 -1
  49. package/src/lib/webhooks/body.ts +70 -0
  50. package/src/lib/webhooks/index.ts +7 -1
  51. package/src/lib/webhooks/verify.ts +17 -9
  52. package/src/modules/payment_gateways/__tests__/types.test.ts +37 -0
  53. package/src/modules/payment_gateways/types.ts +15 -0
  54. package/src/modules/widgets/extension-points.ts +104 -1
@@ -12,6 +12,13 @@ import {
12
12
  type ResolvedJoin,
13
13
  } from './join-utils'
14
14
  import { resolveSearchConfig } from '../search/config'
15
+ import {
16
+ createSearchTokenAvailability,
17
+ isSearchFilterOp,
18
+ type SearchTokenAvailability,
19
+ type SearchTokenProbeDb,
20
+ type SearchTokenProbeQueryBuilder,
21
+ } from '../search/availability'
15
22
  import { tokenizeText } from '../search/tokenize'
16
23
  import { runBeforeQueryPipeline, runAfterQueryPipeline, type QueryExtensionContext } from './query-extension-runner'
17
24
  import {
@@ -207,8 +214,8 @@ function computeCustomFieldScore(cfg: Record<string, unknown>, kind: string, ent
207
214
  */
208
215
  export class BasicQueryEngine implements QueryEngine {
209
216
  private columnCache = new Map<string, boolean>()
210
- private tableCache = new Map<string, boolean>()
211
217
  private searchAliasSeq = 0
218
+ private searchAvailabilityInstance: SearchTokenAvailability | null = null
212
219
 
213
220
  constructor(
214
221
  private em: EntityManager,
@@ -231,6 +238,22 @@ export class BasicQueryEngine implements QueryEngine {
231
238
  throw new Error('BasicQueryEngine requires an EntityManager exposing getKysely() (MikroORM v7)')
232
239
  }
233
240
 
241
+ private searchAvailability(): SearchTokenAvailability {
242
+ if (!this.searchAvailabilityInstance) {
243
+ this.searchAvailabilityInstance = createSearchTokenAvailability({
244
+ getDb: () => this.getDb() as unknown as SearchTokenProbeDb,
245
+ getConfig: resolveSearchConfig,
246
+ applyOrganizationScope: (query, column, scope) => this.applyOrganizationScope(
247
+ query as unknown as AnyBuilder,
248
+ column,
249
+ scope,
250
+ ) as unknown as SearchTokenProbeQueryBuilder,
251
+ logDebug: (event, payload) => this.logSearchDebug(event, payload),
252
+ })
253
+ }
254
+ return this.searchAvailabilityInstance
255
+ }
256
+
234
257
  async query<T = any>(entity: EntityId, opts: QueryOptions = {}): Promise<QueryResult<T>> {
235
258
  // --- UMES query extension: before-query pipeline ---
236
259
  const ext = opts.extensions
@@ -294,17 +317,20 @@ export class BasicQueryEngine implements QueryEngine {
294
317
  const { baseFilters, joinFilters } = partitionFilters(table, normalizedFilters, joinMap)
295
318
  const cfFilters = normalizedFilters.filter((filter) => String(filter.field).startsWith('cf:'))
296
319
  const searchConfig = resolveSearchConfig()
320
+ const searchFilters = [...baseFilters, ...cfFilters].filter((filter) => isSearchFilterOp(filter.op))
297
321
  // Callers that opt out of automatic tenant/org scoping own the full
298
322
  // visibility predicate. Search-token filtering has its own tenant/org
299
323
  // guards, so it must be disabled on this direct-query path as documented
300
324
  // by QueryOptions.omitAutomaticTenantOrgScope.
301
- const searchEnabled = !skipAutoScope && searchConfig.enabled && await this.tableExists('search_tokens')
302
- const hasSearchTokens = searchEnabled
303
- ? await this.hasSearchTokens(String(entity), opts.tenantId ?? null, orgScope)
325
+ const searchEnabled = !skipAutoScope && await this.searchAvailability().staticEnabled()
326
+ // Probe `search_tokens` only when this query actually searches (#4723): every consumer of
327
+ // `searchActive` sits behind a like/ilike guard, so on a plain list load the answer is never
328
+ // read — and the probe is a `LIMIT 1` the planner can resolve as a seq scan over a large
329
+ // `search_tokens`. The join path below already probes lazily for the same reason.
330
+ const hasSearchTokens = searchEnabled && searchFilters.length
331
+ ? await this.searchAvailability().hasTokens(String(entity), opts.tenantId ?? null, orgScope)
304
332
  : false
305
333
  const searchActive = searchEnabled && hasSearchTokens
306
- const joinSearchAvailability = new Map<string, boolean>()
307
- const searchFilters = [...baseFilters, ...cfFilters].filter((filter) => filter.op === 'like' || filter.op === 'ilike')
308
334
  if (searchFilters.length) {
309
335
  const fields = searchFilters.map((filter) => String(filter.field))
310
336
  this.logSearchDebug('search:init', {
@@ -358,9 +384,8 @@ export class BasicQueryEngine implements QueryEngine {
358
384
  const join = joinMap.get(alias)
359
385
  if (!join?.entityId) continue
360
386
  const hasJoinedTokens = searchEnabled
361
- ? await this.hasSearchTokens(join.entityId, opts.tenantId ?? null, orgScope)
387
+ ? await this.searchAvailability().hasTokens(join.entityId, opts.tenantId ?? null, orgScope)
362
388
  : false
363
- joinSearchAvailability.set(join.entityId, hasJoinedTokens)
364
389
  const fallbackFields = filters
365
390
  .filter((filter) => !hasJoinedTokens || typeof filter.value !== 'string' || tokenizeText(filter.value, searchConfig).hashes.length === 0)
366
391
  .map((filter) => filter.column)
@@ -437,11 +462,7 @@ export class BasicQueryEngine implements QueryEngine {
437
462
  if (!['like', 'ilike'].includes(filter.op)) return { applied: false, builder }
438
463
  if (typeof filter.value !== 'string' || filter.value.trim().length === 0) return { applied: false, builder }
439
464
 
440
- let searchAvailable = joinSearchAvailability.get(join.entityId)
441
- if (searchAvailable === undefined) {
442
- searchAvailable = await this.hasSearchTokens(join.entityId, opts.tenantId ?? null, orgScope)
443
- joinSearchAvailability.set(join.entityId, searchAvailable)
444
- }
465
+ const searchAvailable = await this.searchAvailability().hasTokens(join.entityId, opts.tenantId ?? null, orgScope)
445
466
  if (!searchAvailable) return { applied: false, builder }
446
467
 
447
468
  const tokens = tokenizeText(String(filter.value), searchConfig)
@@ -512,8 +533,8 @@ export class BasicQueryEngine implements QueryEngine {
512
533
  // today's complete selection (base fields + CF projections + extension joins).
513
534
  // `projection: 'sortKeys'` selects only `id` + the sort columns — the slim phase-1
514
535
  // candidate scan used when `requiresPlaintextSort`. Re-running the WHERE/JOIN logic
515
- // twice is cheap: every `columnExists`/`tableExists` check is memoized on
516
- // `this.columnCache`/`this.tableCache`, so the second pass hits no extra DB calls.
536
+ // twice is cheap: every `columnExists` check is memoized on `this.columnCache`,
537
+ // so the second pass hits no extra DB calls.
517
538
  const buildQuery = async (projection: 'full' | 'sortKeys'): Promise<BuiltQuery> => {
518
539
  const isSortKeysProjection = projection === 'sortKeys'
519
540
  let q: AnyBuilder = db.selectFrom(table as any)
@@ -1177,51 +1198,6 @@ export class BasicQueryEngine implements QueryEngine {
1177
1198
  return present
1178
1199
  }
1179
1200
 
1180
- private async tableExists(table: string): Promise<boolean> {
1181
- if (this.tableCache.has(table)) return this.tableCache.get(table) ?? false
1182
- const db = this.getDb()
1183
- const exists = await db
1184
- .selectFrom('information_schema.tables' as any)
1185
- .select(sql<number>`1`.as('one'))
1186
- .where('table_name' as any, '=', table)
1187
- .limit(1)
1188
- .executeTakeFirst()
1189
- const present = !!exists
1190
- this.tableCache.set(table, present)
1191
- return present
1192
- }
1193
-
1194
- private async hasSearchTokens(
1195
- entity: string,
1196
- tenantId: string | null,
1197
- orgScope?: { ids: string[]; includeNull: boolean } | null
1198
- ): Promise<boolean> {
1199
- try {
1200
- const db = this.getDb()
1201
- let query: AnyBuilder = db
1202
- .selectFrom('search_tokens' as any)
1203
- .select(sql<number>`1`.as('one'))
1204
- .where('entity_type' as any, '=', entity)
1205
- .limit(1)
1206
- if (tenantId !== undefined) {
1207
- query = query.where(sql<boolean>`tenant_id is not distinct from ${tenantId}`)
1208
- }
1209
- if (orgScope) {
1210
- query = this.applyOrganizationScope(query, 'search_tokens.organization_id', orgScope)
1211
- }
1212
- const row = await query.executeTakeFirst()
1213
- return !!row
1214
- } catch (err) {
1215
- this.logSearchDebug('search:has-tokens-error', {
1216
- entity,
1217
- tenantId,
1218
- organizationScope: orgScope,
1219
- error: err instanceof Error ? err.message : String(err),
1220
- })
1221
- return false
1222
- }
1223
- }
1224
-
1225
1201
  private applySearchTokens(
1226
1202
  q: AnyBuilder,
1227
1203
  opts: {
@@ -108,11 +108,10 @@ export type QueryOptions = {
108
108
  * and MUST fail closed when the authenticated principal lacks a resolvable tenant/org, otherwise
109
109
  * queries return cross-tenant rows.
110
110
  *
111
- * When this flag is set, the hybrid query engine delegates to the basic engine, which means
112
- * custom-field (`cf:*`) filters/sorts, `search_tokens` fulltext filtering, and vector-search
113
- * branches are BYPASSED. Only use this on entities whose scoping does not match the standard
114
- * `organization_id = X AND tenant_id = Y` shape and which do not rely on custom-field/search
115
- * features.
111
+ * When this flag is set, the hybrid query engine delegates to the basic engine. The basic engine
112
+ * still applies `cf:*` filters/sorts, but `search_tokens` fulltext filtering, the JSONB index read
113
+ * path, and the vector-search branch are BYPASSED. Only use this on entities whose scoping does
114
+ * not match the standard `organization_id = X AND tenant_id = Y` shape.
116
115
  */
117
116
  omitAutomaticTenantOrgScope?: boolean
118
117
  // Soft-delete behavior: when false (default), rows with non-null deleted_at
@@ -0,0 +1,211 @@
1
+ import {
2
+ clearSearchTokenPresenceCache,
3
+ createSearchTokenAvailability,
4
+ hasSearchFilter,
5
+ isSearchFilterOp,
6
+ type OrganizationScope,
7
+ type SearchTokenProbeQueryBuilder,
8
+ } from '../availability'
9
+
10
+ beforeEach(() => {
11
+ clearSearchTokenPresenceCache()
12
+ delete process.env.OM_SEARCH_TOKEN_PRESENCE_CACHE_MS
13
+ })
14
+
15
+ afterAll(() => {
16
+ delete process.env.OM_SEARCH_TOKEN_PRESENCE_CACHE_MS
17
+ })
18
+
19
+ type ProbeLog = { table: string; wheres: unknown[][] }
20
+
21
+ function createFakeDb(options?: { rowsByTable?: Record<string, object | undefined>; failTables?: string[] }) {
22
+ const probes: ProbeLog[] = []
23
+ const db = {
24
+ selectFrom(table: string) {
25
+ const log: ProbeLog = { table, wheres: [] }
26
+ probes.push(log)
27
+ const chain = {
28
+ select: () => chain,
29
+ where: (...args: unknown[]) => {
30
+ log.wheres.push(args)
31
+ return chain
32
+ },
33
+ limit: () => chain,
34
+ executeTakeFirst: async () => {
35
+ if (options?.failTables?.includes(table)) throw new Error(`probe failed for ${table}`)
36
+ return options?.rowsByTable?.[table]
37
+ },
38
+ }
39
+ return chain
40
+ },
41
+ }
42
+ return { db, probes }
43
+ }
44
+
45
+ function buildAvailability(options?: Parameters<typeof createFakeDb>[0] & { enabled?: boolean }) {
46
+ const { db, probes } = createFakeDb(options)
47
+ const logDebug = jest.fn()
48
+ const applyOrganizationScope = jest.fn((query: SearchTokenProbeQueryBuilder) => query)
49
+ const availability = createSearchTokenAvailability({
50
+ getDb: () => db,
51
+ getConfig: () => ({ enabled: options?.enabled ?? true }),
52
+ applyOrganizationScope,
53
+ logDebug,
54
+ })
55
+ return { availability, probes, logDebug, applyOrganizationScope }
56
+ }
57
+
58
+ const countProbes = (probes: ProbeLog[], table: string) => probes.filter((probe) => probe.table === table).length
59
+
60
+ describe('hasSearchFilter / isSearchFilterOp', () => {
61
+ test('recognizes like and ilike only', () => {
62
+ expect(isSearchFilterOp('like')).toBe(true)
63
+ expect(isSearchFilterOp('ilike')).toBe(true)
64
+ expect(isSearchFilterOp('eq')).toBe(false)
65
+ expect(hasSearchFilter([{ op: 'eq' }, { op: 'in' }])).toBe(false)
66
+ expect(hasSearchFilter([{ op: 'eq' }, { op: 'ilike' }])).toBe(true)
67
+ expect(hasSearchFilter([])).toBe(false)
68
+ })
69
+ })
70
+
71
+ describe('createSearchTokenAvailability', () => {
72
+ test('staticEnabled probes the table once per instance and rechecks config each call', async () => {
73
+ const { availability, probes } = buildAvailability({
74
+ rowsByTable: { 'information_schema.tables': { one: 1 } },
75
+ })
76
+ await expect(availability.staticEnabled()).resolves.toBe(true)
77
+ await expect(availability.staticEnabled()).resolves.toBe(true)
78
+ expect(countProbes(probes, 'information_schema.tables')).toBe(1)
79
+ })
80
+
81
+ test('staticEnabled short-circuits on disabled config without probing', async () => {
82
+ const { availability, probes } = buildAvailability({ enabled: false })
83
+ await expect(availability.staticEnabled()).resolves.toBe(false)
84
+ expect(probes).toHaveLength(0)
85
+ })
86
+
87
+ test('staticEnabled retries after a failed table probe instead of caching the rejection', async () => {
88
+ const options = { rowsByTable: { 'information_schema.tables': { one: 1 } }, failTables: ['information_schema.tables'] }
89
+ const { availability, probes } = buildAvailability(options)
90
+ await expect(availability.staticEnabled()).rejects.toThrow('probe failed')
91
+ options.failTables.length = 0
92
+ await expect(availability.staticEnabled()).resolves.toBe(true)
93
+ expect(countProbes(probes, 'information_schema.tables')).toBe(2)
94
+ })
95
+
96
+ test('hasTokens memoizes per (entity, tenant, org) key', async () => {
97
+ const { availability, probes } = buildAvailability({ rowsByTable: { search_tokens: { one: 1 } } })
98
+ const orgScope: OrganizationScope = { ids: ['org1'], includeNull: false }
99
+
100
+ await expect(availability.hasTokens('example:todo', 't1', orgScope)).resolves.toBe(true)
101
+ await expect(availability.hasTokens('example:todo', 't1', orgScope)).resolves.toBe(true)
102
+ expect(countProbes(probes, 'search_tokens')).toBe(1)
103
+
104
+ await availability.hasTokens('example:todo', 't2', orgScope)
105
+ await availability.hasTokens('example:other', 't1', orgScope)
106
+ await availability.hasTokens('example:todo', 't1', { ids: ['org2'], includeNull: false })
107
+ expect(countProbes(probes, 'search_tokens')).toBe(4)
108
+ })
109
+
110
+ test('hasTokens applies the injected organization scope', async () => {
111
+ const { availability, applyOrganizationScope } = buildAvailability()
112
+ const orgScope: OrganizationScope = { ids: ['org1'], includeNull: true }
113
+ await availability.hasTokens('example:todo', 't1', orgScope)
114
+ expect(applyOrganizationScope).toHaveBeenCalledWith(expect.anything(), 'search_tokens.organization_id', orgScope)
115
+
116
+ applyOrganizationScope.mockClear()
117
+ await availability.hasTokens('example:todo', 't1', null)
118
+ expect(applyOrganizationScope).not.toHaveBeenCalled()
119
+ })
120
+
121
+ test('hasTokens resolves false and logs search:has-tokens-error on probe failure', async () => {
122
+ const { availability, logDebug } = buildAvailability({ failTables: ['search_tokens'] })
123
+ await expect(availability.hasTokens('example:todo', 't1', null)).resolves.toBe(false)
124
+ expect(logDebug).toHaveBeenCalledWith('search:has-tokens-error', expect.objectContaining({
125
+ entity: 'example:todo',
126
+ tenantId: 't1',
127
+ error: expect.stringContaining('probe failed'),
128
+ }))
129
+ })
130
+
131
+ test('anySourceHasTokens stops at the first source with tokens and logs each probed source', async () => {
132
+ const { availability, probes, logDebug } = buildAvailability({ rowsByTable: { search_tokens: { one: 1 } } })
133
+ const result = await availability.anySourceHasTokens(
134
+ [
135
+ { entity: 'example:todo', recordIdColumn: 'b.id' },
136
+ { entity: 'example:other', recordIdColumn: 'cfs0.entity_id' },
137
+ ],
138
+ 't1',
139
+ null,
140
+ )
141
+ expect(result).toBe(true)
142
+ expect(countProbes(probes, 'search_tokens')).toBe(1)
143
+ expect(logDebug).toHaveBeenCalledTimes(1)
144
+ expect(logDebug).toHaveBeenCalledWith('search:source-has-tokens', expect.objectContaining({
145
+ entity: 'example:todo',
146
+ recordIdColumn: 'b.id',
147
+ hasTokens: true,
148
+ }))
149
+ })
150
+
151
+ test('process-level TTL cache serves later resolver instances without re-probing', async () => {
152
+ const first = buildAvailability({ rowsByTable: { search_tokens: { one: 1 } } })
153
+ await expect(first.availability.hasTokens('example:todo', 't1', null)).resolves.toBe(true)
154
+ expect(countProbes(first.probes, 'search_tokens')).toBe(1)
155
+
156
+ const second = buildAvailability()
157
+ await expect(second.availability.hasTokens('example:todo', 't1', null)).resolves.toBe(true)
158
+ expect(countProbes(second.probes, 'search_tokens')).toBe(0)
159
+ })
160
+
161
+ test('process-level TTL cache expires and can be disabled', async () => {
162
+ const nowSpy = jest.spyOn(Date, 'now').mockReturnValue(1_000_000)
163
+ try {
164
+ const first = buildAvailability()
165
+ await first.availability.hasTokens('example:todo', 't1', null)
166
+
167
+ nowSpy.mockReturnValue(1_000_000 + 30_001)
168
+ const second = buildAvailability()
169
+ await second.availability.hasTokens('example:todo', 't1', null)
170
+ expect(countProbes(second.probes, 'search_tokens')).toBe(1)
171
+
172
+ process.env.OM_SEARCH_TOKEN_PRESENCE_CACHE_MS = '0'
173
+ const third = buildAvailability()
174
+ await third.availability.hasTokens('example:todo', 't1', null)
175
+ expect(countProbes(third.probes, 'search_tokens')).toBe(1)
176
+ } finally {
177
+ nowSpy.mockRestore()
178
+ }
179
+ })
180
+
181
+ test('error-driven false results are not TTL-cached', async () => {
182
+ const first = buildAvailability({ failTables: ['search_tokens'] })
183
+ await expect(first.availability.hasTokens('example:todo', 't1', null)).resolves.toBe(false)
184
+
185
+ const second = buildAvailability({ rowsByTable: { search_tokens: { one: 1 } } })
186
+ await expect(second.availability.hasTokens('example:todo', 't1', null)).resolves.toBe(true)
187
+ expect(countProbes(second.probes, 'search_tokens')).toBe(1)
188
+ })
189
+
190
+ test('probe uses index-friendly tenant predicates instead of IS NOT DISTINCT FROM', async () => {
191
+ const { availability, probes } = buildAvailability()
192
+ await availability.hasTokens('example:todo', 't1', null)
193
+ await availability.hasTokens('example:todo', null, null)
194
+ const [scoped, global] = probes.filter((probe) => probe.table === 'search_tokens')
195
+ expect(scoped.wheres).toContainEqual(['tenant_id', '=', 't1'])
196
+ expect(global.wheres).toContainEqual(['tenant_id', 'is', null])
197
+ })
198
+
199
+ test('anySourceHasTokens sweeps all sources when none has tokens, sharing the memo with hasTokens', async () => {
200
+ const { availability, probes } = buildAvailability()
201
+ const sources = [
202
+ { entity: 'example:todo', recordIdColumn: 'b.id' },
203
+ { entity: 'example:other', recordIdColumn: 'cfs0.entity_id' },
204
+ ]
205
+ await expect(availability.anySourceHasTokens(sources, 't1', null)).resolves.toBe(false)
206
+ expect(countProbes(probes, 'search_tokens')).toBe(2)
207
+
208
+ await expect(availability.hasTokens('example:todo', 't1', null)).resolves.toBe(false)
209
+ expect(countProbes(probes, 'search_tokens')).toBe(2)
210
+ })
211
+ })
@@ -1,8 +1,12 @@
1
1
  import {
2
+ DEFAULT_SEARCH_MAX_FIELD_CHARS,
3
+ DEFAULT_SEARCH_MAX_TOKENS_PER_FIELD,
4
+ DEFAULT_SEARCH_MAX_TOKENS_PER_RECORD,
2
5
  DEFAULT_SEARCH_MIN_TOKEN_LENGTH,
3
6
  isSearchFieldBlocklisted,
4
7
  resolveSearchConfig,
5
8
  resolveSearchMinTokenLength,
9
+ resolveSearchTokenLimits,
6
10
  } from '../config'
7
11
 
8
12
  describe('resolveSearchMinTokenLength', () => {
@@ -109,6 +113,61 @@ describe('OM_SEARCH_FIELD_BLOCKLIST parsing', () => {
109
113
  })
110
114
  })
111
115
 
116
+ describe('search token limits', () => {
117
+ const variableNames = [
118
+ 'OM_SEARCH_MAX_FIELD_CHARS',
119
+ 'OM_SEARCH_MAX_TOKENS_PER_FIELD',
120
+ 'OM_SEARCH_MAX_TOKENS_PER_RECORD',
121
+ ] as const
122
+ const originalValues = Object.fromEntries(variableNames.map((name) => [name, process.env[name]]))
123
+
124
+ afterEach(() => {
125
+ for (const name of variableNames) {
126
+ const original = originalValues[name]
127
+ if (original === undefined) delete process.env[name]
128
+ else process.env[name] = original
129
+ }
130
+ })
131
+
132
+ it('uses safe defaults when limits are unset', () => {
133
+ for (const name of variableNames) delete process.env[name]
134
+
135
+ expect(resolveSearchTokenLimits(resolveSearchConfig())).toEqual({
136
+ maxFieldChars: DEFAULT_SEARCH_MAX_FIELD_CHARS,
137
+ maxTokensPerField: DEFAULT_SEARCH_MAX_TOKENS_PER_FIELD,
138
+ maxTokensPerRecord: DEFAULT_SEARCH_MAX_TOKENS_PER_RECORD,
139
+ })
140
+ })
141
+
142
+ it('accepts zero to disable individual limits', () => {
143
+ for (const name of variableNames) process.env[name] = '0'
144
+
145
+ expect(resolveSearchTokenLimits(resolveSearchConfig())).toEqual({
146
+ maxFieldChars: 0,
147
+ maxTokensPerField: 0,
148
+ maxTokensPerRecord: 0,
149
+ })
150
+ })
151
+
152
+ it('normalizes invalid custom config values to defaults', () => {
153
+ expect(resolveSearchTokenLimits({
154
+ enabled: true,
155
+ minTokenLength: 3,
156
+ enablePartials: true,
157
+ hashAlgorithm: 'sha256',
158
+ storeRawTokens: false,
159
+ blocklistedFields: [],
160
+ maxFieldChars: Number.NaN,
161
+ maxTokensPerField: -1,
162
+ maxTokensPerRecord: 4.8,
163
+ })).toEqual({
164
+ maxFieldChars: DEFAULT_SEARCH_MAX_FIELD_CHARS,
165
+ maxTokensPerField: DEFAULT_SEARCH_MAX_TOKENS_PER_FIELD,
166
+ maxTokensPerRecord: 4,
167
+ })
168
+ })
169
+ })
170
+
112
171
  describe('isSearchFieldBlocklisted', () => {
113
172
  const baseConfig = {
114
173
  enabled: true,
@@ -0,0 +1,49 @@
1
+ import type { SearchConfig } from '../config'
2
+ import { tokenizeText } from '../tokenize'
3
+
4
+ const baseConfig: SearchConfig = {
5
+ enabled: true,
6
+ minTokenLength: 3,
7
+ enablePartials: true,
8
+ hashAlgorithm: 'sha256',
9
+ storeRawTokens: false,
10
+ blocklistedFields: [],
11
+ }
12
+
13
+ describe('tokenizeText limits', () => {
14
+ test('truncates oversized field text before tokenizing', () => {
15
+ const config = { ...baseConfig, enablePartials: false, maxFieldChars: 10 }
16
+
17
+ const { tokens } = tokenizeText('aaaa bbbb cccc', config)
18
+
19
+ expect(tokens).toEqual(['aaaa', 'bbbb'])
20
+ })
21
+
22
+ test('bounds prefix expansion while collecting tokens', () => {
23
+ const config = { ...baseConfig, maxFieldChars: 100, maxTokensPerField: 5 }
24
+
25
+ const { tokens, hashes } = tokenizeText('a'.repeat(100), config)
26
+
27
+ expect(tokens).toEqual(['aaa', 'aaaa', 'aaaaa', 'aaaaaa', 'aaaaaaa'])
28
+ expect(hashes).toHaveLength(5)
29
+ })
30
+
31
+ test('applies safe defaults to legacy configs without limit fields', () => {
32
+ const { tokens } = tokenizeText('a'.repeat(50_000), baseConfig)
33
+
34
+ expect(tokens).toHaveLength(5_000)
35
+ })
36
+
37
+ test('allows a limit to be disabled explicitly', () => {
38
+ const config = {
39
+ ...baseConfig,
40
+ enablePartials: false,
41
+ maxFieldChars: 0,
42
+ maxTokensPerField: 0,
43
+ }
44
+
45
+ const { tokens } = tokenizeText('alpha beta gamma delta', config)
46
+
47
+ expect(tokens).toEqual(['alpha', 'beta', 'gamma', 'delta'])
48
+ })
49
+ })