@open-mercato/shared 0.8.1-develop.7294.1.0ef99fc518 → 0.8.1-develop.7296.1.2111d779db

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -20,13 +20,16 @@ import {
20
20
  type SearchTokenProbeQueryBuilder,
21
21
  } from '../search/availability'
22
22
  import { tokenizeText } from '../search/tokenize'
23
+ import { buildContainmentPatterns } from '../search/containment'
23
24
  import { fieldNameCandidates } from './encrypted-sort'
24
25
  import { isTenantDataEncryptionEnabled } from '../encryption/toggles'
25
26
  import { runBeforeQueryPipeline, runAfterQueryPipeline, type QueryExtensionContext } from './query-extension-runner'
26
27
  import {
27
28
  buildCustomFieldDefinitionIndexFromRows,
29
+ normalizeDefinitionKey,
28
30
  resolveCfDefIndexOrgCandidates,
29
31
  type CustomFieldDefinitionRow,
32
+ type CustomFieldDefinitionSummary,
30
33
  type ResolvedCustomFieldDefinitions,
31
34
  } from '../crud/custom-field-definition-index'
32
35
  import { warnOnCiphertextLikeFallback } from './ciphertext-search-warning'
@@ -293,6 +296,25 @@ function buildFilterableCustomFieldJoins(
293
296
  })
294
297
  }
295
298
 
299
+ /** Custom field kinds stored numerically (`custom_field_values.value_int`/`value_float`) — sort numerically, not as text (#5674). */
300
+ const NUMERIC_CF_SORT_KINDS = new Set(['integer', 'float'])
301
+
302
+ /**
303
+ * Pick the winning `kind` for a `cf:` sort key from an already-resolved
304
+ * definition list — org-scoped beats tenant-wide beats global, same
305
+ * precedence as the dedicated sort-kind lookup (#5674 review).
306
+ */
307
+ function pickCfSortKind(defs: CustomFieldDefinitionSummary[] | undefined): string | null {
308
+ if (!defs || !defs.length) return null
309
+ let winner: { kind: string; specificity: number } | null = null
310
+ for (const def of defs) {
311
+ if (!def.kind) continue
312
+ const specificity = def.organizationId != null ? 2 : def.tenantId != null ? 1 : 0
313
+ if (!winner || specificity >= winner.specificity) winner = { kind: def.kind, specificity }
314
+ }
315
+ return winner?.kind ?? null
316
+ }
317
+
296
318
  function computeCustomFieldScore(cfg: Record<string, unknown>, kind: string, entityIndex: number) {
297
319
  const listVisibleScore = cfg.listVisible === false ? 0 : 1
298
320
  const formEditableScore = cfg.formEditable === false ? 0 : 1
@@ -446,9 +468,9 @@ export class BasicQueryEngine implements QueryEngine {
446
468
  ? await this.searchAvailability().hasTokens(String(entity), opts.tenantId ?? null, orgScope)
447
469
  : false
448
470
  const searchActive = searchEnabled && hasSearchTokens
449
- // Opt-in via OM_SEARCH_USE_ILIKE_FOR_NON_ENCRYPTED_FIELDS (default false: the pre-existing
450
- // rewrite-everything behavior is kept). When enabled, base-column like/ilike is rerouted
451
- // through search tokens ONLY for encrypted columns, where
471
+ // Gated on OM_SEARCH_USE_ILIKE_FOR_NON_ENCRYPTED_FIELDS (default false per #5383; set it to
472
+ // true to opt into the #5803 fix ahead of that follow-up). When enabled, base-column
473
+ // like/ilike is rerouted through search tokens ONLY for encrypted columns, where
452
474
  // ILIKE against ciphertext cannot match. On a plaintext column SQL ILIKE is exact, and the token
453
475
  // rewrite silently changes the result set: tokenization splits on non-alphanumerics and drops
454
476
  // tokens shorter than minTokenLength, so a document-number search like "ZK 1/2026" degrades to
@@ -617,6 +639,27 @@ export class BasicQueryEngine implements QueryEngine {
617
639
  })
618
640
  }
619
641
  }
642
+ // A PLAINTEXT base column the gate above kept off the token path: apply the declared
643
+ // containment once per word so the token subquery's word-order-independent matching survives
644
+ // the reroute (#5803 / TC-RESO-009). Chained `where`s ARE the AND the token
645
+ // `having count(distinct)` did. `fieldName` is required, so this never fires for a JOINed
646
+ // column — `applyJoinFilters` calls `applyFilterOp` without it (below), and that predicate
647
+ // keeps its pre-existing literal-containment behavior unchanged.
648
+ if (
649
+ (op === 'like' || op === 'ilike') &&
650
+ typeof value === 'string' &&
651
+ searchActive &&
652
+ fieldName &&
653
+ encryptedLikeFields !== null &&
654
+ !isEncryptedLikeField(encryptedLikeFields, fieldName)
655
+ ) {
656
+ const patterns = buildContainmentPatterns(value)
657
+ if (patterns.length > 1) {
658
+ let next = builder
659
+ for (const pattern of patterns) next = this.applyColumnOp(next, column, op, pattern)
660
+ return next
661
+ }
662
+ }
620
663
  return this.applyColumnOp(builder, column, op, value)
621
664
  }
622
665
 
@@ -703,6 +746,7 @@ export class BasicQueryEngine implements QueryEngine {
703
746
  hasJoinedAggregates: boolean
704
747
  cfJsonAliases: Set<string>
705
748
  cfMultiAliasByAlias: Map<string, string>
749
+ cfSortAliases: string[]
706
750
  resolvedCustomFieldDefinitions: ResolvedCustomFieldDefinitions | undefined
707
751
  }
708
752
 
@@ -994,10 +1038,70 @@ export class BasicQueryEngine implements QueryEngine {
994
1038
  }
995
1039
  }
996
1040
 
1041
+ // A `cf:` sort needs to know the field's declared `kind` so the ORDER BY
1042
+ // expression can cast numeric kinds instead of ordering by jsonb text
1043
+ // (#5674). Org-scoped beats tenant-wide beats global for the same key —
1044
+ // a same-key definition from an unrelated organization in the tenant
1045
+ // never gets to decide the kind (#5674 review).
1046
+ const cfSortKeys = Array.from(new Set(
1047
+ resolvedSorts.filter((sort) => sort.field.startsWith('cf:')).map((sort) => sort.field.slice(3))
1048
+ ))
1049
+ const cfSortKinds = new Map<string, string>()
1050
+ if (cfSortKeys.length > 0) {
1051
+ // `includeCustomFields === true` already resolved every definition for
1052
+ // these entities into `resolvedCustomFieldDefinitions` above — reuse it
1053
+ // instead of a second `custom_field_defs` round trip (#5674 review).
1054
+ const keysNeedingQuery: string[] = []
1055
+ for (const key of cfSortKeys) {
1056
+ const kind = resolvedCustomFieldDefinitions
1057
+ ? pickCfSortKind(resolvedCustomFieldDefinitions.index.get(normalizeDefinitionKey(key)))
1058
+ : null
1059
+ if (kind) cfSortKinds.set(key, kind)
1060
+ else keysNeedingQuery.push(key)
1061
+ }
1062
+ if (keysNeedingQuery.length > 0) {
1063
+ const sortEntityIds = Array.from(new Set(
1064
+ keysNeedingQuery
1065
+ .map((key) => keySource.get(key)?.entityId)
1066
+ .filter((id): id is EntityId => Boolean(id))
1067
+ .map((id) => String(id))
1068
+ ))
1069
+ if (sortEntityIds.length > 0) {
1070
+ const cfSortOrgCandidates = resolveCfDefIndexOrgCandidates(opts.organizationIds, opts.organizationId ?? null)
1071
+ const kindRows = await db
1072
+ .selectFrom('custom_field_defs' as any)
1073
+ .select(['key' as any, 'kind' as any, 'organization_id' as any, 'tenant_id' as any])
1074
+ .where('entity_id' as any, 'in', sortEntityIds)
1075
+ .where('key' as any, 'in', keysNeedingQuery)
1076
+ .where('is_active' as any, '=', true)
1077
+ .where((eb: any) => eb.or([
1078
+ eb('tenant_id' as any, '=', tenantId),
1079
+ eb('tenant_id' as any, 'is', null),
1080
+ ]))
1081
+ .where((eb: any) => cfSortOrgCandidates.length
1082
+ ? eb.or([
1083
+ eb('organization_id' as any, 'is', null),
1084
+ eb('organization_id' as any, 'in', cfSortOrgCandidates),
1085
+ ])
1086
+ : eb('organization_id' as any, 'is', null))
1087
+ .execute() as Array<{ key: string; kind: string | null; organization_id: string | null; tenant_id: string | null }>
1088
+ const winners = new Map<string, { kind: string; specificity: number }>()
1089
+ for (const row of kindRows) {
1090
+ if (!row.kind) continue
1091
+ const specificity = row.organization_id != null ? 2 : row.tenant_id != null ? 1 : 0
1092
+ const existing = winners.get(row.key)
1093
+ if (!existing || specificity >= existing.specificity) winners.set(row.key, { kind: row.kind, specificity })
1094
+ }
1095
+ for (const [key, entry] of winners) cfSortKinds.set(key, entry.kind)
1096
+ }
1097
+ }
1098
+ }
1099
+
997
1100
  const cfValueExprByKey: Record<string, RawBuilder<string | null>> = {}
998
1101
  const cfSelectedAliases: string[] = []
999
1102
  const cfJsonAliases = new Set<string>()
1000
1103
  const cfMultiAliasByAlias = new Map<string, string>()
1104
+ const cfSortAliases: string[] = []
1001
1105
  for (const key of cfKeys) {
1002
1106
  const source = keySource.get(key)
1003
1107
  if (!source) continue
@@ -1127,6 +1231,12 @@ export class BasicQueryEngine implements QueryEngine {
1127
1231
  // field covered by an encryption map such a leaf therefore compares against
1128
1232
  // ciphertext and will not match.
1129
1233
  //
1234
+ // This also means the #5803 plaintext-containment split (lib/search/containment) does not
1235
+ // apply here: `buildColumnOpExpression` below keeps a multi-word `like`/`ilike` leaf as one
1236
+ // literal pattern, so an OR-grouped search (e.g. `customers/api/people`'s multi-field
1237
+ // fallback) loses word-order independence on Basic while the Hybrid engine's OR groups
1238
+ // apply the split. Pre-existing divergence between the two engines; not tracked by an issue.
1239
+ //
1130
1240
  // The count shape never populates cfValueExprByKey (it joins no cf tables), so
1131
1241
  // its applicability test is key resolution itself — the same condition that
1132
1242
  // gates the full shape's expression map — and a cf leaf compiles to a
@@ -1200,30 +1310,57 @@ export class BasicQueryEngine implements QueryEngine {
1200
1310
  }
1201
1311
  }
1202
1312
 
1203
- // Sorting: base fields and cf:* (use aggregated alias for cf)
1313
+ // Sorting: base fields and cf:* (a dedicated scalar alias, never the jsonb
1314
+ // projection alias — ordering by `to_jsonb(...)` compares arrays after every
1315
+ // scalar string regardless of contents, and a numeric kind must cast to
1316
+ // numeric instead of ordering the text CASE expression lexicographically (#5674)).
1317
+ let lastEmittedSortField: string | null = null
1204
1318
  for (const s of isCountProjection ? [] : resolvedSorts) {
1205
1319
  if (s.field.startsWith('cf:')) {
1206
1320
  const key = s.field.slice(3)
1207
- const alias = sanitize(`cf:${key}`)
1208
- // Ensure included in projection to sort by
1209
- if (!cfSelectedAliases.includes(alias)) {
1210
- const expr = cfValueExprByKey[key]
1211
- if (expr) {
1212
- q = q.select(sql<string | null>`max(${expr})`.as(alias))
1213
- cfSelectedAliases.push(alias)
1321
+ const sortAlias = sanitize(`cf:${key}__sort`)
1322
+ if (!cfSortAliases.includes(sortAlias)) {
1323
+ const source = keySource.get(key)
1324
+ const kind = cfSortKinds.get(key)
1325
+ let sortExpr: RawBuilder<unknown> | null = null
1326
+ if (source && kind && NUMERIC_CF_SORT_KINDS.has(kind)) {
1327
+ const sourceAliasSafe = sanitize(source.alias || 'src')
1328
+ const keyAliasSafe = sanitize(key)
1329
+ const valAlias = `cfv_${sourceAliasSafe}_${keyAliasSafe}`
1330
+ const numericColumn = kind === 'integer' ? 'value_int' : 'value_float'
1331
+ sortExpr = sql<string | null>`max((${sql.ref(`${valAlias}.${numericColumn}`)})::numeric)`
1332
+ } else {
1333
+ const expr = cfValueExprByKey[key]
1334
+ if (expr) sortExpr = sql<string | null>`max(${expr})`
1335
+ }
1336
+ if (sortExpr) {
1337
+ q = q.select(sortExpr.as(sortAlias))
1338
+ cfSortAliases.push(sortAlias)
1214
1339
  }
1215
1340
  }
1216
- // Only order by an alias the projection actually carries. A key that
1217
- // resolved to no expression is dropped rather than emitted as an
1218
- // unselected alias, which is what the base-column branch above already
1219
- // does when `resolveBaseColumn` returns null.
1220
- if (!requiresPlaintextSort && cfSelectedAliases.includes(alias)) {
1221
- q = q.orderBy(alias, (s.dir ?? 'asc') as any)
1341
+ // Only order by an alias the query actually selects. A key that resolved
1342
+ // to no expression is dropped rather than emitted as an unselected alias,
1343
+ // which is what the base-column branch below already does when
1344
+ // `resolveBaseColumn` returns null.
1345
+ if (!requiresPlaintextSort && cfSortAliases.includes(sortAlias)) {
1346
+ const direction = sql.raw((s.dir ?? 'asc') === 'desc' ? 'desc' : 'asc')
1347
+ q = q.orderBy(sql`${sql.ref(sortAlias)} ${direction} NULLS LAST`)
1348
+ lastEmittedSortField = s.field
1222
1349
  }
1223
1350
  } else {
1224
- if (!requiresPlaintextSort) q = q.orderBy(qualify(s.field), (s.dir ?? 'asc') as any)
1351
+ if (!requiresPlaintextSort) {
1352
+ q = q.orderBy(qualify(s.field), (s.dir ?? 'asc') as any)
1353
+ lastEmittedSortField = s.field
1354
+ }
1225
1355
  }
1226
1356
  }
1357
+ // Stable tiebreak so ties (or NULLs) don't reorder arbitrarily across pages
1358
+ // (#5674). Only appended when a sort term actually made it into the ORDER
1359
+ // BY — an entirely unresolved cf sort must stay a no-op, not gain an
1360
+ // incidental `id` ordering the caller never asked for.
1361
+ if (!isCountProjection && lastEmittedSortField !== null && lastEmittedSortField !== 'id') {
1362
+ q = q.orderBy(qualify('id'), 'asc' as any)
1363
+ }
1227
1364
 
1228
1365
  // Deduplicate if we joined CFs or extensions by grouping on base id. The count
1229
1366
  // shape has neither, and must stay barrier-free for its LIMIT to bind.
@@ -1235,7 +1372,7 @@ export class BasicQueryEngine implements QueryEngine {
1235
1372
  q = q.groupBy(`${table}.id`)
1236
1373
  }
1237
1374
 
1238
- return { builder: q, hasJoinedAggregates, cfJsonAliases, cfMultiAliasByAlias, resolvedCustomFieldDefinitions }
1375
+ return { builder: q, hasJoinedAggregates, cfJsonAliases, cfMultiAliasByAlias, cfSortAliases, resolvedCustomFieldDefinitions }
1239
1376
  }
1240
1377
 
1241
1378
  // Pagination
@@ -1247,6 +1384,7 @@ export class BasicQueryEngine implements QueryEngine {
1247
1384
  hasJoinedAggregates,
1248
1385
  cfJsonAliases,
1249
1386
  cfMultiAliasByAlias,
1387
+ cfSortAliases,
1250
1388
  resolvedCustomFieldDefinitions,
1251
1389
  } = await buildQuery('full')
1252
1390
 
@@ -1330,6 +1468,19 @@ export class BasicQueryEngine implements QueryEngine {
1330
1468
  }
1331
1469
  }
1332
1470
 
1471
+ // The `cf:<key>__sort` alias exists only to give a `cf:` sort's ORDER BY
1472
+ // something to reference (and, for a numeric kind, to host the cast) — it
1473
+ // is never a requested output field. Strip it before rows leave this
1474
+ // function so it doesn't leak into API responses as a phantom custom field
1475
+ // `<key>__sort` (unlike `__is_multi`, which `normalizeCfJsonAliases`
1476
+ // already deletes) (#5674 review).
1477
+ const stripCfSortAliases = (rows: ResultRow[]) => {
1478
+ if (cfSortAliases.length === 0) return
1479
+ for (const row of rows) {
1480
+ for (const alias of cfSortAliases) delete row[alias]
1481
+ }
1482
+ }
1483
+
1333
1484
  let pagedItems: ResultRow[]
1334
1485
  let encryptedSortRowCapWarning: EncryptedSortRowCapWarning | undefined
1335
1486
 
@@ -1372,6 +1523,7 @@ export class BasicQueryEngine implements QueryEngine {
1372
1523
  } else {
1373
1524
  const pageRows = await qFull.where(qualify('id'), 'in', pageIds).execute() as ResultRow[]
1374
1525
  normalizeCfJsonAliases(pageRows)
1526
+ stripCfSortAliases(pageRows)
1375
1527
  const decryptedPageRows = decryptPayload
1376
1528
  ? await mapWithConcurrency(pageRows, DECRYPT_CONCURRENCY, decryptRow)
1377
1529
  : pageRows
@@ -1384,6 +1536,7 @@ export class BasicQueryEngine implements QueryEngine {
1384
1536
  const dataQuery = qFull.limit(pageSize).offset((page - 1) * pageSize)
1385
1537
  const items = await dataQuery.execute() as ResultRow[]
1386
1538
  normalizeCfJsonAliases(items)
1539
+ stripCfSortAliases(items)
1387
1540
  pagedItems = decryptPayload
1388
1541
  ? await mapWithConcurrency(items, DECRYPT_CONCURRENCY, decryptRow)
1389
1542
  : items
@@ -113,6 +113,33 @@ describe('OM_SEARCH_FIELD_BLOCKLIST parsing', () => {
113
113
  })
114
114
  })
115
115
 
116
+ describe('OM_SEARCH_USE_ILIKE_FOR_NON_ENCRYPTED_FIELDS', () => {
117
+ const originalValue = process.env.OM_SEARCH_USE_ILIKE_FOR_NON_ENCRYPTED_FIELDS
118
+
119
+ afterEach(() => {
120
+ if (originalValue === undefined) {
121
+ delete process.env.OM_SEARCH_USE_ILIKE_FOR_NON_ENCRYPTED_FIELDS
122
+ } else {
123
+ process.env.OM_SEARCH_USE_ILIKE_FOR_NON_ENCRYPTED_FIELDS = originalValue
124
+ }
125
+ })
126
+
127
+ // #5383: the switch stays off by default until tokenization is made ILIKE-equivalent, so the
128
+ // rewrite-everything behavior is unchanged for a deployment that does not opt in. #5803 is the
129
+ // correctness gap this switch closes when a deployment opts in ahead of that follow-up.
130
+ it('defaults to off so the legacy rewrite is unchanged for every column', () => {
131
+ delete process.env.OM_SEARCH_USE_ILIKE_FOR_NON_ENCRYPTED_FIELDS
132
+
133
+ expect(resolveSearchConfig().useIlikeForNonEncryptedFields).toBe(false)
134
+ })
135
+
136
+ it('can be switched on to apply a declared ilike on a plaintext column as written', () => {
137
+ process.env.OM_SEARCH_USE_ILIKE_FOR_NON_ENCRYPTED_FIELDS = 'true'
138
+
139
+ expect(resolveSearchConfig().useIlikeForNonEncryptedFields).toBe(true)
140
+ })
141
+ })
142
+
116
143
  describe('search token limits', () => {
117
144
  const variableNames = [
118
145
  'OM_SEARCH_MAX_FIELD_CHARS',
@@ -0,0 +1,61 @@
1
+ import { buildContainmentPatterns, MAX_CONTAINMENT_WORDS } from '../containment'
2
+
3
+ describe('buildContainmentPatterns (#5803)', () => {
4
+ test('splits a multi-word contains pattern into one pattern per word', () => {
5
+ // The token subquery matched every token in any order with anything between them. Reproducing
6
+ // that on SQL means ANDing per word, which is what TC-RESO-009 exercises through the API:
7
+ // `?search=Warehouse <stamp>` has to keep matching `Warehouse A <stamp>`.
8
+ expect(buildContainmentPatterns('%Warehouse 1757%')).toEqual(['%Warehouse%', '%1757%'])
9
+ })
10
+
11
+ test('collapses runs of whitespace rather than emitting empty patterns', () => {
12
+ expect(buildContainmentPatterns('%John Smith%')).toEqual(['%John%', '%Smith%'])
13
+ })
14
+
15
+ test('leaves a single-word pattern exactly as the caller declared it', () => {
16
+ // The reported #5803 case: the distinguishing fragment must reach SQL untouched, or the exact
17
+ // row cannot come back at all.
18
+ expect(buildContainmentPatterns('%2026-08%')).toEqual(['%2026-08%'])
19
+ })
20
+
21
+ test('leaves a term too short to tokenize alone', () => {
22
+ expect(buildContainmentPatterns('%08%')).toEqual(['%08%'])
23
+ })
24
+
25
+ test('keeps escaped wildcards attached to their word', () => {
26
+ // escapeLikePattern turns a literal `%` into `\%`; splitting must not treat it as a wildcard
27
+ // and must not tear the escape off its word.
28
+ expect(buildContainmentPatterns('%50\\% off%')).toEqual(['%50\\%%', '%off%'])
29
+ })
30
+
31
+ test('does not split a structured pattern carrying its own wildcards', () => {
32
+ // A caller that hand-built `%a% b%` asked for that exact shape; re-splitting it would silently
33
+ // rewrite a predicate this helper has no business reinterpreting.
34
+ expect(buildContainmentPatterns('%a% b%')).toEqual(['%a% b%'])
35
+ })
36
+
37
+ test('does not split an anchored pattern', () => {
38
+ // `startsWith` / `endsWith` terms are anchored on purpose; per-word ANDing would drop the
39
+ // anchor and widen the match.
40
+ expect(buildContainmentPatterns('John Smith%')).toEqual(['John Smith%'])
41
+ expect(buildContainmentPatterns('%John Smith')).toEqual(['%John Smith'])
42
+ })
43
+
44
+ test('treats a trailing escaped percent as a literal, not as the closing wildcard', () => {
45
+ expect(buildContainmentPatterns('%John Smith\\%')).toEqual(['%John Smith\\%'])
46
+ })
47
+
48
+ test('splits a term at the word cap', () => {
49
+ const words = Array.from({ length: MAX_CONTAINMENT_WORDS }, (_, index) => `w${index}`)
50
+ expect(buildContainmentPatterns(`%${words.join(' ')}%`)).toEqual(words.map((word) => `%${word}%`))
51
+ })
52
+
53
+ test('falls back to the single verbatim pattern above the word cap', () => {
54
+ // Most list routes declare `search` as an unbounded string and this repo ships no trigram
55
+ // index, so an unbounded per-word AND would let one request compile into an unbounded number
56
+ // of sequential-scan predicates. Past the cap, this returns to the pre-split behavior instead.
57
+ const words = Array.from({ length: MAX_CONTAINMENT_WORDS + 1 }, (_, index) => `w${index}`)
58
+ const pattern = `%${words.join(' ')}%`
59
+ expect(buildContainmentPatterns(pattern)).toEqual([pattern])
60
+ })
61
+ })
@@ -9,12 +9,34 @@ export type SearchConfig = {
9
9
  hashAlgorithm: 'sha256' | 'sha1' | 'md5'
10
10
  storeRawTokens: boolean
11
11
  /**
12
- * When true, a like/ilike on a PLAINTEXT base column runs as exact SQL ILIKE instead of being
13
- * rewritten into an approximate search-token match; encrypted columns always keep the token
14
- * path (ILIKE against ciphertext cannot match). Off by default: token matching can be faster
15
- * than an unanchored ILIKE, which may need a full scan without a trigram index — but it is
16
- * approximate (fragments under minTokenLength vanish, so `ZK 1/2026` degrades to its year and
17
- * an all-short term drops the predicate). Flip it on when list search must be exact.
12
+ * When true, a like/ilike on a PLAINTEXT base column runs as SQL ILIKE — one containment
13
+ * predicate per word of the term, ANDed — instead of being rewritten into an approximate
14
+ * search-token match; encrypted columns always keep the token path (ILIKE against ciphertext
15
+ * cannot match).
16
+ *
17
+ * Off by default, per #5383: the token store is expected to become faster than ILIKE once
18
+ * tokenization is made semantically equivalent to it, so the plan there is to keep this switch
19
+ * off until that follow-up lands rather than trade performance for correctness by default. #5803
20
+ * documents the correctness gap this switch closes when enabled: the token rewrite is lossy in a
21
+ * way that silently returns the WRONG record rather than merely extra ones (tokenization splits
22
+ * on non-alphanumerics and drops fragments under minTokenLength, so `2026-08` and `2026-01` both
23
+ * reduce to {202, 2026} and a picker offers the neighbouring period; a term that tokenizes to
24
+ * nothing (`08`) drops the predicate entirely and matches every row) — a deployment that hits
25
+ * that gap before #5383 lands can opt in here.
26
+ *
27
+ * Per-word ANDing (see lib/search/containment) is a trade-off, not a strict improvement, over
28
+ * the single-literal ILIKE #4622 originally introduced: the token subquery matched a value
29
+ * carrying every token in any order with anything between them, so `?search=Warehouse 1757`
30
+ * must keep matching `Warehouse A 1757` — a single verbatim `ILIKE '%Warehouse 1757%'` would
31
+ * not, and TC-RESO-009 pins that as required behavior. The same word-order independence also
32
+ * widens multi-word document-number searches: `?search=ZK 1/2026` now also matches
33
+ * `ZK 11/2026` and `1/2026 ZK`, where the old single-literal ILIKE matched neither.
34
+ *
35
+ * Set `OM_SEARCH_USE_ILIKE_FOR_NON_ENCRYPTED_FIELDS=true` to opt into declared-column ILIKE
36
+ * ahead of #5383 — worth doing when the #5803 wrong-record symptom is hit in practice. Leaving it
37
+ * unset keeps the legacy rewrite-everything behavior, including the token index's prefix matching
38
+ * (`?search=ware` matching `Warehouse` when `enablePartials` is on, which literal containment
39
+ * gives only where the fragment really is a substring).
18
40
  */
19
41
  useIlikeForNonEncryptedFields?: boolean
20
42
  blocklistedFields: string[]
@@ -0,0 +1,66 @@
1
+ /**
2
+ * Splits a `contains` like/ilike pattern into one pattern per whitespace-separated word, so a
3
+ * plaintext column taken off the hashed-token path keeps the word-order-independent matching the
4
+ * token index provided.
5
+ *
6
+ * The token path matches a value when it carries EVERY token of the term, in any order and with
7
+ * anything in between: `?search=Warehouse 1757` matches `Warehouse A 1757`. A single verbatim
8
+ * `name ILIKE '%Warehouse 1757%'` is literal substring containment, so the `A ` sitting between the
9
+ * two words defeats it — that is a capability every list grid has today, and #5803's fix must not
10
+ * take it away (`TC-RESO-009` pins it).
11
+ *
12
+ * ANDing one containment predicate per word reproduces the token semantics exactly on a column the
13
+ * engine can read, without the token path's two lossy steps: nothing is dropped for being shorter
14
+ * than `minTokenLength` (`?search=08` filters instead of matching every row) and nothing is split on
15
+ * non-alphanumerics (`?search=2026-08` no longer collapses onto `2026-01`).
16
+ *
17
+ * Splitting is deliberately narrow — a pattern is only split when it is unambiguously the
18
+ * `%term%` shape `buildIlikeTerm(value, 'contains')` produces:
19
+ *
20
+ * - it opens and closes with a wildcard `%` (a trailing `\%` is an escaped literal, not a wildcard);
21
+ * - the term between them carries no unescaped `%` or `_`, so a hand-built structured pattern such
22
+ * as `%a% b%` is left exactly as the caller wrote it;
23
+ * - the term holds at least two words.
24
+ *
25
+ * Anything else returns the input unchanged as a single pattern, which is the caller's existing
26
+ * behavior.
27
+ *
28
+ * The split is also capped at {@link MAX_CONTAINMENT_WORDS} words. Most list routes declare
29
+ * `search` as an unbounded string, this repository ships no trigram index for the resulting
30
+ * `ILIKE`, and the hybrid engine's `$or` groups multiply the split across every leaf — so an
31
+ * attacker-controlled term with thousands of words would otherwise compile into thousands of
32
+ * sequential-scan predicates from one request. A term at or under the cap keeps the per-word AND
33
+ * semantics; over the cap it falls back to the single verbatim pattern, which is bounded and was
34
+ * this helper's own behavior before the split existed.
35
+ */
36
+ export const MAX_CONTAINMENT_WORDS = 10
37
+
38
+ export function buildContainmentPatterns(pattern: string): string[] {
39
+ if (!isWrappedContainsPattern(pattern)) return [pattern]
40
+ const term = pattern.slice(1, -1)
41
+ if (hasUnescapedWildcard(term)) return [pattern]
42
+ const words = term.split(/\s+/).filter((word) => word.length > 0)
43
+ if (words.length < 2 || words.length > MAX_CONTAINMENT_WORDS) return [pattern]
44
+ return words.map((word) => `%${word}%`)
45
+ }
46
+
47
+ function isWrappedContainsPattern(pattern: string): boolean {
48
+ if (pattern.length < 2) return false
49
+ if (!pattern.startsWith('%') || !pattern.endsWith('%')) return false
50
+ return countTrailingBackslashes(pattern, pattern.length - 2) % 2 === 0
51
+ }
52
+
53
+ function hasUnescapedWildcard(term: string): boolean {
54
+ for (let index = 0; index < term.length; index += 1) {
55
+ const char = term[index]
56
+ if (char !== '%' && char !== '_') continue
57
+ if (countTrailingBackslashes(term, index - 1) % 2 === 0) return true
58
+ }
59
+ return false
60
+ }
61
+
62
+ function countTrailingBackslashes(value: string, fromIndex: number): number {
63
+ let count = 0
64
+ for (let index = fromIndex; index >= 0 && value[index] === '\\'; index -= 1) count += 1
65
+ return count
66
+ }