@open-mercato/shared 0.8.1-develop.7294.1.0ef99fc518 → 0.8.1-develop.7296.1.2111d779db
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.turbo/turbo-build.log +1 -1
- package/AGENTS.md +6 -3
- package/dist/lib/query/encrypted-sort.js +4 -1
- package/dist/lib/query/encrypted-sort.js.map +2 -2
- package/dist/lib/query/engine.js +97 -10
- package/dist/lib/query/engine.js.map +3 -3
- package/dist/lib/search/config.js.map +2 -2
- package/dist/lib/search/containment.js +32 -0
- package/dist/lib/search/containment.js.map +7 -0
- package/dist/lib/version.js +1 -1
- package/dist/lib/version.js.map +1 -1
- package/package.json +2 -2
- package/src/lib/query/__tests__/engine.test.ts +208 -16
- package/src/lib/query/encrypted-sort.ts +8 -1
- package/src/lib/query/engine.ts +172 -19
- package/src/lib/search/__tests__/config.test.ts +27 -0
- package/src/lib/search/__tests__/containment.test.ts +61 -0
- package/src/lib/search/config.ts +28 -6
- package/src/lib/search/containment.ts +66 -0
package/src/lib/query/engine.ts
CHANGED
|
@@ -20,13 +20,16 @@ import {
|
|
|
20
20
|
type SearchTokenProbeQueryBuilder,
|
|
21
21
|
} from '../search/availability'
|
|
22
22
|
import { tokenizeText } from '../search/tokenize'
|
|
23
|
+
import { buildContainmentPatterns } from '../search/containment'
|
|
23
24
|
import { fieldNameCandidates } from './encrypted-sort'
|
|
24
25
|
import { isTenantDataEncryptionEnabled } from '../encryption/toggles'
|
|
25
26
|
import { runBeforeQueryPipeline, runAfterQueryPipeline, type QueryExtensionContext } from './query-extension-runner'
|
|
26
27
|
import {
|
|
27
28
|
buildCustomFieldDefinitionIndexFromRows,
|
|
29
|
+
normalizeDefinitionKey,
|
|
28
30
|
resolveCfDefIndexOrgCandidates,
|
|
29
31
|
type CustomFieldDefinitionRow,
|
|
32
|
+
type CustomFieldDefinitionSummary,
|
|
30
33
|
type ResolvedCustomFieldDefinitions,
|
|
31
34
|
} from '../crud/custom-field-definition-index'
|
|
32
35
|
import { warnOnCiphertextLikeFallback } from './ciphertext-search-warning'
|
|
@@ -293,6 +296,25 @@ function buildFilterableCustomFieldJoins(
|
|
|
293
296
|
})
|
|
294
297
|
}
|
|
295
298
|
|
|
299
|
+
/** Custom field kinds stored numerically (`custom_field_values.value_int`/`value_float`) — sort numerically, not as text (#5674). */
|
|
300
|
+
const NUMERIC_CF_SORT_KINDS = new Set(['integer', 'float'])
|
|
301
|
+
|
|
302
|
+
/**
|
|
303
|
+
* Pick the winning `kind` for a `cf:` sort key from an already-resolved
|
|
304
|
+
* definition list — org-scoped beats tenant-wide beats global, same
|
|
305
|
+
* precedence as the dedicated sort-kind lookup (#5674 review).
|
|
306
|
+
*/
|
|
307
|
+
function pickCfSortKind(defs: CustomFieldDefinitionSummary[] | undefined): string | null {
|
|
308
|
+
if (!defs || !defs.length) return null
|
|
309
|
+
let winner: { kind: string; specificity: number } | null = null
|
|
310
|
+
for (const def of defs) {
|
|
311
|
+
if (!def.kind) continue
|
|
312
|
+
const specificity = def.organizationId != null ? 2 : def.tenantId != null ? 1 : 0
|
|
313
|
+
if (!winner || specificity >= winner.specificity) winner = { kind: def.kind, specificity }
|
|
314
|
+
}
|
|
315
|
+
return winner?.kind ?? null
|
|
316
|
+
}
|
|
317
|
+
|
|
296
318
|
function computeCustomFieldScore(cfg: Record<string, unknown>, kind: string, entityIndex: number) {
|
|
297
319
|
const listVisibleScore = cfg.listVisible === false ? 0 : 1
|
|
298
320
|
const formEditableScore = cfg.formEditable === false ? 0 : 1
|
|
@@ -446,9 +468,9 @@ export class BasicQueryEngine implements QueryEngine {
|
|
|
446
468
|
? await this.searchAvailability().hasTokens(String(entity), opts.tenantId ?? null, orgScope)
|
|
447
469
|
: false
|
|
448
470
|
const searchActive = searchEnabled && hasSearchTokens
|
|
449
|
-
//
|
|
450
|
-
//
|
|
451
|
-
// through search tokens ONLY for encrypted columns, where
|
|
471
|
+
// Gated on OM_SEARCH_USE_ILIKE_FOR_NON_ENCRYPTED_FIELDS (default false per #5383; set it to
|
|
472
|
+
// true to opt into the #5803 fix ahead of that follow-up). When enabled, base-column
|
|
473
|
+
// like/ilike is rerouted through search tokens ONLY for encrypted columns, where
|
|
452
474
|
// ILIKE against ciphertext cannot match. On a plaintext column SQL ILIKE is exact, and the token
|
|
453
475
|
// rewrite silently changes the result set: tokenization splits on non-alphanumerics and drops
|
|
454
476
|
// tokens shorter than minTokenLength, so a document-number search like "ZK 1/2026" degrades to
|
|
@@ -617,6 +639,27 @@ export class BasicQueryEngine implements QueryEngine {
|
|
|
617
639
|
})
|
|
618
640
|
}
|
|
619
641
|
}
|
|
642
|
+
// A PLAINTEXT base column the gate above kept off the token path: apply the declared
|
|
643
|
+
// containment once per word so the token subquery's word-order-independent matching survives
|
|
644
|
+
// the reroute (#5803 / TC-RESO-009). Chained `where`s ARE the AND the token
|
|
645
|
+
// `having count(distinct)` did. `fieldName` is required, so this never fires for a JOINed
|
|
646
|
+
// column — `applyJoinFilters` calls `applyFilterOp` without it (below), and that predicate
|
|
647
|
+
// keeps its pre-existing literal-containment behavior unchanged.
|
|
648
|
+
if (
|
|
649
|
+
(op === 'like' || op === 'ilike') &&
|
|
650
|
+
typeof value === 'string' &&
|
|
651
|
+
searchActive &&
|
|
652
|
+
fieldName &&
|
|
653
|
+
encryptedLikeFields !== null &&
|
|
654
|
+
!isEncryptedLikeField(encryptedLikeFields, fieldName)
|
|
655
|
+
) {
|
|
656
|
+
const patterns = buildContainmentPatterns(value)
|
|
657
|
+
if (patterns.length > 1) {
|
|
658
|
+
let next = builder
|
|
659
|
+
for (const pattern of patterns) next = this.applyColumnOp(next, column, op, pattern)
|
|
660
|
+
return next
|
|
661
|
+
}
|
|
662
|
+
}
|
|
620
663
|
return this.applyColumnOp(builder, column, op, value)
|
|
621
664
|
}
|
|
622
665
|
|
|
@@ -703,6 +746,7 @@ export class BasicQueryEngine implements QueryEngine {
|
|
|
703
746
|
hasJoinedAggregates: boolean
|
|
704
747
|
cfJsonAliases: Set<string>
|
|
705
748
|
cfMultiAliasByAlias: Map<string, string>
|
|
749
|
+
cfSortAliases: string[]
|
|
706
750
|
resolvedCustomFieldDefinitions: ResolvedCustomFieldDefinitions | undefined
|
|
707
751
|
}
|
|
708
752
|
|
|
@@ -994,10 +1038,70 @@ export class BasicQueryEngine implements QueryEngine {
|
|
|
994
1038
|
}
|
|
995
1039
|
}
|
|
996
1040
|
|
|
1041
|
+
// A `cf:` sort needs to know the field's declared `kind` so the ORDER BY
|
|
1042
|
+
// expression can cast numeric kinds instead of ordering by jsonb text
|
|
1043
|
+
// (#5674). Org-scoped beats tenant-wide beats global for the same key —
|
|
1044
|
+
// a same-key definition from an unrelated organization in the tenant
|
|
1045
|
+
// never gets to decide the kind (#5674 review).
|
|
1046
|
+
const cfSortKeys = Array.from(new Set(
|
|
1047
|
+
resolvedSorts.filter((sort) => sort.field.startsWith('cf:')).map((sort) => sort.field.slice(3))
|
|
1048
|
+
))
|
|
1049
|
+
const cfSortKinds = new Map<string, string>()
|
|
1050
|
+
if (cfSortKeys.length > 0) {
|
|
1051
|
+
// `includeCustomFields === true` already resolved every definition for
|
|
1052
|
+
// these entities into `resolvedCustomFieldDefinitions` above — reuse it
|
|
1053
|
+
// instead of a second `custom_field_defs` round trip (#5674 review).
|
|
1054
|
+
const keysNeedingQuery: string[] = []
|
|
1055
|
+
for (const key of cfSortKeys) {
|
|
1056
|
+
const kind = resolvedCustomFieldDefinitions
|
|
1057
|
+
? pickCfSortKind(resolvedCustomFieldDefinitions.index.get(normalizeDefinitionKey(key)))
|
|
1058
|
+
: null
|
|
1059
|
+
if (kind) cfSortKinds.set(key, kind)
|
|
1060
|
+
else keysNeedingQuery.push(key)
|
|
1061
|
+
}
|
|
1062
|
+
if (keysNeedingQuery.length > 0) {
|
|
1063
|
+
const sortEntityIds = Array.from(new Set(
|
|
1064
|
+
keysNeedingQuery
|
|
1065
|
+
.map((key) => keySource.get(key)?.entityId)
|
|
1066
|
+
.filter((id): id is EntityId => Boolean(id))
|
|
1067
|
+
.map((id) => String(id))
|
|
1068
|
+
))
|
|
1069
|
+
if (sortEntityIds.length > 0) {
|
|
1070
|
+
const cfSortOrgCandidates = resolveCfDefIndexOrgCandidates(opts.organizationIds, opts.organizationId ?? null)
|
|
1071
|
+
const kindRows = await db
|
|
1072
|
+
.selectFrom('custom_field_defs' as any)
|
|
1073
|
+
.select(['key' as any, 'kind' as any, 'organization_id' as any, 'tenant_id' as any])
|
|
1074
|
+
.where('entity_id' as any, 'in', sortEntityIds)
|
|
1075
|
+
.where('key' as any, 'in', keysNeedingQuery)
|
|
1076
|
+
.where('is_active' as any, '=', true)
|
|
1077
|
+
.where((eb: any) => eb.or([
|
|
1078
|
+
eb('tenant_id' as any, '=', tenantId),
|
|
1079
|
+
eb('tenant_id' as any, 'is', null),
|
|
1080
|
+
]))
|
|
1081
|
+
.where((eb: any) => cfSortOrgCandidates.length
|
|
1082
|
+
? eb.or([
|
|
1083
|
+
eb('organization_id' as any, 'is', null),
|
|
1084
|
+
eb('organization_id' as any, 'in', cfSortOrgCandidates),
|
|
1085
|
+
])
|
|
1086
|
+
: eb('organization_id' as any, 'is', null))
|
|
1087
|
+
.execute() as Array<{ key: string; kind: string | null; organization_id: string | null; tenant_id: string | null }>
|
|
1088
|
+
const winners = new Map<string, { kind: string; specificity: number }>()
|
|
1089
|
+
for (const row of kindRows) {
|
|
1090
|
+
if (!row.kind) continue
|
|
1091
|
+
const specificity = row.organization_id != null ? 2 : row.tenant_id != null ? 1 : 0
|
|
1092
|
+
const existing = winners.get(row.key)
|
|
1093
|
+
if (!existing || specificity >= existing.specificity) winners.set(row.key, { kind: row.kind, specificity })
|
|
1094
|
+
}
|
|
1095
|
+
for (const [key, entry] of winners) cfSortKinds.set(key, entry.kind)
|
|
1096
|
+
}
|
|
1097
|
+
}
|
|
1098
|
+
}
|
|
1099
|
+
|
|
997
1100
|
const cfValueExprByKey: Record<string, RawBuilder<string | null>> = {}
|
|
998
1101
|
const cfSelectedAliases: string[] = []
|
|
999
1102
|
const cfJsonAliases = new Set<string>()
|
|
1000
1103
|
const cfMultiAliasByAlias = new Map<string, string>()
|
|
1104
|
+
const cfSortAliases: string[] = []
|
|
1001
1105
|
for (const key of cfKeys) {
|
|
1002
1106
|
const source = keySource.get(key)
|
|
1003
1107
|
if (!source) continue
|
|
@@ -1127,6 +1231,12 @@ export class BasicQueryEngine implements QueryEngine {
|
|
|
1127
1231
|
// field covered by an encryption map such a leaf therefore compares against
|
|
1128
1232
|
// ciphertext and will not match.
|
|
1129
1233
|
//
|
|
1234
|
+
// This also means the #5803 plaintext-containment split (lib/search/containment) does not
|
|
1235
|
+
// apply here: `buildColumnOpExpression` below keeps a multi-word `like`/`ilike` leaf as one
|
|
1236
|
+
// literal pattern, so an OR-grouped search (e.g. `customers/api/people`'s multi-field
|
|
1237
|
+
// fallback) loses word-order independence on Basic while the Hybrid engine's OR groups
|
|
1238
|
+
// apply the split. Pre-existing divergence between the two engines; not tracked by an issue.
|
|
1239
|
+
//
|
|
1130
1240
|
// The count shape never populates cfValueExprByKey (it joins no cf tables), so
|
|
1131
1241
|
// its applicability test is key resolution itself — the same condition that
|
|
1132
1242
|
// gates the full shape's expression map — and a cf leaf compiles to a
|
|
@@ -1200,30 +1310,57 @@ export class BasicQueryEngine implements QueryEngine {
|
|
|
1200
1310
|
}
|
|
1201
1311
|
}
|
|
1202
1312
|
|
|
1203
|
-
// Sorting: base fields and cf:* (
|
|
1313
|
+
// Sorting: base fields and cf:* (a dedicated scalar alias, never the jsonb
|
|
1314
|
+
// projection alias — ordering by `to_jsonb(...)` compares arrays after every
|
|
1315
|
+
// scalar string regardless of contents, and a numeric kind must cast to
|
|
1316
|
+
// numeric instead of ordering the text CASE expression lexicographically (#5674)).
|
|
1317
|
+
let lastEmittedSortField: string | null = null
|
|
1204
1318
|
for (const s of isCountProjection ? [] : resolvedSorts) {
|
|
1205
1319
|
if (s.field.startsWith('cf:')) {
|
|
1206
1320
|
const key = s.field.slice(3)
|
|
1207
|
-
const
|
|
1208
|
-
|
|
1209
|
-
|
|
1210
|
-
const
|
|
1211
|
-
|
|
1212
|
-
|
|
1213
|
-
|
|
1321
|
+
const sortAlias = sanitize(`cf:${key}__sort`)
|
|
1322
|
+
if (!cfSortAliases.includes(sortAlias)) {
|
|
1323
|
+
const source = keySource.get(key)
|
|
1324
|
+
const kind = cfSortKinds.get(key)
|
|
1325
|
+
let sortExpr: RawBuilder<unknown> | null = null
|
|
1326
|
+
if (source && kind && NUMERIC_CF_SORT_KINDS.has(kind)) {
|
|
1327
|
+
const sourceAliasSafe = sanitize(source.alias || 'src')
|
|
1328
|
+
const keyAliasSafe = sanitize(key)
|
|
1329
|
+
const valAlias = `cfv_${sourceAliasSafe}_${keyAliasSafe}`
|
|
1330
|
+
const numericColumn = kind === 'integer' ? 'value_int' : 'value_float'
|
|
1331
|
+
sortExpr = sql<string | null>`max((${sql.ref(`${valAlias}.${numericColumn}`)})::numeric)`
|
|
1332
|
+
} else {
|
|
1333
|
+
const expr = cfValueExprByKey[key]
|
|
1334
|
+
if (expr) sortExpr = sql<string | null>`max(${expr})`
|
|
1335
|
+
}
|
|
1336
|
+
if (sortExpr) {
|
|
1337
|
+
q = q.select(sortExpr.as(sortAlias))
|
|
1338
|
+
cfSortAliases.push(sortAlias)
|
|
1214
1339
|
}
|
|
1215
1340
|
}
|
|
1216
|
-
// Only order by an alias the
|
|
1217
|
-
//
|
|
1218
|
-
//
|
|
1219
|
-
//
|
|
1220
|
-
if (!requiresPlaintextSort &&
|
|
1221
|
-
|
|
1341
|
+
// Only order by an alias the query actually selects. A key that resolved
|
|
1342
|
+
// to no expression is dropped rather than emitted as an unselected alias,
|
|
1343
|
+
// which is what the base-column branch below already does when
|
|
1344
|
+
// `resolveBaseColumn` returns null.
|
|
1345
|
+
if (!requiresPlaintextSort && cfSortAliases.includes(sortAlias)) {
|
|
1346
|
+
const direction = sql.raw((s.dir ?? 'asc') === 'desc' ? 'desc' : 'asc')
|
|
1347
|
+
q = q.orderBy(sql`${sql.ref(sortAlias)} ${direction} NULLS LAST`)
|
|
1348
|
+
lastEmittedSortField = s.field
|
|
1222
1349
|
}
|
|
1223
1350
|
} else {
|
|
1224
|
-
if (!requiresPlaintextSort)
|
|
1351
|
+
if (!requiresPlaintextSort) {
|
|
1352
|
+
q = q.orderBy(qualify(s.field), (s.dir ?? 'asc') as any)
|
|
1353
|
+
lastEmittedSortField = s.field
|
|
1354
|
+
}
|
|
1225
1355
|
}
|
|
1226
1356
|
}
|
|
1357
|
+
// Stable tiebreak so ties (or NULLs) don't reorder arbitrarily across pages
|
|
1358
|
+
// (#5674). Only appended when a sort term actually made it into the ORDER
|
|
1359
|
+
// BY — an entirely unresolved cf sort must stay a no-op, not gain an
|
|
1360
|
+
// incidental `id` ordering the caller never asked for.
|
|
1361
|
+
if (!isCountProjection && lastEmittedSortField !== null && lastEmittedSortField !== 'id') {
|
|
1362
|
+
q = q.orderBy(qualify('id'), 'asc' as any)
|
|
1363
|
+
}
|
|
1227
1364
|
|
|
1228
1365
|
// Deduplicate if we joined CFs or extensions by grouping on base id. The count
|
|
1229
1366
|
// shape has neither, and must stay barrier-free for its LIMIT to bind.
|
|
@@ -1235,7 +1372,7 @@ export class BasicQueryEngine implements QueryEngine {
|
|
|
1235
1372
|
q = q.groupBy(`${table}.id`)
|
|
1236
1373
|
}
|
|
1237
1374
|
|
|
1238
|
-
return { builder: q, hasJoinedAggregates, cfJsonAliases, cfMultiAliasByAlias, resolvedCustomFieldDefinitions }
|
|
1375
|
+
return { builder: q, hasJoinedAggregates, cfJsonAliases, cfMultiAliasByAlias, cfSortAliases, resolvedCustomFieldDefinitions }
|
|
1239
1376
|
}
|
|
1240
1377
|
|
|
1241
1378
|
// Pagination
|
|
@@ -1247,6 +1384,7 @@ export class BasicQueryEngine implements QueryEngine {
|
|
|
1247
1384
|
hasJoinedAggregates,
|
|
1248
1385
|
cfJsonAliases,
|
|
1249
1386
|
cfMultiAliasByAlias,
|
|
1387
|
+
cfSortAliases,
|
|
1250
1388
|
resolvedCustomFieldDefinitions,
|
|
1251
1389
|
} = await buildQuery('full')
|
|
1252
1390
|
|
|
@@ -1330,6 +1468,19 @@ export class BasicQueryEngine implements QueryEngine {
|
|
|
1330
1468
|
}
|
|
1331
1469
|
}
|
|
1332
1470
|
|
|
1471
|
+
// The `cf:<key>__sort` alias exists only to give a `cf:` sort's ORDER BY
|
|
1472
|
+
// something to reference (and, for a numeric kind, to host the cast) — it
|
|
1473
|
+
// is never a requested output field. Strip it before rows leave this
|
|
1474
|
+
// function so it doesn't leak into API responses as a phantom custom field
|
|
1475
|
+
// `<key>__sort` (unlike `__is_multi`, which `normalizeCfJsonAliases`
|
|
1476
|
+
// already deletes) (#5674 review).
|
|
1477
|
+
const stripCfSortAliases = (rows: ResultRow[]) => {
|
|
1478
|
+
if (cfSortAliases.length === 0) return
|
|
1479
|
+
for (const row of rows) {
|
|
1480
|
+
for (const alias of cfSortAliases) delete row[alias]
|
|
1481
|
+
}
|
|
1482
|
+
}
|
|
1483
|
+
|
|
1333
1484
|
let pagedItems: ResultRow[]
|
|
1334
1485
|
let encryptedSortRowCapWarning: EncryptedSortRowCapWarning | undefined
|
|
1335
1486
|
|
|
@@ -1372,6 +1523,7 @@ export class BasicQueryEngine implements QueryEngine {
|
|
|
1372
1523
|
} else {
|
|
1373
1524
|
const pageRows = await qFull.where(qualify('id'), 'in', pageIds).execute() as ResultRow[]
|
|
1374
1525
|
normalizeCfJsonAliases(pageRows)
|
|
1526
|
+
stripCfSortAliases(pageRows)
|
|
1375
1527
|
const decryptedPageRows = decryptPayload
|
|
1376
1528
|
? await mapWithConcurrency(pageRows, DECRYPT_CONCURRENCY, decryptRow)
|
|
1377
1529
|
: pageRows
|
|
@@ -1384,6 +1536,7 @@ export class BasicQueryEngine implements QueryEngine {
|
|
|
1384
1536
|
const dataQuery = qFull.limit(pageSize).offset((page - 1) * pageSize)
|
|
1385
1537
|
const items = await dataQuery.execute() as ResultRow[]
|
|
1386
1538
|
normalizeCfJsonAliases(items)
|
|
1539
|
+
stripCfSortAliases(items)
|
|
1387
1540
|
pagedItems = decryptPayload
|
|
1388
1541
|
? await mapWithConcurrency(items, DECRYPT_CONCURRENCY, decryptRow)
|
|
1389
1542
|
: items
|
|
@@ -113,6 +113,33 @@ describe('OM_SEARCH_FIELD_BLOCKLIST parsing', () => {
|
|
|
113
113
|
})
|
|
114
114
|
})
|
|
115
115
|
|
|
116
|
+
describe('OM_SEARCH_USE_ILIKE_FOR_NON_ENCRYPTED_FIELDS', () => {
|
|
117
|
+
const originalValue = process.env.OM_SEARCH_USE_ILIKE_FOR_NON_ENCRYPTED_FIELDS
|
|
118
|
+
|
|
119
|
+
afterEach(() => {
|
|
120
|
+
if (originalValue === undefined) {
|
|
121
|
+
delete process.env.OM_SEARCH_USE_ILIKE_FOR_NON_ENCRYPTED_FIELDS
|
|
122
|
+
} else {
|
|
123
|
+
process.env.OM_SEARCH_USE_ILIKE_FOR_NON_ENCRYPTED_FIELDS = originalValue
|
|
124
|
+
}
|
|
125
|
+
})
|
|
126
|
+
|
|
127
|
+
// #5383: the switch stays off by default until tokenization is made ILIKE-equivalent, so the
|
|
128
|
+
// rewrite-everything behavior is unchanged for a deployment that does not opt in. #5803 is the
|
|
129
|
+
// correctness gap this switch closes when a deployment opts in ahead of that follow-up.
|
|
130
|
+
it('defaults to off so the legacy rewrite is unchanged for every column', () => {
|
|
131
|
+
delete process.env.OM_SEARCH_USE_ILIKE_FOR_NON_ENCRYPTED_FIELDS
|
|
132
|
+
|
|
133
|
+
expect(resolveSearchConfig().useIlikeForNonEncryptedFields).toBe(false)
|
|
134
|
+
})
|
|
135
|
+
|
|
136
|
+
it('can be switched on to apply a declared ilike on a plaintext column as written', () => {
|
|
137
|
+
process.env.OM_SEARCH_USE_ILIKE_FOR_NON_ENCRYPTED_FIELDS = 'true'
|
|
138
|
+
|
|
139
|
+
expect(resolveSearchConfig().useIlikeForNonEncryptedFields).toBe(true)
|
|
140
|
+
})
|
|
141
|
+
})
|
|
142
|
+
|
|
116
143
|
describe('search token limits', () => {
|
|
117
144
|
const variableNames = [
|
|
118
145
|
'OM_SEARCH_MAX_FIELD_CHARS',
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
import { buildContainmentPatterns, MAX_CONTAINMENT_WORDS } from '../containment'
|
|
2
|
+
|
|
3
|
+
describe('buildContainmentPatterns (#5803)', () => {
|
|
4
|
+
test('splits a multi-word contains pattern into one pattern per word', () => {
|
|
5
|
+
// The token subquery matched every token in any order with anything between them. Reproducing
|
|
6
|
+
// that on SQL means ANDing per word, which is what TC-RESO-009 exercises through the API:
|
|
7
|
+
// `?search=Warehouse <stamp>` has to keep matching `Warehouse A <stamp>`.
|
|
8
|
+
expect(buildContainmentPatterns('%Warehouse 1757%')).toEqual(['%Warehouse%', '%1757%'])
|
|
9
|
+
})
|
|
10
|
+
|
|
11
|
+
test('collapses runs of whitespace rather than emitting empty patterns', () => {
|
|
12
|
+
expect(buildContainmentPatterns('%John Smith%')).toEqual(['%John%', '%Smith%'])
|
|
13
|
+
})
|
|
14
|
+
|
|
15
|
+
test('leaves a single-word pattern exactly as the caller declared it', () => {
|
|
16
|
+
// The reported #5803 case: the distinguishing fragment must reach SQL untouched, or the exact
|
|
17
|
+
// row cannot come back at all.
|
|
18
|
+
expect(buildContainmentPatterns('%2026-08%')).toEqual(['%2026-08%'])
|
|
19
|
+
})
|
|
20
|
+
|
|
21
|
+
test('leaves a term too short to tokenize alone', () => {
|
|
22
|
+
expect(buildContainmentPatterns('%08%')).toEqual(['%08%'])
|
|
23
|
+
})
|
|
24
|
+
|
|
25
|
+
test('keeps escaped wildcards attached to their word', () => {
|
|
26
|
+
// escapeLikePattern turns a literal `%` into `\%`; splitting must not treat it as a wildcard
|
|
27
|
+
// and must not tear the escape off its word.
|
|
28
|
+
expect(buildContainmentPatterns('%50\\% off%')).toEqual(['%50\\%%', '%off%'])
|
|
29
|
+
})
|
|
30
|
+
|
|
31
|
+
test('does not split a structured pattern carrying its own wildcards', () => {
|
|
32
|
+
// A caller that hand-built `%a% b%` asked for that exact shape; re-splitting it would silently
|
|
33
|
+
// rewrite a predicate this helper has no business reinterpreting.
|
|
34
|
+
expect(buildContainmentPatterns('%a% b%')).toEqual(['%a% b%'])
|
|
35
|
+
})
|
|
36
|
+
|
|
37
|
+
test('does not split an anchored pattern', () => {
|
|
38
|
+
// `startsWith` / `endsWith` terms are anchored on purpose; per-word ANDing would drop the
|
|
39
|
+
// anchor and widen the match.
|
|
40
|
+
expect(buildContainmentPatterns('John Smith%')).toEqual(['John Smith%'])
|
|
41
|
+
expect(buildContainmentPatterns('%John Smith')).toEqual(['%John Smith'])
|
|
42
|
+
})
|
|
43
|
+
|
|
44
|
+
test('treats a trailing escaped percent as a literal, not as the closing wildcard', () => {
|
|
45
|
+
expect(buildContainmentPatterns('%John Smith\\%')).toEqual(['%John Smith\\%'])
|
|
46
|
+
})
|
|
47
|
+
|
|
48
|
+
test('splits a term at the word cap', () => {
|
|
49
|
+
const words = Array.from({ length: MAX_CONTAINMENT_WORDS }, (_, index) => `w${index}`)
|
|
50
|
+
expect(buildContainmentPatterns(`%${words.join(' ')}%`)).toEqual(words.map((word) => `%${word}%`))
|
|
51
|
+
})
|
|
52
|
+
|
|
53
|
+
test('falls back to the single verbatim pattern above the word cap', () => {
|
|
54
|
+
// Most list routes declare `search` as an unbounded string and this repo ships no trigram
|
|
55
|
+
// index, so an unbounded per-word AND would let one request compile into an unbounded number
|
|
56
|
+
// of sequential-scan predicates. Past the cap, this returns to the pre-split behavior instead.
|
|
57
|
+
const words = Array.from({ length: MAX_CONTAINMENT_WORDS + 1 }, (_, index) => `w${index}`)
|
|
58
|
+
const pattern = `%${words.join(' ')}%`
|
|
59
|
+
expect(buildContainmentPatterns(pattern)).toEqual([pattern])
|
|
60
|
+
})
|
|
61
|
+
})
|
package/src/lib/search/config.ts
CHANGED
|
@@ -9,12 +9,34 @@ export type SearchConfig = {
|
|
|
9
9
|
hashAlgorithm: 'sha256' | 'sha1' | 'md5'
|
|
10
10
|
storeRawTokens: boolean
|
|
11
11
|
/**
|
|
12
|
-
* When true, a like/ilike on a PLAINTEXT base column runs as
|
|
13
|
-
*
|
|
14
|
-
*
|
|
15
|
-
*
|
|
16
|
-
*
|
|
17
|
-
*
|
|
12
|
+
* When true, a like/ilike on a PLAINTEXT base column runs as SQL ILIKE — one containment
|
|
13
|
+
* predicate per word of the term, ANDed — instead of being rewritten into an approximate
|
|
14
|
+
* search-token match; encrypted columns always keep the token path (ILIKE against ciphertext
|
|
15
|
+
* cannot match).
|
|
16
|
+
*
|
|
17
|
+
* Off by default, per #5383: the token store is expected to become faster than ILIKE once
|
|
18
|
+
* tokenization is made semantically equivalent to it, so the plan there is to keep this switch
|
|
19
|
+
* off until that follow-up lands rather than trade performance for correctness by default. #5803
|
|
20
|
+
* documents the correctness gap this switch closes when enabled: the token rewrite is lossy in a
|
|
21
|
+
* way that silently returns the WRONG record rather than merely extra ones (tokenization splits
|
|
22
|
+
* on non-alphanumerics and drops fragments under minTokenLength, so `2026-08` and `2026-01` both
|
|
23
|
+
* reduce to {202, 2026} and a picker offers the neighbouring period; a term that tokenizes to
|
|
24
|
+
* nothing (`08`) drops the predicate entirely and matches every row) — a deployment that hits
|
|
25
|
+
* that gap before #5383 lands can opt in here.
|
|
26
|
+
*
|
|
27
|
+
* Per-word ANDing (see lib/search/containment) is a trade-off, not a strict improvement, over
|
|
28
|
+
* the single-literal ILIKE #4622 originally introduced: the token subquery matched a value
|
|
29
|
+
* carrying every token in any order with anything between them, so `?search=Warehouse 1757`
|
|
30
|
+
* must keep matching `Warehouse A 1757` — a single verbatim `ILIKE '%Warehouse 1757%'` would
|
|
31
|
+
* not, and TC-RESO-009 pins that as required behavior. The same word-order independence also
|
|
32
|
+
* widens multi-word document-number searches: `?search=ZK 1/2026` now also matches
|
|
33
|
+
* `ZK 11/2026` and `1/2026 ZK`, where the old single-literal ILIKE matched neither.
|
|
34
|
+
*
|
|
35
|
+
* Set `OM_SEARCH_USE_ILIKE_FOR_NON_ENCRYPTED_FIELDS=true` to opt into declared-column ILIKE
|
|
36
|
+
* ahead of #5383 — worth doing when the #5803 wrong-record symptom is hit in practice. Leaving it
|
|
37
|
+
* unset keeps the legacy rewrite-everything behavior, including the token index's prefix matching
|
|
38
|
+
* (`?search=ware` matching `Warehouse` when `enablePartials` is on, which literal containment
|
|
39
|
+
* gives only where the fragment really is a substring).
|
|
18
40
|
*/
|
|
19
41
|
useIlikeForNonEncryptedFields?: boolean
|
|
20
42
|
blocklistedFields: string[]
|
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Splits a `contains` like/ilike pattern into one pattern per whitespace-separated word, so a
|
|
3
|
+
* plaintext column taken off the hashed-token path keeps the word-order-independent matching the
|
|
4
|
+
* token index provided.
|
|
5
|
+
*
|
|
6
|
+
* The token path matches a value when it carries EVERY token of the term, in any order and with
|
|
7
|
+
* anything in between: `?search=Warehouse 1757` matches `Warehouse A 1757`. A single verbatim
|
|
8
|
+
* `name ILIKE '%Warehouse 1757%'` is literal substring containment, so the `A ` sitting between the
|
|
9
|
+
* two words defeats it — that is a capability every list grid has today, and #5803's fix must not
|
|
10
|
+
* take it away (`TC-RESO-009` pins it).
|
|
11
|
+
*
|
|
12
|
+
* ANDing one containment predicate per word reproduces the token semantics exactly on a column the
|
|
13
|
+
* engine can read, without the token path's two lossy steps: nothing is dropped for being shorter
|
|
14
|
+
* than `minTokenLength` (`?search=08` filters instead of matching every row) and nothing is split on
|
|
15
|
+
* non-alphanumerics (`?search=2026-08` no longer collapses onto `2026-01`).
|
|
16
|
+
*
|
|
17
|
+
* Splitting is deliberately narrow — a pattern is only split when it is unambiguously the
|
|
18
|
+
* `%term%` shape `buildIlikeTerm(value, 'contains')` produces:
|
|
19
|
+
*
|
|
20
|
+
* - it opens and closes with a wildcard `%` (a trailing `\%` is an escaped literal, not a wildcard);
|
|
21
|
+
* - the term between them carries no unescaped `%` or `_`, so a hand-built structured pattern such
|
|
22
|
+
* as `%a% b%` is left exactly as the caller wrote it;
|
|
23
|
+
* - the term holds at least two words.
|
|
24
|
+
*
|
|
25
|
+
* Anything else returns the input unchanged as a single pattern, which is the caller's existing
|
|
26
|
+
* behavior.
|
|
27
|
+
*
|
|
28
|
+
* The split is also capped at {@link MAX_CONTAINMENT_WORDS} words. Most list routes declare
|
|
29
|
+
* `search` as an unbounded string, this repository ships no trigram index for the resulting
|
|
30
|
+
* `ILIKE`, and the hybrid engine's `$or` groups multiply the split across every leaf — so an
|
|
31
|
+
* attacker-controlled term with thousands of words would otherwise compile into thousands of
|
|
32
|
+
* sequential-scan predicates from one request. A term at or under the cap keeps the per-word AND
|
|
33
|
+
* semantics; over the cap it falls back to the single verbatim pattern, which is bounded and was
|
|
34
|
+
* this helper's own behavior before the split existed.
|
|
35
|
+
*/
|
|
36
|
+
export const MAX_CONTAINMENT_WORDS = 10
|
|
37
|
+
|
|
38
|
+
export function buildContainmentPatterns(pattern: string): string[] {
|
|
39
|
+
if (!isWrappedContainsPattern(pattern)) return [pattern]
|
|
40
|
+
const term = pattern.slice(1, -1)
|
|
41
|
+
if (hasUnescapedWildcard(term)) return [pattern]
|
|
42
|
+
const words = term.split(/\s+/).filter((word) => word.length > 0)
|
|
43
|
+
if (words.length < 2 || words.length > MAX_CONTAINMENT_WORDS) return [pattern]
|
|
44
|
+
return words.map((word) => `%${word}%`)
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
function isWrappedContainsPattern(pattern: string): boolean {
|
|
48
|
+
if (pattern.length < 2) return false
|
|
49
|
+
if (!pattern.startsWith('%') || !pattern.endsWith('%')) return false
|
|
50
|
+
return countTrailingBackslashes(pattern, pattern.length - 2) % 2 === 0
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
function hasUnescapedWildcard(term: string): boolean {
|
|
54
|
+
for (let index = 0; index < term.length; index += 1) {
|
|
55
|
+
const char = term[index]
|
|
56
|
+
if (char !== '%' && char !== '_') continue
|
|
57
|
+
if (countTrailingBackslashes(term, index - 1) % 2 === 0) return true
|
|
58
|
+
}
|
|
59
|
+
return false
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
function countTrailingBackslashes(value: string, fromIndex: number): number {
|
|
63
|
+
let count = 0
|
|
64
|
+
for (let index = fromIndex; index >= 0 && value[index] === '\\'; index -= 1) count += 1
|
|
65
|
+
return count
|
|
66
|
+
}
|