@open-mercato/shared 0.8.1-develop.7275.1.f772b944d4 → 0.8.1-develop.7295.1.d0e0e33014
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.turbo/turbo-build.log +2 -1
- package/AGENTS.md +6 -3
- package/build.mjs +4 -1
- package/dist/lib/crud/errors.js +6 -1
- package/dist/lib/crud/errors.js.map +2 -2
- package/dist/lib/crud/factory.js +22 -13
- package/dist/lib/crud/factory.js.map +2 -2
- package/dist/lib/custom-fields/kinds.js +113 -0
- package/dist/lib/custom-fields/kinds.js.map +7 -0
- package/dist/lib/encryption/indexDoc.js +12 -9
- package/dist/lib/encryption/indexDoc.js.map +2 -2
- package/dist/lib/location/countries.generated.js +265 -0
- package/dist/lib/location/countries.generated.js.map +7 -0
- package/dist/lib/location/countries.js +2 -13
- package/dist/lib/location/countries.js.map +2 -2
- package/dist/lib/navigation/pageReload.js +11 -0
- package/dist/lib/navigation/pageReload.js.map +7 -0
- package/dist/lib/query/encrypted-sort.js +4 -1
- package/dist/lib/query/encrypted-sort.js.map +2 -2
- package/dist/lib/query/engine.js +97 -10
- package/dist/lib/query/engine.js.map +3 -3
- package/dist/lib/schedule/interval.js +33 -0
- package/dist/lib/schedule/interval.js.map +7 -0
- package/dist/lib/schedule/invalidScheduleValue.js +19 -0
- package/dist/lib/schedule/invalidScheduleValue.js.map +7 -0
- package/dist/lib/search/config.js.map +2 -2
- package/dist/lib/search/containment.js +32 -0
- package/dist/lib/search/containment.js.map +7 -0
- package/dist/lib/version.js +1 -1
- package/dist/lib/version.js.map +1 -1
- package/package.json +3 -2
- package/scripts/generate-countries.mjs +75 -0
- package/src/lib/crud/__tests__/crud-factory.test.ts +177 -0
- package/src/lib/crud/__tests__/errors.test.ts +31 -1
- package/src/lib/crud/errors.ts +16 -0
- package/src/lib/crud/factory.ts +37 -15
- package/src/lib/custom-fields/__tests__/kinds.test.ts +208 -0
- package/src/lib/custom-fields/kinds.ts +205 -0
- package/src/lib/encryption/__tests__/indexDoc.custom-field-kinds.test.ts +127 -0
- package/src/lib/encryption/__tests__/indexDoc.test.ts +39 -0
- package/src/lib/encryption/indexDoc.ts +16 -7
- package/src/lib/location/__tests__/countries.test.ts +25 -0
- package/src/lib/location/countries.generated.ts +264 -0
- package/src/lib/location/countries.ts +2 -23
- package/src/lib/navigation/__tests__/pageReload.test.ts +58 -0
- package/src/lib/navigation/pageReload.ts +43 -0
- package/src/lib/query/__tests__/engine.test.ts +208 -16
- package/src/lib/query/encrypted-sort.ts +8 -1
- package/src/lib/query/engine.ts +172 -19
- package/src/lib/schedule/__tests__/interval.test.ts +37 -0
- package/src/lib/schedule/interval.ts +45 -0
- package/src/lib/schedule/invalidScheduleValue.ts +33 -0
- package/src/lib/search/__tests__/config.test.ts +27 -0
- package/src/lib/search/__tests__/containment.test.ts +61 -0
- package/src/lib/search/config.ts +28 -6
- package/src/lib/search/containment.ts +66 -0
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Canonical interval-format rules for recurring schedules.
|
|
3
|
+
*
|
|
4
|
+
* Both the scheduler runtime and the API validators that reject a schedule
|
|
5
|
+
* before it is persisted read the format from here, so the documented format
|
|
6
|
+
* (`<number><unit>`, e.g. `15m`, `1h`, `24h`) cannot drift between the layer
|
|
7
|
+
* that accepts a value and the layer that has to run it.
|
|
8
|
+
*/
|
|
9
|
+
export const MIN_SCHEDULE_INTERVAL_MS = 60 * 1000
|
|
10
|
+
|
|
11
|
+
export const SCHEDULE_INTERVAL_PATTERN = /^(\d+)(s|m|h|d)$/
|
|
12
|
+
|
|
13
|
+
const UNIT_MULTIPLIERS: Record<string, number> = {
|
|
14
|
+
s: 1000,
|
|
15
|
+
m: 60 * 1000,
|
|
16
|
+
h: 60 * 60 * 1000,
|
|
17
|
+
d: 24 * 60 * 60 * 1000,
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
export type ScheduleIntervalUnit = 's' | 'm' | 'h' | 'd'
|
|
21
|
+
|
|
22
|
+
export type ScheduleIntervalParts = {
|
|
23
|
+
amount: number
|
|
24
|
+
unit: ScheduleIntervalUnit
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
export function matchScheduleInterval(interval: string): ScheduleIntervalParts | null {
|
|
28
|
+
const match = SCHEDULE_INTERVAL_PATTERN.exec(interval)
|
|
29
|
+
if (!match) return null
|
|
30
|
+
return { amount: Number.parseInt(match[1], 10), unit: match[2] as ScheduleIntervalUnit }
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
export function parseScheduleInterval(interval: string): number {
|
|
34
|
+
const parts = matchScheduleInterval(interval)
|
|
35
|
+
if (!parts) {
|
|
36
|
+
throw new Error(`Invalid interval format: ${interval}. Expected format: <number><unit> (e.g., 15m, 2h, 1d)`)
|
|
37
|
+
}
|
|
38
|
+
return parts.amount * UNIT_MULTIPLIERS[parts.unit]
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
export function isValidScheduleInterval(interval: string): boolean {
|
|
42
|
+
const parts = matchScheduleInterval(interval)
|
|
43
|
+
if (!parts) return false
|
|
44
|
+
return parts.amount * UNIT_MULTIPLIERS[parts.unit] >= MIN_SCHEDULE_INTERVAL_MS
|
|
45
|
+
}
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
export const INVALID_SCHEDULE_VALUE_CODE = 'INVALID_SCHEDULE_VALUE'
|
|
2
|
+
|
|
3
|
+
export type ScheduleValueKind = 'cron' | 'interval'
|
|
4
|
+
|
|
5
|
+
export type InvalidScheduleValueError = Error & {
|
|
6
|
+
code: typeof INVALID_SCHEDULE_VALUE_CODE
|
|
7
|
+
scheduleType: ScheduleValueKind
|
|
8
|
+
scheduleValue: string
|
|
9
|
+
}
|
|
10
|
+
|
|
11
|
+
export function createInvalidScheduleValueError(
|
|
12
|
+
scheduleType: ScheduleValueKind,
|
|
13
|
+
scheduleValue: string,
|
|
14
|
+
message: string,
|
|
15
|
+
): InvalidScheduleValueError {
|
|
16
|
+
return Object.assign(new Error(message), {
|
|
17
|
+
code: INVALID_SCHEDULE_VALUE_CODE,
|
|
18
|
+
scheduleType,
|
|
19
|
+
scheduleValue,
|
|
20
|
+
} as const)
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
/**
|
|
24
|
+
* Structural check rather than `instanceof`: the error crosses a package
|
|
25
|
+
* boundary (and a production bundle), where a duplicated class identity would
|
|
26
|
+
* make `instanceof` silently false.
|
|
27
|
+
*/
|
|
28
|
+
export function isInvalidScheduleValueError(error: unknown): error is InvalidScheduleValueError {
|
|
29
|
+
if (typeof error !== 'object' || error === null) return false
|
|
30
|
+
const candidate = error as { code?: unknown; scheduleType?: unknown }
|
|
31
|
+
return candidate.code === INVALID_SCHEDULE_VALUE_CODE
|
|
32
|
+
&& (candidate.scheduleType === 'cron' || candidate.scheduleType === 'interval')
|
|
33
|
+
}
|
|
@@ -113,6 +113,33 @@ describe('OM_SEARCH_FIELD_BLOCKLIST parsing', () => {
|
|
|
113
113
|
})
|
|
114
114
|
})
|
|
115
115
|
|
|
116
|
+
describe('OM_SEARCH_USE_ILIKE_FOR_NON_ENCRYPTED_FIELDS', () => {
|
|
117
|
+
const originalValue = process.env.OM_SEARCH_USE_ILIKE_FOR_NON_ENCRYPTED_FIELDS
|
|
118
|
+
|
|
119
|
+
afterEach(() => {
|
|
120
|
+
if (originalValue === undefined) {
|
|
121
|
+
delete process.env.OM_SEARCH_USE_ILIKE_FOR_NON_ENCRYPTED_FIELDS
|
|
122
|
+
} else {
|
|
123
|
+
process.env.OM_SEARCH_USE_ILIKE_FOR_NON_ENCRYPTED_FIELDS = originalValue
|
|
124
|
+
}
|
|
125
|
+
})
|
|
126
|
+
|
|
127
|
+
// #5383: the switch stays off by default until tokenization is made ILIKE-equivalent, so the
|
|
128
|
+
// rewrite-everything behavior is unchanged for a deployment that does not opt in. #5803 is the
|
|
129
|
+
// correctness gap this switch closes when a deployment opts in ahead of that follow-up.
|
|
130
|
+
it('defaults to off so the legacy rewrite is unchanged for every column', () => {
|
|
131
|
+
delete process.env.OM_SEARCH_USE_ILIKE_FOR_NON_ENCRYPTED_FIELDS
|
|
132
|
+
|
|
133
|
+
expect(resolveSearchConfig().useIlikeForNonEncryptedFields).toBe(false)
|
|
134
|
+
})
|
|
135
|
+
|
|
136
|
+
it('can be switched on to apply a declared ilike on a plaintext column as written', () => {
|
|
137
|
+
process.env.OM_SEARCH_USE_ILIKE_FOR_NON_ENCRYPTED_FIELDS = 'true'
|
|
138
|
+
|
|
139
|
+
expect(resolveSearchConfig().useIlikeForNonEncryptedFields).toBe(true)
|
|
140
|
+
})
|
|
141
|
+
})
|
|
142
|
+
|
|
116
143
|
describe('search token limits', () => {
|
|
117
144
|
const variableNames = [
|
|
118
145
|
'OM_SEARCH_MAX_FIELD_CHARS',
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
import { buildContainmentPatterns, MAX_CONTAINMENT_WORDS } from '../containment'
|
|
2
|
+
|
|
3
|
+
describe('buildContainmentPatterns (#5803)', () => {
|
|
4
|
+
test('splits a multi-word contains pattern into one pattern per word', () => {
|
|
5
|
+
// The token subquery matched every token in any order with anything between them. Reproducing
|
|
6
|
+
// that on SQL means ANDing per word, which is what TC-RESO-009 exercises through the API:
|
|
7
|
+
// `?search=Warehouse <stamp>` has to keep matching `Warehouse A <stamp>`.
|
|
8
|
+
expect(buildContainmentPatterns('%Warehouse 1757%')).toEqual(['%Warehouse%', '%1757%'])
|
|
9
|
+
})
|
|
10
|
+
|
|
11
|
+
test('collapses runs of whitespace rather than emitting empty patterns', () => {
|
|
12
|
+
expect(buildContainmentPatterns('%John Smith%')).toEqual(['%John%', '%Smith%'])
|
|
13
|
+
})
|
|
14
|
+
|
|
15
|
+
test('leaves a single-word pattern exactly as the caller declared it', () => {
|
|
16
|
+
// The reported #5803 case: the distinguishing fragment must reach SQL untouched, or the exact
|
|
17
|
+
// row cannot come back at all.
|
|
18
|
+
expect(buildContainmentPatterns('%2026-08%')).toEqual(['%2026-08%'])
|
|
19
|
+
})
|
|
20
|
+
|
|
21
|
+
test('leaves a term too short to tokenize alone', () => {
|
|
22
|
+
expect(buildContainmentPatterns('%08%')).toEqual(['%08%'])
|
|
23
|
+
})
|
|
24
|
+
|
|
25
|
+
test('keeps escaped wildcards attached to their word', () => {
|
|
26
|
+
// escapeLikePattern turns a literal `%` into `\%`; splitting must not treat it as a wildcard
|
|
27
|
+
// and must not tear the escape off its word.
|
|
28
|
+
expect(buildContainmentPatterns('%50\\% off%')).toEqual(['%50\\%%', '%off%'])
|
|
29
|
+
})
|
|
30
|
+
|
|
31
|
+
test('does not split a structured pattern carrying its own wildcards', () => {
|
|
32
|
+
// A caller that hand-built `%a% b%` asked for that exact shape; re-splitting it would silently
|
|
33
|
+
// rewrite a predicate this helper has no business reinterpreting.
|
|
34
|
+
expect(buildContainmentPatterns('%a% b%')).toEqual(['%a% b%'])
|
|
35
|
+
})
|
|
36
|
+
|
|
37
|
+
test('does not split an anchored pattern', () => {
|
|
38
|
+
// `startsWith` / `endsWith` terms are anchored on purpose; per-word ANDing would drop the
|
|
39
|
+
// anchor and widen the match.
|
|
40
|
+
expect(buildContainmentPatterns('John Smith%')).toEqual(['John Smith%'])
|
|
41
|
+
expect(buildContainmentPatterns('%John Smith')).toEqual(['%John Smith'])
|
|
42
|
+
})
|
|
43
|
+
|
|
44
|
+
test('treats a trailing escaped percent as a literal, not as the closing wildcard', () => {
|
|
45
|
+
expect(buildContainmentPatterns('%John Smith\\%')).toEqual(['%John Smith\\%'])
|
|
46
|
+
})
|
|
47
|
+
|
|
48
|
+
test('splits a term at the word cap', () => {
|
|
49
|
+
const words = Array.from({ length: MAX_CONTAINMENT_WORDS }, (_, index) => `w${index}`)
|
|
50
|
+
expect(buildContainmentPatterns(`%${words.join(' ')}%`)).toEqual(words.map((word) => `%${word}%`))
|
|
51
|
+
})
|
|
52
|
+
|
|
53
|
+
test('falls back to the single verbatim pattern above the word cap', () => {
|
|
54
|
+
// Most list routes declare `search` as an unbounded string and this repo ships no trigram
|
|
55
|
+
// index, so an unbounded per-word AND would let one request compile into an unbounded number
|
|
56
|
+
// of sequential-scan predicates. Past the cap, this returns to the pre-split behavior instead.
|
|
57
|
+
const words = Array.from({ length: MAX_CONTAINMENT_WORDS + 1 }, (_, index) => `w${index}`)
|
|
58
|
+
const pattern = `%${words.join(' ')}%`
|
|
59
|
+
expect(buildContainmentPatterns(pattern)).toEqual([pattern])
|
|
60
|
+
})
|
|
61
|
+
})
|
package/src/lib/search/config.ts
CHANGED
|
@@ -9,12 +9,34 @@ export type SearchConfig = {
|
|
|
9
9
|
hashAlgorithm: 'sha256' | 'sha1' | 'md5'
|
|
10
10
|
storeRawTokens: boolean
|
|
11
11
|
/**
|
|
12
|
-
* When true, a like/ilike on a PLAINTEXT base column runs as
|
|
13
|
-
*
|
|
14
|
-
*
|
|
15
|
-
*
|
|
16
|
-
*
|
|
17
|
-
*
|
|
12
|
+
* When true, a like/ilike on a PLAINTEXT base column runs as SQL ILIKE — one containment
|
|
13
|
+
* predicate per word of the term, ANDed — instead of being rewritten into an approximate
|
|
14
|
+
* search-token match; encrypted columns always keep the token path (ILIKE against ciphertext
|
|
15
|
+
* cannot match).
|
|
16
|
+
*
|
|
17
|
+
* Off by default, per #5383: the token store is expected to become faster than ILIKE once
|
|
18
|
+
* tokenization is made semantically equivalent to it, so the plan there is to keep this switch
|
|
19
|
+
* off until that follow-up lands rather than trade performance for correctness by default. #5803
|
|
20
|
+
* documents the correctness gap this switch closes when enabled: the token rewrite is lossy in a
|
|
21
|
+
* way that silently returns the WRONG record rather than merely extra ones (tokenization splits
|
|
22
|
+
* on non-alphanumerics and drops fragments under minTokenLength, so `2026-08` and `2026-01` both
|
|
23
|
+
* reduce to {202, 2026} and a picker offers the neighbouring period; a term that tokenizes to
|
|
24
|
+
* nothing (`08`) drops the predicate entirely and matches every row) — a deployment that hits
|
|
25
|
+
* that gap before #5383 lands can opt in here.
|
|
26
|
+
*
|
|
27
|
+
* Per-word ANDing (see lib/search/containment) is a trade-off, not a strict improvement, over
|
|
28
|
+
* the single-literal ILIKE #4622 originally introduced: the token subquery matched a value
|
|
29
|
+
* carrying every token in any order with anything between them, so `?search=Warehouse 1757`
|
|
30
|
+
* must keep matching `Warehouse A 1757` — a single verbatim `ILIKE '%Warehouse 1757%'` would
|
|
31
|
+
* not, and TC-RESO-009 pins that as required behavior. The same word-order independence also
|
|
32
|
+
* widens multi-word document-number searches: `?search=ZK 1/2026` now also matches
|
|
33
|
+
* `ZK 11/2026` and `1/2026 ZK`, where the old single-literal ILIKE matched neither.
|
|
34
|
+
*
|
|
35
|
+
* Set `OM_SEARCH_USE_ILIKE_FOR_NON_ENCRYPTED_FIELDS=true` to opt into declared-column ILIKE
|
|
36
|
+
* ahead of #5383 — worth doing when the #5803 wrong-record symptom is hit in practice. Leaving it
|
|
37
|
+
* unset keeps the legacy rewrite-everything behavior, including the token index's prefix matching
|
|
38
|
+
* (`?search=ware` matching `Warehouse` when `enablePartials` is on, which literal containment
|
|
39
|
+
* gives only where the fragment really is a substring).
|
|
18
40
|
*/
|
|
19
41
|
useIlikeForNonEncryptedFields?: boolean
|
|
20
42
|
blocklistedFields: string[]
|
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Splits a `contains` like/ilike pattern into one pattern per whitespace-separated word, so a
|
|
3
|
+
* plaintext column taken off the hashed-token path keeps the word-order-independent matching the
|
|
4
|
+
* token index provided.
|
|
5
|
+
*
|
|
6
|
+
* The token path matches a value when it carries EVERY token of the term, in any order and with
|
|
7
|
+
* anything in between: `?search=Warehouse 1757` matches `Warehouse A 1757`. A single verbatim
|
|
8
|
+
* `name ILIKE '%Warehouse 1757%'` is literal substring containment, so the `A ` sitting between the
|
|
9
|
+
* two words defeats it — that is a capability every list grid has today, and #5803's fix must not
|
|
10
|
+
* take it away (`TC-RESO-009` pins it).
|
|
11
|
+
*
|
|
12
|
+
* ANDing one containment predicate per word reproduces the token semantics exactly on a column the
|
|
13
|
+
* engine can read, without the token path's two lossy steps: nothing is dropped for being shorter
|
|
14
|
+
* than `minTokenLength` (`?search=08` filters instead of matching every row) and nothing is split on
|
|
15
|
+
* non-alphanumerics (`?search=2026-08` no longer collapses onto `2026-01`).
|
|
16
|
+
*
|
|
17
|
+
* Splitting is deliberately narrow — a pattern is only split when it is unambiguously the
|
|
18
|
+
* `%term%` shape `buildIlikeTerm(value, 'contains')` produces:
|
|
19
|
+
*
|
|
20
|
+
* - it opens and closes with a wildcard `%` (a trailing `\%` is an escaped literal, not a wildcard);
|
|
21
|
+
* - the term between them carries no unescaped `%` or `_`, so a hand-built structured pattern such
|
|
22
|
+
* as `%a% b%` is left exactly as the caller wrote it;
|
|
23
|
+
* - the term holds at least two words.
|
|
24
|
+
*
|
|
25
|
+
* Anything else returns the input unchanged as a single pattern, which is the caller's existing
|
|
26
|
+
* behavior.
|
|
27
|
+
*
|
|
28
|
+
* The split is also capped at {@link MAX_CONTAINMENT_WORDS} words. Most list routes declare
|
|
29
|
+
* `search` as an unbounded string, this repository ships no trigram index for the resulting
|
|
30
|
+
* `ILIKE`, and the hybrid engine's `$or` groups multiply the split across every leaf — so an
|
|
31
|
+
* attacker-controlled term with thousands of words would otherwise compile into thousands of
|
|
32
|
+
* sequential-scan predicates from one request. A term at or under the cap keeps the per-word AND
|
|
33
|
+
* semantics; over the cap it falls back to the single verbatim pattern, which is bounded and was
|
|
34
|
+
* this helper's own behavior before the split existed.
|
|
35
|
+
*/
|
|
36
|
+
export const MAX_CONTAINMENT_WORDS = 10
|
|
37
|
+
|
|
38
|
+
export function buildContainmentPatterns(pattern: string): string[] {
|
|
39
|
+
if (!isWrappedContainsPattern(pattern)) return [pattern]
|
|
40
|
+
const term = pattern.slice(1, -1)
|
|
41
|
+
if (hasUnescapedWildcard(term)) return [pattern]
|
|
42
|
+
const words = term.split(/\s+/).filter((word) => word.length > 0)
|
|
43
|
+
if (words.length < 2 || words.length > MAX_CONTAINMENT_WORDS) return [pattern]
|
|
44
|
+
return words.map((word) => `%${word}%`)
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
function isWrappedContainsPattern(pattern: string): boolean {
|
|
48
|
+
if (pattern.length < 2) return false
|
|
49
|
+
if (!pattern.startsWith('%') || !pattern.endsWith('%')) return false
|
|
50
|
+
return countTrailingBackslashes(pattern, pattern.length - 2) % 2 === 0
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
function hasUnescapedWildcard(term: string): boolean {
|
|
54
|
+
for (let index = 0; index < term.length; index += 1) {
|
|
55
|
+
const char = term[index]
|
|
56
|
+
if (char !== '%' && char !== '_') continue
|
|
57
|
+
if (countTrailingBackslashes(term, index - 1) % 2 === 0) return true
|
|
58
|
+
}
|
|
59
|
+
return false
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
function countTrailingBackslashes(value: string, fromIndex: number): number {
|
|
63
|
+
let count = 0
|
|
64
|
+
for (let index = fromIndex; index >= 0 && value[index] === '\\'; index -= 1) count += 1
|
|
65
|
+
return count
|
|
66
|
+
}
|