@open-mercato/shared 0.8.1-develop.7275.1.f772b944d4 → 0.8.1-develop.7295.1.d0e0e33014

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (56) hide show
  1. package/.turbo/turbo-build.log +2 -1
  2. package/AGENTS.md +6 -3
  3. package/build.mjs +4 -1
  4. package/dist/lib/crud/errors.js +6 -1
  5. package/dist/lib/crud/errors.js.map +2 -2
  6. package/dist/lib/crud/factory.js +22 -13
  7. package/dist/lib/crud/factory.js.map +2 -2
  8. package/dist/lib/custom-fields/kinds.js +113 -0
  9. package/dist/lib/custom-fields/kinds.js.map +7 -0
  10. package/dist/lib/encryption/indexDoc.js +12 -9
  11. package/dist/lib/encryption/indexDoc.js.map +2 -2
  12. package/dist/lib/location/countries.generated.js +265 -0
  13. package/dist/lib/location/countries.generated.js.map +7 -0
  14. package/dist/lib/location/countries.js +2 -13
  15. package/dist/lib/location/countries.js.map +2 -2
  16. package/dist/lib/navigation/pageReload.js +11 -0
  17. package/dist/lib/navigation/pageReload.js.map +7 -0
  18. package/dist/lib/query/encrypted-sort.js +4 -1
  19. package/dist/lib/query/encrypted-sort.js.map +2 -2
  20. package/dist/lib/query/engine.js +97 -10
  21. package/dist/lib/query/engine.js.map +3 -3
  22. package/dist/lib/schedule/interval.js +33 -0
  23. package/dist/lib/schedule/interval.js.map +7 -0
  24. package/dist/lib/schedule/invalidScheduleValue.js +19 -0
  25. package/dist/lib/schedule/invalidScheduleValue.js.map +7 -0
  26. package/dist/lib/search/config.js.map +2 -2
  27. package/dist/lib/search/containment.js +32 -0
  28. package/dist/lib/search/containment.js.map +7 -0
  29. package/dist/lib/version.js +1 -1
  30. package/dist/lib/version.js.map +1 -1
  31. package/package.json +3 -2
  32. package/scripts/generate-countries.mjs +75 -0
  33. package/src/lib/crud/__tests__/crud-factory.test.ts +177 -0
  34. package/src/lib/crud/__tests__/errors.test.ts +31 -1
  35. package/src/lib/crud/errors.ts +16 -0
  36. package/src/lib/crud/factory.ts +37 -15
  37. package/src/lib/custom-fields/__tests__/kinds.test.ts +208 -0
  38. package/src/lib/custom-fields/kinds.ts +205 -0
  39. package/src/lib/encryption/__tests__/indexDoc.custom-field-kinds.test.ts +127 -0
  40. package/src/lib/encryption/__tests__/indexDoc.test.ts +39 -0
  41. package/src/lib/encryption/indexDoc.ts +16 -7
  42. package/src/lib/location/__tests__/countries.test.ts +25 -0
  43. package/src/lib/location/countries.generated.ts +264 -0
  44. package/src/lib/location/countries.ts +2 -23
  45. package/src/lib/navigation/__tests__/pageReload.test.ts +58 -0
  46. package/src/lib/navigation/pageReload.ts +43 -0
  47. package/src/lib/query/__tests__/engine.test.ts +208 -16
  48. package/src/lib/query/encrypted-sort.ts +8 -1
  49. package/src/lib/query/engine.ts +172 -19
  50. package/src/lib/schedule/__tests__/interval.test.ts +37 -0
  51. package/src/lib/schedule/interval.ts +45 -0
  52. package/src/lib/schedule/invalidScheduleValue.ts +33 -0
  53. package/src/lib/search/__tests__/config.test.ts +27 -0
  54. package/src/lib/search/__tests__/containment.test.ts +61 -0
  55. package/src/lib/search/config.ts +28 -6
  56. package/src/lib/search/containment.ts +66 -0
@@ -0,0 +1,45 @@
1
+ /**
2
+ * Canonical interval-format rules for recurring schedules.
3
+ *
4
+ * Both the scheduler runtime and the API validators that reject a schedule
5
+ * before it is persisted read the format from here, so the documented format
6
+ * (`<number><unit>`, e.g. `15m`, `1h`, `24h`) cannot drift between the layer
7
+ * that accepts a value and the layer that has to run it.
8
+ */
9
+ export const MIN_SCHEDULE_INTERVAL_MS = 60 * 1000
10
+
11
+ export const SCHEDULE_INTERVAL_PATTERN = /^(\d+)(s|m|h|d)$/
12
+
13
+ const UNIT_MULTIPLIERS: Record<string, number> = {
14
+ s: 1000,
15
+ m: 60 * 1000,
16
+ h: 60 * 60 * 1000,
17
+ d: 24 * 60 * 60 * 1000,
18
+ }
19
+
20
+ export type ScheduleIntervalUnit = 's' | 'm' | 'h' | 'd'
21
+
22
+ export type ScheduleIntervalParts = {
23
+ amount: number
24
+ unit: ScheduleIntervalUnit
25
+ }
26
+
27
+ export function matchScheduleInterval(interval: string): ScheduleIntervalParts | null {
28
+ const match = SCHEDULE_INTERVAL_PATTERN.exec(interval)
29
+ if (!match) return null
30
+ return { amount: Number.parseInt(match[1], 10), unit: match[2] as ScheduleIntervalUnit }
31
+ }
32
+
33
+ export function parseScheduleInterval(interval: string): number {
34
+ const parts = matchScheduleInterval(interval)
35
+ if (!parts) {
36
+ throw new Error(`Invalid interval format: ${interval}. Expected format: <number><unit> (e.g., 15m, 2h, 1d)`)
37
+ }
38
+ return parts.amount * UNIT_MULTIPLIERS[parts.unit]
39
+ }
40
+
41
+ export function isValidScheduleInterval(interval: string): boolean {
42
+ const parts = matchScheduleInterval(interval)
43
+ if (!parts) return false
44
+ return parts.amount * UNIT_MULTIPLIERS[parts.unit] >= MIN_SCHEDULE_INTERVAL_MS
45
+ }
@@ -0,0 +1,33 @@
1
+ export const INVALID_SCHEDULE_VALUE_CODE = 'INVALID_SCHEDULE_VALUE'
2
+
3
+ export type ScheduleValueKind = 'cron' | 'interval'
4
+
5
+ export type InvalidScheduleValueError = Error & {
6
+ code: typeof INVALID_SCHEDULE_VALUE_CODE
7
+ scheduleType: ScheduleValueKind
8
+ scheduleValue: string
9
+ }
10
+
11
+ export function createInvalidScheduleValueError(
12
+ scheduleType: ScheduleValueKind,
13
+ scheduleValue: string,
14
+ message: string,
15
+ ): InvalidScheduleValueError {
16
+ return Object.assign(new Error(message), {
17
+ code: INVALID_SCHEDULE_VALUE_CODE,
18
+ scheduleType,
19
+ scheduleValue,
20
+ } as const)
21
+ }
22
+
23
+ /**
24
+ * Structural check rather than `instanceof`: the error crosses a package
25
+ * boundary (and a production bundle), where a duplicated class identity would
26
+ * make `instanceof` silently false.
27
+ */
28
+ export function isInvalidScheduleValueError(error: unknown): error is InvalidScheduleValueError {
29
+ if (typeof error !== 'object' || error === null) return false
30
+ const candidate = error as { code?: unknown; scheduleType?: unknown }
31
+ return candidate.code === INVALID_SCHEDULE_VALUE_CODE
32
+ && (candidate.scheduleType === 'cron' || candidate.scheduleType === 'interval')
33
+ }
@@ -113,6 +113,33 @@ describe('OM_SEARCH_FIELD_BLOCKLIST parsing', () => {
113
113
  })
114
114
  })
115
115
 
116
+ describe('OM_SEARCH_USE_ILIKE_FOR_NON_ENCRYPTED_FIELDS', () => {
117
+ const originalValue = process.env.OM_SEARCH_USE_ILIKE_FOR_NON_ENCRYPTED_FIELDS
118
+
119
+ afterEach(() => {
120
+ if (originalValue === undefined) {
121
+ delete process.env.OM_SEARCH_USE_ILIKE_FOR_NON_ENCRYPTED_FIELDS
122
+ } else {
123
+ process.env.OM_SEARCH_USE_ILIKE_FOR_NON_ENCRYPTED_FIELDS = originalValue
124
+ }
125
+ })
126
+
127
+ // #5383: the switch stays off by default until tokenization is made ILIKE-equivalent, so the
128
+ // rewrite-everything behavior is unchanged for a deployment that does not opt in. #5803 is the
129
+ // correctness gap this switch closes when a deployment opts in ahead of that follow-up.
130
+ it('defaults to off so the legacy rewrite is unchanged for every column', () => {
131
+ delete process.env.OM_SEARCH_USE_ILIKE_FOR_NON_ENCRYPTED_FIELDS
132
+
133
+ expect(resolveSearchConfig().useIlikeForNonEncryptedFields).toBe(false)
134
+ })
135
+
136
+ it('can be switched on to apply a declared ilike on a plaintext column as written', () => {
137
+ process.env.OM_SEARCH_USE_ILIKE_FOR_NON_ENCRYPTED_FIELDS = 'true'
138
+
139
+ expect(resolveSearchConfig().useIlikeForNonEncryptedFields).toBe(true)
140
+ })
141
+ })
142
+
116
143
  describe('search token limits', () => {
117
144
  const variableNames = [
118
145
  'OM_SEARCH_MAX_FIELD_CHARS',
@@ -0,0 +1,61 @@
1
+ import { buildContainmentPatterns, MAX_CONTAINMENT_WORDS } from '../containment'
2
+
3
+ describe('buildContainmentPatterns (#5803)', () => {
4
+ test('splits a multi-word contains pattern into one pattern per word', () => {
5
+ // The token subquery matched every token in any order with anything between them. Reproducing
6
+ // that on SQL means ANDing per word, which is what TC-RESO-009 exercises through the API:
7
+ // `?search=Warehouse <stamp>` has to keep matching `Warehouse A <stamp>`.
8
+ expect(buildContainmentPatterns('%Warehouse 1757%')).toEqual(['%Warehouse%', '%1757%'])
9
+ })
10
+
11
+ test('collapses runs of whitespace rather than emitting empty patterns', () => {
12
+ expect(buildContainmentPatterns('%John Smith%')).toEqual(['%John%', '%Smith%'])
13
+ })
14
+
15
+ test('leaves a single-word pattern exactly as the caller declared it', () => {
16
+ // The reported #5803 case: the distinguishing fragment must reach SQL untouched, or the exact
17
+ // row cannot come back at all.
18
+ expect(buildContainmentPatterns('%2026-08%')).toEqual(['%2026-08%'])
19
+ })
20
+
21
+ test('leaves a term too short to tokenize alone', () => {
22
+ expect(buildContainmentPatterns('%08%')).toEqual(['%08%'])
23
+ })
24
+
25
+ test('keeps escaped wildcards attached to their word', () => {
26
+ // escapeLikePattern turns a literal `%` into `\%`; splitting must not treat it as a wildcard
27
+ // and must not tear the escape off its word.
28
+ expect(buildContainmentPatterns('%50\\% off%')).toEqual(['%50\\%%', '%off%'])
29
+ })
30
+
31
+ test('does not split a structured pattern carrying its own wildcards', () => {
32
+ // A caller that hand-built `%a% b%` asked for that exact shape; re-splitting it would silently
33
+ // rewrite a predicate this helper has no business reinterpreting.
34
+ expect(buildContainmentPatterns('%a% b%')).toEqual(['%a% b%'])
35
+ })
36
+
37
+ test('does not split an anchored pattern', () => {
38
+ // `startsWith` / `endsWith` terms are anchored on purpose; per-word ANDing would drop the
39
+ // anchor and widen the match.
40
+ expect(buildContainmentPatterns('John Smith%')).toEqual(['John Smith%'])
41
+ expect(buildContainmentPatterns('%John Smith')).toEqual(['%John Smith'])
42
+ })
43
+
44
+ test('treats a trailing escaped percent as a literal, not as the closing wildcard', () => {
45
+ expect(buildContainmentPatterns('%John Smith\\%')).toEqual(['%John Smith\\%'])
46
+ })
47
+
48
+ test('splits a term at the word cap', () => {
49
+ const words = Array.from({ length: MAX_CONTAINMENT_WORDS }, (_, index) => `w${index}`)
50
+ expect(buildContainmentPatterns(`%${words.join(' ')}%`)).toEqual(words.map((word) => `%${word}%`))
51
+ })
52
+
53
+ test('falls back to the single verbatim pattern above the word cap', () => {
54
+ // Most list routes declare `search` as an unbounded string and this repo ships no trigram
55
+ // index, so an unbounded per-word AND would let one request compile into an unbounded number
56
+ // of sequential-scan predicates. Past the cap, this returns to the pre-split behavior instead.
57
+ const words = Array.from({ length: MAX_CONTAINMENT_WORDS + 1 }, (_, index) => `w${index}`)
58
+ const pattern = `%${words.join(' ')}%`
59
+ expect(buildContainmentPatterns(pattern)).toEqual([pattern])
60
+ })
61
+ })
@@ -9,12 +9,34 @@ export type SearchConfig = {
9
9
  hashAlgorithm: 'sha256' | 'sha1' | 'md5'
10
10
  storeRawTokens: boolean
11
11
  /**
12
- * When true, a like/ilike on a PLAINTEXT base column runs as exact SQL ILIKE instead of being
13
- * rewritten into an approximate search-token match; encrypted columns always keep the token
14
- * path (ILIKE against ciphertext cannot match). Off by default: token matching can be faster
15
- * than an unanchored ILIKE, which may need a full scan without a trigram index — but it is
16
- * approximate (fragments under minTokenLength vanish, so `ZK 1/2026` degrades to its year and
17
- * an all-short term drops the predicate). Flip it on when list search must be exact.
12
+ * When true, a like/ilike on a PLAINTEXT base column runs as SQL ILIKE — one containment
13
+ * predicate per word of the term, ANDed — instead of being rewritten into an approximate
14
+ * search-token match; encrypted columns always keep the token path (ILIKE against ciphertext
15
+ * cannot match).
16
+ *
17
+ * Off by default, per #5383: the token store is expected to become faster than ILIKE once
18
+ * tokenization is made semantically equivalent to it, so the plan there is to keep this switch
19
+ * off until that follow-up lands rather than trade performance for correctness by default. #5803
20
+ * documents the correctness gap this switch closes when enabled: the token rewrite is lossy in a
21
+ * way that silently returns the WRONG record rather than merely extra ones (tokenization splits
22
+ * on non-alphanumerics and drops fragments under minTokenLength, so `2026-08` and `2026-01` both
23
+ * reduce to {202, 2026} and a picker offers the neighbouring period; a term that tokenizes to
24
+ * nothing (`08`) drops the predicate entirely and matches every row) — a deployment that hits
25
+ * that gap before #5383 lands can opt in here.
26
+ *
27
+ * Per-word ANDing (see lib/search/containment) is a trade-off, not a strict improvement, over
28
+ * the single-literal ILIKE #4622 originally introduced: the token subquery matched a value
29
+ * carrying every token in any order with anything between them, so `?search=Warehouse 1757`
30
+ * must keep matching `Warehouse A 1757` — a single verbatim `ILIKE '%Warehouse 1757%'` would
31
+ * not, and TC-RESO-009 pins that as required behavior. The same word-order independence also
32
+ * widens multi-word document-number searches: `?search=ZK 1/2026` now also matches
33
+ * `ZK 11/2026` and `1/2026 ZK`, where the old single-literal ILIKE matched neither.
34
+ *
35
+ * Set `OM_SEARCH_USE_ILIKE_FOR_NON_ENCRYPTED_FIELDS=true` to opt into declared-column ILIKE
36
+ * ahead of #5383 — worth doing when the #5803 wrong-record symptom is hit in practice. Leaving it
37
+ * unset keeps the legacy rewrite-everything behavior, including the token index's prefix matching
38
+ * (`?search=ware` matching `Warehouse` when `enablePartials` is on, which literal containment
39
+ * gives only where the fragment really is a substring).
18
40
  */
19
41
  useIlikeForNonEncryptedFields?: boolean
20
42
  blocklistedFields: string[]
@@ -0,0 +1,66 @@
1
+ /**
2
+ * Splits a `contains` like/ilike pattern into one pattern per whitespace-separated word, so a
3
+ * plaintext column taken off the hashed-token path keeps the word-order-independent matching the
4
+ * token index provided.
5
+ *
6
+ * The token path matches a value when it carries EVERY token of the term, in any order and with
7
+ * anything in between: `?search=Warehouse 1757` matches `Warehouse A 1757`. A single verbatim
8
+ * `name ILIKE '%Warehouse 1757%'` is literal substring containment, so the `A ` sitting between the
9
+ * two words defeats it — that is a capability every list grid has today, and #5803's fix must not
10
+ * take it away (`TC-RESO-009` pins it).
11
+ *
12
+ * ANDing one containment predicate per word reproduces the token semantics exactly on a column the
13
+ * engine can read, without the token path's two lossy steps: nothing is dropped for being shorter
14
+ * than `minTokenLength` (`?search=08` filters instead of matching every row) and nothing is split on
15
+ * non-alphanumerics (`?search=2026-08` no longer collapses onto `2026-01`).
16
+ *
17
+ * Splitting is deliberately narrow — a pattern is only split when it is unambiguously the
18
+ * `%term%` shape `buildIlikeTerm(value, 'contains')` produces:
19
+ *
20
+ * - it opens and closes with a wildcard `%` (a trailing `\%` is an escaped literal, not a wildcard);
21
+ * - the term between them carries no unescaped `%` or `_`, so a hand-built structured pattern such
22
+ * as `%a% b%` is left exactly as the caller wrote it;
23
+ * - the term holds at least two words.
24
+ *
25
+ * Anything else returns the input unchanged as a single pattern, which is the caller's existing
26
+ * behavior.
27
+ *
28
+ * The split is also capped at {@link MAX_CONTAINMENT_WORDS} words. Most list routes declare
29
+ * `search` as an unbounded string, this repository ships no trigram index for the resulting
30
+ * `ILIKE`, and the hybrid engine's `$or` groups multiply the split across every leaf — so an
31
+ * attacker-controlled term with thousands of words would otherwise compile into thousands of
32
+ * sequential-scan predicates from one request. A term at or under the cap keeps the per-word AND
33
+ * semantics; over the cap it falls back to the single verbatim pattern, which is bounded and was
34
+ * this helper's own behavior before the split existed.
35
+ */
36
+ export const MAX_CONTAINMENT_WORDS = 10
37
+
38
+ export function buildContainmentPatterns(pattern: string): string[] {
39
+ if (!isWrappedContainsPattern(pattern)) return [pattern]
40
+ const term = pattern.slice(1, -1)
41
+ if (hasUnescapedWildcard(term)) return [pattern]
42
+ const words = term.split(/\s+/).filter((word) => word.length > 0)
43
+ if (words.length < 2 || words.length > MAX_CONTAINMENT_WORDS) return [pattern]
44
+ return words.map((word) => `%${word}%`)
45
+ }
46
+
47
+ function isWrappedContainsPattern(pattern: string): boolean {
48
+ if (pattern.length < 2) return false
49
+ if (!pattern.startsWith('%') || !pattern.endsWith('%')) return false
50
+ return countTrailingBackslashes(pattern, pattern.length - 2) % 2 === 0
51
+ }
52
+
53
+ function hasUnescapedWildcard(term: string): boolean {
54
+ for (let index = 0; index < term.length; index += 1) {
55
+ const char = term[index]
56
+ if (char !== '%' && char !== '_') continue
57
+ if (countTrailingBackslashes(term, index - 1) % 2 === 0) return true
58
+ }
59
+ return false
60
+ }
61
+
62
+ function countTrailingBackslashes(value: string, fromIndex: number): number {
63
+ let count = 0
64
+ for (let index = fromIndex; index >= 0 && value[index] === '\\'; index -= 1) count += 1
65
+ return count
66
+ }