@open-mercato/shared 0.6.7 → 0.6.8-develop.6874.1.982d6097d8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (195) hide show
  1. package/.turbo/turbo-build.log +1 -1
  2. package/AGENTS.md +57 -6
  3. package/build.mjs +2 -4
  4. package/dist/lib/ai/safety-identifier.js +32 -0
  5. package/dist/lib/ai/safety-identifier.js.map +7 -0
  6. package/dist/lib/auth/organizationScope.js +34 -0
  7. package/dist/lib/auth/organizationScope.js.map +7 -0
  8. package/dist/lib/auth/server.js +15 -2
  9. package/dist/lib/auth/server.js.map +2 -2
  10. package/dist/lib/bootstrap/clientOnlyModules.js +55 -0
  11. package/dist/lib/bootstrap/clientOnlyModules.js.map +7 -0
  12. package/dist/lib/bootstrap/dynamicLoader.js +117 -43
  13. package/dist/lib/bootstrap/dynamicLoader.js.map +2 -2
  14. package/dist/lib/commands/command-bus.js +5 -1
  15. package/dist/lib/commands/command-bus.js.map +2 -2
  16. package/dist/lib/commands/command-interceptor-runner.js +2 -2
  17. package/dist/lib/commands/command-interceptor-runner.js.map +2 -2
  18. package/dist/lib/crud/enricher-runner.js +2 -11
  19. package/dist/lib/crud/enricher-runner.js.map +2 -2
  20. package/dist/lib/crud/factory.js +72 -9
  21. package/dist/lib/crud/factory.js.map +2 -2
  22. package/dist/lib/crud/interceptor-runner.js +2 -2
  23. package/dist/lib/crud/interceptor-runner.js.map +2 -2
  24. package/dist/lib/crud/mutation-guard-registry.js +2 -2
  25. package/dist/lib/crud/mutation-guard-registry.js.map +2 -2
  26. package/dist/lib/crud/types.js +4 -0
  27. package/dist/lib/crud/types.js.map +3 -3
  28. package/dist/lib/data/consistency.js +19 -0
  29. package/dist/lib/data/consistency.js.map +7 -0
  30. package/dist/lib/data/engine.js +37 -11
  31. package/dist/lib/data/engine.js.map +2 -2
  32. package/dist/lib/db/entityMetadata.js +26 -0
  33. package/dist/lib/db/entityMetadata.js.map +7 -0
  34. package/dist/lib/db/pg-errors.js +42 -0
  35. package/dist/lib/db/pg-errors.js.map +2 -2
  36. package/dist/lib/di/container.js +48 -6
  37. package/dist/lib/di/container.js.map +2 -2
  38. package/dist/lib/email/send.js +56 -0
  39. package/dist/lib/email/send.js.map +2 -2
  40. package/dist/lib/encryption/subscriber.js +2 -10
  41. package/dist/lib/encryption/subscriber.js.map +2 -2
  42. package/dist/lib/encryption/tenantDataEncryptionService.js +52 -17
  43. package/dist/lib/encryption/tenantDataEncryptionService.js.map +2 -2
  44. package/dist/lib/i18n/config.js +1 -1
  45. package/dist/lib/i18n/config.js.map +2 -2
  46. package/dist/lib/i18n/server.js +4 -4
  47. package/dist/lib/i18n/server.js.map +3 -3
  48. package/dist/lib/logger/extension.js +29 -0
  49. package/dist/lib/logger/extension.js.map +7 -0
  50. package/dist/lib/logger/index.js +47 -3
  51. package/dist/lib/logger/index.js.map +3 -3
  52. package/dist/lib/modules/registry.js +5 -1
  53. package/dist/lib/modules/registry.js.map +2 -2
  54. package/dist/lib/modules/surfaceFingerprint.js +47 -0
  55. package/dist/lib/modules/surfaceFingerprint.js.map +7 -0
  56. package/dist/lib/query/ciphertext-search-warning.js +45 -0
  57. package/dist/lib/query/ciphertext-search-warning.js.map +7 -0
  58. package/dist/lib/query/engine.js +54 -40
  59. package/dist/lib/query/engine.js.map +2 -2
  60. package/dist/lib/query/types.js.map +1 -1
  61. package/dist/lib/redis/connection.js +7 -2
  62. package/dist/lib/redis/connection.js.map +2 -2
  63. package/dist/lib/search/auto-indexing.js +14 -0
  64. package/dist/lib/search/auto-indexing.js.map +7 -0
  65. package/dist/lib/search/availability.js +107 -0
  66. package/dist/lib/search/availability.js.map +7 -0
  67. package/dist/lib/search/config.js +61 -2
  68. package/dist/lib/search/config.js.map +2 -2
  69. package/dist/lib/search/tokenLookup.js +46 -0
  70. package/dist/lib/search/tokenLookup.js.map +7 -0
  71. package/dist/lib/search/tokenize.js +23 -11
  72. package/dist/lib/search/tokenize.js.map +2 -2
  73. package/dist/lib/telemetry/runtime.js +35 -0
  74. package/dist/lib/telemetry/runtime.js.map +7 -0
  75. package/dist/lib/version.js +1 -1
  76. package/dist/lib/version.js.map +1 -1
  77. package/dist/lib/webhooks/body.js +57 -0
  78. package/dist/lib/webhooks/body.js.map +7 -0
  79. package/dist/lib/webhooks/inbound-types.js +1 -0
  80. package/dist/lib/webhooks/inbound-types.js.map +7 -0
  81. package/dist/lib/webhooks/index.js +13 -1
  82. package/dist/lib/webhooks/index.js.map +2 -2
  83. package/dist/lib/webhooks/verify.js +12 -7
  84. package/dist/lib/webhooks/verify.js.map +2 -2
  85. package/dist/modules/analytics.js.map +2 -2
  86. package/dist/modules/overrides.js +50 -1
  87. package/dist/modules/overrides.js.map +2 -2
  88. package/dist/modules/payment_gateways/types.js +8 -1
  89. package/dist/modules/payment_gateways/types.js.map +2 -2
  90. package/dist/modules/registry.js +1 -0
  91. package/dist/modules/registry.js.map +2 -2
  92. package/dist/modules/widgets/extension-points.js +96 -0
  93. package/dist/modules/widgets/extension-points.js.map +7 -0
  94. package/dist/security/enabledModulesRegistry.js +56 -11
  95. package/dist/security/enabledModulesRegistry.js.map +2 -2
  96. package/dist/security/featurePolicy.js +62 -0
  97. package/dist/security/featurePolicy.js.map +7 -0
  98. package/jest.config.cjs +2 -1
  99. package/package.json +19 -9
  100. package/scripts/versionSource.cjs +65 -0
  101. package/src/lib/__tests__/versionSource.test.ts +46 -0
  102. package/src/lib/ai/__tests__/llm-provider-contract.test.ts +59 -0
  103. package/src/lib/ai/__tests__/safety-identifier.test.ts +71 -0
  104. package/src/lib/ai/llm-provider.ts +33 -0
  105. package/src/lib/ai/safety-identifier.ts +73 -0
  106. package/src/lib/auth/__tests__/organizationScope.test.ts +107 -0
  107. package/src/lib/auth/__tests__/server.test.ts +23 -0
  108. package/src/lib/auth/organizationScope.ts +65 -0
  109. package/src/lib/auth/server.ts +23 -2
  110. package/src/lib/bootstrap/__tests__/clientOnlyModules.test.ts +189 -0
  111. package/src/lib/bootstrap/__tests__/dynamicLoader.aliasResolver.test.ts +98 -0
  112. package/src/lib/bootstrap/__tests__/dynamicLoader.appDi.test.ts +211 -0
  113. package/src/lib/bootstrap/__tests__/dynamicLoader.commandInterceptors.test.ts +63 -8
  114. package/src/lib/bootstrap/__tests__/dynamicLoader.generatedCacheRecovery.test.ts +256 -0
  115. package/src/lib/bootstrap/__tests__/dynamicLoader.tsconfig.test.ts +21 -1
  116. package/src/lib/bootstrap/clientOnlyModules.ts +85 -0
  117. package/src/lib/bootstrap/dynamicLoader.ts +213 -65
  118. package/src/lib/commands/__tests__/command-interceptor-runner.test.ts +28 -0
  119. package/src/lib/commands/command-bus.ts +5 -1
  120. package/src/lib/commands/command-interceptor-runner.ts +2 -2
  121. package/src/lib/crud/__tests__/crud-factory.cache-user-scope.test.ts +171 -0
  122. package/src/lib/crud/__tests__/crud-factory.test.ts +351 -0
  123. package/src/lib/crud/__tests__/mutation-guard-registry.test.ts +26 -0
  124. package/src/lib/crud/enricher-runner.ts +2 -11
  125. package/src/lib/crud/factory.ts +82 -6
  126. package/src/lib/crud/interceptor-runner.ts +2 -2
  127. package/src/lib/crud/mutation-guard-registry.ts +2 -2
  128. package/src/lib/crud/types.ts +3 -0
  129. package/src/lib/data/__tests__/consistency.test.ts +38 -0
  130. package/src/lib/data/__tests__/engine.bulk-suppress.test.ts +18 -3
  131. package/src/lib/data/consistency.ts +17 -0
  132. package/src/lib/data/engine.ts +50 -16
  133. package/src/lib/db/__tests__/entityMetadata.test.ts +38 -0
  134. package/src/lib/db/__tests__/pg-errors.test.ts +48 -0
  135. package/src/lib/db/entityMetadata.ts +43 -0
  136. package/src/lib/db/pg-errors.ts +58 -0
  137. package/src/lib/di/__tests__/container-app-di-absent.test.ts +155 -0
  138. package/src/lib/di/__tests__/container-app-di.test.ts +173 -0
  139. package/src/lib/di/__tests__/registrar-error-log.test.ts +154 -0
  140. package/src/lib/di/container.ts +72 -8
  141. package/src/lib/email/__tests__/send.test.ts +37 -1
  142. package/src/lib/email/send.ts +79 -0
  143. package/src/lib/encryption/__tests__/subscriber.test.ts +54 -2
  144. package/src/lib/encryption/__tests__/tenantDataEncryptionService.test.ts +58 -0
  145. package/src/lib/encryption/subscriber.ts +2 -14
  146. package/src/lib/encryption/tenantDataEncryptionService.ts +64 -18
  147. package/src/lib/i18n/__tests__/config.test.ts +8 -0
  148. package/src/lib/i18n/__tests__/server-dictionary-cache.test.ts +30 -0
  149. package/src/lib/i18n/__tests__/server-unbootstrapped.test.ts +29 -0
  150. package/src/lib/i18n/config.ts +2 -2
  151. package/src/lib/i18n/server.ts +6 -2
  152. package/src/lib/logger/__tests__/logger.test.ts +77 -0
  153. package/src/lib/logger/extension.ts +59 -0
  154. package/src/lib/logger/index.ts +59 -1
  155. package/src/lib/modules/__tests__/surfaceFingerprint.test.ts +122 -0
  156. package/src/lib/modules/registry.ts +10 -0
  157. package/src/lib/modules/surfaceFingerprint.ts +87 -0
  158. package/src/lib/query/__tests__/ciphertext-search-warning.test.ts +178 -0
  159. package/src/lib/query/__tests__/engine.test.ts +65 -13
  160. package/src/lib/query/ciphertext-search-warning.ts +95 -0
  161. package/src/lib/query/engine.ts +75 -58
  162. package/src/lib/query/types.ts +4 -5
  163. package/src/lib/redis/__tests__/connection.test.ts +16 -0
  164. package/src/lib/redis/connection.ts +10 -7
  165. package/src/lib/search/__tests__/availability.test.ts +211 -0
  166. package/src/lib/search/__tests__/config.test.ts +177 -0
  167. package/src/lib/search/__tests__/tokenLookup.test.ts +206 -0
  168. package/src/lib/search/__tests__/tokenize.test.ts +49 -0
  169. package/src/lib/search/auto-indexing.ts +22 -0
  170. package/src/lib/search/availability.ts +232 -0
  171. package/src/lib/search/config.ts +106 -8
  172. package/src/lib/search/tokenLookup.ts +133 -0
  173. package/src/lib/search/tokenize.ts +33 -11
  174. package/src/lib/telemetry/runtime.ts +72 -0
  175. package/src/lib/webhooks/__tests__/body.test.ts +95 -0
  176. package/src/lib/webhooks/__tests__/verify.test.ts +20 -1
  177. package/src/lib/webhooks/body.ts +70 -0
  178. package/src/lib/webhooks/inbound-types.ts +125 -0
  179. package/src/lib/webhooks/index.ts +19 -1
  180. package/src/lib/webhooks/verify.ts +17 -9
  181. package/src/modules/__tests__/nav-group-order-override.test.ts +183 -0
  182. package/src/modules/__tests__/registry.test.ts +10 -2
  183. package/src/modules/analytics.ts +7 -0
  184. package/src/modules/customer-auth.ts +1 -0
  185. package/src/modules/encryption.ts +3 -0
  186. package/src/modules/navigation/backendChrome.ts +15 -0
  187. package/src/modules/overrides.ts +103 -0
  188. package/src/modules/payment_gateways/__tests__/types.test.ts +37 -0
  189. package/src/modules/payment_gateways/types.ts +41 -0
  190. package/src/modules/registry.ts +3 -0
  191. package/src/modules/widgets/__tests__/extension-points.test.ts +101 -0
  192. package/src/modules/widgets/extension-points.ts +524 -0
  193. package/src/security/__tests__/featurePolicy.test.ts +166 -0
  194. package/src/security/enabledModulesRegistry.ts +64 -12
  195. package/src/security/featurePolicy.ts +89 -0
@@ -0,0 +1,232 @@
1
+ import { sql } from 'kysely'
2
+ import { parseNumberWithDefault } from '@open-mercato/shared/lib/number'
3
+
4
+ export type OrganizationScope = { ids: string[]; includeNull: boolean }
5
+
6
+ export type SearchTokenSourceRef = { entity: string; recordIdColumn?: string }
7
+
8
+ type ProbeExpression = object | string | readonly string[] | null
9
+
10
+ export type SearchTokenProbeQueryBuilder = {
11
+ select: (selection: ProbeExpression) => SearchTokenProbeQueryBuilder
12
+ where: (column: ProbeExpression, operator?: string, value?: ProbeExpression) => SearchTokenProbeQueryBuilder
13
+ limit: (count: number) => SearchTokenProbeQueryBuilder
14
+ executeTakeFirst: () => Promise<object | undefined>
15
+ }
16
+
17
+ export type SearchTokenProbeDb = { selectFrom: (table: string) => SearchTokenProbeQueryBuilder }
18
+
19
+ export type SearchTokenAvailabilityDebugPayload = {
20
+ entity: string
21
+ tenantId: string | null
22
+ organizationScope?: OrganizationScope | null
23
+ recordIdColumn?: string
24
+ hasTokens?: boolean
25
+ error?: string
26
+ }
27
+
28
+ export type SearchTokenAvailabilityDeps = {
29
+ getDb: () => SearchTokenProbeDb
30
+ getConfig: () => { enabled: boolean }
31
+ applyOrganizationScope: (
32
+ query: SearchTokenProbeQueryBuilder,
33
+ column: string,
34
+ scope: OrganizationScope,
35
+ ) => SearchTokenProbeQueryBuilder
36
+ logDebug: (event: string, payload: SearchTokenAvailabilityDebugPayload) => void
37
+ }
38
+
39
+ export type SearchTokenAvailability = {
40
+ /**
41
+ * The cheap, statically-known half of the decision: search is configured on
42
+ * AND the `search_tokens` table exists. Safe to resolve eagerly (consumers
43
+ * like custom-field source attachment need it before any filter is
44
+ * inspected); the table probe is memoized per instance.
45
+ */
46
+ staticEnabled: () => Promise<boolean>
47
+ /**
48
+ * The expensive half: does `search_tokens` hold any row for this
49
+ * (entity, tenant, organization scope)? Historically the `LIMIT 1` probe
50
+ * could degenerate into a seq scan on a large table (#4723), so callers MUST
51
+ * only ask when the query actually carries a like/ilike filter — use
52
+ * `hasSearchFilter` for the gate. Three layers keep it cheap for the end
53
+ * user: index-usable predicates (no `IS NOT DISTINCT FROM`) served by the
54
+ * dedicated `search_tokens_presence_idx (entity_type, tenant_id,
55
+ * organization_id)` prefix — which makes the MISS as cheap as the hit — a
56
+ * per-request instance memo, and a process-level TTL cache
57
+ * (`OM_SEARCH_TOKEN_PRESENCE_CACHE_MS`, default 30s) that amortizes the
58
+ * probe across requests. Probe errors log `search:has-tokens-error`,
59
+ * resolve to `false`, and are never TTL-cached.
60
+ */
61
+ hasTokens: (entity: string, tenantId: string | null, orgScope?: OrganizationScope | null) => Promise<boolean>
62
+ /** First-hit sweep over token sources; logs `search:source-has-tokens` per probed source. */
63
+ anySourceHasTokens: (
64
+ sources: SearchTokenSourceRef[],
65
+ tenantId: string | null,
66
+ orgScope?: OrganizationScope | null,
67
+ ) => Promise<boolean>
68
+ }
69
+
70
+ export function isSearchFilterOp(op: string | null | undefined): boolean {
71
+ return op === 'like' || op === 'ilike'
72
+ }
73
+
74
+ /**
75
+ * The single definition of "this query actually searches". Every consumer of
76
+ * the token-availability answer sits behind a like/ilike guard, so when this
77
+ * returns `false` the `hasTokens` probe's answer would never be read — gate
78
+ * the probe on it.
79
+ */
80
+ export function hasSearchFilter(filters: ReadonlyArray<{ op?: string | null }>): boolean {
81
+ return filters.some((filter) => isSearchFilterOp(filter.op))
82
+ }
83
+
84
+ function orgScopeKey(scope: OrganizationScope | null | undefined): string {
85
+ if (!scope) return 'none'
86
+ return `${scope.includeNull ? '1' : '0'}:${[...scope.ids].sort((left, right) => left.localeCompare(right)).join(',')}`
87
+ }
88
+
89
+ const PRESENCE_CACHE_DEFAULT_TTL_MS = 30_000
90
+ const PRESENCE_CACHE_MAX_ENTRIES = 10_000
91
+
92
+ // Process-level TTL cache for the token-presence answer. Module-level (process-global) on
93
+ // purpose: `createRequestContainer` builds fresh engines — and with them fresh resolver
94
+ // instances — per request, so an instance-scoped memo alone re-pays the probe on every
95
+ // request. On a large `search_tokens` one probe can be pathologically expensive (#4723),
96
+ // so the answer is amortized across requests here and only re-checked once per TTL.
97
+ // Staleness contract: a stale `false` keeps like/ilike on the plain-column fallback for up
98
+ // to the TTL after an entity's first tokens are written; a stale `true` routes search
99
+ // through an emptied token set for up to the TTL after a purge. Both converge within the
100
+ // TTL; set OM_SEARCH_TOKEN_PRESENCE_CACHE_MS=0 to disable and probe per request again.
101
+ const presenceCache = new Map<string, { value: boolean; expiresAt: number }>()
102
+
103
+ function resolvePresenceCacheTtlMs(): number {
104
+ return parseNumberWithDefault(process.env.OM_SEARCH_TOKEN_PRESENCE_CACHE_MS, PRESENCE_CACHE_DEFAULT_TTL_MS, { integer: true, min: 0 })
105
+ }
106
+
107
+ function storePresence(key: string, value: boolean, ttlMs: number): void {
108
+ if (presenceCache.size >= PRESENCE_CACHE_MAX_ENTRIES) {
109
+ const now = Date.now()
110
+ for (const [entryKey, entry] of presenceCache) {
111
+ if (entry.expiresAt <= now) presenceCache.delete(entryKey)
112
+ }
113
+ if (presenceCache.size >= PRESENCE_CACHE_MAX_ENTRIES) presenceCache.clear()
114
+ }
115
+ presenceCache.set(key, { value, expiresAt: Date.now() + ttlMs })
116
+ }
117
+
118
+ export function clearSearchTokenPresenceCache(): void {
119
+ presenceCache.clear()
120
+ }
121
+
122
+ /**
123
+ * One place that answers "is token search usable here?" for both query
124
+ * engines, instead of each hand-assembling config + table-existence +
125
+ * token-presence probes with private duplicate helpers and ad-hoc memo maps.
126
+ *
127
+ * Memoization is per instance; engines are constructed per request
128
+ * (`createRequestContainer`), so entries never outlive a request — the same
129
+ * staleness contract the engines' previous per-query join maps had. A
130
+ * rejected table probe is evicted so the next call retries instead of
131
+ * observing a poisoned cache entry.
132
+ */
133
+ export function createSearchTokenAvailability(deps: SearchTokenAvailabilityDeps): SearchTokenAvailability {
134
+ const tablePresence = new Map<string, Promise<boolean>>()
135
+ const tokenPresence = new Map<string, Promise<boolean>>()
136
+
137
+ const tableExists = (table: string): Promise<boolean> => {
138
+ const cached = tablePresence.get(table)
139
+ if (cached) return cached
140
+ const probe = (async () => {
141
+ const row = await deps.getDb()
142
+ .selectFrom('information_schema.tables')
143
+ .select(sql<number>`1`.as('one'))
144
+ .where('table_name', '=', table)
145
+ .limit(1)
146
+ .executeTakeFirst()
147
+ return !!row
148
+ })()
149
+ tablePresence.set(table, probe)
150
+ probe.catch(() => tablePresence.delete(table))
151
+ return probe
152
+ }
153
+
154
+ const probeTokens = async (
155
+ entity: string,
156
+ tenantId: string | null,
157
+ orgScope?: OrganizationScope | null,
158
+ ): Promise<boolean> => {
159
+ let query = deps.getDb()
160
+ .selectFrom('search_tokens')
161
+ .select(sql<number>`1`.as('one'))
162
+ .where('entity_type', '=', entity)
163
+ // Deliberately `= / IS NULL` instead of `IS NOT DISTINCT FROM` (identical semantics
164
+ // for a string|null tenant): the latter cannot serve as an index condition, which is
165
+ // part of why the planner degraded this probe to a seq scan on large tables (#4723).
166
+ // With plain predicates the probe is a pure prefix seek on
167
+ // `search_tokens_presence_idx (entity_type, tenant_id, organization_id)`, making the
168
+ // miss as cheap as the hit.
169
+ query = tenantId == null
170
+ ? query.where('tenant_id', 'is', null)
171
+ : query.where('tenant_id', '=', tenantId)
172
+ if (orgScope) {
173
+ query = deps.applyOrganizationScope(query, 'search_tokens.organization_id', orgScope)
174
+ }
175
+ const row = await query.limit(1).executeTakeFirst()
176
+ return !!row
177
+ }
178
+
179
+ const hasTokens = (
180
+ entity: string,
181
+ tenantId: string | null,
182
+ orgScope?: OrganizationScope | null,
183
+ ): Promise<boolean> => {
184
+ const key = `${entity}|${tenantId ?? '__null__'}|${orgScopeKey(orgScope)}`
185
+ const ttlMs = resolvePresenceCacheTtlMs()
186
+ if (ttlMs > 0) {
187
+ const entry = presenceCache.get(key)
188
+ if (entry && entry.expiresAt > Date.now()) return Promise.resolve(entry.value)
189
+ }
190
+ const cached = tokenPresence.get(key)
191
+ if (cached) return cached
192
+ const probe = (async () => {
193
+ try {
194
+ const value = await probeTokens(entity, tenantId, orgScope)
195
+ // Only genuine probe results enter the process-level cache — caching an
196
+ // error-driven `false` would pin degraded search for a full TTL after a
197
+ // transient DB failure.
198
+ if (ttlMs > 0) storePresence(key, value, ttlMs)
199
+ return value
200
+ } catch (err) {
201
+ deps.logDebug('search:has-tokens-error', {
202
+ entity,
203
+ tenantId,
204
+ organizationScope: orgScope,
205
+ error: err instanceof Error ? err.message : String(err),
206
+ })
207
+ return false
208
+ }
209
+ })()
210
+ tokenPresence.set(key, probe)
211
+ return probe
212
+ }
213
+
214
+ return {
215
+ staticEnabled: async () => deps.getConfig().enabled && await tableExists('search_tokens'),
216
+ hasTokens,
217
+ anySourceHasTokens: async (sources, tenantId, orgScope) => {
218
+ for (const source of sources) {
219
+ const ok = await hasTokens(source.entity, tenantId, orgScope)
220
+ deps.logDebug('search:source-has-tokens', {
221
+ entity: source.entity,
222
+ recordIdColumn: source.recordIdColumn,
223
+ tenantId,
224
+ organizationScope: orgScope,
225
+ hasTokens: ok,
226
+ })
227
+ if (ok) return true
228
+ }
229
+ return false
230
+ },
231
+ }
232
+ }
@@ -1,5 +1,6 @@
1
1
  import { parseBooleanWithDefault } from '@open-mercato/shared/lib/boolean'
2
2
  import { parseNumberWithDefault } from '@open-mercato/shared/lib/number'
3
+ import { parseCommaSeparatedList } from '@open-mercato/shared/lib/string'
3
4
 
4
5
  export type SearchConfig = {
5
6
  enabled: boolean
@@ -8,12 +9,27 @@ export type SearchConfig = {
8
9
  hashAlgorithm: 'sha256' | 'sha1' | 'md5'
9
10
  storeRawTokens: boolean
10
11
  blocklistedFields: string[]
12
+ entityBlocklistedFields?: Record<string, string[]>
13
+ maxFieldChars?: number
14
+ maxTokensPerField?: number
15
+ maxTokensPerRecord?: number
11
16
  }
12
17
 
13
18
  export const DEFAULT_SEARCH_MIN_TOKEN_LENGTH = 3
19
+ export const DEFAULT_SEARCH_MAX_FIELD_CHARS = 20_000
20
+ export const DEFAULT_SEARCH_MAX_TOKENS_PER_FIELD = 5_000
21
+ export const DEFAULT_SEARCH_MAX_TOKENS_PER_RECORD = 20_000
22
+
23
+ export type SearchTokenLimits = {
24
+ maxFieldChars: number
25
+ maxTokensPerField: number
26
+ maxTokensPerRecord: number
27
+ }
14
28
 
15
29
  const DEFAULT_BLOCKLIST = ['password', 'token', 'secret', 'hash']
16
30
 
31
+ const ENTITY_BLOCKLIST_SEPARATOR = '@'
32
+
17
33
  function parseBoolean(raw: string | undefined, fallback: boolean): boolean {
18
34
  return parseBooleanWithDefault(raw, fallback)
19
35
  }
@@ -22,6 +38,19 @@ function parseNumber(raw: string | undefined, fallback: number, min = 1): number
22
38
  return parseNumberWithDefault(raw, fallback, { integer: true, min })
23
39
  }
24
40
 
41
+ export function resolveSearchTokenLimits(config: SearchConfig): SearchTokenLimits {
42
+ const resolveLimit = (value: number | undefined, fallback: number): number => {
43
+ if (value === undefined) return fallback
44
+ if (!Number.isFinite(value) || value < 0) return fallback
45
+ return Math.trunc(value)
46
+ }
47
+ return {
48
+ maxFieldChars: resolveLimit(config.maxFieldChars, DEFAULT_SEARCH_MAX_FIELD_CHARS),
49
+ maxTokensPerField: resolveLimit(config.maxTokensPerField, DEFAULT_SEARCH_MAX_TOKENS_PER_FIELD),
50
+ maxTokensPerRecord: resolveLimit(config.maxTokensPerRecord, DEFAULT_SEARCH_MAX_TOKENS_PER_RECORD),
51
+ }
52
+ }
53
+
25
54
  function parseHashAlgorithm(raw: string | undefined): 'sha256' | 'sha1' | 'md5' {
26
55
  const value = (raw ?? '').trim().toLowerCase()
27
56
  if (value === 'sha1') return 'sha1'
@@ -29,24 +58,93 @@ function parseHashAlgorithm(raw: string | undefined): 'sha256' | 'sha1' | 'md5'
29
58
  return 'sha256'
30
59
  }
31
60
 
61
+ /**
62
+ * Parses `OM_SEARCH_FIELD_BLOCKLIST` into a global list plus per-entity-type lists.
63
+ *
64
+ * Why: a deployment often needs to keep one large free-text column out of the token
65
+ * index (e-mail bodies on `customers:customer_interaction`) while still indexing the
66
+ * same-named column elsewhere. A flat global list cannot express that.
67
+ *
68
+ * How to apply: entries are comma-separated; an entry may carry an optional
69
+ * `entityType@` prefix — `body` blocks the field everywhere, while
70
+ * `customers:customer_interaction@body` blocks it only for that entity type. Entries
71
+ * whose field part is empty are ignored so malformed env input cannot break indexing.
72
+ */
73
+ function parseFieldBlocklist(raw: string | undefined): {
74
+ global: string[]
75
+ byEntity: Record<string, string[]>
76
+ } {
77
+ const global: string[] = []
78
+ const byEntity = new Map<string, string[]>()
79
+
80
+ for (const rawEntry of parseCommaSeparatedList(raw)) {
81
+ const entry = rawEntry.toLowerCase()
82
+ const separatorIndex = entry.indexOf(ENTITY_BLOCKLIST_SEPARATOR)
83
+ const entityType = separatorIndex >= 0 ? entry.slice(0, separatorIndex).trim() : ''
84
+ const field = separatorIndex >= 0 ? entry.slice(separatorIndex + 1).trim() : entry
85
+ if (!field.length) continue
86
+
87
+ if (!entityType.length) {
88
+ if (!global.includes(field)) global.push(field)
89
+ continue
90
+ }
91
+
92
+ const scoped = byEntity.get(entityType) ?? []
93
+ if (!scoped.includes(field)) scoped.push(field)
94
+ byEntity.set(entityType, scoped)
95
+ }
96
+
97
+ for (const fallback of DEFAULT_BLOCKLIST) {
98
+ if (!global.includes(fallback)) global.push(fallback)
99
+ }
100
+
101
+ const scopedBlocklist = Object.create(null) as Record<string, string[]>
102
+ for (const [entityType, fields] of byEntity) scopedBlocklist[entityType] = fields
103
+
104
+ return { global, byEntity: scopedBlocklist }
105
+ }
106
+
32
107
  export function resolveSearchConfig(): SearchConfig {
108
+ const blocklist = parseFieldBlocklist(process.env.OM_SEARCH_FIELD_BLOCKLIST)
33
109
  return {
34
110
  enabled: parseBoolean(process.env.OM_SEARCH_ENABLED, true),
35
111
  minTokenLength: resolveSearchMinTokenLength(),
36
112
  enablePartials: parseBoolean(process.env.OM_SEARCH_ENABLE_PARTIAL, true),
37
113
  hashAlgorithm: parseHashAlgorithm(process.env.OM_SEARCH_HASH_ALGO),
38
114
  storeRawTokens: parseBoolean(process.env.OM_SEARCH_STORE_RAW_TOKENS, false),
39
- blocklistedFields: (process.env.OM_SEARCH_FIELD_BLOCKLIST ?? '')
40
- .split(',')
41
- .map((entry) => entry.trim())
42
- .filter((entry) => entry.length > 0)
43
- .filter((value, index, arr) => arr.indexOf(value) === index)
44
- .map((entry) => entry.toLowerCase())
45
- .concat(DEFAULT_BLOCKLIST)
46
- .filter((value, index, arr) => arr.indexOf(value) === index),
115
+ blocklistedFields: blocklist.global,
116
+ entityBlocklistedFields: blocklist.byEntity,
117
+ maxFieldChars: parseNumber(process.env.OM_SEARCH_MAX_FIELD_CHARS, DEFAULT_SEARCH_MAX_FIELD_CHARS, 0),
118
+ maxTokensPerField: parseNumber(process.env.OM_SEARCH_MAX_TOKENS_PER_FIELD, DEFAULT_SEARCH_MAX_TOKENS_PER_FIELD, 0),
119
+ maxTokensPerRecord: parseNumber(process.env.OM_SEARCH_MAX_TOKENS_PER_RECORD, DEFAULT_SEARCH_MAX_TOKENS_PER_RECORD, 0),
47
120
  }
48
121
  }
49
122
 
123
+ /**
124
+ * Single matcher for "should this field be kept out of the search index?".
125
+ *
126
+ * Why: the per-field token path and the `search_text` aggregate previously each
127
+ * decided this on their own, and the aggregate simply never consulted the config —
128
+ * so a blocklisted column's text came back into the index under the aggregate's
129
+ * field name (#4624). Both paths now share this function so they cannot drift.
130
+ *
131
+ * How to apply: pass the document's field name and the entity type being indexed;
132
+ * `entityType` may be omitted when unknown, in which case only global entries apply.
133
+ * Matching keeps the historical substring semantics (`fieldName.includes(pattern)`).
134
+ */
135
+ export function isSearchFieldBlocklisted(
136
+ field: string,
137
+ entityType: string | null | undefined,
138
+ config: SearchConfig,
139
+ ): boolean {
140
+ const lower = field.toLowerCase()
141
+ if (config.blocklistedFields.some((blocked) => lower.includes(blocked))) return true
142
+ if (!entityType) return false
143
+ const scoped = config.entityBlocklistedFields?.[entityType.trim().toLowerCase()]
144
+ if (!Array.isArray(scoped) || !scoped.length) return false
145
+ return scoped.some((blocked) => lower.includes(blocked))
146
+ }
147
+
50
148
  /**
51
149
  * Browser-safe accessor for the minimum search token length.
52
150
  *
@@ -0,0 +1,133 @@
1
+ import { type Kysely, sql } from 'kysely'
2
+ import { resolveSearchConfig, type SearchConfig } from './config'
3
+ import { tokenizeText } from './tokenize'
4
+
5
+ export type SearchTokenDatabase = {
6
+ search_tokens: {
7
+ entity_id: string
8
+ entity_type: string
9
+ field: string
10
+ token_hash: string
11
+ tenant_id: string | null
12
+ organization_id: string | null
13
+ }
14
+ }
15
+
16
+ /**
17
+ * Tenant/organization scoping for a `search_tokens` lookup.
18
+ *
19
+ * `undefined` and `null` are NOT interchangeable:
20
+ * - `undefined` omits the predicate entirely (the caller owns visibility).
21
+ * - `null` emits a null-safe predicate that matches only globally scoped rows.
22
+ */
23
+ export type SearchTokenScope = {
24
+ tenantId?: string | null
25
+ organizationId?: string | null
26
+ organizationIds?: readonly string[] | null
27
+ }
28
+
29
+ /**
30
+ * Why a lookup could not produce an id set. Callers MUST NOT read these as
31
+ * "nothing matched" — the token index was never consulted, so the caller's own
32
+ * predicate (usually an `ilike`) is still the authoritative one.
33
+ */
34
+ export type SearchTokenLookupSkipReason = 'empty-query' | 'search-disabled' | 'no-tokens'
35
+
36
+ export type SearchTokenLookupResult =
37
+ | { matched: true; ids: string[] }
38
+ | { matched: false; reason: SearchTokenLookupSkipReason }
39
+
40
+ export type FindEntityIdsBySearchTokensInput = {
41
+ db: Kysely<SearchTokenDatabase>
42
+ entityType: string
43
+ query: string
44
+ fields?: readonly string[] | null
45
+ scope?: SearchTokenScope
46
+ config?: SearchConfig
47
+ }
48
+
49
+ /**
50
+ * Resolve the record ids whose indexed `search_tokens` cover every token in
51
+ * `query`.
52
+ *
53
+ * This is the encryption-safe replacement for `ilike` filtering on columns an
54
+ * encryption map covers: the stored column holds ciphertext, so
55
+ * `ilike '%term%'` silently matches nothing, while the token index stores
56
+ * hashes of the plaintext and keeps matching. See issue #2990.
57
+ *
58
+ * Matching requires ALL query tokens to be present on the record. The token
59
+ * search strategy in `@open-mercato/search` uses a looser match ratio; list
60
+ * endpoints want the stricter behavior so a two-word query narrows rather than
61
+ * widens the result set.
62
+ */
63
+ export async function findEntityIdsBySearchTokens({
64
+ db,
65
+ entityType,
66
+ query,
67
+ fields,
68
+ scope,
69
+ config,
70
+ }: FindEntityIdsBySearchTokensInput): Promise<SearchTokenLookupResult> {
71
+ const trimmed = query.trim()
72
+ if (!trimmed) return { matched: false, reason: 'empty-query' }
73
+
74
+ const searchConfig = config ?? resolveSearchConfig()
75
+ if (!searchConfig.enabled) return { matched: false, reason: 'search-disabled' }
76
+
77
+ const { hashes } = tokenizeText(trimmed, searchConfig)
78
+ if (!hashes.length) return { matched: false, reason: 'no-tokens' }
79
+
80
+ let builder = db
81
+ .selectFrom('search_tokens')
82
+ .select('entity_id')
83
+ .where('entity_type', '=', entityType)
84
+ .where('token_hash', 'in', hashes)
85
+
86
+ const scopedFields = (fields ?? []).filter((field) => typeof field === 'string' && field.length > 0)
87
+ if (scopedFields.length === 1) {
88
+ builder = builder.where('field', '=', scopedFields[0])
89
+ } else if (scopedFields.length > 1) {
90
+ builder = builder.where('field', 'in', Array.from(scopedFields))
91
+ }
92
+
93
+ if (scope?.tenantId !== undefined) {
94
+ builder = builder.where(sql<boolean>`tenant_id is not distinct from ${scope.tenantId}`)
95
+ }
96
+
97
+ if (scope?.organizationId !== undefined) {
98
+ builder = scope.organizationId === null
99
+ ? builder.where(sql<boolean>`organization_id is not distinct from ${null}`)
100
+ : builder.where('organization_id', '=', scope.organizationId)
101
+ } else if (scope?.organizationIds?.length) {
102
+ builder = builder.where('organization_id', 'in', Array.from(scope.organizationIds))
103
+ }
104
+
105
+ const rows = (await builder
106
+ .groupBy('entity_id')
107
+ .having(sql<boolean>`count(distinct token_hash) >= ${hashes.length}`)
108
+ .execute()) as Array<{ entity_id?: unknown }>
109
+
110
+ const ids = rows
111
+ .map((row) => (typeof row.entity_id === 'string' ? row.entity_id : null))
112
+ .filter((id): id is string => typeof id === 'string' && id.length > 0)
113
+
114
+ return { matched: true, ids }
115
+ }
116
+
117
+ /**
118
+ * Legacy-shaped adapter for call sites that predate
119
+ * {@link SearchTokenLookupResult}: `null` for a blank query, `[]` for every
120
+ * other non-answer, otherwise the matched ids.
121
+ *
122
+ * Prefer {@link findEntityIdsBySearchTokens} in new code — the discriminated
123
+ * result distinguishes "the index says nothing matched" from "the index was
124
+ * never consulted", and that distinction is exactly what a `null`/`[]` pair
125
+ * loses.
126
+ */
127
+ export async function findEntityIdsBySearchTokensCompat(
128
+ input: FindEntityIdsBySearchTokensInput,
129
+ ): Promise<string[] | null> {
130
+ const result = await findEntityIdsBySearchTokens(input)
131
+ if (result.matched) return result.ids
132
+ return result.reason === 'empty-query' ? null : []
133
+ }
@@ -1,5 +1,5 @@
1
1
  import crypto from 'crypto'
2
- import { resolveSearchConfig, type SearchConfig } from './config'
2
+ import { resolveSearchConfig, resolveSearchTokenLimits, type SearchConfig } from './config'
3
3
 
4
4
  export type TokenizationResult = {
5
5
  tokens: string[]
@@ -20,13 +20,28 @@ function splitTokens(text: string, minLength: number): string[] {
20
20
  .filter((token) => token.length >= minLength)
21
21
  }
22
22
 
23
- function expandToken(token: string, config: SearchConfig): string[] {
24
- if (!config.enablePartials) return [token]
25
- const results: string[] = []
26
- for (let i = config.minTokenLength; i <= token.length; i += 1) {
27
- results.push(token.slice(0, i))
23
+ function appendExpandedToken(
24
+ token: string,
25
+ config: SearchConfig,
26
+ seen: Set<string>,
27
+ tokens: string[],
28
+ limit: number,
29
+ ): void {
30
+ const append = (candidate: string): boolean => {
31
+ if (seen.has(candidate)) return tokens.length < limit
32
+ seen.add(candidate)
33
+ tokens.push(candidate)
34
+ return tokens.length < limit
35
+ }
36
+
37
+ if (!config.enablePartials) {
38
+ append(token)
39
+ return
40
+ }
41
+
42
+ for (let length = config.minTokenLength; length <= token.length; length += 1) {
43
+ if (!append(token.slice(0, length))) return
28
44
  }
29
- return results
30
45
  }
31
46
 
32
47
  export function hashToken(token: string, config?: SearchConfig): string {
@@ -36,10 +51,17 @@ export function hashToken(token: string, config?: SearchConfig): string {
36
51
 
37
52
  export function tokenizeText(text: string, config?: SearchConfig): TokenizationResult {
38
53
  const cfg = config ?? resolveSearchConfig()
39
- const baseTokens = splitTokens(text, cfg.minTokenLength)
40
- const expanded = baseTokens.flatMap((token) => expandToken(token, cfg))
41
- const unique = Array.from(new Set(expanded))
42
- const tokens = unique.filter((token) => token.length >= cfg.minTokenLength)
54
+ const limits = resolveSearchTokenLimits(cfg)
55
+ const boundedText = limits.maxFieldChars > 0 ? text.slice(0, limits.maxFieldChars) : text
56
+ const tokenLimit = limits.maxTokensPerField > 0 ? limits.maxTokensPerField : Number.POSITIVE_INFINITY
57
+ const seen = new Set<string>()
58
+ const tokens: string[] = []
59
+
60
+ for (const token of splitTokens(boundedText, cfg.minTokenLength)) {
61
+ if (tokens.length >= tokenLimit) break
62
+ appendExpandedToken(token, cfg, seen, tokens, tokenLimit)
63
+ }
64
+
43
65
  const hashes = tokens.map((token) => hashToken(token, cfg))
44
66
  return { tokens, hashes }
45
67
  }
@@ -0,0 +1,72 @@
1
+ export type TelemetryTraceCarrier = Record<string, string>
2
+
3
+ export type TelemetryRuntime = {
4
+ /**
5
+ * True only when the active SDK may safely use the process-global W3C
6
+ * propagator for cross-boundary extraction.
7
+ */
8
+ canUseGlobalTracePropagation(): boolean
9
+ captureTraceContext(): TelemetryTraceCarrier
10
+ continueTrace<T>(
11
+ carrier: TelemetryTraceCarrier | undefined,
12
+ name: string,
13
+ fn: () => T,
14
+ options?: { kind?: 'internal' | 'server' | 'client' | 'producer' | 'consumer' },
15
+ ): T
16
+ recordHttpDuration(method: string, route: string, status: number, startedAt: number): void
17
+ reportError(
18
+ error: unknown,
19
+ context?: {
20
+ module?: string
21
+ attributes?: Record<string, string | number | boolean | undefined>
22
+ },
23
+ ): void
24
+ shutdown(): Promise<void>
25
+ }
26
+
27
+ const GLOBAL_KEY = Symbol.for('@open-mercato/shared.telemetryRuntime')
28
+ const ENABLED_BACKENDS = new Set(['console', 'signoz', 'newrelic', 'otlp'])
29
+
30
+ type TelemetryRuntimeStore = {
31
+ active?: TelemetryRuntime
32
+ }
33
+
34
+ function store(): TelemetryRuntimeStore {
35
+ const globalStore = globalThis as unknown as Record<symbol, TelemetryRuntimeStore | undefined>
36
+ let current = globalStore[GLOBAL_KEY]
37
+ if (!current) {
38
+ current = {}
39
+ globalStore[GLOBAL_KEY] = current
40
+ }
41
+ return current
42
+ }
43
+
44
+ /**
45
+ * This check is intentionally owned by shared code so hosts can decide whether
46
+ * to dynamically import the telemetry package without evaluating that package.
47
+ */
48
+ export function isTelemetryBackendEnabled(raw?: string): boolean {
49
+ const value = raw ?? (
50
+ typeof process === 'undefined'
51
+ ? undefined
52
+ : process.env.TELEMETRY_BACKEND
53
+ )
54
+ return ENABLED_BACKENDS.has((value ?? '').trim().toLowerCase())
55
+ }
56
+
57
+ export function registerTelemetryRuntime(runtime: TelemetryRuntime): () => void {
58
+ store().active = runtime
59
+ return () => {
60
+ const current = store()
61
+ if (current.active === runtime) current.active = undefined
62
+ }
63
+ }
64
+
65
+ export function getTelemetryRuntime(): TelemetryRuntime | undefined {
66
+ return store().active
67
+ }
68
+
69
+ /** Test-only: clear the process-wide telemetry bridge. */
70
+ export function resetTelemetryRuntime(): void {
71
+ store().active = undefined
72
+ }