@open-mercato/shared 0.8.1-develop.7294.1.0ef99fc518 → 0.8.1-develop.7296.1.2111d779db

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "version": 3,
3
3
  "sources": ["../../../src/lib/search/config.ts"],
4
- "sourcesContent": ["import { parseBooleanWithDefault } from '@open-mercato/shared/lib/boolean'\nimport { parseNumberWithDefault } from '@open-mercato/shared/lib/number'\nimport { parseCommaSeparatedList } from '@open-mercato/shared/lib/string'\n\nexport type SearchConfig = {\n enabled: boolean\n minTokenLength: number\n enablePartials: boolean\n hashAlgorithm: 'sha256' | 'sha1' | 'md5'\n storeRawTokens: boolean\n /**\n * When true, a like/ilike on a PLAINTEXT base column runs as exact SQL ILIKE instead of being\n * rewritten into an approximate search-token match; encrypted columns always keep the token\n * path (ILIKE against ciphertext cannot match). Off by default: token matching can be faster\n * than an unanchored ILIKE, which may need a full scan without a trigram index \u2014 but it is\n * approximate (fragments under minTokenLength vanish, so `ZK 1/2026` degrades to its year and\n * an all-short term drops the predicate). Flip it on when list search must be exact.\n */\n useIlikeForNonEncryptedFields?: boolean\n blocklistedFields: string[]\n entityBlocklistedFields?: Record<string, string[]>\n maxFieldChars?: number\n maxTokensPerField?: number\n /**\n * Ceiling on token rows across all fields of one record; `0` disables it. The budget is spent in\n * the order the document's own keys iterate in, so on an over-budget record *which* fields stay\n * searchable depends on that key order \u2014 see `buildSearchTokenRows` in\n * `@open-mercato/core/modules/query_index/lib/search-tokens` before recomputing expected tokens\n * from a document that did not come straight from the indexer.\n */\n maxTokensPerRecord?: number\n}\n\nexport const DEFAULT_SEARCH_MIN_TOKEN_LENGTH = 3\nexport const DEFAULT_SEARCH_MAX_FIELD_CHARS = 20_000\nexport const DEFAULT_SEARCH_MAX_TOKENS_PER_FIELD = 5_000\nexport const DEFAULT_SEARCH_MAX_TOKENS_PER_RECORD = 20_000\n\nexport type SearchTokenLimits = {\n maxFieldChars: number\n maxTokensPerField: number\n maxTokensPerRecord: number\n}\n\nconst DEFAULT_BLOCKLIST = ['password', 'token', 'secret', 'hash']\n\nconst ENTITY_BLOCKLIST_SEPARATOR = '@'\n\nfunction parseBoolean(raw: string | undefined, fallback: boolean): boolean {\n return parseBooleanWithDefault(raw, fallback)\n}\n\nfunction parseNumber(raw: string | undefined, fallback: number, min = 1): number {\n return parseNumberWithDefault(raw, fallback, { integer: true, min })\n}\n\nexport function resolveSearchTokenLimits(config: SearchConfig): SearchTokenLimits {\n const resolveLimit = (value: number | undefined, fallback: number): number => {\n if (value === undefined) return fallback\n if (!Number.isFinite(value) || value < 0) return fallback\n return Math.trunc(value)\n }\n return {\n maxFieldChars: resolveLimit(config.maxFieldChars, DEFAULT_SEARCH_MAX_FIELD_CHARS),\n maxTokensPerField: resolveLimit(config.maxTokensPerField, DEFAULT_SEARCH_MAX_TOKENS_PER_FIELD),\n maxTokensPerRecord: resolveLimit(config.maxTokensPerRecord, DEFAULT_SEARCH_MAX_TOKENS_PER_RECORD),\n }\n}\n\nfunction parseHashAlgorithm(raw: string | undefined): 'sha256' | 'sha1' | 'md5' {\n const value = (raw ?? '').trim().toLowerCase()\n if (value === 'sha1') return 'sha1'\n if (value === 'md5') return 'md5'\n return 'sha256'\n}\n\n/**\n * Parses `OM_SEARCH_FIELD_BLOCKLIST` into a global list plus per-entity-type lists.\n *\n * Why: a deployment often needs to keep one large free-text column out of the token\n * index (e-mail bodies on `customers:customer_interaction`) while still indexing the\n * same-named column elsewhere. A flat global list cannot express that.\n *\n * How to apply: entries are comma-separated; an entry may carry an optional\n * `entityType@` prefix \u2014 `body` blocks the field everywhere, while\n * `customers:customer_interaction@body` blocks it only for that entity type. Entries\n * whose field part is empty are ignored so malformed env input cannot break indexing.\n */\nfunction parseFieldBlocklist(raw: string | undefined): {\n global: string[]\n byEntity: Record<string, string[]>\n} {\n const global: string[] = []\n const byEntity = new Map<string, string[]>()\n\n for (const rawEntry of parseCommaSeparatedList(raw)) {\n const entry = rawEntry.toLowerCase()\n const separatorIndex = entry.indexOf(ENTITY_BLOCKLIST_SEPARATOR)\n const entityType = separatorIndex >= 0 ? entry.slice(0, separatorIndex).trim() : ''\n const field = separatorIndex >= 0 ? entry.slice(separatorIndex + 1).trim() : entry\n if (!field.length) continue\n\n if (!entityType.length) {\n if (!global.includes(field)) global.push(field)\n continue\n }\n\n const scoped = byEntity.get(entityType) ?? []\n if (!scoped.includes(field)) scoped.push(field)\n byEntity.set(entityType, scoped)\n }\n\n for (const fallback of DEFAULT_BLOCKLIST) {\n if (!global.includes(fallback)) global.push(fallback)\n }\n\n const scopedBlocklist = Object.create(null) as Record<string, string[]>\n for (const [entityType, fields] of byEntity) scopedBlocklist[entityType] = fields\n\n return { global, byEntity: scopedBlocklist }\n}\n\nexport function resolveSearchConfig(): SearchConfig {\n const blocklist = parseFieldBlocklist(process.env.OM_SEARCH_FIELD_BLOCKLIST)\n return {\n enabled: parseBoolean(process.env.OM_SEARCH_ENABLED, true),\n minTokenLength: resolveSearchMinTokenLength(),\n enablePartials: parseBoolean(process.env.OM_SEARCH_ENABLE_PARTIAL, true),\n hashAlgorithm: parseHashAlgorithm(process.env.OM_SEARCH_HASH_ALGO),\n storeRawTokens: parseBoolean(process.env.OM_SEARCH_STORE_RAW_TOKENS, false),\n useIlikeForNonEncryptedFields: parseBoolean(process.env.OM_SEARCH_USE_ILIKE_FOR_NON_ENCRYPTED_FIELDS, false),\n blocklistedFields: blocklist.global,\n entityBlocklistedFields: blocklist.byEntity,\n maxFieldChars: parseNumber(process.env.OM_SEARCH_MAX_FIELD_CHARS, DEFAULT_SEARCH_MAX_FIELD_CHARS, 0),\n maxTokensPerField: parseNumber(process.env.OM_SEARCH_MAX_TOKENS_PER_FIELD, DEFAULT_SEARCH_MAX_TOKENS_PER_FIELD, 0),\n maxTokensPerRecord: parseNumber(process.env.OM_SEARCH_MAX_TOKENS_PER_RECORD, DEFAULT_SEARCH_MAX_TOKENS_PER_RECORD, 0),\n }\n}\n\n/**\n * Single matcher for \"should this field be kept out of the search index?\".\n *\n * Why: the per-field token path and the `search_text` aggregate previously each\n * decided this on their own, and the aggregate simply never consulted the config \u2014\n * so a blocklisted column's text came back into the index under the aggregate's\n * field name (#4624). Both paths now share this function so they cannot drift.\n *\n * How to apply: pass the document's field name and the entity type being indexed;\n * `entityType` may be omitted when unknown, in which case only global entries apply.\n * Matching keeps the historical substring semantics (`fieldName.includes(pattern)`).\n */\nexport function isSearchFieldBlocklisted(\n field: string,\n entityType: string | null | undefined,\n config: SearchConfig,\n): boolean {\n const lower = field.toLowerCase()\n if (config.blocklistedFields.some((blocked) => lower.includes(blocked))) return true\n if (!entityType) return false\n const scoped = config.entityBlocklistedFields?.[entityType.trim().toLowerCase()]\n if (!Array.isArray(scoped) || !scoped.length) return false\n return scoped.some((blocked) => lower.includes(blocked))\n}\n\n/**\n * Browser-safe accessor for the minimum search token length.\n *\n * Why: client components (e.g. global search dialog) must mirror the server-side\n * tokenizer's `minTokenLength` so the UI gates the request before hitting an\n * empty result set. Pulling the value through this single helper keeps the env\n * contract (`OM_SEARCH_MIN_LEN`) authoritative on both sides.\n *\n * How to apply: call from anywhere \u2014 server, client (when the host app exposes\n * `OM_SEARCH_MIN_LEN` through `next.config.ts`'s `env` block), or tests.\n */\nexport function resolveSearchMinTokenLength(): number {\n return parseNumber(process.env.OM_SEARCH_MIN_LEN, DEFAULT_SEARCH_MIN_TOKEN_LENGTH, 1)\n}\n"],
5
- "mappings": "AAAA,SAAS,+BAA+B;AACxC,SAAS,8BAA8B;AACvC,SAAS,+BAA+B;AA+BjC,MAAM,kCAAkC;AACxC,MAAM,iCAAiC;AACvC,MAAM,sCAAsC;AAC5C,MAAM,uCAAuC;AAQpD,MAAM,oBAAoB,CAAC,YAAY,SAAS,UAAU,MAAM;AAEhE,MAAM,6BAA6B;AAEnC,SAAS,aAAa,KAAyB,UAA4B;AACzE,SAAO,wBAAwB,KAAK,QAAQ;AAC9C;AAEA,SAAS,YAAY,KAAyB,UAAkB,MAAM,GAAW;AAC/E,SAAO,uBAAuB,KAAK,UAAU,EAAE,SAAS,MAAM,IAAI,CAAC;AACrE;AAEO,SAAS,yBAAyB,QAAyC;AAChF,QAAM,eAAe,CAAC,OAA2B,aAA6B;AAC5E,QAAI,UAAU,OAAW,QAAO;AAChC,QAAI,CAAC,OAAO,SAAS,KAAK,KAAK,QAAQ,EAAG,QAAO;AACjD,WAAO,KAAK,MAAM,KAAK;AAAA,EACzB;AACA,SAAO;AAAA,IACL,eAAe,aAAa,OAAO,eAAe,8BAA8B;AAAA,IAChF,mBAAmB,aAAa,OAAO,mBAAmB,mCAAmC;AAAA,IAC7F,oBAAoB,aAAa,OAAO,oBAAoB,oCAAoC;AAAA,EAClG;AACF;AAEA,SAAS,mBAAmB,KAAoD;AAC9E,QAAM,SAAS,OAAO,IAAI,KAAK,EAAE,YAAY;AAC7C,MAAI,UAAU,OAAQ,QAAO;AAC7B,MAAI,UAAU,MAAO,QAAO;AAC5B,SAAO;AACT;AAcA,SAAS,oBAAoB,KAG3B;AACA,QAAM,SAAmB,CAAC;AAC1B,QAAM,WAAW,oBAAI,IAAsB;AAE3C,aAAW,YAAY,wBAAwB,GAAG,GAAG;AACnD,UAAM,QAAQ,SAAS,YAAY;AACnC,UAAM,iBAAiB,MAAM,QAAQ,0BAA0B;AAC/D,UAAM,aAAa,kBAAkB,IAAI,MAAM,MAAM,GAAG,cAAc,EAAE,KAAK,IAAI;AACjF,UAAM,QAAQ,kBAAkB,IAAI,MAAM,MAAM,iBAAiB,CAAC,EAAE,KAAK,IAAI;AAC7E,QAAI,CAAC,MAAM,OAAQ;AAEnB,QAAI,CAAC,WAAW,QAAQ;AACtB,UAAI,CAAC,OAAO,SAAS,KAAK,EAAG,QAAO,KAAK,KAAK;AAC9C;AAAA,IACF;AAEA,UAAM,SAAS,SAAS,IAAI,UAAU,KAAK,CAAC;AAC5C,QAAI,CAAC,OAAO,SAAS,KAAK,EAAG,QAAO,KAAK,KAAK;AAC9C,aAAS,IAAI,YAAY,MAAM;AAAA,EACjC;AAEA,aAAW,YAAY,mBAAmB;AACxC,QAAI,CAAC,OAAO,SAAS,QAAQ,EAAG,QAAO,KAAK,QAAQ;AAAA,EACtD;AAEA,QAAM,kBAAkB,uBAAO,OAAO,IAAI;AAC1C,aAAW,CAAC,YAAY,MAAM,KAAK,SAAU,iBAAgB,UAAU,IAAI;AAE3E,SAAO,EAAE,QAAQ,UAAU,gBAAgB;AAC7C;AAEO,SAAS,sBAAoC;AAClD,QAAM,YAAY,oBAAoB,QAAQ,IAAI,yBAAyB;AAC3E,SAAO;AAAA,IACL,SAAS,aAAa,QAAQ,IAAI,mBAAmB,IAAI;AAAA,IACzD,gBAAgB,4BAA4B;AAAA,IAC5C,gBAAgB,aAAa,QAAQ,IAAI,0BAA0B,IAAI;AAAA,IACvE,eAAe,mBAAmB,QAAQ,IAAI,mBAAmB;AAAA,IACjE,gBAAgB,aAAa,QAAQ,IAAI,4BAA4B,KAAK;AAAA,IAC1E,+BAA+B,aAAa,QAAQ,IAAI,8CAA8C,KAAK;AAAA,IAC3G,mBAAmB,UAAU;AAAA,IAC7B,yBAAyB,UAAU;AAAA,IACnC,eAAe,YAAY,QAAQ,IAAI,2BAA2B,gCAAgC,CAAC;AAAA,IACnG,mBAAmB,YAAY,QAAQ,IAAI,gCAAgC,qCAAqC,CAAC;AAAA,IACjH,oBAAoB,YAAY,QAAQ,IAAI,iCAAiC,sCAAsC,CAAC;AAAA,EACtH;AACF;AAcO,SAAS,yBACd,OACA,YACA,QACS;AACT,QAAM,QAAQ,MAAM,YAAY;AAChC,MAAI,OAAO,kBAAkB,KAAK,CAAC,YAAY,MAAM,SAAS,OAAO,CAAC,EAAG,QAAO;AAChF,MAAI,CAAC,WAAY,QAAO;AACxB,QAAM,SAAS,OAAO,0BAA0B,WAAW,KAAK,EAAE,YAAY,CAAC;AAC/E,MAAI,CAAC,MAAM,QAAQ,MAAM,KAAK,CAAC,OAAO,OAAQ,QAAO;AACrD,SAAO,OAAO,KAAK,CAAC,YAAY,MAAM,SAAS,OAAO,CAAC;AACzD;AAaO,SAAS,8BAAsC;AACpD,SAAO,YAAY,QAAQ,IAAI,mBAAmB,iCAAiC,CAAC;AACtF;",
4
+ "sourcesContent": ["import { parseBooleanWithDefault } from '@open-mercato/shared/lib/boolean'\nimport { parseNumberWithDefault } from '@open-mercato/shared/lib/number'\nimport { parseCommaSeparatedList } from '@open-mercato/shared/lib/string'\n\nexport type SearchConfig = {\n enabled: boolean\n minTokenLength: number\n enablePartials: boolean\n hashAlgorithm: 'sha256' | 'sha1' | 'md5'\n storeRawTokens: boolean\n /**\n * When true, a like/ilike on a PLAINTEXT base column runs as SQL ILIKE \u2014 one containment\n * predicate per word of the term, ANDed \u2014 instead of being rewritten into an approximate\n * search-token match; encrypted columns always keep the token path (ILIKE against ciphertext\n * cannot match).\n *\n * Off by default, per #5383: the token store is expected to become faster than ILIKE once\n * tokenization is made semantically equivalent to it, so the plan there is to keep this switch\n * off until that follow-up lands rather than trade performance for correctness by default. #5803\n * documents the correctness gap this switch closes when enabled: the token rewrite is lossy in a\n * way that silently returns the WRONG record rather than merely extra ones (tokenization splits\n * on non-alphanumerics and drops fragments under minTokenLength, so `2026-08` and `2026-01` both\n * reduce to {202, 2026} and a picker offers the neighbouring period; a term that tokenizes to\n * nothing (`08`) drops the predicate entirely and matches every row) \u2014 a deployment that hits\n * that gap before #5383 lands can opt in here.\n *\n * Per-word ANDing (see lib/search/containment) is a trade-off, not a strict improvement, over\n * the single-literal ILIKE #4622 originally introduced: the token subquery matched a value\n * carrying every token in any order with anything between them, so `?search=Warehouse 1757`\n * must keep matching `Warehouse A 1757` \u2014 a single verbatim `ILIKE '%Warehouse 1757%'` would\n * not, and TC-RESO-009 pins that as required behavior. The same word-order independence also\n * widens multi-word document-number searches: `?search=ZK 1/2026` now also matches\n * `ZK 11/2026` and `1/2026 ZK`, where the old single-literal ILIKE matched neither.\n *\n * Set `OM_SEARCH_USE_ILIKE_FOR_NON_ENCRYPTED_FIELDS=true` to opt into declared-column ILIKE\n * ahead of #5383 \u2014 worth doing when the #5803 wrong-record symptom is hit in practice. Leaving it\n * unset keeps the legacy rewrite-everything behavior, including the token index's prefix matching\n * (`?search=ware` matching `Warehouse` when `enablePartials` is on, which literal containment\n * gives only where the fragment really is a substring).\n */\n useIlikeForNonEncryptedFields?: boolean\n blocklistedFields: string[]\n entityBlocklistedFields?: Record<string, string[]>\n maxFieldChars?: number\n maxTokensPerField?: number\n /**\n * Ceiling on token rows across all fields of one record; `0` disables it. The budget is spent in\n * the order the document's own keys iterate in, so on an over-budget record *which* fields stay\n * searchable depends on that key order \u2014 see `buildSearchTokenRows` in\n * `@open-mercato/core/modules/query_index/lib/search-tokens` before recomputing expected tokens\n * from a document that did not come straight from the indexer.\n */\n maxTokensPerRecord?: number\n}\n\nexport const DEFAULT_SEARCH_MIN_TOKEN_LENGTH = 3\nexport const DEFAULT_SEARCH_MAX_FIELD_CHARS = 20_000\nexport const DEFAULT_SEARCH_MAX_TOKENS_PER_FIELD = 5_000\nexport const DEFAULT_SEARCH_MAX_TOKENS_PER_RECORD = 20_000\n\nexport type SearchTokenLimits = {\n maxFieldChars: number\n maxTokensPerField: number\n maxTokensPerRecord: number\n}\n\nconst DEFAULT_BLOCKLIST = ['password', 'token', 'secret', 'hash']\n\nconst ENTITY_BLOCKLIST_SEPARATOR = '@'\n\nfunction parseBoolean(raw: string | undefined, fallback: boolean): boolean {\n return parseBooleanWithDefault(raw, fallback)\n}\n\nfunction parseNumber(raw: string | undefined, fallback: number, min = 1): number {\n return parseNumberWithDefault(raw, fallback, { integer: true, min })\n}\n\nexport function resolveSearchTokenLimits(config: SearchConfig): SearchTokenLimits {\n const resolveLimit = (value: number | undefined, fallback: number): number => {\n if (value === undefined) return fallback\n if (!Number.isFinite(value) || value < 0) return fallback\n return Math.trunc(value)\n }\n return {\n maxFieldChars: resolveLimit(config.maxFieldChars, DEFAULT_SEARCH_MAX_FIELD_CHARS),\n maxTokensPerField: resolveLimit(config.maxTokensPerField, DEFAULT_SEARCH_MAX_TOKENS_PER_FIELD),\n maxTokensPerRecord: resolveLimit(config.maxTokensPerRecord, DEFAULT_SEARCH_MAX_TOKENS_PER_RECORD),\n }\n}\n\nfunction parseHashAlgorithm(raw: string | undefined): 'sha256' | 'sha1' | 'md5' {\n const value = (raw ?? '').trim().toLowerCase()\n if (value === 'sha1') return 'sha1'\n if (value === 'md5') return 'md5'\n return 'sha256'\n}\n\n/**\n * Parses `OM_SEARCH_FIELD_BLOCKLIST` into a global list plus per-entity-type lists.\n *\n * Why: a deployment often needs to keep one large free-text column out of the token\n * index (e-mail bodies on `customers:customer_interaction`) while still indexing the\n * same-named column elsewhere. A flat global list cannot express that.\n *\n * How to apply: entries are comma-separated; an entry may carry an optional\n * `entityType@` prefix \u2014 `body` blocks the field everywhere, while\n * `customers:customer_interaction@body` blocks it only for that entity type. Entries\n * whose field part is empty are ignored so malformed env input cannot break indexing.\n */\nfunction parseFieldBlocklist(raw: string | undefined): {\n global: string[]\n byEntity: Record<string, string[]>\n} {\n const global: string[] = []\n const byEntity = new Map<string, string[]>()\n\n for (const rawEntry of parseCommaSeparatedList(raw)) {\n const entry = rawEntry.toLowerCase()\n const separatorIndex = entry.indexOf(ENTITY_BLOCKLIST_SEPARATOR)\n const entityType = separatorIndex >= 0 ? entry.slice(0, separatorIndex).trim() : ''\n const field = separatorIndex >= 0 ? entry.slice(separatorIndex + 1).trim() : entry\n if (!field.length) continue\n\n if (!entityType.length) {\n if (!global.includes(field)) global.push(field)\n continue\n }\n\n const scoped = byEntity.get(entityType) ?? []\n if (!scoped.includes(field)) scoped.push(field)\n byEntity.set(entityType, scoped)\n }\n\n for (const fallback of DEFAULT_BLOCKLIST) {\n if (!global.includes(fallback)) global.push(fallback)\n }\n\n const scopedBlocklist = Object.create(null) as Record<string, string[]>\n for (const [entityType, fields] of byEntity) scopedBlocklist[entityType] = fields\n\n return { global, byEntity: scopedBlocklist }\n}\n\nexport function resolveSearchConfig(): SearchConfig {\n const blocklist = parseFieldBlocklist(process.env.OM_SEARCH_FIELD_BLOCKLIST)\n return {\n enabled: parseBoolean(process.env.OM_SEARCH_ENABLED, true),\n minTokenLength: resolveSearchMinTokenLength(),\n enablePartials: parseBoolean(process.env.OM_SEARCH_ENABLE_PARTIAL, true),\n hashAlgorithm: parseHashAlgorithm(process.env.OM_SEARCH_HASH_ALGO),\n storeRawTokens: parseBoolean(process.env.OM_SEARCH_STORE_RAW_TOKENS, false),\n useIlikeForNonEncryptedFields: parseBoolean(process.env.OM_SEARCH_USE_ILIKE_FOR_NON_ENCRYPTED_FIELDS, false),\n blocklistedFields: blocklist.global,\n entityBlocklistedFields: blocklist.byEntity,\n maxFieldChars: parseNumber(process.env.OM_SEARCH_MAX_FIELD_CHARS, DEFAULT_SEARCH_MAX_FIELD_CHARS, 0),\n maxTokensPerField: parseNumber(process.env.OM_SEARCH_MAX_TOKENS_PER_FIELD, DEFAULT_SEARCH_MAX_TOKENS_PER_FIELD, 0),\n maxTokensPerRecord: parseNumber(process.env.OM_SEARCH_MAX_TOKENS_PER_RECORD, DEFAULT_SEARCH_MAX_TOKENS_PER_RECORD, 0),\n }\n}\n\n/**\n * Single matcher for \"should this field be kept out of the search index?\".\n *\n * Why: the per-field token path and the `search_text` aggregate previously each\n * decided this on their own, and the aggregate simply never consulted the config \u2014\n * so a blocklisted column's text came back into the index under the aggregate's\n * field name (#4624). Both paths now share this function so they cannot drift.\n *\n * How to apply: pass the document's field name and the entity type being indexed;\n * `entityType` may be omitted when unknown, in which case only global entries apply.\n * Matching keeps the historical substring semantics (`fieldName.includes(pattern)`).\n */\nexport function isSearchFieldBlocklisted(\n field: string,\n entityType: string | null | undefined,\n config: SearchConfig,\n): boolean {\n const lower = field.toLowerCase()\n if (config.blocklistedFields.some((blocked) => lower.includes(blocked))) return true\n if (!entityType) return false\n const scoped = config.entityBlocklistedFields?.[entityType.trim().toLowerCase()]\n if (!Array.isArray(scoped) || !scoped.length) return false\n return scoped.some((blocked) => lower.includes(blocked))\n}\n\n/**\n * Browser-safe accessor for the minimum search token length.\n *\n * Why: client components (e.g. global search dialog) must mirror the server-side\n * tokenizer's `minTokenLength` so the UI gates the request before hitting an\n * empty result set. Pulling the value through this single helper keeps the env\n * contract (`OM_SEARCH_MIN_LEN`) authoritative on both sides.\n *\n * How to apply: call from anywhere \u2014 server, client (when the host app exposes\n * `OM_SEARCH_MIN_LEN` through `next.config.ts`'s `env` block), or tests.\n */\nexport function resolveSearchMinTokenLength(): number {\n return parseNumber(process.env.OM_SEARCH_MIN_LEN, DEFAULT_SEARCH_MIN_TOKEN_LENGTH, 1)\n}\n"],
5
+ "mappings": "AAAA,SAAS,+BAA+B;AACxC,SAAS,8BAA8B;AACvC,SAAS,+BAA+B;AAqDjC,MAAM,kCAAkC;AACxC,MAAM,iCAAiC;AACvC,MAAM,sCAAsC;AAC5C,MAAM,uCAAuC;AAQpD,MAAM,oBAAoB,CAAC,YAAY,SAAS,UAAU,MAAM;AAEhE,MAAM,6BAA6B;AAEnC,SAAS,aAAa,KAAyB,UAA4B;AACzE,SAAO,wBAAwB,KAAK,QAAQ;AAC9C;AAEA,SAAS,YAAY,KAAyB,UAAkB,MAAM,GAAW;AAC/E,SAAO,uBAAuB,KAAK,UAAU,EAAE,SAAS,MAAM,IAAI,CAAC;AACrE;AAEO,SAAS,yBAAyB,QAAyC;AAChF,QAAM,eAAe,CAAC,OAA2B,aAA6B;AAC5E,QAAI,UAAU,OAAW,QAAO;AAChC,QAAI,CAAC,OAAO,SAAS,KAAK,KAAK,QAAQ,EAAG,QAAO;AACjD,WAAO,KAAK,MAAM,KAAK;AAAA,EACzB;AACA,SAAO;AAAA,IACL,eAAe,aAAa,OAAO,eAAe,8BAA8B;AAAA,IAChF,mBAAmB,aAAa,OAAO,mBAAmB,mCAAmC;AAAA,IAC7F,oBAAoB,aAAa,OAAO,oBAAoB,oCAAoC;AAAA,EAClG;AACF;AAEA,SAAS,mBAAmB,KAAoD;AAC9E,QAAM,SAAS,OAAO,IAAI,KAAK,EAAE,YAAY;AAC7C,MAAI,UAAU,OAAQ,QAAO;AAC7B,MAAI,UAAU,MAAO,QAAO;AAC5B,SAAO;AACT;AAcA,SAAS,oBAAoB,KAG3B;AACA,QAAM,SAAmB,CAAC;AAC1B,QAAM,WAAW,oBAAI,IAAsB;AAE3C,aAAW,YAAY,wBAAwB,GAAG,GAAG;AACnD,UAAM,QAAQ,SAAS,YAAY;AACnC,UAAM,iBAAiB,MAAM,QAAQ,0BAA0B;AAC/D,UAAM,aAAa,kBAAkB,IAAI,MAAM,MAAM,GAAG,cAAc,EAAE,KAAK,IAAI;AACjF,UAAM,QAAQ,kBAAkB,IAAI,MAAM,MAAM,iBAAiB,CAAC,EAAE,KAAK,IAAI;AAC7E,QAAI,CAAC,MAAM,OAAQ;AAEnB,QAAI,CAAC,WAAW,QAAQ;AACtB,UAAI,CAAC,OAAO,SAAS,KAAK,EAAG,QAAO,KAAK,KAAK;AAC9C;AAAA,IACF;AAEA,UAAM,SAAS,SAAS,IAAI,UAAU,KAAK,CAAC;AAC5C,QAAI,CAAC,OAAO,SAAS,KAAK,EAAG,QAAO,KAAK,KAAK;AAC9C,aAAS,IAAI,YAAY,MAAM;AAAA,EACjC;AAEA,aAAW,YAAY,mBAAmB;AACxC,QAAI,CAAC,OAAO,SAAS,QAAQ,EAAG,QAAO,KAAK,QAAQ;AAAA,EACtD;AAEA,QAAM,kBAAkB,uBAAO,OAAO,IAAI;AAC1C,aAAW,CAAC,YAAY,MAAM,KAAK,SAAU,iBAAgB,UAAU,IAAI;AAE3E,SAAO,EAAE,QAAQ,UAAU,gBAAgB;AAC7C;AAEO,SAAS,sBAAoC;AAClD,QAAM,YAAY,oBAAoB,QAAQ,IAAI,yBAAyB;AAC3E,SAAO;AAAA,IACL,SAAS,aAAa,QAAQ,IAAI,mBAAmB,IAAI;AAAA,IACzD,gBAAgB,4BAA4B;AAAA,IAC5C,gBAAgB,aAAa,QAAQ,IAAI,0BAA0B,IAAI;AAAA,IACvE,eAAe,mBAAmB,QAAQ,IAAI,mBAAmB;AAAA,IACjE,gBAAgB,aAAa,QAAQ,IAAI,4BAA4B,KAAK;AAAA,IAC1E,+BAA+B,aAAa,QAAQ,IAAI,8CAA8C,KAAK;AAAA,IAC3G,mBAAmB,UAAU;AAAA,IAC7B,yBAAyB,UAAU;AAAA,IACnC,eAAe,YAAY,QAAQ,IAAI,2BAA2B,gCAAgC,CAAC;AAAA,IACnG,mBAAmB,YAAY,QAAQ,IAAI,gCAAgC,qCAAqC,CAAC;AAAA,IACjH,oBAAoB,YAAY,QAAQ,IAAI,iCAAiC,sCAAsC,CAAC;AAAA,EACtH;AACF;AAcO,SAAS,yBACd,OACA,YACA,QACS;AACT,QAAM,QAAQ,MAAM,YAAY;AAChC,MAAI,OAAO,kBAAkB,KAAK,CAAC,YAAY,MAAM,SAAS,OAAO,CAAC,EAAG,QAAO;AAChF,MAAI,CAAC,WAAY,QAAO;AACxB,QAAM,SAAS,OAAO,0BAA0B,WAAW,KAAK,EAAE,YAAY,CAAC;AAC/E,MAAI,CAAC,MAAM,QAAQ,MAAM,KAAK,CAAC,OAAO,OAAQ,QAAO;AACrD,SAAO,OAAO,KAAK,CAAC,YAAY,MAAM,SAAS,OAAO,CAAC;AACzD;AAaO,SAAS,8BAAsC;AACpD,SAAO,YAAY,QAAQ,IAAI,mBAAmB,iCAAiC,CAAC;AACtF;",
6
6
  "names": []
7
7
  }
@@ -0,0 +1,32 @@
1
+ const MAX_CONTAINMENT_WORDS = 10;
2
+ function buildContainmentPatterns(pattern) {
3
+ if (!isWrappedContainsPattern(pattern)) return [pattern];
4
+ const term = pattern.slice(1, -1);
5
+ if (hasUnescapedWildcard(term)) return [pattern];
6
+ const words = term.split(/\s+/).filter((word) => word.length > 0);
7
+ if (words.length < 2 || words.length > MAX_CONTAINMENT_WORDS) return [pattern];
8
+ return words.map((word) => `%${word}%`);
9
+ }
10
+ function isWrappedContainsPattern(pattern) {
11
+ if (pattern.length < 2) return false;
12
+ if (!pattern.startsWith("%") || !pattern.endsWith("%")) return false;
13
+ return countTrailingBackslashes(pattern, pattern.length - 2) % 2 === 0;
14
+ }
15
+ function hasUnescapedWildcard(term) {
16
+ for (let index = 0; index < term.length; index += 1) {
17
+ const char = term[index];
18
+ if (char !== "%" && char !== "_") continue;
19
+ if (countTrailingBackslashes(term, index - 1) % 2 === 0) return true;
20
+ }
21
+ return false;
22
+ }
23
+ function countTrailingBackslashes(value, fromIndex) {
24
+ let count = 0;
25
+ for (let index = fromIndex; index >= 0 && value[index] === "\\"; index -= 1) count += 1;
26
+ return count;
27
+ }
28
+ export {
29
+ MAX_CONTAINMENT_WORDS,
30
+ buildContainmentPatterns
31
+ };
32
+ //# sourceMappingURL=containment.js.map
@@ -0,0 +1,7 @@
1
+ {
2
+ "version": 3,
3
+ "sources": ["../../../src/lib/search/containment.ts"],
4
+ "sourcesContent": ["/**\n * Splits a `contains` like/ilike pattern into one pattern per whitespace-separated word, so a\n * plaintext column taken off the hashed-token path keeps the word-order-independent matching the\n * token index provided.\n *\n * The token path matches a value when it carries EVERY token of the term, in any order and with\n * anything in between: `?search=Warehouse 1757` matches `Warehouse A 1757`. A single verbatim\n * `name ILIKE '%Warehouse 1757%'` is literal substring containment, so the `A ` sitting between the\n * two words defeats it \u2014 that is a capability every list grid has today, and #5803's fix must not\n * take it away (`TC-RESO-009` pins it).\n *\n * ANDing one containment predicate per word reproduces the token semantics exactly on a column the\n * engine can read, without the token path's two lossy steps: nothing is dropped for being shorter\n * than `minTokenLength` (`?search=08` filters instead of matching every row) and nothing is split on\n * non-alphanumerics (`?search=2026-08` no longer collapses onto `2026-01`).\n *\n * Splitting is deliberately narrow \u2014 a pattern is only split when it is unambiguously the\n * `%term%` shape `buildIlikeTerm(value, 'contains')` produces:\n *\n * - it opens and closes with a wildcard `%` (a trailing `\\%` is an escaped literal, not a wildcard);\n * - the term between them carries no unescaped `%` or `_`, so a hand-built structured pattern such\n * as `%a% b%` is left exactly as the caller wrote it;\n * - the term holds at least two words.\n *\n * Anything else returns the input unchanged as a single pattern, which is the caller's existing\n * behavior.\n *\n * The split is also capped at {@link MAX_CONTAINMENT_WORDS} words. Most list routes declare\n * `search` as an unbounded string, this repository ships no trigram index for the resulting\n * `ILIKE`, and the hybrid engine's `$or` groups multiply the split across every leaf \u2014 so an\n * attacker-controlled term with thousands of words would otherwise compile into thousands of\n * sequential-scan predicates from one request. A term at or under the cap keeps the per-word AND\n * semantics; over the cap it falls back to the single verbatim pattern, which is bounded and was\n * this helper's own behavior before the split existed.\n */\nexport const MAX_CONTAINMENT_WORDS = 10\n\nexport function buildContainmentPatterns(pattern: string): string[] {\n if (!isWrappedContainsPattern(pattern)) return [pattern]\n const term = pattern.slice(1, -1)\n if (hasUnescapedWildcard(term)) return [pattern]\n const words = term.split(/\\s+/).filter((word) => word.length > 0)\n if (words.length < 2 || words.length > MAX_CONTAINMENT_WORDS) return [pattern]\n return words.map((word) => `%${word}%`)\n}\n\nfunction isWrappedContainsPattern(pattern: string): boolean {\n if (pattern.length < 2) return false\n if (!pattern.startsWith('%') || !pattern.endsWith('%')) return false\n return countTrailingBackslashes(pattern, pattern.length - 2) % 2 === 0\n}\n\nfunction hasUnescapedWildcard(term: string): boolean {\n for (let index = 0; index < term.length; index += 1) {\n const char = term[index]\n if (char !== '%' && char !== '_') continue\n if (countTrailingBackslashes(term, index - 1) % 2 === 0) return true\n }\n return false\n}\n\nfunction countTrailingBackslashes(value: string, fromIndex: number): number {\n let count = 0\n for (let index = fromIndex; index >= 0 && value[index] === '\\\\'; index -= 1) count += 1\n return count\n}\n"],
5
+ "mappings": "AAmCO,MAAM,wBAAwB;AAE9B,SAAS,yBAAyB,SAA2B;AAClE,MAAI,CAAC,yBAAyB,OAAO,EAAG,QAAO,CAAC,OAAO;AACvD,QAAM,OAAO,QAAQ,MAAM,GAAG,EAAE;AAChC,MAAI,qBAAqB,IAAI,EAAG,QAAO,CAAC,OAAO;AAC/C,QAAM,QAAQ,KAAK,MAAM,KAAK,EAAE,OAAO,CAAC,SAAS,KAAK,SAAS,CAAC;AAChE,MAAI,MAAM,SAAS,KAAK,MAAM,SAAS,sBAAuB,QAAO,CAAC,OAAO;AAC7E,SAAO,MAAM,IAAI,CAAC,SAAS,IAAI,IAAI,GAAG;AACxC;AAEA,SAAS,yBAAyB,SAA0B;AAC1D,MAAI,QAAQ,SAAS,EAAG,QAAO;AAC/B,MAAI,CAAC,QAAQ,WAAW,GAAG,KAAK,CAAC,QAAQ,SAAS,GAAG,EAAG,QAAO;AAC/D,SAAO,yBAAyB,SAAS,QAAQ,SAAS,CAAC,IAAI,MAAM;AACvE;AAEA,SAAS,qBAAqB,MAAuB;AACnD,WAAS,QAAQ,GAAG,QAAQ,KAAK,QAAQ,SAAS,GAAG;AACnD,UAAM,OAAO,KAAK,KAAK;AACvB,QAAI,SAAS,OAAO,SAAS,IAAK;AAClC,QAAI,yBAAyB,MAAM,QAAQ,CAAC,IAAI,MAAM,EAAG,QAAO;AAAA,EAClE;AACA,SAAO;AACT;AAEA,SAAS,yBAAyB,OAAe,WAA2B;AAC1E,MAAI,QAAQ;AACZ,WAAS,QAAQ,WAAW,SAAS,KAAK,MAAM,KAAK,MAAM,MAAM,SAAS,EAAG,UAAS;AACtF,SAAO;AACT;",
6
+ "names": []
7
+ }
@@ -1,4 +1,4 @@
1
- const APP_VERSION = "0.8.1-develop.7294.1.0ef99fc518";
1
+ const APP_VERSION = "0.8.1-develop.7296.1.2111d779db";
2
2
  const appVersion = APP_VERSION;
3
3
  export {
4
4
  APP_VERSION,
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "version": 3,
3
3
  "sources": ["../../src/lib/version.ts"],
4
- "sourcesContent": ["// Build-time generated version\nexport const APP_VERSION = '0.8.1-develop.7294.1.0ef99fc518';\nexport const appVersion = APP_VERSION;\n"],
4
+ "sourcesContent": ["// Build-time generated version\nexport const APP_VERSION = '0.8.1-develop.7296.1.2111d779db';\nexport const appVersion = APP_VERSION;\n"],
5
5
  "mappings": "AACO,MAAM,cAAc;AACpB,MAAM,aAAa;",
6
6
  "names": []
7
7
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@open-mercato/shared",
3
- "version": "0.8.1-develop.7294.1.0ef99fc518",
3
+ "version": "0.8.1-develop.7296.1.2111d779db",
4
4
  "license": "MIT",
5
5
  "type": "module",
6
6
  "main": "./dist/index.js",
@@ -113,7 +113,7 @@
113
113
  "@mikro-orm/core": "^7.1.14",
114
114
  "@mikro-orm/decorators": "^7.1.14",
115
115
  "@mikro-orm/postgresql": "^7.1.14",
116
- "@open-mercato/cache": "0.8.1-develop.7294.1.0ef99fc518",
116
+ "@open-mercato/cache": "0.8.1-develop.7296.1.2111d779db",
117
117
  "@types/html-to-text": "^9.0.4",
118
118
  "@types/sanitize-html": "^2.16.1",
119
119
  "dotenv": "^17.4.2",
@@ -287,7 +287,13 @@ describe('BasicQueryEngine (Kysely)', () => {
287
287
  })
288
288
  expect(hasTenantFilter).toBe(true)
289
289
  const baseCall = fakeDb._calls.find((b: any) => b._ops.table === 'users')
290
- const hasCfOrder = baseCall._ops.orderBys.some((o: any) => o[0] === 'cf_vip')
290
+ // The sort rides its own dedicated `__sort` alias, not the jsonb projection
291
+ // alias `cf_vip` — see the dedicated-alias test above (#5674).
292
+ const hasCfOrder = baseCall._ops.orderBys.some((o: any) => {
293
+ const expr = o[0]
294
+ return typeof expr?.toOperationNode === 'function'
295
+ && JSON.stringify(expr.toOperationNode()).includes('cf_vip__sort')
296
+ })
291
297
  expect(hasCfOrder).toBe(true)
292
298
  const hasExtJoin = baseCall._ops.joins.length > 0
293
299
  expect(hasExtJoin).toBe(true)
@@ -303,13 +309,102 @@ describe('BasicQueryEngine (Kysely)', () => {
303
309
  tenantId: 't1',
304
310
  })
305
311
  const baseCall = fakeDb._calls.find((b: any) => b._ops.table === 'users')
306
- expect(baseCall._ops.orderBys).toContainEqual(['cf_vip', 'asc'])
312
+ // The sort rides its own dedicated scalar alias, never the jsonb projection
313
+ // alias — ordering by `to_jsonb(...)` compares arrays after every scalar
314
+ // string regardless of contents (#5674). Trailing entry is the stable `id`
315
+ // tiebreak appended when the sort didn't already end on `id`.
316
+ expect(baseCall._ops.orderBys).toHaveLength(2)
317
+ const [sortExpr] = baseCall._ops.orderBys[0]
318
+ expect(JSON.stringify(sortExpr.toOperationNode())).toContain('cf_vip__sort')
319
+ expect(baseCall._ops.orderBys[1]).toEqual(['users.id', 'asc'])
307
320
  // Ordering by an alias the query never selected is a Postgres 42703, so the
308
321
  // sort has to bring its own projection and joins along (#5521).
309
- expect(selectAliases(baseCall)).toContain('cf_vip')
322
+ expect(selectAliases(baseCall)).toContain('cf_vip__sort')
310
323
  expect(baseCall._ops.joins.length).toBeGreaterThan(0)
311
324
  })
312
325
 
326
+ test('a numeric-kind cf sort casts to numeric instead of ordering as text (#5674)', async () => {
327
+ const fakeDb = createFakeKysely({
328
+ custom_field_defs: [
329
+ { key: 'rank', entity_id: 'auth:user', is_active: true, config_json: '{}', kind: 'float' },
330
+ ],
331
+ })
332
+ const engine = new BasicQueryEngine({} as any, () => fakeDb as any)
333
+ await engine.query('auth:user', {
334
+ fields: ['id', 'email'],
335
+ sort: [{ field: 'cf:rank', dir: SortDir.Asc }],
336
+ organizationId: '1',
337
+ tenantId: 't1',
338
+ })
339
+ const baseCall = fakeDb._calls.find((b: any) => b._ops.table === 'users')
340
+ // The cast lives on the dedicated sort projection, not the ORDER BY clause,
341
+ // which just references that projection's alias by name.
342
+ const sortSelect = baseCall._ops.selects.find((s: any) => String(s?.alias ?? '') === 'cf_rank__sort')
343
+ expect(sortSelect).toBeTruthy()
344
+ const serialized = JSON.stringify(sortSelect.toOperationNode())
345
+ expect(serialized).toContain('::numeric')
346
+ expect(serialized).toContain('value_float')
347
+ expect(baseCall._ops.orderBys).toHaveLength(2)
348
+ const [orderExpr] = baseCall._ops.orderBys[0]
349
+ expect(JSON.stringify(orderExpr.toOperationNode())).toContain('NULLS LAST')
350
+ expect(baseCall._ops.orderBys[1]).toEqual(['users.id', 'asc'])
351
+ })
352
+
353
+ test('an encrypted base sort combined with a cf: sort still orders by the cf value, and the __sort alias never leaks into returned rows (#5674)', async () => {
354
+ const fakeDb = createFakeKysely({
355
+ users: [
356
+ // '1' and '2' decrypt to the same email — only the cf:vip tiebreak can
357
+ // put them in the right relative order. '3' decrypts to a later email
358
+ // so it sorts last regardless of its cf:vip value.
359
+ { id: '1', tenant_id: 't1', organization_id: 'org1', email: 'cipher-1', cf_vip__sort: 'b' },
360
+ { id: '2', tenant_id: 't1', organization_id: 'org1', email: 'cipher-2', cf_vip__sort: 'a' },
361
+ { id: '3', tenant_id: 't1', organization_id: 'org1', email: 'cipher-3', cf_vip__sort: 'z' },
362
+ ],
363
+ 'information_schema.columns': [
364
+ { table_name: 'users', column_name: 'id' },
365
+ { table_name: 'users', column_name: 'tenant_id' },
366
+ { table_name: 'users', column_name: 'organization_id' },
367
+ { table_name: 'users', column_name: 'deleted_at' },
368
+ { table_name: 'users', column_name: 'email' },
369
+ ],
370
+ })
371
+ const emailById: Record<string, string> = {
372
+ '1': 'dup@example.com',
373
+ '2': 'dup@example.com',
374
+ '3': 'zzz@example.com',
375
+ }
376
+ const engine = new BasicQueryEngine(
377
+ {} as any,
378
+ () => fakeDb as any,
379
+ () => ({
380
+ isEnabled: () => true,
381
+ getEncryptedFieldNames: async () => ['email'],
382
+ decryptEntityPayload: async (_entityId, payload) => ({
383
+ email: emailById[String(payload.id)],
384
+ }),
385
+ }),
386
+ )
387
+
388
+ const result = await engine.query('auth:user', {
389
+ tenantId: 't1',
390
+ organizationId: 'org1',
391
+ fields: ['id', 'email'],
392
+ sort: [{ field: 'email', dir: SortDir.Asc }, { field: 'cf:vip', dir: SortDir.Asc }],
393
+ page: { page: 1, pageSize: 3 },
394
+ })
395
+
396
+ // Before the fix, `sortRowsInMemory` read `cf:vip` through candidates that
397
+ // never included the dedicated `cf_vip__sort` projection alias, so the value
398
+ // came back `undefined` and the cf: sort silently dropped out of the
399
+ // ordering — '1' and '2' would then only tie-break by `id`.
400
+ expect(result.items.map((item: any) => item.id)).toEqual(['2', '1', '3'])
401
+ // The synthetic sort alias is internal-only — it must never leak into a
402
+ // returned row as a phantom custom field `vip__sort` (#5674 review).
403
+ for (const item of result.items) {
404
+ expect(item).not.toHaveProperty('cf_vip__sort')
405
+ }
406
+ })
407
+
313
408
  test('a cf sort that resolves to no definition is dropped, not ordered by', async () => {
314
409
  const fakeDb = createFakeKysely()
315
410
  const engine = new BasicQueryEngine({} as any, () => fakeDb as any)
@@ -322,8 +417,9 @@ describe('BasicQueryEngine (Kysely)', () => {
322
417
  const baseCall = fakeDb._calls.find((b: any) => b._ops.table === 'users')
323
418
  // Dropping an unresolvable sort is what the base-column branch already does;
324
419
  // the alternative here was an ORDER BY over a column that is never selected.
420
+ // No cf sort survived, so there is nothing to tiebreak either.
325
421
  expect(baseCall._ops.orderBys).toEqual([])
326
- expect(selectAliases(baseCall)).not.toContain('cf_no_such_key')
422
+ expect(selectAliases(baseCall)).not.toContain('cf_no_such_key__sort')
327
423
  })
328
424
 
329
425
  test('customFieldSources join additional profiles for custom fields', async () => {
@@ -1041,7 +1137,12 @@ describe('BasicQueryEngine (Kysely)', () => {
1041
1137
  })
1042
1138
 
1043
1139
  const baseCall = fakeDb._calls.find((call: any) => call._ops.table === 'customer_entities')
1044
- expect(baseCall._ops.orderBys).toEqual([['customer_entities.display_name', 'asc']])
1140
+ // A trailing `id` tiebreak is appended when the sort didn't already end on
1141
+ // `id`, so ties/NULLs don't reorder arbitrarily across pages (#5674).
1142
+ expect(baseCall._ops.orderBys).toEqual([
1143
+ ['customer_entities.display_name', 'asc'],
1144
+ ['customer_entities.id', 'asc'],
1145
+ ])
1045
1146
  expect(baseCall._ops.limits).toBe(10)
1046
1147
  expect(baseCall._ops.offsets).toBe(10)
1047
1148
  })
@@ -1302,9 +1403,9 @@ describe('BasicQueryEngine entity-extension joins', () => {
1302
1403
  })
1303
1404
 
1304
1405
  describe('BasicQueryEngine like/ilike routing by column encryption', () => {
1305
- // The gate is opt-in: OM_SEARCH_USE_ILIKE_FOR_NON_ENCRYPTED_FIELDS defaults to false and the
1306
- // legacy rewrite-everything behavior stays. These cases flip it on; the last one pins the
1307
- // default off.
1406
+ // The gate stays off by default per #5383, so these cases opt in explicitly to pin the
1407
+ // declared-ILIKE behavior a deployment gets by setting the switch; the last one flips it back
1408
+ // off and pins the legacy rewrite-everything behavior that is the shipped default.
1308
1409
  beforeEach(() => {
1309
1410
  process.env.OM_SEARCH_USE_ILIKE_FOR_NON_ENCRYPTED_FIELDS = 'true'
1310
1411
  })
@@ -1327,7 +1428,15 @@ describe('BasicQueryEngine like/ilike routing by column encryption', () => {
1327
1428
  ],
1328
1429
  })
1329
1430
 
1330
- test('a plaintext base column keeps exact SQL ILIKE even when tokens are available', async () => {
1431
+ const ilikePatternsFor = (fakeDb: any, table: string, column: string): string[] => {
1432
+ const baseCall = fakeDb._calls.find((b: any) => b._ops.table === table)
1433
+ if (!baseCall) return []
1434
+ return baseCall._ops.wheres
1435
+ .filter((w: any) => Array.isArray(w) && String(w[0]).includes(column) && w[1] === 'ilike')
1436
+ .map((w: any) => String(w[2]))
1437
+ }
1438
+
1439
+ test('a plaintext base column runs as SQL ILIKE even when tokens are available', async () => {
1331
1440
  const fakeDb = fakeDbWithTokens()
1332
1441
  const engine = new BasicQueryEngine(
1333
1442
  {} as any,
@@ -1344,12 +1453,33 @@ describe('BasicQueryEngine like/ilike routing by column encryption', () => {
1344
1453
  })
1345
1454
 
1346
1455
  expect(applySearchTokensSpy).not.toHaveBeenCalled()
1347
- const baseCall = fakeDb._calls.find((b: any) => b._ops.table === 'customer_entities')
1348
- expect(baseCall).toBeTruthy()
1349
- const ilikeWhere = baseCall._ops.wheres.some(
1350
- (w: any) => Array.isArray(w) && String(w[0]).includes('display_name') && w[1] === 'ilike' && w[2] === '%ZK 1/2026%',
1456
+ // Multi-word terms are ANDed per word rather than matched as one literal substring -- see the
1457
+ // TC-RESO-009 case below for why that distinction is the whole point of the reroute.
1458
+ expect(ilikePatternsFor(fakeDb, 'customer_entities', 'display_name')).toEqual(['%ZK%', '%1/2026%'])
1459
+ })
1460
+
1461
+ // #5803 regression guard, reported by CI on the first shape of this fix. The token subquery
1462
+ // matched a value carrying EVERY token in any order with anything in between, so
1463
+ // `?search=Warehouse <stamp>` matched `Warehouse A <stamp>`; TC-RESO-009 pins that as required
1464
+ // behavior. One verbatim `ILIKE '%Warehouse <stamp>%'` would not match -- the `A ` sits between
1465
+ // the words -- so the reroute has to AND one containment predicate per word to be a fix rather
1466
+ // than a trade of one broken search for another.
1467
+ test('a multi-word term matches words in order-independent positions, as the token path did', async () => {
1468
+ const fakeDb = fakeDbWithTokens()
1469
+ const engine = new BasicQueryEngine(
1470
+ {} as any,
1471
+ () => fakeDb as any,
1472
+ () => ({ getEncryptedFieldNames: async () => [] }) as any,
1351
1473
  )
1352
- expect(ilikeWhere).toBe(true)
1474
+
1475
+ await engine.query('customers:customer_entity', {
1476
+ tenantId: 't1',
1477
+ fields: ['id'],
1478
+ filters: { display_name: { $ilike: '%Warehouse 1757%' } },
1479
+ page: { page: 1, pageSize: 10 },
1480
+ })
1481
+
1482
+ expect(ilikePatternsFor(fakeDb, 'customer_entities', 'display_name')).toEqual(['%Warehouse%', '%1757%'])
1353
1483
  })
1354
1484
 
1355
1485
  test('an encrypted base column still routes through search tokens', async () => {
@@ -1433,8 +1563,70 @@ describe('BasicQueryEngine like/ilike routing by column encryption', () => {
1433
1563
  expect(applySearchTokensSpy).toHaveBeenCalled()
1434
1564
  })
1435
1565
 
1436
- test('with the flag off (default) the token rewrite is kept even for plaintext columns', () => {
1437
- delete process.env.OM_SEARCH_USE_ILIKE_FOR_NON_ENCRYPTED_FIELDS
1566
+ // #5803: the reported regression. `2026-08` and `2026-01` tokenize to the identical set
1567
+ // {202, 2026} because the fragment that tells them apart is shorter than minTokenLength, so the
1568
+ // token rewrite answered a search for one period with the other one. The declared predicate has
1569
+ // to reach SQL verbatim for the exact row to come back at all.
1570
+ test('a search term whose distinguishing fragment is dropped by the tokenizer still reaches SQL verbatim', async () => {
1571
+ const fakeDb = createFakeKysely({
1572
+ accounting_periods: [],
1573
+ search_tokens: [{ one: 1 }],
1574
+ 'information_schema.tables': [{ table_name: 'search_tokens' }],
1575
+ 'information_schema.columns': [
1576
+ { table_name: 'accounting_periods', column_name: 'tenant_id' },
1577
+ { table_name: 'accounting_periods', column_name: 'period_code' },
1578
+ ],
1579
+ })
1580
+ const engine = new BasicQueryEngine(
1581
+ {} as any,
1582
+ () => fakeDb as any,
1583
+ () => ({ getEncryptedFieldNames: async () => [] }) as any,
1584
+ )
1585
+ const applySearchTokensSpy = jest.spyOn(engine as any, 'applySearchTokens')
1586
+
1587
+ await engine.query('accounting:accounting_period', {
1588
+ tenantId: 't1',
1589
+ fields: ['id'],
1590
+ filters: { period_code: { $ilike: '%2026-08%' } },
1591
+ page: { page: 1, pageSize: 10 },
1592
+ })
1593
+
1594
+ expect(applySearchTokensSpy).not.toHaveBeenCalled()
1595
+ const baseCall = fakeDb._calls.find((b: any) => b._ops.table === 'accounting_periods')
1596
+ expect(baseCall).toBeTruthy()
1597
+ const ilikeWhere = baseCall._ops.wheres.some(
1598
+ (w: any) => Array.isArray(w) && String(w[0]).includes('period_code') && w[1] === 'ilike' && w[2] === '%2026-08%',
1599
+ )
1600
+ expect(ilikeWhere).toBe(true)
1601
+ })
1602
+
1603
+ // #5803: `08` tokenizes to nothing at all, and the token path answered that with "no predicate"
1604
+ // -- every row in the table. A filter that matches everything is never the honest reading of a
1605
+ // declared containment predicate.
1606
+ test('a term too short to tokenize filters instead of being dropped', async () => {
1607
+ const fakeDb = fakeDbWithTokens()
1608
+ const engine = new BasicQueryEngine(
1609
+ {} as any,
1610
+ () => fakeDb as any,
1611
+ () => ({ getEncryptedFieldNames: async () => [] }) as any,
1612
+ )
1613
+
1614
+ await engine.query('customers:customer_entity', {
1615
+ tenantId: 't1',
1616
+ fields: ['id'],
1617
+ filters: { display_name: { $ilike: '%08%' } },
1618
+ page: { page: 1, pageSize: 10 },
1619
+ })
1620
+
1621
+ const baseCall = fakeDb._calls.find((b: any) => b._ops.table === 'customer_entities')
1622
+ const ilikeWhere = baseCall._ops.wheres.some(
1623
+ (w: any) => Array.isArray(w) && String(w[0]).includes('display_name') && w[1] === 'ilike' && w[2] === '%08%',
1624
+ )
1625
+ expect(ilikeWhere).toBe(true)
1626
+ })
1627
+
1628
+ test('with the flag explicitly off the token rewrite is kept even for plaintext columns', () => {
1629
+ process.env.OM_SEARCH_USE_ILIKE_FOR_NON_ENCRYPTED_FIELDS = 'false'
1438
1630
  const fakeDb = fakeDbWithTokens()
1439
1631
  const engine = new BasicQueryEngine(
1440
1632
  {} as any,
@@ -20,7 +20,14 @@ export function fieldNameCandidates(field: string): string[] {
20
20
  const raw = String(field || '').trim()
21
21
  if (!raw) return []
22
22
  const candidates = [raw, toSnakeCase(raw), toCamelCase(raw)]
23
- if (raw.startsWith('cf:')) candidates.push(raw.replace(/[^a-zA-Z0-9_]/g, '_'))
23
+ if (raw.startsWith('cf:')) {
24
+ const sanitized = raw.replace(/[^a-zA-Z0-9_]/g, '_')
25
+ // The ORM query engine's plaintext-sort candidate scan projects a `cf:`
26
+ // sort key only under its dedicated `<alias>__sort` projection alias, never
27
+ // the plain jsonb-aggregate alias — so the in-memory sort needs that exact
28
+ // name to find the value on the phase-1 candidate rows (#5674 review).
29
+ candidates.push(sanitized, `${sanitized}__sort`)
30
+ }
24
31
  return Array.from(new Set(candidates))
25
32
  }
26
33