@volter/twin-algolia 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,150 @@
1
+ // Algolia FILTER + FACET evaluator — net-new, PORTED CONCEPTUALLY (not copied) from
2
+ // pinecone-filter.ts's discipline (a real interpreter over real stored data, never a name/
3
+ // keyword match) — the SYNTAX is entirely different because Algolia's is its own: `filters` is
4
+ // a STRING that must be tokenized (Mongo-style JSON it is not), and `facetFilters` arrives as a
5
+ // nested array structure (outer array = AND, an inner array = OR) that needs no string parsing
6
+ // at all — two genuinely different evaluators sharing one file because they gate the same
7
+ // `search`/`browse`/`delete`-style operations together (mirrors pinecone-filter.ts's `matchesFilter`
8
+ // covering both query and delete-by-filter).
9
+ //
10
+ // Modeled `filters` grammar (v1, honest cut — see spec-sources.json): `field OP value` numeric
11
+ // comparisons (`> >= < <= = !=`, value a number literal), `field:value` facet/string equality
12
+ // (quoted or bare value), `NOT field:value` facet negation, and `AND`/`OR` combinators (case-
13
+ // sensitive keywords, exactly like the real product) with AND binding tighter than OR (real
14
+ // Algolia operator precedence) — evaluated LEFT-TO-RIGHT across OR-separated AND-groups.
15
+ // Parenthesized precedence override, `_tags`/`tag:` shorthand, and geo filters are NOT modeled
16
+ // (todo: `filters.parentheses`/`filters.tag_shorthand`/`filters.geo` — an honest v1 cut, not an
17
+ // oversight).
18
+ export type FacetFilterClause = string | string[];
19
+ export type FacetFilters = FacetFilterClause[];
20
+
21
+ function parseFacetToken(token: string): { field: string; value: string; negate: boolean } | undefined {
22
+ const idx = token.indexOf(':');
23
+ if (idx < 0) return undefined;
24
+ const field = token.slice(0, idx).trim();
25
+ let value = token.slice(idx + 1).trim();
26
+ let negate = false;
27
+ if (value.startsWith('-')) {
28
+ negate = true;
29
+ value = value.slice(1);
30
+ }
31
+ if ((value.startsWith('"') && value.endsWith('"')) || (value.startsWith("'") && value.endsWith("'"))) {
32
+ value = value.slice(1, -1);
33
+ }
34
+ return { field, value, negate };
35
+ }
36
+
37
+ function facetValueMatches(recordValue: unknown, wanted: string): boolean {
38
+ if (Array.isArray(recordValue)) return recordValue.some((v) => String(v) === wanted);
39
+ if (recordValue === undefined || recordValue === null) return false;
40
+ return String(recordValue) === wanted;
41
+ }
42
+
43
+ /** Evaluate a single `facetFilters` clause (a bare "field:value" string, or an OR-array of them)
44
+ * against a record. An absent/empty facetFilters matches everything. */
45
+ export function matchesFacetFilters(record: Record<string, unknown>, facetFilters: FacetFilters | undefined | null): boolean {
46
+ if (!facetFilters || facetFilters.length === 0) return true;
47
+ for (const clause of facetFilters) {
48
+ const items = Array.isArray(clause) ? clause : [clause];
49
+ // outer array element: AND with the rest; an array-of-strings ELEMENT is itself an OR group.
50
+ const orMatch = items.some((token) => {
51
+ const parsed = parseFacetToken(token);
52
+ if (!parsed) return false;
53
+ const matched = facetValueMatches(record[parsed.field], parsed.value);
54
+ return parsed.negate ? !matched : matched;
55
+ });
56
+ if (!orMatch) return false;
57
+ }
58
+ return true;
59
+ }
60
+
61
+ // ── `filters` string grammar ────────────────────────────────────────────────────────────────
62
+ type Comparison = { field: string; op: '>' | '>=' | '<' | '<=' | '=' | '!='; value: number };
63
+ type FacetClause = { field: string; value: string; negate: boolean };
64
+ type Clause = { kind: 'compare'; c: Comparison } | { kind: 'facet'; c: FacetClause };
65
+
66
+ const COMPARE_RE = /^(\S+)\s*(>=|<=|!=|>|<|=)\s*(-?\d+(?:\.\d+)?)$/;
67
+
68
+ function parseClause(raw: string): Clause | undefined {
69
+ const trimmed = raw.trim();
70
+ if (trimmed.length === 0) return undefined;
71
+ let text = trimmed;
72
+ let negate = false;
73
+ if (text.startsWith('NOT ')) {
74
+ negate = true;
75
+ text = text.slice(4).trim();
76
+ }
77
+ const cmp = COMPARE_RE.exec(text);
78
+ if (cmp) {
79
+ const [, field, op, num] = cmp;
80
+ return { kind: 'compare', c: { field: field!, op: (negate ? negateOp(op as Comparison['op']) : op) as Comparison['op'], value: Number(num) } };
81
+ }
82
+ const facet = parseFacetToken(text);
83
+ if (facet) return { kind: 'facet', c: { field: facet.field, value: facet.value, negate: negate !== facet.negate } };
84
+ return undefined;
85
+ }
86
+
87
+ function negateOp(op: Comparison['op']): Comparison['op'] {
88
+ const table: Record<Comparison['op'], Comparison['op']> = { '>': '<=', '>=': '<', '<': '>=', '<=': '>', '=': '!=', '!=': '=' };
89
+ return table[op];
90
+ }
91
+
92
+ function evalClause(record: Record<string, unknown>, clause: Clause): boolean {
93
+ if (clause.kind === 'facet') {
94
+ const matched = facetValueMatches(record[clause.c.field], clause.c.value);
95
+ return clause.c.negate ? !matched : matched;
96
+ }
97
+ const raw = record[clause.c.field];
98
+ if (typeof raw !== 'number') return false;
99
+ switch (clause.c.op) {
100
+ case '>':
101
+ return raw > clause.c.value;
102
+ case '>=':
103
+ return raw >= clause.c.value;
104
+ case '<':
105
+ return raw < clause.c.value;
106
+ case '<=':
107
+ return raw <= clause.c.value;
108
+ case '=':
109
+ return raw === clause.c.value;
110
+ case '!=':
111
+ return raw !== clause.c.value;
112
+ }
113
+ }
114
+
115
+ /** Evaluate an Algolia `filters` STRING against a stored record: AND binds tighter than OR,
116
+ * evaluated as OR-of-AND-groups (no parenthesized precedence override — v1 honest cut). An
117
+ * absent/empty/whitespace-only filter string matches everything. */
118
+ export function matchesAlgoliaFilters(record: Record<string, unknown>, filters: string | undefined | null): boolean {
119
+ if (!filters || filters.trim().length === 0) return true;
120
+ const orGroups = filters.split(/\s+OR\s+/);
121
+ return orGroups.some((group) =>
122
+ group
123
+ .split(/\s+AND\s+/)
124
+ .map((raw) => parseClause(raw))
125
+ .every((clause) => (clause ? evalClause(record, clause) : true)),
126
+ );
127
+ }
128
+
129
+ /**
130
+ * REAL facet-count aggregation over an already-matched-and-filtered record set (conjunctive —
131
+ * counts reflect the final hit set, not a per-facet disjunctive relaxation — disjunctive facet
132
+ * counting is `facets.disjunctive_facets`, todo). Multi-value (array) facet attributes count
133
+ * each element once per record — a genuine tally, not a name check.
134
+ */
135
+ export function computeFacetCounts(records: ReadonlyArray<Record<string, unknown>>, facetAttributes: readonly string[]): Record<string, Record<string, number>> {
136
+ const out: Record<string, Record<string, number>> = {};
137
+ for (const attr of facetAttributes) {
138
+ const counts: Record<string, number> = {};
139
+ for (const r of records) {
140
+ const v = r[attr];
141
+ const values = Array.isArray(v) ? v : v === undefined || v === null ? [] : [v];
142
+ for (const one of values) {
143
+ const key = String(one);
144
+ counts[key] = (counts[key] ?? 0) + 1;
145
+ }
146
+ }
147
+ out[attr] = counts;
148
+ }
149
+ return out;
150
+ }
@@ -0,0 +1,204 @@
1
+ // Algolia SEARCH ENGINE — net-new, no clone (ADDING_A_TWIN.md's "real computed op as strong
2
+ // done" pattern; the analog of pinecone-similarity.ts's real brute-force top-K, ported
3
+ // CONCEPTUALLY not copied — the shape of the problem differs completely: Algolia ranks TEXT
4
+ // records by a tie-break criteria chain, not vectors by a distance metric). This is REAL
5
+ // compute, not a stub: deterministic ASCII/Unicode tokenization, a real Damerau-Levenshtein
6
+ // edit-distance typo matcher (<=2), and a faithful SUBSET of Algolia's documented default
7
+ // `ranking` tie-break chain (`typo`, `words` (folded into typo-match — AND semantics, see
8
+ // below), `proximity`, `attribute`, `exact`, `custom`) — every criterion below is GENUINELY
9
+ // COMPUTED against the stored record values and searchableAttributes configuration, not a
10
+ // name/keyword match or a marker.
11
+ //
12
+ // WHAT IS NOT YET RANKED (`algolia.search.ranking_criteria_coverage`, todo): the real product
13
+ // ships an additional `geo`/`filters`-as-ranking-criterion slot, and on top a proprietary
14
+ // NeuralSearch/AI re-rank pass. What IS modeled (typo/proximity/attribute/exact/custom) is exact,
15
+ // deterministic, correct math over the modeled criteria — a faithful subset, not a lesser
16
+ // approximation of what it does model (mirrors pinecone-similarity.ts's framing).
17
+ //
18
+ // MATCH SEMANTICS (v1 simplification, documented rather than silently guessed): a record
19
+ // matches a non-empty query iff there EXISTS at least one searchable attribute whose token
20
+ // list satisfies EVERY query word (exact, <=2-edit-distance typo, or same synonym group) —
21
+ // i.e. AND-across-query-words, matched WITHIN one attribute (not unioned across attributes).
22
+ // Real Algolia's default `removeWordsIfNoResults`/optional-words progressive relaxation is
23
+ // NOT modeled (see `search.optional_words` / `search.remove_words_if_no_results`, todo) — this
24
+ // twin's strict-AND match is the honest, simpler default behavior, not the full adaptive
25
+ // algorithm. Attribute VALUES are read top-level only (dot-path nested attribute addressing is
26
+ // not modeled, consistent with the v1 slice's flat-record framing used throughout this pack).
27
+ export type CustomRankingRule = { attribute: string; direction: 'asc' | 'desc' };
28
+
29
+ export type SearchRecord = Record<string, unknown> & { objectID: string };
30
+
31
+ /** Deterministic ASCII/Unicode tokenizer: lowercase, split on runs of non letter/number
32
+ * characters. This is the line `algolia.search.language_processing` (todo) will move: no
33
+ * stemming/plural-folding/CJK segmentation/language normalization yet — just a real,
34
+ * deterministic word-boundary split. */
35
+ export function tokenize(text: string): string[] {
36
+ return text.toLowerCase().match(/[\p{L}\p{N}]+/gu) ?? [];
37
+ }
38
+
39
+ /** Real Damerau-Levenshtein edit distance (insertions/deletions/substitutions/adjacent
40
+ * transpositions) — a genuine DP table, not an approximation. */
41
+ export function damerauLevenshtein(a: string, b: string): number {
42
+ const al = a.length;
43
+ const bl = b.length;
44
+ if (al === 0) return bl;
45
+ if (bl === 0) return al;
46
+ const d: number[][] = Array.from({ length: al + 1 }, () => new Array<number>(bl + 1).fill(0));
47
+ for (let i = 0; i <= al; i++) d[i]![0] = i;
48
+ for (let j = 0; j <= bl; j++) d[0]![j] = j;
49
+ for (let i = 1; i <= al; i++) {
50
+ for (let j = 1; j <= bl; j++) {
51
+ const cost = a[i - 1] === b[j - 1] ? 0 : 1;
52
+ let v = Math.min(
53
+ d[i - 1]![j]! + 1, // deletion
54
+ d[i]![j - 1]! + 1, // insertion
55
+ d[i - 1]![j - 1]! + cost, // substitution
56
+ );
57
+ if (i > 1 && j > 1 && a[i - 1] === b[j - 2] && a[i - 2] === b[j - 1]) {
58
+ v = Math.min(v, d[i - 2]![j - 2]! + cost); // adjacent transposition
59
+ }
60
+ d[i]![j] = v;
61
+ }
62
+ }
63
+ return d[al]![bl]!;
64
+ }
65
+
66
+ const MAX_TYPOS = 2;
67
+
68
+ /** All values a group of interchangeable synonym words maps to each other, lowercase. A word's
69
+ * group includes itself. Basic bidirectional `type:"synonym"` groups only — one-way/altCorrection/
70
+ * placeholder synonyms are `search.optional_words`-adjacent todos (synonyms.one_way etc). */
71
+ export function buildSynonymGroups(synonymRecords: ReadonlyArray<{ type?: string; synonyms?: readonly string[] }>): string[][] {
72
+ const groups: string[][] = [];
73
+ for (const s of synonymRecords) {
74
+ if (s.type !== 'synonym' || !Array.isArray(s.synonyms) || s.synonyms.length < 2) continue;
75
+ groups.push(s.synonyms.map((w) => w.toLowerCase()));
76
+ }
77
+ return groups;
78
+ }
79
+
80
+ function sameSynonymGroup(a: string, b: string, groups: readonly string[][]): boolean {
81
+ for (const g of groups) if (g.includes(a) && g.includes(b)) return true;
82
+ return false;
83
+ }
84
+
85
+ /** Distance between a query word and a candidate token: 0 for exact or same-synonym-group,
86
+ * else the Damerau-Levenshtein edit distance capped at MAX_TYPOS, else `undefined` (no match
87
+ * at all — the record does not satisfy this query word via this token). */
88
+ function wordDistance(queryWord: string, token: string, synonymGroups: readonly string[][]): number | undefined {
89
+ if (queryWord === token) return 0;
90
+ if (sameSynonymGroup(queryWord, token, synonymGroups)) return 0;
91
+ const d = damerauLevenshtein(queryWord, token);
92
+ return d <= MAX_TYPOS ? d : undefined;
93
+ }
94
+
95
+ /** Flatten one attribute's stored value into a token list (string, array-of-strings/values, or
96
+ * a scalar stringified) — non-string/array/number/boolean values (objects, null, undefined)
97
+ * contribute no tokens. */
98
+ function tokensForAttributeValue(value: unknown): string[] {
99
+ if (typeof value === 'string') return tokenize(value);
100
+ if (typeof value === 'number' || typeof value === 'boolean') return tokenize(String(value));
101
+ if (Array.isArray(value)) return value.flatMap((v) => tokensForAttributeValue(v));
102
+ return [];
103
+ }
104
+
105
+ type AttributeMatch = { attributeIndex: number; typoTotal: number; exactCount: number; proximity: number };
106
+
107
+ /** Try to match every query word within ONE attribute's token list; undefined if any query word
108
+ * finds no token in this attribute within MAX_TYPOS. */
109
+ function matchWithinAttribute(tokens: readonly string[], queryWords: readonly string[], synonymGroups: readonly string[][], attributeIndex: number): AttributeMatch | undefined {
110
+ if (tokens.length === 0) return undefined;
111
+ const positions: number[] = [];
112
+ let typoTotal = 0;
113
+ let exactCount = 0;
114
+ for (const qw of queryWords) {
115
+ let best: { idx: number; typos: number } | undefined;
116
+ for (let i = 0; i < tokens.length; i++) {
117
+ const d = wordDistance(qw, tokens[i]!, synonymGroups);
118
+ if (d !== undefined && (!best || d < best.typos)) best = { idx: i, typos: d };
119
+ }
120
+ if (!best) return undefined;
121
+ positions.push(best.idx);
122
+ typoTotal += best.typos;
123
+ if (best.typos === 0) exactCount++;
124
+ }
125
+ const proximity = positions.length > 0 ? Math.max(...positions) - Math.min(...positions) : 0;
126
+ return { attributeIndex, typoTotal, exactCount, proximity };
127
+ }
128
+
129
+ export type RankedHit<T extends SearchRecord = SearchRecord> = {
130
+ record: T;
131
+ typoTotal: number;
132
+ attributeIndex: number;
133
+ proximity: number;
134
+ exactCount: number;
135
+ };
136
+
137
+ /** Default searchable-attribute order when settings.searchableAttributes is unset: every
138
+ * top-level field except objectID, in a deterministic sorted order. Real Algolia treats every
139
+ * attribute as equally searchable when unconfigured (no attribute-priority distinction) — this
140
+ * twin's sorted-key stand-in only matters for tie-breaking when the caller HAS configured an
141
+ * explicit order (search.attributes_to_retrieve / settings.searchable_attributes_affects_search). */
142
+ function defaultSearchableAttributes(records: readonly SearchRecord[]): string[] {
143
+ const keys = new Set<string>();
144
+ for (const r of records) for (const k of Object.keys(r)) if (k !== 'objectID') keys.add(k);
145
+ return [...keys].sort();
146
+ }
147
+
148
+ function customRankingCompare(a: SearchRecord, b: SearchRecord, rules: readonly CustomRankingRule[]): number {
149
+ for (const rule of rules) {
150
+ const av = a[rule.attribute];
151
+ const bv = b[rule.attribute];
152
+ const an = typeof av === 'number' ? av : Number.NEGATIVE_INFINITY;
153
+ const bn = typeof bv === 'number' ? bv : Number.NEGATIVE_INFINITY;
154
+ if (an === bn) continue;
155
+ return rule.direction === 'desc' ? bn - an : an - bn;
156
+ }
157
+ return 0;
158
+ }
159
+
160
+ /**
161
+ * Match + rank records against a query — the pack's signature strong-`done`. Returns every
162
+ * MATCHING record, best-first, per the faithful-subset tie-break chain: typo -> attribute
163
+ * (earlier configured attribute wins) -> proximity -> exact -> customRanking -> objectID
164
+ * (final deterministic tie-break). An empty/whitespace query matches every record (real
165
+ * Algolia's browse-all-via-empty-query convention) with every criterion tied, so ordering then
166
+ * falls straight to customRanking / objectID.
167
+ */
168
+ export function rankRecords<T extends SearchRecord>(
169
+ records: readonly T[],
170
+ query: string,
171
+ opts: { searchableAttributes?: readonly string[]; customRanking?: readonly CustomRankingRule[]; synonymGroups?: readonly string[][] } = {},
172
+ ): RankedHit<T>[] {
173
+ const attrs = opts.searchableAttributes && opts.searchableAttributes.length > 0 ? opts.searchableAttributes : defaultSearchableAttributes(records);
174
+ const customRanking = opts.customRanking ?? [];
175
+ const synonymGroups = opts.synonymGroups ?? [];
176
+ const queryWords = tokenize(query);
177
+
178
+ const hits: RankedHit<T>[] = [];
179
+ for (const record of records) {
180
+ if (queryWords.length === 0) {
181
+ hits.push({ record, typoTotal: 0, attributeIndex: 0, proximity: 0, exactCount: 0 });
182
+ continue;
183
+ }
184
+ let best: AttributeMatch | undefined;
185
+ for (let i = 0; i < attrs.length; i++) {
186
+ const tokens = tokensForAttributeValue(record[attrs[i]!]);
187
+ const m = matchWithinAttribute(tokens, queryWords, synonymGroups, i);
188
+ if (!m) continue;
189
+ if (!best || m.typoTotal < best.typoTotal || (m.typoTotal === best.typoTotal && m.attributeIndex < best.attributeIndex)) best = m;
190
+ }
191
+ if (best) hits.push({ record, typoTotal: best.typoTotal, attributeIndex: best.attributeIndex, proximity: best.proximity, exactCount: best.exactCount });
192
+ }
193
+
194
+ hits.sort((a, b) => {
195
+ if (a.typoTotal !== b.typoTotal) return a.typoTotal - b.typoTotal;
196
+ if (a.attributeIndex !== b.attributeIndex) return a.attributeIndex - b.attributeIndex;
197
+ if (a.proximity !== b.proximity) return a.proximity - b.proximity;
198
+ if (a.exactCount !== b.exactCount) return b.exactCount - a.exactCount;
199
+ const custom = customRankingCompare(a.record, b.record, customRanking);
200
+ if (custom !== 0) return custom;
201
+ return a.record.objectID < b.record.objectID ? -1 : a.record.objectID > b.record.objectID ? 1 : 0;
202
+ });
203
+ return hits;
204
+ }
@@ -0,0 +1,43 @@
1
+ // Algolia twin HTTP server — ONE process serving every host shape a real `algoliasearch` SDK (or
2
+ // raw REST caller) might target, so an unmodified SDK works end-to-end against a single local
3
+ // process. Unlike pinecone's control/data HOST split, Algolia's REST paths already carry the
4
+ // index name (`/1/indexes/{indexName}/...`), so this server does zero host-based dispatch — the
5
+ // literal HTTP `Host` header (or, for the SDK-integration test's proxy-rewrite pattern, an
6
+ // `x-algolia-target-host` stash) is threaded through purely for `routeAlgoliaSurface`'s host-
7
+ // tolerance proof (see algolia-twin.ts's header); routing itself is 100% path-driven.
8
+ //
9
+ // Writable by default; pass readOnly to reject writes (per-operation — see algolia-twin.ts).
10
+ //
11
+ // FETCH-FIRST (runtime contract R12b): the serve path is the plain fetch below, built from the
12
+ // kernel's ONE adaptation (`createTwinFetchFromHandler`) with the host threading as its per-request
13
+ // `extras`; the server is one line of Bun.serve around that same closure.
14
+ import { serveHttp } from '@volter/world-core';
15
+ import { handleAlgoliaTwinRequest } from './algolia-twin.ts';
16
+ import { createTwinFetchFromHandler, statefulTwinManifest } from '@volter/world-core';
17
+
18
+ /** Options every Algolia-twin HTTP surface needs, independent of who owns the socket. */
19
+ export interface AlgoliaTwinFetchOptions {
20
+ root?: string;
21
+ readOnly?: boolean;
22
+ }
23
+
24
+ export function createAlgoliaTwinFetch(options: AlgoliaTwinFetchOptions = {}): (request: Request) => Promise<Response> {
25
+ return createTwinFetchFromHandler(handleAlgoliaTwinRequest, {
26
+ ...options,
27
+ manifest: statefulTwinManifest({ vendor: 'algolia', twinOf: 'Algolia hosted search', stores: 'indexes and their records (CRUD + batch), searched by a real local engine' }),
28
+ // `Headers` iteration already lower-cases every name, so the adapter's header map matches the
29
+ // one this server used to build by hand.
30
+ extras: (request, url) => ({
31
+ host: request.headers.get('x-algolia-target-host') || request.headers.get('host') || url.host,
32
+ }),
33
+ });
34
+ }
35
+
36
+ export async function createAlgoliaTwinServer(options: { root?: string; port?: number; readOnly?: boolean } = {}): Promise<{ port: number; stop: () => void }> {
37
+ const server = await serveHttp({
38
+ port: options.port ?? 0,
39
+ idleTimeout: 60,
40
+ fetch: createAlgoliaTwinFetch(options),
41
+ });
42
+ return { port: server.port ?? options.port ?? 0, stop: () => server.stop(true) };
43
+ }