@volter/twin-upstashvector 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,525 @@
1
+ // UPSTASH VECTOR METADATA FILTERING — a REAL tokenizer, recursive-descent parser and evaluator for
2
+ // the filter language documented at upstash.com/docs/vector/features/filtering.
3
+ //
4
+ // This is a genuine language implementation, not a substring match: `filter` strings arrive as
5
+ // opaque text on `/query`, `/delete` and `/update`, and a twin that "supported filtering" by
6
+ // checking `metadata[k] === v` for the one shape its own tests used would be exactly the fake
7
+ // success this repo forbids. The grammar below is parsed to an AST and the AST is evaluated
8
+ // against each candidate's metadata, so an expression the tests never anticipated still works and
9
+ // a malformed one is REJECTED with a parse error rather than silently matching everything.
10
+ //
11
+ // ── GROUNDED (2026-08-19) ─────────────────────────────────────────────────────────────────────
12
+ // upstash.com/docs/vector/features/filtering, quoted where it is quotable:
13
+ // • operators: `=`, `!=`, `<`, `>`, `<=`, `>=`, `GLOB`, `NOT GLOB`, `IN`, `NOT IN`, `CONTAINS`,
14
+ // `NOT CONTAINS`, `HAS FIELD`, `HAS NOT FIELD`, combined with `AND` / `OR`;
15
+ // • "Nested objects can be at arbitrary depths, so more than one `.` accessor can be used in the
16
+ // same identifier";
17
+ // • "individual array elements can also be filtered by referencing them with the `[]` accessor by
18
+ // their indexes", and "it is possible to index from the back using the `#` character with
19
+ // negative values" (`arr[#-1]` is the last element);
20
+ // • "The string literals can be either single or double quoted";
21
+ // • "Boolean literals are represented as `1` or `0`" — while the docs' own `!=` example uses
22
+ // `is_capital != true`, so BOTH spellings are accepted here (see `parseOperand`);
23
+ // • "`AND` will have higher precedence than `OR`" when no parentheses are given — which is why
24
+ // `parseOr` sits above `parseAnd` below rather than the two sharing one precedence level.
25
+ //
26
+ // ── WHAT IS DELIBERATELY NOT CLAIMED ──────────────────────────────────────────────────────────
27
+ // Upstash does not publish the literal text of its parse errors, and this pack has no live index to
28
+ // probe (unlike its sibling upstash, which live-probed every string). So `FilterError.message`
29
+ // is TWIN-AUTHORED and says so; what is faithful — and what the capability manifest asserts — is
30
+ // that a malformed filter is REJECTED with the vendor's documented `{error,status}` envelope at
31
+ // HTTP 400 rather than being ignored. A filter that silently matched everything would turn a typo
32
+ // into a data leak, which is the failure this module exists to prevent.
33
+
34
+ /** A filter that could not be parsed. The handler maps this to the vendor's 400 envelope. */
35
+ export class FilterError extends Error {
36
+ constructor(message: string) {
37
+ super(message);
38
+ this.name = 'FilterError';
39
+ }
40
+ }
41
+
42
+ // ─────────────────────────────────────────────────────────────────────────────────────────────
43
+ // TOKENIZER
44
+ // ─────────────────────────────────────────────────────────────────────────────────────────────
45
+
46
+ export type TokenKind = 'ident' | 'string' | 'number' | 'op' | 'word' | 'punct';
47
+ export type Token = { kind: TokenKind; value: string; pos: number };
48
+
49
+ const OPERATOR_CHARS = new Set(['=', '!', '<', '>']);
50
+ /** Reserved words. Case-INSENSITIVE in the docs' examples (`and` and `AND` both appear). */
51
+ const WORDS = new Set(['AND', 'OR', 'NOT', 'IN', 'GLOB', 'CONTAINS', 'HAS', 'FIELD']);
52
+
53
+ /**
54
+ * Split a filter string into tokens.
55
+ *
56
+ * Identifiers carry their `.`/`[]` accessors as ONE token (`geography.coordinates.latitude`,
57
+ * `major_industries[0]`, `tags[#-1]`) because the path is resolved as a unit at evaluation time;
58
+ * splitting them would force the parser to re-join them and lose the distinction between the
59
+ * accessor dot and a decimal point.
60
+ */
61
+ export function tokenizeFilter(input: string): Token[] {
62
+ const tokens: Token[] = [];
63
+ let i = 0;
64
+ while (i < input.length) {
65
+ const c = input[i]!;
66
+ if (c === ' ' || c === '\t' || c === '\n' || c === '\r') { i++; continue; }
67
+ const pos = i;
68
+ if (c === '(' || c === ')' || c === ',') { tokens.push({ kind: 'punct', value: c, pos }); i++; continue; }
69
+ if (c === "'" || c === '"') {
70
+ // A quoted string. The closing quote must match the opening one, so an apostrophe inside a
71
+ // double-quoted literal is ordinary text.
72
+ const quote = c;
73
+ i++;
74
+ let out = '';
75
+ let closed = false;
76
+ while (i < input.length) {
77
+ const ch = input[i]!;
78
+ // NO BACKSLASH ESCAPE. An earlier draft treated `\x` as an escape for `x`; §9 round 1
79
+ // showed that is both undocumented by Upstash and a source of SILENT WRONG ANSWERS:
80
+ // with it, `path = 'C:\Users'` matched a vector whose stored path was `C:Users` and did
81
+ // NOT match the one holding `C:\Users` — a confidently wrong row at HTTP 200. Upstash's
82
+ // filter language is SQLite-family (`GLOB`, `IN`, `HAS FIELD`), and no member of that
83
+ // family gives backslash a special meaning inside a string literal. A backslash is an
84
+ // ordinary character; a quote is closed only by its own kind, which is what the two
85
+ // quote styles are for.
86
+ if (ch === quote) { closed = true; i++; break; }
87
+ out += ch;
88
+ i++;
89
+ }
90
+ if (!closed) throw new FilterError(`upstashvector: unterminated string literal starting at position ${pos} in filter`);
91
+ tokens.push({ kind: 'string', value: out, pos });
92
+ continue;
93
+ }
94
+ if (OPERATOR_CHARS.has(c)) {
95
+ const two = input.slice(i, i + 2);
96
+ if (two === '!=' || two === '<=' || two === '>=') { tokens.push({ kind: 'op', value: two, pos }); i += 2; continue; }
97
+ if (c === '=' || c === '<' || c === '>') { tokens.push({ kind: 'op', value: c, pos }); i++; continue; }
98
+ throw new FilterError(`upstashvector: unexpected character '${c}' at position ${pos} in filter (did you mean '!='?)`);
99
+ }
100
+ // A number: an optional sign is handled by the operand parser, not here, so that `a>-1` still
101
+ // tokenizes as `a` `>` `-1`.
102
+ if (/[0-9]/.test(c) || (c === '-' && /[0-9.]/.test(input[i + 1] ?? ''))) {
103
+ let out = c;
104
+ i++;
105
+ while (i < input.length && /[0-9._eE+-]/.test(input[i]!)) {
106
+ // Stop before an `e+`/`e-` that is not part of an exponent, so `a=1 e` does not swallow.
107
+ const ch = input[i]!;
108
+ if ((ch === '+' || ch === '-') && !/[eE]/.test(out[out.length - 1] ?? '')) break;
109
+ out += ch;
110
+ i++;
111
+ }
112
+ if (!Number.isFinite(Number(out))) throw new FilterError(`upstashvector: '${out}' at position ${pos} is not a valid number in filter`);
113
+ tokens.push({ kind: 'number', value: out, pos });
114
+ continue;
115
+ }
116
+ if (/[A-Za-z_]/.test(c)) {
117
+ let out = '';
118
+ while (i < input.length && /[A-Za-z0-9_.[\]#-]/.test(input[i]!)) {
119
+ // `-` is legal inside a bracket subscript (`[#-1]`) but is NOT an identifier character
120
+ // otherwise — otherwise `a-b` would read as one name and a typo would become a field.
121
+ if (input[i] === '-' && !out.includes('[')) break;
122
+ if (input[i] === '-' && out.lastIndexOf('[') < out.lastIndexOf(']')) break;
123
+ out += input[i]!;
124
+ i++;
125
+ }
126
+ tokens.push({ kind: WORDS.has(out.toUpperCase()) ? 'word' : 'ident', value: out, pos });
127
+ continue;
128
+ }
129
+ throw new FilterError(`upstashvector: unexpected character '${c}' at position ${pos} in filter`);
130
+ }
131
+ return tokens;
132
+ }
133
+
134
+ // ─────────────────────────────────────────────────────────────────────────────────────────────
135
+ // AST
136
+ // ─────────────────────────────────────────────────────────────────────────────────────────────
137
+
138
+ export type FilterValue = string | number | boolean;
139
+ export type ComparisonOp = '=' | '!=' | '<' | '>' | '<=' | '>=' | 'GLOB' | 'NOT GLOB' | 'CONTAINS' | 'NOT CONTAINS';
140
+
141
+ export type FilterNode =
142
+ | { kind: 'and'; left: FilterNode; right: FilterNode }
143
+ | { kind: 'or'; left: FilterNode; right: FilterNode }
144
+ | { kind: 'compare'; path: string; op: ComparisonOp; value: FilterValue }
145
+ | { kind: 'in'; path: string; negated: boolean; values: FilterValue[] }
146
+ | { kind: 'hasField'; path: string; negated: boolean };
147
+
148
+ // ─────────────────────────────────────────────────────────────────────────────────────────────
149
+ // PARSER (recursive descent; AND binds tighter than OR, per the docs)
150
+ // ─────────────────────────────────────────────────────────────────────────────────────────────
151
+
152
+ class Parser {
153
+ private at = 0;
154
+ constructor(private readonly tokens: Token[], private readonly source: string) {}
155
+
156
+ private peek(): Token | undefined { return this.tokens[this.at]; }
157
+ private next(): Token | undefined { return this.tokens[this.at++]; }
158
+ private isWord(word: string, offset = 0): boolean {
159
+ const t = this.tokens[this.at + offset];
160
+ return t !== undefined && t.kind === 'word' && t.value.toUpperCase() === word;
161
+ }
162
+ private expectWord(word: string): void {
163
+ if (!this.isWord(word)) throw new FilterError(`upstashvector: expected ${word} in filter '${this.source}'`);
164
+ this.at++;
165
+ }
166
+
167
+ parse(): FilterNode {
168
+ if (this.tokens.length === 0) throw new FilterError('upstashvector: empty filter expression');
169
+ const node = this.parseOr();
170
+ const rest = this.peek();
171
+ if (rest !== undefined) {
172
+ throw new FilterError(`upstashvector: unexpected '${rest.value}' at position ${rest.pos} in filter '${this.source}'`);
173
+ }
174
+ // Validate every metadata PATH eagerly, at parse time.
175
+ //
176
+ // §9 round 1 found the bug this closes: a path like `a[0` (an unterminated subscript)
177
+ // tokenizes and parses fine, and only `parsePath` — called from `resolvePath` DURING
178
+ // EVALUATION — rejects it. That made the failure DATA-DEPENDENT: with rows present the
179
+ // FilterError escaped as an internal HTTP 500, and against an empty namespace the evaluator
180
+ // was never reached at all, so the same malformed filter answered 200 `[]`. A filter is either
181
+ // well-formed or it is not; that cannot depend on how many vectors happen to be stored.
182
+ validatePaths(node);
183
+ return node;
184
+ }
185
+
186
+ /** OR binds LOOSEST — "AND will have higher precedence than OR" (docs). */
187
+ private parseOr(): FilterNode {
188
+ let left = this.parseAnd();
189
+ while (this.isWord('OR')) {
190
+ this.at++;
191
+ left = { kind: 'or', left, right: this.parseAnd() };
192
+ }
193
+ return left;
194
+ }
195
+
196
+ private parseAnd(): FilterNode {
197
+ let left = this.parseUnary();
198
+ while (this.isWord('AND')) {
199
+ this.at++;
200
+ left = { kind: 'and', left, right: this.parseUnary() };
201
+ }
202
+ return left;
203
+ }
204
+
205
+ private parseUnary(): FilterNode {
206
+ const t = this.peek();
207
+ if (t === undefined) throw new FilterError(`upstashvector: unexpected end of filter '${this.source}'`);
208
+ if (t.kind === 'punct' && t.value === '(') {
209
+ this.at++;
210
+ const inner = this.parseOr();
211
+ const close = this.next();
212
+ if (close === undefined || close.value !== ')') throw new FilterError(`upstashvector: unbalanced parenthesis in filter '${this.source}'`);
213
+ return inner;
214
+ }
215
+ // `HAS FIELD x` / `HAS NOT FIELD x` — the only prefix-form operators in the language.
216
+ if (this.isWord('HAS')) {
217
+ this.at++;
218
+ let negated = false;
219
+ if (this.isWord('NOT')) { this.at++; negated = true; }
220
+ this.expectWord('FIELD');
221
+ const ident = this.next();
222
+ if (ident === undefined || ident.kind !== 'ident') {
223
+ throw new FilterError(`upstashvector: HAS ${negated ? 'NOT ' : ''}FIELD needs a metadata key in filter '${this.source}'`);
224
+ }
225
+ return { kind: 'hasField', path: ident.value, negated };
226
+ }
227
+ return this.parseComparison();
228
+ }
229
+
230
+ private parseComparison(): FilterNode {
231
+ const ident = this.next();
232
+ if (ident === undefined || ident.kind !== 'ident') {
233
+ throw new FilterError(`upstashvector: expected a metadata key but found '${ident?.value ?? 'end of input'}' in filter '${this.source}'`);
234
+ }
235
+ const path = ident.value;
236
+ let negated = false;
237
+ if (this.isWord('NOT')) { this.at++; negated = true; }
238
+ const op = this.next();
239
+ if (op === undefined) throw new FilterError(`upstashvector: '${path}' has no operator in filter '${this.source}'`);
240
+
241
+ if (op.kind === 'word' && op.value.toUpperCase() === 'IN') {
242
+ return { kind: 'in', path, negated, values: this.parseList() };
243
+ }
244
+ if (op.kind === 'word' && op.value.toUpperCase() === 'GLOB') {
245
+ return { kind: 'compare', path, op: negated ? 'NOT GLOB' : 'GLOB', value: this.parseOperand() };
246
+ }
247
+ if (op.kind === 'word' && op.value.toUpperCase() === 'CONTAINS') {
248
+ return { kind: 'compare', path, op: negated ? 'NOT CONTAINS' : 'CONTAINS', value: this.parseOperand() };
249
+ }
250
+ if (negated) {
251
+ // `NOT` only prefixes IN / GLOB / CONTAINS; `a NOT = 1` is not the language.
252
+ throw new FilterError(`upstashvector: NOT must be followed by IN, GLOB or CONTAINS in filter '${this.source}'`);
253
+ }
254
+ if (op.kind !== 'op') {
255
+ throw new FilterError(`upstashvector: '${op.value}' is not a comparison operator in filter '${this.source}'`);
256
+ }
257
+ return { kind: 'compare', path, op: op.value as ComparisonOp, value: this.parseOperand() };
258
+ }
259
+
260
+ private parseList(): FilterValue[] {
261
+ const open = this.next();
262
+ if (open === undefined || open.value !== '(') throw new FilterError(`upstashvector: IN needs a parenthesised list in filter '${this.source}'`);
263
+ const values: FilterValue[] = [];
264
+ for (;;) {
265
+ const t = this.peek();
266
+ if (t === undefined) throw new FilterError(`upstashvector: unterminated IN list in filter '${this.source}'`);
267
+ if (t.kind === 'punct' && t.value === ')') { this.at++; break; }
268
+ values.push(this.parseOperand());
269
+ const sep = this.peek();
270
+ if (sep !== undefined && sep.kind === 'punct' && sep.value === ',') { this.at++; continue; }
271
+ if (sep !== undefined && sep.kind === 'punct' && sep.value === ')') { this.at++; break; }
272
+ throw new FilterError(`upstashvector: expected ',' or ')' in IN list in filter '${this.source}'`);
273
+ }
274
+ if (values.length === 0) throw new FilterError(`upstashvector: empty IN list in filter '${this.source}'`);
275
+ return values;
276
+ }
277
+
278
+ /**
279
+ * A literal operand.
280
+ *
281
+ * BOOLEANS: the docs say "Boolean literals are represented as `1` or `0`", but their own `!=`
282
+ * example is `is_capital != true`. Both are accepted; `1`/`0` stay NUMBERS (so `population > 0`
283
+ * still compares numerically) and are matched against a boolean field by `looseEquals`.
284
+ */
285
+ private parseOperand(): FilterValue {
286
+ const t = this.next();
287
+ if (t === undefined) throw new FilterError(`upstashvector: expected a value in filter '${this.source}'`);
288
+ if (t.kind === 'string') return t.value;
289
+ if (t.kind === 'number') return Number(t.value);
290
+ if (t.kind === 'ident' || t.kind === 'word') {
291
+ const lower = t.value.toLowerCase();
292
+ if (lower === 'true') return true;
293
+ if (lower === 'false') return false;
294
+ // A bare word is not a literal — an unquoted string is the single commonest filter typo, and
295
+ // matching it against the field name would silently compare a value to itself.
296
+ throw new FilterError(`upstashvector: '${t.value}' at position ${t.pos} is not a literal — string values must be quoted in filter '${this.source}'`);
297
+ }
298
+ throw new FilterError(`upstashvector: unexpected '${t.value}' where a value was expected in filter '${this.source}'`);
299
+ }
300
+ }
301
+
302
+ /**
303
+ * Walk the AST and parse every metadata path, so a malformed one is a PARSE error.
304
+ * See the note in `Parser.parse` for the data-dependent-status bug this closes.
305
+ */
306
+ function validatePaths(node: FilterNode): void {
307
+ switch (node.kind) {
308
+ case 'and': case 'or':
309
+ validatePaths(node.left);
310
+ validatePaths(node.right);
311
+ return;
312
+ case 'compare': case 'in': case 'hasField':
313
+ parsePath(node.path); // throws FilterError on a malformed path
314
+ }
315
+ }
316
+
317
+ /** Parse a filter string into an AST. Throws `FilterError` on anything malformed. */
318
+ export function parseFilter(filter: string): FilterNode {
319
+ return new Parser(tokenizeFilter(filter), filter).parse();
320
+ }
321
+
322
+ // ─────────────────────────────────────────────────────────────────────────────────────────────
323
+ // PATH RESOLUTION
324
+ // ─────────────────────────────────────────────────────────────────────────────────────────────
325
+
326
+ /** One step of a metadata path: an object key, or an array subscript. */
327
+ type PathStep = { key: string } | { index: number; fromEnd: boolean };
328
+
329
+ /**
330
+ * Split `geography.coordinates[0]` / `tags[#-1]` into steps.
331
+ *
332
+ * `[#-1]` is the docs' "index from the back using the `#` character with negative values": the
333
+ * `#` marks end-relative addressing and the value that follows is the offset, so `[#-1]` is the
334
+ * LAST element. A plain `[0]` is head-relative.
335
+ */
336
+ export function parsePath(path: string): PathStep[] {
337
+ const steps: PathStep[] = [];
338
+ let buf = '';
339
+ let i = 0;
340
+ const flush = () => { if (buf !== '') { steps.push({ key: buf }); buf = ''; } };
341
+ while (i < path.length) {
342
+ const c = path[i]!;
343
+ if (c === '.') { flush(); i++; continue; }
344
+ if (c === '[') {
345
+ flush();
346
+ const close = path.indexOf(']', i);
347
+ if (close === -1) throw new FilterError(`upstashvector: unterminated '[' in metadata path '${path}'`);
348
+ const raw = path.slice(i + 1, close);
349
+ const fromEnd = raw.startsWith('#');
350
+ const digits = fromEnd ? raw.slice(1) : raw;
351
+ // `Number('')` is 0, so a BARE `a[]` used to parse as `a[0]` — silently addressing the first
352
+ // element of an array the caller never named an index for. Require an explicit integer
353
+ // literal rather than trusting the numeric coercion.
354
+ if (!/^-?\d+$/.test(digits)) {
355
+ throw new FilterError(`upstashvector: '${raw}' is not an array index in metadata path '${path}'`);
356
+ }
357
+ const n = Number(digits);
358
+ if (!Number.isInteger(n)) throw new FilterError(`upstashvector: '${raw}' is not an array index in metadata path '${path}'`);
359
+ steps.push({ index: n, fromEnd });
360
+ i = close + 1;
361
+ continue;
362
+ }
363
+ buf += c;
364
+ i++;
365
+ }
366
+ flush();
367
+ if (steps.length === 0) throw new FilterError(`upstashvector: empty metadata path in filter`);
368
+ return steps;
369
+ }
370
+
371
+ /** The sentinel for "this path is not present". Distinct from a stored `null`, which IS present. */
372
+ export const MISSING: unique symbol = Symbol('upstashvector.missing');
373
+
374
+ /** Resolve a metadata path. Returns `MISSING` when any step is absent. */
375
+ export function resolvePath(metadata: unknown, path: string): unknown | typeof MISSING {
376
+ let cur: unknown = metadata;
377
+ for (const step of parsePath(path)) {
378
+ if (cur === null || cur === undefined) return MISSING;
379
+ if ('key' in step) {
380
+ if (typeof cur !== 'object' || Array.isArray(cur)) return MISSING;
381
+ const rec = cur as Record<string, unknown>;
382
+ if (!Object.prototype.hasOwnProperty.call(rec, step.key)) return MISSING;
383
+ cur = rec[step.key];
384
+ continue;
385
+ }
386
+ if (!Array.isArray(cur)) return MISSING;
387
+ // `[#-1]` addresses from the end; `[0]` from the start. A negative head-relative index is not
388
+ // the documented spelling, so it simply misses rather than silently wrapping.
389
+ const idx = step.fromEnd ? cur.length + step.index : step.index;
390
+ if (!Number.isInteger(idx) || idx < 0 || idx >= cur.length) return MISSING;
391
+ cur = cur[idx];
392
+ }
393
+ return cur;
394
+ }
395
+
396
+ // ─────────────────────────────────────────────────────────────────────────────────────────────
397
+ // GLOB
398
+ // ─────────────────────────────────────────────────────────────────────────────────────────────
399
+
400
+ /**
401
+ * SQLite-style GLOB: `*` any run, `?` one character, `[abc]` a class, `[^abc]` a negated class,
402
+ * `[a-z]` a range. Case-SENSITIVE (that is what distinguishes GLOB from LIKE).
403
+ *
404
+ * Implemented as a backtracking matcher rather than by translating to a RegExp: a translation has
405
+ * to escape every regex metacharacter that is ORDINARY in a glob (`.`, `+`, `(`, `$`, …), and
406
+ * missing one turns a filter into a different filter — a silent wrong-answer bug rather than an
407
+ * error. The docs' own example, `city GLOB '?[sz]*[^m-z]'`, exercises all four constructs.
408
+ */
409
+ export function globMatch(pattern: string, subject: string): boolean {
410
+ const walk = (p: number, s: number): boolean => {
411
+ if (p === pattern.length) return s === subject.length;
412
+ const pc = pattern[p]!;
413
+ if (pc === '*') {
414
+ // Collapse runs of `*` so `a**b` costs no more than `a*b`.
415
+ let q = p;
416
+ while (q < pattern.length && pattern[q] === '*') q++;
417
+ for (let k = s; k <= subject.length; k++) if (walk(q, k)) return true;
418
+ return false;
419
+ }
420
+ if (s === subject.length) return false;
421
+ if (pc === '?') return walk(p + 1, s + 1);
422
+ if (pc === '[') {
423
+ const close = classEnd(pattern, p);
424
+ if (close === -1) return pattern[p] === subject[s] && walk(p + 1, s + 1); // an unclosed '[' is a literal
425
+ const negated = pattern[p + 1] === '^';
426
+ const body = pattern.slice(p + 1 + (negated ? 1 : 0), close);
427
+ return inClass(body, subject[s]!) !== negated && walk(close + 1, s + 1);
428
+ }
429
+ return pc === subject[s] && walk(p + 1, s + 1);
430
+ };
431
+ return walk(0, 0);
432
+ }
433
+
434
+ /** Index of the `]` closing the class opened at `open`, or -1. A `]` FIRST in the class is literal. */
435
+ function classEnd(pattern: string, open: number): number {
436
+ let i = open + 1;
437
+ if (pattern[i] === '^') i++;
438
+ if (pattern[i] === ']') i++; // a leading ']' is a literal member, not the terminator
439
+ for (; i < pattern.length; i++) if (pattern[i] === ']') return i;
440
+ return -1;
441
+ }
442
+
443
+ function inClass(body: string, ch: string): boolean {
444
+ for (let i = 0; i < body.length; i++) {
445
+ // A range needs a `-` with members on BOTH sides; a trailing `-` is a literal hyphen.
446
+ if (body[i + 1] === '-' && i + 2 < body.length) {
447
+ const lo = body[i]!;
448
+ const hi = body[i + 2]!;
449
+ if (ch >= lo && ch <= hi) return true;
450
+ i += 2;
451
+ continue;
452
+ }
453
+ if (body[i] === ch) return true;
454
+ }
455
+ return false;
456
+ }
457
+
458
+ // ─────────────────────────────────────────────────────────────────────────────────────────────
459
+ // EVALUATOR
460
+ // ─────────────────────────────────────────────────────────────────────────────────────────────
461
+
462
+ /**
463
+ * Compare a stored metadata value with a filter literal for EQUALITY.
464
+ *
465
+ * `1`/`0` are the docs' own boolean spelling, so a numeric 1 matches a stored `true`. Everything
466
+ * else is compared by JS `===` after that coercion, so `'5'` (a string literal) does NOT equal a
467
+ * stored number 5 — a filter language that silently coerced across types would make
468
+ * `price = '10'` match a `price: 10` field and mask the caller's bug.
469
+ */
470
+ function looseEquals(stored: unknown, literal: FilterValue): boolean {
471
+ if (typeof stored === 'boolean' && typeof literal === 'number') return stored === (literal === 1);
472
+ if (typeof stored === 'boolean' && typeof literal === 'boolean') return stored === literal;
473
+ return stored === literal;
474
+ }
475
+
476
+ /** Ordering comparison. Only like-typed operands are ordered; anything else simply does not match. */
477
+ function ordered(stored: unknown, literal: FilterValue, op: '<' | '>' | '<=' | '>='): boolean {
478
+ if (typeof stored === 'number' && typeof literal === 'number') {
479
+ return op === '<' ? stored < literal : op === '>' ? stored > literal : op === '<=' ? stored <= literal : stored >= literal;
480
+ }
481
+ if (typeof stored === 'string' && typeof literal === 'string') {
482
+ return op === '<' ? stored < literal : op === '>' ? stored > literal : op === '<=' ? stored <= literal : stored >= literal;
483
+ }
484
+ return false;
485
+ }
486
+
487
+ /** Evaluate a parsed filter against one vector's metadata. */
488
+ export function evaluateFilter(node: FilterNode, metadata: unknown): boolean {
489
+ switch (node.kind) {
490
+ case 'and': return evaluateFilter(node.left, metadata) && evaluateFilter(node.right, metadata);
491
+ case 'or': return evaluateFilter(node.left, metadata) || evaluateFilter(node.right, metadata);
492
+ case 'hasField': {
493
+ const present = resolvePath(metadata, node.path) !== MISSING;
494
+ return node.negated ? !present : present;
495
+ }
496
+ case 'in': {
497
+ const v = resolvePath(metadata, node.path);
498
+ // A MISSING field is in no list — and is also not "NOT IN" anything, because the vendor's
499
+ // HAS FIELD operator exists precisely so that absence is asked about explicitly. Treating
500
+ // absence as a NOT IN match would make `country NOT IN ('X')` select vectors with no
501
+ // `country` at all, which is the classic filter-language footgun.
502
+ if (v === MISSING) return false;
503
+ const hit = node.values.some((lit) => looseEquals(v, lit));
504
+ return node.negated ? !hit : hit;
505
+ }
506
+ case 'compare': {
507
+ const v = resolvePath(metadata, node.path);
508
+ if (v === MISSING) return false; // same rule as IN: absence never satisfies a comparison
509
+ switch (node.op) {
510
+ case '=': return looseEquals(v, node.value);
511
+ case '!=': return !looseEquals(v, node.value);
512
+ case '<': case '>': case '<=': case '>=': return ordered(v, node.value, node.op);
513
+ case 'GLOB': return typeof v === 'string' && typeof node.value === 'string' && globMatch(node.value, v);
514
+ case 'NOT GLOB': return typeof v === 'string' && typeof node.value === 'string' && !globMatch(node.value, v);
515
+ case 'CONTAINS': return Array.isArray(v) && v.some((el) => looseEquals(el, node.value));
516
+ case 'NOT CONTAINS': return Array.isArray(v) && !v.some((el) => looseEquals(el, node.value));
517
+ }
518
+ }
519
+ }
520
+ }
521
+
522
+ /** Parse + evaluate in one step. Exported for the handler's single call site. */
523
+ export function matchesFilter(filter: string, metadata: unknown): boolean {
524
+ return evaluateFilter(parseFilter(filter), metadata);
525
+ }
@@ -0,0 +1,86 @@
1
+ // upstashvector twin HTTP server — serve the Upstash Vector REST twin over HTTP so the real
2
+ // `@upstash/vector` SDK (pointed at it with nothing but its own public `url` option) works
3
+ // UNMODIFIED.
4
+ //
5
+ // ── THE INDEX IS CONFIGURED HERE, NOT DISCOVERED ──────────────────────────────────────────────
6
+ // A real Upstash Vector index is created with a fixed `dimension` and `similarityFunction` and can
7
+ // never change them. A twin has no creation step, so those become server options: pass them and the
8
+ // index behaves exactly like one created that way (including the vendor's 422 on a mismatched
9
+ // vector); omit `dimension` and the FIRST upsert locks one in, which is the convenient local
10
+ // default. `similarityFunction` defaults to COSINE, the console's own default.
11
+ //
12
+ // FETCH-FIRST (runtime contract R12b): the surface is the plain `createUpstashVectorTwinFetch` and
13
+ // the SERVER is one line of `Bun.serve` around it. This is a CUSTOM fetch, not the kernel adapter
14
+ // (`createTwinFetchFromHandler`): replies carry ONLY the handler's own headers, and the clock is
15
+ // an injectable seam.
16
+ import { serveHttp } from '@volter/world-core';
17
+ import { handleUpstashVectorTwinRequest } from './upstashvector-twin.ts';
18
+ import { worldNow, statefulTwinManifest} from '@volter/world-core';
19
+ import type { SimilarityFunction } from './upstashvector-store.ts';
20
+
21
+ export type UpstashVectorServerOptions = {
22
+ root?: string;
23
+ port?: number;
24
+ readOnly?: boolean;
25
+ /** The token the twin demands. Omit to accept any non-empty credential (still 401s a missing one). */
26
+ token?: string;
27
+ /** The dimension this index enforces. Omit to let the first upsert lock one in. */
28
+ dimension?: number;
29
+ /** The metric this index ranks with. Defaults to COSINE. */
30
+ similarityFunction?: SimilarityFunction;
31
+ /** The injected clock. Returns the ISO instant this request "happens at". */
32
+ now?: () => string;
33
+ };
34
+
35
+ /** Options every Upstash Vector-twin HTTP surface needs, independent of who owns the socket. */
36
+ export interface UpstashVectorTwinFetchOptions {
37
+ root?: string;
38
+ readOnly?: boolean;
39
+ /** The token the twin demands. Omit to accept any non-empty credential (still 401s a missing one). */
40
+ token?: string;
41
+ /** The dimension this index enforces. Omit to let the first upsert lock one in. */
42
+ dimension?: number;
43
+ /** The metric this index ranks with. Defaults to COSINE. */
44
+ similarityFunction?: SimilarityFunction;
45
+ /** The injected clock. Returns the ISO instant this request "happens at". */
46
+ now?: () => string;
47
+ }
48
+
49
+ export function createUpstashVectorTwinFetch(options: UpstashVectorTwinFetchOptions = {}): (request: Request) => Promise<Response> {
50
+ const readOnly = options.readOnly ?? false;
51
+ const now = options.now ?? (() => worldNow());
52
+ return async function upstashVectorTwinFetch(request: Request): Promise<Response> {
53
+ const url = new URL(request.url);
54
+ // GET /twin — the discovery manifest (education inside the twin).
55
+ if (request.method === 'GET' && url.pathname.replace(/\/+$/, '') === '/twin') {
56
+ return Response.json(statefulTwinManifest({ vendor: 'upstashvector', twinOf: 'the Upstash Vector REST API', stores: 'namespaces and vectors, queried by a real scorer' }));
57
+ }
58
+ const body = request.method !== 'GET' && request.method !== 'HEAD' ? await request.text() : '';
59
+ const headers: Record<string, string> = {};
60
+ request.headers.forEach((value, key) => { headers[key.toLowerCase()] = value; });
61
+ const { status, body: out, headers: outHeaders } = await handleUpstashVectorTwinRequest({
62
+ method: request.method,
63
+ path: url.pathname + (url.search || ''),
64
+ body,
65
+ headers,
66
+ readOnly,
67
+ occurredAt: now(),
68
+ ...(options.root !== undefined ? { root: options.root } : {}),
69
+ ...(options.token !== undefined ? { token: options.token } : {}),
70
+ ...(options.dimension !== undefined ? { dimension: options.dimension } : {}),
71
+ ...(options.similarityFunction !== undefined ? { similarityFunction: options.similarityFunction } : {}),
72
+ });
73
+ // A null body means "no body at all" (HEAD, CORS preflight).
74
+ if (out === null) return new Response(null, { status, headers: { ...(outHeaders ?? {}) } });
75
+ return new Response(JSON.stringify(out), { status, headers: { ...(outHeaders ?? {}) } });
76
+ };
77
+ }
78
+
79
+ export async function createUpstashVectorTwinServer(options: UpstashVectorServerOptions = {}): Promise<{ port: number; stop: () => void }> {
80
+ const server = await serveHttp({
81
+ port: options.port ?? 0,
82
+ idleTimeout: 60,
83
+ fetch: createUpstashVectorTwinFetch(options),
84
+ });
85
+ return { port: server.port ?? options.port ?? 0, stop: () => server.stop(true) };
86
+ }