@volter/twin-turbopuffer 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (52) hide show
  1. package/LICENSE +202 -0
  2. package/README.md +145 -0
  3. package/dist/src/cli.d.ts +2 -0
  4. package/dist/src/cli.js +27 -0
  5. package/dist/src/generated/surface.gen.json +1 -0
  6. package/dist/src/generated/ui.gen.json +1 -0
  7. package/dist/src/index.d.ts +11 -0
  8. package/dist/src/index.js +65 -0
  9. package/dist/src/key-gate.d.ts +3 -0
  10. package/dist/src/key-gate.js +39 -0
  11. package/dist/src/manifest.d.ts +2 -0
  12. package/dist/src/manifest.js +28 -0
  13. package/dist/src/screens/dashboard.d.ts +11 -0
  14. package/dist/src/screens/dashboard.js +191 -0
  15. package/dist/src/semantics/namespaces.d.ts +5 -0
  16. package/dist/src/semantics/namespaces.js +17 -0
  17. package/dist/src/turbopuffer-capabilities.d.ts +6 -0
  18. package/dist/src/turbopuffer-capabilities.js +442 -0
  19. package/dist/src/turbopuffer-conformance.d.ts +8 -0
  20. package/dist/src/turbopuffer-conformance.js +102 -0
  21. package/dist/src/turbopuffer-connector.d.ts +34 -0
  22. package/dist/src/turbopuffer-connector.js +152 -0
  23. package/dist/src/turbopuffer-filter.d.ts +57 -0
  24. package/dist/src/turbopuffer-filter.js +286 -0
  25. package/dist/src/turbopuffer-server.d.ts +23 -0
  26. package/dist/src/turbopuffer-server.js +72 -0
  27. package/dist/src/turbopuffer-stem.d.ts +1 -0
  28. package/dist/src/turbopuffer-stem.js +133 -0
  29. package/dist/src/turbopuffer-store.d.ts +80 -0
  30. package/dist/src/turbopuffer-store.js +1304 -0
  31. package/dist/src/turbopuffer-text.d.ts +44 -0
  32. package/dist/src/turbopuffer-text.js +189 -0
  33. package/dist/src/turbopuffer-twin.d.ts +25 -0
  34. package/dist/src/turbopuffer-twin.js +406 -0
  35. package/package.json +56 -0
  36. package/src/cli.ts +28 -0
  37. package/src/generated/surface.gen.json +1 -0
  38. package/src/generated/ui.gen.json +1 -0
  39. package/src/index.ts +92 -0
  40. package/src/key-gate.ts +39 -0
  41. package/src/manifest.ts +63 -0
  42. package/src/screens/dashboard.tsx +214 -0
  43. package/src/semantics/namespaces.ts +30 -0
  44. package/src/turbopuffer-capabilities.ts +455 -0
  45. package/src/turbopuffer-conformance.ts +101 -0
  46. package/src/turbopuffer-connector.ts +153 -0
  47. package/src/turbopuffer-filter.ts +277 -0
  48. package/src/turbopuffer-server.ts +81 -0
  49. package/src/turbopuffer-stem.ts +104 -0
  50. package/src/turbopuffer-store.ts +1157 -0
  51. package/src/turbopuffer-text.ts +204 -0
  52. package/src/turbopuffer-twin.ts +429 -0
@@ -0,0 +1,153 @@
1
+ // Turbopuffer's half of the real state system (protocol 2): the PERFORM adapter that replays one
2
+ // recorded write against the vendor, and the REFRESH adapter that observes a real account into
3
+ // the tree. Both run over the kernel's injected `RemoteExecute` — the kernel applies the sealed
4
+ // credential (the placeholder `authorization` below is replaced), so this pack never holds a key
5
+ // and issues no network call of its own.
6
+ import { observeResource, observeResources, type PerformContext, type PushOutcome, type RemoteExecute, type TwinAction } from '@volter/world-core';
7
+ import { documentSubjectId, SERVICE } from './turbopuffer-store.ts';
8
+ import { DEFAULT_FTS, type FtsConfig } from './turbopuffer-text.ts';
9
+ import type { AttrConfig } from './turbopuffer-filter.ts';
10
+
11
+ const HEADERS = { accept: 'application/json', 'content-type': 'application/json', authorization: 'Bearer twin' };
12
+ /** Bounds on one refresh, so a huge account cannot turn a sync into an unbounded crawl. */
13
+ export const REFRESH_MAX_NAMESPACE_PAGES = 50;
14
+ export const REFRESH_MAX_DOCUMENTS_PER_NAMESPACE = 100_000;
15
+ const DOC_PAGE = 1000;
16
+
17
+ async function call(execute: RemoteExecute, method: string, path: string, body?: unknown): Promise<Record<string, unknown>> {
18
+ const res = await execute({ method, path, headers: { ...HEADERS }, ...(body === undefined ? {} : { body: JSON.stringify(body) }) });
19
+ // A refusal is thrown, never read as an empty account (adding-a-twin.md §6).
20
+ if (res.status < 200 || res.status >= 300) throw new Error(`turbopuffer ${method} ${path} refused: HTTP ${res.status} ${res.body.slice(0, 200)}`);
21
+ try {
22
+ return JSON.parse(res.body || '{}') as Record<string, unknown>;
23
+ } catch {
24
+ throw new Error(`turbopuffer ${method} ${path} answered a body that is not JSON`);
25
+ }
26
+ }
27
+
28
+ /**
29
+ * Perform ONE recorded entry against the vendor. A write request is recorded whole (its body is the
30
+ * action's `input`), so the perform replays exactly what the caller sent; Turbopuffer ids are
31
+ * caller-chosen, so there is no local→vendor id to rebind.
32
+ */
33
+ export async function performTurbopufferAction(execute: RemoteExecute, action: TwinAction, _ctx: PerformContext): Promise<PushOutcome> {
34
+ const ns = action.subject.id;
35
+ const path = `/v2/namespaces/${encodeURIComponent(ns)}`;
36
+ const input = (action.input ?? {}) as Record<string, unknown>;
37
+ switch (action.operation) {
38
+ case 'namespace.write':
39
+ return { externalId: ns, data: await call(execute, 'POST', path, input) };
40
+ case 'namespace.delete_all':
41
+ return { externalId: ns, data: await call(execute, 'DELETE', path) };
42
+ case 'namespace.update_schema':
43
+ return { externalId: ns, data: await call(execute, 'POST', `/v1/namespaces/${encodeURIComponent(ns)}/schema`, input.schema ?? {}) };
44
+ }
45
+ throw new Error(`turbopuffer perform: no vendor operation for ${action.operation ?? '(unnamed)'} on ${action.subject.type}:${ns}`);
46
+ }
47
+
48
+ /** Map one attribute of the vendor's schema wire into the stored form (lenient: the vendor is authoritative). */
49
+ export function mapSchemaAttr(raw: unknown): AttrConfig {
50
+ const r = (typeof raw === 'string' ? { type: raw } : raw ?? {}) as Record<string, unknown>;
51
+ const fts: FtsConfig | false = r.full_text_search === undefined || r.full_text_search === false || r.full_text_search === null
52
+ ? false
53
+ : { ...DEFAULT_FTS, ...(typeof r.full_text_search === 'object' ? r.full_text_search as Partial<FtsConfig> : {}) };
54
+ return {
55
+ type: typeof r.type === 'string' ? r.type : 'string',
56
+ filterable: typeof r.filterable === 'boolean' ? r.filterable : fts === false,
57
+ full_text_search: fts,
58
+ ...(r.glob === true ? { glob: true } : {}),
59
+ ...(r.regex === true ? { regex: true } : {}),
60
+ };
61
+ }
62
+
63
+ /** The refresh adapter (the descriptor's `stateSystem.refresh`): the D7 pull over the kernel's executor. */
64
+ export async function syncTurbopufferFromRemote(
65
+ execute: RemoteExecute,
66
+ opts: { root?: string; origin?: string; occurredAt?: string } = {},
67
+ ): Promise<{ observed: number; complete?: string[] }> {
68
+ return syncTurbopufferFromReal(execute, opts);
69
+ }
70
+
71
+ /**
72
+ * D7 consumer-facing pull, through the injected executor (the kernel's in production, a fake in
73
+ * tests): list the account's namespaces, then for each read its schema, metadata and every document
74
+ * (paged by id), and OBSERVE them — the kernel diffs and folds. Read-only.
75
+ */
76
+ export async function syncTurbopufferFromReal(
77
+ execute: RemoteExecute,
78
+ opts: { root?: string; occurredAt?: string } = {},
79
+ ): Promise<{ observed: number; complete?: string[] }> {
80
+ const at = opts.occurredAt ?? new Date().toISOString();
81
+ const target = { ...(opts.root !== undefined ? { root: opts.root } : {}), at };
82
+ let observed = 0;
83
+ let truncated = false;
84
+ const names: string[] = [];
85
+ let cursor = '';
86
+ for (let page = 0; page < REFRESH_MAX_NAMESPACE_PAGES; page++) {
87
+ const listed = await call(execute, 'GET', `/v1/namespaces?page_size=1000${cursor ? `&cursor=${encodeURIComponent(cursor)}` : ''}`);
88
+ for (const n of (listed.namespaces as Array<{ id?: unknown }> | undefined) ?? []) if (typeof n.id === 'string') names.push(n.id);
89
+ cursor = typeof listed.next_cursor === 'string' ? listed.next_cursor : '';
90
+ if (cursor === '') break;
91
+ }
92
+ if (cursor !== '') truncated = true;
93
+ for (const ns of names) {
94
+ const enc = encodeURIComponent(ns);
95
+ const meta = await call(execute, 'GET', `/v2/namespaces/${enc}/metadata`);
96
+ const wireSchema = (meta.schema ?? {}) as Record<string, unknown>;
97
+ const schema: Record<string, AttrConfig> = {};
98
+ let distance: string | null = null;
99
+ let dims: number | null = null;
100
+ for (const [attr, raw] of Object.entries(wireSchema)) {
101
+ schema[attr] = mapSchemaAttr(raw);
102
+ const m = /^\[(\d+)\]f(16|32)$/.exec(schema[attr]!.type);
103
+ if (attr === 'vector' && m) {
104
+ dims = Number(m[1]);
105
+ const ann = (raw as Record<string, unknown> | null)?.ann as Record<string, unknown> | undefined;
106
+ if (typeof ann?.distance_metric === 'string') distance = ann.distance_metric;
107
+ }
108
+ }
109
+ observeResource(SERVICE, {
110
+ type: 'namespace',
111
+ id: ns,
112
+ fields: {
113
+ name: ns, schema, distance_metric: distance, vector_dims: dims,
114
+ created_at: typeof meta.created_at === 'string' ? meta.created_at : at,
115
+ updated_at: typeof meta.updated_at === 'string' ? meta.updated_at : at,
116
+ last_write_at: typeof meta.last_write_at === 'string' ? meta.last_write_at : typeof meta.updated_at === 'string' ? meta.updated_at : at,
117
+ gone: false,
118
+ },
119
+ }, target);
120
+ observed++;
121
+ let last: string | number | null = null;
122
+ for (let seen = 0; seen < REFRESH_MAX_DOCUMENTS_PER_NAMESPACE;) {
123
+ const page = await call(execute, 'POST', `/v2/namespaces/${enc}/query`, {
124
+ rank_by: ['id', 'asc'],
125
+ top_k: DOC_PAGE,
126
+ include_attributes: true,
127
+ ...(last === null ? {} : { filters: ['id', 'Gt', last] }),
128
+ });
129
+ const rows = (page.rows as Array<Record<string, unknown>> | undefined) ?? [];
130
+ const docs: Array<{ type: string; id: string; fields: Record<string, unknown> }> = [];
131
+ for (const row of rows) {
132
+ const { id, vector, $dist: _d, ...attributes } = row;
133
+ if (typeof id !== 'string' && typeof id !== 'number') continue;
134
+ for (const k of Object.keys(attributes)) if (attributes[k] === null) delete attributes[k];
135
+ docs.push({
136
+ type: 'document',
137
+ id: documentSubjectId(ns, id),
138
+ fields: { namespace: ns, doc_id: id, attributes, vector: Array.isArray(vector) ? vector : null },
139
+ });
140
+ last = id;
141
+ }
142
+ if (docs.length > 0) observeResources(SERVICE, docs, target);
143
+ observed += docs.length;
144
+ seen += rows.length;
145
+ if (rows.length < DOC_PAGE) break;
146
+ if (seen >= REFRESH_MAX_DOCUMENTS_PER_NAMESPACE) truncated = true;
147
+ }
148
+ }
149
+ // A listing cut short by the bounds is not the whole account, so it never claims completeness
150
+ // (the kernel removes what a COMPLETE refresh no longer saw).
151
+ return { observed, ...(truncated ? {} : { complete: ['namespace', 'document'] }) };
152
+ }
153
+
@@ -0,0 +1,277 @@
1
+ // THE FILTER LANGUAGE — Turbopuffer's array-encoded `filters` (a SQL WHERE clause as JSON), compiled
2
+ // into a predicate over the namespace's documents.
3
+ //
4
+ // The operator set is the one the installed SDK declares (`Filter` in
5
+ // @turbopuffer/turbopuffer@2.8.0 src/resources/custom.ts). Modelled: Eq, NotEq, In, NotIn,
6
+ // Contains, NotContains, ContainsAny, NotContainsAny, Lt, Lte, Gt, Gte, AnyLt, AnyLte, AnyGt,
7
+ // AnyGte, Glob, NotGlob, IGlob, NotIGlob, ContainsAllTokens, ContainsAnyToken, And, Or, Not.
8
+ // Refused with a 400 naming the gap: Regex, Fuzzy, ContainsTokenSequence.
9
+ //
10
+ // Semantics a caller depends on (Dub's provider relies on the first one explicitly): an attribute a
11
+ // document does not carry reads as null, so `NotIn` / `NotEq` / `NotContainsAny` MATCH it and
12
+ // `In` / `Eq x` / `ContainsAny` do not. Comparison operators are refused on an attribute whose
13
+ // schema says `filterable: false` (the default for a full-text-search attribute, per the SDK's
14
+ // `AttributeSchemaConfig.full_text_search` doc); the token operators read the BM25 index instead
15
+ // and require `full_text_search` on the attribute.
16
+ import { MODELED_TOKENIZERS, queryTerms, tokenize, tokensMatch, type FtsConfig } from './turbopuffer-text.ts';
17
+
18
+ /** A vendor-shaped API failure: served as `{ "status": "error", "error": message }`. */
19
+ export class TurbopufferError extends Error {
20
+ constructor(readonly status: number, message: string) {
21
+ super(message);
22
+ this.name = 'TurbopufferError';
23
+ }
24
+ }
25
+
26
+ /** 400: a request outside the shape turbopuffer's pages give it (a malformed filter, schema, write or query, a value of
27
+ * the wrong type, a parameter out of its documented range). The messages are the twin's; the pages name the rules, not
28
+ * the text of their refusals. */
29
+ export function badRequest(message: string): TurbopufferError {
30
+ return new TurbopufferError(400, message);
31
+ }
32
+
33
+ export type AttrConfig = {
34
+ /** Transient: an `ann.distance_metric` on a vector schema entry, moved onto the namespace. */
35
+ ann_distance_metric?: string;
36
+ type: string;
37
+ filterable: boolean;
38
+ full_text_search: FtsConfig | false;
39
+ glob?: boolean;
40
+ regex?: boolean;
41
+ fuzzy?: boolean;
42
+ /** a `{}f16` attribute's `sparse_knn` config (`distance_metric: dot_product`) */
43
+ sparse_knn?: { distance_metric: string };
44
+ /** a `[][N]f32` attribute's `ann: {late_interaction: true}` (false: stored without the index) */
45
+ late_interaction?: boolean;
46
+ /** native embedding: the model, the vector attribute the embeddings are stored in, and its dimensions */
47
+ embed?: { model: string; attribute: string; dims: number };
48
+ };
49
+
50
+ export type Doc = {
51
+ /** `n:<id>` or `s:<id>` — the id's JSON type is part of its identity (7 and "7" are two docs). */
52
+ key: string;
53
+ id: string | number;
54
+ attributes: Record<string, unknown>;
55
+ vector: number[] | null;
56
+ };
57
+
58
+ export type FilterContext = {
59
+ schema: Record<string, AttrConfig>;
60
+ /** Per (attribute, doc key) token cache, shared across one query. */
61
+ tokenCache: Map<string, string[]>;
62
+ };
63
+
64
+ export type Predicate = (doc: Doc) => boolean;
65
+
66
+ const COMPARISON = new Set(['Eq', 'NotEq', 'In', 'NotIn', 'Contains', 'NotContains', 'ContainsAny', 'NotContainsAny', 'Lt', 'Lte', 'Gt', 'Gte', 'AnyLt', 'AnyLte', 'AnyGt', 'AnyGte', 'Glob', 'NotGlob', 'IGlob', 'NotIGlob']);
67
+ const TOKEN_OPS = new Set(['ContainsAllTokens', 'ContainsAnyToken']);
68
+ const UNMODELED_OPS: Record<string, string> = {
69
+ ContainsTokenSequence: 'turbopuffer.filters.contains_token_sequence',
70
+ };
71
+
72
+ /**
73
+ * A namespace observed from a real account may carry a full-text config the twin does not model
74
+ * (the vendor default `word_v4`, stemming, another language). Such an attribute is refused at
75
+ * query time rather than tokenized as if it were `word_v2`.
76
+ */
77
+ export function assertModelledFts(attr: string, fts: FtsConfig): void {
78
+ if (!MODELED_TOKENIZERS.includes(fts.tokenizer)) throw badRequest(`turbopuffer twin: attribute "${attr}" uses tokenizer "${fts.tokenizer}", which this twin does not model yet (turbopuffer.fts.tokenizers is the filed gap)`);
79
+ if (fts.language !== 'english') throw badRequest(`turbopuffer twin: attribute "${attr}" uses language "${fts.language}", which this twin does not model (turbopuffer.fts.languages is the filed gap)`);
80
+ }
81
+
82
+ export function attributeValue(doc: Doc, attr: string): unknown {
83
+ if (attr === 'id') return doc.id;
84
+ if (attr === 'vector') return doc.vector;
85
+ const v = doc.attributes[attr];
86
+ return v === undefined ? null : v;
87
+ }
88
+
89
+ function sameValue(a: unknown, b: unknown): boolean {
90
+ if (a === b) return true;
91
+ if (a === null || b === null || typeof a !== 'object' || typeof b !== 'object') return false;
92
+ return JSON.stringify(a) === JSON.stringify(b);
93
+ }
94
+
95
+ /** -1/0/1 for two comparable scalars (number↔number, string↔string), NaN otherwise. */
96
+ export function compareScalars(a: unknown, b: unknown): number {
97
+ if (typeof a === 'number' && typeof b === 'number') return a < b ? -1 : a > b ? 1 : 0;
98
+ if (typeof a === 'string' && typeof b === 'string') return a < b ? -1 : a > b ? 1 : 0;
99
+ if (typeof a === 'boolean' && typeof b === 'boolean') return a === b ? 0 : a ? 1 : -1;
100
+ return Number.NaN;
101
+ }
102
+
103
+ function globRegex(pattern: string, insensitive: boolean): RegExp {
104
+ let out = '';
105
+ for (let i = 0; i < pattern.length; i++) {
106
+ const c = pattern[i]!;
107
+ if (c === '*') out += '.*';
108
+ else if (c === '?') out += '.';
109
+ else if (c === '[') {
110
+ const close = pattern.indexOf(']', i + 1);
111
+ if (close === -1) { out += '\\['; continue; }
112
+ let body = pattern.slice(i + 1, close);
113
+ if (body.startsWith('!')) body = `^${body.slice(1)}`;
114
+ out += `[${body.replace(/\\/g, '\\\\')}]`;
115
+ i = close;
116
+ } else out += c.replace(/[.*+?^${}()|[\]\\/]/g, '\\$&');
117
+ }
118
+ return new RegExp(`^${out}$`, insensitive ? 'is' : 's');
119
+ }
120
+
121
+ function describe(f: unknown): string {
122
+ const s = JSON.stringify(f);
123
+ return s.length > 120 ? `${s.slice(0, 117)}...` : s;
124
+ }
125
+
126
+ /** Compile a filter into a predicate, refusing a malformed one with a 400. */
127
+ export function compileFilter(f: unknown, ctx: FilterContext): Predicate {
128
+ if (!Array.isArray(f) || f.length === 0) throw badRequest(`invalid filter: expected a non-empty array, got ${describe(f)}`);
129
+ const head = f[0];
130
+ if (head === 'And' || head === 'Or') {
131
+ if (f.length !== 2 || !Array.isArray(f[1])) throw badRequest(`invalid filter: ${head} takes one array of filters, got ${describe(f)}`);
132
+ const parts = (f[1] as unknown[]).map((p) => compileFilter(p, ctx));
133
+ return head === 'And' ? (d) => parts.every((p) => p(d)) : (d) => parts.some((p) => p(d));
134
+ }
135
+ if (head === 'Not') {
136
+ if (f.length !== 2) throw badRequest(`invalid filter: Not takes one filter, got ${describe(f)}`);
137
+ const inner = compileFilter(f[1], ctx);
138
+ return (d) => !inner(d);
139
+ }
140
+ if (typeof head !== 'string') throw badRequest(`invalid filter: expected an attribute name, got ${describe(f)}`);
141
+ const attr = head;
142
+ const op = f[1];
143
+ if (typeof op !== 'string') throw badRequest(`invalid filter: expected an operator for attribute "${attr}", got ${describe(f)}`);
144
+ const gap = UNMODELED_OPS[op];
145
+ if (gap !== undefined) throw badRequest(`turbopuffer twin: the ${op} filter is real Turbopuffer surface this twin does not model yet (${gap} is the filed gap)`);
146
+ const config = ctx.schema[attr];
147
+
148
+ if (TOKEN_OPS.has(op)) {
149
+ if (f.length < 3 || f.length > 4) throw badRequest(`invalid filter: ${op} takes a query and optional params, got ${describe(f)}`);
150
+ if (config === undefined || config.full_text_search === false) throw badRequest(`invalid filter: ${op} requires full-text search to be enabled on attribute "${attr}"`);
151
+ const fts = config.full_text_search;
152
+ assertModelledFts(attr, fts);
153
+ const params = (f[3] ?? {}) as Record<string, unknown>;
154
+ if (typeof params !== 'object' || params === null || Array.isArray(params)) throw badRequest(`invalid filter: ${op} params must be an object`);
155
+ for (const k of Object.keys(params)) if (k !== 'last_as_prefix') throw badRequest(`invalid filter: unknown ${op} param "${k}"`);
156
+ const query = f[2];
157
+ if (typeof query !== 'string' && !(Array.isArray(query) && query.every((q) => typeof q === 'string'))) throw badRequest(`invalid filter: ${op} query must be a string or an array of strings`);
158
+ const terms = queryTerms(query, fts, params.last_as_prefix === true);
159
+ const mode = op === 'ContainsAllTokens' ? 'all' : 'any';
160
+ return (d) => tokensMatch(docTokens(ctx, attr, fts, d), terms, mode);
161
+ }
162
+
163
+ if (op === 'Regex') return regexFilter(attr, config, f);
164
+ if (op === 'Fuzzy') return fuzzyFilter(attr, config, f);
165
+ if (!COMPARISON.has(op)) throw badRequest(`invalid filter: unknown operator "${op}" in ${describe(f)}`);
166
+ if (f.length !== 3) throw badRequest(`invalid filter: ${op} takes exactly one value, got ${describe(f)}`);
167
+ // Glob "Requires the glob (or for backwards compatibility, filterable) schema attribute" (https://turbopuffer.com/docs/query)
168
+ const globbed = /^(Not)?I?Glob$/.test(op) && config?.glob === true;
169
+ if (config !== undefined && !config.filterable && !globbed && attr !== 'id') {
170
+ throw badRequest(`invalid filter: attribute "${attr}" is not filterable (set filterable: true in its schema)`);
171
+ }
172
+ const value = f[2];
173
+ const needsArray = op === 'In' || op === 'NotIn' || op === 'ContainsAny' || op === 'NotContainsAny';
174
+ if (needsArray && !Array.isArray(value)) throw badRequest(`invalid filter: ${op} requires an array value, got ${describe(f)}`);
175
+ const list = needsArray ? (value as unknown[]) : [];
176
+ // Ids order numbers before strings (the order rank_by [id, asc] serves), so paging by id never skips a type.
177
+ const cmp = attr === 'id' ? compareIdValues : compareScalars;
178
+ switch (op) {
179
+ case 'Eq': return (d) => sameValue(attributeValue(d, attr), value);
180
+ case 'NotEq': return (d) => !sameValue(attributeValue(d, attr), value);
181
+ case 'In': return (d) => { const v = attributeValue(d, attr); return v !== null && list.some((x) => sameValue(v, x)); };
182
+ case 'NotIn': return (d) => { const v = attributeValue(d, attr); return v === null || !list.some((x) => sameValue(v, x)); };
183
+ case 'Contains': return (d) => { const v = attributeValue(d, attr); return Array.isArray(v) && v.some((x) => sameValue(x, value)); };
184
+ case 'NotContains': return (d) => { const v = attributeValue(d, attr); return !(Array.isArray(v) && v.some((x) => sameValue(x, value))); };
185
+ case 'ContainsAny': return (d) => { const v = attributeValue(d, attr); return Array.isArray(v) && v.some((x) => list.some((y) => sameValue(x, y))); };
186
+ case 'NotContainsAny': return (d) => { const v = attributeValue(d, attr); return !(Array.isArray(v) && v.some((x) => list.some((y) => sameValue(x, y)))); };
187
+ case 'Lt': return (d) => cmp(attributeValue(d, attr), value) < 0;
188
+ case 'Lte': return (d) => cmp(attributeValue(d, attr), value) <= 0;
189
+ case 'Gt': return (d) => cmp(attributeValue(d, attr), value) > 0;
190
+ case 'Gte': return (d) => cmp(attributeValue(d, attr), value) >= 0;
191
+ case 'AnyLt': return (d) => anyOf(attributeValue(d, attr), (x) => compareScalars(x, value) < 0);
192
+ case 'AnyLte': return (d) => anyOf(attributeValue(d, attr), (x) => compareScalars(x, value) <= 0);
193
+ case 'AnyGt': return (d) => anyOf(attributeValue(d, attr), (x) => compareScalars(x, value) > 0);
194
+ case 'AnyGte': return (d) => anyOf(attributeValue(d, attr), (x) => compareScalars(x, value) >= 0);
195
+ case 'Glob': case 'NotGlob': case 'IGlob': case 'NotIGlob': {
196
+ if (typeof value !== 'string') throw badRequest(`invalid filter: ${op} requires a string pattern`);
197
+ let re: RegExp;
198
+ try { re = globRegex(value, op === 'IGlob' || op === 'NotIGlob'); } catch { throw badRequest(`invalid filter: ${op} pattern ${JSON.stringify(value)} is not a valid glob`); }
199
+ const negate = op.startsWith('Not');
200
+ return (d) => { const v = attributeValue(d, attr); const hit = typeof v === 'string' && re.test(v); return negate ? !hit : hit; };
201
+ }
202
+ }
203
+ throw badRequest(`invalid filter: unknown operator "${op}"`);
204
+ }
205
+
206
+ /** Regex (https://turbopuffer.com/docs/query): "Regular expression match against `string` attribute values. Requires the
207
+ * regex schema attribute to be enabled before use"; its example `["text", "Regex", "\\w+fish"]` "matches "swordfish",
208
+ * "pufferfish", "clownfish"", so a pattern matches anywhere in the value. Where the docs stop: the twin runs the pattern
209
+ * as a JavaScript regular expression (Unicode mode), which agrees with Rust's regex syntax for the common classes. */
210
+ function regexFilter(attr: string, config: AttrConfig | undefined, f: unknown[]): Predicate {
211
+ if (config === undefined || !config.regex) throw badRequest(`invalid filter: Regex requires regex: true in the schema of attribute "${attr}"`);
212
+ if (f.length !== 3 || typeof f[2] !== 'string') throw badRequest('invalid filter: Regex takes a pattern string');
213
+ let re: RegExp;
214
+ try { re = new RegExp(f[2], 'u'); } catch { throw badRequest(`invalid filter: Regex pattern ${JSON.stringify(f[2])} is not a valid regular expression`); }
215
+ return (d) => { const v = attributeValue(d, attr); return typeof v === 'string' && re.test(v); };
216
+ }
217
+
218
+ /** Fuzzy (https://turbopuffer.com/docs/query and /docs/fts, Fuzzy matching): "Fuzzy substring match against `string` or
219
+ * `[]string` attribute values. Requires the fuzzy schema attribute". `max_edit_distance` "sets how many edits to
220
+ * tolerate by the query length in characters (uses Levenshtein distance). `distance` can be `0`, `1`, or `2`, and
221
+ * `min_query_chars` must be at least 3 · (`distance` + 1). Queries shorter than the first `min_query_chars` threshold
222
+ * return no matches"; `case_sensitive` "defaults to `true`", and "A missing or added character, incorrect character,
223
+ * missing or added diacritic (e.g. ü), or case difference will add 1 to the edit distance". The twin measures the
224
+ * fewest edits between the query and any substring of the value, over NFC characters. */
225
+ function fuzzyFilter(attr: string, config: AttrConfig | undefined, f: unknown[]): Predicate {
226
+ if (config === undefined || !config.fuzzy) throw badRequest(`invalid filter: Fuzzy requires fuzzy: true in the schema of attribute "${attr}"`);
227
+ if (f.length !== 4 || typeof f[2] !== 'string' || f[3] === null || typeof f[3] !== 'object' || Array.isArray(f[3])) throw badRequest('invalid filter: Fuzzy takes a query string and { max_edit_distance, case_sensitive? }');
228
+ const opts = f[3] as Record<string, unknown>;
229
+ const steps = opts.max_edit_distance;
230
+ if (!Array.isArray(steps) || steps.length === 0) throw badRequest('invalid filter: Fuzzy requires max_edit_distance');
231
+ const table = steps.map((s) => {
232
+ const e = s as Record<string, unknown>;
233
+ if (e === null || typeof e !== 'object' || ![0, 1, 2].includes(e.distance as number) || typeof e.min_query_chars !== 'number' || e.min_query_chars < 3 * ((e.distance as number) + 1)) {
234
+ throw badRequest('invalid filter: each max_edit_distance entry is {min_query_chars, distance}, distance 0, 1 or 2 and min_query_chars at least 3 · (distance + 1)');
235
+ }
236
+ return { min: e.min_query_chars as number, distance: e.distance as number };
237
+ }).sort((a, b) => a.min - b.min);
238
+ const sensitive = opts.case_sensitive !== false;
239
+ const norm = (s: string): string[] => [...(sensitive ? s.normalize('NFC') : s.normalize('NFC').toLowerCase())];
240
+ const q = norm(f[2]);
241
+ const allowed = table.filter((t) => q.length >= t.min).at(-1)?.distance;
242
+ if (allowed === undefined) return () => false;
243
+ const within = (value: string): boolean => substringDistance(q, norm(value)) <= allowed;
244
+ return (d) => { const v = attributeValue(d, attr); return typeof v === 'string' ? within(v) : Array.isArray(v) && v.some((x) => typeof x === 'string' && within(x)); };
245
+ }
246
+
247
+ /** The fewest Levenshtein edits turning the query into some substring of the text (Sellers' algorithm). */
248
+ function substringDistance(q: string[], t: string[]): number {
249
+ let prev = new Array(t.length + 1).fill(0);
250
+ for (let i = 1; i <= q.length; i++) {
251
+ const cur = [i];
252
+ for (let j = 1; j <= t.length; j++) cur[j] = Math.min(prev[j] + 1, cur[j - 1]! + 1, prev[j - 1] + (q[i - 1] === t[j - 1] ? 0 : 1));
253
+ prev = cur;
254
+ }
255
+ return Math.min(...prev);
256
+ }
257
+
258
+ function compareIdValues(a: unknown, b: unknown): number {
259
+ const ta = typeof a; const tb = typeof b;
260
+ if ((ta !== 'number' && ta !== 'string') || (tb !== 'number' && tb !== 'string')) return Number.NaN;
261
+ if (ta !== tb) return ta === 'number' ? -1 : 1;
262
+ return compareScalars(a, b);
263
+ }
264
+
265
+ function anyOf(v: unknown, test: (x: unknown) => boolean): boolean {
266
+ return Array.isArray(v) && v.some(test);
267
+ }
268
+
269
+ export function docTokens(ctx: FilterContext, attr: string, fts: FtsConfig, d: Doc): string[] {
270
+ const key = `${attr}\u0000${d.key}`;
271
+ let tokens = ctx.tokenCache.get(key);
272
+ if (tokens === undefined) {
273
+ tokens = tokenize(attributeValue(d, attr), fts);
274
+ ctx.tokenCache.set(key, tokens);
275
+ }
276
+ return tokens;
277
+ }
@@ -0,0 +1,81 @@
1
+ // FETCH-FIRST (runtime contract R12b): the pack's HTTP surface is a plain `(Request) => Promise<Response>`, and the
2
+ // local server is the kernel's serve seam around the SAME closure. The SDK never gzips a request body unless the
3
+ // caller opts into `compression: true` (client.ts: `options.compression === undefined ? false`), so the handler's
4
+ // text read of the body is faithful.
5
+ import { compileSurface, createDerivedFetch, createTwinFetchFromHandler, matchOperation, semanticsContext, serveHttp, statefulTwinManifest, type DerivedFetch, type DerivedOperation, type DerivedSurface } from '@volter/world-core';
6
+ import surface from './generated/surface.gen.json' with { type: 'json' };
7
+ import { keyGate } from './key-gate.ts';
8
+ import { manifest } from './manifest.ts';
9
+ import { DASHBOARD_HOST, dashboard } from './screens/dashboard.tsx';
10
+ import { turbopufferHandlers } from './semantics/namespaces.ts';
11
+ import { handleTurbopufferTwinRequest, normalizeTurbopufferPath } from './turbopuffer-twin.ts';
12
+
13
+ export interface TurbopufferTwinFetchOptions {
14
+ root?: string;
15
+ readOnly?: boolean;
16
+ /** The API key the twin demands. Omit to accept any non-empty key (a missing one still 401s). */
17
+ token?: string;
18
+ }
19
+
20
+ // the operation a context is opened under for the dashboard's pages, which answer no operation of the spec
21
+ const DASHBOARD: DerivedOperation = { id: 'TurbopufferDashboard', method: 'GET', path: '/dashboard', class: 'action' };
22
+ const routes = compileSurface(surface as DerivedSurface);
23
+
24
+ /**
25
+ * The pack's wire (docs/contributing/architecture.md, "Protocol 3"): the twin's discovery door (`GET /twin`) in
26
+ * front, turbopuffer's dashboard on turbopuffer.com (screens/dashboard.tsx: sign-in and API keys), then the key gate
27
+ * (key-gate.ts) and the derived dispatch over turbopuffer's spec. A leading region segment (`/aws-us-east-1/v2/...`) is
28
+ * dropped first: it is how a World points an unmodified client here through the SDK's own `TURBOPUFFER_BASE_URL`,
29
+ * whose `{region}` placeholder the client fills in (the pack descriptor's `endpointEnv`). The operations the twin
30
+ * serves go to their handlers (semantics/namespaces.ts); every other one, and a path the spec does not have, is the
31
+ * gap.
32
+ */
33
+ export function createTurbopufferTwinFetch(options: TurbopufferTwinFetchOptions = {}): DerivedFetch {
34
+ const answer = createTwinFetchFromHandler(handleTurbopufferTwinRequest, {
35
+ ...(options.root !== undefined ? { root: options.root } : {}),
36
+ ...(options.readOnly !== undefined ? { readOnly: options.readOnly } : {}),
37
+ ...(options.token !== undefined ? { handlerOptions: { token: options.token } } : {}),
38
+ manifest: statefulTwinManifest({
39
+ vendor: 'turbopuffer',
40
+ twinOf: 'the Turbopuffer API (v2 namespaces: write, query and multi-query, deleteAll, metadata, schema, namespace listing, cache warm hint)',
41
+ stores: 'namespaces (schema, distance metric) and their documents; queries filter, BM25-rank, order and count them deterministically',
42
+ identity: 'Send `Authorization: Bearer <any non-empty key>`, as the SDK does from TURBOPUFFER_API_KEY. A leading region path segment (/aws-us-east-1/v2/...) is accepted, which is how TURBOPUFFER_BASE_URL=<twin>/{region} reaches the twin.',
43
+ }),
44
+ });
45
+ const derived = createDerivedFetch({
46
+ surface: surface as DerivedSurface,
47
+ handlers: turbopufferHandlers(answer),
48
+ gap: (request) => turbopufferGap(request),
49
+ });
50
+ const scope = options.root !== undefined ? { root: options.root } : {};
51
+ const contextFor = (request: Request, operation: DerivedOperation = DASHBOARD) => semanticsContext(manifest, request, operation, scope);
52
+ return Object.assign(async (request: Request): Promise<Response> => {
53
+ const url = new URL(request.url);
54
+ // the vendor host a redirected request names (the injector's fetch path, a hosted World), else its own
55
+ const host = (request.headers.get('x-volter-twin-original-host') ?? url.host).split(':')[0]!.toLowerCase();
56
+ if (host === DASHBOARD_HOST) return (await dashboard(request, (r) => contextFor(r))) ?? new Response('Not Found', { status: 404, headers: { 'content-type': 'text/plain' } });
57
+ if (request.method === 'GET' && url.pathname.replace(/\/+$/, '') === '/twin') return answer(request);
58
+ const path = normalizeTurbopufferPath(url.pathname);
59
+ const routed = path === url.pathname ? request : new Request(Object.assign(url, { pathname: path }), request);
60
+ const matched = matchOperation(routes, routed.method, new URL(routed.url).pathname, new URL(routed.url).searchParams, routed.headers);
61
+ const refused = matched ? keyGate(await contextFor(new Request(routed.url), matched.operation), routed, matched.operation.id) : undefined;
62
+ return refused ?? derived(routed);
63
+ }, { owners: () => derived.owners() });
64
+ }
65
+
66
+ /** What the twin answers a request it has no operation for, in turbopuffer's error body
67
+ * (https://turbopuffer.com/docs/auth, "Error responses"). Where the documentation stops: no page prints the answer
68
+ * to an unknown path or an operation the twin does not model; 404 is the twin's. */
69
+ function turbopufferGap(request: Request): Response {
70
+ const url = new URL(request.url);
71
+ return Response.json({ status: 'error', error: `not found: ${request.method} ${url.pathname}` }, { status: 404 });
72
+ }
73
+
74
+ export async function createTurbopufferTwinServer(options: TurbopufferTwinFetchOptions & { port?: number } = {}): Promise<{ port: number; stop: () => void }> {
75
+ const server = await serveHttp({
76
+ port: options.port ?? 0,
77
+ idleTimeout: 60,
78
+ fetch: createTurbopufferTwinFetch(options),
79
+ });
80
+ return { port: server.port ?? options.port ?? 0, stop: () => server.stop(true) };
81
+ }
@@ -0,0 +1,104 @@
1
+ // ENGLISH STEMMING — `full_text_search: {stemming: true}` with `language: "english"`: "Language-specific stemming for
2
+ // the text" (https://turbopuffer.com/docs/write, full_text_search). turbopuffer's analyzer (alyze, which "Includes a
3
+ // complete analyzer implementation, with support for lowercasing, ASCII case folding, stemming & stopword removal",
4
+ // github.com/turbopuffer/alyze src/analyze/mod.rs) stems with the `rust_stemmers` crate's Snowball algorithms; for
5
+ // English that is the Snowball English ("Porter2") stemmer, written out here from its published definition
6
+ // (https://snowballstem.org/algorithms/english/stemmer.html). Other languages' stemmers are not modelled.
7
+
8
+ const VOWELS = new Set(['a', 'e', 'i', 'o', 'u', 'y']);
9
+ const isVowel = (w: string, i: number): boolean => VOWELS.has(w[i]!);
10
+ const DOUBLES = ['bb', 'dd', 'ff', 'gg', 'mm', 'nn', 'pp', 'rr', 'tt'];
11
+ const LI_ENDINGS = new Set(['c', 'd', 'e', 'g', 'h', 'k', 'm', 'n', 'r', 't']);
12
+
13
+ const EXCEPTIONS: Record<string, string> = {
14
+ skis: 'ski', skies: 'sky', dying: 'die', lying: 'lie', tying: 'tie', idly: 'idl', gently: 'gentl', ugly: 'ugli',
15
+ early: 'earli', only: 'onli', singly: 'singl', sky: 'sky', news: 'news', howe: 'howe', atlas: 'atlas', cosmos: 'cosmos',
16
+ bias: 'bias', andes: 'andes',
17
+ };
18
+ const AFTER_1A = new Set(['inning', 'outing', 'canning', 'herring', 'earring', 'proceed', 'exceed', 'succeed']);
19
+
20
+ /** The start of the region after the first non-vowel following a vowel, from `from`. */
21
+ function regionAfter(w: string, from: number): number {
22
+ for (let i = from + 1; i < w.length; i++) if (!isVowel(w, i) && isVowel(w, i - 1)) return i + 1;
23
+ return w.length;
24
+ }
25
+
26
+ /** A short syllable ending at index `i` (inclusive). */
27
+ function shortSyllableAt(w: string, i: number): boolean {
28
+ if (i === 1) return isVowel(w, 0) && !isVowel(w, 1);
29
+ if (i < 2) return false;
30
+ return !isVowel(w, i - 2) && isVowel(w, i - 1) && !isVowel(w, i) && !['w', 'x', 'Y'].includes(w[i]!);
31
+ }
32
+
33
+ /** Step 2's rewrite of a suffix in R1: ogi to og after l, li deleted after a valid li-ending, the rest by the table. */
34
+ function step2(w: string, s2: string, table: Record<string, string>): string {
35
+ if (s2 === 'ogi') return w[w.length - 4] === 'l' ? `${w.slice(0, -3)}og` : w;
36
+ if (s2 === 'li') return LI_ENDINGS.has(w[w.length - 3] ?? '') ? w.slice(0, -2) : w;
37
+ return w.slice(0, -s2.length) + table[s2]!;
38
+ }
39
+
40
+ /** Step 3's rewrite of a suffix in R1 by the table (ative, which needs R2, is the caller's). */
41
+ function step3(w: string, s3: string, table: Record<string, string>): string {
42
+ return w.slice(0, -s3.length) + table[s3]!;
43
+ }
44
+
45
+ export function stemEnglish(word: string): string {
46
+ if (word.length <= 2) return word;
47
+ if (EXCEPTIONS[word] !== undefined) return EXCEPTIONS[word]!;
48
+ let w = word.startsWith("'") ? word.slice(1) : word;
49
+ // initial y, and y after a vowel, are consonants: marked Y
50
+ w = w.replace(/^y/, 'Y').replace(/([aeiouy])y/g, '$1Y');
51
+ let r1 = /^(gener|commun|arsen)/.test(w) ? /^(gener|commun|arsen)/.exec(w)![0].length : regionAfter(w, 0);
52
+ const r2 = (): number => regionAfter(w, r1);
53
+ const inR1 = (suffix: string): boolean => w.length - suffix.length >= r1;
54
+ const inR2 = (suffix: string): boolean => w.length - suffix.length >= r2();
55
+ const ends = (s: string): boolean => w.endsWith(s);
56
+ const cut = (n: number): void => { w = w.slice(0, w.length - n); };
57
+ const longest = (list: string[]): string | undefined => list.filter(ends).sort((a, b) => b.length - a.length)[0];
58
+ const hasVowelBefore = (end: number): boolean => { for (let i = 0; i < end; i++) if (isVowel(w, i)) return true; return false; };
59
+
60
+ // Step 0
61
+ const s0 = longest(["'s'", "'s", "'"]);
62
+ if (s0) cut(s0.length);
63
+ // Step 1a
64
+ const s1a = longest(['sses', 'ied', 'ies', 'us', 'ss', 's']);
65
+ if (s1a === 'sses') cut(2);
66
+ else if (s1a === 'ied' || s1a === 'ies') { cut(3); w += w.length > 1 ? 'i' : 'ie'; }
67
+ else if (s1a === 's') { if (hasVowelBefore(w.length - 2)) cut(1); }
68
+ if (AFTER_1A.has(w)) return w;
69
+ // Step 1b
70
+ const s1b = longest(['eedly', 'eed', 'ingly', 'edly', 'ing', 'ed']);
71
+ if (s1b === 'eed' || s1b === 'eedly') { if (inR1(s1b)) { cut(s1b.length); w += 'ee'; } }
72
+ else if (s1b && hasVowelBefore(w.length - s1b.length)) {
73
+ cut(s1b.length);
74
+ if (ends('at') || ends('bl') || ends('iz')) w += 'e';
75
+ else if (DOUBLES.some(ends)) cut(1);
76
+ else if (r1 >= w.length && shortSyllableAt(w, w.length - 1)) w += 'e';
77
+ }
78
+ // Step 1c
79
+ if ((ends('y') || ends('Y')) && w.length > 2 && !isVowel(w, w.length - 2)) w = `${w.slice(0, -1)}i`;
80
+ // Step 2
81
+ const STEP2: Record<string, string> = {
82
+ tional: 'tion', enci: 'ence', anci: 'ance', abli: 'able', entli: 'ent', izer: 'ize', ization: 'ize', ational: 'ate',
83
+ ation: 'ate', ator: 'ate', alism: 'al', aliti: 'al', alli: 'al', fulness: 'ful', ousli: 'ous', ousness: 'ous',
84
+ iveness: 'ive', iviti: 'ive', biliti: 'ble', bli: 'ble', ogi: 'og', fulli: 'ful', lessli: 'less', li: '',
85
+ };
86
+ const s2 = longest(Object.keys(STEP2));
87
+ if (s2 && inR1(s2)) w = step2(w, s2, STEP2);
88
+ // Step 3
89
+ const STEP3: Record<string, string> = { tional: 'tion', ational: 'ate', alize: 'al', icate: 'ic', iciti: 'ic', ical: 'ic', ful: '', ness: '', ative: '' };
90
+ const s3 = longest(Object.keys(STEP3));
91
+ if (s3 && inR1(s3)) w = s3 === 'ative' ? (inR2(s3) ? w.slice(0, -5) : w) : step3(w, s3, STEP3);
92
+ // Step 4
93
+ const s4 = longest(['al', 'ance', 'ence', 'er', 'ic', 'able', 'ible', 'ant', 'ement', 'ment', 'ent', 'ism', 'ate', 'iti', 'ous', 'ive', 'ize', 'ion']);
94
+ if (s4 && inR2(s4)) {
95
+ if (s4 === 'ion') { if (['s', 't'].includes(w[w.length - 4] ?? '')) cut(3); }
96
+ else cut(s4.length);
97
+ }
98
+ // Step 5
99
+ if (ends('e')) {
100
+ if (inR2('e') || (inR1('e') && !shortSyllableAt(w, w.length - 2))) cut(1);
101
+ } else if (ends('l') && inR2('l') && w[w.length - 2] === 'l') cut(1);
102
+ r1 = 0;
103
+ return w.replace(/Y/g, 'y');
104
+ }