@volter/twin-turbopuffer 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (52) hide show
  1. package/LICENSE +202 -0
  2. package/README.md +145 -0
  3. package/dist/src/cli.d.ts +2 -0
  4. package/dist/src/cli.js +27 -0
  5. package/dist/src/generated/surface.gen.json +1 -0
  6. package/dist/src/generated/ui.gen.json +1 -0
  7. package/dist/src/index.d.ts +11 -0
  8. package/dist/src/index.js +65 -0
  9. package/dist/src/key-gate.d.ts +3 -0
  10. package/dist/src/key-gate.js +39 -0
  11. package/dist/src/manifest.d.ts +2 -0
  12. package/dist/src/manifest.js +28 -0
  13. package/dist/src/screens/dashboard.d.ts +11 -0
  14. package/dist/src/screens/dashboard.js +191 -0
  15. package/dist/src/semantics/namespaces.d.ts +5 -0
  16. package/dist/src/semantics/namespaces.js +17 -0
  17. package/dist/src/turbopuffer-capabilities.d.ts +6 -0
  18. package/dist/src/turbopuffer-capabilities.js +442 -0
  19. package/dist/src/turbopuffer-conformance.d.ts +8 -0
  20. package/dist/src/turbopuffer-conformance.js +102 -0
  21. package/dist/src/turbopuffer-connector.d.ts +34 -0
  22. package/dist/src/turbopuffer-connector.js +152 -0
  23. package/dist/src/turbopuffer-filter.d.ts +57 -0
  24. package/dist/src/turbopuffer-filter.js +286 -0
  25. package/dist/src/turbopuffer-server.d.ts +23 -0
  26. package/dist/src/turbopuffer-server.js +72 -0
  27. package/dist/src/turbopuffer-stem.d.ts +1 -0
  28. package/dist/src/turbopuffer-stem.js +133 -0
  29. package/dist/src/turbopuffer-store.d.ts +80 -0
  30. package/dist/src/turbopuffer-store.js +1304 -0
  31. package/dist/src/turbopuffer-text.d.ts +44 -0
  32. package/dist/src/turbopuffer-text.js +189 -0
  33. package/dist/src/turbopuffer-twin.d.ts +25 -0
  34. package/dist/src/turbopuffer-twin.js +406 -0
  35. package/package.json +56 -0
  36. package/src/cli.ts +28 -0
  37. package/src/generated/surface.gen.json +1 -0
  38. package/src/generated/ui.gen.json +1 -0
  39. package/src/index.ts +92 -0
  40. package/src/key-gate.ts +39 -0
  41. package/src/manifest.ts +63 -0
  42. package/src/screens/dashboard.tsx +214 -0
  43. package/src/semantics/namespaces.ts +30 -0
  44. package/src/turbopuffer-capabilities.ts +455 -0
  45. package/src/turbopuffer-conformance.ts +101 -0
  46. package/src/turbopuffer-connector.ts +153 -0
  47. package/src/turbopuffer-filter.ts +277 -0
  48. package/src/turbopuffer-server.ts +81 -0
  49. package/src/turbopuffer-stem.ts +104 -0
  50. package/src/turbopuffer-store.ts +1157 -0
  51. package/src/turbopuffer-text.ts +204 -0
  52. package/src/turbopuffer-twin.ts +429 -0
@@ -0,0 +1,44 @@
1
+ export declare const TOKENIZERS: readonly ["pre_tokenized_array", "word_v0", "word_v1", "word_v2", "word_v3", "word_v4"];
2
+ /** The tokenizers this twin models. Every other one in `TOKENIZERS` is refused with a 400. */
3
+ export declare const MODELED_TOKENIZERS: readonly string[];
4
+ /** The vendor's default tokenizer (namespaces.ts: "Defaults to `word_v4`"). */
5
+ export declare const DEFAULT_TOKENIZER = "word_v4";
6
+ export type FtsConfig = {
7
+ tokenizer: string;
8
+ case_sensitive: boolean;
9
+ remove_stopwords: boolean;
10
+ stemming: boolean;
11
+ ascii_folding: boolean;
12
+ language: string;
13
+ max_token_length: number;
14
+ k1: number;
15
+ b: number;
16
+ k3?: number;
17
+ };
18
+ export declare const DEFAULT_FTS: FtsConfig;
19
+ /**
20
+ * Tokenize one value (a string, or a `[]string`, whose elements are tokenized and concatenated).
21
+ * `keepStopwords` is used for the final prefix token of a `last_as_prefix` query, which must not
22
+ * vanish because the user has so far typed only `an` of `anna`.
23
+ */
24
+ export declare function tokenize(value: unknown, cfg: FtsConfig, opts?: {
25
+ keepStopwords?: boolean;
26
+ }): string[];
27
+ /** A query's terms, split into the exact terms and the optional trailing prefix. */
28
+ export type QueryTerms = {
29
+ exact: string[];
30
+ prefix: string | null;
31
+ };
32
+ export declare function queryTerms(query: unknown, cfg: FtsConfig, lastAsPrefix: boolean): QueryTerms;
33
+ /** Does a document's token list satisfy the terms? `all` for ContainsAllTokens, else any. */
34
+ export declare function tokensMatch(docTokens: readonly string[], terms: QueryTerms, mode: 'all' | 'any'): boolean;
35
+ /** The per-attribute corpus statistics BM25 needs, computed once per query over the namespace. */
36
+ export type Corpus = {
37
+ n: number;
38
+ avgdl: number;
39
+ df: Map<string, number>;
40
+ docs: Map<string, string[]>;
41
+ };
42
+ export declare function buildCorpus(entries: Iterable<[string, unknown]>, cfg: FtsConfig): Corpus;
43
+ /** BM25 of one document against the query terms; 0 means "no term matched". */
44
+ export declare function bm25Score(key: string, corpus: Corpus, terms: QueryTerms, cfg: FtsConfig): number;
@@ -0,0 +1,189 @@
1
+ // FULL-TEXT SEARCH — the tokenizer and the BM25 scorer behind `rank_by: [attr, 'BM25', q]` and the
2
+ // `ContainsAllTokens` / `ContainsAnyToken` filters.
3
+ //
4
+ // ── WHAT IS GROUNDED, AND WHERE THE EVIDENCE STOPS ──────────────────────────────────────────
5
+ // Grounded in the installed `@turbopuffer/turbopuffer@2.8.0` (src/resources/namespaces.ts,
6
+ // `FullTextSearchConfig`): the defaults — `k1` 1.2, `b` 0.75, case-insensitive, `language`
7
+ // english, `stemming` false, `ascii_folding` false, `max_token_length` 39
8
+ // BYTES (tokens longer are filtered out), and "by default, BM25-enabled attributes are not
9
+ // filterable". `remove_stopwords` defaults to false: "Defaults to false (i.e. keep common words)"
10
+ // (https://turbopuffer.com/docs/write, full_text_search) since "`remove_stopwords` now defaults to `false`"
11
+ // (https://turbopuffer.com/docs/roadmap, January 2026); the spec's description, and the client's doc comment, still
12
+ // say true. Grounded in Dub's provider (apps/web/lib/api/partners/search/providers/
13
+ // turbopuffer.ts, measured against the real service): `word_v2` splits a URL into its
14
+ // components (so `scottdigital` matches `https://www.scottdigital-42.techcorp.io`), and "a
15
+ // single-token `last_as_prefix` query scores every match at exactly 1".
16
+ //
17
+ // EXTRAPOLATED (deterministic, not the vendor's exact algorithm): `word_v2` is modelled as
18
+ // maximal runs of Unicode letters, digits, marks and `_`; the English stopword list is the
19
+ // classic Lucene/tantivy 33-word list; BM25 is the textbook Okapi form with the Lucene IDF
20
+ // `ln(1 + (N − n + 0.5)/(n + 0.5))` over the namespace's documents that carry the attribute, and a
21
+ // `last_as_prefix` final token contributes a constant 1 per matching document. Scores are
22
+ // therefore deterministic and ordered sensibly, but they are not byte-equal to Turbopuffer's.
23
+ // Tokenizers other than `word_v2` and `pre_tokenized_array`, stemming, and non-English languages
24
+ // are refused (400) rather than silently approximated.
25
+ import { stemEnglish } from "./turbopuffer-stem.js";
26
+ export const TOKENIZERS = ['pre_tokenized_array', 'word_v0', 'word_v1', 'word_v2', 'word_v3', 'word_v4'];
27
+ /** The tokenizers this twin models. Every other one in `TOKENIZERS` is refused with a 400. */
28
+ export const MODELED_TOKENIZERS = ['word_v4', 'word_v3', 'word_v2', 'pre_tokenized_array'];
29
+ /** The vendor's default tokenizer (namespaces.ts: "Defaults to `word_v4`"). */
30
+ export const DEFAULT_TOKENIZER = 'word_v4';
31
+ export const DEFAULT_FTS = {
32
+ tokenizer: DEFAULT_TOKENIZER,
33
+ case_sensitive: false,
34
+ remove_stopwords: false,
35
+ stemming: false,
36
+ ascii_folding: false,
37
+ language: 'english',
38
+ max_token_length: 39,
39
+ k1: 1.2,
40
+ b: 0.75,
41
+ };
42
+ const ENGLISH_STOPWORDS = new Set([
43
+ 'a', 'an', 'and', 'are', 'as', 'at', 'be', 'but', 'by', 'for', 'if', 'in', 'into', 'is', 'it', 'no', 'not',
44
+ 'of', 'on', 'or', 'such', 'that', 'the', 'their', 'then', 'there', 'these', 'they', 'this', 'to', 'was',
45
+ 'will', 'with',
46
+ ]);
47
+ const WORD = /[\p{L}\p{N}\p{M}_]+/gu;
48
+ const encoder = new TextEncoder();
49
+ function normalizeToken(token, cfg) {
50
+ let t = cfg.case_sensitive ? token : token.toLowerCase();
51
+ if (cfg.ascii_folding)
52
+ t = t.normalize('NFD').replace(/\p{M}+/gu, '');
53
+ return t;
54
+ }
55
+ /**
56
+ * Tokenize one value (a string, or a `[]string`, whose elements are tokenized and concatenated).
57
+ * `keepStopwords` is used for the final prefix token of a `last_as_prefix` query, which must not
58
+ * vanish because the user has so far typed only `an` of `anna`.
59
+ */
60
+ export function tokenize(value, cfg, opts = {}) {
61
+ const parts = Array.isArray(value) ? value.filter((v) => typeof v === 'string') : typeof value === 'string' ? [value] : [];
62
+ if (cfg.tokenizer === 'word_v4' || cfg.tokenizer === 'word_v3')
63
+ return parts.flatMap((part) => uax29Tokens(part, cfg, opts));
64
+ const out = [];
65
+ for (const part of parts) {
66
+ const raw = cfg.tokenizer === 'pre_tokenized_array' ? [part] : (part.match(WORD) ?? []);
67
+ for (const r of raw) {
68
+ const t = normalizeToken(r, cfg);
69
+ if (t === '' || encoder.encode(t).length > cfg.max_token_length)
70
+ continue;
71
+ if (cfg.remove_stopwords && !opts.keepStopwords && ENGLISH_STOPWORDS.has(t))
72
+ continue;
73
+ out.push(cfg.stemming ? stemEnglish(t) : t);
74
+ }
75
+ }
76
+ return out;
77
+ }
78
+ // ── word_v4 and word_v3 ─────────────────────────────────────────────────────────────────────
79
+ // "The `word_v4` and `word_v3` tokenizers use Unicode v17.0 text segmentation rules (UAX #29) for accurate
80
+ // segmentation across most languages, scripts, and emojis. `word_v4` is the current default for new namespaces; it
81
+ // behaves like `word_v3`, but is roughly 3x faster and fixes a few tokenization edge cases. It's powered by our
82
+ // open-source alyze library" (https://turbopuffer.com/docs/fts, Tokenizers). Modelled on alyze at
83
+ // 1de437c8604b751f26e7061070b9d2b35aebdc73 (github.com/turbopuffer/alyze, src/analyze/mod.rs, src/uax29/word/mod.rs):
84
+ // - word boundaries by UAX #29, and only "word-like" segments become tokens: a segment "is 'word-like' if it contains
85
+ // any char that is ALetter, HebrewLetter, or Numeric, Ideographic or Extended_Pictographic, Other_Number general
86
+ // category, [or] a character whose Script is something meaningful … as opposed to Script=Common/Inherited/Unknown";
87
+ // - then, in alyze's order: the maximum token length (alyze admits a token within the limit in bytes or in
88
+ // characters, `s.len() <= max || s.chars().nth(max).is_none()`), lowercasing unless case_sensitive, stopword removal
89
+ // (the Lucene English list, as word_v2's), and ASCII folding, lowercased again.
90
+ // Where the evidence stops: the twin segments with the host's Intl.Segmenter (ICU's UAX #29, with runs of ideographs
91
+ // and hiragana split back into characters as UAX #29's rules leave them; ICU's dictionary segmentation of Thai, Lao,
92
+ // Khmer and Burmese is kept, which alyze may not do), whose Unicode version is
93
+ // the runtime's, not 17.0; alyze's own DFA may differ from ICU on edge cases, and the "few tokenization edge cases"
94
+ // word_v4 fixes over word_v3 are not documented, so the two are one tokenizer here; ASCII folding strips combining
95
+ // marks after NFD, where alyze's `ascii_fold` may map more characters (ß, æ).
96
+ const segmenter = new Intl.Segmenter('und', { granularity: 'word' });
97
+ const WORD_LIKE = /[\p{L}\p{Nd}\p{No}\p{Extended_Pictographic}\p{Ideographic}]|[^\p{Script=Common}\p{Script=Inherited}\p{Script=Unknown}]/u;
98
+ function uax29Tokens(text, cfg, opts) {
99
+ const out = [];
100
+ // ICU joins runs of ideographs and hiragana by dictionary; UAX #29's rules break between them (neither is ALetter,
101
+ // and no rule joins them), so such a run is split back into its characters
102
+ const segments = [...segmenter.segment(text)].flatMap(({ segment }) => (/^[\p{Ideographic}\p{Script=Hiragana}]+$/u.test(segment) ? [...segment] : [segment]));
103
+ for (const segment of segments) {
104
+ if (!WORD_LIKE.test(segment))
105
+ continue;
106
+ if (!(encoder.encode(segment).length <= cfg.max_token_length || [...segment].length <= cfg.max_token_length))
107
+ continue;
108
+ let t = cfg.case_sensitive ? segment : segment.toLowerCase();
109
+ if (cfg.remove_stopwords && !opts.keepStopwords && ENGLISH_STOPWORDS.has(t))
110
+ continue;
111
+ if (cfg.stemming)
112
+ t = stemEnglish(t);
113
+ out.push(cfg.ascii_folding && /[^\x00-\x7f]/.test(t) ? foldToken(t, cfg) : t);
114
+ }
115
+ return out;
116
+ }
117
+ /** alyze's ASCII folding of a non-ASCII token, lowercased again unless case_sensitive ("ASCII folding can produce
118
+ * uppercase ASCII characters, so we'll lowercase again if case folding is enabled", alyze src/analyze/mod.rs). */
119
+ function foldToken(t, cfg) {
120
+ const folded = t.normalize('NFD').replace(/\p{M}+/gu, '');
121
+ return cfg.case_sensitive ? folded : folded.toLowerCase();
122
+ }
123
+ export function queryTerms(query, cfg, lastAsPrefix) {
124
+ if (!lastAsPrefix)
125
+ return { exact: [...new Set(tokenize(query, cfg))], prefix: null };
126
+ // The prefix is the LAST token of the raw input, kept even when it is a stopword.
127
+ const all = tokenize(query, cfg, { keepStopwords: true });
128
+ if (all.length === 0)
129
+ return { exact: [], prefix: null };
130
+ const prefix = all[all.length - 1];
131
+ const head = cfg.remove_stopwords ? all.slice(0, -1).filter((t) => !ENGLISH_STOPWORDS.has(t)) : all.slice(0, -1);
132
+ return { exact: [...new Set(head)], prefix };
133
+ }
134
+ /** Does a document's token list satisfy the terms? `all` for ContainsAllTokens, else any. */
135
+ export function tokensMatch(docTokens, terms, mode) {
136
+ const set = new Set(docTokens);
137
+ const checks = terms.exact.map((t) => set.has(t));
138
+ if (terms.prefix !== null) {
139
+ const p = terms.prefix;
140
+ checks.push(docTokens.some((t) => t.startsWith(p)));
141
+ }
142
+ // A query whose every token is dropped (all stopwords, all over-length) matches nothing, in
143
+ // either mode — the same documents BM25 would score above zero. The vendor's answer here is
144
+ // unverified; this keeps the twin's count and its ranked search in agreement.
145
+ if (checks.length === 0)
146
+ return false;
147
+ return mode === 'all' ? checks.every(Boolean) : checks.some(Boolean);
148
+ }
149
+ export function buildCorpus(entries, cfg) {
150
+ const docs = new Map();
151
+ const df = new Map();
152
+ let total = 0;
153
+ for (const [key, value] of entries) {
154
+ if (value === null || value === undefined)
155
+ continue;
156
+ const tokens = tokenize(value, cfg);
157
+ docs.set(key, tokens);
158
+ total += tokens.length;
159
+ for (const t of new Set(tokens))
160
+ df.set(t, (df.get(t) ?? 0) + 1);
161
+ }
162
+ return { n: docs.size, avgdl: docs.size === 0 ? 0 : total / docs.size, df, docs };
163
+ }
164
+ /** BM25 of one document against the query terms; 0 means "no term matched". */
165
+ export function bm25Score(key, corpus, terms, cfg) {
166
+ const tokens = corpus.docs.get(key);
167
+ if (tokens === undefined || tokens.length === 0)
168
+ return 0;
169
+ let score = 0;
170
+ const dl = tokens.length;
171
+ for (const term of terms.exact) {
172
+ let tf = 0;
173
+ for (const t of tokens)
174
+ if (t === term)
175
+ tf++;
176
+ if (tf === 0)
177
+ continue;
178
+ const n = corpus.df.get(term) ?? 0;
179
+ const idf = Math.log(1 + (corpus.n - n + 0.5) / (n + 0.5));
180
+ const norm = corpus.avgdl === 0 ? 1 : 1 - cfg.b + cfg.b * (dl / corpus.avgdl);
181
+ score += idf * ((tf * (cfg.k1 + 1)) / (tf + cfg.k1 * norm));
182
+ }
183
+ if (terms.prefix !== null) {
184
+ const p = terms.prefix;
185
+ if (tokens.some((t) => t.startsWith(p)))
186
+ score += 1;
187
+ }
188
+ return score;
189
+ }
@@ -0,0 +1,25 @@
1
+ export type TurbopufferTwinRequest = {
2
+ method: string;
3
+ path: string;
4
+ body?: string;
5
+ headers?: Record<string, string>;
6
+ root?: string;
7
+ occurredAt?: string;
8
+ readOnly?: boolean;
9
+ /** The API key the twin demands. Omit to accept any non-empty key (a missing one still 401s). */
10
+ token?: string;
11
+ };
12
+ export type TurbopufferTwinResponse = {
13
+ status: number;
14
+ body: unknown;
15
+ headers?: Record<string, string>;
16
+ };
17
+ /** `Authorization: Bearer <key>` (client.ts `authHeaders`). An empty key is no key. */
18
+ export declare function extractTurbopufferApiKey(headers: Record<string, string> | undefined): string | null;
19
+ /** Collapse duplicate slashes and drop a leading region segment (see the header). */
20
+ export declare function normalizeTurbopufferPath(rawPath: string): string;
21
+ export declare function handleTurbopufferTwinRequest(req: TurbopufferTwinRequest): Promise<TurbopufferTwinResponse>;
22
+ export declare function turbopufferTwinSnapshot(): {
23
+ resourceTypes: readonly string[];
24
+ implementedEndpoints: readonly string[];
25
+ };