@volter/twin-turbopuffer 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +202 -0
- package/README.md +145 -0
- package/dist/src/cli.d.ts +2 -0
- package/dist/src/cli.js +27 -0
- package/dist/src/generated/surface.gen.json +1 -0
- package/dist/src/generated/ui.gen.json +1 -0
- package/dist/src/index.d.ts +11 -0
- package/dist/src/index.js +65 -0
- package/dist/src/key-gate.d.ts +3 -0
- package/dist/src/key-gate.js +39 -0
- package/dist/src/manifest.d.ts +2 -0
- package/dist/src/manifest.js +28 -0
- package/dist/src/screens/dashboard.d.ts +11 -0
- package/dist/src/screens/dashboard.js +191 -0
- package/dist/src/semantics/namespaces.d.ts +5 -0
- package/dist/src/semantics/namespaces.js +17 -0
- package/dist/src/turbopuffer-capabilities.d.ts +6 -0
- package/dist/src/turbopuffer-capabilities.js +442 -0
- package/dist/src/turbopuffer-conformance.d.ts +8 -0
- package/dist/src/turbopuffer-conformance.js +102 -0
- package/dist/src/turbopuffer-connector.d.ts +34 -0
- package/dist/src/turbopuffer-connector.js +152 -0
- package/dist/src/turbopuffer-filter.d.ts +57 -0
- package/dist/src/turbopuffer-filter.js +286 -0
- package/dist/src/turbopuffer-server.d.ts +23 -0
- package/dist/src/turbopuffer-server.js +72 -0
- package/dist/src/turbopuffer-stem.d.ts +1 -0
- package/dist/src/turbopuffer-stem.js +133 -0
- package/dist/src/turbopuffer-store.d.ts +80 -0
- package/dist/src/turbopuffer-store.js +1304 -0
- package/dist/src/turbopuffer-text.d.ts +44 -0
- package/dist/src/turbopuffer-text.js +189 -0
- package/dist/src/turbopuffer-twin.d.ts +25 -0
- package/dist/src/turbopuffer-twin.js +406 -0
- package/package.json +56 -0
- package/src/cli.ts +28 -0
- package/src/generated/surface.gen.json +1 -0
- package/src/generated/ui.gen.json +1 -0
- package/src/index.ts +92 -0
- package/src/key-gate.ts +39 -0
- package/src/manifest.ts +63 -0
- package/src/screens/dashboard.tsx +214 -0
- package/src/semantics/namespaces.ts +30 -0
- package/src/turbopuffer-capabilities.ts +455 -0
- package/src/turbopuffer-conformance.ts +101 -0
- package/src/turbopuffer-connector.ts +153 -0
- package/src/turbopuffer-filter.ts +277 -0
- package/src/turbopuffer-server.ts +81 -0
- package/src/turbopuffer-stem.ts +104 -0
- package/src/turbopuffer-store.ts +1157 -0
- package/src/turbopuffer-text.ts +204 -0
- package/src/turbopuffer-twin.ts +429 -0
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
export declare const TOKENIZERS: readonly ["pre_tokenized_array", "word_v0", "word_v1", "word_v2", "word_v3", "word_v4"];
|
|
2
|
+
/** The tokenizers this twin models. Every other one in `TOKENIZERS` is refused with a 400. */
|
|
3
|
+
export declare const MODELED_TOKENIZERS: readonly string[];
|
|
4
|
+
/** The vendor's default tokenizer (namespaces.ts: "Defaults to `word_v4`"). */
|
|
5
|
+
export declare const DEFAULT_TOKENIZER = "word_v4";
|
|
6
|
+
export type FtsConfig = {
|
|
7
|
+
tokenizer: string;
|
|
8
|
+
case_sensitive: boolean;
|
|
9
|
+
remove_stopwords: boolean;
|
|
10
|
+
stemming: boolean;
|
|
11
|
+
ascii_folding: boolean;
|
|
12
|
+
language: string;
|
|
13
|
+
max_token_length: number;
|
|
14
|
+
k1: number;
|
|
15
|
+
b: number;
|
|
16
|
+
k3?: number;
|
|
17
|
+
};
|
|
18
|
+
export declare const DEFAULT_FTS: FtsConfig;
|
|
19
|
+
/**
|
|
20
|
+
* Tokenize one value (a string, or a `[]string`, whose elements are tokenized and concatenated).
|
|
21
|
+
* `keepStopwords` is used for the final prefix token of a `last_as_prefix` query, which must not
|
|
22
|
+
* vanish because the user has so far typed only `an` of `anna`.
|
|
23
|
+
*/
|
|
24
|
+
export declare function tokenize(value: unknown, cfg: FtsConfig, opts?: {
|
|
25
|
+
keepStopwords?: boolean;
|
|
26
|
+
}): string[];
|
|
27
|
+
/** A query's terms, split into the exact terms and the optional trailing prefix. */
|
|
28
|
+
export type QueryTerms = {
|
|
29
|
+
exact: string[];
|
|
30
|
+
prefix: string | null;
|
|
31
|
+
};
|
|
32
|
+
export declare function queryTerms(query: unknown, cfg: FtsConfig, lastAsPrefix: boolean): QueryTerms;
|
|
33
|
+
/** Does a document's token list satisfy the terms? `all` for ContainsAllTokens, else any. */
|
|
34
|
+
export declare function tokensMatch(docTokens: readonly string[], terms: QueryTerms, mode: 'all' | 'any'): boolean;
|
|
35
|
+
/** The per-attribute corpus statistics BM25 needs, computed once per query over the namespace. */
|
|
36
|
+
export type Corpus = {
|
|
37
|
+
n: number;
|
|
38
|
+
avgdl: number;
|
|
39
|
+
df: Map<string, number>;
|
|
40
|
+
docs: Map<string, string[]>;
|
|
41
|
+
};
|
|
42
|
+
export declare function buildCorpus(entries: Iterable<[string, unknown]>, cfg: FtsConfig): Corpus;
|
|
43
|
+
/** BM25 of one document against the query terms; 0 means "no term matched". */
|
|
44
|
+
export declare function bm25Score(key: string, corpus: Corpus, terms: QueryTerms, cfg: FtsConfig): number;
|
|
@@ -0,0 +1,189 @@
|
|
|
1
|
+
// FULL-TEXT SEARCH — the tokenizer and the BM25 scorer behind `rank_by: [attr, 'BM25', q]` and the
|
|
2
|
+
// `ContainsAllTokens` / `ContainsAnyToken` filters.
|
|
3
|
+
//
|
|
4
|
+
// ── WHAT IS GROUNDED, AND WHERE THE EVIDENCE STOPS ──────────────────────────────────────────
|
|
5
|
+
// Grounded in the installed `@turbopuffer/turbopuffer@2.8.0` (src/resources/namespaces.ts,
|
|
6
|
+
// `FullTextSearchConfig`): the defaults — `k1` 1.2, `b` 0.75, case-insensitive, `language`
|
|
7
|
+
// english, `stemming` false, `ascii_folding` false, `max_token_length` 39
|
|
8
|
+
// BYTES (tokens longer are filtered out), and "by default, BM25-enabled attributes are not
|
|
9
|
+
// filterable". `remove_stopwords` defaults to false: "Defaults to false (i.e. keep common words)"
|
|
10
|
+
// (https://turbopuffer.com/docs/write, full_text_search) since "`remove_stopwords` now defaults to `false`"
|
|
11
|
+
// (https://turbopuffer.com/docs/roadmap, January 2026); the spec's description, and the client's doc comment, still
|
|
12
|
+
// say true. Grounded in Dub's provider (apps/web/lib/api/partners/search/providers/
|
|
13
|
+
// turbopuffer.ts, measured against the real service): `word_v2` splits a URL into its
|
|
14
|
+
// components (so `scottdigital` matches `https://www.scottdigital-42.techcorp.io`), and "a
|
|
15
|
+
// single-token `last_as_prefix` query scores every match at exactly 1".
|
|
16
|
+
//
|
|
17
|
+
// EXTRAPOLATED (deterministic, not the vendor's exact algorithm): `word_v2` is modelled as
|
|
18
|
+
// maximal runs of Unicode letters, digits, marks and `_`; the English stopword list is the
|
|
19
|
+
// classic Lucene/tantivy 33-word list; BM25 is the textbook Okapi form with the Lucene IDF
|
|
20
|
+
// `ln(1 + (N − n + 0.5)/(n + 0.5))` over the namespace's documents that carry the attribute, and a
|
|
21
|
+
// `last_as_prefix` final token contributes a constant 1 per matching document. Scores are
|
|
22
|
+
// therefore deterministic and ordered sensibly, but they are not byte-equal to Turbopuffer's.
|
|
23
|
+
// Tokenizers other than `word_v2` and `pre_tokenized_array`, stemming, and non-English languages
|
|
24
|
+
// are refused (400) rather than silently approximated.
|
|
25
|
+
import { stemEnglish } from "./turbopuffer-stem.js";
|
|
26
|
+
export const TOKENIZERS = ['pre_tokenized_array', 'word_v0', 'word_v1', 'word_v2', 'word_v3', 'word_v4'];
|
|
27
|
+
/** The tokenizers this twin models. Every other one in `TOKENIZERS` is refused with a 400. */
|
|
28
|
+
export const MODELED_TOKENIZERS = ['word_v4', 'word_v3', 'word_v2', 'pre_tokenized_array'];
|
|
29
|
+
/** The vendor's default tokenizer (namespaces.ts: "Defaults to `word_v4`"). */
|
|
30
|
+
export const DEFAULT_TOKENIZER = 'word_v4';
|
|
31
|
+
export const DEFAULT_FTS = {
|
|
32
|
+
tokenizer: DEFAULT_TOKENIZER,
|
|
33
|
+
case_sensitive: false,
|
|
34
|
+
remove_stopwords: false,
|
|
35
|
+
stemming: false,
|
|
36
|
+
ascii_folding: false,
|
|
37
|
+
language: 'english',
|
|
38
|
+
max_token_length: 39,
|
|
39
|
+
k1: 1.2,
|
|
40
|
+
b: 0.75,
|
|
41
|
+
};
|
|
42
|
+
const ENGLISH_STOPWORDS = new Set([
|
|
43
|
+
'a', 'an', 'and', 'are', 'as', 'at', 'be', 'but', 'by', 'for', 'if', 'in', 'into', 'is', 'it', 'no', 'not',
|
|
44
|
+
'of', 'on', 'or', 'such', 'that', 'the', 'their', 'then', 'there', 'these', 'they', 'this', 'to', 'was',
|
|
45
|
+
'will', 'with',
|
|
46
|
+
]);
|
|
47
|
+
const WORD = /[\p{L}\p{N}\p{M}_]+/gu;
|
|
48
|
+
const encoder = new TextEncoder();
|
|
49
|
+
function normalizeToken(token, cfg) {
|
|
50
|
+
let t = cfg.case_sensitive ? token : token.toLowerCase();
|
|
51
|
+
if (cfg.ascii_folding)
|
|
52
|
+
t = t.normalize('NFD').replace(/\p{M}+/gu, '');
|
|
53
|
+
return t;
|
|
54
|
+
}
|
|
55
|
+
/**
|
|
56
|
+
* Tokenize one value (a string, or a `[]string`, whose elements are tokenized and concatenated).
|
|
57
|
+
* `keepStopwords` is used for the final prefix token of a `last_as_prefix` query, which must not
|
|
58
|
+
* vanish because the user has so far typed only `an` of `anna`.
|
|
59
|
+
*/
|
|
60
|
+
export function tokenize(value, cfg, opts = {}) {
|
|
61
|
+
const parts = Array.isArray(value) ? value.filter((v) => typeof v === 'string') : typeof value === 'string' ? [value] : [];
|
|
62
|
+
if (cfg.tokenizer === 'word_v4' || cfg.tokenizer === 'word_v3')
|
|
63
|
+
return parts.flatMap((part) => uax29Tokens(part, cfg, opts));
|
|
64
|
+
const out = [];
|
|
65
|
+
for (const part of parts) {
|
|
66
|
+
const raw = cfg.tokenizer === 'pre_tokenized_array' ? [part] : (part.match(WORD) ?? []);
|
|
67
|
+
for (const r of raw) {
|
|
68
|
+
const t = normalizeToken(r, cfg);
|
|
69
|
+
if (t === '' || encoder.encode(t).length > cfg.max_token_length)
|
|
70
|
+
continue;
|
|
71
|
+
if (cfg.remove_stopwords && !opts.keepStopwords && ENGLISH_STOPWORDS.has(t))
|
|
72
|
+
continue;
|
|
73
|
+
out.push(cfg.stemming ? stemEnglish(t) : t);
|
|
74
|
+
}
|
|
75
|
+
}
|
|
76
|
+
return out;
|
|
77
|
+
}
|
|
78
|
+
// ── word_v4 and word_v3 ─────────────────────────────────────────────────────────────────────
|
|
79
|
+
// "The `word_v4` and `word_v3` tokenizers use Unicode v17.0 text segmentation rules (UAX #29) for accurate
|
|
80
|
+
// segmentation across most languages, scripts, and emojis. `word_v4` is the current default for new namespaces; it
|
|
81
|
+
// behaves like `word_v3`, but is roughly 3x faster and fixes a few tokenization edge cases. It's powered by our
|
|
82
|
+
// open-source alyze library" (https://turbopuffer.com/docs/fts, Tokenizers). Modelled on alyze at
|
|
83
|
+
// 1de437c8604b751f26e7061070b9d2b35aebdc73 (github.com/turbopuffer/alyze, src/analyze/mod.rs, src/uax29/word/mod.rs):
|
|
84
|
+
// - word boundaries by UAX #29, and only "word-like" segments become tokens: a segment "is 'word-like' if it contains
|
|
85
|
+
// any char that is ALetter, HebrewLetter, or Numeric, Ideographic or Extended_Pictographic, Other_Number general
|
|
86
|
+
// category, [or] a character whose Script is something meaningful … as opposed to Script=Common/Inherited/Unknown";
|
|
87
|
+
// - then, in alyze's order: the maximum token length (alyze admits a token within the limit in bytes or in
|
|
88
|
+
// characters, `s.len() <= max || s.chars().nth(max).is_none()`), lowercasing unless case_sensitive, stopword removal
|
|
89
|
+
// (the Lucene English list, as word_v2's), and ASCII folding, lowercased again.
|
|
90
|
+
// Where the evidence stops: the twin segments with the host's Intl.Segmenter (ICU's UAX #29, with runs of ideographs
|
|
91
|
+
// and hiragana split back into characters as UAX #29's rules leave them; ICU's dictionary segmentation of Thai, Lao,
|
|
92
|
+
// Khmer and Burmese is kept, which alyze may not do), whose Unicode version is
|
|
93
|
+
// the runtime's, not 17.0; alyze's own DFA may differ from ICU on edge cases, and the "few tokenization edge cases"
|
|
94
|
+
// word_v4 fixes over word_v3 are not documented, so the two are one tokenizer here; ASCII folding strips combining
|
|
95
|
+
// marks after NFD, where alyze's `ascii_fold` may map more characters (ß, æ).
|
|
96
|
+
const segmenter = new Intl.Segmenter('und', { granularity: 'word' });
|
|
97
|
+
const WORD_LIKE = /[\p{L}\p{Nd}\p{No}\p{Extended_Pictographic}\p{Ideographic}]|[^\p{Script=Common}\p{Script=Inherited}\p{Script=Unknown}]/u;
|
|
98
|
+
function uax29Tokens(text, cfg, opts) {
|
|
99
|
+
const out = [];
|
|
100
|
+
// ICU joins runs of ideographs and hiragana by dictionary; UAX #29's rules break between them (neither is ALetter,
|
|
101
|
+
// and no rule joins them), so such a run is split back into its characters
|
|
102
|
+
const segments = [...segmenter.segment(text)].flatMap(({ segment }) => (/^[\p{Ideographic}\p{Script=Hiragana}]+$/u.test(segment) ? [...segment] : [segment]));
|
|
103
|
+
for (const segment of segments) {
|
|
104
|
+
if (!WORD_LIKE.test(segment))
|
|
105
|
+
continue;
|
|
106
|
+
if (!(encoder.encode(segment).length <= cfg.max_token_length || [...segment].length <= cfg.max_token_length))
|
|
107
|
+
continue;
|
|
108
|
+
let t = cfg.case_sensitive ? segment : segment.toLowerCase();
|
|
109
|
+
if (cfg.remove_stopwords && !opts.keepStopwords && ENGLISH_STOPWORDS.has(t))
|
|
110
|
+
continue;
|
|
111
|
+
if (cfg.stemming)
|
|
112
|
+
t = stemEnglish(t);
|
|
113
|
+
out.push(cfg.ascii_folding && /[^\x00-\x7f]/.test(t) ? foldToken(t, cfg) : t);
|
|
114
|
+
}
|
|
115
|
+
return out;
|
|
116
|
+
}
|
|
117
|
+
/** alyze's ASCII folding of a non-ASCII token, lowercased again unless case_sensitive ("ASCII folding can produce
|
|
118
|
+
* uppercase ASCII characters, so we'll lowercase again if case folding is enabled", alyze src/analyze/mod.rs). */
|
|
119
|
+
function foldToken(t, cfg) {
|
|
120
|
+
const folded = t.normalize('NFD').replace(/\p{M}+/gu, '');
|
|
121
|
+
return cfg.case_sensitive ? folded : folded.toLowerCase();
|
|
122
|
+
}
|
|
123
|
+
export function queryTerms(query, cfg, lastAsPrefix) {
|
|
124
|
+
if (!lastAsPrefix)
|
|
125
|
+
return { exact: [...new Set(tokenize(query, cfg))], prefix: null };
|
|
126
|
+
// The prefix is the LAST token of the raw input, kept even when it is a stopword.
|
|
127
|
+
const all = tokenize(query, cfg, { keepStopwords: true });
|
|
128
|
+
if (all.length === 0)
|
|
129
|
+
return { exact: [], prefix: null };
|
|
130
|
+
const prefix = all[all.length - 1];
|
|
131
|
+
const head = cfg.remove_stopwords ? all.slice(0, -1).filter((t) => !ENGLISH_STOPWORDS.has(t)) : all.slice(0, -1);
|
|
132
|
+
return { exact: [...new Set(head)], prefix };
|
|
133
|
+
}
|
|
134
|
+
/** Does a document's token list satisfy the terms? `all` for ContainsAllTokens, else any. */
|
|
135
|
+
export function tokensMatch(docTokens, terms, mode) {
|
|
136
|
+
const set = new Set(docTokens);
|
|
137
|
+
const checks = terms.exact.map((t) => set.has(t));
|
|
138
|
+
if (terms.prefix !== null) {
|
|
139
|
+
const p = terms.prefix;
|
|
140
|
+
checks.push(docTokens.some((t) => t.startsWith(p)));
|
|
141
|
+
}
|
|
142
|
+
// A query whose every token is dropped (all stopwords, all over-length) matches nothing, in
|
|
143
|
+
// either mode — the same documents BM25 would score above zero. The vendor's answer here is
|
|
144
|
+
// unverified; this keeps the twin's count and its ranked search in agreement.
|
|
145
|
+
if (checks.length === 0)
|
|
146
|
+
return false;
|
|
147
|
+
return mode === 'all' ? checks.every(Boolean) : checks.some(Boolean);
|
|
148
|
+
}
|
|
149
|
+
export function buildCorpus(entries, cfg) {
|
|
150
|
+
const docs = new Map();
|
|
151
|
+
const df = new Map();
|
|
152
|
+
let total = 0;
|
|
153
|
+
for (const [key, value] of entries) {
|
|
154
|
+
if (value === null || value === undefined)
|
|
155
|
+
continue;
|
|
156
|
+
const tokens = tokenize(value, cfg);
|
|
157
|
+
docs.set(key, tokens);
|
|
158
|
+
total += tokens.length;
|
|
159
|
+
for (const t of new Set(tokens))
|
|
160
|
+
df.set(t, (df.get(t) ?? 0) + 1);
|
|
161
|
+
}
|
|
162
|
+
return { n: docs.size, avgdl: docs.size === 0 ? 0 : total / docs.size, df, docs };
|
|
163
|
+
}
|
|
164
|
+
/** BM25 of one document against the query terms; 0 means "no term matched". */
|
|
165
|
+
export function bm25Score(key, corpus, terms, cfg) {
|
|
166
|
+
const tokens = corpus.docs.get(key);
|
|
167
|
+
if (tokens === undefined || tokens.length === 0)
|
|
168
|
+
return 0;
|
|
169
|
+
let score = 0;
|
|
170
|
+
const dl = tokens.length;
|
|
171
|
+
for (const term of terms.exact) {
|
|
172
|
+
let tf = 0;
|
|
173
|
+
for (const t of tokens)
|
|
174
|
+
if (t === term)
|
|
175
|
+
tf++;
|
|
176
|
+
if (tf === 0)
|
|
177
|
+
continue;
|
|
178
|
+
const n = corpus.df.get(term) ?? 0;
|
|
179
|
+
const idf = Math.log(1 + (corpus.n - n + 0.5) / (n + 0.5));
|
|
180
|
+
const norm = corpus.avgdl === 0 ? 1 : 1 - cfg.b + cfg.b * (dl / corpus.avgdl);
|
|
181
|
+
score += idf * ((tf * (cfg.k1 + 1)) / (tf + cfg.k1 * norm));
|
|
182
|
+
}
|
|
183
|
+
if (terms.prefix !== null) {
|
|
184
|
+
const p = terms.prefix;
|
|
185
|
+
if (tokens.some((t) => t.startsWith(p)))
|
|
186
|
+
score += 1;
|
|
187
|
+
}
|
|
188
|
+
return score;
|
|
189
|
+
}
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
export type TurbopufferTwinRequest = {
|
|
2
|
+
method: string;
|
|
3
|
+
path: string;
|
|
4
|
+
body?: string;
|
|
5
|
+
headers?: Record<string, string>;
|
|
6
|
+
root?: string;
|
|
7
|
+
occurredAt?: string;
|
|
8
|
+
readOnly?: boolean;
|
|
9
|
+
/** The API key the twin demands. Omit to accept any non-empty key (a missing one still 401s). */
|
|
10
|
+
token?: string;
|
|
11
|
+
};
|
|
12
|
+
export type TurbopufferTwinResponse = {
|
|
13
|
+
status: number;
|
|
14
|
+
body: unknown;
|
|
15
|
+
headers?: Record<string, string>;
|
|
16
|
+
};
|
|
17
|
+
/** `Authorization: Bearer <key>` (client.ts `authHeaders`). An empty key is no key. */
|
|
18
|
+
export declare function extractTurbopufferApiKey(headers: Record<string, string> | undefined): string | null;
|
|
19
|
+
/** Collapse duplicate slashes and drop a leading region segment (see the header). */
|
|
20
|
+
export declare function normalizeTurbopufferPath(rawPath: string): string;
|
|
21
|
+
export declare function handleTurbopufferTwinRequest(req: TurbopufferTwinRequest): Promise<TurbopufferTwinResponse>;
|
|
22
|
+
export declare function turbopufferTwinSnapshot(): {
|
|
23
|
+
resourceTypes: readonly string[];
|
|
24
|
+
implementedEndpoints: readonly string[];
|
|
25
|
+
};
|