@volter/twin-upstashvector 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +202 -0
- package/README.md +233 -0
- package/dist/src/cli.d.ts +2 -0
- package/dist/src/cli.js +45 -0
- package/dist/src/index.d.ts +14 -0
- package/dist/src/index.js +106 -0
- package/dist/src/upstashvector-budget.d.ts +84 -0
- package/dist/src/upstashvector-budget.js +443 -0
- package/dist/src/upstashvector-capabilities.d.ts +4 -0
- package/dist/src/upstashvector-capabilities.js +1108 -0
- package/dist/src/upstashvector-conformance.d.ts +7 -0
- package/dist/src/upstashvector-conformance.js +163 -0
- package/dist/src/upstashvector-connector.d.ts +109 -0
- package/dist/src/upstashvector-connector.js +286 -0
- package/dist/src/upstashvector-filter.d.ts +80 -0
- package/dist/src/upstashvector-filter.js +564 -0
- package/dist/src/upstashvector-server.d.ts +32 -0
- package/dist/src/upstashvector-server.js +56 -0
- package/dist/src/upstashvector-store.d.ts +248 -0
- package/dist/src/upstashvector-store.js +883 -0
- package/dist/src/upstashvector-twin.d.ts +67 -0
- package/dist/src/upstashvector-twin.js +287 -0
- package/package.json +51 -0
- package/src/cli.ts +47 -0
- package/src/index.ts +193 -0
- package/src/upstashvector-budget.ts +489 -0
- package/src/upstashvector-capabilities.ts +1242 -0
- package/src/upstashvector-conformance.ts +175 -0
- package/src/upstashvector-connector.ts +328 -0
- package/src/upstashvector-filter.ts +525 -0
- package/src/upstashvector-server.ts +86 -0
- package/src/upstashvector-store.ts +944 -0
- package/src/upstashvector-twin.ts +347 -0
|
@@ -0,0 +1,248 @@
|
|
|
1
|
+
import { matchesFilter } from './upstashvector-filter.js';
|
|
2
|
+
export declare const SERVICE = "upstashvector";
|
|
3
|
+
/** The default namespace is named `""` (empty string) — upstash.com/docs/vector/features/namespaces. */
|
|
4
|
+
export declare const DEFAULT_NAMESPACE = "";
|
|
5
|
+
/** The three metrics an Upstash Vector index can be created with (`/info`.similarityFunction). */
|
|
6
|
+
export declare const SIMILARITY_FUNCTIONS: readonly ["COSINE", "EUCLIDEAN", "DOT_PRODUCT"];
|
|
7
|
+
export type SimilarityFunction = typeof SIMILARITY_FUNCTIONS[number];
|
|
8
|
+
/** The `/info`.indexType values. This twin models DENSE only; see `UNSUPPORTED_SPARSE`. */
|
|
9
|
+
export declare const DEFAULT_SIMILARITY: SimilarityFunction;
|
|
10
|
+
/**
|
|
11
|
+
* A vendor-shaped API failure: a message and the HTTP status it is served with.
|
|
12
|
+
*
|
|
13
|
+
* The status codes are the ones upstash.com/docs/vector/api/get-started enumerates — 400 for
|
|
14
|
+
* "syntax/command errors", 401 for auth, 405 for an unsupported method — plus the 422 the upsert
|
|
15
|
+
* endpoint's own error example shows for a dimension mismatch.
|
|
16
|
+
*/
|
|
17
|
+
export declare class VectorApiError extends Error {
|
|
18
|
+
readonly status: number;
|
|
19
|
+
constructor(message: string, status: number);
|
|
20
|
+
}
|
|
21
|
+
/** Thrown when a write is attempted against a read-only twin. Mapped to HTTP 405 by the handler. */
|
|
22
|
+
export declare class ReadOnlyError extends Error {
|
|
23
|
+
constructor();
|
|
24
|
+
}
|
|
25
|
+
/** Everything a request needs. `root` scopes the kernel state; `occurredAt` stamps the actions. */
|
|
26
|
+
export type VectorContext = {
|
|
27
|
+
root?: string;
|
|
28
|
+
occurredAt: string;
|
|
29
|
+
readOnly?: boolean;
|
|
30
|
+
/** The dimension this index was created with. Omit to let the FIRST upsert lock one in. */
|
|
31
|
+
dimension?: number;
|
|
32
|
+
/** The metric this index was created with. Defaults to COSINE (the console's own default). */
|
|
33
|
+
similarityFunction?: SimilarityFunction;
|
|
34
|
+
};
|
|
35
|
+
/** One stored vector, in the twin's own storage encoding. */
|
|
36
|
+
export type VectorRow = {
|
|
37
|
+
ns: string;
|
|
38
|
+
vid: string;
|
|
39
|
+
values: number[];
|
|
40
|
+
metadata: Record<string, unknown> | null;
|
|
41
|
+
data: string | null;
|
|
42
|
+
/** A per-subject WRITE ORDINAL — see `nextRev`. Never leaves this module. */
|
|
43
|
+
_rev?: number;
|
|
44
|
+
};
|
|
45
|
+
/** Σ aᵢ·bᵢ. */
|
|
46
|
+
export declare function dotProduct(a: readonly number[], b: readonly number[]): number;
|
|
47
|
+
/** Σ (aᵢ−bᵢ)² — the SQUARED euclidean distance, which is what the vendor's formula takes. */
|
|
48
|
+
export declare function squaredDistance(a: readonly number[], b: readonly number[]): number;
|
|
49
|
+
/**
|
|
50
|
+
* dot(a,b) / (‖a‖·‖b‖).
|
|
51
|
+
*
|
|
52
|
+
* ZERO-VECTOR EDGE (twin-defined, and the vendor does not document it): the true cosine of a
|
|
53
|
+
* zero-length vector is undefined — the denominator is 0. Returning the raw `0/0` would put a
|
|
54
|
+
* `NaN` into the score, which `JSON.stringify` renders as `null`, so a caller would receive a
|
|
55
|
+
* result row whose `score` key was null and no error anywhere. This returns 0 (a normalized score
|
|
56
|
+
* of exactly 0.5, "no information") instead, so the response stays well-typed. Documented in the
|
|
57
|
+
* README's ## Coverage as an unverified edge rather than presented as vendor behaviour.
|
|
58
|
+
*/
|
|
59
|
+
export declare function cosineSimilarity(a: readonly number[], b: readonly number[]): number;
|
|
60
|
+
/**
|
|
61
|
+
* The NORMALIZED score Upstash returns, per metric. Quoted verbatim from
|
|
62
|
+
* upstash.com/docs/vector/features/similarityfunctions:
|
|
63
|
+
* COSINE "(1 + cosine_similarity(v1, v2)) / 2"
|
|
64
|
+
* EUCLIDEAN "1 / (1 + squared_distance(v1, v2))"
|
|
65
|
+
* DOT_PRODUCT "(1 + dot_product(v1, v2)) / 2"
|
|
66
|
+
* All three are monotonically increasing in similarity, so a higher score is always a better match.
|
|
67
|
+
*/
|
|
68
|
+
export declare function similarityScore(fn: SimilarityFunction, a: readonly number[], b: readonly number[]): number;
|
|
69
|
+
/**
|
|
70
|
+
* A vector's kernel subject id.
|
|
71
|
+
*
|
|
72
|
+
* Both components are percent-encoded, which is what makes the join UNAMBIGUOUS: `encodeURIComponent`
|
|
73
|
+
* escapes `:` to `%3A`, so a namespace named `a` holding id `b:c` and a namespace named `a:b`
|
|
74
|
+
* holding id `c` produce different subject ids. A naive `vec:${ns}:${id}` would map both to
|
|
75
|
+
* `vec:a:b:c` — one vector silently overwriting the other, across namespaces, which is precisely
|
|
76
|
+
* the isolation namespaces exist to provide.
|
|
77
|
+
*/
|
|
78
|
+
export declare function vectorSubjectId(ns: string, vid: string): string;
|
|
79
|
+
/** A namespace's kernel subject id. */
|
|
80
|
+
export declare function namespaceSubjectId(ns: string): string;
|
|
81
|
+
/** The single index-configuration subject. */
|
|
82
|
+
export declare const CONFIG_SUBJECT_ID = "config:index";
|
|
83
|
+
/** The kernel subject types this twin writes. Mirrored by the conformance check. */
|
|
84
|
+
export declare const UPSTASHVECTOR_RESOURCE_TYPES: readonly ["vector", "namespace", "config"];
|
|
85
|
+
export type UpstashVectorResourceType = typeof UPSTASHVECTOR_RESOURCE_TYPES[number];
|
|
86
|
+
/**
|
|
87
|
+
* A synchronous in-memory image of the whole index for the duration of ONE request.
|
|
88
|
+
*
|
|
89
|
+
* Seeded from `projectResources` (the kernel IS the source of truth); every mutation records what
|
|
90
|
+
* it touched, and `flush` writes exactly those subjects back. Nothing here outlives the request —
|
|
91
|
+
* no cache, no singleton, no state a later request could inherit.
|
|
92
|
+
*/
|
|
93
|
+
export declare class IndexSpace {
|
|
94
|
+
readonly ctx: VectorContext;
|
|
95
|
+
private readonly rows;
|
|
96
|
+
private readonly namespaces;
|
|
97
|
+
private readonly touched;
|
|
98
|
+
/** subjectId → the last write ordinal seen, INCLUDING for subjects currently deleted. */
|
|
99
|
+
private readonly revs;
|
|
100
|
+
private dimension;
|
|
101
|
+
private similarity;
|
|
102
|
+
private configDirty;
|
|
103
|
+
/** True when stored vectors disagree on length, so no dimension can be honestly inferred. */
|
|
104
|
+
private mixedDimensions;
|
|
105
|
+
constructor(ctx: VectorContext);
|
|
106
|
+
private assertWritable;
|
|
107
|
+
/** The next write ordinal for a subject. Survives deletes — see the constructor's note. */
|
|
108
|
+
private nextRev;
|
|
109
|
+
/** The dimension this index enforces, or `null` while no upsert has locked one in yet. */
|
|
110
|
+
get indexDimension(): number | null;
|
|
111
|
+
get similarityFunction(): SimilarityFunction;
|
|
112
|
+
/**
|
|
113
|
+
* Check a vector's length against the index, LOCKING the dimension in on the first upsert.
|
|
114
|
+
*
|
|
115
|
+
* The vendor's index has a dimension chosen at creation time and cannot change it. A local twin
|
|
116
|
+
* has no creation step, so the first upsert plays that role; from then on the check is exactly
|
|
117
|
+
* as strict as the vendor's, including the error, which is the ONE message in this pack quoted
|
|
118
|
+
* verbatim from the vendor's own docs (the /upsert endpoint's 422 example).
|
|
119
|
+
*/
|
|
120
|
+
private checkDimension;
|
|
121
|
+
/** Every namespace that currently exists, default first then lexicographic (deterministic). */
|
|
122
|
+
listNamespaces(): string[];
|
|
123
|
+
hasNamespace(ns: string): boolean;
|
|
124
|
+
private ensureNamespace;
|
|
125
|
+
/**
|
|
126
|
+
* `POST|DELETE /delete-namespace/{ns}` — remove a namespace AND everything in it.
|
|
127
|
+
*
|
|
128
|
+
* Distinct from `/reset/{ns}`, which empties a namespace but leaves it existing. Keeping both
|
|
129
|
+
* behaviours distinct is why namespaces are tracked as their own kernel subject rather than
|
|
130
|
+
* derived from "does any vector mention this name" — a derived model would make the two
|
|
131
|
+
* endpoints indistinguishable, which would quietly drop a real part of the vendor's surface.
|
|
132
|
+
*
|
|
133
|
+
* TWIN-DEFINED (unverified): deleting the DEFAULT namespace is refused. The default namespace
|
|
134
|
+
* has no name to address and `/reset` already empties it, so a request to remove it is far more
|
|
135
|
+
* likely to be a bug than an intent. Stated in the README's ## Coverage.
|
|
136
|
+
*/
|
|
137
|
+
deleteNamespace(ns: string): void;
|
|
138
|
+
/** Every live vector in one namespace, in a DETERMINISTIC order (id ascending). */
|
|
139
|
+
vectorsIn(ns: string): VectorRow[];
|
|
140
|
+
getVector(ns: string, vid: string): VectorRow | undefined;
|
|
141
|
+
private putVector;
|
|
142
|
+
private removeVector;
|
|
143
|
+
/**
|
|
144
|
+
* `POST /upsert[/{ns}]` — insert or REPLACE whole vectors.
|
|
145
|
+
*
|
|
146
|
+
* Upsert REPLACES: the vendor's `/update` endpoint exists precisely because `/upsert` does not
|
|
147
|
+
* merge, so an upsert without `metadata` clears any metadata the id previously had. Getting this
|
|
148
|
+
* backwards would make `/update`'s whole reason for existing invisible.
|
|
149
|
+
*
|
|
150
|
+
* Returns the vendor's `{"result": "Success"}` payload string.
|
|
151
|
+
*/
|
|
152
|
+
upsert(ns: string, items: unknown[]): 'Success';
|
|
153
|
+
/** Read one `/upsert` element: `{id, vector, metadata?, data?}`. */
|
|
154
|
+
private readUpsertItem;
|
|
155
|
+
/**
|
|
156
|
+
* `POST /update[/{ns}]` — partially modify ONE existing vector.
|
|
157
|
+
*
|
|
158
|
+
* `metadataUpdateMode` (from the vendor's update endpoint): `OVERWRITE` (the default) replaces
|
|
159
|
+
* the metadata object wholesale; `PATCH` merges the supplied keys over what is there.
|
|
160
|
+
* Returns `{updated: 0|1}` — 0 when the id does not exist, which is a normal answer, not an error.
|
|
161
|
+
*/
|
|
162
|
+
update(ns: string, raw: unknown): {
|
|
163
|
+
updated: number;
|
|
164
|
+
};
|
|
165
|
+
/**
|
|
166
|
+
* `POST /query[/{ns}]` — rank the namespace's vectors against a query vector.
|
|
167
|
+
*
|
|
168
|
+
* The ORDER: score DESCENDING, ties broken by id ASCENDING. The tie-break is TWIN-DEFINED (the
|
|
169
|
+
* vendor does not specify one) and exists so a verify that seeds two equidistant vectors gets a
|
|
170
|
+
* reproducible answer instead of whatever the map iteration happened to yield.
|
|
171
|
+
*/
|
|
172
|
+
query(ns: string, raw: unknown): Array<Record<string, unknown>>;
|
|
173
|
+
/**
|
|
174
|
+
* `POST /fetch[/{ns}]` — look vectors up BY ID (or by id prefix), unranked.
|
|
175
|
+
*
|
|
176
|
+
* By ids: "Array elements can be `null` if no such vector exists with the provided id" (the
|
|
177
|
+
* vendor's own fetch doc), and the result is positionally aligned with the request. By prefix:
|
|
178
|
+
* only the matches, so no nulls.
|
|
179
|
+
*/
|
|
180
|
+
fetch(ns: string, raw: unknown): Array<Record<string, unknown> | null>;
|
|
181
|
+
/**
|
|
182
|
+
* `POST /range[/{ns}]` — paginate the namespace in id order.
|
|
183
|
+
*
|
|
184
|
+
* The cursor is a decimal OFFSET as a string (the vendor's own example advances `""` → `"2"`),
|
|
185
|
+
* and `nextCursor` is `""` when the walk is finished. Ordering is id-ascending, which is what
|
|
186
|
+
* makes the offset cursor stable across pages.
|
|
187
|
+
*/
|
|
188
|
+
range(ns: string, raw: unknown): {
|
|
189
|
+
nextCursor: string;
|
|
190
|
+
vectors: Array<Record<string, unknown>>;
|
|
191
|
+
};
|
|
192
|
+
/**
|
|
193
|
+
* `POST|DELETE /delete[/{ns}]` — by ids, by id prefix, or by metadata filter.
|
|
194
|
+
*
|
|
195
|
+
* Returns `{deleted: N}`, the vendor's own shape, where N counts vectors that ACTUALLY existed —
|
|
196
|
+
* deleting an unknown id is not an error and contributes 0.
|
|
197
|
+
*/
|
|
198
|
+
delete(ns: string, raw: unknown): {
|
|
199
|
+
deleted: number;
|
|
200
|
+
};
|
|
201
|
+
/**
|
|
202
|
+
* `POST|DELETE /reset[/{ns}]` and `/reset?all` — empty a namespace (or every namespace).
|
|
203
|
+
*
|
|
204
|
+
* Emptying, NOT removing: the namespaces themselves survive a reset. `?all` is how the SDK's
|
|
205
|
+
* `reset({all:true})` spells it (`ResetCommand` appends the bare `?all` query flag).
|
|
206
|
+
*/
|
|
207
|
+
reset(opts: {
|
|
208
|
+
namespace?: string;
|
|
209
|
+
all?: boolean;
|
|
210
|
+
}): 'Success';
|
|
211
|
+
/**
|
|
212
|
+
* `GET|POST /info` — index-wide counts and configuration.
|
|
213
|
+
*
|
|
214
|
+
* Shape quoted from upstash.com/docs/vector/api/endpoints/info. `pendingVectorCount` is always 0
|
|
215
|
+
* here and that is HONEST rather than a stub: the vendor's pending count reflects vectors still
|
|
216
|
+
* being indexed asynchronously, and this twin's upsert is synchronous, so nothing is ever
|
|
217
|
+
* pending. `indexSize` is a byte estimate; the vendor's exact accounting is not published, so
|
|
218
|
+
* this reports a computed, deterministic size (see `estimateIndexSize`) rather than a constant.
|
|
219
|
+
*/
|
|
220
|
+
info(): Record<string, unknown>;
|
|
221
|
+
/**
|
|
222
|
+
* `indexSize`, computed with the VENDOR'S OWN PUBLISHED FORMULA.
|
|
223
|
+
*
|
|
224
|
+
* upstash.com/docs/vector/help/faq states it verbatim: "Each dimension is estimated to be 4
|
|
225
|
+
* bytes, resulting in vector storage being calculated as vector count * dimension count * 4
|
|
226
|
+
* bytes", and adds that the storage charge combines that with metadata (up to 48 KB/vector) and
|
|
227
|
+
* data (up to 1 MB/vector). So this is `dimensions x 4` per vector plus the encoded size of its
|
|
228
|
+
* metadata and data — not a twin invention. (An earlier draft said the vendor published no
|
|
229
|
+
* accounting and also counted the id's bytes, which the formula does not include; both were
|
|
230
|
+
* corrected once the FAQ was found during the §9 pass.)
|
|
231
|
+
*
|
|
232
|
+
* It stays an ESTIMATE in the vendor's own sense of the word, and it is a real function of real
|
|
233
|
+
* state — it grows when you add vectors and shrinks when you delete them.
|
|
234
|
+
*/
|
|
235
|
+
private estimateIndexSize;
|
|
236
|
+
/** Did this run write anything? */
|
|
237
|
+
get dirty(): boolean;
|
|
238
|
+
/**
|
|
239
|
+
* Write every touched subject back to the kernel action log.
|
|
240
|
+
*
|
|
241
|
+
* The kernel MERGES fields, so a delete writes EVERY field back to its "nothing here" value —
|
|
242
|
+
* leaving `values`/`metadata` behind would let a later re-upsert of the same id inherit a dead
|
|
243
|
+
* coordinate array if it ever wrote a partial field set.
|
|
244
|
+
*/
|
|
245
|
+
flush(): Promise<void>;
|
|
246
|
+
}
|
|
247
|
+
/** Re-exported so the handler can map a filter parse failure without importing the filter module. */
|
|
248
|
+
export { matchesFilter };
|