agentfootprint 9.1.0 → 9.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +15 -0
- package/dist/adapters/memory/pgVector.js +731 -0
- package/dist/adapters/memory/pgVector.js.map +1 -0
- package/dist/adapters/memory/s3Vectors.js +628 -0
- package/dist/adapters/memory/s3Vectors.js.map +1 -0
- package/dist/adapters/memory/sqliteVector.js +16 -79
- package/dist/adapters/memory/sqliteVector.js.map +1 -1
- package/dist/core/Agent.js +344 -18
- package/dist/core/Agent.js.map +1 -1
- package/dist/core/LLMCall.js +17 -0
- package/dist/core/LLMCall.js.map +1 -1
- package/dist/core/RunnerBase.js +22 -6
- package/dist/core/RunnerBase.js.map +1 -1
- package/dist/core/agent/AgentBuilder.js +20 -0
- package/dist/core/agent/AgentBuilder.js.map +1 -1
- package/dist/core/conversation.js +139 -0
- package/dist/core/conversation.js.map +1 -0
- package/dist/core/runCheckpoint.js +60 -2
- package/dist/core/runCheckpoint.js.map +1 -1
- package/dist/embedders/index.js +240 -76
- package/dist/embedders/index.js.map +1 -1
- package/dist/esm/adapters/memory/pgVector.d.ts +243 -0
- package/dist/esm/adapters/memory/pgVector.js +726 -0
- package/dist/esm/adapters/memory/pgVector.js.map +1 -0
- package/dist/esm/adapters/memory/s3Vectors.d.ts +208 -0
- package/dist/esm/adapters/memory/s3Vectors.js +624 -0
- package/dist/esm/adapters/memory/s3Vectors.js.map +1 -0
- package/dist/esm/adapters/memory/sqliteVector.d.ts +5 -27
- package/dist/esm/adapters/memory/sqliteVector.js +10 -73
- package/dist/esm/adapters/memory/sqliteVector.js.map +1 -1
- package/dist/esm/core/Agent.d.ts +218 -3
- package/dist/esm/core/Agent.js +345 -19
- package/dist/esm/core/Agent.js.map +1 -1
- package/dist/esm/core/LLMCall.d.ts +9 -0
- package/dist/esm/core/LLMCall.js +17 -0
- package/dist/esm/core/LLMCall.js.map +1 -1
- package/dist/esm/core/RunnerBase.d.ts +22 -6
- package/dist/esm/core/RunnerBase.js +22 -6
- package/dist/esm/core/RunnerBase.js.map +1 -1
- package/dist/esm/core/agent/AgentBuilder.d.ts +5 -0
- package/dist/esm/core/agent/AgentBuilder.js +20 -0
- package/dist/esm/core/agent/AgentBuilder.js.map +1 -1
- package/dist/esm/core/agent/types.d.ts +45 -3
- package/dist/esm/core/conversation.d.ts +96 -0
- package/dist/esm/core/conversation.js +133 -0
- package/dist/esm/core/conversation.js.map +1 -0
- package/dist/esm/core/runCheckpoint.d.ts +85 -1
- package/dist/esm/core/runCheckpoint.js +57 -1
- package/dist/esm/core/runCheckpoint.js.map +1 -1
- package/dist/esm/embedders/index.d.ts +99 -19
- package/dist/esm/embedders/index.js +240 -76
- package/dist/esm/embedders/index.js.map +1 -1
- package/dist/esm/hosting/standingAgent.d.ts +6 -2
- package/dist/esm/hosting/standingAgent.js +36 -27
- package/dist/esm/hosting/standingAgent.js.map +1 -1
- package/dist/esm/index.d.ts +2 -1
- package/dist/esm/index.js +5 -1
- package/dist/esm/index.js.map +1 -1
- package/dist/esm/lib/embedderMismatch.d.ts +67 -0
- package/dist/esm/lib/embedderMismatch.js +96 -0
- package/dist/esm/lib/embedderMismatch.js.map +1 -0
- package/dist/esm/lib/rag/defineRAG.d.ts +28 -6
- package/dist/esm/lib/rag/defineRAG.js +39 -5
- package/dist/esm/lib/rag/defineRAG.js.map +1 -1
- package/dist/esm/memory/define.js +31 -1
- package/dist/esm/memory/define.js.map +1 -1
- package/dist/esm/memory/define.types.d.ts +13 -2
- package/dist/esm/memory/define.types.js.map +1 -1
- package/dist/esm/memory/embedding/loadRelevant.d.ts +10 -2
- package/dist/esm/memory/embedding/loadRelevant.js +36 -18
- package/dist/esm/memory/embedding/loadRelevant.js.map +1 -1
- package/dist/esm/memory/pipeline/semantic.d.ts +13 -2
- package/dist/esm/memory/pipeline/semantic.js +11 -5
- package/dist/esm/memory/pipeline/semantic.js.map +1 -1
- package/dist/esm/memory/store/capability.d.ts +25 -6
- package/dist/esm/memory/store/capability.js +68 -3
- package/dist/esm/memory/store/capability.js.map +1 -1
- package/dist/esm/memory/store/index.d.ts +1 -0
- package/dist/esm/memory/store/index.js +4 -0
- package/dist/esm/memory/store/index.js.map +1 -1
- package/dist/esm/memory/store/types.d.ts +40 -0
- package/dist/esm/memory-providers.d.ts +7 -4
- package/dist/esm/memory-providers.js +18 -4
- package/dist/esm/memory-providers.js.map +1 -1
- package/dist/hosting/standingAgent.js +36 -27
- package/dist/hosting/standingAgent.js.map +1 -1
- package/dist/index.js +9 -1
- package/dist/index.js.map +1 -1
- package/dist/lib/embedderMismatch.js +103 -0
- package/dist/lib/embedderMismatch.js.map +1 -0
- package/dist/lib/rag/defineRAG.js +39 -5
- package/dist/lib/rag/defineRAG.js.map +1 -1
- package/dist/memory/define.js +31 -1
- package/dist/memory/define.js.map +1 -1
- package/dist/memory/define.types.js.map +1 -1
- package/dist/memory/embedding/loadRelevant.js +36 -18
- package/dist/memory/embedding/loadRelevant.js.map +1 -1
- package/dist/memory/pipeline/semantic.js +11 -5
- package/dist/memory/pipeline/semantic.js.map +1 -1
- package/dist/memory/store/capability.js +70 -4
- package/dist/memory/store/capability.js.map +1 -1
- package/dist/memory/store/index.js +7 -1
- package/dist/memory/store/index.js.map +1 -1
- package/dist/memory-providers.js +22 -5
- package/dist/memory-providers.js.map +1 -1
- package/dist/types/adapters/memory/pgVector.d.ts +244 -0
- package/dist/types/adapters/memory/pgVector.d.ts.map +1 -0
- package/dist/types/adapters/memory/s3Vectors.d.ts +209 -0
- package/dist/types/adapters/memory/s3Vectors.d.ts.map +1 -0
- package/dist/types/adapters/memory/sqliteVector.d.ts +5 -27
- package/dist/types/adapters/memory/sqliteVector.d.ts.map +1 -1
- package/dist/types/core/Agent.d.ts +218 -3
- package/dist/types/core/Agent.d.ts.map +1 -1
- package/dist/types/core/LLMCall.d.ts +9 -0
- package/dist/types/core/LLMCall.d.ts.map +1 -1
- package/dist/types/core/RunnerBase.d.ts +22 -6
- package/dist/types/core/RunnerBase.d.ts.map +1 -1
- package/dist/types/core/agent/AgentBuilder.d.ts +5 -0
- package/dist/types/core/agent/AgentBuilder.d.ts.map +1 -1
- package/dist/types/core/agent/types.d.ts +45 -3
- package/dist/types/core/agent/types.d.ts.map +1 -1
- package/dist/types/core/conversation.d.ts +97 -0
- package/dist/types/core/conversation.d.ts.map +1 -0
- package/dist/types/core/runCheckpoint.d.ts +85 -1
- package/dist/types/core/runCheckpoint.d.ts.map +1 -1
- package/dist/types/embedders/index.d.ts +99 -19
- package/dist/types/embedders/index.d.ts.map +1 -1
- package/dist/types/hosting/standingAgent.d.ts +6 -2
- package/dist/types/hosting/standingAgent.d.ts.map +1 -1
- package/dist/types/index.d.ts +2 -1
- package/dist/types/index.d.ts.map +1 -1
- package/dist/types/lib/embedderMismatch.d.ts +68 -0
- package/dist/types/lib/embedderMismatch.d.ts.map +1 -0
- package/dist/types/lib/rag/defineRAG.d.ts +28 -6
- package/dist/types/lib/rag/defineRAG.d.ts.map +1 -1
- package/dist/types/memory/define.d.ts.map +1 -1
- package/dist/types/memory/define.types.d.ts +13 -2
- package/dist/types/memory/define.types.d.ts.map +1 -1
- package/dist/types/memory/embedding/loadRelevant.d.ts +10 -2
- package/dist/types/memory/embedding/loadRelevant.d.ts.map +1 -1
- package/dist/types/memory/pipeline/semantic.d.ts +13 -2
- package/dist/types/memory/pipeline/semantic.d.ts.map +1 -1
- package/dist/types/memory/store/capability.d.ts +25 -6
- package/dist/types/memory/store/capability.d.ts.map +1 -1
- package/dist/types/memory/store/index.d.ts +1 -0
- package/dist/types/memory/store/index.d.ts.map +1 -1
- package/dist/types/memory/store/types.d.ts +40 -0
- package/dist/types/memory/store/types.d.ts.map +1 -1
- package/dist/types/memory-providers.d.ts +7 -4
- package/dist/types/memory-providers.d.ts.map +1 -1
- package/package.json +9 -1
|
@@ -8,7 +8,8 @@
|
|
|
8
8
|
* you use and agentfootprint stays dependency-free.
|
|
9
9
|
*
|
|
10
10
|
* openaiEmbedder() — hosted; needs OPENAI_API_KEY; no extra install (fetch).
|
|
11
|
-
* bedrockEmbedder() — hosted on AWS; Titan Text Embeddings
|
|
11
|
+
* bedrockEmbedder() — hosted on AWS; Titan Text Embeddings and Cohere Embed
|
|
12
|
+
* v3 (the model id picks the body shape); credentials
|
|
12
13
|
* come from the AWS chain, so no key option at all.
|
|
13
14
|
* peer dep: @aws-sdk/client-bedrock-runtime.
|
|
14
15
|
* localEmbedder() — on-device sentence-transformer; no key; offline after a
|
|
@@ -111,25 +112,85 @@ export interface BedrockRuntimeSdkModule {
|
|
|
111
112
|
}) => BedrockRuntimeLikeClient;
|
|
112
113
|
readonly InvokeModelCommand?: new (input: unknown) => unknown;
|
|
113
114
|
}
|
|
115
|
+
/**
|
|
116
|
+
* The request/response SHAPE a Bedrock embedding model speaks (9.3.0).
|
|
117
|
+
*
|
|
118
|
+
* `InvokeModel` is one operation with a vendor-specific body on both sides:
|
|
119
|
+
* Titan takes `{ inputText }` and answers `{ embedding }`, Cohere takes
|
|
120
|
+
* `{ texts, input_type }` and answers `{ embeddings }`. One model id therefore
|
|
121
|
+
* does not describe one call, and until 9.3.0 this factory sent Titan's body to
|
|
122
|
+
* everything — so a Cohere model id was accepted at construction (with
|
|
123
|
+
* `dimensions`) and failed at the first embed, against the real service, with a
|
|
124
|
+
* validation error from AWS rather than a sentence from here.
|
|
125
|
+
*/
|
|
126
|
+
export type BedrockEmbeddingFamily = 'titan' | 'cohere';
|
|
127
|
+
/**
|
|
128
|
+
* Cohere's `input_type`, which is a real parameter and not a hint: the v3
|
|
129
|
+
* models embed a QUERY and a DOCUMENT into deliberately different places, and
|
|
130
|
+
* the two are meant to be compared with each other. Sending one value for both
|
|
131
|
+
* halves is a measurable loss of retrieval quality, not a style choice — and
|
|
132
|
+
* Cohere requires the field, so there is no "unset" to fall back to.
|
|
133
|
+
*/
|
|
134
|
+
export type CohereInputType = 'search_document' | 'search_query';
|
|
114
135
|
export interface BedrockEmbedderOptions {
|
|
115
136
|
/**
|
|
116
137
|
* Bedrock model id. Default `'amazon.titan-embed-text-v2:0'`.
|
|
117
138
|
*
|
|
118
|
-
*
|
|
119
|
-
*
|
|
120
|
-
*
|
|
121
|
-
*
|
|
139
|
+
* Four are known by name — Titan V2, Titan V1, and Cohere Embed English /
|
|
140
|
+
* Multilingual v3 (see {@link BEDROCK_EMBEDDING_MODELS}) — and each brings
|
|
141
|
+
* its own body shape, vector length and input window. An id that WRAPS one of
|
|
142
|
+
* those (a cross-region inference profile `us.amazon.titan-embed-text-v2:0`,
|
|
143
|
+
* or an ARN ending in the model id) is resolved to the model it names.
|
|
144
|
+
*
|
|
145
|
+
* Anything else is a model this library has never met: pass `dimensions`
|
|
146
|
+
* with it (its vector length is not something this can know) and `family` if
|
|
147
|
+
* it is not Titan-shaped.
|
|
122
148
|
*/
|
|
123
149
|
readonly model?: string;
|
|
124
150
|
/**
|
|
125
|
-
* Vector length to request.
|
|
151
|
+
* Vector length to request.
|
|
152
|
+
*
|
|
153
|
+
* Titan V2 is the one CONFIGURABLE model — 1024 (default), 512 or 256 — and
|
|
126
154
|
* the value is SENT to the model AND reported as `.dimensions`, so the two
|
|
127
|
-
* can never disagree.
|
|
155
|
+
* can never disagree. Every other known model has ONE size, and asking for a
|
|
156
|
+
* different one is refused rather than reported: `.dimensions` is what a
|
|
157
|
+
* vector store fingerprints on, and a wrong one corrupts it silently.
|
|
128
158
|
*
|
|
129
|
-
* Required for a model outside {@link
|
|
130
|
-
* `.dimensions` silently corrupts a vector store.
|
|
159
|
+
* Required for a model outside {@link BEDROCK_EMBEDDING_MODELS}.
|
|
131
160
|
*/
|
|
132
161
|
readonly dimensions?: number;
|
|
162
|
+
/**
|
|
163
|
+
* The body shape to speak, when the model id does not say (9.3.0).
|
|
164
|
+
*
|
|
165
|
+
* Inferred for every known model and for anything that wraps one, so this is
|
|
166
|
+
* only for a model id this library has never met — a provisioned-throughput
|
|
167
|
+
* ARN, a custom deployment. Unknown and unstated, the body is **Titan's**,
|
|
168
|
+
* which is the shape every release before 9.3.0 sent to everything.
|
|
169
|
+
*
|
|
170
|
+
* Stating a family that contradicts a known model id is refused by name.
|
|
171
|
+
*/
|
|
172
|
+
readonly family?: BedrockEmbeddingFamily;
|
|
173
|
+
/**
|
|
174
|
+
* Pin Cohere's `input_type` instead of deriving it from the call (9.3.0).
|
|
175
|
+
*
|
|
176
|
+
* Unset — the default — `embed()` sends `'search_query'` and `embedBatch()`
|
|
177
|
+
* sends `'search_document'`, because that is what this library's own two
|
|
178
|
+
* call sites are: retrieval embeds ONE question (`loadRelevant`), indexing
|
|
179
|
+
* embeds MANY passages (`indexDocuments`, `embedMessages`). Pin it when your
|
|
180
|
+
* own code uses the two calls differently — embedding a single document, say,
|
|
181
|
+
* or scoring a batch of queries.
|
|
182
|
+
*
|
|
183
|
+
* Ignored by Titan, which has no such parameter.
|
|
184
|
+
*/
|
|
185
|
+
readonly inputType?: CohereInputType;
|
|
186
|
+
/**
|
|
187
|
+
* The longest input this model reads whole, in CHARACTERS
|
|
188
|
+
* ({@link Embedder.maxInputChars}). Declared for every known model from its
|
|
189
|
+
* documented token window; this option is how a model this library does not
|
|
190
|
+
* know states its own, rather than declaring none and leaving the indexer's
|
|
191
|
+
* conservative default in place. An explicit value always wins.
|
|
192
|
+
*/
|
|
193
|
+
readonly maxInputChars?: number;
|
|
133
194
|
/** AWS region. Passed to the SDK client when this factory builds one. */
|
|
134
195
|
readonly region?: string;
|
|
135
196
|
/** A pre-built Bedrock runtime client, so one SDK config serves the whole app. */
|
|
@@ -167,19 +228,31 @@ export interface BedrockEmbedderOptions {
|
|
|
167
228
|
* q8 and an fp32 build of one model are near-identical spaces, and "near" is
|
|
168
229
|
* exactly the difference that surfaces as a mysteriously worse ranking.)
|
|
169
230
|
*
|
|
231
|
+
* ── One operation, two body shapes (9.3.0) ───────────────────────────────
|
|
232
|
+
* `InvokeModel` is a single API over vendor-specific JSON. Titan takes
|
|
233
|
+
* `{ inputText }` and answers `{ embedding }`; Cohere takes
|
|
234
|
+
* `{ texts, input_type }` and answers `{ embeddings }`, embeds up to
|
|
235
|
+
* {@link COHERE_MAX_TEXTS_PER_CALL} of them per call, and distinguishes a
|
|
236
|
+
* QUERY from a DOCUMENT. So the model id selects a FAMILY
|
|
237
|
+
* ({@link BedrockEmbeddingFamily}), and the family owns the request, the
|
|
238
|
+
* response and the batching. Before this, one shape was sent to everything —
|
|
239
|
+
* a Cohere id constructed fine and failed at the first embed.
|
|
240
|
+
*
|
|
170
241
|
* ── The input ceiling (9.1.0) ────────────────────────────────────────────
|
|
171
|
-
* `.maxInputChars`
|
|
172
|
-
*
|
|
173
|
-
*
|
|
174
|
-
* the indexer's own default, which was measured on an on-device model
|
|
175
|
-
*
|
|
176
|
-
*
|
|
177
|
-
*
|
|
178
|
-
*
|
|
242
|
+
* `.maxInputChars` is per MODEL, from its documented token window converted at
|
|
243
|
+
* the stated {@link CHARS_PER_TOKEN} assumption of 4 characters per token:
|
|
244
|
+
* **32,000** for both Titan text-embedding models (8,192 tokens — sixteen
|
|
245
|
+
* times the indexer's own default, which was measured on an on-device model),
|
|
246
|
+
* and **2,000** for Cohere Embed v3 (512 tokens). Those two numbers are why the
|
|
247
|
+
* ceiling cannot be a per-vendor constant: the same 2,500-character chunk is
|
|
248
|
+
* read whole by Titan and truncated by Cohere. Dense text (code, tables, CJK)
|
|
249
|
+
* tokenises tighter than the assumption; pass an explicit `maxChunkChars` for
|
|
250
|
+
* such a corpus, and it wins over this number.
|
|
179
251
|
*
|
|
180
252
|
* @throws if `model` is unknown and `dimensions` was not supplied; if
|
|
181
|
-
* `dimensions` is
|
|
182
|
-
*
|
|
253
|
+
* `dimensions` is a size the model does not produce; if `family`
|
|
254
|
+
* contradicts a known model id; or if the SDK is missing and no
|
|
255
|
+
* `client` / `_client` / `_sdk` was passed.
|
|
183
256
|
*
|
|
184
257
|
* @example
|
|
185
258
|
* ```ts
|
|
@@ -190,6 +263,13 @@ export interface BedrockEmbedderOptions {
|
|
|
190
263
|
* const embedder = bedrockEmbedder({ region: 'us-east-1', dimensions: 512 });
|
|
191
264
|
* await indexFolder('./docs', { to: sqliteVectorStore({ file: './corpus.db' }), embedder });
|
|
192
265
|
* ```
|
|
266
|
+
*
|
|
267
|
+
* @example A Cohere embedding model on the same runtime
|
|
268
|
+
* ```ts
|
|
269
|
+
* // Body shape, response field, batch size and 512-token window all follow
|
|
270
|
+
* // from the model id — nothing else changes at the call site.
|
|
271
|
+
* const embedder = bedrockEmbedder({ model: 'cohere.embed-english-v3' });
|
|
272
|
+
* ```
|
|
193
273
|
*/
|
|
194
274
|
export declare function bedrockEmbedder(options?: BedrockEmbedderOptions): Embedder;
|
|
195
275
|
/**
|
|
@@ -127,39 +127,99 @@ export function openaiEmbedder(options = {}) {
|
|
|
127
127
|
};
|
|
128
128
|
}
|
|
129
129
|
/**
|
|
130
|
-
*
|
|
131
|
-
*
|
|
130
|
+
* The Bedrock embedding models this library knows by name — three facts each,
|
|
131
|
+
* in one table, because they arrive together and drift apart when they are
|
|
132
|
+
* kept apart (9.3.0; two Titan-only tables until then).
|
|
132
133
|
*
|
|
133
|
-
* Titan Text Embeddings **V2** is configurable — 1024
|
|
134
|
-
* Titan Embeddings G1 – Text (**V1**) has one size, 1536,
|
|
135
|
-
* parameter at all.
|
|
136
|
-
*
|
|
137
|
-
*
|
|
138
|
-
* `openaiEmbedder` applies, for the same reason: a store that trusts
|
|
134
|
+
* **Size.** Titan Text Embeddings **V2** is the configurable one — 1024
|
|
135
|
+
* (default), 512, 256. Titan Embeddings G1 – Text (**V1**) has one size, 1536,
|
|
136
|
+
* and no `dimensions` parameter at all. Cohere Embed v3 (English and
|
|
137
|
+
* Multilingual) returns 1024 and likewise takes no size. A model outside this
|
|
138
|
+
* table has a length this library cannot know and must state its own — the same
|
|
139
|
+
* rule `openaiEmbedder` applies, for the same reason: a store that trusts
|
|
139
140
|
* `.dimensions` and gets a different length back corrupts silently.
|
|
141
|
+
*
|
|
142
|
+
* **Window.** Both Titan text-embedding models accept **8,192 tokens** —
|
|
143
|
+
* sixteen times the on-device model the shipped default ceiling was measured
|
|
144
|
+
* on. Cohere Embed v3 accepts **512**, and that number is the whole argument
|
|
145
|
+
* for a per-MODEL ceiling rather than a per-VENDOR one: 512 tokens is ~2,000
|
|
146
|
+
* characters, so the same corpus that is embedded whole by Titan is silently
|
|
147
|
+
* truncated by Cohere at a quarter of the chunk. Declared, the indexers cut to
|
|
148
|
+
* fit; guessed from a sibling model, they would not.
|
|
149
|
+
*
|
|
150
|
+
* **Family.** The body shape (see {@link BedrockEmbeddingFamily}) — the fact
|
|
151
|
+
* whose absence made a Cohere model id constructible and unusable before 9.3.0.
|
|
140
152
|
*/
|
|
141
|
-
const
|
|
142
|
-
'amazon.titan-embed-text-v2:0':
|
|
143
|
-
|
|
153
|
+
const BEDROCK_EMBEDDING_MODELS = {
|
|
154
|
+
'amazon.titan-embed-text-v2:0': {
|
|
155
|
+
family: 'titan',
|
|
156
|
+
dimensions: 1024,
|
|
157
|
+
sizes: [1024, 512, 256],
|
|
158
|
+
maxInputTokens: 8192,
|
|
159
|
+
},
|
|
160
|
+
'amazon.titan-embed-text-v1': { family: 'titan', dimensions: 1536, maxInputTokens: 8192 },
|
|
161
|
+
'cohere.embed-english-v3': { family: 'cohere', dimensions: 1024, maxInputTokens: 512 },
|
|
162
|
+
'cohere.embed-multilingual-v3': { family: 'cohere', dimensions: 1024, maxInputTokens: 512 },
|
|
144
163
|
};
|
|
145
164
|
/**
|
|
146
|
-
*
|
|
165
|
+
* Titan V2's configurable sizes, kept as a constant because an id this table
|
|
166
|
+
* has never seen can still be recognisably Titan V2 (a variant released after
|
|
167
|
+
* this version) — and for such an id the size must still be SENT, or the model
|
|
168
|
+
* returns 1024 while `.dimensions` reports 512.
|
|
169
|
+
*/
|
|
170
|
+
const TITAN_V2_SIZES = [1024, 512, 256];
|
|
171
|
+
/**
|
|
172
|
+
* How many texts Cohere embeds in ONE `InvokeModel` call.
|
|
147
173
|
*
|
|
148
|
-
*
|
|
149
|
-
*
|
|
150
|
-
*
|
|
151
|
-
*
|
|
152
|
-
*
|
|
174
|
+
* **96**, the documented maximum, and the reason `embedBatch` is a real batch
|
|
175
|
+
* here and N calls for Titan: a 500-chunk corpus is 6 round-trips instead of
|
|
176
|
+
* 500. Chunked at exactly the documented number rather than under it, because
|
|
177
|
+
* unlike a byte limit this one is a count the caller can see — and a batch that
|
|
178
|
+
* silently used half the allowance would be a cost nobody asked for.
|
|
179
|
+
*/
|
|
180
|
+
const COHERE_MAX_TEXTS_PER_CALL = 96;
|
|
181
|
+
/**
|
|
182
|
+
* The body shape sent to a model id this library has never met.
|
|
153
183
|
*
|
|
154
|
-
*
|
|
155
|
-
*
|
|
156
|
-
*
|
|
184
|
+
* Titan's — because it is the shape EVERY release before 9.3.0 sent to every
|
|
185
|
+
* model, so an unknown id keeps doing exactly what it did. `family` is how a
|
|
186
|
+
* caller says otherwise, and the read-side refusal names that option.
|
|
157
187
|
*/
|
|
158
|
-
const
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
188
|
+
const DEFAULT_BEDROCK_FAMILY = 'titan';
|
|
189
|
+
/**
|
|
190
|
+
* The facts for a model id, including ids that WRAP a known one.
|
|
191
|
+
*
|
|
192
|
+
* A cross-region inference profile (`us.amazon.titan-embed-text-v2:0`) and an
|
|
193
|
+
* inference-profile ARN both END in the model id they route to, so containment
|
|
194
|
+
* resolves them exactly rather than by guess — and an id that names its model is
|
|
195
|
+
* not "unknown" just because it has a prefix. Before 9.3.0 those were refused
|
|
196
|
+
* as unknown models, which is why widening this is safe: it accepts what used
|
|
197
|
+
* to throw.
|
|
198
|
+
*/
|
|
199
|
+
function bedrockFactsFor(model) {
|
|
200
|
+
const exact = BEDROCK_EMBEDDING_MODELS[model];
|
|
201
|
+
if (exact)
|
|
202
|
+
return exact;
|
|
203
|
+
for (const [id, facts] of Object.entries(BEDROCK_EMBEDDING_MODELS)) {
|
|
204
|
+
if (model.includes(id))
|
|
205
|
+
return facts;
|
|
206
|
+
}
|
|
207
|
+
return undefined;
|
|
208
|
+
}
|
|
209
|
+
/**
|
|
210
|
+
* The family a model id NAMES, or `undefined` when it names none.
|
|
211
|
+
*
|
|
212
|
+
* Substring, not prefix: the id may be wrapped by a region prefix or an ARN.
|
|
213
|
+
* An id that says neither gets Titan's body — the shape every release before
|
|
214
|
+
* 9.3.0 sent to everything — and `family` is how you say otherwise.
|
|
215
|
+
*/
|
|
216
|
+
function inferBedrockFamily(model) {
|
|
217
|
+
if (model.includes('cohere.embed'))
|
|
218
|
+
return 'cohere';
|
|
219
|
+
if (model.includes('titan-embed'))
|
|
220
|
+
return 'titan';
|
|
221
|
+
return undefined;
|
|
222
|
+
}
|
|
163
223
|
/**
|
|
164
224
|
* Amazon Bedrock's hosted embeddings, through `InvokeModel`.
|
|
165
225
|
*
|
|
@@ -188,19 +248,31 @@ const TITAN_V2_SUPPORTED = [1024, 512, 256];
|
|
|
188
248
|
* q8 and an fp32 build of one model are near-identical spaces, and "near" is
|
|
189
249
|
* exactly the difference that surfaces as a mysteriously worse ranking.)
|
|
190
250
|
*
|
|
251
|
+
* ── One operation, two body shapes (9.3.0) ───────────────────────────────
|
|
252
|
+
* `InvokeModel` is a single API over vendor-specific JSON. Titan takes
|
|
253
|
+
* `{ inputText }` and answers `{ embedding }`; Cohere takes
|
|
254
|
+
* `{ texts, input_type }` and answers `{ embeddings }`, embeds up to
|
|
255
|
+
* {@link COHERE_MAX_TEXTS_PER_CALL} of them per call, and distinguishes a
|
|
256
|
+
* QUERY from a DOCUMENT. So the model id selects a FAMILY
|
|
257
|
+
* ({@link BedrockEmbeddingFamily}), and the family owns the request, the
|
|
258
|
+
* response and the batching. Before this, one shape was sent to everything —
|
|
259
|
+
* a Cohere id constructed fine and failed at the first embed.
|
|
260
|
+
*
|
|
191
261
|
* ── The input ceiling (9.1.0) ────────────────────────────────────────────
|
|
192
|
-
* `.maxInputChars`
|
|
193
|
-
*
|
|
194
|
-
*
|
|
195
|
-
* the indexer's own default, which was measured on an on-device model
|
|
196
|
-
*
|
|
197
|
-
*
|
|
198
|
-
*
|
|
199
|
-
*
|
|
262
|
+
* `.maxInputChars` is per MODEL, from its documented token window converted at
|
|
263
|
+
* the stated {@link CHARS_PER_TOKEN} assumption of 4 characters per token:
|
|
264
|
+
* **32,000** for both Titan text-embedding models (8,192 tokens — sixteen
|
|
265
|
+
* times the indexer's own default, which was measured on an on-device model),
|
|
266
|
+
* and **2,000** for Cohere Embed v3 (512 tokens). Those two numbers are why the
|
|
267
|
+
* ceiling cannot be a per-vendor constant: the same 2,500-character chunk is
|
|
268
|
+
* read whole by Titan and truncated by Cohere. Dense text (code, tables, CJK)
|
|
269
|
+
* tokenises tighter than the assumption; pass an explicit `maxChunkChars` for
|
|
270
|
+
* such a corpus, and it wins over this number.
|
|
200
271
|
*
|
|
201
272
|
* @throws if `model` is unknown and `dimensions` was not supplied; if
|
|
202
|
-
* `dimensions` is
|
|
203
|
-
*
|
|
273
|
+
* `dimensions` is a size the model does not produce; if `family`
|
|
274
|
+
* contradicts a known model id; or if the SDK is missing and no
|
|
275
|
+
* `client` / `_client` / `_sdk` was passed.
|
|
204
276
|
*
|
|
205
277
|
* @example
|
|
206
278
|
* ```ts
|
|
@@ -211,24 +283,61 @@ const TITAN_V2_SUPPORTED = [1024, 512, 256];
|
|
|
211
283
|
* const embedder = bedrockEmbedder({ region: 'us-east-1', dimensions: 512 });
|
|
212
284
|
* await indexFolder('./docs', { to: sqliteVectorStore({ file: './corpus.db' }), embedder });
|
|
213
285
|
* ```
|
|
286
|
+
*
|
|
287
|
+
* @example A Cohere embedding model on the same runtime
|
|
288
|
+
* ```ts
|
|
289
|
+
* // Body shape, response field, batch size and 512-token window all follow
|
|
290
|
+
* // from the model id — nothing else changes at the call site.
|
|
291
|
+
* const embedder = bedrockEmbedder({ model: 'cohere.embed-english-v3' });
|
|
292
|
+
* ```
|
|
214
293
|
*/
|
|
215
294
|
export function bedrockEmbedder(options = {}) {
|
|
216
295
|
const model = options.model ?? 'amazon.titan-embed-text-v2:0';
|
|
217
|
-
const
|
|
296
|
+
const facts = bedrockFactsFor(model);
|
|
297
|
+
const named = inferBedrockFamily(model);
|
|
298
|
+
if (options.family !== undefined && facts !== undefined && options.family !== facts.family) {
|
|
299
|
+
throw new Error(`bedrockEmbedder: model '${model}' is a ${facts.family} model, and \`family: ` +
|
|
300
|
+
`'${options.family}'\` says otherwise. The two cannot both be right, and guessing ` +
|
|
301
|
+
`which line is the mistake would decide the request body — the one thing a wrong ` +
|
|
302
|
+
`answer here breaks at the first embed. Drop \`family\` (the model id already says ` +
|
|
303
|
+
`it), or name the model you meant.`);
|
|
304
|
+
}
|
|
305
|
+
// Explicit, then what the id names, then Titan — which is the body every
|
|
306
|
+
// release before 9.3.0 sent to every model, so an id this library has never
|
|
307
|
+
// met keeps behaving exactly as it did.
|
|
308
|
+
const family = options.family ?? facts?.family ?? named ?? DEFAULT_BEDROCK_FAMILY;
|
|
218
309
|
const requested = options.dimensions;
|
|
219
|
-
const dimensions = requested ??
|
|
310
|
+
const dimensions = requested ?? facts?.dimensions;
|
|
220
311
|
if (dimensions === undefined) {
|
|
221
312
|
throw new Error(`bedrockEmbedder: unknown model '${model}' — its vector length is not something this ` +
|
|
222
313
|
`library can know, and reporting a wrong .dimensions silently corrupts a vector store. ` +
|
|
223
314
|
`Pass { dimensions } with the length that model returns.`);
|
|
224
315
|
}
|
|
225
|
-
|
|
226
|
-
|
|
316
|
+
// A model this table has never seen can still be recognisably Titan V2, and
|
|
317
|
+
// for one of those the requested size must still be validated and SENT.
|
|
318
|
+
const sizes = facts?.sizes ??
|
|
319
|
+
(family === 'titan' && model.includes('titan-embed-text-v2') ? TITAN_V2_SIZES : undefined);
|
|
320
|
+
if (requested !== undefined && sizes !== undefined && !sizes.includes(requested)) {
|
|
321
|
+
throw new Error(`bedrockEmbedder: '${model}' returns ${sizes.join(', ')} ` +
|
|
227
322
|
`dimensions — received ${String(requested)}. Asking for a size the model does not ` +
|
|
228
323
|
`produce would store vectors of a length that disagrees with the .dimensions this ` +
|
|
229
324
|
`embedder reports.`);
|
|
230
325
|
}
|
|
231
|
-
|
|
326
|
+
if (requested !== undefined &&
|
|
327
|
+
sizes === undefined &&
|
|
328
|
+
facts !== undefined &&
|
|
329
|
+
requested !== facts.dimensions) {
|
|
330
|
+
throw new Error(`bedrockEmbedder: '${model}' returns ${facts.dimensions}-dimension vectors and takes no ` +
|
|
331
|
+
`size parameter — received ${String(requested)}, which would be reported as ` +
|
|
332
|
+
`.dimensions and never be the length that comes back. Drop \`dimensions\`: this ` +
|
|
333
|
+
`factory already knows this model's size.`);
|
|
334
|
+
}
|
|
335
|
+
const maxInputChars = options.maxInputChars ?? charsFor(facts?.maxInputTokens);
|
|
336
|
+
// Only a size the model actually TAKES is sent. Titan V2 defaults to 1024 on
|
|
337
|
+
// its own side, V1 and Cohere reject the field — so a caller who asked for
|
|
338
|
+
// nothing gets a request body identical to the one before this option
|
|
339
|
+
// existed.
|
|
340
|
+
const sendSize = requested !== undefined && sizes !== undefined;
|
|
232
341
|
let connection;
|
|
233
342
|
/**
|
|
234
343
|
* Resolve the client + command constructor, once, on first embed.
|
|
@@ -271,20 +380,26 @@ export function bedrockEmbedder(options = {}) {
|
|
|
271
380
|
return connection;
|
|
272
381
|
};
|
|
273
382
|
/**
|
|
274
|
-
*
|
|
275
|
-
*
|
|
276
|
-
*
|
|
277
|
-
*
|
|
383
|
+
* ONE `InvokeModel` round-trip, with the family's body on the way in and the
|
|
384
|
+
* family's field on the way out.
|
|
385
|
+
*
|
|
386
|
+
* Takes a slice of texts rather than one, because how many fit in a call is
|
|
387
|
+
* itself a family fact: Titan embeds a single `inputText`, Cohere embeds up
|
|
388
|
+
* to {@link COHERE_MAX_TEXTS_PER_CALL}. The callers below never have to know
|
|
389
|
+
* which — they hand over texts and get one vector per text back, in order.
|
|
278
390
|
*/
|
|
279
|
-
async function invoke(
|
|
391
|
+
async function invoke(texts, inputType, signal) {
|
|
280
392
|
const conn = connect();
|
|
281
|
-
const body = JSON.stringify(
|
|
282
|
-
|
|
283
|
-
|
|
284
|
-
|
|
285
|
-
|
|
286
|
-
|
|
287
|
-
|
|
393
|
+
const body = JSON.stringify(family === 'cohere'
|
|
394
|
+
? {
|
|
395
|
+
texts: [...texts],
|
|
396
|
+
// Required by the model, so there is no "unset" — see `inputType`.
|
|
397
|
+
input_type: options.inputType ?? inputType,
|
|
398
|
+
}
|
|
399
|
+
: {
|
|
400
|
+
inputText: texts[0],
|
|
401
|
+
...(sendSize && { dimensions: requested }),
|
|
402
|
+
});
|
|
288
403
|
const command = new conn.Command({
|
|
289
404
|
modelId: model,
|
|
290
405
|
contentType: 'application/json',
|
|
@@ -294,7 +409,29 @@ export function bedrockEmbedder(options = {}) {
|
|
|
294
409
|
const out = await (signal
|
|
295
410
|
? conn.client.send(command, { abortSignal: signal })
|
|
296
411
|
: conn.client.send(command));
|
|
297
|
-
return
|
|
412
|
+
return readEmbeddings(out, model, family, texts.length, facts === undefined);
|
|
413
|
+
}
|
|
414
|
+
/**
|
|
415
|
+
* The calls one batch becomes: Titan one text at a time, Cohere in chunks.
|
|
416
|
+
*
|
|
417
|
+
* Sequential rather than parallel either way: callers of the batch path are
|
|
418
|
+
* usually indexing a whole corpus, and `indexCorpus` already fans out over
|
|
419
|
+
* batches with its own bounded parallelism and retry. Racing N more requests
|
|
420
|
+
* inside one of those branches is how a corpus index meets a throttling
|
|
421
|
+
* error.
|
|
422
|
+
*
|
|
423
|
+
* The abort is checked between calls as well as passed into each one, so an
|
|
424
|
+
* aborted batch stops at the next boundary instead of embedding the rest of a
|
|
425
|
+
* corpus nobody is waiting for.
|
|
426
|
+
*/
|
|
427
|
+
async function invokeAll(texts, inputType, signal) {
|
|
428
|
+
const perCall = family === 'cohere' ? COHERE_MAX_TEXTS_PER_CALL : 1;
|
|
429
|
+
const out = [];
|
|
430
|
+
for (let i = 0; i < texts.length; i += perCall) {
|
|
431
|
+
signal?.throwIfAborted();
|
|
432
|
+
out.push(...(await invoke(texts.slice(i, i + perCall), inputType, signal)));
|
|
433
|
+
}
|
|
434
|
+
return out;
|
|
298
435
|
}
|
|
299
436
|
return {
|
|
300
437
|
dimensions,
|
|
@@ -304,24 +441,16 @@ export function bedrockEmbedder(options = {}) {
|
|
|
304
441
|
// than one it cannot stand behind.
|
|
305
442
|
...(maxInputChars !== undefined && { maxInputChars }),
|
|
306
443
|
async embed({ text, signal }) {
|
|
307
|
-
|
|
444
|
+
// ONE text is this library's query shape (`loadRelevant` embeds the
|
|
445
|
+
// question) — so Cohere is told it is a query unless `inputType` says
|
|
446
|
+
// otherwise. Titan ignores the argument entirely.
|
|
447
|
+
return (await invoke([text], 'search_query', signal))[0];
|
|
308
448
|
},
|
|
309
449
|
async embedBatch({ texts, signal }) {
|
|
310
|
-
//
|
|
311
|
-
//
|
|
312
|
-
//
|
|
313
|
-
|
|
314
|
-
// branches is how a corpus index meets a throttling error.
|
|
315
|
-
//
|
|
316
|
-
// The abort is checked between calls as well as passed into each one,
|
|
317
|
-
// so an aborted batch stops at the next boundary instead of embedding
|
|
318
|
-
// the rest of a corpus nobody is waiting for.
|
|
319
|
-
const out = [];
|
|
320
|
-
for (const text of texts) {
|
|
321
|
-
signal?.throwIfAborted();
|
|
322
|
-
out.push(await invoke(text, signal));
|
|
323
|
-
}
|
|
324
|
-
return out;
|
|
450
|
+
// MANY texts is this library's document shape (`indexDocuments`,
|
|
451
|
+
// `embedMessages`), and Cohere embeds a document into a different place
|
|
452
|
+
// from a query on purpose.
|
|
453
|
+
return invokeAll(texts, 'search_document', signal);
|
|
325
454
|
},
|
|
326
455
|
};
|
|
327
456
|
}
|
|
@@ -342,15 +471,26 @@ function loadBedrockRuntimeSdk() {
|
|
|
342
471
|
}
|
|
343
472
|
}
|
|
344
473
|
/**
|
|
345
|
-
* Pull the
|
|
474
|
+
* Pull the vectors out of an `InvokeModel` response, by FAMILY.
|
|
346
475
|
*
|
|
347
476
|
* The SDK hands back `body` as a `Uint8Array`; a hand-rolled or mock client
|
|
348
477
|
* may hand back the decoded object or a string. All three are read, and
|
|
349
478
|
* anything else is refused by NAME rather than returning an empty vector —
|
|
350
479
|
* an embedder that silently returns `[]` writes a row a store will never
|
|
351
480
|
* rank, which is indistinguishable from "the corpus does not mention that".
|
|
481
|
+
*
|
|
482
|
+
* The field differs with the family (`embedding` vs `embeddings`), so the
|
|
483
|
+
* refusal has to as well: told the wrong family, the response is perfectly
|
|
484
|
+
* valid and this is the only place that can say so — which is why the message
|
|
485
|
+
* names `family` when the model id was not one this library knows.
|
|
486
|
+
*
|
|
487
|
+
* @param expected how many vectors were asked for. A count mismatch is refused
|
|
488
|
+
* rather than returned short: the caller pairs vectors with texts by
|
|
489
|
+
* POSITION, so a missing one does not go missing — it silently re-labels
|
|
490
|
+
* every passage after it.
|
|
491
|
+
* @param guessedFamily whether the family was a fallback rather than a fact.
|
|
352
492
|
*/
|
|
353
|
-
function
|
|
493
|
+
function readEmbeddings(response, model, family, expected, guessedFamily) {
|
|
354
494
|
const body = response?.body ?? response;
|
|
355
495
|
let parsed = body;
|
|
356
496
|
if (body instanceof Uint8Array) {
|
|
@@ -359,13 +499,37 @@ function readEmbedding(response, model) {
|
|
|
359
499
|
else if (typeof body === 'string') {
|
|
360
500
|
parsed = JSON.parse(body);
|
|
361
501
|
}
|
|
362
|
-
const
|
|
363
|
-
|
|
364
|
-
|
|
502
|
+
const field = family === 'cohere' ? 'embeddings' : 'embedding';
|
|
503
|
+
const raw = parsed?.[field];
|
|
504
|
+
// Cohere answers `{ embeddings: [[...]] }`, or `{ embeddings: { float: [[...]] } }`
|
|
505
|
+
// when a caller asked for typed embeddings. This never asks, and reads both.
|
|
506
|
+
const rows = family === 'cohere'
|
|
507
|
+
? Array.isArray(raw)
|
|
508
|
+
? raw
|
|
509
|
+
: raw?.float
|
|
510
|
+
: [raw];
|
|
511
|
+
const vectors = Array.isArray(rows) &&
|
|
512
|
+
rows.every((row) => Array.isArray(row) && row.every((n) => typeof n === 'number'))
|
|
513
|
+
? rows
|
|
514
|
+
: undefined;
|
|
515
|
+
if (vectors === undefined) {
|
|
516
|
+
throw new Error(`bedrockEmbedder: '${model}' returned no \`${field}\` array. Bedrock answered with ` +
|
|
365
517
|
`${describeShape(parsed)}, which this adapter cannot read — check the model id names an ` +
|
|
366
|
-
`EMBEDDING model (a text-generation model answers a different shape).`
|
|
518
|
+
`EMBEDDING model (a text-generation model answers a different shape).` +
|
|
519
|
+
(guessedFamily
|
|
520
|
+
? `\n This model id is not one this library knows, so it was sent the ${family} ` +
|
|
521
|
+
`request body. If it is a ${family === 'titan' ? 'Cohere' : 'Titan'} model, pass ` +
|
|
522
|
+
`\`family: '${family === 'titan' ? 'cohere' : 'titan'}'\` — the request body and ` +
|
|
523
|
+
`the response field both differ per family.`
|
|
524
|
+
: ''));
|
|
525
|
+
}
|
|
526
|
+
if (vectors.length !== expected) {
|
|
527
|
+
throw new Error(`bedrockEmbedder: '${model}' was sent ${expected} text(s) and answered with ` +
|
|
528
|
+
`${vectors.length} vector(s). Vectors are paired with texts by POSITION, so a short ` +
|
|
529
|
+
`answer would not lose one passage — it would attach every later vector to the wrong ` +
|
|
530
|
+
`passage, and the corpus would rank confidently and wrongly forever.`);
|
|
367
531
|
}
|
|
368
|
-
return
|
|
532
|
+
return vectors;
|
|
369
533
|
}
|
|
370
534
|
/** Describe a response by shape, never by content — it may carry customer text. */
|
|
371
535
|
function describeShape(value) {
|