akm-cli 0.9.15-beta.1 → 0.9.15-beta.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +175 -13
- package/dist/akm +54 -1
- package/dist/akm-migrate +34 -1
- package/dist/cli.js +37 -1
- package/dist/commands/improve/locks.js +3 -2
- package/dist/commands/sources/installed-stashes.js +58 -16
- package/dist/commands/sources/stash-cli.js +17 -0
- package/dist/core/config/schema/embedding.js +41 -0
- package/dist/core/file-lock.js +49 -15
- package/dist/core/parent-watchdog.js +64 -0
- package/dist/core/run-lock.js +13 -2
- package/dist/indexer/index-rebuild-lock.js +4 -4
- package/dist/indexer/index-written-assets.js +9 -1
- package/dist/indexer/indexer.js +79 -16
- package/dist/indexer/materialize-embeddings.js +299 -33
- package/dist/indexer/search/search-source.js +23 -1
- package/dist/llm/embedders/remote.js +340 -42
- package/dist/scripts/akm-migrate-node.js +398 -77
- package/dist/scripts/akm-migrate.js +398 -77
- package/dist/storage/repositories/embedding-salvage-repository.js +184 -0
- package/dist/storage/repositories/index-schema.js +16 -0
- package/dist/tasks/run/run-native-task.js +23 -1
- package/docs/migration/release-notes/0.9.15.md +85 -4
- package/docs/migration/release-notes/README.md +3 -2
- package/docs/reference/cli.md +28 -3
- package/docs/reference/configuration.md +67 -15
- package/package.json +1 -1
- package/schemas/akm-config.json +39 -0
|
@@ -7,9 +7,10 @@
|
|
|
7
7
|
* Calls the configured `/embeddings` endpoint and L2-normalizes the returned
|
|
8
8
|
* vectors so the scoring pipeline's L2-to-cosine conversion is correct.
|
|
9
9
|
*/
|
|
10
|
-
import { fetchWithTimeout, isHttpUrl, readBodyWithByteCap } from "../../core/common.js";
|
|
10
|
+
import { abortableDelay, backoffDelay, fetchWithTimeout, isHttpUrl, readBodyWithByteCap } from "../../core/common.js";
|
|
11
11
|
import { concurrentMap } from "../../core/concurrent.js";
|
|
12
12
|
import { resolveSecret } from "../../core/config/config.js";
|
|
13
|
+
import { ENV_REFERENCE_PATTERN, SECRET_STORE_REFERENCE_PATTERN } from "../../core/config/schema/primitives.js";
|
|
13
14
|
import { defaultConcurrencyForEndpoint } from "../../core/loopback.js";
|
|
14
15
|
import { redactErrorBody, redactSensitiveText } from "../../core/redaction.js";
|
|
15
16
|
import { warnVerbose } from "../../core/warn.js";
|
|
@@ -24,7 +25,8 @@ import { resolveSecretFromStore } from "../../sources/snapshot-fetchers/secret-s
|
|
|
24
25
|
export const DEFAULT_REMOTE_BATCH_SIZE = 100;
|
|
25
26
|
/**
|
|
26
27
|
* Conservative default token budget per HTTP request when the config gives
|
|
27
|
-
* no better number (`maxTokens`
|
|
28
|
+
* no better number (`maxTokens` — see #956 for why `contextLength`
|
|
29
|
+
* no longer feeds this). #874's measurements:
|
|
28
30
|
* a batch of 100 small docs (~400 KB, ~100K tokens) took 14.8s against a
|
|
29
31
|
* healthy local endpoint — half the 30s request timeout — and a single
|
|
30
32
|
* 128 KB (~24K token) document alone was rejected by the endpoint as
|
|
@@ -36,6 +38,110 @@ export const DEFAULT_TOKEN_BUDGET = 8000;
|
|
|
36
38
|
export function estimateTokenCount(text) {
|
|
37
39
|
return Math.round(text.length / 4);
|
|
38
40
|
}
|
|
41
|
+
/**
|
|
42
|
+
* Default per-document embedding cap (`embedding.maxInputTokens`, #956)
|
|
43
|
+
* — the materializer truncates a document's embedded text to
|
|
44
|
+
* this cap (head only) instead of skipping it outright, so one oversized
|
|
45
|
+
* entry can no longer fail a whole batch. Fragments are not embedded at all
|
|
46
|
+
* (only the entry's own search text is), so this is the only lever on how
|
|
47
|
+
* much of a large document contributes to its vector.
|
|
48
|
+
*/
|
|
49
|
+
export const DEFAULT_MAX_INPUT_TOKENS = 512;
|
|
50
|
+
/**
|
|
51
|
+
* Truncate `text` to at most `maxTokens` (estimated via
|
|
52
|
+
* {@link estimateTokenCount}, the same 4-chars≈1-token rule the batching
|
|
53
|
+
* budget uses), keeping only its head. The cut never splits a UTF-16
|
|
54
|
+
* surrogate pair. Text already at or under the cap is returned unchanged
|
|
55
|
+
* (`truncated: false`) — including empty text, which is never itself
|
|
56
|
+
* "truncated".
|
|
57
|
+
*/
|
|
58
|
+
export function capEmbeddingText(text, maxTokens) {
|
|
59
|
+
if (estimateTokenCount(text) <= maxTokens)
|
|
60
|
+
return { text, truncated: false };
|
|
61
|
+
const charBudget = Math.max(0, maxTokens * 4);
|
|
62
|
+
let cut = Math.min(charBudget, text.length);
|
|
63
|
+
if (cut > 0 && cut < text.length) {
|
|
64
|
+
const code = text.charCodeAt(cut);
|
|
65
|
+
// A low surrogate (0xDC00-0xDFFF) at the cut point means its high
|
|
66
|
+
// surrogate is the character just before it — back off one position so
|
|
67
|
+
// the pair stays together rather than yielding a lone surrogate.
|
|
68
|
+
if (code >= 0xdc00 && code <= 0xdfff)
|
|
69
|
+
cut -= 1;
|
|
70
|
+
}
|
|
71
|
+
return { text: text.slice(0, cut), truncated: true };
|
|
72
|
+
}
|
|
73
|
+
/**
|
|
74
|
+
* Default per-request timeout when `embedding.timeoutMs` is unset (#954).
|
|
75
|
+
* The prior fixed 30s cut off exactly the field-report case: a
|
|
76
|
+
* local model server on an 8000-token (`DEFAULT_TOKEN_BUDGET`) batch
|
|
77
|
+
* legitimately takes longer than that, and the timeout fired mid-response
|
|
78
|
+
* with no retry — every batch it hit was silently dropped for the rest of
|
|
79
|
+
* an hours-long run. 120s comfortably covers a slow local batch while still
|
|
80
|
+
* bounding a genuinely dead endpoint to a few minutes, not forever.
|
|
81
|
+
*/
|
|
82
|
+
export const DEFAULT_EMBEDDING_TIMEOUT_MS = 120_000;
|
|
83
|
+
/** Resolve the effective per-request timeout: `embedding.timeoutMs` when set, else the default above. */
|
|
84
|
+
export function resolveEmbeddingTimeoutMs(config) {
|
|
85
|
+
return config.timeoutMs ?? DEFAULT_EMBEDDING_TIMEOUT_MS;
|
|
86
|
+
}
|
|
87
|
+
/**
|
|
88
|
+
* Scale the per-request timeout down for a smaller-than-budget request
|
|
89
|
+
* (#954, field-report follow-up): `embedding.timeoutMs` /
|
|
90
|
+
* {@link resolveEmbeddingTimeoutMs} is the budget for a request at the FULL
|
|
91
|
+
* token budget; a batch using only a fraction of it gets a proportionally
|
|
92
|
+
* smaller timeout, floored at 30s and never above the configured
|
|
93
|
+
* `timeoutMs` itself, so a dead server is detected in seconds on the common
|
|
94
|
+
* case of small documents instead of always waiting out the full configured
|
|
95
|
+
* budget.
|
|
96
|
+
*/
|
|
97
|
+
export function scaleEmbeddingTimeoutMs(timeoutMs, requestTokens, tokenBudget) {
|
|
98
|
+
const scaled = tokenBudget > 0 ? timeoutMs * (requestTokens / tokenBudget) : timeoutMs;
|
|
99
|
+
return Math.min(Math.max(scaled, 30_000), timeoutMs);
|
|
100
|
+
}
|
|
101
|
+
/**
|
|
102
|
+
* True when `err` is a request- or body-read timeout (#954) —
|
|
103
|
+
* `fetchWithTimeout`'s connection/header timeout ("Request timed out
|
|
104
|
+
* after...") or `readBodyWithByteCap`'s body-phase `BodyReadTimeoutError`.
|
|
105
|
+
* Only this failure mode gets the back-off-and-retry treatment:
|
|
106
|
+
* the field evidence was specifically that a timed-out request keeps
|
|
107
|
+
* computing server-side, so abandoning it immediately (the prior
|
|
108
|
+
* behavior) just grows the provider's queue further. A genuine network/HTTP
|
|
109
|
+
* failure (connection refused, malformed response, a real error response)
|
|
110
|
+
* has no such still-in-flight hazard and keeps the original
|
|
111
|
+
* skip-immediately behavior.
|
|
112
|
+
*/
|
|
113
|
+
export function isEmbeddingTimeoutError(err) {
|
|
114
|
+
if (!(err instanceof Error))
|
|
115
|
+
return false;
|
|
116
|
+
if (err.name === "BodyReadTimeoutError")
|
|
117
|
+
return true;
|
|
118
|
+
return err.message.startsWith("Request timed out after ");
|
|
119
|
+
}
|
|
120
|
+
/** TEST-ONLY seam: override the backoff base/max so retry-backoff tests run fast without waiting real seconds. */
|
|
121
|
+
let embeddingTimeoutBackoffOverrideForTests;
|
|
122
|
+
/** TEST-ONLY. Pass undefined to restore the real 5s/60s backoff. */
|
|
123
|
+
export function _setEmbeddingTimeoutBackoffForTests(config) {
|
|
124
|
+
embeddingTimeoutBackoffOverrideForTests = config;
|
|
125
|
+
}
|
|
126
|
+
/**
|
|
127
|
+
* Backoff before the single same-size retry on a request timeout
|
|
128
|
+
* (#954) — reuses the same jittered
|
|
129
|
+
* exponential formula {@link backoffDelay} uses for the rest of the
|
|
130
|
+
* codebase's retry paths, at a base of "5s, doubling, capped at 60s".
|
|
131
|
+
*
|
|
132
|
+
* `timeoutAttempt` (#954, field-report follow-up) is how many times the
|
|
133
|
+
* SAME-SIZE-retry-then-split chain has already split before reaching this
|
|
134
|
+
* size — 0 at the top level. The first-timeout backoff for each successive,
|
|
135
|
+
* smaller size after a split grows with it (5s, then doubling, capped at
|
|
136
|
+
* 60s) instead of every split resetting to a flat ~5s: the field evidence
|
|
137
|
+
* was that an abandoned request keeps computing server-side, so a server
|
|
138
|
+
* already draining a whole chain of abandoned requests needs progressively
|
|
139
|
+
* more room, not the same fixed pause at every size.
|
|
140
|
+
*/
|
|
141
|
+
export function embeddingTimeoutRetryBackoffMs(timeoutAttempt = 0) {
|
|
142
|
+
const { baseMs, maxMs } = embeddingTimeoutBackoffOverrideForTests ?? { baseMs: 5_000, maxMs: 60_000 };
|
|
143
|
+
return backoffDelay(timeoutAttempt, baseMs, maxMs);
|
|
144
|
+
}
|
|
39
145
|
/**
|
|
40
146
|
* Distinguishes a batch rejected because it exceeded the endpoint's context
|
|
41
147
|
* window from every other failure mode (network error, 5xx, malformed
|
|
@@ -50,8 +156,15 @@ export class ContextExceededError extends Error {
|
|
|
50
156
|
this.name = "ContextExceededError";
|
|
51
157
|
}
|
|
52
158
|
}
|
|
53
|
-
/**
|
|
54
|
-
|
|
159
|
+
/**
|
|
160
|
+
* Patterns providers use to report a request too large for the model's
|
|
161
|
+
* context window. `input is too large to process`/`physical batch size`/
|
|
162
|
+
* `ubatch` (#954) cover llama.cpp's own physical-batch
|
|
163
|
+
* rejection (HTTP 500, e.g. "input is too large to process. increase the
|
|
164
|
+
* physical batch size"), which was previously an unrecognized generic
|
|
165
|
+
* failure — the whole batch was dropped instead of split and retried.
|
|
166
|
+
*/
|
|
167
|
+
const CONTEXT_EXCEEDED_PATTERN = /exceed_context_size_error|context size|context length|too many tokens|input is too large to process|physical batch size|ubatch/i;
|
|
55
168
|
/**
|
|
56
169
|
* True when an HTTP failure means "this request's input is too large for the
|
|
57
170
|
* endpoint's context window" rather than some other failure. HTTP 413
|
|
@@ -65,16 +178,25 @@ export function isContextExceededResponse(status, body) {
|
|
|
65
178
|
}
|
|
66
179
|
/**
|
|
67
180
|
* Resolve the effective in-flight request window for `RemoteEmbedder.embedBatch`.
|
|
68
|
-
*
|
|
69
|
-
* via the shared `defaultConcurrencyForEndpoint`
|
|
70
|
-
* the same lowest-common-denominator rule
|
|
71
|
-
* (`src/indexer/indexer.ts`) uses.
|
|
72
|
-
*
|
|
73
|
-
* `embedding.
|
|
74
|
-
*
|
|
75
|
-
*
|
|
181
|
+
* Default (unset `embedding.concurrency`): 1 for a loopback endpoint, 2 for a
|
|
182
|
+
* remote one, via the shared `defaultConcurrencyForEndpoint`
|
|
183
|
+
* (`src/core/loopback.ts`), the same lowest-common-denominator rule
|
|
184
|
+
* `getDefaultLlmConcurrency` (`src/indexer/indexer.ts`) uses.
|
|
185
|
+
*
|
|
186
|
+
* `embedding.concurrency` (#954) overrides this default in
|
|
187
|
+
* either direction, bounded 1-16 at the config schema — added after field
|
|
188
|
+
* evidence that a multi-slot local server (llama.cpp `--parallel N`, vLLM)
|
|
189
|
+
* genuinely serves parallel requests and was left idle by the fixed default.
|
|
190
|
+
* Request SIZE remains the first throughput lever regardless:
|
|
191
|
+
* `embedding.batchSize` (document cap) and `embedding.maxTokens` (request
|
|
192
|
+
* token budget — see #956; `contextLength` no longer feeds it)
|
|
193
|
+
* reach a larger batch per request, which is where most of the win is for a
|
|
194
|
+
* single-slot server — a 32-input batch takes about the same wall time as
|
|
195
|
+
* one input against a healthy endpoint.
|
|
76
196
|
*/
|
|
77
197
|
export function resolveEmbeddingConcurrency(config) {
|
|
198
|
+
if (typeof config.concurrency === "number")
|
|
199
|
+
return config.concurrency;
|
|
78
200
|
return defaultConcurrencyForEndpoint(config.endpoint);
|
|
79
201
|
}
|
|
80
202
|
/**
|
|
@@ -138,6 +260,7 @@ export class RemoteEmbedder {
|
|
|
138
260
|
if (ollamaOpts) {
|
|
139
261
|
body.options = ollamaOpts;
|
|
140
262
|
}
|
|
263
|
+
const timeoutMs = resolveEmbeddingTimeoutMs(this.config);
|
|
141
264
|
// `signal` MUST go through fetchWithTimeout's dedicated 4th parameter, not
|
|
142
265
|
// the RequestInit: fetchWithTimeout replaces `opts.signal` with its own
|
|
143
266
|
// controller (`{ ...opts, signal: controller.signal }`), so a signal passed
|
|
@@ -146,16 +269,16 @@ export class RemoteEmbedder {
|
|
|
146
269
|
method: "POST",
|
|
147
270
|
headers,
|
|
148
271
|
body: JSON.stringify(body),
|
|
149
|
-
},
|
|
272
|
+
}, timeoutMs, signal);
|
|
150
273
|
if (!response.ok) {
|
|
151
|
-
const errBody = await readBodyWithByteCap(response, undefined, { bodyTimeoutMs:
|
|
274
|
+
const errBody = await readBodyWithByteCap(response, undefined, { bodyTimeoutMs: timeoutMs, signal }).catch((err) => {
|
|
152
275
|
if (signal?.aborted)
|
|
153
276
|
throw err;
|
|
154
277
|
return "";
|
|
155
278
|
});
|
|
156
279
|
throw new Error(`Embedding request failed (${response.status}): ${this.safeErrorBody(errBody)}`);
|
|
157
280
|
}
|
|
158
|
-
const json = JSON.parse(await readBodyWithByteCap(response, undefined, { bodyTimeoutMs:
|
|
281
|
+
const json = JSON.parse(await readBodyWithByteCap(response, undefined, { bodyTimeoutMs: timeoutMs, signal }));
|
|
159
282
|
if (!json.data?.[0]?.embedding) {
|
|
160
283
|
throw new Error(`Unexpected embedding response format: missing data[0].embedding.${embeddingEndpointPathHint(this.endpoint)}`);
|
|
161
284
|
}
|
|
@@ -197,8 +320,25 @@ export class RemoteEmbedder {
|
|
|
197
320
|
* {@link isContextExceededResponse}) is split in half and retried
|
|
198
321
|
* recursively rather than skipped outright, down to individual documents; a
|
|
199
322
|
* single document that still fails this way becomes a genuine
|
|
200
|
-
* `context-window-exceeded` skip.
|
|
201
|
-
*
|
|
323
|
+
* `context-window-exceeded` skip.
|
|
324
|
+
*
|
|
325
|
+
* A request TIMEOUT (see {@link isEmbeddingTimeoutError}) never drops the
|
|
326
|
+
* batch outright (#954): the field evidence
|
|
327
|
+
* was that akm abandoning a timed-out request does not stop the server
|
|
328
|
+
* from still computing it, so immediately skipping (or immediately
|
|
329
|
+
* splitting, the prior behavior) just let the provider's queue grow
|
|
330
|
+
* while every following batch died the same way. Instead, on a timeout,
|
|
331
|
+
* this backs off ({@link embeddingTimeoutRetryBackoffMs}) and retries the
|
|
332
|
+
* SAME request once; a second timeout splits it in half (like a
|
|
333
|
+
* context-size rejection) and retries each half the same way, down to
|
|
334
|
+
* single documents — a single document that times out twice is finally
|
|
335
|
+
* skipped. Every other failure (network error, malformed response, a
|
|
336
|
+
* non-timeout HTTP failure) keeps the original skip-the-whole-batch-
|
|
337
|
+
* immediately behavior, at any size. The per-request timeout itself also
|
|
338
|
+
* scales down with the request's estimated size via
|
|
339
|
+
* {@link scaleEmbeddingTimeoutMs}, so a dead server is detected in seconds
|
|
340
|
+
* on a small batch rather than always waiting out the full configured
|
|
341
|
+
* `embedding.timeoutMs`.
|
|
202
342
|
*/
|
|
203
343
|
async embedBatch(texts, signal, onSkip, onBatch) {
|
|
204
344
|
if (texts.length === 0)
|
|
@@ -206,9 +346,15 @@ export class RemoteEmbedder {
|
|
|
206
346
|
const results = new Array(texts.length).fill(undefined);
|
|
207
347
|
const headers = this.buildHeaders();
|
|
208
348
|
const ollamaOpts = resolveOllamaOptions(this.config);
|
|
209
|
-
|
|
349
|
+
// #956: `contextLength` is Ollama's `num_ctx` ONLY (see
|
|
350
|
+
// resolveOllamaOptions below) — it used to double as this client-side
|
|
351
|
+
// request budget too, so a config author setting it for one purpose
|
|
352
|
+
// silently changed the other. `maxTokens` is the sole knob for the
|
|
353
|
+
// request budget now; unset falls back to DEFAULT_TOKEN_BUDGET.
|
|
354
|
+
const tokenBudget = this.config.maxTokens ?? DEFAULT_TOKEN_BUDGET;
|
|
210
355
|
const maxCount = this.config.batchSize ?? DEFAULT_REMOTE_BATCH_SIZE;
|
|
211
356
|
const textBatches = buildTokenBoundedBatches(texts, tokenBudget, maxCount);
|
|
357
|
+
const configuredTimeoutMs = resolveEmbeddingTimeoutMs(this.config);
|
|
212
358
|
// Stops the pool from claiming any FURTHER provider batch once the
|
|
213
359
|
// caller's onBatch has failed once (the materializer's transaction
|
|
214
360
|
// failed, so a subsequent commit would just fail again) — dispatching
|
|
@@ -242,7 +388,7 @@ export class RemoteEmbedder {
|
|
|
242
388
|
// fabricated "batch-request-failed" skip (#954). Checked (and rethrown)
|
|
243
389
|
// once the pool drains, the same way `signal?.aborted` is today.
|
|
244
390
|
let firstOnBatchError;
|
|
245
|
-
const commitBatch = (indices, embeddings, model) => {
|
|
391
|
+
const commitBatch = (indices, embeddings, model, outcome) => {
|
|
246
392
|
if (!onBatch)
|
|
247
393
|
return;
|
|
248
394
|
// Once persistence has failed once, an already in-flight batch that
|
|
@@ -252,7 +398,7 @@ export class RemoteEmbedder {
|
|
|
252
398
|
if (firstOnBatchError !== undefined)
|
|
253
399
|
return;
|
|
254
400
|
try {
|
|
255
|
-
onBatch(indices, embeddings, model);
|
|
401
|
+
onBatch(indices, embeddings, model, outcome);
|
|
256
402
|
}
|
|
257
403
|
catch (err) {
|
|
258
404
|
firstOnBatchError = err;
|
|
@@ -260,22 +406,51 @@ export class RemoteEmbedder {
|
|
|
260
406
|
}
|
|
261
407
|
};
|
|
262
408
|
// Requests a single provider batch (by index list), recursing on a
|
|
263
|
-
// context-size rejection
|
|
264
|
-
//
|
|
265
|
-
//
|
|
266
|
-
//
|
|
267
|
-
//
|
|
268
|
-
|
|
409
|
+
// context-size rejection OR a repeated timeout (#954). Never
|
|
410
|
+
// throws except to propagate a genuine caller abort — every other
|
|
411
|
+
// outcome (success or a non-abort failure) resolves normally after
|
|
412
|
+
// reporting via onSkip/onBatch. `onBatch` fires only once this
|
|
413
|
+
// try/catch has already settled success vs. failure, so a throw from it
|
|
414
|
+
// is never caught and reclassified by this block.
|
|
415
|
+
//
|
|
416
|
+
// `isTimeoutRetry` marks the SECOND attempt at this exact `indices`
|
|
417
|
+
// (after the one same-size backoff-and-retry) — a second
|
|
418
|
+
// timeout at that point splits or terminally skips rather than backing
|
|
419
|
+
// off again.
|
|
420
|
+
//
|
|
421
|
+
// `timeoutAttempt` (#954, field-report follow-up) counts how many splits
|
|
422
|
+
// down the timeout-retry chain this call is: 0 at the top level, then
|
|
423
|
+
// +1 each time a second timeout at one size splits into two smaller
|
|
424
|
+
// requests. It is the `attempt` fed to {@link embeddingTimeoutRetryBackoffMs}
|
|
425
|
+
// so each successive size's first-timeout backoff is longer than the
|
|
426
|
+
// last (5s, doubling, capped at 60s) instead of every split restarting
|
|
427
|
+
// at the same ~5s delay — a server still draining a whole run of
|
|
428
|
+
// abandoned requests needs more room the deeper the chain goes, not the
|
|
429
|
+
// same fixed pause every time.
|
|
430
|
+
const requestAndCommit = async (indices, batchIndex, isTimeoutRetry = false, timeoutAttempt = 0) => {
|
|
431
|
+
// A concurrent batch may have tripped the circuit breaker (or failed
|
|
432
|
+
// onBatch) while this call was queued behind a split or a backoff —
|
|
433
|
+
// never let a deeper recursive call make a request that can no longer
|
|
434
|
+
// be reported, mirroring how the pool below never claims a
|
|
435
|
+
// not-yet-started textBatch once dispatch has stopped.
|
|
436
|
+
if (dispatchAbort.signal.aborted)
|
|
437
|
+
return;
|
|
269
438
|
const batch = indices.map((i) => texts[i]);
|
|
439
|
+
const requestTokens = batch.reduce((sum, text) => sum + estimateTokenCount(text), 0);
|
|
440
|
+
const requestTimeoutMs = scaleEmbeddingTimeoutMs(configuredTimeoutMs, requestTokens, tokenBudget);
|
|
441
|
+
const requestStart = Date.now();
|
|
270
442
|
let batchEmbeddings;
|
|
271
443
|
let responseModel;
|
|
444
|
+
let outcome;
|
|
445
|
+
let failureReason;
|
|
272
446
|
try {
|
|
273
|
-
const { vectors, model } = await this.requestBatch(batch, headers, ollamaOpts, signal);
|
|
447
|
+
const { vectors, model } = await this.requestBatch(batch, headers, ollamaOpts, requestTimeoutMs, signal);
|
|
274
448
|
for (let k = 0; k < indices.length; k++) {
|
|
275
449
|
results[indices[k]] = vectors[k];
|
|
276
450
|
}
|
|
277
451
|
batchEmbeddings = indices.map((i) => results[i]);
|
|
278
452
|
responseModel = model;
|
|
453
|
+
outcome = "stored";
|
|
279
454
|
}
|
|
280
455
|
catch (err) {
|
|
281
456
|
// A caller abort must still propagate — it is not a "this batch
|
|
@@ -284,21 +459,101 @@ export class RemoteEmbedder {
|
|
|
284
459
|
throw err;
|
|
285
460
|
if (err instanceof ContextExceededError && indices.length > 1) {
|
|
286
461
|
const mid = Math.ceil(indices.length / 2);
|
|
287
|
-
await requestAndCommit(indices.slice(0, mid));
|
|
288
|
-
await requestAndCommit(indices.slice(mid));
|
|
462
|
+
await requestAndCommit(indices.slice(0, mid), batchIndex, false, timeoutAttempt);
|
|
463
|
+
await requestAndCommit(indices.slice(mid), batchIndex, false, timeoutAttempt);
|
|
464
|
+
return;
|
|
465
|
+
}
|
|
466
|
+
const timedOut = isEmbeddingTimeoutError(err);
|
|
467
|
+
if (timedOut && !isTimeoutRetry) {
|
|
468
|
+
// First timeout at this size: back off so the provider can drain
|
|
469
|
+
// the abandoned request, then retry the SAME request once before
|
|
470
|
+
// ever splitting or skipping (#954). The backoff grows with
|
|
471
|
+
// `timeoutAttempt`, not a flat ~5s every time, so a chain of
|
|
472
|
+
// splits down to smaller and smaller requests gives the provider
|
|
473
|
+
// proportionally more room to drain each time.
|
|
474
|
+
const backoffMs = embeddingTimeoutRetryBackoffMs(timeoutAttempt);
|
|
475
|
+
warnVerbose(`[embed] batch of ${batch.length} document(s) timed out after ${requestTimeoutMs}ms; retrying once after a ${Math.round(backoffMs)}ms backoff`);
|
|
476
|
+
// Default-level notice (#954 field-report follow-up), not just the
|
|
477
|
+
// verbose line above — a run silently waiting out a multi-minute
|
|
478
|
+
// back-off looked identical to a hang otherwise. Nothing has
|
|
479
|
+
// failed or succeeded yet, so there is nothing to persist:
|
|
480
|
+
// `embeddings` are all `undefined` and the materializer's onBatch
|
|
481
|
+
// must not touch storage for this event.
|
|
482
|
+
commitBatch(indices, indices.map(() => undefined), undefined, {
|
|
483
|
+
batchIndex,
|
|
484
|
+
batchCount: textBatches.length,
|
|
485
|
+
docCount: indices.length,
|
|
486
|
+
requestTokens,
|
|
487
|
+
elapsedMs: backoffMs,
|
|
488
|
+
outcome: "retrying",
|
|
489
|
+
reason: "timed out",
|
|
490
|
+
});
|
|
491
|
+
await abortableDelay(backoffMs, signal, "embedding interrupted during retry backoff");
|
|
492
|
+
if (!dispatchAbort.signal.aborted) {
|
|
493
|
+
return requestAndCommit(indices, batchIndex, true, timeoutAttempt);
|
|
494
|
+
}
|
|
495
|
+
// Dispatch was stopped (by another batch's circuit-breaker trip)
|
|
496
|
+
// while this one was backing off — fall through and skip below
|
|
497
|
+
// instead of issuing a request that can no longer be reported.
|
|
498
|
+
}
|
|
499
|
+
else if (timedOut && indices.length > 1) {
|
|
500
|
+
// Timed out again on the retry: split rather than skip outright —
|
|
501
|
+
// the provider may still fit it once it is smaller (#954),
|
|
502
|
+
// the same treatment a context-size rejection gets.
|
|
503
|
+
// Each half's OWN first-timeout backoff (#954 follow-up) starts
|
|
504
|
+
// one attempt further down the chain than this size's did.
|
|
505
|
+
const mid = Math.ceil(indices.length / 2);
|
|
506
|
+
await requestAndCommit(indices.slice(0, mid), batchIndex, false, timeoutAttempt + 1);
|
|
507
|
+
await requestAndCommit(indices.slice(mid), batchIndex, false, timeoutAttempt + 1);
|
|
289
508
|
return;
|
|
290
509
|
}
|
|
291
510
|
const message = err instanceof Error ? err.message : String(err);
|
|
292
511
|
const reason = err instanceof ContextExceededError ? "context-window-exceeded" : "batch-request-failed";
|
|
293
|
-
|
|
294
|
-
for
|
|
295
|
-
|
|
512
|
+
const failureKind = reason === "batch-request-failed" ? (timedOut ? "timeout" : "network-error") : undefined;
|
|
513
|
+
// Default-level visibility for a failed batch (not verbose-only) is
|
|
514
|
+
// still guaranteed here — just not via warn(). The `commitBatch` call
|
|
515
|
+
// below carries `outcome: "failed"` and this `message` as `reason`
|
|
516
|
+
// through `onBatch`, and materialize-embeddings.ts's per-batch line
|
|
517
|
+
// (also default-level) prints it from there. A warn() call here used
|
|
518
|
+
// to print the identical event a second time on stderr — the same
|
|
519
|
+
// class of double-print bug fixed for the truncation/re-embed-reason
|
|
520
|
+
// lines in materialize-embeddings.ts (#954, field-report follow-up).
|
|
521
|
+
// Per-entry batch-mapping detail stays verbose-only
|
|
522
|
+
// (materialize-embeddings.ts).
|
|
523
|
+
let stopRequested = false;
|
|
524
|
+
for (const [k, idx] of indices.entries()) {
|
|
525
|
+
if (onSkip?.({
|
|
526
|
+
index: idx,
|
|
527
|
+
reason,
|
|
528
|
+
message,
|
|
529
|
+
batchStart: k === 0,
|
|
530
|
+
batchSize: indices.length,
|
|
531
|
+
failureKind,
|
|
532
|
+
}) === false)
|
|
533
|
+
stopRequested = true;
|
|
296
534
|
}
|
|
535
|
+
// #954: the caller's circuit breaker asked to stop —
|
|
536
|
+
// gate further dispatch through the SAME dispatchAbort controller
|
|
537
|
+
// the onBatch-throw path above uses, but resolve this call normally
|
|
538
|
+
// (never reject) with whatever results already landed, since this is
|
|
539
|
+
// a policy decision, not a persistence failure.
|
|
540
|
+
if (stopRequested)
|
|
541
|
+
stopDispatch();
|
|
297
542
|
batchEmbeddings = indices.map(() => undefined);
|
|
543
|
+
outcome = "failed";
|
|
544
|
+
failureReason = message;
|
|
298
545
|
}
|
|
299
|
-
commitBatch(indices, batchEmbeddings, responseModel
|
|
546
|
+
commitBatch(indices, batchEmbeddings, responseModel, {
|
|
547
|
+
batchIndex,
|
|
548
|
+
batchCount: textBatches.length,
|
|
549
|
+
docCount: indices.length,
|
|
550
|
+
requestTokens,
|
|
551
|
+
elapsedMs: Date.now() - requestStart,
|
|
552
|
+
outcome,
|
|
553
|
+
reason: failureReason,
|
|
554
|
+
});
|
|
300
555
|
};
|
|
301
|
-
const runProviderBatch = async (textBatch) => {
|
|
556
|
+
const runProviderBatch = async (textBatch, batchIndex) => {
|
|
302
557
|
if (textBatch.oversized) {
|
|
303
558
|
const idx = textBatch.indices[0];
|
|
304
559
|
const estTokens = estimateTokenCount(texts[idx]);
|
|
@@ -306,19 +561,37 @@ export class RemoteEmbedder {
|
|
|
306
561
|
index: idx,
|
|
307
562
|
reason: "context-window-exceeded",
|
|
308
563
|
message: `Document estimated at ${estTokens} tokens exceeds the ${tokenBudget}-token embedding budget; skipped.`,
|
|
564
|
+
batchStart: true,
|
|
565
|
+
batchSize: 1,
|
|
566
|
+
});
|
|
567
|
+
// Never made a request — excluded from the default-level per-batch
|
|
568
|
+
// line (there is no request outcome to report), but still counted
|
|
569
|
+
// in the run's oversized-skip total via `onSkip` above.
|
|
570
|
+
commitBatch([idx], [undefined], undefined, {
|
|
571
|
+
batchIndex,
|
|
572
|
+
batchCount: textBatches.length,
|
|
573
|
+
docCount: 1,
|
|
574
|
+
requestTokens: estTokens,
|
|
575
|
+
elapsedMs: 0,
|
|
576
|
+
outcome: "failed",
|
|
577
|
+
reason: "oversized",
|
|
309
578
|
});
|
|
310
|
-
commitBatch([idx], [undefined]);
|
|
311
579
|
return;
|
|
312
580
|
}
|
|
313
|
-
await requestAndCommit(textBatch.indices);
|
|
581
|
+
await requestAndCommit(textBatch.indices, batchIndex);
|
|
314
582
|
};
|
|
315
583
|
const concurrency = resolveEmbeddingConcurrency(this.config);
|
|
316
584
|
// concurrentMap swallows a thrown fn (per-item, results discarded here —
|
|
317
585
|
// requestAndCommit only throws to signal a caller abort), so abort must
|
|
318
586
|
// be re-checked once the pool has drained rather than relying on the
|
|
319
587
|
// throw itself to escape. Dispatch is gated on `dispatchAbort`, not the
|
|
320
|
-
// caller's `signal` directly — see the comment above.
|
|
321
|
-
|
|
588
|
+
// caller's `signal` directly — see the comment above. `batchIndex` is
|
|
589
|
+
// 1-based (matches the human-readable "batch N/Total" line) and comes
|
|
590
|
+
// from `concurrentMap`'s own 0-based item index, not a separately
|
|
591
|
+
// tracked counter, so it stays correct under concurrency.
|
|
592
|
+
await concurrentMap(textBatches, (textBatch, i) => runProviderBatch(textBatch, i + 1), concurrency, {
|
|
593
|
+
signal: dispatchAbort.signal,
|
|
594
|
+
});
|
|
322
595
|
if (callerAbortListener)
|
|
323
596
|
signal?.removeEventListener("abort", callerAbortListener);
|
|
324
597
|
if (signal?.aborted) {
|
|
@@ -335,8 +608,14 @@ export class RemoteEmbedder {
|
|
|
335
608
|
* used by the embedding-fingerprint canary to verify a config-string
|
|
336
609
|
* rename against what the endpoint actually served, not just re-assert the
|
|
337
610
|
* configured string. Throws on any failure.
|
|
611
|
+
*
|
|
612
|
+
* `timeoutMs` is the caller's ALREADY-SCALED per-request timeout (#954
|
|
613
|
+
* — see {@link scaleEmbeddingTimeoutMs}), not re-resolved here:
|
|
614
|
+
* `embedBatch` computes it per request from that request's own size so a
|
|
615
|
+
* split-down retry gets a smaller, size-appropriate timeout rather than
|
|
616
|
+
* always the full configured `embedding.timeoutMs`.
|
|
338
617
|
*/
|
|
339
|
-
async requestBatch(batch, headers, ollamaOpts, signal) {
|
|
618
|
+
async requestBatch(batch, headers, ollamaOpts, timeoutMs, signal) {
|
|
340
619
|
const body = {
|
|
341
620
|
input: batch,
|
|
342
621
|
model: this.model,
|
|
@@ -353,9 +632,9 @@ export class RemoteEmbedder {
|
|
|
353
632
|
method: "POST",
|
|
354
633
|
headers,
|
|
355
634
|
body: JSON.stringify(body),
|
|
356
|
-
},
|
|
635
|
+
}, timeoutMs, signal);
|
|
357
636
|
if (!response.ok) {
|
|
358
|
-
const respBody = await readBodyWithByteCap(response, undefined, { bodyTimeoutMs:
|
|
637
|
+
const respBody = await readBodyWithByteCap(response, undefined, { bodyTimeoutMs: timeoutMs, signal }).catch((err) => {
|
|
359
638
|
if (signal?.aborted)
|
|
360
639
|
throw err;
|
|
361
640
|
return "";
|
|
@@ -366,7 +645,7 @@ export class RemoteEmbedder {
|
|
|
366
645
|
}
|
|
367
646
|
throw new Error(message);
|
|
368
647
|
}
|
|
369
|
-
const json = JSON.parse(await readBodyWithByteCap(response, undefined, { bodyTimeoutMs:
|
|
648
|
+
const json = JSON.parse(await readBodyWithByteCap(response, undefined, { bodyTimeoutMs: timeoutMs, signal }));
|
|
370
649
|
if (!json.data || json.data.length !== batch.length) {
|
|
371
650
|
throw new Error(`Unexpected embedding batch response: expected ${batch.length} embeddings, got ${json.data?.length ?? 0}.${embeddingEndpointPathHint(this.endpoint)}`);
|
|
372
651
|
}
|
|
@@ -471,3 +750,22 @@ function resolveOllamaOptions(config) {
|
|
|
471
750
|
export function hasRemoteEndpoint(config) {
|
|
472
751
|
return isHttpUrl(config.endpoint);
|
|
473
752
|
}
|
|
753
|
+
/**
|
|
754
|
+
* Describe WHERE an `embedding.apiKey` came from, never its value — the
|
|
755
|
+
* actionable outcome of the #953 field gap: `resolveSecret` throws on an
|
|
756
|
+
* unresolvable `secret://` reference, so a keyless request can only mean
|
|
757
|
+
* `embedding.apiKey` was absent from the config the run actually loaded (a
|
|
758
|
+
* different config root, scope, or a config edited after the run started).
|
|
759
|
+
* A default-level progress line naming the credential's SOURCE (this
|
|
760
|
+
* helper), printed once before the first provider request, lets a field run
|
|
761
|
+
* self-diagnose that without ever surfacing the secret itself.
|
|
762
|
+
*/
|
|
763
|
+
export function describeEmbeddingCredential(apiKey) {
|
|
764
|
+
if (!apiKey)
|
|
765
|
+
return "none configured";
|
|
766
|
+
if (SECRET_STORE_REFERENCE_PATTERN.test(apiKey))
|
|
767
|
+
return `${apiKey} (store)`;
|
|
768
|
+
if (ENV_REFERENCE_PATTERN.test(apiKey))
|
|
769
|
+
return `${apiKey} (env)`;
|
|
770
|
+
return "literal apiKey";
|
|
771
|
+
}
|