akm-cli 0.9.15-beta.1 → 0.9.15-beta.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +234 -13
- package/dist/akm +54 -1
- package/dist/akm-migrate +34 -1
- package/dist/cli.js +40 -4
- package/dist/commands/health/checks.js +11 -2
- package/dist/commands/health/scheduler-binary.js +120 -0
- package/dist/commands/health.js +9 -0
- package/dist/commands/improve/locks.js +3 -2
- package/dist/commands/sources/installed-stashes.js +58 -16
- package/dist/commands/sources/stash-cli.js +17 -0
- package/dist/core/config/schema/embedding.js +41 -1
- package/dist/core/errors.js +1 -0
- package/dist/core/file-lock.js +49 -15
- package/dist/core/parent-watchdog.js +64 -0
- package/dist/core/run-lock.js +13 -2
- package/dist/indexer/index-rebuild-lock.js +4 -4
- package/dist/indexer/index-written-assets.js +9 -1
- package/dist/indexer/indexer.js +123 -19
- package/dist/indexer/materialize-embeddings.js +345 -37
- package/dist/indexer/search/search-source.js +23 -1
- package/dist/llm/embedders/remote.js +443 -47
- package/dist/scripts/akm-migrate-node.js +454 -85
- package/dist/scripts/akm-migrate.js +454 -85
- package/dist/storage/repositories/embedding-salvage-repository.js +184 -0
- package/dist/storage/repositories/index-schema.js +16 -0
- package/dist/tasks/run/run-native-task.js +23 -1
- package/docs/migration/release-notes/0.9.15.md +103 -4
- package/docs/migration/release-notes/README.md +3 -2
- package/docs/reference/cli.md +55 -8
- package/docs/reference/configuration.md +83 -15
- package/package.json +1 -1
- package/schemas/akm-config.json +36 -9
|
@@ -7,9 +7,10 @@
|
|
|
7
7
|
* Calls the configured `/embeddings` endpoint and L2-normalizes the returned
|
|
8
8
|
* vectors so the scoring pipeline's L2-to-cosine conversion is correct.
|
|
9
9
|
*/
|
|
10
|
-
import { fetchWithTimeout, isHttpUrl, readBodyWithByteCap } from "../../core/common.js";
|
|
10
|
+
import { abortableDelay, backoffDelay, fetchWithTimeout, isHttpUrl, readBodyWithByteCap } from "../../core/common.js";
|
|
11
11
|
import { concurrentMap } from "../../core/concurrent.js";
|
|
12
12
|
import { resolveSecret } from "../../core/config/config.js";
|
|
13
|
+
import { ENV_REFERENCE_PATTERN, SECRET_STORE_REFERENCE_PATTERN } from "../../core/config/schema/primitives.js";
|
|
13
14
|
import { defaultConcurrencyForEndpoint } from "../../core/loopback.js";
|
|
14
15
|
import { redactErrorBody, redactSensitiveText } from "../../core/redaction.js";
|
|
15
16
|
import { warnVerbose } from "../../core/warn.js";
|
|
@@ -24,18 +25,130 @@ import { resolveSecretFromStore } from "../../sources/snapshot-fetchers/secret-s
|
|
|
24
25
|
export const DEFAULT_REMOTE_BATCH_SIZE = 100;
|
|
25
26
|
/**
|
|
26
27
|
* Conservative default token budget per HTTP request when the config gives
|
|
27
|
-
* no better number (`maxTokens`
|
|
28
|
+
* no better number (`maxTokens` — see #956 for why `contextLength`
|
|
29
|
+
* no longer feeds this). #874's measurements:
|
|
28
30
|
* a batch of 100 small docs (~400 KB, ~100K tokens) took 14.8s against a
|
|
29
31
|
* healthy local endpoint — half the 30s request timeout — and a single
|
|
30
32
|
* 128 KB (~24K token) document alone was rejected by the endpoint as
|
|
31
|
-
* exceeding its context size.
|
|
32
|
-
*
|
|
33
|
+
* exceeding its context size.
|
|
34
|
+
*
|
|
35
|
+
* Lowered from 8000 to 6000 (#954, field report on beta.1): the 4-chars-
|
|
36
|
+
* per-token estimator undercounts dense technical text by 7-55%, so 8000
|
|
37
|
+
* against an 8192-token llama.cpp embedder regularly landed real requests
|
|
38
|
+
* over the endpoint's context window. 6000 is the value the field confirmed
|
|
39
|
+
* stops that steady trickle of rejections; `embedBatch`'s run-scoped
|
|
40
|
+
* adaptive budget below still shrinks further, for an endpoint where even
|
|
41
|
+
* this is not enough.
|
|
33
42
|
*/
|
|
34
|
-
export const DEFAULT_TOKEN_BUDGET =
|
|
43
|
+
export const DEFAULT_TOKEN_BUDGET = 6000;
|
|
35
44
|
/** Cheap token estimator: 4 chars ≈ 1 token. Used in verbose logging and error messages. */
|
|
36
45
|
export function estimateTokenCount(text) {
|
|
37
46
|
return Math.round(text.length / 4);
|
|
38
47
|
}
|
|
48
|
+
/**
|
|
49
|
+
* Default per-document embedding cap (`embedding.maxInputTokens`, #956)
|
|
50
|
+
* — the materializer truncates a document's embedded text to
|
|
51
|
+
* this cap (head only) instead of skipping it outright, so one oversized
|
|
52
|
+
* entry can no longer fail a whole batch. Fragments are not embedded at all
|
|
53
|
+
* (only the entry's own search text is), so this is the only lever on how
|
|
54
|
+
* much of a large document contributes to its vector.
|
|
55
|
+
*/
|
|
56
|
+
export const DEFAULT_MAX_INPUT_TOKENS = 512;
|
|
57
|
+
/**
|
|
58
|
+
* Truncate `text` to at most `maxTokens` (estimated via
|
|
59
|
+
* {@link estimateTokenCount}, the same 4-chars≈1-token rule the batching
|
|
60
|
+
* budget uses), keeping only its head. The cut never splits a UTF-16
|
|
61
|
+
* surrogate pair. Text already at or under the cap is returned unchanged
|
|
62
|
+
* (`truncated: false`) — including empty text, which is never itself
|
|
63
|
+
* "truncated".
|
|
64
|
+
*/
|
|
65
|
+
export function capEmbeddingText(text, maxTokens) {
|
|
66
|
+
if (estimateTokenCount(text) <= maxTokens)
|
|
67
|
+
return { text, truncated: false };
|
|
68
|
+
const charBudget = Math.max(0, maxTokens * 4);
|
|
69
|
+
let cut = Math.min(charBudget, text.length);
|
|
70
|
+
if (cut > 0 && cut < text.length) {
|
|
71
|
+
const code = text.charCodeAt(cut);
|
|
72
|
+
// A low surrogate (0xDC00-0xDFFF) at the cut point means its high
|
|
73
|
+
// surrogate is the character just before it — back off one position so
|
|
74
|
+
// the pair stays together rather than yielding a lone surrogate.
|
|
75
|
+
if (code >= 0xdc00 && code <= 0xdfff)
|
|
76
|
+
cut -= 1;
|
|
77
|
+
}
|
|
78
|
+
return { text: text.slice(0, cut), truncated: true };
|
|
79
|
+
}
|
|
80
|
+
/**
|
|
81
|
+
* Default per-request timeout when `embedding.timeoutMs` is unset (#954).
|
|
82
|
+
* The prior fixed 30s cut off exactly the field-report case: a
|
|
83
|
+
* local model server on a full-budget (`DEFAULT_TOKEN_BUDGET`) batch
|
|
84
|
+
* legitimately takes longer than that, and the timeout fired mid-response
|
|
85
|
+
* with no retry — every batch it hit was silently dropped for the rest of
|
|
86
|
+
* an hours-long run. 120s comfortably covers a slow local batch while still
|
|
87
|
+
* bounding a genuinely dead endpoint to a few minutes, not forever.
|
|
88
|
+
*/
|
|
89
|
+
export const DEFAULT_EMBEDDING_TIMEOUT_MS = 120_000;
|
|
90
|
+
/** Resolve the effective per-request timeout: `embedding.timeoutMs` when set, else the default above. */
|
|
91
|
+
export function resolveEmbeddingTimeoutMs(config) {
|
|
92
|
+
return config.timeoutMs ?? DEFAULT_EMBEDDING_TIMEOUT_MS;
|
|
93
|
+
}
|
|
94
|
+
/**
|
|
95
|
+
* Scale the per-request timeout down for a smaller-than-budget request
|
|
96
|
+
* (#954, field-report follow-up): `embedding.timeoutMs` /
|
|
97
|
+
* {@link resolveEmbeddingTimeoutMs} is the budget for a request at the FULL
|
|
98
|
+
* token budget; a batch using only a fraction of it gets a proportionally
|
|
99
|
+
* smaller timeout, floored at 30s and never above the configured
|
|
100
|
+
* `timeoutMs` itself, so a dead server is detected in seconds on the common
|
|
101
|
+
* case of small documents instead of always waiting out the full configured
|
|
102
|
+
* budget.
|
|
103
|
+
*/
|
|
104
|
+
export function scaleEmbeddingTimeoutMs(timeoutMs, requestTokens, tokenBudget) {
|
|
105
|
+
const scaled = tokenBudget > 0 ? timeoutMs * (requestTokens / tokenBudget) : timeoutMs;
|
|
106
|
+
return Math.min(Math.max(scaled, 30_000), timeoutMs);
|
|
107
|
+
}
|
|
108
|
+
/**
|
|
109
|
+
* True when `err` is a request- or body-read timeout (#954) —
|
|
110
|
+
* `fetchWithTimeout`'s connection/header timeout ("Request timed out
|
|
111
|
+
* after...") or `readBodyWithByteCap`'s body-phase `BodyReadTimeoutError`.
|
|
112
|
+
* Only this failure mode gets the back-off-and-retry treatment:
|
|
113
|
+
* the field evidence was specifically that a timed-out request keeps
|
|
114
|
+
* computing server-side, so abandoning it immediately (the prior
|
|
115
|
+
* behavior) just grows the provider's queue further. A genuine network/HTTP
|
|
116
|
+
* failure (connection refused, malformed response, a real error response)
|
|
117
|
+
* has no such still-in-flight hazard and keeps the original
|
|
118
|
+
* skip-immediately behavior.
|
|
119
|
+
*/
|
|
120
|
+
export function isEmbeddingTimeoutError(err) {
|
|
121
|
+
if (!(err instanceof Error))
|
|
122
|
+
return false;
|
|
123
|
+
if (err.name === "BodyReadTimeoutError")
|
|
124
|
+
return true;
|
|
125
|
+
return err.message.startsWith("Request timed out after ");
|
|
126
|
+
}
|
|
127
|
+
/** TEST-ONLY seam: override the backoff base/max so retry-backoff tests run fast without waiting real seconds. */
|
|
128
|
+
let embeddingTimeoutBackoffOverrideForTests;
|
|
129
|
+
/** TEST-ONLY. Pass undefined to restore the real 5s/60s backoff. */
|
|
130
|
+
export function _setEmbeddingTimeoutBackoffForTests(config) {
|
|
131
|
+
embeddingTimeoutBackoffOverrideForTests = config;
|
|
132
|
+
}
|
|
133
|
+
/**
|
|
134
|
+
* Backoff before the single same-size retry on a request timeout
|
|
135
|
+
* (#954) — reuses the same jittered
|
|
136
|
+
* exponential formula {@link backoffDelay} uses for the rest of the
|
|
137
|
+
* codebase's retry paths, at a base of "5s, doubling, capped at 60s".
|
|
138
|
+
*
|
|
139
|
+
* `timeoutAttempt` (#954, field-report follow-up) is how many times the
|
|
140
|
+
* SAME-SIZE-retry-then-split chain has already split before reaching this
|
|
141
|
+
* size — 0 at the top level. The first-timeout backoff for each successive,
|
|
142
|
+
* smaller size after a split grows with it (5s, then doubling, capped at
|
|
143
|
+
* 60s) instead of every split resetting to a flat ~5s: the field evidence
|
|
144
|
+
* was that an abandoned request keeps computing server-side, so a server
|
|
145
|
+
* already draining a whole chain of abandoned requests needs progressively
|
|
146
|
+
* more room, not the same fixed pause at every size.
|
|
147
|
+
*/
|
|
148
|
+
export function embeddingTimeoutRetryBackoffMs(timeoutAttempt = 0) {
|
|
149
|
+
const { baseMs, maxMs } = embeddingTimeoutBackoffOverrideForTests ?? { baseMs: 5_000, maxMs: 60_000 };
|
|
150
|
+
return backoffDelay(timeoutAttempt, baseMs, maxMs);
|
|
151
|
+
}
|
|
39
152
|
/**
|
|
40
153
|
* Distinguishes a batch rejected because it exceeded the endpoint's context
|
|
41
154
|
* window from every other failure mode (network error, 5xx, malformed
|
|
@@ -50,8 +163,15 @@ export class ContextExceededError extends Error {
|
|
|
50
163
|
this.name = "ContextExceededError";
|
|
51
164
|
}
|
|
52
165
|
}
|
|
53
|
-
/**
|
|
54
|
-
|
|
166
|
+
/**
|
|
167
|
+
* Patterns providers use to report a request too large for the model's
|
|
168
|
+
* context window. `input is too large to process`/`physical batch size`/
|
|
169
|
+
* `ubatch` (#954) cover llama.cpp's own physical-batch
|
|
170
|
+
* rejection (HTTP 500, e.g. "input is too large to process. increase the
|
|
171
|
+
* physical batch size"), which was previously an unrecognized generic
|
|
172
|
+
* failure — the whole batch was dropped instead of split and retried.
|
|
173
|
+
*/
|
|
174
|
+
const CONTEXT_EXCEEDED_PATTERN = /exceed_context_size_error|context size|context length|too many tokens|input is too large to process|physical batch size|ubatch/i;
|
|
55
175
|
/**
|
|
56
176
|
* True when an HTTP failure means "this request's input is too large for the
|
|
57
177
|
* endpoint's context window" rather than some other failure. HTTP 413
|
|
@@ -65,16 +185,25 @@ export function isContextExceededResponse(status, body) {
|
|
|
65
185
|
}
|
|
66
186
|
/**
|
|
67
187
|
* Resolve the effective in-flight request window for `RemoteEmbedder.embedBatch`.
|
|
68
|
-
*
|
|
69
|
-
* via the shared `defaultConcurrencyForEndpoint`
|
|
70
|
-
* the same lowest-common-denominator rule
|
|
71
|
-
* (`src/indexer/indexer.ts`) uses.
|
|
72
|
-
*
|
|
73
|
-
* `embedding.
|
|
74
|
-
*
|
|
75
|
-
*
|
|
188
|
+
* Default (unset `embedding.concurrency`): 1 for a loopback endpoint, 2 for a
|
|
189
|
+
* remote one, via the shared `defaultConcurrencyForEndpoint`
|
|
190
|
+
* (`src/core/loopback.ts`), the same lowest-common-denominator rule
|
|
191
|
+
* `getDefaultLlmConcurrency` (`src/indexer/indexer.ts`) uses.
|
|
192
|
+
*
|
|
193
|
+
* `embedding.concurrency` (#954) overrides this default in
|
|
194
|
+
* either direction, bounded 1-16 at the config schema — added after field
|
|
195
|
+
* evidence that a multi-slot local server (llama.cpp `--parallel N`, vLLM)
|
|
196
|
+
* genuinely serves parallel requests and was left idle by the fixed default.
|
|
197
|
+
* Request SIZE remains the first throughput lever regardless:
|
|
198
|
+
* `embedding.batchSize` (document cap) and `embedding.maxTokens` (request
|
|
199
|
+
* token budget — see #956; `contextLength` no longer feeds it)
|
|
200
|
+
* reach a larger batch per request, which is where most of the win is for a
|
|
201
|
+
* single-slot server — a 32-input batch takes about the same wall time as
|
|
202
|
+
* one input against a healthy endpoint.
|
|
76
203
|
*/
|
|
77
204
|
export function resolveEmbeddingConcurrency(config) {
|
|
205
|
+
if (typeof config.concurrency === "number")
|
|
206
|
+
return config.concurrency;
|
|
78
207
|
return defaultConcurrencyForEndpoint(config.endpoint);
|
|
79
208
|
}
|
|
80
209
|
/**
|
|
@@ -113,6 +242,21 @@ export function buildTokenBoundedBatches(texts, tokenBudget, maxCount) {
|
|
|
113
242
|
flush();
|
|
114
243
|
return batches;
|
|
115
244
|
}
|
|
245
|
+
/**
|
|
246
|
+
* Shrink factor applied to the effective request budget on the first
|
|
247
|
+
* context-size rejection of an `embedBatch` run (#954, field report on
|
|
248
|
+
* beta.1): one 25% cut absorbs the estimator's measured undercount without
|
|
249
|
+
* repeatedly re-shrinking mid-run — see the "shrink at most once" rule on
|
|
250
|
+
* {@link RemoteEmbedder.embedBatch}.
|
|
251
|
+
*/
|
|
252
|
+
const ADAPTIVE_BUDGET_SHRINK_FACTOR = 0.75;
|
|
253
|
+
/**
|
|
254
|
+
* Floor on the adaptive-budget shrink above, as a multiple of
|
|
255
|
+
* `maxInputTokens` (#954): a request budget below twice the per-document cap
|
|
256
|
+
* could no longer batch more than one document per request, defeating the
|
|
257
|
+
* point of batching at all.
|
|
258
|
+
*/
|
|
259
|
+
const ADAPTIVE_BUDGET_FLOOR_MULTIPLIER = 2;
|
|
116
260
|
export class RemoteEmbedder {
|
|
117
261
|
config;
|
|
118
262
|
endpoint;
|
|
@@ -138,6 +282,7 @@ export class RemoteEmbedder {
|
|
|
138
282
|
if (ollamaOpts) {
|
|
139
283
|
body.options = ollamaOpts;
|
|
140
284
|
}
|
|
285
|
+
const timeoutMs = resolveEmbeddingTimeoutMs(this.config);
|
|
141
286
|
// `signal` MUST go through fetchWithTimeout's dedicated 4th parameter, not
|
|
142
287
|
// the RequestInit: fetchWithTimeout replaces `opts.signal` with its own
|
|
143
288
|
// controller (`{ ...opts, signal: controller.signal }`), so a signal passed
|
|
@@ -146,16 +291,16 @@ export class RemoteEmbedder {
|
|
|
146
291
|
method: "POST",
|
|
147
292
|
headers,
|
|
148
293
|
body: JSON.stringify(body),
|
|
149
|
-
},
|
|
294
|
+
}, timeoutMs, signal);
|
|
150
295
|
if (!response.ok) {
|
|
151
|
-
const errBody = await readBodyWithByteCap(response, undefined, { bodyTimeoutMs:
|
|
296
|
+
const errBody = await readBodyWithByteCap(response, undefined, { bodyTimeoutMs: timeoutMs, signal }).catch((err) => {
|
|
152
297
|
if (signal?.aborted)
|
|
153
298
|
throw err;
|
|
154
299
|
return "";
|
|
155
300
|
});
|
|
156
301
|
throw new Error(`Embedding request failed (${response.status}): ${this.safeErrorBody(errBody)}`);
|
|
157
302
|
}
|
|
158
|
-
const json = JSON.parse(await readBodyWithByteCap(response, undefined, { bodyTimeoutMs:
|
|
303
|
+
const json = JSON.parse(await readBodyWithByteCap(response, undefined, { bodyTimeoutMs: timeoutMs, signal }));
|
|
159
304
|
if (!json.data?.[0]?.embedding) {
|
|
160
305
|
throw new Error(`Unexpected embedding response format: missing data[0].embedding.${embeddingEndpointPathHint(this.endpoint)}`);
|
|
161
306
|
}
|
|
@@ -197,8 +342,37 @@ export class RemoteEmbedder {
|
|
|
197
342
|
* {@link isContextExceededResponse}) is split in half and retried
|
|
198
343
|
* recursively rather than skipped outright, down to individual documents; a
|
|
199
344
|
* single document that still fails this way becomes a genuine
|
|
200
|
-
* `context-window-exceeded` skip.
|
|
201
|
-
*
|
|
345
|
+
* `context-window-exceeded` skip.
|
|
346
|
+
*
|
|
347
|
+
* A request TIMEOUT (see {@link isEmbeddingTimeoutError}) never drops the
|
|
348
|
+
* batch outright (#954): the field evidence
|
|
349
|
+
* was that akm abandoning a timed-out request does not stop the server
|
|
350
|
+
* from still computing it, so immediately skipping (or immediately
|
|
351
|
+
* splitting, the prior behavior) just let the provider's queue grow
|
|
352
|
+
* while every following batch died the same way. Instead, on a timeout,
|
|
353
|
+
* this backs off ({@link embeddingTimeoutRetryBackoffMs}) and retries the
|
|
354
|
+
* SAME request once; a second timeout splits it in half (like a
|
|
355
|
+
* context-size rejection) and retries each half the same way, down to
|
|
356
|
+
* single documents — a single document that times out twice is finally
|
|
357
|
+
* skipped. Every other failure (network error, malformed response, a
|
|
358
|
+
* non-timeout HTTP failure) keeps the original skip-the-whole-batch-
|
|
359
|
+
* immediately behavior, at any size. The per-request timeout itself also
|
|
360
|
+
* scales down with the request's estimated size via
|
|
361
|
+
* {@link scaleEmbeddingTimeoutMs}, so a dead server is detected in seconds
|
|
362
|
+
* on a small batch rather than always waiting out the full configured
|
|
363
|
+
* `embedding.timeoutMs`.
|
|
364
|
+
*
|
|
365
|
+
* Run-scoped adaptive budget (#954, field report on beta.1): the FIRST
|
|
366
|
+
* context-size rejection of the run shrinks the effective request budget
|
|
367
|
+
* by {@link ADAPTIVE_BUDGET_SHRINK_FACTOR} (floored at
|
|
368
|
+
* {@link ADAPTIVE_BUDGET_FLOOR_MULTIPLIER} times `maxInputTokens`) for
|
|
369
|
+
* every batch not yet dispatched — the still-planned tail of `texts` is
|
|
370
|
+
* re-batched with `buildTokenBoundedBatches` at the smaller budget, and a
|
|
371
|
+
* `budget-lowered` `onBatch` event reports it once. This never touches the
|
|
372
|
+
* split-and-retry of the rejected batch itself (above), and never fires a
|
|
373
|
+
* second time in the same run even if a later batch is also rejected — a
|
|
374
|
+
* static configured budget that is simply too big for the endpoint should
|
|
375
|
+
* self-correct once, not ratchet down forever.
|
|
202
376
|
*/
|
|
203
377
|
async embedBatch(texts, signal, onSkip, onBatch) {
|
|
204
378
|
if (texts.length === 0)
|
|
@@ -206,9 +380,64 @@ export class RemoteEmbedder {
|
|
|
206
380
|
const results = new Array(texts.length).fill(undefined);
|
|
207
381
|
const headers = this.buildHeaders();
|
|
208
382
|
const ollamaOpts = resolveOllamaOptions(this.config);
|
|
209
|
-
|
|
383
|
+
// #956: `contextLength` is Ollama's `num_ctx` ONLY (see
|
|
384
|
+
// resolveOllamaOptions below) — it used to double as this client-side
|
|
385
|
+
// request budget too, so a config author setting it for one purpose
|
|
386
|
+
// silently changed the other. `maxTokens` is the sole knob for the
|
|
387
|
+
// request budget now; unset falls back to DEFAULT_TOKEN_BUDGET.
|
|
388
|
+
//
|
|
389
|
+
// `effectiveTokenBudget` (#954) starts at the configured/default value
|
|
390
|
+
// and MAY shrink once, on the run's first context-size rejection — see
|
|
391
|
+
// `maybeShrinkBudget` below. `textBatches` is mutated in place (spliced)
|
|
392
|
+
// by that shrink rather than reassigned, so the in-flight
|
|
393
|
+
// `concurrentMap` pool below (which reads this same array by reference)
|
|
394
|
+
// picks up the re-planned tail without restarting.
|
|
395
|
+
let effectiveTokenBudget = this.config.maxTokens ?? DEFAULT_TOKEN_BUDGET;
|
|
210
396
|
const maxCount = this.config.batchSize ?? DEFAULT_REMOTE_BATCH_SIZE;
|
|
211
|
-
const
|
|
397
|
+
const maxInputTokens = this.config.maxInputTokens ?? DEFAULT_MAX_INPUT_TOKENS;
|
|
398
|
+
const textBatches = buildTokenBoundedBatches(texts, effectiveTokenBudget, maxCount);
|
|
399
|
+
const configuredTimeoutMs = resolveEmbeddingTimeoutMs(this.config);
|
|
400
|
+
// How many of `textBatches` concurrentMap has already claimed (its own
|
|
401
|
+
// `nextIndex`, mirrored here so a budget shrink knows where the
|
|
402
|
+
// not-yet-dispatched tail begins). Assigned, not incremented, at the top
|
|
403
|
+
// of `runProviderBatch` — batches are claimed in strictly increasing
|
|
404
|
+
// order, so the highest `batchIndex` seen so far IS the claimed count.
|
|
405
|
+
let dispatchedBatchCount = 0;
|
|
406
|
+
// Set once the run's first context-size rejection has shrunk the budget
|
|
407
|
+
// (#954) — guards `maybeShrinkBudget` so it never fires twice.
|
|
408
|
+
let budgetShrunk = false;
|
|
409
|
+
// On the FIRST context-size rejection of this `embedBatch` call, shrink
|
|
410
|
+
// `effectiveTokenBudget` and re-plan every batch `concurrentMap` has not
|
|
411
|
+
// yet claimed from the smaller budget. Never touches `rejectedIndices`
|
|
412
|
+
// itself — the caller's own split-and-retry handles that batch — and is
|
|
413
|
+
// a no-op after the first call (`budgetShrunk`).
|
|
414
|
+
const maybeShrinkBudget = (rejectedIndices, rejectedBatchIndex, rejectedRequestTokens) => {
|
|
415
|
+
if (budgetShrunk)
|
|
416
|
+
return;
|
|
417
|
+
budgetShrunk = true;
|
|
418
|
+
const floor = ADAPTIVE_BUDGET_FLOOR_MULTIPLIER * maxInputTokens;
|
|
419
|
+
effectiveTokenBudget = Math.max(Math.round(effectiveTokenBudget * ADAPTIVE_BUDGET_SHRINK_FACTOR), floor);
|
|
420
|
+
const notYetDispatched = textBatches.slice(dispatchedBatchCount);
|
|
421
|
+
const remainingIndices = notYetDispatched.flatMap((batch) => batch.indices);
|
|
422
|
+
if (remainingIndices.length > 0) {
|
|
423
|
+
const remainingTexts = remainingIndices.map((i) => texts[i]);
|
|
424
|
+
const replanned = buildTokenBoundedBatches(remainingTexts, effectiveTokenBudget, maxCount).map((batch) => ({
|
|
425
|
+
indices: batch.indices.map((localIndex) => remainingIndices[localIndex]),
|
|
426
|
+
oversized: batch.oversized,
|
|
427
|
+
}));
|
|
428
|
+
textBatches.splice(dispatchedBatchCount, textBatches.length - dispatchedBatchCount, ...replanned);
|
|
429
|
+
}
|
|
430
|
+
warnVerbose(`[embed] provider rejected a ${rejectedRequestTokens}-token request as over its context; request budget lowered to ${effectiveTokenBudget.toLocaleString()} for the rest of this run`);
|
|
431
|
+
commitBatch(rejectedIndices, rejectedIndices.map(() => undefined), undefined, {
|
|
432
|
+
batchIndex: rejectedBatchIndex,
|
|
433
|
+
batchCount: textBatches.length,
|
|
434
|
+
docCount: rejectedIndices.length,
|
|
435
|
+
requestTokens: rejectedRequestTokens,
|
|
436
|
+
elapsedMs: 0,
|
|
437
|
+
outcome: "budget-lowered",
|
|
438
|
+
reason: `provider rejected ${rejectedRequestTokens.toLocaleString()} tokens as over its context; request budget lowered to ${effectiveTokenBudget.toLocaleString()} for the rest of this run`,
|
|
439
|
+
});
|
|
440
|
+
};
|
|
212
441
|
// Stops the pool from claiming any FURTHER provider batch once the
|
|
213
442
|
// caller's onBatch has failed once (the materializer's transaction
|
|
214
443
|
// failed, so a subsequent commit would just fail again) — dispatching
|
|
@@ -242,7 +471,7 @@ export class RemoteEmbedder {
|
|
|
242
471
|
// fabricated "batch-request-failed" skip (#954). Checked (and rethrown)
|
|
243
472
|
// once the pool drains, the same way `signal?.aborted` is today.
|
|
244
473
|
let firstOnBatchError;
|
|
245
|
-
const commitBatch = (indices, embeddings, model) => {
|
|
474
|
+
const commitBatch = (indices, embeddings, model, outcome) => {
|
|
246
475
|
if (!onBatch)
|
|
247
476
|
return;
|
|
248
477
|
// Once persistence has failed once, an already in-flight batch that
|
|
@@ -252,7 +481,7 @@ export class RemoteEmbedder {
|
|
|
252
481
|
if (firstOnBatchError !== undefined)
|
|
253
482
|
return;
|
|
254
483
|
try {
|
|
255
|
-
onBatch(indices, embeddings, model);
|
|
484
|
+
onBatch(indices, embeddings, model, outcome);
|
|
256
485
|
}
|
|
257
486
|
catch (err) {
|
|
258
487
|
firstOnBatchError = err;
|
|
@@ -260,65 +489,207 @@ export class RemoteEmbedder {
|
|
|
260
489
|
}
|
|
261
490
|
};
|
|
262
491
|
// Requests a single provider batch (by index list), recursing on a
|
|
263
|
-
// context-size rejection
|
|
264
|
-
//
|
|
265
|
-
//
|
|
266
|
-
//
|
|
267
|
-
//
|
|
268
|
-
|
|
492
|
+
// context-size rejection OR a repeated timeout (#954). Never
|
|
493
|
+
// throws except to propagate a genuine caller abort — every other
|
|
494
|
+
// outcome (success or a non-abort failure) resolves normally after
|
|
495
|
+
// reporting via onSkip/onBatch. `onBatch` fires only once this
|
|
496
|
+
// try/catch has already settled success vs. failure, so a throw from it
|
|
497
|
+
// is never caught and reclassified by this block.
|
|
498
|
+
//
|
|
499
|
+
// `isTimeoutRetry` marks the SECOND attempt at this exact `indices`
|
|
500
|
+
// (after the one same-size backoff-and-retry) — a second
|
|
501
|
+
// timeout at that point splits or terminally skips rather than backing
|
|
502
|
+
// off again.
|
|
503
|
+
//
|
|
504
|
+
// `timeoutAttempt` (#954, field-report follow-up) counts how many splits
|
|
505
|
+
// down the timeout-retry chain this call is: 0 at the top level, then
|
|
506
|
+
// +1 each time a second timeout at one size splits into two smaller
|
|
507
|
+
// requests. It is the `attempt` fed to {@link embeddingTimeoutRetryBackoffMs}
|
|
508
|
+
// so each successive size's first-timeout backoff is longer than the
|
|
509
|
+
// last (5s, doubling, capped at 60s) instead of every split restarting
|
|
510
|
+
// at the same ~5s delay — a server still draining a whole run of
|
|
511
|
+
// abandoned requests needs more room the deeper the chain goes, not the
|
|
512
|
+
// same fixed pause every time.
|
|
513
|
+
const requestAndCommit = async (indices, batchIndex, isTimeoutRetry = false, timeoutAttempt = 0) => {
|
|
514
|
+
// A concurrent batch may have tripped the circuit breaker (or failed
|
|
515
|
+
// onBatch) while this call was queued behind a split or a backoff —
|
|
516
|
+
// never let a deeper recursive call make a request that can no longer
|
|
517
|
+
// be reported, mirroring how the pool below never claims a
|
|
518
|
+
// not-yet-started textBatch once dispatch has stopped.
|
|
519
|
+
if (dispatchAbort.signal.aborted)
|
|
520
|
+
return;
|
|
269
521
|
const batch = indices.map((i) => texts[i]);
|
|
522
|
+
const requestTokens = batch.reduce((sum, text) => sum + estimateTokenCount(text), 0);
|
|
523
|
+
const requestTimeoutMs = scaleEmbeddingTimeoutMs(configuredTimeoutMs, requestTokens, effectiveTokenBudget);
|
|
524
|
+
const requestStart = Date.now();
|
|
270
525
|
let batchEmbeddings;
|
|
271
526
|
let responseModel;
|
|
527
|
+
let outcome;
|
|
528
|
+
let failureReason;
|
|
272
529
|
try {
|
|
273
|
-
const { vectors, model } = await this.requestBatch(batch, headers, ollamaOpts, signal);
|
|
530
|
+
const { vectors, model } = await this.requestBatch(batch, headers, ollamaOpts, requestTimeoutMs, signal);
|
|
274
531
|
for (let k = 0; k < indices.length; k++) {
|
|
275
532
|
results[indices[k]] = vectors[k];
|
|
276
533
|
}
|
|
277
534
|
batchEmbeddings = indices.map((i) => results[i]);
|
|
278
535
|
responseModel = model;
|
|
536
|
+
outcome = "stored";
|
|
279
537
|
}
|
|
280
538
|
catch (err) {
|
|
281
539
|
// A caller abort must still propagate — it is not a "this batch
|
|
282
540
|
// failed" condition, it means stop entirely.
|
|
283
541
|
if (signal?.aborted)
|
|
284
542
|
throw err;
|
|
543
|
+
if (err instanceof ContextExceededError) {
|
|
544
|
+
// #954: the run's FIRST context-size rejection (any size) shrinks
|
|
545
|
+
// the budget for everything not yet dispatched; a no-op after the
|
|
546
|
+
// first call. Deliberately BEFORE the split below — it must fire
|
|
547
|
+
// for a single-document rejection too (which never reaches the
|
|
548
|
+
// `indices.length > 1` split branch), and it never touches this
|
|
549
|
+
// batch's own split-and-retry.
|
|
550
|
+
maybeShrinkBudget(indices, batchIndex, requestTokens);
|
|
551
|
+
}
|
|
285
552
|
if (err instanceof ContextExceededError && indices.length > 1) {
|
|
286
553
|
const mid = Math.ceil(indices.length / 2);
|
|
287
|
-
await requestAndCommit(indices.slice(0, mid));
|
|
288
|
-
await requestAndCommit(indices.slice(mid));
|
|
554
|
+
await requestAndCommit(indices.slice(0, mid), batchIndex, false, timeoutAttempt);
|
|
555
|
+
await requestAndCommit(indices.slice(mid), batchIndex, false, timeoutAttempt);
|
|
556
|
+
return;
|
|
557
|
+
}
|
|
558
|
+
const timedOut = isEmbeddingTimeoutError(err);
|
|
559
|
+
if (timedOut && !isTimeoutRetry) {
|
|
560
|
+
// First timeout at this size: back off so the provider can drain
|
|
561
|
+
// the abandoned request, then retry the SAME request once before
|
|
562
|
+
// ever splitting or skipping (#954). The backoff grows with
|
|
563
|
+
// `timeoutAttempt`, not a flat ~5s every time, so a chain of
|
|
564
|
+
// splits down to smaller and smaller requests gives the provider
|
|
565
|
+
// proportionally more room to drain each time.
|
|
566
|
+
const backoffMs = embeddingTimeoutRetryBackoffMs(timeoutAttempt);
|
|
567
|
+
warnVerbose(`[embed] batch of ${batch.length} document(s) timed out after ${requestTimeoutMs}ms; retrying once after a ${Math.round(backoffMs)}ms backoff`);
|
|
568
|
+
// Default-level notice (#954 field-report follow-up), not just the
|
|
569
|
+
// verbose line above — a run silently waiting out a multi-minute
|
|
570
|
+
// back-off looked identical to a hang otherwise. Nothing has
|
|
571
|
+
// failed or succeeded yet, so there is nothing to persist:
|
|
572
|
+
// `embeddings` are all `undefined` and the materializer's onBatch
|
|
573
|
+
// must not touch storage for this event.
|
|
574
|
+
commitBatch(indices, indices.map(() => undefined), undefined, {
|
|
575
|
+
batchIndex,
|
|
576
|
+
batchCount: textBatches.length,
|
|
577
|
+
docCount: indices.length,
|
|
578
|
+
requestTokens,
|
|
579
|
+
elapsedMs: backoffMs,
|
|
580
|
+
outcome: "retrying",
|
|
581
|
+
reason: "timed out",
|
|
582
|
+
});
|
|
583
|
+
await abortableDelay(backoffMs, signal, "embedding interrupted during retry backoff");
|
|
584
|
+
if (!dispatchAbort.signal.aborted) {
|
|
585
|
+
return requestAndCommit(indices, batchIndex, true, timeoutAttempt);
|
|
586
|
+
}
|
|
587
|
+
// Dispatch was stopped (by another batch's circuit-breaker trip)
|
|
588
|
+
// while this one was backing off — fall through and skip below
|
|
589
|
+
// instead of issuing a request that can no longer be reported.
|
|
590
|
+
}
|
|
591
|
+
else if (timedOut && indices.length > 1) {
|
|
592
|
+
// Timed out again on the retry: split rather than skip outright —
|
|
593
|
+
// the provider may still fit it once it is smaller (#954),
|
|
594
|
+
// the same treatment a context-size rejection gets.
|
|
595
|
+
// Each half's OWN first-timeout backoff (#954 follow-up) starts
|
|
596
|
+
// one attempt further down the chain than this size's did.
|
|
597
|
+
const mid = Math.ceil(indices.length / 2);
|
|
598
|
+
await requestAndCommit(indices.slice(0, mid), batchIndex, false, timeoutAttempt + 1);
|
|
599
|
+
await requestAndCommit(indices.slice(mid), batchIndex, false, timeoutAttempt + 1);
|
|
289
600
|
return;
|
|
290
601
|
}
|
|
291
602
|
const message = err instanceof Error ? err.message : String(err);
|
|
292
603
|
const reason = err instanceof ContextExceededError ? "context-window-exceeded" : "batch-request-failed";
|
|
293
|
-
|
|
294
|
-
for
|
|
295
|
-
|
|
604
|
+
const failureKind = reason === "batch-request-failed" ? (timedOut ? "timeout" : "network-error") : undefined;
|
|
605
|
+
// Default-level visibility for a failed batch (not verbose-only) is
|
|
606
|
+
// still guaranteed here — just not via warn(). The `commitBatch` call
|
|
607
|
+
// below carries `outcome: "failed"` and this `message` as `reason`
|
|
608
|
+
// through `onBatch`, and materialize-embeddings.ts's per-batch line
|
|
609
|
+
// (also default-level) prints it from there. A warn() call here used
|
|
610
|
+
// to print the identical event a second time on stderr — the same
|
|
611
|
+
// class of double-print bug fixed for the truncation/re-embed-reason
|
|
612
|
+
// lines in materialize-embeddings.ts (#954, field-report follow-up).
|
|
613
|
+
// Per-entry batch-mapping detail stays verbose-only
|
|
614
|
+
// (materialize-embeddings.ts).
|
|
615
|
+
let stopRequested = false;
|
|
616
|
+
for (const [k, idx] of indices.entries()) {
|
|
617
|
+
if (onSkip?.({
|
|
618
|
+
index: idx,
|
|
619
|
+
reason,
|
|
620
|
+
message,
|
|
621
|
+
batchStart: k === 0,
|
|
622
|
+
batchSize: indices.length,
|
|
623
|
+
failureKind,
|
|
624
|
+
}) === false)
|
|
625
|
+
stopRequested = true;
|
|
296
626
|
}
|
|
627
|
+
// #954: the caller's circuit breaker asked to stop —
|
|
628
|
+
// gate further dispatch through the SAME dispatchAbort controller
|
|
629
|
+
// the onBatch-throw path above uses, but resolve this call normally
|
|
630
|
+
// (never reject) with whatever results already landed, since this is
|
|
631
|
+
// a policy decision, not a persistence failure.
|
|
632
|
+
if (stopRequested)
|
|
633
|
+
stopDispatch();
|
|
297
634
|
batchEmbeddings = indices.map(() => undefined);
|
|
635
|
+
outcome = "failed";
|
|
636
|
+
failureReason = message;
|
|
298
637
|
}
|
|
299
|
-
commitBatch(indices, batchEmbeddings, responseModel
|
|
638
|
+
commitBatch(indices, batchEmbeddings, responseModel, {
|
|
639
|
+
batchIndex,
|
|
640
|
+
batchCount: textBatches.length,
|
|
641
|
+
docCount: indices.length,
|
|
642
|
+
requestTokens,
|
|
643
|
+
elapsedMs: Date.now() - requestStart,
|
|
644
|
+
outcome,
|
|
645
|
+
reason: failureReason,
|
|
646
|
+
});
|
|
300
647
|
};
|
|
301
|
-
const runProviderBatch = async (textBatch) => {
|
|
648
|
+
const runProviderBatch = async (textBatch, batchIndex) => {
|
|
649
|
+
// Claimed in strictly increasing order by `concurrentMap` below, so
|
|
650
|
+
// the highest `batchIndex` seen so far is exactly how many batches it
|
|
651
|
+
// has claimed (#954) — see `maybeShrinkBudget`'s doc comment above.
|
|
652
|
+
// Assigned synchronously at entry, before any `await`, so this always
|
|
653
|
+
// matches `concurrentMap`'s own `nextIndex` at the moment of claim.
|
|
654
|
+
dispatchedBatchCount = batchIndex;
|
|
302
655
|
if (textBatch.oversized) {
|
|
303
656
|
const idx = textBatch.indices[0];
|
|
304
657
|
const estTokens = estimateTokenCount(texts[idx]);
|
|
305
658
|
onSkip?.({
|
|
306
659
|
index: idx,
|
|
307
660
|
reason: "context-window-exceeded",
|
|
308
|
-
message: `Document estimated at ${estTokens} tokens exceeds the ${
|
|
661
|
+
message: `Document estimated at ${estTokens} tokens exceeds the ${effectiveTokenBudget}-token embedding budget; skipped.`,
|
|
662
|
+
batchStart: true,
|
|
663
|
+
batchSize: 1,
|
|
664
|
+
});
|
|
665
|
+
// Never made a request — excluded from the default-level per-batch
|
|
666
|
+
// line (there is no request outcome to report), but still counted
|
|
667
|
+
// in the run's oversized-skip total via `onSkip` above.
|
|
668
|
+
commitBatch([idx], [undefined], undefined, {
|
|
669
|
+
batchIndex,
|
|
670
|
+
batchCount: textBatches.length,
|
|
671
|
+
docCount: 1,
|
|
672
|
+
requestTokens: estTokens,
|
|
673
|
+
elapsedMs: 0,
|
|
674
|
+
outcome: "failed",
|
|
675
|
+
reason: "oversized",
|
|
309
676
|
});
|
|
310
|
-
commitBatch([idx], [undefined]);
|
|
311
677
|
return;
|
|
312
678
|
}
|
|
313
|
-
await requestAndCommit(textBatch.indices);
|
|
679
|
+
await requestAndCommit(textBatch.indices, batchIndex);
|
|
314
680
|
};
|
|
315
681
|
const concurrency = resolveEmbeddingConcurrency(this.config);
|
|
316
682
|
// concurrentMap swallows a thrown fn (per-item, results discarded here —
|
|
317
683
|
// requestAndCommit only throws to signal a caller abort), so abort must
|
|
318
684
|
// be re-checked once the pool has drained rather than relying on the
|
|
319
685
|
// throw itself to escape. Dispatch is gated on `dispatchAbort`, not the
|
|
320
|
-
// caller's `signal` directly — see the comment above.
|
|
321
|
-
|
|
686
|
+
// caller's `signal` directly — see the comment above. `batchIndex` is
|
|
687
|
+
// 1-based (matches the human-readable "batch N/Total" line) and comes
|
|
688
|
+
// from `concurrentMap`'s own 0-based item index, not a separately
|
|
689
|
+
// tracked counter, so it stays correct under concurrency.
|
|
690
|
+
await concurrentMap(textBatches, (textBatch, i) => runProviderBatch(textBatch, i + 1), concurrency, {
|
|
691
|
+
signal: dispatchAbort.signal,
|
|
692
|
+
});
|
|
322
693
|
if (callerAbortListener)
|
|
323
694
|
signal?.removeEventListener("abort", callerAbortListener);
|
|
324
695
|
if (signal?.aborted) {
|
|
@@ -335,8 +706,14 @@ export class RemoteEmbedder {
|
|
|
335
706
|
* used by the embedding-fingerprint canary to verify a config-string
|
|
336
707
|
* rename against what the endpoint actually served, not just re-assert the
|
|
337
708
|
* configured string. Throws on any failure.
|
|
709
|
+
*
|
|
710
|
+
* `timeoutMs` is the caller's ALREADY-SCALED per-request timeout (#954
|
|
711
|
+
* — see {@link scaleEmbeddingTimeoutMs}), not re-resolved here:
|
|
712
|
+
* `embedBatch` computes it per request from that request's own size so a
|
|
713
|
+
* split-down retry gets a smaller, size-appropriate timeout rather than
|
|
714
|
+
* always the full configured `embedding.timeoutMs`.
|
|
338
715
|
*/
|
|
339
|
-
async requestBatch(batch, headers, ollamaOpts, signal) {
|
|
716
|
+
async requestBatch(batch, headers, ollamaOpts, timeoutMs, signal) {
|
|
340
717
|
const body = {
|
|
341
718
|
input: batch,
|
|
342
719
|
model: this.model,
|
|
@@ -353,9 +730,9 @@ export class RemoteEmbedder {
|
|
|
353
730
|
method: "POST",
|
|
354
731
|
headers,
|
|
355
732
|
body: JSON.stringify(body),
|
|
356
|
-
},
|
|
733
|
+
}, timeoutMs, signal);
|
|
357
734
|
if (!response.ok) {
|
|
358
|
-
const respBody = await readBodyWithByteCap(response, undefined, { bodyTimeoutMs:
|
|
735
|
+
const respBody = await readBodyWithByteCap(response, undefined, { bodyTimeoutMs: timeoutMs, signal }).catch((err) => {
|
|
359
736
|
if (signal?.aborted)
|
|
360
737
|
throw err;
|
|
361
738
|
return "";
|
|
@@ -366,7 +743,7 @@ export class RemoteEmbedder {
|
|
|
366
743
|
}
|
|
367
744
|
throw new Error(message);
|
|
368
745
|
}
|
|
369
|
-
const json = JSON.parse(await readBodyWithByteCap(response, undefined, { bodyTimeoutMs:
|
|
746
|
+
const json = JSON.parse(await readBodyWithByteCap(response, undefined, { bodyTimeoutMs: timeoutMs, signal }));
|
|
370
747
|
if (!json.data || json.data.length !== batch.length) {
|
|
371
748
|
throw new Error(`Unexpected embedding batch response: expected ${batch.length} embeddings, got ${json.data?.length ?? 0}.${embeddingEndpointPathHint(this.endpoint)}`);
|
|
372
749
|
}
|
|
@@ -471,3 +848,22 @@ function resolveOllamaOptions(config) {
|
|
|
471
848
|
export function hasRemoteEndpoint(config) {
|
|
472
849
|
return isHttpUrl(config.endpoint);
|
|
473
850
|
}
|
|
851
|
+
/**
|
|
852
|
+
* Describe WHERE an `embedding.apiKey` came from, never its value — the
|
|
853
|
+
* actionable outcome of the #953 field gap: `resolveSecret` throws on an
|
|
854
|
+
* unresolvable `secret://` reference, so a keyless request can only mean
|
|
855
|
+
* `embedding.apiKey` was absent from the config the run actually loaded (a
|
|
856
|
+
* different config root, scope, or a config edited after the run started).
|
|
857
|
+
* A default-level progress line naming the credential's SOURCE (this
|
|
858
|
+
* helper), printed once before the first provider request, lets a field run
|
|
859
|
+
* self-diagnose that without ever surfacing the secret itself.
|
|
860
|
+
*/
|
|
861
|
+
export function describeEmbeddingCredential(apiKey) {
|
|
862
|
+
if (!apiKey)
|
|
863
|
+
return "none configured";
|
|
864
|
+
if (SECRET_STORE_REFERENCE_PATTERN.test(apiKey))
|
|
865
|
+
return `${apiKey} (store)`;
|
|
866
|
+
if (ENV_REFERENCE_PATTERN.test(apiKey))
|
|
867
|
+
return `${apiKey} (env)`;
|
|
868
|
+
return "literal apiKey";
|
|
869
|
+
}
|