akm-cli 0.9.15-beta.1 → 0.9.15-beta.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -7,9 +7,10 @@
7
7
  * Calls the configured `/embeddings` endpoint and L2-normalizes the returned
8
8
  * vectors so the scoring pipeline's L2-to-cosine conversion is correct.
9
9
  */
10
- import { fetchWithTimeout, isHttpUrl, readBodyWithByteCap } from "../../core/common.js";
10
+ import { abortableDelay, backoffDelay, fetchWithTimeout, isHttpUrl, readBodyWithByteCap } from "../../core/common.js";
11
11
  import { concurrentMap } from "../../core/concurrent.js";
12
12
  import { resolveSecret } from "../../core/config/config.js";
13
+ import { ENV_REFERENCE_PATTERN, SECRET_STORE_REFERENCE_PATTERN } from "../../core/config/schema/primitives.js";
13
14
  import { defaultConcurrencyForEndpoint } from "../../core/loopback.js";
14
15
  import { redactErrorBody, redactSensitiveText } from "../../core/redaction.js";
15
16
  import { warnVerbose } from "../../core/warn.js";
@@ -24,18 +25,130 @@ import { resolveSecretFromStore } from "../../sources/snapshot-fetchers/secret-s
24
25
  export const DEFAULT_REMOTE_BATCH_SIZE = 100;
25
26
  /**
26
27
  * Conservative default token budget per HTTP request when the config gives
27
- * no better number (`maxTokens` or `contextLength`). #874's measurements:
28
+ * no better number (`maxTokens` — see #956 for why `contextLength`
29
+ * no longer feeds this). #874's measurements:
28
30
  * a batch of 100 small docs (~400 KB, ~100K tokens) took 14.8s against a
29
31
  * healthy local endpoint — half the 30s request timeout — and a single
30
32
  * 128 KB (~24K token) document alone was rejected by the endpoint as
31
- * exceeding its context size. 8000 tokens keeps a batch's estimated size
32
- * comfortably under both the timeout and common local-model context windows.
33
+ * exceeding its context size.
34
+ *
35
+ * Lowered from 8000 to 6000 (#954, field report on beta.1): the 4-chars-
36
+ * per-token estimator undercounts dense technical text by 7-55%, so 8000
37
+ * against an 8192-token llama.cpp embedder regularly landed real requests
38
+ * over the endpoint's context window. 6000 is the value the field confirmed
39
+ * stops that steady trickle of rejections; `embedBatch`'s run-scoped
40
+ * adaptive budget below still shrinks further, for an endpoint where even
41
+ * this is not enough.
33
42
  */
34
- export const DEFAULT_TOKEN_BUDGET = 8000;
43
+ export const DEFAULT_TOKEN_BUDGET = 6000;
35
44
  /** Cheap token estimator: 4 chars ≈ 1 token. Used in verbose logging and error messages. */
36
45
  export function estimateTokenCount(text) {
37
46
  return Math.round(text.length / 4);
38
47
  }
48
+ /**
49
+ * Default per-document embedding cap (`embedding.maxInputTokens`, #956)
50
+ * — the materializer truncates a document's embedded text to
51
+ * this cap (head only) instead of skipping it outright, so one oversized
52
+ * entry can no longer fail a whole batch. Fragments are not embedded at all
53
+ * (only the entry's own search text is), so this is the only lever on how
54
+ * much of a large document contributes to its vector.
55
+ */
56
+ export const DEFAULT_MAX_INPUT_TOKENS = 512;
57
+ /**
58
+ * Truncate `text` to at most `maxTokens` (estimated via
59
+ * {@link estimateTokenCount}, the same 4-chars≈1-token rule the batching
60
+ * budget uses), keeping only its head. The cut never splits a UTF-16
61
+ * surrogate pair. Text already at or under the cap is returned unchanged
62
+ * (`truncated: false`) — including empty text, which is never itself
63
+ * "truncated".
64
+ */
65
+ export function capEmbeddingText(text, maxTokens) {
66
+ if (estimateTokenCount(text) <= maxTokens)
67
+ return { text, truncated: false };
68
+ const charBudget = Math.max(0, maxTokens * 4);
69
+ let cut = Math.min(charBudget, text.length);
70
+ if (cut > 0 && cut < text.length) {
71
+ const code = text.charCodeAt(cut);
72
+ // A low surrogate (0xDC00-0xDFFF) at the cut point means its high
73
+ // surrogate is the character just before it — back off one position so
74
+ // the pair stays together rather than yielding a lone surrogate.
75
+ if (code >= 0xdc00 && code <= 0xdfff)
76
+ cut -= 1;
77
+ }
78
+ return { text: text.slice(0, cut), truncated: true };
79
+ }
80
+ /**
81
+ * Default per-request timeout when `embedding.timeoutMs` is unset (#954).
82
+ * The prior fixed 30s cut off exactly the field-report case: a
83
+ * local model server on a full-budget (`DEFAULT_TOKEN_BUDGET`) batch
84
+ * legitimately takes longer than that, and the timeout fired mid-response
85
+ * with no retry — every batch it hit was silently dropped for the rest of
86
+ * an hours-long run. 120s comfortably covers a slow local batch while still
87
+ * bounding a genuinely dead endpoint to a few minutes, not forever.
88
+ */
89
+ export const DEFAULT_EMBEDDING_TIMEOUT_MS = 120_000;
90
+ /** Resolve the effective per-request timeout: `embedding.timeoutMs` when set, else the default above. */
91
+ export function resolveEmbeddingTimeoutMs(config) {
92
+ return config.timeoutMs ?? DEFAULT_EMBEDDING_TIMEOUT_MS;
93
+ }
94
+ /**
95
+ * Scale the per-request timeout down for a smaller-than-budget request
96
+ * (#954, field-report follow-up): `embedding.timeoutMs` /
97
+ * {@link resolveEmbeddingTimeoutMs} is the budget for a request at the FULL
98
+ * token budget; a batch using only a fraction of it gets a proportionally
99
+ * smaller timeout, floored at 30s and never above the configured
100
+ * `timeoutMs` itself, so a dead server is detected in seconds on the common
101
+ * case of small documents instead of always waiting out the full configured
102
+ * budget.
103
+ */
104
+ export function scaleEmbeddingTimeoutMs(timeoutMs, requestTokens, tokenBudget) {
105
+ const scaled = tokenBudget > 0 ? timeoutMs * (requestTokens / tokenBudget) : timeoutMs;
106
+ return Math.min(Math.max(scaled, 30_000), timeoutMs);
107
+ }
108
+ /**
109
+ * True when `err` is a request- or body-read timeout (#954) —
110
+ * `fetchWithTimeout`'s connection/header timeout ("Request timed out
111
+ * after...") or `readBodyWithByteCap`'s body-phase `BodyReadTimeoutError`.
112
+ * Only this failure mode gets the back-off-and-retry treatment:
113
+ * the field evidence was specifically that a timed-out request keeps
114
+ * computing server-side, so abandoning it immediately (the prior
115
+ * behavior) just grows the provider's queue further. A genuine network/HTTP
116
+ * failure (connection refused, malformed response, a real error response)
117
+ * has no such still-in-flight hazard and keeps the original
118
+ * skip-immediately behavior.
119
+ */
120
+ export function isEmbeddingTimeoutError(err) {
121
+ if (!(err instanceof Error))
122
+ return false;
123
+ if (err.name === "BodyReadTimeoutError")
124
+ return true;
125
+ return err.message.startsWith("Request timed out after ");
126
+ }
127
+ /** TEST-ONLY seam: override the backoff base/max so retry-backoff tests run fast without waiting real seconds. */
128
+ let embeddingTimeoutBackoffOverrideForTests;
129
+ /** TEST-ONLY. Pass undefined to restore the real 5s/60s backoff. */
130
+ export function _setEmbeddingTimeoutBackoffForTests(config) {
131
+ embeddingTimeoutBackoffOverrideForTests = config;
132
+ }
133
+ /**
134
+ * Backoff before the single same-size retry on a request timeout
135
+ * (#954) — reuses the same jittered
136
+ * exponential formula {@link backoffDelay} uses for the rest of the
137
+ * codebase's retry paths, at a base of "5s, doubling, capped at 60s".
138
+ *
139
+ * `timeoutAttempt` (#954, field-report follow-up) is how many times the
140
+ * SAME-SIZE-retry-then-split chain has already split before reaching this
141
+ * size — 0 at the top level. The first-timeout backoff for each successive,
142
+ * smaller size after a split grows with it (5s, then doubling, capped at
143
+ * 60s) instead of every split resetting to a flat ~5s: the field evidence
144
+ * was that an abandoned request keeps computing server-side, so a server
145
+ * already draining a whole chain of abandoned requests needs progressively
146
+ * more room, not the same fixed pause at every size.
147
+ */
148
+ export function embeddingTimeoutRetryBackoffMs(timeoutAttempt = 0) {
149
+ const { baseMs, maxMs } = embeddingTimeoutBackoffOverrideForTests ?? { baseMs: 5_000, maxMs: 60_000 };
150
+ return backoffDelay(timeoutAttempt, baseMs, maxMs);
151
+ }
39
152
  /**
40
153
  * Distinguishes a batch rejected because it exceeded the endpoint's context
41
154
  * window from every other failure mode (network error, 5xx, malformed
@@ -50,8 +163,15 @@ export class ContextExceededError extends Error {
50
163
  this.name = "ContextExceededError";
51
164
  }
52
165
  }
53
- /** Patterns providers use to report a request too large for the model's context window. */
54
- const CONTEXT_EXCEEDED_PATTERN = /exceed_context_size_error|context size|context length|too many tokens/i;
166
+ /**
167
+ * Patterns providers use to report a request too large for the model's
168
+ * context window. `input is too large to process`/`physical batch size`/
169
+ * `ubatch` (#954) cover llama.cpp's own physical-batch
170
+ * rejection (HTTP 500, e.g. "input is too large to process. increase the
171
+ * physical batch size"), which was previously an unrecognized generic
172
+ * failure — the whole batch was dropped instead of split and retried.
173
+ */
174
+ const CONTEXT_EXCEEDED_PATTERN = /exceed_context_size_error|context size|context length|too many tokens|input is too large to process|physical batch size|ubatch/i;
55
175
  /**
56
176
  * True when an HTTP failure means "this request's input is too large for the
57
177
  * endpoint's context window" rather than some other failure. HTTP 413
@@ -65,16 +185,25 @@ export function isContextExceededResponse(status, body) {
65
185
  }
66
186
  /**
67
187
  * Resolve the effective in-flight request window for `RemoteEmbedder.embedBatch`.
68
- * FIXED — no config override: 1 for a loopback endpoint, 2 for a remote one,
69
- * via the shared `defaultConcurrencyForEndpoint` (`src/core/loopback.ts`),
70
- * the same lowest-common-denominator rule `getDefaultLlmConcurrency`
71
- * (`src/indexer/indexer.ts`) uses. The actual throughput knob is request
72
- * SIZE, not request count: `embedding.batchSize` (document cap) and
73
- * `embedding.maxTokens`/`contextLength` (token budget) reach a larger batch
74
- * per request, which is where most of the win is — a 32-input batch takes
75
- * about the same wall time as one input against a healthy endpoint.
188
+ * Default (unset `embedding.concurrency`): 1 for a loopback endpoint, 2 for a
189
+ * remote one, via the shared `defaultConcurrencyForEndpoint`
190
+ * (`src/core/loopback.ts`), the same lowest-common-denominator rule
191
+ * `getDefaultLlmConcurrency` (`src/indexer/indexer.ts`) uses.
192
+ *
193
+ * `embedding.concurrency` (#954) overrides this default in
194
+ * either direction, bounded 1-16 at the config schema — added after field
195
+ * evidence that a multi-slot local server (llama.cpp `--parallel N`, vLLM)
196
+ * genuinely serves parallel requests and was left idle by the fixed default.
197
+ * Request SIZE remains the first throughput lever regardless:
198
+ * `embedding.batchSize` (document cap) and `embedding.maxTokens` (request
199
+ * token budget — see #956; `contextLength` no longer feeds it)
200
+ * reach a larger batch per request, which is where most of the win is for a
201
+ * single-slot server — a 32-input batch takes about the same wall time as
202
+ * one input against a healthy endpoint.
76
203
  */
77
204
  export function resolveEmbeddingConcurrency(config) {
205
+ if (typeof config.concurrency === "number")
206
+ return config.concurrency;
78
207
  return defaultConcurrencyForEndpoint(config.endpoint);
79
208
  }
80
209
  /**
@@ -113,6 +242,21 @@ export function buildTokenBoundedBatches(texts, tokenBudget, maxCount) {
113
242
  flush();
114
243
  return batches;
115
244
  }
245
+ /**
246
+ * Shrink factor applied to the effective request budget on the first
247
+ * context-size rejection of an `embedBatch` run (#954, field report on
248
+ * beta.1): one 25% cut absorbs the estimator's measured undercount without
249
+ * repeatedly re-shrinking mid-run — see the "shrink at most once" rule on
250
+ * {@link RemoteEmbedder.embedBatch}.
251
+ */
252
+ const ADAPTIVE_BUDGET_SHRINK_FACTOR = 0.75;
253
+ /**
254
+ * Floor on the adaptive-budget shrink above, as a multiple of
255
+ * `maxInputTokens` (#954): a request budget below twice the per-document cap
256
+ * could no longer batch more than one document per request, defeating the
257
+ * point of batching at all.
258
+ */
259
+ const ADAPTIVE_BUDGET_FLOOR_MULTIPLIER = 2;
116
260
  export class RemoteEmbedder {
117
261
  config;
118
262
  endpoint;
@@ -138,6 +282,7 @@ export class RemoteEmbedder {
138
282
  if (ollamaOpts) {
139
283
  body.options = ollamaOpts;
140
284
  }
285
+ const timeoutMs = resolveEmbeddingTimeoutMs(this.config);
141
286
  // `signal` MUST go through fetchWithTimeout's dedicated 4th parameter, not
142
287
  // the RequestInit: fetchWithTimeout replaces `opts.signal` with its own
143
288
  // controller (`{ ...opts, signal: controller.signal }`), so a signal passed
@@ -146,16 +291,16 @@ export class RemoteEmbedder {
146
291
  method: "POST",
147
292
  headers,
148
293
  body: JSON.stringify(body),
149
- }, 30_000, signal);
294
+ }, timeoutMs, signal);
150
295
  if (!response.ok) {
151
- const errBody = await readBodyWithByteCap(response, undefined, { bodyTimeoutMs: 30_000, signal }).catch((err) => {
296
+ const errBody = await readBodyWithByteCap(response, undefined, { bodyTimeoutMs: timeoutMs, signal }).catch((err) => {
152
297
  if (signal?.aborted)
153
298
  throw err;
154
299
  return "";
155
300
  });
156
301
  throw new Error(`Embedding request failed (${response.status}): ${this.safeErrorBody(errBody)}`);
157
302
  }
158
- const json = JSON.parse(await readBodyWithByteCap(response, undefined, { bodyTimeoutMs: 30_000, signal }));
303
+ const json = JSON.parse(await readBodyWithByteCap(response, undefined, { bodyTimeoutMs: timeoutMs, signal }));
159
304
  if (!json.data?.[0]?.embedding) {
160
305
  throw new Error(`Unexpected embedding response format: missing data[0].embedding.${embeddingEndpointPathHint(this.endpoint)}`);
161
306
  }
@@ -197,8 +342,37 @@ export class RemoteEmbedder {
197
342
  * {@link isContextExceededResponse}) is split in half and retried
198
343
  * recursively rather than skipped outright, down to individual documents; a
199
344
  * single document that still fails this way becomes a genuine
200
- * `context-window-exceeded` skip. Every other failure (network error, 5xx,
201
- * malformed response) keeps the original skip-the-whole-batch behavior.
345
+ * `context-window-exceeded` skip.
346
+ *
347
+ * A request TIMEOUT (see {@link isEmbeddingTimeoutError}) never drops the
348
+ * batch outright (#954): the field evidence
349
+ * was that akm abandoning a timed-out request does not stop the server
350
+ * from still computing it, so immediately skipping (or immediately
351
+ * splitting, the prior behavior) just let the provider's queue grow
352
+ * while every following batch died the same way. Instead, on a timeout,
353
+ * this backs off ({@link embeddingTimeoutRetryBackoffMs}) and retries the
354
+ * SAME request once; a second timeout splits it in half (like a
355
+ * context-size rejection) and retries each half the same way, down to
356
+ * single documents — a single document that times out twice is finally
357
+ * skipped. Every other failure (network error, malformed response, a
358
+ * non-timeout HTTP failure) keeps the original skip-the-whole-batch-
359
+ * immediately behavior, at any size. The per-request timeout itself also
360
+ * scales down with the request's estimated size via
361
+ * {@link scaleEmbeddingTimeoutMs}, so a dead server is detected in seconds
362
+ * on a small batch rather than always waiting out the full configured
363
+ * `embedding.timeoutMs`.
364
+ *
365
+ * Run-scoped adaptive budget (#954, field report on beta.1): the FIRST
366
+ * context-size rejection of the run shrinks the effective request budget
367
+ * by {@link ADAPTIVE_BUDGET_SHRINK_FACTOR} (floored at
368
+ * {@link ADAPTIVE_BUDGET_FLOOR_MULTIPLIER} times `maxInputTokens`) for
369
+ * every batch not yet dispatched — the still-planned tail of `texts` is
370
+ * re-batched with `buildTokenBoundedBatches` at the smaller budget, and a
371
+ * `budget-lowered` `onBatch` event reports it once. This never touches the
372
+ * split-and-retry of the rejected batch itself (above), and never fires a
373
+ * second time in the same run even if a later batch is also rejected — a
374
+ * static configured budget that is simply too big for the endpoint should
375
+ * self-correct once, not ratchet down forever.
202
376
  */
203
377
  async embedBatch(texts, signal, onSkip, onBatch) {
204
378
  if (texts.length === 0)
@@ -206,9 +380,64 @@ export class RemoteEmbedder {
206
380
  const results = new Array(texts.length).fill(undefined);
207
381
  const headers = this.buildHeaders();
208
382
  const ollamaOpts = resolveOllamaOptions(this.config);
209
- const tokenBudget = this.config.maxTokens ?? this.config.contextLength ?? DEFAULT_TOKEN_BUDGET;
383
+ // #956: `contextLength` is Ollama's `num_ctx` ONLY (see
384
+ // resolveOllamaOptions below) — it used to double as this client-side
385
+ // request budget too, so a config author setting it for one purpose
386
+ // silently changed the other. `maxTokens` is the sole knob for the
387
+ // request budget now; unset falls back to DEFAULT_TOKEN_BUDGET.
388
+ //
389
+ // `effectiveTokenBudget` (#954) starts at the configured/default value
390
+ // and MAY shrink once, on the run's first context-size rejection — see
391
+ // `maybeShrinkBudget` below. `textBatches` is mutated in place (spliced)
392
+ // by that shrink rather than reassigned, so the in-flight
393
+ // `concurrentMap` pool below (which reads this same array by reference)
394
+ // picks up the re-planned tail without restarting.
395
+ let effectiveTokenBudget = this.config.maxTokens ?? DEFAULT_TOKEN_BUDGET;
210
396
  const maxCount = this.config.batchSize ?? DEFAULT_REMOTE_BATCH_SIZE;
211
- const textBatches = buildTokenBoundedBatches(texts, tokenBudget, maxCount);
397
+ const maxInputTokens = this.config.maxInputTokens ?? DEFAULT_MAX_INPUT_TOKENS;
398
+ const textBatches = buildTokenBoundedBatches(texts, effectiveTokenBudget, maxCount);
399
+ const configuredTimeoutMs = resolveEmbeddingTimeoutMs(this.config);
400
+ // How many of `textBatches` concurrentMap has already claimed (its own
401
+ // `nextIndex`, mirrored here so a budget shrink knows where the
402
+ // not-yet-dispatched tail begins). Assigned, not incremented, at the top
403
+ // of `runProviderBatch` — batches are claimed in strictly increasing
404
+ // order, so the highest `batchIndex` seen so far IS the claimed count.
405
+ let dispatchedBatchCount = 0;
406
+ // Set once the run's first context-size rejection has shrunk the budget
407
+ // (#954) — guards `maybeShrinkBudget` so it never fires twice.
408
+ let budgetShrunk = false;
409
+ // On the FIRST context-size rejection of this `embedBatch` call, shrink
410
+ // `effectiveTokenBudget` and re-plan every batch `concurrentMap` has not
411
+ // yet claimed from the smaller budget. Never touches `rejectedIndices`
412
+ // itself — the caller's own split-and-retry handles that batch — and is
413
+ // a no-op after the first call (`budgetShrunk`).
414
+ const maybeShrinkBudget = (rejectedIndices, rejectedBatchIndex, rejectedRequestTokens) => {
415
+ if (budgetShrunk)
416
+ return;
417
+ budgetShrunk = true;
418
+ const floor = ADAPTIVE_BUDGET_FLOOR_MULTIPLIER * maxInputTokens;
419
+ effectiveTokenBudget = Math.max(Math.round(effectiveTokenBudget * ADAPTIVE_BUDGET_SHRINK_FACTOR), floor);
420
+ const notYetDispatched = textBatches.slice(dispatchedBatchCount);
421
+ const remainingIndices = notYetDispatched.flatMap((batch) => batch.indices);
422
+ if (remainingIndices.length > 0) {
423
+ const remainingTexts = remainingIndices.map((i) => texts[i]);
424
+ const replanned = buildTokenBoundedBatches(remainingTexts, effectiveTokenBudget, maxCount).map((batch) => ({
425
+ indices: batch.indices.map((localIndex) => remainingIndices[localIndex]),
426
+ oversized: batch.oversized,
427
+ }));
428
+ textBatches.splice(dispatchedBatchCount, textBatches.length - dispatchedBatchCount, ...replanned);
429
+ }
430
+ warnVerbose(`[embed] provider rejected a ${rejectedRequestTokens}-token request as over its context; request budget lowered to ${effectiveTokenBudget.toLocaleString()} for the rest of this run`);
431
+ commitBatch(rejectedIndices, rejectedIndices.map(() => undefined), undefined, {
432
+ batchIndex: rejectedBatchIndex,
433
+ batchCount: textBatches.length,
434
+ docCount: rejectedIndices.length,
435
+ requestTokens: rejectedRequestTokens,
436
+ elapsedMs: 0,
437
+ outcome: "budget-lowered",
438
+ reason: `provider rejected ${rejectedRequestTokens.toLocaleString()} tokens as over its context; request budget lowered to ${effectiveTokenBudget.toLocaleString()} for the rest of this run`,
439
+ });
440
+ };
212
441
  // Stops the pool from claiming any FURTHER provider batch once the
213
442
  // caller's onBatch has failed once (the materializer's transaction
214
443
  // failed, so a subsequent commit would just fail again) — dispatching
@@ -242,7 +471,7 @@ export class RemoteEmbedder {
242
471
  // fabricated "batch-request-failed" skip (#954). Checked (and rethrown)
243
472
  // once the pool drains, the same way `signal?.aborted` is today.
244
473
  let firstOnBatchError;
245
- const commitBatch = (indices, embeddings, model) => {
474
+ const commitBatch = (indices, embeddings, model, outcome) => {
246
475
  if (!onBatch)
247
476
  return;
248
477
  // Once persistence has failed once, an already in-flight batch that
@@ -252,7 +481,7 @@ export class RemoteEmbedder {
252
481
  if (firstOnBatchError !== undefined)
253
482
  return;
254
483
  try {
255
- onBatch(indices, embeddings, model);
484
+ onBatch(indices, embeddings, model, outcome);
256
485
  }
257
486
  catch (err) {
258
487
  firstOnBatchError = err;
@@ -260,65 +489,207 @@ export class RemoteEmbedder {
260
489
  }
261
490
  };
262
491
  // Requests a single provider batch (by index list), recursing on a
263
- // context-size rejection. Never throws except to propagate a genuine
264
- // caller abort — every other outcome (success or a non-abort failure)
265
- // resolves normally after reporting via onSkip/onBatch. `onBatch` fires
266
- // only once this try/catch has already settled success vs. failure, so a
267
- // throw from it is never caught and reclassified by this block.
268
- const requestAndCommit = async (indices) => {
492
+ // context-size rejection OR a repeated timeout (#954). Never
493
+ // throws except to propagate a genuine caller abort — every other
494
+ // outcome (success or a non-abort failure) resolves normally after
495
+ // reporting via onSkip/onBatch. `onBatch` fires only once this
496
+ // try/catch has already settled success vs. failure, so a throw from it
497
+ // is never caught and reclassified by this block.
498
+ //
499
+ // `isTimeoutRetry` marks the SECOND attempt at this exact `indices`
500
+ // (after the one same-size backoff-and-retry) — a second
501
+ // timeout at that point splits or terminally skips rather than backing
502
+ // off again.
503
+ //
504
+ // `timeoutAttempt` (#954, field-report follow-up) counts how many splits
505
+ // down the timeout-retry chain this call is: 0 at the top level, then
506
+ // +1 each time a second timeout at one size splits into two smaller
507
+ // requests. It is the `attempt` fed to {@link embeddingTimeoutRetryBackoffMs}
508
+ // so each successive size's first-timeout backoff is longer than the
509
+ // last (5s, doubling, capped at 60s) instead of every split restarting
510
+ // at the same ~5s delay — a server still draining a whole run of
511
+ // abandoned requests needs more room the deeper the chain goes, not the
512
+ // same fixed pause every time.
513
+ const requestAndCommit = async (indices, batchIndex, isTimeoutRetry = false, timeoutAttempt = 0) => {
514
+ // A concurrent batch may have tripped the circuit breaker (or failed
515
+ // onBatch) while this call was queued behind a split or a backoff —
516
+ // never let a deeper recursive call make a request that can no longer
517
+ // be reported, mirroring how the pool below never claims a
518
+ // not-yet-started textBatch once dispatch has stopped.
519
+ if (dispatchAbort.signal.aborted)
520
+ return;
269
521
  const batch = indices.map((i) => texts[i]);
522
+ const requestTokens = batch.reduce((sum, text) => sum + estimateTokenCount(text), 0);
523
+ const requestTimeoutMs = scaleEmbeddingTimeoutMs(configuredTimeoutMs, requestTokens, effectiveTokenBudget);
524
+ const requestStart = Date.now();
270
525
  let batchEmbeddings;
271
526
  let responseModel;
527
+ let outcome;
528
+ let failureReason;
272
529
  try {
273
- const { vectors, model } = await this.requestBatch(batch, headers, ollamaOpts, signal);
530
+ const { vectors, model } = await this.requestBatch(batch, headers, ollamaOpts, requestTimeoutMs, signal);
274
531
  for (let k = 0; k < indices.length; k++) {
275
532
  results[indices[k]] = vectors[k];
276
533
  }
277
534
  batchEmbeddings = indices.map((i) => results[i]);
278
535
  responseModel = model;
536
+ outcome = "stored";
279
537
  }
280
538
  catch (err) {
281
539
  // A caller abort must still propagate — it is not a "this batch
282
540
  // failed" condition, it means stop entirely.
283
541
  if (signal?.aborted)
284
542
  throw err;
543
+ if (err instanceof ContextExceededError) {
544
+ // #954: the run's FIRST context-size rejection (any size) shrinks
545
+ // the budget for everything not yet dispatched; a no-op after the
546
+ // first call. Deliberately BEFORE the split below — it must fire
547
+ // for a single-document rejection too (which never reaches the
548
+ // `indices.length > 1` split branch), and it never touches this
549
+ // batch's own split-and-retry.
550
+ maybeShrinkBudget(indices, batchIndex, requestTokens);
551
+ }
285
552
  if (err instanceof ContextExceededError && indices.length > 1) {
286
553
  const mid = Math.ceil(indices.length / 2);
287
- await requestAndCommit(indices.slice(0, mid));
288
- await requestAndCommit(indices.slice(mid));
554
+ await requestAndCommit(indices.slice(0, mid), batchIndex, false, timeoutAttempt);
555
+ await requestAndCommit(indices.slice(mid), batchIndex, false, timeoutAttempt);
556
+ return;
557
+ }
558
+ const timedOut = isEmbeddingTimeoutError(err);
559
+ if (timedOut && !isTimeoutRetry) {
560
+ // First timeout at this size: back off so the provider can drain
561
+ // the abandoned request, then retry the SAME request once before
562
+ // ever splitting or skipping (#954). The backoff grows with
563
+ // `timeoutAttempt`, not a flat ~5s every time, so a chain of
564
+ // splits down to smaller and smaller requests gives the provider
565
+ // proportionally more room to drain each time.
566
+ const backoffMs = embeddingTimeoutRetryBackoffMs(timeoutAttempt);
567
+ warnVerbose(`[embed] batch of ${batch.length} document(s) timed out after ${requestTimeoutMs}ms; retrying once after a ${Math.round(backoffMs)}ms backoff`);
568
+ // Default-level notice (#954 field-report follow-up), not just the
569
+ // verbose line above — a run silently waiting out a multi-minute
570
+ // back-off looked identical to a hang otherwise. Nothing has
571
+ // failed or succeeded yet, so there is nothing to persist:
572
+ // `embeddings` are all `undefined` and the materializer's onBatch
573
+ // must not touch storage for this event.
574
+ commitBatch(indices, indices.map(() => undefined), undefined, {
575
+ batchIndex,
576
+ batchCount: textBatches.length,
577
+ docCount: indices.length,
578
+ requestTokens,
579
+ elapsedMs: backoffMs,
580
+ outcome: "retrying",
581
+ reason: "timed out",
582
+ });
583
+ await abortableDelay(backoffMs, signal, "embedding interrupted during retry backoff");
584
+ if (!dispatchAbort.signal.aborted) {
585
+ return requestAndCommit(indices, batchIndex, true, timeoutAttempt);
586
+ }
587
+ // Dispatch was stopped (by another batch's circuit-breaker trip)
588
+ // while this one was backing off — fall through and skip below
589
+ // instead of issuing a request that can no longer be reported.
590
+ }
591
+ else if (timedOut && indices.length > 1) {
592
+ // Timed out again on the retry: split rather than skip outright —
593
+ // the provider may still fit it once it is smaller (#954),
594
+ // the same treatment a context-size rejection gets.
595
+ // Each half's OWN first-timeout backoff (#954 follow-up) starts
596
+ // one attempt further down the chain than this size's did.
597
+ const mid = Math.ceil(indices.length / 2);
598
+ await requestAndCommit(indices.slice(0, mid), batchIndex, false, timeoutAttempt + 1);
599
+ await requestAndCommit(indices.slice(mid), batchIndex, false, timeoutAttempt + 1);
289
600
  return;
290
601
  }
291
602
  const message = err instanceof Error ? err.message : String(err);
292
603
  const reason = err instanceof ContextExceededError ? "context-window-exceeded" : "batch-request-failed";
293
- warnVerbose(`[embed] batch of ${batch.length} document(s) failed and was skipped: ${message}`);
294
- for (const idx of indices) {
295
- onSkip?.({ index: idx, reason, message });
604
+ const failureKind = reason === "batch-request-failed" ? (timedOut ? "timeout" : "network-error") : undefined;
605
+ // Default-level visibility for a failed batch (not verbose-only) is
606
+ // still guaranteed here — just not via warn(). The `commitBatch` call
607
+ // below carries `outcome: "failed"` and this `message` as `reason`
608
+ // through `onBatch`, and materialize-embeddings.ts's per-batch line
609
+ // (also default-level) prints it from there. A warn() call here used
610
+ // to print the identical event a second time on stderr — the same
611
+ // class of double-print bug fixed for the truncation/re-embed-reason
612
+ // lines in materialize-embeddings.ts (#954, field-report follow-up).
613
+ // Per-entry batch-mapping detail stays verbose-only
614
+ // (materialize-embeddings.ts).
615
+ let stopRequested = false;
616
+ for (const [k, idx] of indices.entries()) {
617
+ if (onSkip?.({
618
+ index: idx,
619
+ reason,
620
+ message,
621
+ batchStart: k === 0,
622
+ batchSize: indices.length,
623
+ failureKind,
624
+ }) === false)
625
+ stopRequested = true;
296
626
  }
627
+ // #954: the caller's circuit breaker asked to stop —
628
+ // gate further dispatch through the SAME dispatchAbort controller
629
+ // the onBatch-throw path above uses, but resolve this call normally
630
+ // (never reject) with whatever results already landed, since this is
631
+ // a policy decision, not a persistence failure.
632
+ if (stopRequested)
633
+ stopDispatch();
297
634
  batchEmbeddings = indices.map(() => undefined);
635
+ outcome = "failed";
636
+ failureReason = message;
298
637
  }
299
- commitBatch(indices, batchEmbeddings, responseModel);
638
+ commitBatch(indices, batchEmbeddings, responseModel, {
639
+ batchIndex,
640
+ batchCount: textBatches.length,
641
+ docCount: indices.length,
642
+ requestTokens,
643
+ elapsedMs: Date.now() - requestStart,
644
+ outcome,
645
+ reason: failureReason,
646
+ });
300
647
  };
301
- const runProviderBatch = async (textBatch) => {
648
+ const runProviderBatch = async (textBatch, batchIndex) => {
649
+ // Claimed in strictly increasing order by `concurrentMap` below, so
650
+ // the highest `batchIndex` seen so far is exactly how many batches it
651
+ // has claimed (#954) — see `maybeShrinkBudget`'s doc comment above.
652
+ // Assigned synchronously at entry, before any `await`, so this always
653
+ // matches `concurrentMap`'s own `nextIndex` at the moment of claim.
654
+ dispatchedBatchCount = batchIndex;
302
655
  if (textBatch.oversized) {
303
656
  const idx = textBatch.indices[0];
304
657
  const estTokens = estimateTokenCount(texts[idx]);
305
658
  onSkip?.({
306
659
  index: idx,
307
660
  reason: "context-window-exceeded",
308
- message: `Document estimated at ${estTokens} tokens exceeds the ${tokenBudget}-token embedding budget; skipped.`,
661
+ message: `Document estimated at ${estTokens} tokens exceeds the ${effectiveTokenBudget}-token embedding budget; skipped.`,
662
+ batchStart: true,
663
+ batchSize: 1,
664
+ });
665
+ // Never made a request — excluded from the default-level per-batch
666
+ // line (there is no request outcome to report), but still counted
667
+ // in the run's oversized-skip total via `onSkip` above.
668
+ commitBatch([idx], [undefined], undefined, {
669
+ batchIndex,
670
+ batchCount: textBatches.length,
671
+ docCount: 1,
672
+ requestTokens: estTokens,
673
+ elapsedMs: 0,
674
+ outcome: "failed",
675
+ reason: "oversized",
309
676
  });
310
- commitBatch([idx], [undefined]);
311
677
  return;
312
678
  }
313
- await requestAndCommit(textBatch.indices);
679
+ await requestAndCommit(textBatch.indices, batchIndex);
314
680
  };
315
681
  const concurrency = resolveEmbeddingConcurrency(this.config);
316
682
  // concurrentMap swallows a thrown fn (per-item, results discarded here —
317
683
  // requestAndCommit only throws to signal a caller abort), so abort must
318
684
  // be re-checked once the pool has drained rather than relying on the
319
685
  // throw itself to escape. Dispatch is gated on `dispatchAbort`, not the
320
- // caller's `signal` directly — see the comment above.
321
- await concurrentMap(textBatches, runProviderBatch, concurrency, { signal: dispatchAbort.signal });
686
+ // caller's `signal` directly — see the comment above. `batchIndex` is
687
+ // 1-based (matches the human-readable "batch N/Total" line) and comes
688
+ // from `concurrentMap`'s own 0-based item index, not a separately
689
+ // tracked counter, so it stays correct under concurrency.
690
+ await concurrentMap(textBatches, (textBatch, i) => runProviderBatch(textBatch, i + 1), concurrency, {
691
+ signal: dispatchAbort.signal,
692
+ });
322
693
  if (callerAbortListener)
323
694
  signal?.removeEventListener("abort", callerAbortListener);
324
695
  if (signal?.aborted) {
@@ -335,8 +706,14 @@ export class RemoteEmbedder {
335
706
  * used by the embedding-fingerprint canary to verify a config-string
336
707
  * rename against what the endpoint actually served, not just re-assert the
337
708
  * configured string. Throws on any failure.
709
+ *
710
+ * `timeoutMs` is the caller's ALREADY-SCALED per-request timeout (#954
711
+ * — see {@link scaleEmbeddingTimeoutMs}), not re-resolved here:
712
+ * `embedBatch` computes it per request from that request's own size so a
713
+ * split-down retry gets a smaller, size-appropriate timeout rather than
714
+ * always the full configured `embedding.timeoutMs`.
338
715
  */
339
- async requestBatch(batch, headers, ollamaOpts, signal) {
716
+ async requestBatch(batch, headers, ollamaOpts, timeoutMs, signal) {
340
717
  const body = {
341
718
  input: batch,
342
719
  model: this.model,
@@ -353,9 +730,9 @@ export class RemoteEmbedder {
353
730
  method: "POST",
354
731
  headers,
355
732
  body: JSON.stringify(body),
356
- }, 30_000, signal);
733
+ }, timeoutMs, signal);
357
734
  if (!response.ok) {
358
- const respBody = await readBodyWithByteCap(response, undefined, { bodyTimeoutMs: 30_000, signal }).catch((err) => {
735
+ const respBody = await readBodyWithByteCap(response, undefined, { bodyTimeoutMs: timeoutMs, signal }).catch((err) => {
359
736
  if (signal?.aborted)
360
737
  throw err;
361
738
  return "";
@@ -366,7 +743,7 @@ export class RemoteEmbedder {
366
743
  }
367
744
  throw new Error(message);
368
745
  }
369
- const json = JSON.parse(await readBodyWithByteCap(response, undefined, { bodyTimeoutMs: 30_000, signal }));
746
+ const json = JSON.parse(await readBodyWithByteCap(response, undefined, { bodyTimeoutMs: timeoutMs, signal }));
370
747
  if (!json.data || json.data.length !== batch.length) {
371
748
  throw new Error(`Unexpected embedding batch response: expected ${batch.length} embeddings, got ${json.data?.length ?? 0}.${embeddingEndpointPathHint(this.endpoint)}`);
372
749
  }
@@ -471,3 +848,22 @@ function resolveOllamaOptions(config) {
471
848
  export function hasRemoteEndpoint(config) {
472
849
  return isHttpUrl(config.endpoint);
473
850
  }
851
+ /**
852
+ * Describe WHERE an `embedding.apiKey` came from, never its value — the
853
+ * actionable outcome of the #953 field gap: `resolveSecret` throws on an
854
+ * unresolvable `secret://` reference, so a keyless request can only mean
855
+ * `embedding.apiKey` was absent from the config the run actually loaded (a
856
+ * different config root, scope, or a config edited after the run started).
857
+ * A default-level progress line naming the credential's SOURCE (this
858
+ * helper), printed once before the first provider request, lets a field run
859
+ * self-diagnose that without ever surfacing the secret itself.
860
+ */
861
+ export function describeEmbeddingCredential(apiKey) {
862
+ if (!apiKey)
863
+ return "none configured";
864
+ if (SECRET_STORE_REFERENCE_PATTERN.test(apiKey))
865
+ return `${apiKey} (store)`;
866
+ if (ENV_REFERENCE_PATTERN.test(apiKey))
867
+ return `${apiKey} (env)`;
868
+ return "literal apiKey";
869
+ }