akm-cli 0.9.15-beta.1 → 0.9.15-beta.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -7,9 +7,10 @@
7
7
  * Calls the configured `/embeddings` endpoint and L2-normalizes the returned
8
8
  * vectors so the scoring pipeline's L2-to-cosine conversion is correct.
9
9
  */
10
- import { fetchWithTimeout, isHttpUrl, readBodyWithByteCap } from "../../core/common.js";
10
+ import { abortableDelay, backoffDelay, fetchWithTimeout, isHttpUrl, readBodyWithByteCap } from "../../core/common.js";
11
11
  import { concurrentMap } from "../../core/concurrent.js";
12
12
  import { resolveSecret } from "../../core/config/config.js";
13
+ import { ENV_REFERENCE_PATTERN, SECRET_STORE_REFERENCE_PATTERN } from "../../core/config/schema/primitives.js";
13
14
  import { defaultConcurrencyForEndpoint } from "../../core/loopback.js";
14
15
  import { redactErrorBody, redactSensitiveText } from "../../core/redaction.js";
15
16
  import { warnVerbose } from "../../core/warn.js";
@@ -24,7 +25,8 @@ import { resolveSecretFromStore } from "../../sources/snapshot-fetchers/secret-s
24
25
  export const DEFAULT_REMOTE_BATCH_SIZE = 100;
25
26
  /**
26
27
  * Conservative default token budget per HTTP request when the config gives
27
- * no better number (`maxTokens` or `contextLength`). #874's measurements:
28
+ * no better number (`maxTokens` — see #956 for why `contextLength`
29
+ * no longer feeds this). #874's measurements:
28
30
  * a batch of 100 small docs (~400 KB, ~100K tokens) took 14.8s against a
29
31
  * healthy local endpoint — half the 30s request timeout — and a single
30
32
  * 128 KB (~24K token) document alone was rejected by the endpoint as
@@ -36,6 +38,110 @@ export const DEFAULT_TOKEN_BUDGET = 8000;
36
38
  export function estimateTokenCount(text) {
37
39
  return Math.round(text.length / 4);
38
40
  }
41
+ /**
42
+ * Default per-document embedding cap (`embedding.maxInputTokens`, #956)
43
+ * — the materializer truncates a document's embedded text to
44
+ * this cap (head only) instead of skipping it outright, so one oversized
45
+ * entry can no longer fail a whole batch. Fragments are not embedded at all
46
+ * (only the entry's own search text is), so this is the only lever on how
47
+ * much of a large document contributes to its vector.
48
+ */
49
+ export const DEFAULT_MAX_INPUT_TOKENS = 512;
50
+ /**
51
+ * Truncate `text` to at most `maxTokens` (estimated via
52
+ * {@link estimateTokenCount}, the same 4-chars≈1-token rule the batching
53
+ * budget uses), keeping only its head. The cut never splits a UTF-16
54
+ * surrogate pair. Text already at or under the cap is returned unchanged
55
+ * (`truncated: false`) — including empty text, which is never itself
56
+ * "truncated".
57
+ */
58
+ export function capEmbeddingText(text, maxTokens) {
59
+ if (estimateTokenCount(text) <= maxTokens)
60
+ return { text, truncated: false };
61
+ const charBudget = Math.max(0, maxTokens * 4);
62
+ let cut = Math.min(charBudget, text.length);
63
+ if (cut > 0 && cut < text.length) {
64
+ const code = text.charCodeAt(cut);
65
+ // A low surrogate (0xDC00-0xDFFF) at the cut point means its high
66
+ // surrogate is the character just before it — back off one position so
67
+ // the pair stays together rather than yielding a lone surrogate.
68
+ if (code >= 0xdc00 && code <= 0xdfff)
69
+ cut -= 1;
70
+ }
71
+ return { text: text.slice(0, cut), truncated: true };
72
+ }
73
+ /**
74
+ * Default per-request timeout when `embedding.timeoutMs` is unset (#954).
75
+ * The prior fixed 30s cut off exactly the field-report case: a
76
+ * local model server on an 8000-token (`DEFAULT_TOKEN_BUDGET`) batch
77
+ * legitimately takes longer than that, and the timeout fired mid-response
78
+ * with no retry — every batch it hit was silently dropped for the rest of
79
+ * an hours-long run. 120s comfortably covers a slow local batch while still
80
+ * bounding a genuinely dead endpoint to a few minutes, not forever.
81
+ */
82
+ export const DEFAULT_EMBEDDING_TIMEOUT_MS = 120_000;
83
+ /** Resolve the effective per-request timeout: `embedding.timeoutMs` when set, else the default above. */
84
+ export function resolveEmbeddingTimeoutMs(config) {
85
+ return config.timeoutMs ?? DEFAULT_EMBEDDING_TIMEOUT_MS;
86
+ }
87
+ /**
88
+ * Scale the per-request timeout down for a smaller-than-budget request
89
+ * (#954, field-report follow-up): `embedding.timeoutMs` /
90
+ * {@link resolveEmbeddingTimeoutMs} is the budget for a request at the FULL
91
+ * token budget; a batch using only a fraction of it gets a proportionally
92
+ * smaller timeout, floored at 30s and never above the configured
93
+ * `timeoutMs` itself, so a dead server is detected in seconds on the common
94
+ * case of small documents instead of always waiting out the full configured
95
+ * budget.
96
+ */
97
+ export function scaleEmbeddingTimeoutMs(timeoutMs, requestTokens, tokenBudget) {
98
+ const scaled = tokenBudget > 0 ? timeoutMs * (requestTokens / tokenBudget) : timeoutMs;
99
+ return Math.min(Math.max(scaled, 30_000), timeoutMs);
100
+ }
101
+ /**
102
+ * True when `err` is a request- or body-read timeout (#954) —
103
+ * `fetchWithTimeout`'s connection/header timeout ("Request timed out
104
+ * after...") or `readBodyWithByteCap`'s body-phase `BodyReadTimeoutError`.
105
+ * Only this failure mode gets the back-off-and-retry treatment:
106
+ * the field evidence was specifically that a timed-out request keeps
107
+ * computing server-side, so abandoning it immediately (the prior
108
+ * behavior) just grows the provider's queue further. A genuine network/HTTP
109
+ * failure (connection refused, malformed response, a real error response)
110
+ * has no such still-in-flight hazard and keeps the original
111
+ * skip-immediately behavior.
112
+ */
113
+ export function isEmbeddingTimeoutError(err) {
114
+ if (!(err instanceof Error))
115
+ return false;
116
+ if (err.name === "BodyReadTimeoutError")
117
+ return true;
118
+ return err.message.startsWith("Request timed out after ");
119
+ }
120
+ /** TEST-ONLY seam: override the backoff base/max so retry-backoff tests run fast without waiting real seconds. */
121
+ let embeddingTimeoutBackoffOverrideForTests;
122
+ /** TEST-ONLY. Pass undefined to restore the real 5s/60s backoff. */
123
+ export function _setEmbeddingTimeoutBackoffForTests(config) {
124
+ embeddingTimeoutBackoffOverrideForTests = config;
125
+ }
126
+ /**
127
+ * Backoff before the single same-size retry on a request timeout
128
+ * (#954) — reuses the same jittered
129
+ * exponential formula {@link backoffDelay} uses for the rest of the
130
+ * codebase's retry paths, at a base of "5s, doubling, capped at 60s".
131
+ *
132
+ * `timeoutAttempt` (#954, field-report follow-up) is how many times the
133
+ * SAME-SIZE-retry-then-split chain has already split before reaching this
134
+ * size — 0 at the top level. The first-timeout backoff for each successive,
135
+ * smaller size after a split grows with it (5s, then doubling, capped at
136
+ * 60s) instead of every split resetting to a flat ~5s: the field evidence
137
+ * was that an abandoned request keeps computing server-side, so a server
138
+ * already draining a whole chain of abandoned requests needs progressively
139
+ * more room, not the same fixed pause at every size.
140
+ */
141
+ export function embeddingTimeoutRetryBackoffMs(timeoutAttempt = 0) {
142
+ const { baseMs, maxMs } = embeddingTimeoutBackoffOverrideForTests ?? { baseMs: 5_000, maxMs: 60_000 };
143
+ return backoffDelay(timeoutAttempt, baseMs, maxMs);
144
+ }
39
145
  /**
40
146
  * Distinguishes a batch rejected because it exceeded the endpoint's context
41
147
  * window from every other failure mode (network error, 5xx, malformed
@@ -50,8 +156,15 @@ export class ContextExceededError extends Error {
50
156
  this.name = "ContextExceededError";
51
157
  }
52
158
  }
53
- /** Patterns providers use to report a request too large for the model's context window. */
54
- const CONTEXT_EXCEEDED_PATTERN = /exceed_context_size_error|context size|context length|too many tokens/i;
159
+ /**
160
+ * Patterns providers use to report a request too large for the model's
161
+ * context window. `input is too large to process`/`physical batch size`/
162
+ * `ubatch` (#954) cover llama.cpp's own physical-batch
163
+ * rejection (HTTP 500, e.g. "input is too large to process. increase the
164
+ * physical batch size"), which was previously an unrecognized generic
165
+ * failure — the whole batch was dropped instead of split and retried.
166
+ */
167
+ const CONTEXT_EXCEEDED_PATTERN = /exceed_context_size_error|context size|context length|too many tokens|input is too large to process|physical batch size|ubatch/i;
55
168
  /**
56
169
  * True when an HTTP failure means "this request's input is too large for the
57
170
  * endpoint's context window" rather than some other failure. HTTP 413
@@ -65,16 +178,25 @@ export function isContextExceededResponse(status, body) {
65
178
  }
66
179
  /**
67
180
  * Resolve the effective in-flight request window for `RemoteEmbedder.embedBatch`.
68
- * FIXED — no config override: 1 for a loopback endpoint, 2 for a remote one,
69
- * via the shared `defaultConcurrencyForEndpoint` (`src/core/loopback.ts`),
70
- * the same lowest-common-denominator rule `getDefaultLlmConcurrency`
71
- * (`src/indexer/indexer.ts`) uses. The actual throughput knob is request
72
- * SIZE, not request count: `embedding.batchSize` (document cap) and
73
- * `embedding.maxTokens`/`contextLength` (token budget) reach a larger batch
74
- * per request, which is where most of the win is — a 32-input batch takes
75
- * about the same wall time as one input against a healthy endpoint.
181
+ * Default (unset `embedding.concurrency`): 1 for a loopback endpoint, 2 for a
182
+ * remote one, via the shared `defaultConcurrencyForEndpoint`
183
+ * (`src/core/loopback.ts`), the same lowest-common-denominator rule
184
+ * `getDefaultLlmConcurrency` (`src/indexer/indexer.ts`) uses.
185
+ *
186
+ * `embedding.concurrency` (#954) overrides this default in
187
+ * either direction, bounded 1-16 at the config schema — added after field
188
+ * evidence that a multi-slot local server (llama.cpp `--parallel N`, vLLM)
189
+ * genuinely serves parallel requests and was left idle by the fixed default.
190
+ * Request SIZE remains the first throughput lever regardless:
191
+ * `embedding.batchSize` (document cap) and `embedding.maxTokens` (request
192
+ * token budget — see #956; `contextLength` no longer feeds it)
193
+ * reach a larger batch per request, which is where most of the win is for a
194
+ * single-slot server — a 32-input batch takes about the same wall time as
195
+ * one input against a healthy endpoint.
76
196
  */
77
197
  export function resolveEmbeddingConcurrency(config) {
198
+ if (typeof config.concurrency === "number")
199
+ return config.concurrency;
78
200
  return defaultConcurrencyForEndpoint(config.endpoint);
79
201
  }
80
202
  /**
@@ -138,6 +260,7 @@ export class RemoteEmbedder {
138
260
  if (ollamaOpts) {
139
261
  body.options = ollamaOpts;
140
262
  }
263
+ const timeoutMs = resolveEmbeddingTimeoutMs(this.config);
141
264
  // `signal` MUST go through fetchWithTimeout's dedicated 4th parameter, not
142
265
  // the RequestInit: fetchWithTimeout replaces `opts.signal` with its own
143
266
  // controller (`{ ...opts, signal: controller.signal }`), so a signal passed
@@ -146,16 +269,16 @@ export class RemoteEmbedder {
146
269
  method: "POST",
147
270
  headers,
148
271
  body: JSON.stringify(body),
149
- }, 30_000, signal);
272
+ }, timeoutMs, signal);
150
273
  if (!response.ok) {
151
- const errBody = await readBodyWithByteCap(response, undefined, { bodyTimeoutMs: 30_000, signal }).catch((err) => {
274
+ const errBody = await readBodyWithByteCap(response, undefined, { bodyTimeoutMs: timeoutMs, signal }).catch((err) => {
152
275
  if (signal?.aborted)
153
276
  throw err;
154
277
  return "";
155
278
  });
156
279
  throw new Error(`Embedding request failed (${response.status}): ${this.safeErrorBody(errBody)}`);
157
280
  }
158
- const json = JSON.parse(await readBodyWithByteCap(response, undefined, { bodyTimeoutMs: 30_000, signal }));
281
+ const json = JSON.parse(await readBodyWithByteCap(response, undefined, { bodyTimeoutMs: timeoutMs, signal }));
159
282
  if (!json.data?.[0]?.embedding) {
160
283
  throw new Error(`Unexpected embedding response format: missing data[0].embedding.${embeddingEndpointPathHint(this.endpoint)}`);
161
284
  }
@@ -197,8 +320,25 @@ export class RemoteEmbedder {
197
320
  * {@link isContextExceededResponse}) is split in half and retried
198
321
  * recursively rather than skipped outright, down to individual documents; a
199
322
  * single document that still fails this way becomes a genuine
200
- * `context-window-exceeded` skip. Every other failure (network error, 5xx,
201
- * malformed response) keeps the original skip-the-whole-batch behavior.
323
+ * `context-window-exceeded` skip.
324
+ *
325
+ * A request TIMEOUT (see {@link isEmbeddingTimeoutError}) never drops the
326
+ * batch outright (#954): the field evidence
327
+ * was that akm abandoning a timed-out request does not stop the server
328
+ * from still computing it, so immediately skipping (or immediately
329
+ * splitting, the prior behavior) just let the provider's queue grow
330
+ * while every following batch died the same way. Instead, on a timeout,
331
+ * this backs off ({@link embeddingTimeoutRetryBackoffMs}) and retries the
332
+ * SAME request once; a second timeout splits it in half (like a
333
+ * context-size rejection) and retries each half the same way, down to
334
+ * single documents — a single document that times out twice is finally
335
+ * skipped. Every other failure (network error, malformed response, a
336
+ * non-timeout HTTP failure) keeps the original skip-the-whole-batch-
337
+ * immediately behavior, at any size. The per-request timeout itself also
338
+ * scales down with the request's estimated size via
339
+ * {@link scaleEmbeddingTimeoutMs}, so a dead server is detected in seconds
340
+ * on a small batch rather than always waiting out the full configured
341
+ * `embedding.timeoutMs`.
202
342
  */
203
343
  async embedBatch(texts, signal, onSkip, onBatch) {
204
344
  if (texts.length === 0)
@@ -206,9 +346,15 @@ export class RemoteEmbedder {
206
346
  const results = new Array(texts.length).fill(undefined);
207
347
  const headers = this.buildHeaders();
208
348
  const ollamaOpts = resolveOllamaOptions(this.config);
209
- const tokenBudget = this.config.maxTokens ?? this.config.contextLength ?? DEFAULT_TOKEN_BUDGET;
349
+ // #956: `contextLength` is Ollama's `num_ctx` ONLY (see
350
+ // resolveOllamaOptions below) — it used to double as this client-side
351
+ // request budget too, so a config author setting it for one purpose
352
+ // silently changed the other. `maxTokens` is the sole knob for the
353
+ // request budget now; unset falls back to DEFAULT_TOKEN_BUDGET.
354
+ const tokenBudget = this.config.maxTokens ?? DEFAULT_TOKEN_BUDGET;
210
355
  const maxCount = this.config.batchSize ?? DEFAULT_REMOTE_BATCH_SIZE;
211
356
  const textBatches = buildTokenBoundedBatches(texts, tokenBudget, maxCount);
357
+ const configuredTimeoutMs = resolveEmbeddingTimeoutMs(this.config);
212
358
  // Stops the pool from claiming any FURTHER provider batch once the
213
359
  // caller's onBatch has failed once (the materializer's transaction
214
360
  // failed, so a subsequent commit would just fail again) — dispatching
@@ -242,7 +388,7 @@ export class RemoteEmbedder {
242
388
  // fabricated "batch-request-failed" skip (#954). Checked (and rethrown)
243
389
  // once the pool drains, the same way `signal?.aborted` is today.
244
390
  let firstOnBatchError;
245
- const commitBatch = (indices, embeddings, model) => {
391
+ const commitBatch = (indices, embeddings, model, outcome) => {
246
392
  if (!onBatch)
247
393
  return;
248
394
  // Once persistence has failed once, an already in-flight batch that
@@ -252,7 +398,7 @@ export class RemoteEmbedder {
252
398
  if (firstOnBatchError !== undefined)
253
399
  return;
254
400
  try {
255
- onBatch(indices, embeddings, model);
401
+ onBatch(indices, embeddings, model, outcome);
256
402
  }
257
403
  catch (err) {
258
404
  firstOnBatchError = err;
@@ -260,22 +406,51 @@ export class RemoteEmbedder {
260
406
  }
261
407
  };
262
408
  // Requests a single provider batch (by index list), recursing on a
263
- // context-size rejection. Never throws except to propagate a genuine
264
- // caller abort — every other outcome (success or a non-abort failure)
265
- // resolves normally after reporting via onSkip/onBatch. `onBatch` fires
266
- // only once this try/catch has already settled success vs. failure, so a
267
- // throw from it is never caught and reclassified by this block.
268
- const requestAndCommit = async (indices) => {
409
+ // context-size rejection OR a repeated timeout (#954). Never
410
+ // throws except to propagate a genuine caller abort — every other
411
+ // outcome (success or a non-abort failure) resolves normally after
412
+ // reporting via onSkip/onBatch. `onBatch` fires only once this
413
+ // try/catch has already settled success vs. failure, so a throw from it
414
+ // is never caught and reclassified by this block.
415
+ //
416
+ // `isTimeoutRetry` marks the SECOND attempt at this exact `indices`
417
+ // (after the one same-size backoff-and-retry) — a second
418
+ // timeout at that point splits or terminally skips rather than backing
419
+ // off again.
420
+ //
421
+ // `timeoutAttempt` (#954, field-report follow-up) counts how many splits
422
+ // down the timeout-retry chain this call is: 0 at the top level, then
423
+ // +1 each time a second timeout at one size splits into two smaller
424
+ // requests. It is the `attempt` fed to {@link embeddingTimeoutRetryBackoffMs}
425
+ // so each successive size's first-timeout backoff is longer than the
426
+ // last (5s, doubling, capped at 60s) instead of every split restarting
427
+ // at the same ~5s delay — a server still draining a whole run of
428
+ // abandoned requests needs more room the deeper the chain goes, not the
429
+ // same fixed pause every time.
430
+ const requestAndCommit = async (indices, batchIndex, isTimeoutRetry = false, timeoutAttempt = 0) => {
431
+ // A concurrent batch may have tripped the circuit breaker (or failed
432
+ // onBatch) while this call was queued behind a split or a backoff —
433
+ // never let a deeper recursive call make a request that can no longer
434
+ // be reported, mirroring how the pool below never claims a
435
+ // not-yet-started textBatch once dispatch has stopped.
436
+ if (dispatchAbort.signal.aborted)
437
+ return;
269
438
  const batch = indices.map((i) => texts[i]);
439
+ const requestTokens = batch.reduce((sum, text) => sum + estimateTokenCount(text), 0);
440
+ const requestTimeoutMs = scaleEmbeddingTimeoutMs(configuredTimeoutMs, requestTokens, tokenBudget);
441
+ const requestStart = Date.now();
270
442
  let batchEmbeddings;
271
443
  let responseModel;
444
+ let outcome;
445
+ let failureReason;
272
446
  try {
273
- const { vectors, model } = await this.requestBatch(batch, headers, ollamaOpts, signal);
447
+ const { vectors, model } = await this.requestBatch(batch, headers, ollamaOpts, requestTimeoutMs, signal);
274
448
  for (let k = 0; k < indices.length; k++) {
275
449
  results[indices[k]] = vectors[k];
276
450
  }
277
451
  batchEmbeddings = indices.map((i) => results[i]);
278
452
  responseModel = model;
453
+ outcome = "stored";
279
454
  }
280
455
  catch (err) {
281
456
  // A caller abort must still propagate — it is not a "this batch
@@ -284,21 +459,101 @@ export class RemoteEmbedder {
284
459
  throw err;
285
460
  if (err instanceof ContextExceededError && indices.length > 1) {
286
461
  const mid = Math.ceil(indices.length / 2);
287
- await requestAndCommit(indices.slice(0, mid));
288
- await requestAndCommit(indices.slice(mid));
462
+ await requestAndCommit(indices.slice(0, mid), batchIndex, false, timeoutAttempt);
463
+ await requestAndCommit(indices.slice(mid), batchIndex, false, timeoutAttempt);
464
+ return;
465
+ }
466
+ const timedOut = isEmbeddingTimeoutError(err);
467
+ if (timedOut && !isTimeoutRetry) {
468
+ // First timeout at this size: back off so the provider can drain
469
+ // the abandoned request, then retry the SAME request once before
470
+ // ever splitting or skipping (#954). The backoff grows with
471
+ // `timeoutAttempt`, not a flat ~5s every time, so a chain of
472
+ // splits down to smaller and smaller requests gives the provider
473
+ // proportionally more room to drain each time.
474
+ const backoffMs = embeddingTimeoutRetryBackoffMs(timeoutAttempt);
475
+ warnVerbose(`[embed] batch of ${batch.length} document(s) timed out after ${requestTimeoutMs}ms; retrying once after a ${Math.round(backoffMs)}ms backoff`);
476
+ // Default-level notice (#954 field-report follow-up), not just the
477
+ // verbose line above — a run silently waiting out a multi-minute
478
+ // back-off looked identical to a hang otherwise. Nothing has
479
+ // failed or succeeded yet, so there is nothing to persist:
480
+ // `embeddings` are all `undefined` and the materializer's onBatch
481
+ // must not touch storage for this event.
482
+ commitBatch(indices, indices.map(() => undefined), undefined, {
483
+ batchIndex,
484
+ batchCount: textBatches.length,
485
+ docCount: indices.length,
486
+ requestTokens,
487
+ elapsedMs: backoffMs,
488
+ outcome: "retrying",
489
+ reason: "timed out",
490
+ });
491
+ await abortableDelay(backoffMs, signal, "embedding interrupted during retry backoff");
492
+ if (!dispatchAbort.signal.aborted) {
493
+ return requestAndCommit(indices, batchIndex, true, timeoutAttempt);
494
+ }
495
+ // Dispatch was stopped (by another batch's circuit-breaker trip)
496
+ // while this one was backing off — fall through and skip below
497
+ // instead of issuing a request that can no longer be reported.
498
+ }
499
+ else if (timedOut && indices.length > 1) {
500
+ // Timed out again on the retry: split rather than skip outright —
501
+ // the provider may still fit it once it is smaller (#954),
502
+ // the same treatment a context-size rejection gets.
503
+ // Each half's OWN first-timeout backoff (#954 follow-up) starts
504
+ // one attempt further down the chain than this size's did.
505
+ const mid = Math.ceil(indices.length / 2);
506
+ await requestAndCommit(indices.slice(0, mid), batchIndex, false, timeoutAttempt + 1);
507
+ await requestAndCommit(indices.slice(mid), batchIndex, false, timeoutAttempt + 1);
289
508
  return;
290
509
  }
291
510
  const message = err instanceof Error ? err.message : String(err);
292
511
  const reason = err instanceof ContextExceededError ? "context-window-exceeded" : "batch-request-failed";
293
- warnVerbose(`[embed] batch of ${batch.length} document(s) failed and was skipped: ${message}`);
294
- for (const idx of indices) {
295
- onSkip?.({ index: idx, reason, message });
512
+ const failureKind = reason === "batch-request-failed" ? (timedOut ? "timeout" : "network-error") : undefined;
513
+ // Default-level visibility for a failed batch (not verbose-only) is
514
+ // still guaranteed here — just not via warn(). The `commitBatch` call
515
+ // below carries `outcome: "failed"` and this `message` as `reason`
516
+ // through `onBatch`, and materialize-embeddings.ts's per-batch line
517
+ // (also default-level) prints it from there. A warn() call here used
518
+ // to print the identical event a second time on stderr — the same
519
+ // class of double-print bug fixed for the truncation/re-embed-reason
520
+ // lines in materialize-embeddings.ts (#954, field-report follow-up).
521
+ // Per-entry batch-mapping detail stays verbose-only
522
+ // (materialize-embeddings.ts).
523
+ let stopRequested = false;
524
+ for (const [k, idx] of indices.entries()) {
525
+ if (onSkip?.({
526
+ index: idx,
527
+ reason,
528
+ message,
529
+ batchStart: k === 0,
530
+ batchSize: indices.length,
531
+ failureKind,
532
+ }) === false)
533
+ stopRequested = true;
296
534
  }
535
+ // #954: the caller's circuit breaker asked to stop —
536
+ // gate further dispatch through the SAME dispatchAbort controller
537
+ // the onBatch-throw path above uses, but resolve this call normally
538
+ // (never reject) with whatever results already landed, since this is
539
+ // a policy decision, not a persistence failure.
540
+ if (stopRequested)
541
+ stopDispatch();
297
542
  batchEmbeddings = indices.map(() => undefined);
543
+ outcome = "failed";
544
+ failureReason = message;
298
545
  }
299
- commitBatch(indices, batchEmbeddings, responseModel);
546
+ commitBatch(indices, batchEmbeddings, responseModel, {
547
+ batchIndex,
548
+ batchCount: textBatches.length,
549
+ docCount: indices.length,
550
+ requestTokens,
551
+ elapsedMs: Date.now() - requestStart,
552
+ outcome,
553
+ reason: failureReason,
554
+ });
300
555
  };
301
- const runProviderBatch = async (textBatch) => {
556
+ const runProviderBatch = async (textBatch, batchIndex) => {
302
557
  if (textBatch.oversized) {
303
558
  const idx = textBatch.indices[0];
304
559
  const estTokens = estimateTokenCount(texts[idx]);
@@ -306,19 +561,37 @@ export class RemoteEmbedder {
306
561
  index: idx,
307
562
  reason: "context-window-exceeded",
308
563
  message: `Document estimated at ${estTokens} tokens exceeds the ${tokenBudget}-token embedding budget; skipped.`,
564
+ batchStart: true,
565
+ batchSize: 1,
566
+ });
567
+ // Never made a request — excluded from the default-level per-batch
568
+ // line (there is no request outcome to report), but still counted
569
+ // in the run's oversized-skip total via `onSkip` above.
570
+ commitBatch([idx], [undefined], undefined, {
571
+ batchIndex,
572
+ batchCount: textBatches.length,
573
+ docCount: 1,
574
+ requestTokens: estTokens,
575
+ elapsedMs: 0,
576
+ outcome: "failed",
577
+ reason: "oversized",
309
578
  });
310
- commitBatch([idx], [undefined]);
311
579
  return;
312
580
  }
313
- await requestAndCommit(textBatch.indices);
581
+ await requestAndCommit(textBatch.indices, batchIndex);
314
582
  };
315
583
  const concurrency = resolveEmbeddingConcurrency(this.config);
316
584
  // concurrentMap swallows a thrown fn (per-item, results discarded here —
317
585
  // requestAndCommit only throws to signal a caller abort), so abort must
318
586
  // be re-checked once the pool has drained rather than relying on the
319
587
  // throw itself to escape. Dispatch is gated on `dispatchAbort`, not the
320
- // caller's `signal` directly — see the comment above.
321
- await concurrentMap(textBatches, runProviderBatch, concurrency, { signal: dispatchAbort.signal });
588
+ // caller's `signal` directly — see the comment above. `batchIndex` is
589
+ // 1-based (matches the human-readable "batch N/Total" line) and comes
590
+ // from `concurrentMap`'s own 0-based item index, not a separately
591
+ // tracked counter, so it stays correct under concurrency.
592
+ await concurrentMap(textBatches, (textBatch, i) => runProviderBatch(textBatch, i + 1), concurrency, {
593
+ signal: dispatchAbort.signal,
594
+ });
322
595
  if (callerAbortListener)
323
596
  signal?.removeEventListener("abort", callerAbortListener);
324
597
  if (signal?.aborted) {
@@ -335,8 +608,14 @@ export class RemoteEmbedder {
335
608
  * used by the embedding-fingerprint canary to verify a config-string
336
609
  * rename against what the endpoint actually served, not just re-assert the
337
610
  * configured string. Throws on any failure.
611
+ *
612
+ * `timeoutMs` is the caller's ALREADY-SCALED per-request timeout (#954
613
+ * — see {@link scaleEmbeddingTimeoutMs}), not re-resolved here:
614
+ * `embedBatch` computes it per request from that request's own size so a
615
+ * split-down retry gets a smaller, size-appropriate timeout rather than
616
+ * always the full configured `embedding.timeoutMs`.
338
617
  */
339
- async requestBatch(batch, headers, ollamaOpts, signal) {
618
+ async requestBatch(batch, headers, ollamaOpts, timeoutMs, signal) {
340
619
  const body = {
341
620
  input: batch,
342
621
  model: this.model,
@@ -353,9 +632,9 @@ export class RemoteEmbedder {
353
632
  method: "POST",
354
633
  headers,
355
634
  body: JSON.stringify(body),
356
- }, 30_000, signal);
635
+ }, timeoutMs, signal);
357
636
  if (!response.ok) {
358
- const respBody = await readBodyWithByteCap(response, undefined, { bodyTimeoutMs: 30_000, signal }).catch((err) => {
637
+ const respBody = await readBodyWithByteCap(response, undefined, { bodyTimeoutMs: timeoutMs, signal }).catch((err) => {
359
638
  if (signal?.aborted)
360
639
  throw err;
361
640
  return "";
@@ -366,7 +645,7 @@ export class RemoteEmbedder {
366
645
  }
367
646
  throw new Error(message);
368
647
  }
369
- const json = JSON.parse(await readBodyWithByteCap(response, undefined, { bodyTimeoutMs: 30_000, signal }));
648
+ const json = JSON.parse(await readBodyWithByteCap(response, undefined, { bodyTimeoutMs: timeoutMs, signal }));
370
649
  if (!json.data || json.data.length !== batch.length) {
371
650
  throw new Error(`Unexpected embedding batch response: expected ${batch.length} embeddings, got ${json.data?.length ?? 0}.${embeddingEndpointPathHint(this.endpoint)}`);
372
651
  }
@@ -471,3 +750,22 @@ function resolveOllamaOptions(config) {
471
750
  export function hasRemoteEndpoint(config) {
472
751
  return isHttpUrl(config.endpoint);
473
752
  }
753
+ /**
754
+ * Describe WHERE an `embedding.apiKey` came from, never its value — the
755
+ * actionable outcome of the #953 field gap: `resolveSecret` throws on an
756
+ * unresolvable `secret://` reference, so a keyless request can only mean
757
+ * `embedding.apiKey` was absent from the config the run actually loaded (a
758
+ * different config root, scope, or a config edited after the run started).
759
+ * A default-level progress line naming the credential's SOURCE (this
760
+ * helper), printed once before the first provider request, lets a field run
761
+ * self-diagnose that without ever surfacing the secret itself.
762
+ */
763
+ export function describeEmbeddingCredential(apiKey) {
764
+ if (!apiKey)
765
+ return "none configured";
766
+ if (SECRET_STORE_REFERENCE_PATTERN.test(apiKey))
767
+ return `${apiKey} (store)`;
768
+ if (ENV_REFERENCE_PATTERN.test(apiKey))
769
+ return `${apiKey} (env)`;
770
+ return "literal apiKey";
771
+ }