akm-cli 0.9.15-beta.1 → 0.9.15-beta.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,12 +1,14 @@
1
1
  // This Source Code Form is subject to the terms of the Mozilla Public
2
2
  // License, v. 2.0. If a copy of the MPL was not distributed with this
3
3
  // file, You can obtain one at https://mozilla.org/MPL/2.0/.
4
+ import { getConfigPath } from "../core/paths.js";
4
5
  import { isVerbose, warn, warnVerbose } from "../core/warn.js";
5
6
  import { embedBatch } from "../llm/embedder.js";
6
7
  import { DETERMINISTIC_EMBED_MODEL_ID, isDeterministicEmbedEnabled } from "../llm/embedders/deterministic.js";
7
8
  import { DEFAULT_LOCAL_MODEL } from "../llm/embedders/local.js";
8
- import { buildTokenBoundedBatches, DEFAULT_REMOTE_BATCH_SIZE, DEFAULT_TOKEN_BUDGET, estimateTokenCount, hasRemoteEndpoint, } from "../llm/embedders/remote.js";
9
+ import { buildTokenBoundedBatches, capEmbeddingText, DEFAULT_MAX_INPUT_TOKENS, DEFAULT_REMOTE_BATCH_SIZE, DEFAULT_TOKEN_BUDGET, describeEmbeddingCredential, estimateTokenCount, hasRemoteEndpoint, normalizeEmbeddingEndpoint, } from "../llm/embedders/remote.js";
9
10
  import { cosineSimilarity } from "../llm/embedders/types.js";
11
+ import { purgeEmbeddingSalvage, relabelEmbeddingSalvageFingerprint, reuseSalvagedEmbeddings, } from "../storage/repositories/embedding-salvage-repository.js";
10
12
  import { getEmbeddableEntryCount } from "../storage/repositories/index-entries-repository.js";
11
13
  import { deleteMeta, getMeta, setMeta } from "../storage/repositories/index-meta-repository.js";
12
14
  import { getAllEntriesForEmbedding, getEmbeddingCount, isVecFastPathComplete, isVecFastPathReady, purgeEmbeddings, sampleEmbeddedEntriesForCanary, setVecFastPathReady, upsertEmbedding, } from "../storage/repositories/index-vec-repository.js";
@@ -23,8 +25,14 @@ export function deriveSemanticProviderFingerprint(embedding) {
23
25
  }
24
26
  return `local:${embedding?.localModel ?? DEFAULT_LOCAL_MODEL}`;
25
27
  }
26
- /** How often (in stored entries) to emit a progress line during a large embedding run (#954). */
27
- const PROGRESS_INTERVAL = 500;
28
+ /**
29
+ * The heartbeat text emitted every 15s while a provider request is in
30
+ * flight, and by default (not `--verbose`-only) since silence indistinguishable
31
+ * from a hang was the field report's own symptom (#954).
32
+ */
33
+ export function formatEmbeddingHeartbeat(storedCount, total, failedCount) {
34
+ return `Still generating embeddings: ${storedCount}/${total} stored, ${failedCount} failed; waiting on embedding provider.`;
35
+ }
28
36
  /**
29
37
  * Number of already-embedded entries sampled for the fingerprint-rename
30
38
  * canary (#955) — small and cheap even against a slow local server; a
@@ -38,6 +46,23 @@ const CANARY_SAMPLE_SIZE = 8;
38
46
  * same-model rename rather than a real model change (#955).
39
47
  */
40
48
  const CANARY_SIMILARITY_THRESHOLD = 0.999;
49
+ /**
50
+ * Consecutive transport failures after which the embedding pass stops
51
+ * dispatching further requests and ends the run as a failure rather than
52
+ * grinding through every remaining batch against a dead endpoint (#954).
53
+ * Two independent trip conditions
54
+ * share this threshold — see `onSkip` below: 3 consecutive failures at
55
+ * single-document size (timeout OR network error — a multi-document
56
+ * timeout is not by itself evidence the endpoint is dead, since
57
+ * `RemoteEmbedder.embedBatch` already retries and splits it smaller before
58
+ * ever reporting it as failed at single-document size), or 3 consecutive
59
+ * network errors at ANY size (a network error is never retried, so it is
60
+ * trusted immediately regardless of how large the request was).
61
+ * `context-window-exceeded` never counts — that reason proves the provider
62
+ * IS reachable, and split-and-retry already handles it; it resets both
63
+ * streaks instead.
64
+ */
65
+ const CIRCUIT_BREAKER_THRESHOLD = 3;
41
66
  /**
42
67
  * Pure decision: do stored vectors remain valid against freshly re-embedded
43
68
  * canary samples? The ONE place that computes the canary's similarity
@@ -89,7 +114,7 @@ function medianOf(values) {
89
114
  * on what the server (or local model) actually reported plus the observed
90
115
  * vector width, so a gateway/transport change that keeps returning the same
91
116
  * underlying model can be told apart from a genuine model change without
92
- * relying on the operator's config string (#955 field-review addendum).
117
+ * relying on the operator's config string (#955).
93
118
  * Returns undefined when nothing was actually observed this call (no vector
94
119
  * to measure yet).
95
120
  */
@@ -111,8 +136,14 @@ function deriveObservedEmbeddingIdentity(embedding, observedModel, observedVecto
111
136
  * direct `RemoteEmbedder`) so every embedder branch — remote, local,
112
137
  * deterministic, and test overrides via `_setEmbedderForTests` — is
113
138
  * exercised identically to the main embedding pass.
139
+ *
140
+ * `maxInputTokens` must be the SAME cap the main pass below applies via
141
+ * {@link capEmbeddingText} — the stored vector for each sampled entry was
142
+ * produced from its capped text, so comparing against a fresh vector of the
143
+ * uncapped text would compare unlike inputs for any entry over the cap
144
+ * (#955).
114
145
  */
115
- async function runEmbeddingCanary(db, config, signal) {
146
+ async function runEmbeddingCanary(db, config, signal, maxInputTokens) {
116
147
  const samples = sampleEmbeddedEntriesForCanary(db, CANARY_SAMPLE_SIZE);
117
148
  if (samples.length === 0) {
118
149
  return { outcome: "keep", verified: false, viaIdentityMatch: false };
@@ -121,7 +152,15 @@ async function runEmbeddingCanary(db, config, signal) {
121
152
  const skips = [];
122
153
  let canaryVectors;
123
154
  try {
124
- canaryVectors = await embedBatch(samples.map((sample) => sample.searchText), config.embedding, signal, (skip) => skips.push(skip), (_indices, _embeddings, model) => {
155
+ canaryVectors = await embedBatch(
156
+ // #955: the stored vector for each sample was produced from
157
+ // capEmbeddingText(searchText, maxInputTokens) — the main pass below
158
+ // caps every document before embedding it. The canary must re-embed
159
+ // the SAME capped text, or an entry over the cap compares a fresh
160
+ // vector of a different input against a stored vector of the capped
161
+ // one, and a genuine model match can read as a rebuild-worthy
162
+ // mismatch for reasons unrelated to the model.
163
+ samples.map((sample) => capEmbeddingText(sample.searchText, maxInputTokens).text), config.embedding, signal, (skip) => skips.push(skip), (_indices, _embeddings, model) => {
125
164
  if (model)
126
165
  observedModel = model;
127
166
  });
@@ -176,11 +215,47 @@ function throwIfAborted(signal) {
176
215
  }
177
216
  }
178
217
  export async function generateEmbeddingsForDb(db, config, onProgress, signal, entryIds, opts) {
218
+ // Drift guard (#954): refuse an ambient transaction. Every
219
+ // per-batch `db.transaction()` below is meant to be its own durable commit
220
+ // (#954) — inside an already-open outer transaction it would nest as an
221
+ // unobservable SAVEPOINT instead, so an interruption (competing-process
222
+ // collision, SIGKILL) could lose the whole pass rather than only the batch
223
+ // in flight. This is an internal contract error (a caller bug), not a
224
+ // user-facing failure class: callers with their own transaction (e.g. `akm
225
+ // bundle update`'s unified update transaction) must run the embedding
226
+ // phase on a separate connection AFTER their own transaction commits — see
227
+ // `runEmbeddingPass` in `src/indexer/indexer.ts`.
228
+ if (db.inTransaction) {
229
+ throw new Error("generateEmbeddingsForDb was called with an ambient transaction already open on `db`: per-batch commits " +
230
+ "would become SAVEPOINTs inside it, losing the crash-durability contract per-batch commit exists for. " +
231
+ "Run the embedding phase on a connection with no open transaction.");
232
+ }
179
233
  throwIfAborted(signal);
180
234
  if (config.semanticSearchMode === "off") {
235
+ // #955: salvage is self-emptying only if every path that skips reuse
236
+ // also drains it — otherwise a full rebuild performed with semantic
237
+ // search disabled leaves permanent orphaned rows behind (nothing will
238
+ // ever consume them, since this path never reaches the reuse step).
239
+ purgeEmbeddingSalvage(db);
181
240
  onProgress({ phase: "embeddings", message: "Semantic search disabled; skipping embeddings." });
182
241
  return { success: false, message: "Semantic search is disabled." };
183
242
  }
243
+ // #953 field gap: the actionable outcome is a self-diagnosing run, not a
244
+ // fix (every RemoteEmbedder path already resolves secret:// through one
245
+ // boundary — a keyless request can only mean embedding.apiKey was absent
246
+ // from the config THIS run loaded). One default-level line, before the
247
+ // first provider request of the phase (the canary probe or the main
248
+ // pass, whichever runs first below), naming the endpoint/model/credential
249
+ // SOURCE — never the credential value.
250
+ if (hasRemoteEndpoint(config.embedding ?? {})) {
251
+ const endpoint = normalizeEmbeddingEndpoint(config.embedding?.endpoint ?? "");
252
+ const credential = describeEmbeddingCredential(config.embedding?.apiKey);
253
+ const configFileSuffix = isVerbose() ? `; config: ${getConfigPath()}` : "";
254
+ onProgress({
255
+ phase: "embeddings",
256
+ message: `[embed] endpoint ${endpoint}, model ${config.embedding?.model ?? "unknown"}; credential: ${credential}${configFileSuffix}`,
257
+ });
258
+ }
184
259
  // A targeted call starts from an already-published generation. Preserve its
185
260
  // trust decision in O(1): successful writes for the changed IDs keep a
186
261
  // healthy fast path healthy, but can never promote a generation already
@@ -191,6 +266,10 @@ export async function generateEmbeddingsForDb(db, config, onProgress, signal, en
191
266
  let targetEntryIds = entryIds;
192
267
  /** Set only on an actual rebuild, so the up-front "Re-embedding N entries" line names why. */
193
268
  let rebuildReason;
269
+ // Resolved once and reused by both the canary (below) and the main pass's
270
+ // cap loop (further down) — the same cap must apply to both, or the canary
271
+ // compares a differently-capped text against the stored vector (#955).
272
+ const maxInputTokens = config.embedding?.maxInputTokens ?? DEFAULT_MAX_INPUT_TOKENS;
194
273
  if (opts?.forceReembed) {
195
274
  // `akm index --reembed`: an explicit operator override, skips the canary
196
275
  // entirely. The new fingerprint (and identity, now stale/unknown until
@@ -200,6 +279,9 @@ export async function generateEmbeddingsForDb(db, config, onProgress, signal, en
200
279
  // instead of purging again from zero (#955/#956).
201
280
  db.transaction(() => {
202
281
  purgeEmbeddings(db, { dropVecTable: true });
282
+ // #955: an explicit forced rebuild must re-embed everything, not
283
+ // quietly satisfy some of it from stale salvage.
284
+ purgeEmbeddingSalvage(db);
203
285
  deleteMeta(db, "embeddingDim");
204
286
  setMeta(db, "embeddingFingerprint", currentFingerprint);
205
287
  deleteMeta(db, "embeddingIdentity");
@@ -208,7 +290,7 @@ export async function generateEmbeddingsForDb(db, config, onProgress, signal, en
208
290
  rebuildReason = "forced by --reembed";
209
291
  }
210
292
  else if (storedFingerprint && storedFingerprint !== currentFingerprint) {
211
- const decision = await runEmbeddingCanary(db, config, signal);
293
+ const decision = await runEmbeddingCanary(db, config, signal, maxInputTokens);
212
294
  if (decision.outcome === "unverifiable") {
213
295
  // Destroying a good index because the server happens to be down right
214
296
  // now is worse than leaving a rename unverified until the next run —
@@ -221,6 +303,9 @@ export async function generateEmbeddingsForDb(db, config, onProgress, signal, en
221
303
  if (decision.outcome === "rebuild") {
222
304
  db.transaction(() => {
223
305
  purgeEmbeddings(db, { dropVecTable: true });
306
+ // #955: the stored vectors AND any leftover salvage both belong to
307
+ // a different model now — neither is reusable, so both go.
308
+ purgeEmbeddingSalvage(db);
224
309
  deleteMeta(db, "embeddingDim");
225
310
  setMeta(db, "embeddingFingerprint", currentFingerprint);
226
311
  if (decision.identity)
@@ -238,6 +323,11 @@ export async function generateEmbeddingsForDb(db, config, onProgress, signal, en
238
323
  // immediate write means a crash right after this decision does not
239
324
  // re-run the canary needlessly on the next attempt.
240
325
  setMeta(db, "embeddingFingerprint", currentFingerprint);
326
+ // #955: the model did not actually change, only the fingerprint
327
+ // STRING did (e.g. a gateway rename) — any leftover salvage rows
328
+ // tagged with the OLD string are still valid vectors. Relabel them so
329
+ // the reuse step below (and any later pass) can still find them.
330
+ relabelEmbeddingSalvageFingerprint(db, storedFingerprint, currentFingerprint);
241
331
  if (decision.identity)
242
332
  setMeta(db, "embeddingIdentity", decision.identity);
243
333
  if (decision.verified) {
@@ -253,24 +343,94 @@ export async function generateEmbeddingsForDb(db, config, onProgress, signal, en
253
343
  // lose or verify against — adopt the label silently, no purge line.
254
344
  }
255
345
  }
346
+ else {
347
+ // No rename to verify (either this is the very first
348
+ // pass ever for this db, or the fingerprint already matches the last
349
+ // successful one) — still record it NOW rather than deferring to a
350
+ // fully successful pass, mirroring the rebuild/keep branches above
351
+ // (#955/#956). Without this, an interrupted FIRST-EVER pass left
352
+ // `embeddingFingerprint` unset despite a per-batch commit below (#954)
353
+ // already having durably written real vectors — a later `akm index
354
+ // --full`'s salvage-before-discard step tags rows by this meta
355
+ // (`salvageEmbeddingsBeforeDiscard`) and treats an unset fingerprint as
356
+ // "nothing was ever verified", silently turning genuinely-embedded
357
+ // vectors into a full re-embed instead of a salvage-and-reuse.
358
+ setMeta(db, "embeddingFingerprint", currentFingerprint);
359
+ }
256
360
  try {
257
361
  throwIfAborted(signal);
258
362
  const allEntries = getAllEntriesForEmbedding(db, targetEntryIds);
259
- if (allEntries.length === 0) {
363
+ let vecFailedCount = 0;
364
+ let vecUnavailableCount = 0;
365
+ // #955: before any provider call, hand back vectors salvaged from a
366
+ // full rebuild or a generation bump for entries whose search_text is
367
+ // byte-identical to what was salvaged under the SAME fingerprint — a
368
+ // fingerprint mismatch or a single-byte content change both correctly
369
+ // fall through to the provider below instead.
370
+ const { reusedCount, remaining: candidateEntries } = reuseSalvagedEmbeddings(db, allEntries, currentFingerprint, (entry, embedding) => {
371
+ const result = upsertEmbedding(db, entry.id, embedding);
372
+ if (result.vec === "failed")
373
+ vecFailedCount++;
374
+ if (result.vec === "unavailable")
375
+ vecUnavailableCount++;
376
+ return result.stored;
377
+ });
378
+ if (reusedCount > 0) {
379
+ onProgress({
380
+ phase: "embeddings",
381
+ message: `Reused ${reusedCount} embedding${reusedCount === 1 ? "" : "s"} from the previous generation; embedding ${candidateEntries.length} new.`,
382
+ });
383
+ }
384
+ if (candidateEntries.length === 0) {
260
385
  onProgress({ phase: "embeddings", message: "Embeddings already up to date." });
261
386
  setMeta(db, "embeddingFingerprint", currentFingerprint);
262
- return { success: true };
387
+ if (reusedCount > 0) {
388
+ const vecGenerationComplete = targetEntryIds === undefined ? isVecFastPathComplete(db) : vecFastPathWasReady;
389
+ setVecFastPathReady(db, vecFailedCount === 0 && vecUnavailableCount === 0 && vecGenerationComplete);
390
+ }
391
+ // A pass that completes (even one that did nothing but reuse) purges
392
+ // whatever is left — salvage is consumed by the NEXT pass, never kept
393
+ // around as a second cache.
394
+ purgeEmbeddingSalvage(db);
395
+ return reusedCount > 0 ? { success: true, vecInsertFailures: vecFailedCount } : { success: true };
396
+ }
397
+ // Cap each document's embedded text at
398
+ // embedding.maxInputTokens (default DEFAULT_MAX_INPUT_TOKENS, resolved
399
+ // once above so the canary uses the identical cap) instead of ever
400
+ // failing a whole batch over one oversized entry — truncation keeps the
401
+ // head of the text, unicode-safe. A document is skipped only when its
402
+ // head is empty (the impossible case: nothing left to embed), never
403
+ // merely for being long.
404
+ let truncatedCount = 0;
405
+ const texts = [];
406
+ const pendingEntries = [];
407
+ for (const entry of candidateEntries) {
408
+ const capped = capEmbeddingText(entry.searchText, maxInputTokens);
409
+ if (capped.text.length === 0)
410
+ continue;
411
+ if (capped.truncated)
412
+ truncatedCount++;
413
+ pendingEntries.push(entry);
414
+ texts.push(capped.text);
415
+ }
416
+ if (truncatedCount > 0) {
417
+ // Through onProgress ONLY, not warn() too — onProgress already reaches
418
+ // stderr at the default level in every output mode (#954), and the
419
+ // index CLI's progress handler writes it through info() (log-file
420
+ // aware), so calling warn() as well printed the identical sentence
421
+ // twice in text mode.
422
+ const message = `[embed] ${truncatedCount} entr${truncatedCount === 1 ? "y" : "ies"} truncated to the ${maxInputTokens}-token embedding cap (embedding.maxInputTokens); rerun with a higher cap to embed the full text.`;
423
+ onProgress({ phase: "embeddings", message });
263
424
  }
264
425
  if (rebuildReason) {
265
- const message = `[embed] Re-embedding ${allEntries.length} entr${allEntries.length === 1 ? "y" : "ies"} because ${rebuildReason}`;
266
- warn(message);
426
+ // See the truncation notice above: onProgress ONLY.
427
+ const message = `[embed] Re-embedding ${pendingEntries.length} entr${pendingEntries.length === 1 ? "y" : "ies"} because ${rebuildReason}`;
267
428
  onProgress({ phase: "embeddings", message });
268
429
  }
269
430
  onProgress({
270
431
  phase: "embeddings",
271
- message: `Generating embeddings for ${allEntries.length} entr${allEntries.length === 1 ? "y" : "ies"}.`,
432
+ message: `Generating embeddings for ${pendingEntries.length} entr${pendingEntries.length === 1 ? "y" : "ies"}.`,
272
433
  });
273
- const texts = allEntries.map((entry) => entry.searchText);
274
434
  if (isVerbose()) {
275
435
  // Mirror RemoteEmbedder's actual token-bounded batching (#874) so this
276
436
  // log reflects the real request grouping rather than a fixed count of
@@ -279,7 +439,9 @@ export async function generateEmbeddingsForDb(db, config, onProgress, signal, en
279
439
  // for inference throughput only, never fails/skips), so there's
280
440
  // nothing meaningful to report per-batch for them.
281
441
  if (hasRemoteEndpoint(config.embedding ?? {})) {
282
- const tokenBudget = config.embedding?.maxTokens ?? config.embedding?.contextLength ?? DEFAULT_TOKEN_BUDGET;
442
+ // Mirrors RemoteEmbedder.embedBatch's own tokenBudget resolution
443
+ // (#956: contextLength no longer feeds this).
444
+ const tokenBudget = config.embedding?.maxTokens ?? DEFAULT_TOKEN_BUDGET;
283
445
  const maxCount = config.embedding?.batchSize ?? DEFAULT_REMOTE_BATCH_SIZE;
284
446
  const batches = buildTokenBoundedBatches(texts, tokenBudget, maxCount);
285
447
  const batchNumberByIndex = new Map();
@@ -287,7 +449,7 @@ export async function generateEmbeddingsForDb(db, config, onProgress, signal, en
287
449
  for (const i of batch.indices)
288
450
  batchNumberByIndex.set(i, batchIdx + 1);
289
451
  });
290
- for (const [i, entry] of allEntries.entries()) {
452
+ for (const [i, entry] of pendingEntries.entries()) {
291
453
  const chars = entry.searchText.length;
292
454
  const tokens = estimateTokenCount(entry.searchText);
293
455
  const batch = batches[batchNumberByIndex.get(i) - 1];
@@ -298,7 +460,7 @@ export async function generateEmbeddingsForDb(db, config, onProgress, signal, en
298
460
  }
299
461
  }
300
462
  else {
301
- for (const entry of allEntries) {
463
+ for (const entry of pendingEntries) {
302
464
  warnVerbose(`[embed] ${entry.itemRef} (${entry.searchText.length} chars, est. ${estimateTokenCount(entry.searchText)} tokens)`);
303
465
  }
304
466
  }
@@ -307,15 +469,12 @@ export async function generateEmbeddingsForDb(db, config, onProgress, signal, en
307
469
  let storedCount = 0;
308
470
  let skippedCount = 0;
309
471
  let embedFailedCount = 0;
310
- let vecFailedCount = 0;
311
- let vecUnavailableCount = 0;
312
472
  let storedTokens = 0;
313
- let lastProgressBucket = 0;
314
473
  try {
315
474
  heartbeatTimer = setInterval(() => {
316
475
  onProgress({
317
476
  phase: "embeddings",
318
- message: `Still generating embeddings: ${storedCount}/${allEntries.length} stored; waiting on embedding provider.`,
477
+ message: formatEmbeddingHeartbeat(storedCount, pendingEntries.length, embedFailedCount),
319
478
  });
320
479
  }, 15000);
321
480
  // A failing sub-batch or an oversized document is SKIPPED by embedBatch,
@@ -323,6 +482,42 @@ export async function generateEmbeddingsForDb(db, config, onProgress, signal, en
323
482
  // few bad documents don't discard every other entry's embedding.
324
483
  const skips = [];
325
484
  const embedStart = Date.now();
485
+ // Circuit breaker (#954): stop
486
+ // dispatching further batches once either consecutive-failure streak
487
+ // below reaches CIRCUIT_BREAKER_THRESHOLD — a dead/hung provider used
488
+ // to grind through every remaining batch for hours, one 30s (now
489
+ // configurable, and now backed off/retried/split first — see
490
+ // RemoteEmbedder.embedBatch) timeout at a time, ending in one
491
+ // aggregate warning and `ok: true`. Counted per BATCH
492
+ // (`skip.batchStart`), not per document: a single failed 100-document
493
+ // batch must not look like 100 consecutive failures.
494
+ let consecutiveSingleDocFailures = 0;
495
+ let consecutiveNetworkErrorFailures = 0;
496
+ let circuitBreakerReason;
497
+ const onSkip = (skip) => {
498
+ skips.push(skip);
499
+ if (!skip.batchStart)
500
+ return undefined;
501
+ if (skip.reason === "context-window-exceeded") {
502
+ consecutiveSingleDocFailures = 0;
503
+ consecutiveNetworkErrorFailures = 0;
504
+ return undefined;
505
+ }
506
+ // "batch-request-failed": a timeout only counts once retries have
507
+ // already narrowed it down to a single document (embedBatch backs
508
+ // off, retries, and splits a multi-document timeout before ever
509
+ // reporting it here); a network error counts immediately at any
510
+ // size — it was never retried, so it is trusted right away.
511
+ consecutiveSingleDocFailures = skip.batchSize === 1 ? consecutiveSingleDocFailures + 1 : 0;
512
+ consecutiveNetworkErrorFailures =
513
+ skip.failureKind === "network-error" ? consecutiveNetworkErrorFailures + 1 : 0;
514
+ if (consecutiveSingleDocFailures >= CIRCUIT_BREAKER_THRESHOLD ||
515
+ consecutiveNetworkErrorFailures >= CIRCUIT_BREAKER_THRESHOLD) {
516
+ circuitBreakerReason = skip.message;
517
+ return false;
518
+ }
519
+ return undefined;
520
+ };
326
521
  // Commit each provider batch in its own short transaction as it lands,
327
522
  // rather than buffering the whole run in memory for one transaction at
328
523
  // the very end (#954) — a competing-process lock error or any other
@@ -334,12 +529,56 @@ export async function generateEmbeddingsForDb(db, config, onProgress, signal, en
334
529
  // is kept; every batch from one run shares the same provider/model.
335
530
  let observedModel;
336
531
  let observedVectorLen;
337
- const onBatch = (indices, batchEmbeddings, model) => {
532
+ // Whether the remote provider's endpoint/model/token language is
533
+ // meaningful for this run — the per-batch diagnostic line below is
534
+ // remote-only, same gate the credential diagnostic (#953) above uses.
535
+ const reportPerBatchLine = hasRemoteEndpoint(config.embedding ?? {});
536
+ const onBatch = (indices, batchEmbeddings, model, outcome) => {
537
+ // #954 field-report follow-up: a "retrying" event carries nothing to
538
+ // commit — the request hasn't settled yet — only the notice that a
539
+ // back-off is about to be waited out, default-level so a run is
540
+ // never silently stalled indistinguishably from a hang.
541
+ if (outcome?.outcome === "retrying") {
542
+ if (reportPerBatchLine) {
543
+ onProgress({
544
+ phase: "embeddings",
545
+ message: `[embed] batch ${outcome.batchIndex}/${outcome.batchCount}: ${outcome.docCount} docs, ${outcome.requestTokens.toLocaleString()} tokens → retrying after ${(outcome.elapsedMs / 1000).toFixed(1)} s`,
546
+ });
547
+ }
548
+ return;
549
+ }
550
+ // #954: a "budget-lowered" event is the same kind of notice as
551
+ // "retrying" above — the run's first context-size rejection just
552
+ // shrank the request budget for everything not yet dispatched, but
553
+ // THIS rejected batch's own indices are still being split and
554
+ // retried by the embedder (their real stored/failed outcome lands in
555
+ // a later onBatch call). Nothing here has settled, so it must never
556
+ // touch storage, only report the notice — one line, at most once per
557
+ // run.
558
+ if (outcome?.outcome === "budget-lowered") {
559
+ if (reportPerBatchLine) {
560
+ onProgress({
561
+ phase: "embeddings",
562
+ message: `[embed] batch ${outcome.batchIndex}/${outcome.batchCount}: ${outcome.docCount} docs, ${outcome.requestTokens.toLocaleString()} tokens → ${outcome.reason}`,
563
+ });
564
+ }
565
+ return;
566
+ }
338
567
  if (model)
339
568
  observedModel = model;
569
+ // A batch that delivered at least one real embedding proves the
570
+ // provider is currently answering — reset both circuit-breaker
571
+ // streaks. (A wholly failed batch's `batchEmbeddings` are all
572
+ // `undefined`, per commitBatch's skip path, so this never
573
+ // re-triggers what onSkip just counted moments earlier.)
574
+ if (batchEmbeddings.some((embedding) => embedding !== undefined)) {
575
+ consecutiveSingleDocFailures = 0;
576
+ consecutiveNetworkErrorFailures = 0;
577
+ }
340
578
  db.transaction(() => {
341
579
  for (let k = 0; k < indices.length; k++) {
342
- const entry = allEntries[indices[k]];
580
+ const index = indices[k];
581
+ const entry = pendingEntries[index];
343
582
  if (!entry)
344
583
  continue;
345
584
  const embedding = batchEmbeddings[k];
@@ -352,7 +591,13 @@ export async function generateEmbeddingsForDb(db, config, onProgress, signal, en
352
591
  const result = upsertEmbedding(db, entry.id, embedding);
353
592
  if (result.stored) {
354
593
  storedCount++;
355
- storedTokens += estimateTokenCount(entry.searchText);
594
+ // #954: sum the estimate of the text actually sent —
595
+ // `texts[index]` is the capped string `embedBatch` was handed,
596
+ // parallel to `pendingEntries` by construction above (the
597
+ // `entry` guard covers both) — not `entry.searchText`, which is
598
+ // the pre-cap original and overstates throughput for every
599
+ // entry over the cap.
600
+ storedTokens += estimateTokenCount(texts[index]);
356
601
  }
357
602
  else {
358
603
  skippedCount++;
@@ -363,29 +608,36 @@ export async function generateEmbeddingsForDb(db, config, onProgress, signal, en
363
608
  vecUnavailableCount++;
364
609
  }
365
610
  })();
366
- const bucket = Math.floor(storedCount / PROGRESS_INTERVAL);
367
- if (bucket > lastProgressBucket) {
368
- lastProgressBucket = bucket;
611
+ // Default level, one line per provider batch (#954, field-report
612
+ // follow-up): oversized documents never made a request
613
+ // (`reason === "oversized"`), so there is no batch outcome to
614
+ // report — they are covered by the run's final oversized-skip
615
+ // count and list instead.
616
+ if (outcome && outcome.reason !== "oversized" && reportPerBatchLine) {
617
+ const elapsedSeconds = (outcome.elapsedMs / 1000).toFixed(1);
618
+ const outcomeLabel = outcome.outcome === "stored"
619
+ ? `${outcome.docCount} stored (${elapsedSeconds} s)`
620
+ : `failed: ${outcome.reason}`;
369
621
  onProgress({
370
622
  phase: "embeddings",
371
- message: `Embedded ${storedCount}/${allEntries.length} entries.`,
623
+ message: `[embed] batch ${outcome.batchIndex}/${outcome.batchCount}: ${outcome.docCount} docs, ${outcome.requestTokens.toLocaleString()} tokens → ${outcomeLabel}`,
372
624
  });
373
625
  }
626
+ // Every committed batch, not just every 500 stored entries (#954)
627
+ // — the prior bucketing left a non-verbose run silent
628
+ // for the entire embedding phase on anything smaller than 500
629
+ // entries, indistinguishable from a hang.
630
+ onProgress({
631
+ phase: "embeddings",
632
+ message: `Embedded ${storedCount}/${pendingEntries.length} entries.`,
633
+ });
374
634
  };
375
- await embedBatch(texts, config.embedding, signal, (skip) => skips.push(skip), onBatch);
635
+ await embedBatch(texts, config.embedding, signal, onSkip, onBatch);
376
636
  throwIfAborted(signal);
377
637
  const elapsedSeconds = Math.max((Date.now() - embedStart) / 1000, 0.001);
378
638
  if (skippedCount > 0) {
379
639
  warn(`[embed] ${skippedCount} embedding${skippedCount === 1 ? "" : "s"} skipped (entry deleted between queue and write)`);
380
640
  }
381
- if (embedFailedCount > 0) {
382
- const detail = skips
383
- .slice(0, 20)
384
- .map((skip) => ` - ${allEntries[skip.index]?.itemRef ?? skip.index} (${skip.reason}): ${skip.message}`)
385
- .join("\n");
386
- const more = skips.length > 20 ? `\n ...and ${skips.length - 20} more` : "";
387
- warn(`[embed] ${embedFailedCount} embedding${embedFailedCount === 1 ? "" : "s"} could not be generated and ${embedFailedCount === 1 ? "was" : "were"} skipped:\n${detail}${more}`);
388
- }
389
641
  const vecGenerationComplete = targetEntryIds === undefined ? isVecFastPathComplete(db) : vecFastPathWasReady;
390
642
  setVecFastPathReady(db, vecFailedCount === 0 && vecUnavailableCount === 0 && vecGenerationComplete);
391
643
  if (vecFailedCount > 0) {
@@ -395,14 +647,66 @@ export async function generateEmbeddingsForDb(db, config, onProgress, signal, en
395
647
  }
396
648
  const entriesPerSec = storedCount / elapsedSeconds;
397
649
  const tokensPerSec = storedTokens / elapsedSeconds;
650
+ const totalStored = storedCount + reusedCount;
651
+ // #954, field-report follow-up: the final line
652
+ // reports every outcome, not just what was stored — counts come from
653
+ // the same collected `skips` the circuit breaker already uses,
654
+ // categorized by `reason`/`failureKind`. "oversized skipped" =
655
+ // context-window-exceeded (never fit any request, at any size);
656
+ // "timed out" = a batch-request-failed skip whose last attempt timed
657
+ // out (retries/splits already exhausted before this counted); "failed"
658
+ // = every other batch-request-failed skip (a genuine, never-retried
659
+ // network/HTTP failure).
660
+ const oversizedSkips = skips.filter((skip) => skip.reason === "context-window-exceeded");
661
+ const timedOutSkips = skips.filter((skip) => skip.reason === "batch-request-failed" && skip.failureKind === "timeout");
662
+ const failedSkips = skips.filter((skip) => skip.reason === "batch-request-failed" && skip.failureKind !== "timeout");
663
+ const throughputLine = reusedCount > 0
664
+ ? // #955: report reused and newly-embedded counts separately — the
665
+ // rate figures below are provider throughput only (reuse is a
666
+ // plain DB write, not provider work) and would be misleadingly
667
+ // inflated if reused entries were folded into them.
668
+ `Stored ${totalStored} embedding${totalStored === 1 ? "" : "s"} (${reusedCount} reused, ${storedCount} newly embedded) in ${elapsedSeconds.toFixed(1)}s (${entriesPerSec.toFixed(1)} entries/s, ~${Math.round(tokensPerSec)} tokens/s)`
669
+ : `Stored ${storedCount} embedding${storedCount === 1 ? "" : "s"} in ${elapsedSeconds.toFixed(1)}s (${entriesPerSec.toFixed(1)} entries/s, ~${Math.round(tokensPerSec)} tokens/s)`;
398
670
  onProgress({
399
671
  phase: "embeddings",
400
- message: `Stored ${storedCount} embedding${storedCount === 1 ? "" : "s"} in ${elapsedSeconds.toFixed(1)}s (${entriesPerSec.toFixed(1)} entries/s, ~${Math.round(tokensPerSec)} tokens/s).`,
672
+ message: `${throughputLine}; ${oversizedSkips.length} oversized skipped, ${timedOutSkips.length} timed out, ${failedSkips.length} failed.`,
401
673
  });
674
+ // Bounded itemRef-level detail for every skip category, not just
675
+ // oversized — the aggregate counts above say HOW MANY documents timed
676
+ // out or failed, but give the operator no way to find out WHICH ones
677
+ // short of rerunning with --verbose and re-reading the whole log.
678
+ // Default level caps each list (there is nothing actionable about the
679
+ // 21st identical failure); --verbose prints every one, matching the
680
+ // per-document mapping lines' own verbosity gate above.
681
+ const printSkipList = (label, skipList) => {
682
+ if (skipList.length === 0)
683
+ return;
684
+ const limit = isVerbose() ? skipList.length : 20;
685
+ const listed = skipList
686
+ .slice(0, limit)
687
+ .map((skip) => ` - ${pendingEntries[skip.index]?.itemRef ?? skip.index}: ${skip.message}`)
688
+ .join("\n");
689
+ const more = skipList.length > limit ? `\n ...and ${skipList.length - limit} more` : "";
690
+ onProgress({ phase: "embeddings", message: `[embed] ${label} skipped:\n${listed}${more}` });
691
+ };
692
+ printSkipList("oversized documents", oversizedSkips);
693
+ printSkipList("timed-out documents", timedOutSkips);
694
+ printSkipList("failed documents", failedSkips);
402
695
  setMeta(db, "embeddingFingerprint", currentFingerprint);
403
696
  const observedIdentity = deriveObservedEmbeddingIdentity(config.embedding, observedModel, observedVectorLen);
404
697
  if (observedIdentity)
405
698
  setMeta(db, "embeddingIdentity", observedIdentity);
699
+ // Circuit breaker tripped (#954): committed batches are
700
+ // kept (nothing above discards them), but the pass is not a success —
701
+ // the provider looks dead, not just occasionally flaky.
702
+ if (circuitBreakerReason !== undefined) {
703
+ const message = `embedding provider failed ${CIRCUIT_BREAKER_THRESHOLD} consecutive batches ` +
704
+ `(last: ${circuitBreakerReason}); stopped after ${storedCount} embedding${storedCount === 1 ? "" : "s"} ` +
705
+ "were stored — rerun akm index when the endpoint is healthy";
706
+ warn(`[embed] ${message}`);
707
+ onProgress({ phase: "embeddings", message });
708
+ return { success: false, message, vecInsertFailures: vecFailedCount };
709
+ }
406
710
  // Only a total failure (nothing at all embedded, despite having entries
407
711
  // to embed) turns into a phase failure. Any partial success — the vast
408
712
  // majority of a large bundle embedding fine around a handful of skips —
@@ -416,6 +720,10 @@ export async function generateEmbeddingsForDb(db, config, onProgress, signal, en
416
720
  message: `All ${embedFailedCount} embedding batch(es) failed: ${firstMessage}`,
417
721
  };
418
722
  }
723
+ // A pass that completes without abort or circuit-break purges
724
+ // whatever salvage is left — consumed by this pass's reuse step
725
+ // above, or superseded by what it just embedded.
726
+ purgeEmbeddingSalvage(db);
419
727
  return { success: true, vecInsertFailures: vecFailedCount };
420
728
  }
421
729
  finally {
@@ -238,6 +238,7 @@ export async function ensureSourceCaches(config, options) {
238
238
  const cfg = config ?? loadConfig();
239
239
  const force = options?.force === true;
240
240
  const materialize = options?.materialize !== false;
241
+ const onProgress = options?.onProgress ?? (() => { });
241
242
  // Polymorphic refresh: walk every enabled source through its registered
242
243
  // provider and call `sync()`. Every cache-backed kind (git, website, npm)
243
244
  // refreshes the same way — a bad source warns and is skipped without
@@ -249,6 +250,12 @@ export async function ensureSourceCaches(config, options) {
249
250
  // content ENDS UP; the lock's `localRoot` merely records the result. So this
250
251
  // path correctly uses the provider, not the shared `lockContentRootFor`
251
252
  // resolver that reads/writes use to agree on where content already IS.
253
+ //
254
+ // Two passes: first resolve which sources will actually sync (constructing
255
+ // each provider once, exactly as before), so the progress count ("i/n")
256
+ // reflects real syncs rather than every configured entry including managed/
257
+ // unsyncable ones; then sync them in order, reporting progress per source.
258
+ const toSync = [];
252
259
  for (const entry of getSources(cfg)) {
253
260
  if (entry.enabled === false)
254
261
  continue;
@@ -282,12 +289,27 @@ export async function ensureSourceCaches(config, options) {
282
289
  warnIfSourceUnavailableForRead(entry, provider.name);
283
290
  continue;
284
291
  }
292
+ toSync.push(provider);
293
+ }
294
+ for (const [i, provider] of toSync.entries()) {
295
+ onProgress(`Hydrating source ${i + 1}/${toSync.length}: ${provider.name}`);
296
+ let heartbeat;
285
297
  try {
286
- await provider.sync({ force, secrets: options?.secrets, ensureWebsiteMirror });
298
+ heartbeat = setInterval(() => {
299
+ onProgress(`Still hydrating source ${i + 1}/${toSync.length}: ${provider.name}...`);
300
+ }, 15000);
301
+ // `toSync` only ever holds providers whose `.sync` passed the check
302
+ // above; the optional-chain here is just to satisfy the type (the
303
+ // narrowing does not survive the array round-trip).
304
+ await provider.sync?.({ force, secrets: options?.secrets, ensureWebsiteMirror });
287
305
  }
288
306
  catch (err) {
289
307
  warn(`Warning: failed to refresh ${provider.kind} source "${provider.name}": ${err instanceof Error ? err.message : String(err)}`);
290
308
  }
309
+ finally {
310
+ if (heartbeat)
311
+ clearInterval(heartbeat);
312
+ }
291
313
  }
292
314
  }
293
315
  /**