akm-cli 0.9.15-beta.1 → 0.9.15-beta.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,12 +1,14 @@
1
1
  // This Source Code Form is subject to the terms of the Mozilla Public
2
2
  // License, v. 2.0. If a copy of the MPL was not distributed with this
3
3
  // file, You can obtain one at https://mozilla.org/MPL/2.0/.
4
+ import { getConfigPath } from "../core/paths.js";
4
5
  import { isVerbose, warn, warnVerbose } from "../core/warn.js";
5
6
  import { embedBatch } from "../llm/embedder.js";
6
7
  import { DETERMINISTIC_EMBED_MODEL_ID, isDeterministicEmbedEnabled } from "../llm/embedders/deterministic.js";
7
8
  import { DEFAULT_LOCAL_MODEL } from "../llm/embedders/local.js";
8
- import { buildTokenBoundedBatches, DEFAULT_REMOTE_BATCH_SIZE, DEFAULT_TOKEN_BUDGET, estimateTokenCount, hasRemoteEndpoint, } from "../llm/embedders/remote.js";
9
+ import { buildTokenBoundedBatches, capEmbeddingText, DEFAULT_MAX_INPUT_TOKENS, DEFAULT_REMOTE_BATCH_SIZE, DEFAULT_TOKEN_BUDGET, describeEmbeddingCredential, estimateTokenCount, hasRemoteEndpoint, normalizeEmbeddingEndpoint, } from "../llm/embedders/remote.js";
9
10
  import { cosineSimilarity } from "../llm/embedders/types.js";
11
+ import { purgeEmbeddingSalvage, relabelEmbeddingSalvageFingerprint, reuseSalvagedEmbeddings, } from "../storage/repositories/embedding-salvage-repository.js";
10
12
  import { getEmbeddableEntryCount } from "../storage/repositories/index-entries-repository.js";
11
13
  import { deleteMeta, getMeta, setMeta } from "../storage/repositories/index-meta-repository.js";
12
14
  import { getAllEntriesForEmbedding, getEmbeddingCount, isVecFastPathComplete, isVecFastPathReady, purgeEmbeddings, sampleEmbeddedEntriesForCanary, setVecFastPathReady, upsertEmbedding, } from "../storage/repositories/index-vec-repository.js";
@@ -23,8 +25,14 @@ export function deriveSemanticProviderFingerprint(embedding) {
23
25
  }
24
26
  return `local:${embedding?.localModel ?? DEFAULT_LOCAL_MODEL}`;
25
27
  }
26
- /** How often (in stored entries) to emit a progress line during a large embedding run (#954). */
27
- const PROGRESS_INTERVAL = 500;
28
+ /**
29
+ * The heartbeat text emitted every 15s while a provider request is in
30
+ * flight, and by default (not `--verbose`-only) since silence indistinguishable
31
+ * from a hang was the field report's own symptom (#954).
32
+ */
33
+ export function formatEmbeddingHeartbeat(storedCount, total, failedCount) {
34
+ return `Still generating embeddings: ${storedCount}/${total} stored, ${failedCount} failed; waiting on embedding provider.`;
35
+ }
28
36
  /**
29
37
  * Number of already-embedded entries sampled for the fingerprint-rename
30
38
  * canary (#955) — small and cheap even against a slow local server; a
@@ -38,6 +46,23 @@ const CANARY_SAMPLE_SIZE = 8;
38
46
  * same-model rename rather than a real model change (#955).
39
47
  */
40
48
  const CANARY_SIMILARITY_THRESHOLD = 0.999;
49
+ /**
50
+ * Consecutive transport failures after which the embedding pass stops
51
+ * dispatching further requests and ends the run as a failure rather than
52
+ * grinding through every remaining batch against a dead endpoint (#954).
53
+ * Two independent trip conditions
54
+ * share this threshold — see `onSkip` below: 3 consecutive failures at
55
+ * single-document size (timeout OR network error — a multi-document
56
+ * timeout is not by itself evidence the endpoint is dead, since
57
+ * `RemoteEmbedder.embedBatch` already retries and splits it smaller before
58
+ * ever reporting it as failed at single-document size), or 3 consecutive
59
+ * network errors at ANY size (a network error is never retried, so it is
60
+ * trusted immediately regardless of how large the request was).
61
+ * `context-window-exceeded` never counts — that reason proves the provider
62
+ * IS reachable, and split-and-retry already handles it; it resets both
63
+ * streaks instead.
64
+ */
65
+ const CIRCUIT_BREAKER_THRESHOLD = 3;
41
66
  /**
42
67
  * Pure decision: do stored vectors remain valid against freshly re-embedded
43
68
  * canary samples? The ONE place that computes the canary's similarity
@@ -89,7 +114,7 @@ function medianOf(values) {
89
114
  * on what the server (or local model) actually reported plus the observed
90
115
  * vector width, so a gateway/transport change that keeps returning the same
91
116
  * underlying model can be told apart from a genuine model change without
92
- * relying on the operator's config string (#955 field-review addendum).
117
+ * relying on the operator's config string (#955).
93
118
  * Returns undefined when nothing was actually observed this call (no vector
94
119
  * to measure yet).
95
120
  */
@@ -176,11 +201,47 @@ function throwIfAborted(signal) {
176
201
  }
177
202
  }
178
203
  export async function generateEmbeddingsForDb(db, config, onProgress, signal, entryIds, opts) {
204
+ // Drift guard (#954): refuse an ambient transaction. Every
205
+ // per-batch `db.transaction()` below is meant to be its own durable commit
206
+ // (#954) — inside an already-open outer transaction it would nest as an
207
+ // unobservable SAVEPOINT instead, so an interruption (competing-process
208
+ // collision, SIGKILL) could lose the whole pass rather than only the batch
209
+ // in flight. This is an internal contract error (a caller bug), not a
210
+ // user-facing failure class: callers with their own transaction (e.g. `akm
211
+ // bundle update`'s unified update transaction) must run the embedding
212
+ // phase on a separate connection AFTER their own transaction commits — see
213
+ // `runEmbeddingPass` in `src/indexer/indexer.ts`.
214
+ if (db.inTransaction) {
215
+ throw new Error("generateEmbeddingsForDb was called with an ambient transaction already open on `db`: per-batch commits " +
216
+ "would become SAVEPOINTs inside it, losing the crash-durability contract per-batch commit exists for. " +
217
+ "Run the embedding phase on a connection with no open transaction.");
218
+ }
179
219
  throwIfAborted(signal);
180
220
  if (config.semanticSearchMode === "off") {
221
+ // #955: salvage is self-emptying only if every path that skips reuse
222
+ // also drains it — otherwise a full rebuild performed with semantic
223
+ // search disabled leaves permanent orphaned rows behind (nothing will
224
+ // ever consume them, since this path never reaches the reuse step).
225
+ purgeEmbeddingSalvage(db);
181
226
  onProgress({ phase: "embeddings", message: "Semantic search disabled; skipping embeddings." });
182
227
  return { success: false, message: "Semantic search is disabled." };
183
228
  }
229
+ // #953 field gap: the actionable outcome is a self-diagnosing run, not a
230
+ // fix (every RemoteEmbedder path already resolves secret:// through one
231
+ // boundary — a keyless request can only mean embedding.apiKey was absent
232
+ // from the config THIS run loaded). One default-level line, before the
233
+ // first provider request of the phase (the canary probe or the main
234
+ // pass, whichever runs first below), naming the endpoint/model/credential
235
+ // SOURCE — never the credential value.
236
+ if (hasRemoteEndpoint(config.embedding ?? {})) {
237
+ const endpoint = normalizeEmbeddingEndpoint(config.embedding?.endpoint ?? "");
238
+ const credential = describeEmbeddingCredential(config.embedding?.apiKey);
239
+ const configFileSuffix = isVerbose() ? `; config: ${getConfigPath()}` : "";
240
+ onProgress({
241
+ phase: "embeddings",
242
+ message: `[embed] endpoint ${endpoint}, model ${config.embedding?.model ?? "unknown"}; credential: ${credential}${configFileSuffix}`,
243
+ });
244
+ }
184
245
  // A targeted call starts from an already-published generation. Preserve its
185
246
  // trust decision in O(1): successful writes for the changed IDs keep a
186
247
  // healthy fast path healthy, but can never promote a generation already
@@ -200,6 +261,9 @@ export async function generateEmbeddingsForDb(db, config, onProgress, signal, en
200
261
  // instead of purging again from zero (#955/#956).
201
262
  db.transaction(() => {
202
263
  purgeEmbeddings(db, { dropVecTable: true });
264
+ // #955: an explicit forced rebuild must re-embed everything, not
265
+ // quietly satisfy some of it from stale salvage.
266
+ purgeEmbeddingSalvage(db);
203
267
  deleteMeta(db, "embeddingDim");
204
268
  setMeta(db, "embeddingFingerprint", currentFingerprint);
205
269
  deleteMeta(db, "embeddingIdentity");
@@ -221,6 +285,9 @@ export async function generateEmbeddingsForDb(db, config, onProgress, signal, en
221
285
  if (decision.outcome === "rebuild") {
222
286
  db.transaction(() => {
223
287
  purgeEmbeddings(db, { dropVecTable: true });
288
+ // #955: the stored vectors AND any leftover salvage both belong to
289
+ // a different model now — neither is reusable, so both go.
290
+ purgeEmbeddingSalvage(db);
224
291
  deleteMeta(db, "embeddingDim");
225
292
  setMeta(db, "embeddingFingerprint", currentFingerprint);
226
293
  if (decision.identity)
@@ -238,6 +305,11 @@ export async function generateEmbeddingsForDb(db, config, onProgress, signal, en
238
305
  // immediate write means a crash right after this decision does not
239
306
  // re-run the canary needlessly on the next attempt.
240
307
  setMeta(db, "embeddingFingerprint", currentFingerprint);
308
+ // #955: the model did not actually change, only the fingerprint
309
+ // STRING did (e.g. a gateway rename) — any leftover salvage rows
310
+ // tagged with the OLD string are still valid vectors. Relabel them so
311
+ // the reuse step below (and any later pass) can still find them.
312
+ relabelEmbeddingSalvageFingerprint(db, storedFingerprint, currentFingerprint);
241
313
  if (decision.identity)
242
314
  setMeta(db, "embeddingIdentity", decision.identity);
243
315
  if (decision.verified) {
@@ -253,24 +325,94 @@ export async function generateEmbeddingsForDb(db, config, onProgress, signal, en
253
325
  // lose or verify against — adopt the label silently, no purge line.
254
326
  }
255
327
  }
328
+ else {
329
+ // No rename to verify (either this is the very first
330
+ // pass ever for this db, or the fingerprint already matches the last
331
+ // successful one) — still record it NOW rather than deferring to a
332
+ // fully successful pass, mirroring the rebuild/keep branches above
333
+ // (#955/#956). Without this, an interrupted FIRST-EVER pass left
334
+ // `embeddingFingerprint` unset despite a per-batch commit below (#954)
335
+ // already having durably written real vectors — a later `akm index
336
+ // --full`'s salvage-before-discard step tags rows by this meta
337
+ // (`salvageEmbeddingsBeforeDiscard`) and treats an unset fingerprint as
338
+ // "nothing was ever verified", silently turning genuinely-embedded
339
+ // vectors into a full re-embed instead of a salvage-and-reuse.
340
+ setMeta(db, "embeddingFingerprint", currentFingerprint);
341
+ }
256
342
  try {
257
343
  throwIfAborted(signal);
258
344
  const allEntries = getAllEntriesForEmbedding(db, targetEntryIds);
259
- if (allEntries.length === 0) {
345
+ let vecFailedCount = 0;
346
+ let vecUnavailableCount = 0;
347
+ // #955: before any provider call, hand back vectors salvaged from a
348
+ // full rebuild or a generation bump for entries whose search_text is
349
+ // byte-identical to what was salvaged under the SAME fingerprint — a
350
+ // fingerprint mismatch or a single-byte content change both correctly
351
+ // fall through to the provider below instead.
352
+ const { reusedCount, remaining: candidateEntries } = reuseSalvagedEmbeddings(db, allEntries, currentFingerprint, (entry, embedding) => {
353
+ const result = upsertEmbedding(db, entry.id, embedding);
354
+ if (result.vec === "failed")
355
+ vecFailedCount++;
356
+ if (result.vec === "unavailable")
357
+ vecUnavailableCount++;
358
+ return result.stored;
359
+ });
360
+ if (reusedCount > 0) {
361
+ onProgress({
362
+ phase: "embeddings",
363
+ message: `Reused ${reusedCount} embedding${reusedCount === 1 ? "" : "s"} from the previous generation; embedding ${candidateEntries.length} new.`,
364
+ });
365
+ }
366
+ if (candidateEntries.length === 0) {
260
367
  onProgress({ phase: "embeddings", message: "Embeddings already up to date." });
261
368
  setMeta(db, "embeddingFingerprint", currentFingerprint);
262
- return { success: true };
369
+ if (reusedCount > 0) {
370
+ const vecGenerationComplete = targetEntryIds === undefined ? isVecFastPathComplete(db) : vecFastPathWasReady;
371
+ setVecFastPathReady(db, vecFailedCount === 0 && vecUnavailableCount === 0 && vecGenerationComplete);
372
+ }
373
+ // A pass that completes (even one that did nothing but reuse) purges
374
+ // whatever is left — salvage is consumed by the NEXT pass, never kept
375
+ // around as a second cache.
376
+ purgeEmbeddingSalvage(db);
377
+ return reusedCount > 0 ? { success: true, vecInsertFailures: vecFailedCount } : { success: true };
378
+ }
379
+ // Cap each document's embedded text at
380
+ // embedding.maxInputTokens (default DEFAULT_MAX_INPUT_TOKENS) instead of
381
+ // ever failing a whole batch over one oversized entry — truncation keeps
382
+ // the head of the text, unicode-safe. A document is skipped only when its
383
+ // head is empty (the impossible case: nothing left to embed), never
384
+ // merely for being long.
385
+ const maxInputTokens = config.embedding?.maxInputTokens ?? DEFAULT_MAX_INPUT_TOKENS;
386
+ let truncatedCount = 0;
387
+ const texts = [];
388
+ const pendingEntries = [];
389
+ for (const entry of candidateEntries) {
390
+ const capped = capEmbeddingText(entry.searchText, maxInputTokens);
391
+ if (capped.text.length === 0)
392
+ continue;
393
+ if (capped.truncated)
394
+ truncatedCount++;
395
+ pendingEntries.push(entry);
396
+ texts.push(capped.text);
397
+ }
398
+ if (truncatedCount > 0) {
399
+ // Through onProgress ONLY, not warn() too — onProgress already reaches
400
+ // stderr at the default level in every output mode (#954), and the
401
+ // index CLI's progress handler writes it through info() (log-file
402
+ // aware), so calling warn() as well printed the identical sentence
403
+ // twice in text mode.
404
+ const message = `[embed] ${truncatedCount} entr${truncatedCount === 1 ? "y" : "ies"} truncated to the ${maxInputTokens}-token embedding cap (embedding.maxInputTokens); rerun with a higher cap to embed the full text.`;
405
+ onProgress({ phase: "embeddings", message });
263
406
  }
264
407
  if (rebuildReason) {
265
- const message = `[embed] Re-embedding ${allEntries.length} entr${allEntries.length === 1 ? "y" : "ies"} because ${rebuildReason}`;
266
- warn(message);
408
+ // See the truncation notice above: onProgress ONLY.
409
+ const message = `[embed] Re-embedding ${pendingEntries.length} entr${pendingEntries.length === 1 ? "y" : "ies"} because ${rebuildReason}`;
267
410
  onProgress({ phase: "embeddings", message });
268
411
  }
269
412
  onProgress({
270
413
  phase: "embeddings",
271
- message: `Generating embeddings for ${allEntries.length} entr${allEntries.length === 1 ? "y" : "ies"}.`,
414
+ message: `Generating embeddings for ${pendingEntries.length} entr${pendingEntries.length === 1 ? "y" : "ies"}.`,
272
415
  });
273
- const texts = allEntries.map((entry) => entry.searchText);
274
416
  if (isVerbose()) {
275
417
  // Mirror RemoteEmbedder's actual token-bounded batching (#874) so this
276
418
  // log reflects the real request grouping rather than a fixed count of
@@ -279,7 +421,9 @@ export async function generateEmbeddingsForDb(db, config, onProgress, signal, en
279
421
  // for inference throughput only, never fails/skips), so there's
280
422
  // nothing meaningful to report per-batch for them.
281
423
  if (hasRemoteEndpoint(config.embedding ?? {})) {
282
- const tokenBudget = config.embedding?.maxTokens ?? config.embedding?.contextLength ?? DEFAULT_TOKEN_BUDGET;
424
+ // Mirrors RemoteEmbedder.embedBatch's own tokenBudget resolution
425
+ // (#956: contextLength no longer feeds this).
426
+ const tokenBudget = config.embedding?.maxTokens ?? DEFAULT_TOKEN_BUDGET;
283
427
  const maxCount = config.embedding?.batchSize ?? DEFAULT_REMOTE_BATCH_SIZE;
284
428
  const batches = buildTokenBoundedBatches(texts, tokenBudget, maxCount);
285
429
  const batchNumberByIndex = new Map();
@@ -287,7 +431,7 @@ export async function generateEmbeddingsForDb(db, config, onProgress, signal, en
287
431
  for (const i of batch.indices)
288
432
  batchNumberByIndex.set(i, batchIdx + 1);
289
433
  });
290
- for (const [i, entry] of allEntries.entries()) {
434
+ for (const [i, entry] of pendingEntries.entries()) {
291
435
  const chars = entry.searchText.length;
292
436
  const tokens = estimateTokenCount(entry.searchText);
293
437
  const batch = batches[batchNumberByIndex.get(i) - 1];
@@ -298,7 +442,7 @@ export async function generateEmbeddingsForDb(db, config, onProgress, signal, en
298
442
  }
299
443
  }
300
444
  else {
301
- for (const entry of allEntries) {
445
+ for (const entry of pendingEntries) {
302
446
  warnVerbose(`[embed] ${entry.itemRef} (${entry.searchText.length} chars, est. ${estimateTokenCount(entry.searchText)} tokens)`);
303
447
  }
304
448
  }
@@ -307,15 +451,12 @@ export async function generateEmbeddingsForDb(db, config, onProgress, signal, en
307
451
  let storedCount = 0;
308
452
  let skippedCount = 0;
309
453
  let embedFailedCount = 0;
310
- let vecFailedCount = 0;
311
- let vecUnavailableCount = 0;
312
454
  let storedTokens = 0;
313
- let lastProgressBucket = 0;
314
455
  try {
315
456
  heartbeatTimer = setInterval(() => {
316
457
  onProgress({
317
458
  phase: "embeddings",
318
- message: `Still generating embeddings: ${storedCount}/${allEntries.length} stored; waiting on embedding provider.`,
459
+ message: formatEmbeddingHeartbeat(storedCount, pendingEntries.length, embedFailedCount),
319
460
  });
320
461
  }, 15000);
321
462
  // A failing sub-batch or an oversized document is SKIPPED by embedBatch,
@@ -323,6 +464,42 @@ export async function generateEmbeddingsForDb(db, config, onProgress, signal, en
323
464
  // few bad documents don't discard every other entry's embedding.
324
465
  const skips = [];
325
466
  const embedStart = Date.now();
467
+ // Circuit breaker (#954): stop
468
+ // dispatching further batches once either consecutive-failure streak
469
+ // below reaches CIRCUIT_BREAKER_THRESHOLD — a dead/hung provider used
470
+ // to grind through every remaining batch for hours, one 30s (now
471
+ // configurable, and now backed off/retried/split first — see
472
+ // RemoteEmbedder.embedBatch) timeout at a time, ending in one
473
+ // aggregate warning and `ok: true`. Counted per BATCH
474
+ // (`skip.batchStart`), not per document: a single failed 100-document
475
+ // batch must not look like 100 consecutive failures.
476
+ let consecutiveSingleDocFailures = 0;
477
+ let consecutiveNetworkErrorFailures = 0;
478
+ let circuitBreakerReason;
479
+ const onSkip = (skip) => {
480
+ skips.push(skip);
481
+ if (!skip.batchStart)
482
+ return undefined;
483
+ if (skip.reason === "context-window-exceeded") {
484
+ consecutiveSingleDocFailures = 0;
485
+ consecutiveNetworkErrorFailures = 0;
486
+ return undefined;
487
+ }
488
+ // "batch-request-failed": a timeout only counts once retries have
489
+ // already narrowed it down to a single document (embedBatch backs
490
+ // off, retries, and splits a multi-document timeout before ever
491
+ // reporting it here); a network error counts immediately at any
492
+ // size — it was never retried, so it is trusted right away.
493
+ consecutiveSingleDocFailures = skip.batchSize === 1 ? consecutiveSingleDocFailures + 1 : 0;
494
+ consecutiveNetworkErrorFailures =
495
+ skip.failureKind === "network-error" ? consecutiveNetworkErrorFailures + 1 : 0;
496
+ if (consecutiveSingleDocFailures >= CIRCUIT_BREAKER_THRESHOLD ||
497
+ consecutiveNetworkErrorFailures >= CIRCUIT_BREAKER_THRESHOLD) {
498
+ circuitBreakerReason = skip.message;
499
+ return false;
500
+ }
501
+ return undefined;
502
+ };
326
503
  // Commit each provider batch in its own short transaction as it lands,
327
504
  // rather than buffering the whole run in memory for one transaction at
328
505
  // the very end (#954) — a competing-process lock error or any other
@@ -334,12 +511,38 @@ export async function generateEmbeddingsForDb(db, config, onProgress, signal, en
334
511
  // is kept; every batch from one run shares the same provider/model.
335
512
  let observedModel;
336
513
  let observedVectorLen;
337
- const onBatch = (indices, batchEmbeddings, model) => {
514
+ // Whether the remote provider's endpoint/model/token language is
515
+ // meaningful for this run — the per-batch diagnostic line below is
516
+ // remote-only, same gate the credential diagnostic (#953) above uses.
517
+ const reportPerBatchLine = hasRemoteEndpoint(config.embedding ?? {});
518
+ const onBatch = (indices, batchEmbeddings, model, outcome) => {
519
+ // #954 field-report follow-up: a "retrying" event carries nothing to
520
+ // commit — the request hasn't settled yet — only the notice that a
521
+ // back-off is about to be waited out, default-level so a run is
522
+ // never silently stalled indistinguishably from a hang.
523
+ if (outcome?.outcome === "retrying") {
524
+ if (reportPerBatchLine) {
525
+ onProgress({
526
+ phase: "embeddings",
527
+ message: `[embed] batch ${outcome.batchIndex}/${outcome.batchCount}: ${outcome.docCount} docs, ${outcome.requestTokens.toLocaleString()} tokens → retrying after ${(outcome.elapsedMs / 1000).toFixed(1)} s`,
528
+ });
529
+ }
530
+ return;
531
+ }
338
532
  if (model)
339
533
  observedModel = model;
534
+ // A batch that delivered at least one real embedding proves the
535
+ // provider is currently answering — reset both circuit-breaker
536
+ // streaks. (A wholly failed batch's `batchEmbeddings` are all
537
+ // `undefined`, per commitBatch's skip path, so this never
538
+ // re-triggers what onSkip just counted moments earlier.)
539
+ if (batchEmbeddings.some((embedding) => embedding !== undefined)) {
540
+ consecutiveSingleDocFailures = 0;
541
+ consecutiveNetworkErrorFailures = 0;
542
+ }
340
543
  db.transaction(() => {
341
544
  for (let k = 0; k < indices.length; k++) {
342
- const entry = allEntries[indices[k]];
545
+ const entry = pendingEntries[indices[k]];
343
546
  if (!entry)
344
547
  continue;
345
548
  const embedding = batchEmbeddings[k];
@@ -363,29 +566,36 @@ export async function generateEmbeddingsForDb(db, config, onProgress, signal, en
363
566
  vecUnavailableCount++;
364
567
  }
365
568
  })();
366
- const bucket = Math.floor(storedCount / PROGRESS_INTERVAL);
367
- if (bucket > lastProgressBucket) {
368
- lastProgressBucket = bucket;
569
+ // Default level, one line per provider batch (#954, field-report
570
+ // follow-up): oversized documents never made a request
571
+ // (`reason === "oversized"`), so there is no batch outcome to
572
+ // report — they are covered by the run's final oversized-skip
573
+ // count and list instead.
574
+ if (outcome && outcome.reason !== "oversized" && reportPerBatchLine) {
575
+ const elapsedSeconds = (outcome.elapsedMs / 1000).toFixed(1);
576
+ const outcomeLabel = outcome.outcome === "stored"
577
+ ? `${outcome.docCount} stored (${elapsedSeconds} s)`
578
+ : `failed: ${outcome.reason}`;
369
579
  onProgress({
370
580
  phase: "embeddings",
371
- message: `Embedded ${storedCount}/${allEntries.length} entries.`,
581
+ message: `[embed] batch ${outcome.batchIndex}/${outcome.batchCount}: ${outcome.docCount} docs, ${outcome.requestTokens.toLocaleString()} tokens → ${outcomeLabel}`,
372
582
  });
373
583
  }
584
+ // Every committed batch, not just every 500 stored entries (#954)
585
+ // — the prior bucketing left a non-verbose run silent
586
+ // for the entire embedding phase on anything smaller than 500
587
+ // entries, indistinguishable from a hang.
588
+ onProgress({
589
+ phase: "embeddings",
590
+ message: `Embedded ${storedCount}/${pendingEntries.length} entries.`,
591
+ });
374
592
  };
375
- await embedBatch(texts, config.embedding, signal, (skip) => skips.push(skip), onBatch);
593
+ await embedBatch(texts, config.embedding, signal, onSkip, onBatch);
376
594
  throwIfAborted(signal);
377
595
  const elapsedSeconds = Math.max((Date.now() - embedStart) / 1000, 0.001);
378
596
  if (skippedCount > 0) {
379
597
  warn(`[embed] ${skippedCount} embedding${skippedCount === 1 ? "" : "s"} skipped (entry deleted between queue and write)`);
380
598
  }
381
- if (embedFailedCount > 0) {
382
- const detail = skips
383
- .slice(0, 20)
384
- .map((skip) => ` - ${allEntries[skip.index]?.itemRef ?? skip.index} (${skip.reason}): ${skip.message}`)
385
- .join("\n");
386
- const more = skips.length > 20 ? `\n ...and ${skips.length - 20} more` : "";
387
- warn(`[embed] ${embedFailedCount} embedding${embedFailedCount === 1 ? "" : "s"} could not be generated and ${embedFailedCount === 1 ? "was" : "were"} skipped:\n${detail}${more}`);
388
- }
389
599
  const vecGenerationComplete = targetEntryIds === undefined ? isVecFastPathComplete(db) : vecFastPathWasReady;
390
600
  setVecFastPathReady(db, vecFailedCount === 0 && vecUnavailableCount === 0 && vecGenerationComplete);
391
601
  if (vecFailedCount > 0) {
@@ -395,14 +605,66 @@ export async function generateEmbeddingsForDb(db, config, onProgress, signal, en
395
605
  }
396
606
  const entriesPerSec = storedCount / elapsedSeconds;
397
607
  const tokensPerSec = storedTokens / elapsedSeconds;
608
+ const totalStored = storedCount + reusedCount;
609
+ // #954, field-report follow-up: the final line
610
+ // reports every outcome, not just what was stored — counts come from
611
+ // the same collected `skips` the circuit breaker already uses,
612
+ // categorized by `reason`/`failureKind`. "oversized skipped" =
613
+ // context-window-exceeded (never fit any request, at any size);
614
+ // "timed out" = a batch-request-failed skip whose last attempt timed
615
+ // out (retries/splits already exhausted before this counted); "failed"
616
+ // = every other batch-request-failed skip (a genuine, never-retried
617
+ // network/HTTP failure).
618
+ const oversizedSkips = skips.filter((skip) => skip.reason === "context-window-exceeded");
619
+ const timedOutSkips = skips.filter((skip) => skip.reason === "batch-request-failed" && skip.failureKind === "timeout");
620
+ const failedSkips = skips.filter((skip) => skip.reason === "batch-request-failed" && skip.failureKind !== "timeout");
621
+ const throughputLine = reusedCount > 0
622
+ ? // #955: report reused and newly-embedded counts separately — the
623
+ // rate figures below are provider throughput only (reuse is a
624
+ // plain DB write, not provider work) and would be misleadingly
625
+ // inflated if reused entries were folded into them.
626
+ `Stored ${totalStored} embedding${totalStored === 1 ? "" : "s"} (${reusedCount} reused, ${storedCount} newly embedded) in ${elapsedSeconds.toFixed(1)}s (${entriesPerSec.toFixed(1)} entries/s, ~${Math.round(tokensPerSec)} tokens/s)`
627
+ : `Stored ${storedCount} embedding${storedCount === 1 ? "" : "s"} in ${elapsedSeconds.toFixed(1)}s (${entriesPerSec.toFixed(1)} entries/s, ~${Math.round(tokensPerSec)} tokens/s)`;
398
628
  onProgress({
399
629
  phase: "embeddings",
400
- message: `Stored ${storedCount} embedding${storedCount === 1 ? "" : "s"} in ${elapsedSeconds.toFixed(1)}s (${entriesPerSec.toFixed(1)} entries/s, ~${Math.round(tokensPerSec)} tokens/s).`,
630
+ message: `${throughputLine}; ${oversizedSkips.length} oversized skipped, ${timedOutSkips.length} timed out, ${failedSkips.length} failed.`,
401
631
  });
632
+ // Bounded itemRef-level detail for every skip category, not just
633
+ // oversized — the aggregate counts above say HOW MANY documents timed
634
+ // out or failed, but give the operator no way to find out WHICH ones
635
+ // short of rerunning with --verbose and re-reading the whole log.
636
+ // Default level caps each list (there is nothing actionable about the
637
+ // 21st identical failure); --verbose prints every one, matching the
638
+ // per-document mapping lines' own verbosity gate above.
639
+ const printSkipList = (label, skipList) => {
640
+ if (skipList.length === 0)
641
+ return;
642
+ const limit = isVerbose() ? skipList.length : 20;
643
+ const listed = skipList
644
+ .slice(0, limit)
645
+ .map((skip) => ` - ${pendingEntries[skip.index]?.itemRef ?? skip.index}: ${skip.message}`)
646
+ .join("\n");
647
+ const more = skipList.length > limit ? `\n ...and ${skipList.length - limit} more` : "";
648
+ onProgress({ phase: "embeddings", message: `[embed] ${label} skipped:\n${listed}${more}` });
649
+ };
650
+ printSkipList("oversized documents", oversizedSkips);
651
+ printSkipList("timed-out documents", timedOutSkips);
652
+ printSkipList("failed documents", failedSkips);
402
653
  setMeta(db, "embeddingFingerprint", currentFingerprint);
403
654
  const observedIdentity = deriveObservedEmbeddingIdentity(config.embedding, observedModel, observedVectorLen);
404
655
  if (observedIdentity)
405
656
  setMeta(db, "embeddingIdentity", observedIdentity);
657
+ // Circuit breaker tripped (#954): committed batches are
658
+ // kept (nothing above discards them), but the pass is not a success —
659
+ // the provider looks dead, not just occasionally flaky.
660
+ if (circuitBreakerReason !== undefined) {
661
+ const message = `embedding provider failed ${CIRCUIT_BREAKER_THRESHOLD} consecutive batches ` +
662
+ `(last: ${circuitBreakerReason}); stopped after ${storedCount} embedding${storedCount === 1 ? "" : "s"} ` +
663
+ "were stored — rerun akm index when the endpoint is healthy";
664
+ warn(`[embed] ${message}`);
665
+ onProgress({ phase: "embeddings", message });
666
+ return { success: false, message, vecInsertFailures: vecFailedCount };
667
+ }
406
668
  // Only a total failure (nothing at all embedded, despite having entries
407
669
  // to embed) turns into a phase failure. Any partial success — the vast
408
670
  // majority of a large bundle embedding fine around a handful of skips —
@@ -416,6 +678,10 @@ export async function generateEmbeddingsForDb(db, config, onProgress, signal, en
416
678
  message: `All ${embedFailedCount} embedding batch(es) failed: ${firstMessage}`,
417
679
  };
418
680
  }
681
+ // A pass that completes without abort or circuit-break purges
682
+ // whatever salvage is left — consumed by this pass's reuse step
683
+ // above, or superseded by what it just embedded.
684
+ purgeEmbeddingSalvage(db);
419
685
  return { success: true, vecInsertFailures: vecFailedCount };
420
686
  }
421
687
  finally {
@@ -238,6 +238,7 @@ export async function ensureSourceCaches(config, options) {
238
238
  const cfg = config ?? loadConfig();
239
239
  const force = options?.force === true;
240
240
  const materialize = options?.materialize !== false;
241
+ const onProgress = options?.onProgress ?? (() => { });
241
242
  // Polymorphic refresh: walk every enabled source through its registered
242
243
  // provider and call `sync()`. Every cache-backed kind (git, website, npm)
243
244
  // refreshes the same way — a bad source warns and is skipped without
@@ -249,6 +250,12 @@ export async function ensureSourceCaches(config, options) {
249
250
  // content ENDS UP; the lock's `localRoot` merely records the result. So this
250
251
  // path correctly uses the provider, not the shared `lockContentRootFor`
251
252
  // resolver that reads/writes use to agree on where content already IS.
253
+ //
254
+ // Two passes: first resolve which sources will actually sync (constructing
255
+ // each provider once, exactly as before), so the progress count ("i/n")
256
+ // reflects real syncs rather than every configured entry including managed/
257
+ // unsyncable ones; then sync them in order, reporting progress per source.
258
+ const toSync = [];
252
259
  for (const entry of getSources(cfg)) {
253
260
  if (entry.enabled === false)
254
261
  continue;
@@ -282,12 +289,27 @@ export async function ensureSourceCaches(config, options) {
282
289
  warnIfSourceUnavailableForRead(entry, provider.name);
283
290
  continue;
284
291
  }
292
+ toSync.push(provider);
293
+ }
294
+ for (const [i, provider] of toSync.entries()) {
295
+ onProgress(`Hydrating source ${i + 1}/${toSync.length}: ${provider.name}`);
296
+ let heartbeat;
285
297
  try {
286
- await provider.sync({ force, secrets: options?.secrets, ensureWebsiteMirror });
298
+ heartbeat = setInterval(() => {
299
+ onProgress(`Still hydrating source ${i + 1}/${toSync.length}: ${provider.name}...`);
300
+ }, 15000);
301
+ // `toSync` only ever holds providers whose `.sync` passed the check
302
+ // above; the optional-chain here is just to satisfy the type (the
303
+ // narrowing does not survive the array round-trip).
304
+ await provider.sync?.({ force, secrets: options?.secrets, ensureWebsiteMirror });
287
305
  }
288
306
  catch (err) {
289
307
  warn(`Warning: failed to refresh ${provider.kind} source "${provider.name}": ${err instanceof Error ? err.message : String(err)}`);
290
308
  }
309
+ finally {
310
+ if (heartbeat)
311
+ clearInterval(heartbeat);
312
+ }
291
313
  }
292
314
  }
293
315
  /**