akm-cli 0.9.15-beta.1 → 0.9.15-beta.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +175 -13
- package/dist/akm +54 -1
- package/dist/akm-migrate +34 -1
- package/dist/cli.js +37 -1
- package/dist/commands/improve/locks.js +3 -2
- package/dist/commands/sources/installed-stashes.js +58 -16
- package/dist/commands/sources/stash-cli.js +17 -0
- package/dist/core/config/schema/embedding.js +41 -0
- package/dist/core/file-lock.js +49 -15
- package/dist/core/parent-watchdog.js +64 -0
- package/dist/core/run-lock.js +13 -2
- package/dist/indexer/index-rebuild-lock.js +4 -4
- package/dist/indexer/index-written-assets.js +9 -1
- package/dist/indexer/indexer.js +79 -16
- package/dist/indexer/materialize-embeddings.js +299 -33
- package/dist/indexer/search/search-source.js +23 -1
- package/dist/llm/embedders/remote.js +340 -42
- package/dist/scripts/akm-migrate-node.js +398 -77
- package/dist/scripts/akm-migrate.js +398 -77
- package/dist/storage/repositories/embedding-salvage-repository.js +184 -0
- package/dist/storage/repositories/index-schema.js +16 -0
- package/dist/tasks/run/run-native-task.js +23 -1
- package/docs/migration/release-notes/0.9.15.md +85 -4
- package/docs/migration/release-notes/README.md +3 -2
- package/docs/reference/cli.md +28 -3
- package/docs/reference/configuration.md +67 -15
- package/package.json +1 -1
- package/schemas/akm-config.json +39 -0
|
@@ -1,12 +1,14 @@
|
|
|
1
1
|
// This Source Code Form is subject to the terms of the Mozilla Public
|
|
2
2
|
// License, v. 2.0. If a copy of the MPL was not distributed with this
|
|
3
3
|
// file, You can obtain one at https://mozilla.org/MPL/2.0/.
|
|
4
|
+
import { getConfigPath } from "../core/paths.js";
|
|
4
5
|
import { isVerbose, warn, warnVerbose } from "../core/warn.js";
|
|
5
6
|
import { embedBatch } from "../llm/embedder.js";
|
|
6
7
|
import { DETERMINISTIC_EMBED_MODEL_ID, isDeterministicEmbedEnabled } from "../llm/embedders/deterministic.js";
|
|
7
8
|
import { DEFAULT_LOCAL_MODEL } from "../llm/embedders/local.js";
|
|
8
|
-
import { buildTokenBoundedBatches, DEFAULT_REMOTE_BATCH_SIZE, DEFAULT_TOKEN_BUDGET, estimateTokenCount, hasRemoteEndpoint, } from "../llm/embedders/remote.js";
|
|
9
|
+
import { buildTokenBoundedBatches, capEmbeddingText, DEFAULT_MAX_INPUT_TOKENS, DEFAULT_REMOTE_BATCH_SIZE, DEFAULT_TOKEN_BUDGET, describeEmbeddingCredential, estimateTokenCount, hasRemoteEndpoint, normalizeEmbeddingEndpoint, } from "../llm/embedders/remote.js";
|
|
9
10
|
import { cosineSimilarity } from "../llm/embedders/types.js";
|
|
11
|
+
import { purgeEmbeddingSalvage, relabelEmbeddingSalvageFingerprint, reuseSalvagedEmbeddings, } from "../storage/repositories/embedding-salvage-repository.js";
|
|
10
12
|
import { getEmbeddableEntryCount } from "../storage/repositories/index-entries-repository.js";
|
|
11
13
|
import { deleteMeta, getMeta, setMeta } from "../storage/repositories/index-meta-repository.js";
|
|
12
14
|
import { getAllEntriesForEmbedding, getEmbeddingCount, isVecFastPathComplete, isVecFastPathReady, purgeEmbeddings, sampleEmbeddedEntriesForCanary, setVecFastPathReady, upsertEmbedding, } from "../storage/repositories/index-vec-repository.js";
|
|
@@ -23,8 +25,14 @@ export function deriveSemanticProviderFingerprint(embedding) {
|
|
|
23
25
|
}
|
|
24
26
|
return `local:${embedding?.localModel ?? DEFAULT_LOCAL_MODEL}`;
|
|
25
27
|
}
|
|
26
|
-
/**
|
|
27
|
-
|
|
28
|
+
/**
|
|
29
|
+
* The heartbeat text emitted every 15s while a provider request is in
|
|
30
|
+
* flight, and by default (not `--verbose`-only) since silence indistinguishable
|
|
31
|
+
* from a hang was the field report's own symptom (#954).
|
|
32
|
+
*/
|
|
33
|
+
export function formatEmbeddingHeartbeat(storedCount, total, failedCount) {
|
|
34
|
+
return `Still generating embeddings: ${storedCount}/${total} stored, ${failedCount} failed; waiting on embedding provider.`;
|
|
35
|
+
}
|
|
28
36
|
/**
|
|
29
37
|
* Number of already-embedded entries sampled for the fingerprint-rename
|
|
30
38
|
* canary (#955) — small and cheap even against a slow local server; a
|
|
@@ -38,6 +46,23 @@ const CANARY_SAMPLE_SIZE = 8;
|
|
|
38
46
|
* same-model rename rather than a real model change (#955).
|
|
39
47
|
*/
|
|
40
48
|
const CANARY_SIMILARITY_THRESHOLD = 0.999;
|
|
49
|
+
/**
|
|
50
|
+
* Consecutive transport failures after which the embedding pass stops
|
|
51
|
+
* dispatching further requests and ends the run as a failure rather than
|
|
52
|
+
* grinding through every remaining batch against a dead endpoint (#954).
|
|
53
|
+
* Two independent trip conditions
|
|
54
|
+
* share this threshold — see `onSkip` below: 3 consecutive failures at
|
|
55
|
+
* single-document size (timeout OR network error — a multi-document
|
|
56
|
+
* timeout is not by itself evidence the endpoint is dead, since
|
|
57
|
+
* `RemoteEmbedder.embedBatch` already retries and splits it smaller before
|
|
58
|
+
* ever reporting it as failed at single-document size), or 3 consecutive
|
|
59
|
+
* network errors at ANY size (a network error is never retried, so it is
|
|
60
|
+
* trusted immediately regardless of how large the request was).
|
|
61
|
+
* `context-window-exceeded` never counts — that reason proves the provider
|
|
62
|
+
* IS reachable, and split-and-retry already handles it; it resets both
|
|
63
|
+
* streaks instead.
|
|
64
|
+
*/
|
|
65
|
+
const CIRCUIT_BREAKER_THRESHOLD = 3;
|
|
41
66
|
/**
|
|
42
67
|
* Pure decision: do stored vectors remain valid against freshly re-embedded
|
|
43
68
|
* canary samples? The ONE place that computes the canary's similarity
|
|
@@ -89,7 +114,7 @@ function medianOf(values) {
|
|
|
89
114
|
* on what the server (or local model) actually reported plus the observed
|
|
90
115
|
* vector width, so a gateway/transport change that keeps returning the same
|
|
91
116
|
* underlying model can be told apart from a genuine model change without
|
|
92
|
-
* relying on the operator's config string (#955
|
|
117
|
+
* relying on the operator's config string (#955).
|
|
93
118
|
* Returns undefined when nothing was actually observed this call (no vector
|
|
94
119
|
* to measure yet).
|
|
95
120
|
*/
|
|
@@ -176,11 +201,47 @@ function throwIfAborted(signal) {
|
|
|
176
201
|
}
|
|
177
202
|
}
|
|
178
203
|
export async function generateEmbeddingsForDb(db, config, onProgress, signal, entryIds, opts) {
|
|
204
|
+
// Drift guard (#954): refuse an ambient transaction. Every
|
|
205
|
+
// per-batch `db.transaction()` below is meant to be its own durable commit
|
|
206
|
+
// (#954) — inside an already-open outer transaction it would nest as an
|
|
207
|
+
// unobservable SAVEPOINT instead, so an interruption (competing-process
|
|
208
|
+
// collision, SIGKILL) could lose the whole pass rather than only the batch
|
|
209
|
+
// in flight. This is an internal contract error (a caller bug), not a
|
|
210
|
+
// user-facing failure class: callers with their own transaction (e.g. `akm
|
|
211
|
+
// bundle update`'s unified update transaction) must run the embedding
|
|
212
|
+
// phase on a separate connection AFTER their own transaction commits — see
|
|
213
|
+
// `runEmbeddingPass` in `src/indexer/indexer.ts`.
|
|
214
|
+
if (db.inTransaction) {
|
|
215
|
+
throw new Error("generateEmbeddingsForDb was called with an ambient transaction already open on `db`: per-batch commits " +
|
|
216
|
+
"would become SAVEPOINTs inside it, losing the crash-durability contract per-batch commit exists for. " +
|
|
217
|
+
"Run the embedding phase on a connection with no open transaction.");
|
|
218
|
+
}
|
|
179
219
|
throwIfAborted(signal);
|
|
180
220
|
if (config.semanticSearchMode === "off") {
|
|
221
|
+
// #955: salvage is self-emptying only if every path that skips reuse
|
|
222
|
+
// also drains it — otherwise a full rebuild performed with semantic
|
|
223
|
+
// search disabled leaves permanent orphaned rows behind (nothing will
|
|
224
|
+
// ever consume them, since this path never reaches the reuse step).
|
|
225
|
+
purgeEmbeddingSalvage(db);
|
|
181
226
|
onProgress({ phase: "embeddings", message: "Semantic search disabled; skipping embeddings." });
|
|
182
227
|
return { success: false, message: "Semantic search is disabled." };
|
|
183
228
|
}
|
|
229
|
+
// #953 field gap: the actionable outcome is a self-diagnosing run, not a
|
|
230
|
+
// fix (every RemoteEmbedder path already resolves secret:// through one
|
|
231
|
+
// boundary — a keyless request can only mean embedding.apiKey was absent
|
|
232
|
+
// from the config THIS run loaded). One default-level line, before the
|
|
233
|
+
// first provider request of the phase (the canary probe or the main
|
|
234
|
+
// pass, whichever runs first below), naming the endpoint/model/credential
|
|
235
|
+
// SOURCE — never the credential value.
|
|
236
|
+
if (hasRemoteEndpoint(config.embedding ?? {})) {
|
|
237
|
+
const endpoint = normalizeEmbeddingEndpoint(config.embedding?.endpoint ?? "");
|
|
238
|
+
const credential = describeEmbeddingCredential(config.embedding?.apiKey);
|
|
239
|
+
const configFileSuffix = isVerbose() ? `; config: ${getConfigPath()}` : "";
|
|
240
|
+
onProgress({
|
|
241
|
+
phase: "embeddings",
|
|
242
|
+
message: `[embed] endpoint ${endpoint}, model ${config.embedding?.model ?? "unknown"}; credential: ${credential}${configFileSuffix}`,
|
|
243
|
+
});
|
|
244
|
+
}
|
|
184
245
|
// A targeted call starts from an already-published generation. Preserve its
|
|
185
246
|
// trust decision in O(1): successful writes for the changed IDs keep a
|
|
186
247
|
// healthy fast path healthy, but can never promote a generation already
|
|
@@ -200,6 +261,9 @@ export async function generateEmbeddingsForDb(db, config, onProgress, signal, en
|
|
|
200
261
|
// instead of purging again from zero (#955/#956).
|
|
201
262
|
db.transaction(() => {
|
|
202
263
|
purgeEmbeddings(db, { dropVecTable: true });
|
|
264
|
+
// #955: an explicit forced rebuild must re-embed everything, not
|
|
265
|
+
// quietly satisfy some of it from stale salvage.
|
|
266
|
+
purgeEmbeddingSalvage(db);
|
|
203
267
|
deleteMeta(db, "embeddingDim");
|
|
204
268
|
setMeta(db, "embeddingFingerprint", currentFingerprint);
|
|
205
269
|
deleteMeta(db, "embeddingIdentity");
|
|
@@ -221,6 +285,9 @@ export async function generateEmbeddingsForDb(db, config, onProgress, signal, en
|
|
|
221
285
|
if (decision.outcome === "rebuild") {
|
|
222
286
|
db.transaction(() => {
|
|
223
287
|
purgeEmbeddings(db, { dropVecTable: true });
|
|
288
|
+
// #955: the stored vectors AND any leftover salvage both belong to
|
|
289
|
+
// a different model now — neither is reusable, so both go.
|
|
290
|
+
purgeEmbeddingSalvage(db);
|
|
224
291
|
deleteMeta(db, "embeddingDim");
|
|
225
292
|
setMeta(db, "embeddingFingerprint", currentFingerprint);
|
|
226
293
|
if (decision.identity)
|
|
@@ -238,6 +305,11 @@ export async function generateEmbeddingsForDb(db, config, onProgress, signal, en
|
|
|
238
305
|
// immediate write means a crash right after this decision does not
|
|
239
306
|
// re-run the canary needlessly on the next attempt.
|
|
240
307
|
setMeta(db, "embeddingFingerprint", currentFingerprint);
|
|
308
|
+
// #955: the model did not actually change, only the fingerprint
|
|
309
|
+
// STRING did (e.g. a gateway rename) — any leftover salvage rows
|
|
310
|
+
// tagged with the OLD string are still valid vectors. Relabel them so
|
|
311
|
+
// the reuse step below (and any later pass) can still find them.
|
|
312
|
+
relabelEmbeddingSalvageFingerprint(db, storedFingerprint, currentFingerprint);
|
|
241
313
|
if (decision.identity)
|
|
242
314
|
setMeta(db, "embeddingIdentity", decision.identity);
|
|
243
315
|
if (decision.verified) {
|
|
@@ -253,24 +325,94 @@ export async function generateEmbeddingsForDb(db, config, onProgress, signal, en
|
|
|
253
325
|
// lose or verify against — adopt the label silently, no purge line.
|
|
254
326
|
}
|
|
255
327
|
}
|
|
328
|
+
else {
|
|
329
|
+
// No rename to verify (either this is the very first
|
|
330
|
+
// pass ever for this db, or the fingerprint already matches the last
|
|
331
|
+
// successful one) — still record it NOW rather than deferring to a
|
|
332
|
+
// fully successful pass, mirroring the rebuild/keep branches above
|
|
333
|
+
// (#955/#956). Without this, an interrupted FIRST-EVER pass left
|
|
334
|
+
// `embeddingFingerprint` unset despite a per-batch commit below (#954)
|
|
335
|
+
// already having durably written real vectors — a later `akm index
|
|
336
|
+
// --full`'s salvage-before-discard step tags rows by this meta
|
|
337
|
+
// (`salvageEmbeddingsBeforeDiscard`) and treats an unset fingerprint as
|
|
338
|
+
// "nothing was ever verified", silently turning genuinely-embedded
|
|
339
|
+
// vectors into a full re-embed instead of a salvage-and-reuse.
|
|
340
|
+
setMeta(db, "embeddingFingerprint", currentFingerprint);
|
|
341
|
+
}
|
|
256
342
|
try {
|
|
257
343
|
throwIfAborted(signal);
|
|
258
344
|
const allEntries = getAllEntriesForEmbedding(db, targetEntryIds);
|
|
259
|
-
|
|
345
|
+
let vecFailedCount = 0;
|
|
346
|
+
let vecUnavailableCount = 0;
|
|
347
|
+
// #955: before any provider call, hand back vectors salvaged from a
|
|
348
|
+
// full rebuild or a generation bump for entries whose search_text is
|
|
349
|
+
// byte-identical to what was salvaged under the SAME fingerprint — a
|
|
350
|
+
// fingerprint mismatch or a single-byte content change both correctly
|
|
351
|
+
// fall through to the provider below instead.
|
|
352
|
+
const { reusedCount, remaining: candidateEntries } = reuseSalvagedEmbeddings(db, allEntries, currentFingerprint, (entry, embedding) => {
|
|
353
|
+
const result = upsertEmbedding(db, entry.id, embedding);
|
|
354
|
+
if (result.vec === "failed")
|
|
355
|
+
vecFailedCount++;
|
|
356
|
+
if (result.vec === "unavailable")
|
|
357
|
+
vecUnavailableCount++;
|
|
358
|
+
return result.stored;
|
|
359
|
+
});
|
|
360
|
+
if (reusedCount > 0) {
|
|
361
|
+
onProgress({
|
|
362
|
+
phase: "embeddings",
|
|
363
|
+
message: `Reused ${reusedCount} embedding${reusedCount === 1 ? "" : "s"} from the previous generation; embedding ${candidateEntries.length} new.`,
|
|
364
|
+
});
|
|
365
|
+
}
|
|
366
|
+
if (candidateEntries.length === 0) {
|
|
260
367
|
onProgress({ phase: "embeddings", message: "Embeddings already up to date." });
|
|
261
368
|
setMeta(db, "embeddingFingerprint", currentFingerprint);
|
|
262
|
-
|
|
369
|
+
if (reusedCount > 0) {
|
|
370
|
+
const vecGenerationComplete = targetEntryIds === undefined ? isVecFastPathComplete(db) : vecFastPathWasReady;
|
|
371
|
+
setVecFastPathReady(db, vecFailedCount === 0 && vecUnavailableCount === 0 && vecGenerationComplete);
|
|
372
|
+
}
|
|
373
|
+
// A pass that completes (even one that did nothing but reuse) purges
|
|
374
|
+
// whatever is left — salvage is consumed by the NEXT pass, never kept
|
|
375
|
+
// around as a second cache.
|
|
376
|
+
purgeEmbeddingSalvage(db);
|
|
377
|
+
return reusedCount > 0 ? { success: true, vecInsertFailures: vecFailedCount } : { success: true };
|
|
378
|
+
}
|
|
379
|
+
// Cap each document's embedded text at
|
|
380
|
+
// embedding.maxInputTokens (default DEFAULT_MAX_INPUT_TOKENS) instead of
|
|
381
|
+
// ever failing a whole batch over one oversized entry — truncation keeps
|
|
382
|
+
// the head of the text, unicode-safe. A document is skipped only when its
|
|
383
|
+
// head is empty (the impossible case: nothing left to embed), never
|
|
384
|
+
// merely for being long.
|
|
385
|
+
const maxInputTokens = config.embedding?.maxInputTokens ?? DEFAULT_MAX_INPUT_TOKENS;
|
|
386
|
+
let truncatedCount = 0;
|
|
387
|
+
const texts = [];
|
|
388
|
+
const pendingEntries = [];
|
|
389
|
+
for (const entry of candidateEntries) {
|
|
390
|
+
const capped = capEmbeddingText(entry.searchText, maxInputTokens);
|
|
391
|
+
if (capped.text.length === 0)
|
|
392
|
+
continue;
|
|
393
|
+
if (capped.truncated)
|
|
394
|
+
truncatedCount++;
|
|
395
|
+
pendingEntries.push(entry);
|
|
396
|
+
texts.push(capped.text);
|
|
397
|
+
}
|
|
398
|
+
if (truncatedCount > 0) {
|
|
399
|
+
// Through onProgress ONLY, not warn() too — onProgress already reaches
|
|
400
|
+
// stderr at the default level in every output mode (#954), and the
|
|
401
|
+
// index CLI's progress handler writes it through info() (log-file
|
|
402
|
+
// aware), so calling warn() as well printed the identical sentence
|
|
403
|
+
// twice in text mode.
|
|
404
|
+
const message = `[embed] ${truncatedCount} entr${truncatedCount === 1 ? "y" : "ies"} truncated to the ${maxInputTokens}-token embedding cap (embedding.maxInputTokens); rerun with a higher cap to embed the full text.`;
|
|
405
|
+
onProgress({ phase: "embeddings", message });
|
|
263
406
|
}
|
|
264
407
|
if (rebuildReason) {
|
|
265
|
-
|
|
266
|
-
|
|
408
|
+
// See the truncation notice above: onProgress ONLY.
|
|
409
|
+
const message = `[embed] Re-embedding ${pendingEntries.length} entr${pendingEntries.length === 1 ? "y" : "ies"} because ${rebuildReason}`;
|
|
267
410
|
onProgress({ phase: "embeddings", message });
|
|
268
411
|
}
|
|
269
412
|
onProgress({
|
|
270
413
|
phase: "embeddings",
|
|
271
|
-
message: `Generating embeddings for ${
|
|
414
|
+
message: `Generating embeddings for ${pendingEntries.length} entr${pendingEntries.length === 1 ? "y" : "ies"}.`,
|
|
272
415
|
});
|
|
273
|
-
const texts = allEntries.map((entry) => entry.searchText);
|
|
274
416
|
if (isVerbose()) {
|
|
275
417
|
// Mirror RemoteEmbedder's actual token-bounded batching (#874) so this
|
|
276
418
|
// log reflects the real request grouping rather than a fixed count of
|
|
@@ -279,7 +421,9 @@ export async function generateEmbeddingsForDb(db, config, onProgress, signal, en
|
|
|
279
421
|
// for inference throughput only, never fails/skips), so there's
|
|
280
422
|
// nothing meaningful to report per-batch for them.
|
|
281
423
|
if (hasRemoteEndpoint(config.embedding ?? {})) {
|
|
282
|
-
|
|
424
|
+
// Mirrors RemoteEmbedder.embedBatch's own tokenBudget resolution
|
|
425
|
+
// (#956: contextLength no longer feeds this).
|
|
426
|
+
const tokenBudget = config.embedding?.maxTokens ?? DEFAULT_TOKEN_BUDGET;
|
|
283
427
|
const maxCount = config.embedding?.batchSize ?? DEFAULT_REMOTE_BATCH_SIZE;
|
|
284
428
|
const batches = buildTokenBoundedBatches(texts, tokenBudget, maxCount);
|
|
285
429
|
const batchNumberByIndex = new Map();
|
|
@@ -287,7 +431,7 @@ export async function generateEmbeddingsForDb(db, config, onProgress, signal, en
|
|
|
287
431
|
for (const i of batch.indices)
|
|
288
432
|
batchNumberByIndex.set(i, batchIdx + 1);
|
|
289
433
|
});
|
|
290
|
-
for (const [i, entry] of
|
|
434
|
+
for (const [i, entry] of pendingEntries.entries()) {
|
|
291
435
|
const chars = entry.searchText.length;
|
|
292
436
|
const tokens = estimateTokenCount(entry.searchText);
|
|
293
437
|
const batch = batches[batchNumberByIndex.get(i) - 1];
|
|
@@ -298,7 +442,7 @@ export async function generateEmbeddingsForDb(db, config, onProgress, signal, en
|
|
|
298
442
|
}
|
|
299
443
|
}
|
|
300
444
|
else {
|
|
301
|
-
for (const entry of
|
|
445
|
+
for (const entry of pendingEntries) {
|
|
302
446
|
warnVerbose(`[embed] ${entry.itemRef} (${entry.searchText.length} chars, est. ${estimateTokenCount(entry.searchText)} tokens)`);
|
|
303
447
|
}
|
|
304
448
|
}
|
|
@@ -307,15 +451,12 @@ export async function generateEmbeddingsForDb(db, config, onProgress, signal, en
|
|
|
307
451
|
let storedCount = 0;
|
|
308
452
|
let skippedCount = 0;
|
|
309
453
|
let embedFailedCount = 0;
|
|
310
|
-
let vecFailedCount = 0;
|
|
311
|
-
let vecUnavailableCount = 0;
|
|
312
454
|
let storedTokens = 0;
|
|
313
|
-
let lastProgressBucket = 0;
|
|
314
455
|
try {
|
|
315
456
|
heartbeatTimer = setInterval(() => {
|
|
316
457
|
onProgress({
|
|
317
458
|
phase: "embeddings",
|
|
318
|
-
message:
|
|
459
|
+
message: formatEmbeddingHeartbeat(storedCount, pendingEntries.length, embedFailedCount),
|
|
319
460
|
});
|
|
320
461
|
}, 15000);
|
|
321
462
|
// A failing sub-batch or an oversized document is SKIPPED by embedBatch,
|
|
@@ -323,6 +464,42 @@ export async function generateEmbeddingsForDb(db, config, onProgress, signal, en
|
|
|
323
464
|
// few bad documents don't discard every other entry's embedding.
|
|
324
465
|
const skips = [];
|
|
325
466
|
const embedStart = Date.now();
|
|
467
|
+
// Circuit breaker (#954): stop
|
|
468
|
+
// dispatching further batches once either consecutive-failure streak
|
|
469
|
+
// below reaches CIRCUIT_BREAKER_THRESHOLD — a dead/hung provider used
|
|
470
|
+
// to grind through every remaining batch for hours, one 30s (now
|
|
471
|
+
// configurable, and now backed off/retried/split first — see
|
|
472
|
+
// RemoteEmbedder.embedBatch) timeout at a time, ending in one
|
|
473
|
+
// aggregate warning and `ok: true`. Counted per BATCH
|
|
474
|
+
// (`skip.batchStart`), not per document: a single failed 100-document
|
|
475
|
+
// batch must not look like 100 consecutive failures.
|
|
476
|
+
let consecutiveSingleDocFailures = 0;
|
|
477
|
+
let consecutiveNetworkErrorFailures = 0;
|
|
478
|
+
let circuitBreakerReason;
|
|
479
|
+
const onSkip = (skip) => {
|
|
480
|
+
skips.push(skip);
|
|
481
|
+
if (!skip.batchStart)
|
|
482
|
+
return undefined;
|
|
483
|
+
if (skip.reason === "context-window-exceeded") {
|
|
484
|
+
consecutiveSingleDocFailures = 0;
|
|
485
|
+
consecutiveNetworkErrorFailures = 0;
|
|
486
|
+
return undefined;
|
|
487
|
+
}
|
|
488
|
+
// "batch-request-failed": a timeout only counts once retries have
|
|
489
|
+
// already narrowed it down to a single document (embedBatch backs
|
|
490
|
+
// off, retries, and splits a multi-document timeout before ever
|
|
491
|
+
// reporting it here); a network error counts immediately at any
|
|
492
|
+
// size — it was never retried, so it is trusted right away.
|
|
493
|
+
consecutiveSingleDocFailures = skip.batchSize === 1 ? consecutiveSingleDocFailures + 1 : 0;
|
|
494
|
+
consecutiveNetworkErrorFailures =
|
|
495
|
+
skip.failureKind === "network-error" ? consecutiveNetworkErrorFailures + 1 : 0;
|
|
496
|
+
if (consecutiveSingleDocFailures >= CIRCUIT_BREAKER_THRESHOLD ||
|
|
497
|
+
consecutiveNetworkErrorFailures >= CIRCUIT_BREAKER_THRESHOLD) {
|
|
498
|
+
circuitBreakerReason = skip.message;
|
|
499
|
+
return false;
|
|
500
|
+
}
|
|
501
|
+
return undefined;
|
|
502
|
+
};
|
|
326
503
|
// Commit each provider batch in its own short transaction as it lands,
|
|
327
504
|
// rather than buffering the whole run in memory for one transaction at
|
|
328
505
|
// the very end (#954) — a competing-process lock error or any other
|
|
@@ -334,12 +511,38 @@ export async function generateEmbeddingsForDb(db, config, onProgress, signal, en
|
|
|
334
511
|
// is kept; every batch from one run shares the same provider/model.
|
|
335
512
|
let observedModel;
|
|
336
513
|
let observedVectorLen;
|
|
337
|
-
|
|
514
|
+
// Whether the remote provider's endpoint/model/token language is
|
|
515
|
+
// meaningful for this run — the per-batch diagnostic line below is
|
|
516
|
+
// remote-only, same gate the credential diagnostic (#953) above uses.
|
|
517
|
+
const reportPerBatchLine = hasRemoteEndpoint(config.embedding ?? {});
|
|
518
|
+
const onBatch = (indices, batchEmbeddings, model, outcome) => {
|
|
519
|
+
// #954 field-report follow-up: a "retrying" event carries nothing to
|
|
520
|
+
// commit — the request hasn't settled yet — only the notice that a
|
|
521
|
+
// back-off is about to be waited out, default-level so a run is
|
|
522
|
+
// never silently stalled indistinguishably from a hang.
|
|
523
|
+
if (outcome?.outcome === "retrying") {
|
|
524
|
+
if (reportPerBatchLine) {
|
|
525
|
+
onProgress({
|
|
526
|
+
phase: "embeddings",
|
|
527
|
+
message: `[embed] batch ${outcome.batchIndex}/${outcome.batchCount}: ${outcome.docCount} docs, ${outcome.requestTokens.toLocaleString()} tokens → retrying after ${(outcome.elapsedMs / 1000).toFixed(1)} s`,
|
|
528
|
+
});
|
|
529
|
+
}
|
|
530
|
+
return;
|
|
531
|
+
}
|
|
338
532
|
if (model)
|
|
339
533
|
observedModel = model;
|
|
534
|
+
// A batch that delivered at least one real embedding proves the
|
|
535
|
+
// provider is currently answering — reset both circuit-breaker
|
|
536
|
+
// streaks. (A wholly failed batch's `batchEmbeddings` are all
|
|
537
|
+
// `undefined`, per commitBatch's skip path, so this never
|
|
538
|
+
// re-triggers what onSkip just counted moments earlier.)
|
|
539
|
+
if (batchEmbeddings.some((embedding) => embedding !== undefined)) {
|
|
540
|
+
consecutiveSingleDocFailures = 0;
|
|
541
|
+
consecutiveNetworkErrorFailures = 0;
|
|
542
|
+
}
|
|
340
543
|
db.transaction(() => {
|
|
341
544
|
for (let k = 0; k < indices.length; k++) {
|
|
342
|
-
const entry =
|
|
545
|
+
const entry = pendingEntries[indices[k]];
|
|
343
546
|
if (!entry)
|
|
344
547
|
continue;
|
|
345
548
|
const embedding = batchEmbeddings[k];
|
|
@@ -363,29 +566,36 @@ export async function generateEmbeddingsForDb(db, config, onProgress, signal, en
|
|
|
363
566
|
vecUnavailableCount++;
|
|
364
567
|
}
|
|
365
568
|
})();
|
|
366
|
-
|
|
367
|
-
|
|
368
|
-
|
|
569
|
+
// Default level, one line per provider batch (#954, field-report
|
|
570
|
+
// follow-up): oversized documents never made a request
|
|
571
|
+
// (`reason === "oversized"`), so there is no batch outcome to
|
|
572
|
+
// report — they are covered by the run's final oversized-skip
|
|
573
|
+
// count and list instead.
|
|
574
|
+
if (outcome && outcome.reason !== "oversized" && reportPerBatchLine) {
|
|
575
|
+
const elapsedSeconds = (outcome.elapsedMs / 1000).toFixed(1);
|
|
576
|
+
const outcomeLabel = outcome.outcome === "stored"
|
|
577
|
+
? `${outcome.docCount} stored (${elapsedSeconds} s)`
|
|
578
|
+
: `failed: ${outcome.reason}`;
|
|
369
579
|
onProgress({
|
|
370
580
|
phase: "embeddings",
|
|
371
|
-
message: `
|
|
581
|
+
message: `[embed] batch ${outcome.batchIndex}/${outcome.batchCount}: ${outcome.docCount} docs, ${outcome.requestTokens.toLocaleString()} tokens → ${outcomeLabel}`,
|
|
372
582
|
});
|
|
373
583
|
}
|
|
584
|
+
// Every committed batch, not just every 500 stored entries (#954)
|
|
585
|
+
// — the prior bucketing left a non-verbose run silent
|
|
586
|
+
// for the entire embedding phase on anything smaller than 500
|
|
587
|
+
// entries, indistinguishable from a hang.
|
|
588
|
+
onProgress({
|
|
589
|
+
phase: "embeddings",
|
|
590
|
+
message: `Embedded ${storedCount}/${pendingEntries.length} entries.`,
|
|
591
|
+
});
|
|
374
592
|
};
|
|
375
|
-
await embedBatch(texts, config.embedding, signal,
|
|
593
|
+
await embedBatch(texts, config.embedding, signal, onSkip, onBatch);
|
|
376
594
|
throwIfAborted(signal);
|
|
377
595
|
const elapsedSeconds = Math.max((Date.now() - embedStart) / 1000, 0.001);
|
|
378
596
|
if (skippedCount > 0) {
|
|
379
597
|
warn(`[embed] ${skippedCount} embedding${skippedCount === 1 ? "" : "s"} skipped (entry deleted between queue and write)`);
|
|
380
598
|
}
|
|
381
|
-
if (embedFailedCount > 0) {
|
|
382
|
-
const detail = skips
|
|
383
|
-
.slice(0, 20)
|
|
384
|
-
.map((skip) => ` - ${allEntries[skip.index]?.itemRef ?? skip.index} (${skip.reason}): ${skip.message}`)
|
|
385
|
-
.join("\n");
|
|
386
|
-
const more = skips.length > 20 ? `\n ...and ${skips.length - 20} more` : "";
|
|
387
|
-
warn(`[embed] ${embedFailedCount} embedding${embedFailedCount === 1 ? "" : "s"} could not be generated and ${embedFailedCount === 1 ? "was" : "were"} skipped:\n${detail}${more}`);
|
|
388
|
-
}
|
|
389
599
|
const vecGenerationComplete = targetEntryIds === undefined ? isVecFastPathComplete(db) : vecFastPathWasReady;
|
|
390
600
|
setVecFastPathReady(db, vecFailedCount === 0 && vecUnavailableCount === 0 && vecGenerationComplete);
|
|
391
601
|
if (vecFailedCount > 0) {
|
|
@@ -395,14 +605,66 @@ export async function generateEmbeddingsForDb(db, config, onProgress, signal, en
|
|
|
395
605
|
}
|
|
396
606
|
const entriesPerSec = storedCount / elapsedSeconds;
|
|
397
607
|
const tokensPerSec = storedTokens / elapsedSeconds;
|
|
608
|
+
const totalStored = storedCount + reusedCount;
|
|
609
|
+
// #954, field-report follow-up: the final line
|
|
610
|
+
// reports every outcome, not just what was stored — counts come from
|
|
611
|
+
// the same collected `skips` the circuit breaker already uses,
|
|
612
|
+
// categorized by `reason`/`failureKind`. "oversized skipped" =
|
|
613
|
+
// context-window-exceeded (never fit any request, at any size);
|
|
614
|
+
// "timed out" = a batch-request-failed skip whose last attempt timed
|
|
615
|
+
// out (retries/splits already exhausted before this counted); "failed"
|
|
616
|
+
// = every other batch-request-failed skip (a genuine, never-retried
|
|
617
|
+
// network/HTTP failure).
|
|
618
|
+
const oversizedSkips = skips.filter((skip) => skip.reason === "context-window-exceeded");
|
|
619
|
+
const timedOutSkips = skips.filter((skip) => skip.reason === "batch-request-failed" && skip.failureKind === "timeout");
|
|
620
|
+
const failedSkips = skips.filter((skip) => skip.reason === "batch-request-failed" && skip.failureKind !== "timeout");
|
|
621
|
+
const throughputLine = reusedCount > 0
|
|
622
|
+
? // #955: report reused and newly-embedded counts separately — the
|
|
623
|
+
// rate figures below are provider throughput only (reuse is a
|
|
624
|
+
// plain DB write, not provider work) and would be misleadingly
|
|
625
|
+
// inflated if reused entries were folded into them.
|
|
626
|
+
`Stored ${totalStored} embedding${totalStored === 1 ? "" : "s"} (${reusedCount} reused, ${storedCount} newly embedded) in ${elapsedSeconds.toFixed(1)}s (${entriesPerSec.toFixed(1)} entries/s, ~${Math.round(tokensPerSec)} tokens/s)`
|
|
627
|
+
: `Stored ${storedCount} embedding${storedCount === 1 ? "" : "s"} in ${elapsedSeconds.toFixed(1)}s (${entriesPerSec.toFixed(1)} entries/s, ~${Math.round(tokensPerSec)} tokens/s)`;
|
|
398
628
|
onProgress({
|
|
399
629
|
phase: "embeddings",
|
|
400
|
-
message:
|
|
630
|
+
message: `${throughputLine}; ${oversizedSkips.length} oversized skipped, ${timedOutSkips.length} timed out, ${failedSkips.length} failed.`,
|
|
401
631
|
});
|
|
632
|
+
// Bounded itemRef-level detail for every skip category, not just
|
|
633
|
+
// oversized — the aggregate counts above say HOW MANY documents timed
|
|
634
|
+
// out or failed, but give the operator no way to find out WHICH ones
|
|
635
|
+
// short of rerunning with --verbose and re-reading the whole log.
|
|
636
|
+
// Default level caps each list (there is nothing actionable about the
|
|
637
|
+
// 21st identical failure); --verbose prints every one, matching the
|
|
638
|
+
// per-document mapping lines' own verbosity gate above.
|
|
639
|
+
const printSkipList = (label, skipList) => {
|
|
640
|
+
if (skipList.length === 0)
|
|
641
|
+
return;
|
|
642
|
+
const limit = isVerbose() ? skipList.length : 20;
|
|
643
|
+
const listed = skipList
|
|
644
|
+
.slice(0, limit)
|
|
645
|
+
.map((skip) => ` - ${pendingEntries[skip.index]?.itemRef ?? skip.index}: ${skip.message}`)
|
|
646
|
+
.join("\n");
|
|
647
|
+
const more = skipList.length > limit ? `\n ...and ${skipList.length - limit} more` : "";
|
|
648
|
+
onProgress({ phase: "embeddings", message: `[embed] ${label} skipped:\n${listed}${more}` });
|
|
649
|
+
};
|
|
650
|
+
printSkipList("oversized documents", oversizedSkips);
|
|
651
|
+
printSkipList("timed-out documents", timedOutSkips);
|
|
652
|
+
printSkipList("failed documents", failedSkips);
|
|
402
653
|
setMeta(db, "embeddingFingerprint", currentFingerprint);
|
|
403
654
|
const observedIdentity = deriveObservedEmbeddingIdentity(config.embedding, observedModel, observedVectorLen);
|
|
404
655
|
if (observedIdentity)
|
|
405
656
|
setMeta(db, "embeddingIdentity", observedIdentity);
|
|
657
|
+
// Circuit breaker tripped (#954): committed batches are
|
|
658
|
+
// kept (nothing above discards them), but the pass is not a success —
|
|
659
|
+
// the provider looks dead, not just occasionally flaky.
|
|
660
|
+
if (circuitBreakerReason !== undefined) {
|
|
661
|
+
const message = `embedding provider failed ${CIRCUIT_BREAKER_THRESHOLD} consecutive batches ` +
|
|
662
|
+
`(last: ${circuitBreakerReason}); stopped after ${storedCount} embedding${storedCount === 1 ? "" : "s"} ` +
|
|
663
|
+
"were stored — rerun akm index when the endpoint is healthy";
|
|
664
|
+
warn(`[embed] ${message}`);
|
|
665
|
+
onProgress({ phase: "embeddings", message });
|
|
666
|
+
return { success: false, message, vecInsertFailures: vecFailedCount };
|
|
667
|
+
}
|
|
406
668
|
// Only a total failure (nothing at all embedded, despite having entries
|
|
407
669
|
// to embed) turns into a phase failure. Any partial success — the vast
|
|
408
670
|
// majority of a large bundle embedding fine around a handful of skips —
|
|
@@ -416,6 +678,10 @@ export async function generateEmbeddingsForDb(db, config, onProgress, signal, en
|
|
|
416
678
|
message: `All ${embedFailedCount} embedding batch(es) failed: ${firstMessage}`,
|
|
417
679
|
};
|
|
418
680
|
}
|
|
681
|
+
// A pass that completes without abort or circuit-break purges
|
|
682
|
+
// whatever salvage is left — consumed by this pass's reuse step
|
|
683
|
+
// above, or superseded by what it just embedded.
|
|
684
|
+
purgeEmbeddingSalvage(db);
|
|
419
685
|
return { success: true, vecInsertFailures: vecFailedCount };
|
|
420
686
|
}
|
|
421
687
|
finally {
|
|
@@ -238,6 +238,7 @@ export async function ensureSourceCaches(config, options) {
|
|
|
238
238
|
const cfg = config ?? loadConfig();
|
|
239
239
|
const force = options?.force === true;
|
|
240
240
|
const materialize = options?.materialize !== false;
|
|
241
|
+
const onProgress = options?.onProgress ?? (() => { });
|
|
241
242
|
// Polymorphic refresh: walk every enabled source through its registered
|
|
242
243
|
// provider and call `sync()`. Every cache-backed kind (git, website, npm)
|
|
243
244
|
// refreshes the same way — a bad source warns and is skipped without
|
|
@@ -249,6 +250,12 @@ export async function ensureSourceCaches(config, options) {
|
|
|
249
250
|
// content ENDS UP; the lock's `localRoot` merely records the result. So this
|
|
250
251
|
// path correctly uses the provider, not the shared `lockContentRootFor`
|
|
251
252
|
// resolver that reads/writes use to agree on where content already IS.
|
|
253
|
+
//
|
|
254
|
+
// Two passes: first resolve which sources will actually sync (constructing
|
|
255
|
+
// each provider once, exactly as before), so the progress count ("i/n")
|
|
256
|
+
// reflects real syncs rather than every configured entry including managed/
|
|
257
|
+
// unsyncable ones; then sync them in order, reporting progress per source.
|
|
258
|
+
const toSync = [];
|
|
252
259
|
for (const entry of getSources(cfg)) {
|
|
253
260
|
if (entry.enabled === false)
|
|
254
261
|
continue;
|
|
@@ -282,12 +289,27 @@ export async function ensureSourceCaches(config, options) {
|
|
|
282
289
|
warnIfSourceUnavailableForRead(entry, provider.name);
|
|
283
290
|
continue;
|
|
284
291
|
}
|
|
292
|
+
toSync.push(provider);
|
|
293
|
+
}
|
|
294
|
+
for (const [i, provider] of toSync.entries()) {
|
|
295
|
+
onProgress(`Hydrating source ${i + 1}/${toSync.length}: ${provider.name}`);
|
|
296
|
+
let heartbeat;
|
|
285
297
|
try {
|
|
286
|
-
|
|
298
|
+
heartbeat = setInterval(() => {
|
|
299
|
+
onProgress(`Still hydrating source ${i + 1}/${toSync.length}: ${provider.name}...`);
|
|
300
|
+
}, 15000);
|
|
301
|
+
// `toSync` only ever holds providers whose `.sync` passed the check
|
|
302
|
+
// above; the optional-chain here is just to satisfy the type (the
|
|
303
|
+
// narrowing does not survive the array round-trip).
|
|
304
|
+
await provider.sync?.({ force, secrets: options?.secrets, ensureWebsiteMirror });
|
|
287
305
|
}
|
|
288
306
|
catch (err) {
|
|
289
307
|
warn(`Warning: failed to refresh ${provider.kind} source "${provider.name}": ${err instanceof Error ? err.message : String(err)}`);
|
|
290
308
|
}
|
|
309
|
+
finally {
|
|
310
|
+
if (heartbeat)
|
|
311
|
+
clearInterval(heartbeat);
|
|
312
|
+
}
|
|
291
313
|
}
|
|
292
314
|
}
|
|
293
315
|
/**
|