akm-cli 0.9.16-alpha.1 → 0.9.16

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (147) hide show
  1. package/CHANGELOG.md +56 -132
  2. package/dist/assets/hints/cli-hints-full.md +13 -6
  3. package/dist/assets/tasks/core/index-refresh.yml +1 -1
  4. package/dist/assets/tasks/improve/akm-improve-catchup.yml +3 -6
  5. package/dist/cli/retired-commands.js +0 -4
  6. package/dist/cli/unknown-flags.js +3 -36
  7. package/dist/commands/env/env-binding.js +4 -4
  8. package/dist/commands/env/env-cli.js +3 -3
  9. package/dist/commands/improve/collapse-detector.js +2 -2
  10. package/dist/commands/improve/consolidate.js +4 -6
  11. package/dist/commands/improve/improve-cli.js +20 -15
  12. package/dist/commands/improve/reflect.js +23 -2
  13. package/dist/commands/lint/base-linter.js +9 -0
  14. package/dist/commands/lint/env-key-rules.js +2 -2
  15. package/dist/commands/proposal/propose.js +15 -1
  16. package/dist/commands/proposal/repository.js +3 -12
  17. package/dist/commands/proposal/validators/proposal-quality-validators.js +40 -3
  18. package/dist/commands/proposal/validators/proposal-validators.js +5 -4
  19. package/dist/commands/read/curate.js +44 -34
  20. package/dist/commands/read/search.js +35 -54
  21. package/dist/commands/read/show.js +21 -2
  22. package/dist/commands/registry-cli.js +5 -5
  23. package/dist/commands/sources/add-cli.js +59 -16
  24. package/dist/commands/sources/bundle-cli.js +35 -11
  25. package/dist/commands/sources/bundle-config-ops.js +30 -0
  26. package/dist/commands/sources/dangerous-env-audit.js +4 -4
  27. package/dist/commands/sources/info.js +8 -8
  28. package/dist/commands/sources/installed-stashes.js +55 -61
  29. package/dist/commands/sources/source-add.js +39 -38
  30. package/dist/commands/sources/source-manage.js +34 -12
  31. package/dist/commands/sources/stash-cli.js +111 -119
  32. package/dist/commands/sources/stash-skeleton.js +6 -3
  33. package/dist/commands/tasks/explain.js +4 -1
  34. package/dist/commands/tasks/tasks-cli.js +31 -9
  35. package/dist/commands/tasks/tasks.js +239 -194
  36. package/dist/commands/tasks/validate.js +20 -32
  37. package/dist/core/activation-policy.js +4 -4
  38. package/dist/core/adapter/adapters/akm-adapter.js +8 -35
  39. package/dist/core/adapter/adapters/akm-metadata.js +1 -11
  40. package/dist/core/adapter/execution-source.js +10 -29
  41. package/dist/core/asset/asset-placement.js +0 -35
  42. package/dist/core/config/config-schema.js +64 -8
  43. package/dist/core/config/config-sources.js +96 -2
  44. package/dist/core/config/config.js +190 -24
  45. package/dist/core/config/legacy-source-shape-shim.js +9 -0
  46. package/dist/core/config/schema/embedding.js +30 -7
  47. package/dist/core/config/schema/execution.js +23 -0
  48. package/dist/core/config/schema/experimental.js +1 -1
  49. package/dist/core/config/schema/scheduler.js +20 -0
  50. package/dist/core/config/schema/search.js +10 -12
  51. package/dist/core/config/schema/sources-bundles.js +32 -1
  52. package/dist/core/content-safety.js +52 -0
  53. package/dist/core/errors.js +2 -5
  54. package/dist/core/maintenance-barrier.js +11 -13
  55. package/dist/core/paths.js +11 -0
  56. package/dist/core/run-lock.js +2 -5
  57. package/dist/core/state/migrations.js +1 -26
  58. package/dist/core/state-db.js +27 -63
  59. package/dist/core/type-presentation.js +1 -1
  60. package/dist/core/write-source.js +13 -8
  61. package/dist/indexer/bundle-identity-guard.js +45 -8
  62. package/dist/indexer/ensure-index.js +0 -5
  63. package/dist/indexer/index-db-contention.js +56 -0
  64. package/dist/indexer/index-rebuild-lock.js +73 -0
  65. package/dist/indexer/index-written-assets.js +171 -133
  66. package/dist/indexer/indexer.js +1621 -458
  67. package/dist/indexer/lookup/adapter-concept-owner.js +5 -19
  68. package/dist/indexer/materialize-embeddings.js +785 -0
  69. package/dist/indexer/passes/dir-staleness.js +161 -0
  70. package/dist/indexer/passes/metadata.js +1 -18
  71. package/dist/indexer/scan/drain-dir.js +70 -27
  72. package/dist/indexer/search/db-search.js +89 -373
  73. package/dist/indexer/search/ranking-contributors.js +16 -21
  74. package/dist/indexer/search/ranking.js +57 -135
  75. package/dist/indexer/search/search-source.js +29 -11
  76. package/dist/integrations/agent/execution-lowering.js +3 -2
  77. package/dist/integrations/agent/execution-preparation.js +32 -1
  78. package/dist/integrations/agent/prompts.js +1 -1
  79. package/dist/integrations/agent/request-lowering.js +3 -2
  80. package/dist/llm/client.js +3 -11
  81. package/dist/llm/embedder.js +3 -10
  82. package/dist/llm/embedders/remote.js +104 -133
  83. package/dist/llm/feature-gate.js +2 -4
  84. package/dist/llm/rerank-client.js +3 -3
  85. package/dist/output/html-render.js +2 -1
  86. package/dist/output/shapes/passthrough.js +2 -1
  87. package/dist/output/stdout.js +24 -0
  88. package/dist/output/text/command-format.js +13 -19
  89. package/dist/output/text/helpers.js +1 -1
  90. package/dist/output/text/index.js +2 -5
  91. package/dist/output/text.js +4 -3
  92. package/dist/registry/resolve.js +37 -10
  93. package/dist/scripts/akm-migrate-node.js +15197 -11351
  94. package/dist/scripts/akm-migrate.js +15514 -11668
  95. package/dist/setup/semantic-assets.js +2 -2
  96. package/dist/setup/setup.js +3 -3
  97. package/dist/setup/steps/connection.js +2 -3
  98. package/dist/setup/steps/tasks.js +29 -36
  99. package/dist/sources/providers/git-install.js +17 -11
  100. package/dist/sources/providers/git-provider.js +12 -5
  101. package/dist/sources/providers/git-stash.js +38 -16
  102. package/dist/sources/snapshot-fetchers/website-ingest.js +3 -3
  103. package/dist/storage/repositories/embedding-salvage-repository.js +184 -0
  104. package/dist/storage/repositories/index-connection.js +3 -1
  105. package/dist/storage/repositories/index-entries-repository.js +68 -77
  106. package/dist/storage/repositories/index-entry-schema.js +25 -16
  107. package/dist/storage/repositories/index-fts-repository.js +263 -29
  108. package/dist/storage/repositories/index-meta-repository.js +29 -0
  109. package/dist/storage/repositories/index-schema.js +122 -115
  110. package/dist/storage/repositories/index-utility-repository.js +1 -1
  111. package/dist/storage/repositories/index-vec-repository.js +435 -22
  112. package/dist/tasks/activation-config.js +90 -0
  113. package/dist/tasks/backends/cron.js +9 -0
  114. package/dist/tasks/backends/launchd.js +1 -0
  115. package/dist/tasks/backends/schtasks.js +2 -0
  116. package/dist/tasks/embedded.js +4 -5
  117. package/dist/tasks/scheduler-binding.js +2 -2
  118. package/dist/tasks/scheduler-sync-preview.js +8 -1
  119. package/dist/tasks/scheduler-sync.js +19 -10
  120. package/dist/tasks/source/parse-task-source.js +10 -113
  121. package/dist/tasks/source/project-v4.js +2 -2
  122. package/dist/tasks/source/task-source-v4.js +4 -12
  123. package/dist/tasks/source/task-to-v3.js +4 -12
  124. package/dist/tasks/source/task-to-v4.js +40 -7
  125. package/docs/migration/README.md +1 -0
  126. package/docs/migration/release-notes/0.9.15.md +36 -34
  127. package/docs/migration/release-notes/0.9.16.md +60 -98
  128. package/docs/migration/release-notes/README.md +0 -5
  129. package/docs/migration/v0.9.1-to-v0.9.2.md +6 -9
  130. package/docs/reference/cli.md +124 -122
  131. package/docs/reference/configuration.md +137 -133
  132. package/docs/reference/data-and-telemetry.md +1 -2
  133. package/docs/reference/tasks.md +34 -29
  134. package/package.json +1 -1
  135. package/schemas/akm-config.json +170 -6
  136. package/schemas/akm-task.json +1 -2
  137. package/dist/commands/sources/index-status.js +0 -99
  138. package/dist/core/hash.js +0 -18
  139. package/dist/indexer/drain.js +0 -306
  140. package/dist/indexer/embedding-identity.js +0 -20
  141. package/dist/indexer/enrich.js +0 -260
  142. package/dist/indexer/reconcile.js +0 -890
  143. package/dist/indexer/scan/parse-file.js +0 -66
  144. package/dist/indexer/units/unit.js +0 -159
  145. package/dist/llm/embedders/provider-limits.js +0 -288
  146. package/dist/storage/repositories/files-repository.js +0 -181
  147. package/dist/storage/repositories/units-repository.js +0 -510
@@ -0,0 +1,785 @@
1
+ // This Source Code Form is subject to the terms of the Mozilla Public
2
+ // License, v. 2.0. If a copy of the MPL was not distributed with this
3
+ // file, You can obtain one at https://mozilla.org/MPL/2.0/.
4
+ import { getConfigPath } from "../core/paths.js";
5
+ import { isVerbose, warn, warnVerbose } from "../core/warn.js";
6
+ import { embedBatch } from "../llm/embedder.js";
7
+ import { DETERMINISTIC_EMBED_MODEL_ID, isDeterministicEmbedEnabled } from "../llm/embedders/deterministic.js";
8
+ import { DEFAULT_LOCAL_MODEL } from "../llm/embedders/local.js";
9
+ import { buildTokenBoundedBatches, capEmbeddingText, DEFAULT_MAX_INPUT_TOKENS, DEFAULT_REMOTE_BATCH_SIZE, DEFAULT_TOKEN_BUDGET, describeEmbeddingCredential, estimateTokenCount, hasRemoteEndpoint, normalizeEmbeddingEndpoint, } from "../llm/embedders/remote.js";
10
+ import { cosineSimilarity } from "../llm/embedders/types.js";
11
+ import { purgeEmbeddingSalvage, relabelEmbeddingSalvageFingerprint, reuseSalvagedEmbeddings, } from "../storage/repositories/embedding-salvage-repository.js";
12
+ import { getEmbeddableEntryCount } from "../storage/repositories/index-entries-repository.js";
13
+ import { deleteMeta, getMeta, setMeta } from "../storage/repositories/index-meta-repository.js";
14
+ import { EMBEDDING_DIM } from "../storage/repositories/index-schema.js";
15
+ import { getAllEntriesForEmbedding, getEmbeddingCount, isVecFastPathComplete, isVecFastPathReady, purgeEmbeddings, repairVecFastPath, sampleEmbeddedEntriesForCanary, setVecFastPathReady, upsertEmbedding, } from "../storage/repositories/index-vec-repository.js";
16
+ import { reclassifyIndexDbContention } from "./index-db-contention.js";
17
+ /** Identifies the embedding provider+model+dimension a stored vector was generated with. */
18
+ export function deriveSemanticProviderFingerprint(embedding) {
19
+ if (isDeterministicEmbedEnabled()) {
20
+ return `deterministic:${DETERMINISTIC_EMBED_MODEL_ID}`;
21
+ }
22
+ if (embedding?.endpoint) {
23
+ // Fingerprint keys on vector identity only (model + dimension). The endpoint
24
+ // is transport/routing and has no bearing on vector compatibility, so moving
25
+ // the same model+dimension to a different host must not force a full re-embed.
26
+ return `remote:${embedding.model}|${embedding.dimension ?? "default"}`;
27
+ }
28
+ return `local:${embedding?.localModel ?? DEFAULT_LOCAL_MODEL}`;
29
+ }
30
+ /**
31
+ * The heartbeat text emitted every 15s while a provider request is in
32
+ * flight, and by default (not `--verbose`-only) since silence indistinguishable
33
+ * from a hang was the field report's own symptom (#954).
34
+ */
35
+ export function formatEmbeddingHeartbeat(storedCount, total, failedCount) {
36
+ return `Still generating embeddings: ${storedCount}/${total} stored, ${failedCount} failed; waiting on embedding provider.`;
37
+ }
38
+ /**
39
+ * Number of already-embedded entries sampled for the fingerprint-rename
40
+ * canary (#955) — small and cheap even against a slow local server; a
41
+ * handful of chunks is a strong compatibility signal (a different model
42
+ * cannot plausibly land near-identical vectors by chance).
43
+ */
44
+ const CANARY_SAMPLE_SIZE = 8;
45
+ /**
46
+ * Minimum median cosine similarity between stored and freshly re-embedded
47
+ * canary vectors for a fingerprint-string change to be treated as a
48
+ * same-model rename rather than a real model change (#955).
49
+ */
50
+ const CANARY_SIMILARITY_THRESHOLD = 0.999;
51
+ /**
52
+ * Consecutive transport failures after which the embedding pass stops
53
+ * dispatching further requests and ends the run as a failure rather than
54
+ * grinding through every remaining batch against a dead endpoint (#954).
55
+ * Two independent trip conditions
56
+ * share this threshold — see `onSkip` below: 3 consecutive failures at
57
+ * single-document size (timeout OR network error — a multi-document
58
+ * timeout is not by itself evidence the endpoint is dead, since
59
+ * `RemoteEmbedder.embedBatch` already retries and splits it smaller before
60
+ * ever reporting it as failed at single-document size), or 3 consecutive
61
+ * network errors at ANY size (a network error is never retried, so it is
62
+ * trusted immediately regardless of how large the request was).
63
+ * `context-window-exceeded` never counts — that reason proves the provider
64
+ * IS reachable, and split-and-retry already handles it; it resets both
65
+ * streaks instead.
66
+ */
67
+ const CIRCUIT_BREAKER_THRESHOLD = 3;
68
+ /**
69
+ * Pure decision: do stored vectors remain valid against freshly re-embedded
70
+ * canary samples? The ONE place that computes the canary's similarity
71
+ * numbers — callers must use this result rather than recomputing it (#955).
72
+ *
73
+ * An empty sample means nothing is stored to lose or verify against, so
74
+ * there is nothing to decide — keep.
75
+ *
76
+ * A sample whose re-embed FAILED (`fresh === undefined`, e.g. a provider
77
+ * sub-batch that was skipped) is EXCLUDED from the similarity computation
78
+ * entirely, not scored as zero: a partial provider failure is not evidence
79
+ * of a different model (#955). A dimension mismatch on a successful
80
+ * re-embed still counts as zero similarity via {@link cosineSimilarity}'s
81
+ * own dimension-mismatch guard — that IS evidence. When half or fewer of
82
+ * the sampled entries re-embedded successfully, the sample is too thin to
83
+ * trust either verdict — the outcome is `unverifiable`, the same outcome a
84
+ * total canary failure already produces.
85
+ *
86
+ * Otherwise the MEDIAN pairwise cosine similarity of the verified samples
87
+ * must clear {@link CANARY_SIMILARITY_THRESHOLD}; the median (not the
88
+ * minimum or mean) tolerates one stale or lightly-edited sample without
89
+ * either discarding a real match or being fooled by it.
90
+ */
91
+ export function decideEmbeddingCompatibility(pairs) {
92
+ if (pairs.length === 0)
93
+ return { outcome: "keep", medianSimilarity: undefined, verifiedSamples: 0 };
94
+ const verified = pairs.filter((pair) => pair.fresh !== undefined);
95
+ if (verified.length * 2 <= pairs.length) {
96
+ return { outcome: "unverifiable", medianSimilarity: undefined, verifiedSamples: verified.length };
97
+ }
98
+ const similarities = verified.map((pair) => cosineSimilarity(pair.stored, pair.fresh));
99
+ const medianSimilarity = medianOf(similarities);
100
+ return {
101
+ outcome: medianSimilarity >= CANARY_SIMILARITY_THRESHOLD ? "keep" : "rebuild",
102
+ medianSimilarity,
103
+ verifiedSamples: verified.length,
104
+ };
105
+ }
106
+ function medianOf(values) {
107
+ const sorted = [...values].sort((a, b) => a - b);
108
+ const mid = Math.floor(sorted.length / 2);
109
+ return sorted.length % 2 === 0
110
+ ? (sorted[mid - 1] + sorted[mid]) / 2
111
+ : sorted[mid];
112
+ }
113
+ /**
114
+ * Identity of the embedding vectors actually observed on a run — as opposed
115
+ * to {@link deriveSemanticProviderFingerprint}'s CONFIG-derived string. Keys
116
+ * on what the server (or local model) actually reported plus the observed
117
+ * vector width, so a gateway/transport change that keeps returning the same
118
+ * underlying model can be told apart from a genuine model change without
119
+ * relying on the operator's config string (#955).
120
+ * Returns undefined when nothing was actually observed this call (no vector
121
+ * to measure yet).
122
+ */
123
+ function deriveObservedEmbeddingIdentity(embedding, observedModel, observedVectorLen) {
124
+ if (isDeterministicEmbedEnabled()) {
125
+ return `deterministic:${DETERMINISTIC_EMBED_MODEL_ID}`;
126
+ }
127
+ if (observedVectorLen === undefined)
128
+ return undefined;
129
+ if (embedding?.endpoint) {
130
+ return `remote:${observedModel ?? embedding.model ?? "unknown"}|${observedVectorLen}`;
131
+ }
132
+ return `local:${embedding?.localModel ?? DEFAULT_LOCAL_MODEL}|${observedVectorLen}`;
133
+ }
134
+ /**
135
+ * Run the fingerprint-rename canary: re-embed a small sample of already-
136
+ * stored entries with the CURRENT config and decide whether the stored
137
+ * index survives. Goes through the standard {@link embedBatch} facade (not a
138
+ * direct `RemoteEmbedder`) so every embedder branch — remote, local,
139
+ * deterministic, and test overrides via `_setEmbedderForTests` — is
140
+ * exercised identically to the main embedding pass.
141
+ *
142
+ * `maxInputTokens` must be the SAME cap the main pass below applies via
143
+ * {@link capEmbeddingText} — the stored vector for each sampled entry was
144
+ * produced from its capped text, so comparing against a fresh vector of the
145
+ * uncapped text would compare unlike inputs for any entry over the cap
146
+ * (#955).
147
+ */
148
+ async function runEmbeddingCanary(db, config, signal, maxInputTokens) {
149
+ const samples = sampleEmbeddedEntriesForCanary(db, CANARY_SAMPLE_SIZE);
150
+ if (samples.length === 0) {
151
+ return { outcome: "keep", verified: false, viaIdentityMatch: false };
152
+ }
153
+ let observedModel;
154
+ const skips = [];
155
+ let canaryVectors;
156
+ try {
157
+ canaryVectors = await embedBatch(
158
+ // #955: the stored vector for each sample was produced from
159
+ // capEmbeddingText(searchText, maxInputTokens) — the main pass below
160
+ // caps every document before embedding it. The canary must re-embed
161
+ // the SAME capped text, or an entry over the cap compares a fresh
162
+ // vector of a different input against a stored vector of the capped
163
+ // one, and a genuine model match can read as a rebuild-worthy
164
+ // mismatch for reasons unrelated to the model.
165
+ samples.map((sample) => capEmbeddingText(sample.searchText, maxInputTokens).text), config.embedding, signal, (skip) => skips.push(skip), (_indices, _embeddings, model) => {
166
+ if (model)
167
+ observedModel = model;
168
+ });
169
+ }
170
+ catch (error) {
171
+ const message = error instanceof Error ? error.message : String(error);
172
+ return {
173
+ outcome: "unverifiable",
174
+ message: `could not verify embedding compatibility (${message}); keeping existing vectors — rerun akm index when the endpoint is reachable`,
175
+ };
176
+ }
177
+ const observedVectorLen = canaryVectors.find((vector) => vector !== undefined)?.length;
178
+ const observedIdentity = deriveObservedEmbeddingIdentity(config.embedding, observedModel, observedVectorLen);
179
+ const storedIdentity = getMeta(db, "embeddingIdentity");
180
+ if (storedIdentity && observedIdentity && storedIdentity === observedIdentity) {
181
+ // The server reports the same model identity as last time — no need to
182
+ // even look at the cosines; the config string alone was misleading.
183
+ return { outcome: "keep", verified: true, identity: observedIdentity, viaIdentityMatch: true };
184
+ }
185
+ const pairs = samples.map((sample, i) => ({ stored: sample.vector, fresh: canaryVectors[i] }));
186
+ const decision = decideEmbeddingCompatibility(pairs);
187
+ if (decision.outcome === "unverifiable") {
188
+ // Covers both a total provider failure (RemoteEmbedder skips a failing
189
+ // request rather than throwing, #874, so an unreachable endpoint
190
+ // surfaces here as an all-`undefined` canary result, not a caught
191
+ // exception) and a partial one thin enough that neither verdict can be
192
+ // trusted (#955) — same message path either way.
193
+ const message = skips[0]?.message ?? "embedding provider returned no vectors for the canary sample";
194
+ return {
195
+ outcome: "unverifiable",
196
+ message: `could not verify embedding compatibility (${message}); keeping existing vectors — rerun akm index when the endpoint is reachable`,
197
+ };
198
+ }
199
+ if (decision.outcome === "keep") {
200
+ return {
201
+ outcome: "keep",
202
+ verified: true,
203
+ identity: observedIdentity,
204
+ viaIdentityMatch: false,
205
+ medianSimilarity: decision.medianSimilarity,
206
+ };
207
+ }
208
+ return {
209
+ outcome: "rebuild",
210
+ identity: observedIdentity,
211
+ reason: `vectors differ (median similarity ${decision.medianSimilarity?.toFixed(3)})`,
212
+ };
213
+ }
214
+ function throwIfAborted(signal) {
215
+ if (signal?.aborted) {
216
+ throw signal.reason instanceof Error ? signal.reason : new Error("index interrupted");
217
+ }
218
+ }
219
+ export async function generateEmbeddingsForDb(db, config, onProgress, signal, entryIds, opts) {
220
+ // Drift guard (#954): refuse an ambient transaction. Every
221
+ // per-batch `db.transaction()` below is meant to be its own durable commit
222
+ // (#954) — inside an already-open outer transaction it would nest as an
223
+ // unobservable SAVEPOINT instead, so an interruption (competing-process
224
+ // collision, SIGKILL) could lose the whole pass rather than only the batch
225
+ // in flight. This is an internal contract error (a caller bug), not a
226
+ // user-facing failure class: callers with their own transaction (e.g. `akm
227
+ // bundle update`'s unified update transaction) must run the embedding
228
+ // phase on a separate connection AFTER their own transaction commits — see
229
+ // `runEmbeddingPass` in `src/indexer/indexer.ts`.
230
+ if (db.inTransaction) {
231
+ throw new Error("generateEmbeddingsForDb was called with an ambient transaction already open on `db`: per-batch commits " +
232
+ "would become SAVEPOINTs inside it, losing the crash-durability contract per-batch commit exists for. " +
233
+ "Run the embedding phase on a connection with no open transaction.");
234
+ }
235
+ throwIfAborted(signal);
236
+ if (config.semanticSearchMode === "off") {
237
+ // #955: salvage is self-emptying only if every path that skips reuse
238
+ // also drains it — otherwise a full rebuild performed with semantic
239
+ // search disabled leaves permanent orphaned rows behind (nothing will
240
+ // ever consume them, since this path never reaches the reuse step).
241
+ purgeEmbeddingSalvage(db);
242
+ onProgress({ phase: "embeddings", message: "Semantic search disabled; skipping embeddings." });
243
+ return { success: false, message: "Semantic search is disabled." };
244
+ }
245
+ // #953 field gap: the actionable outcome is a self-diagnosing run, not a
246
+ // fix (every RemoteEmbedder path already resolves secret:// through one
247
+ // boundary — a keyless request can only mean embedding.apiKey was absent
248
+ // from the config THIS run loaded). One default-level line, before the
249
+ // first provider request of the phase (the canary probe or the main
250
+ // pass, whichever runs first below), naming the endpoint/model/credential
251
+ // SOURCE — never the credential value.
252
+ if (hasRemoteEndpoint(config.embedding ?? {})) {
253
+ const endpoint = normalizeEmbeddingEndpoint(config.embedding?.endpoint ?? "");
254
+ const credential = describeEmbeddingCredential(config.embedding?.apiKey);
255
+ const configFileSuffix = isVerbose() ? `; config: ${getConfigPath()}` : "";
256
+ onProgress({
257
+ phase: "embeddings",
258
+ message: `[embed] endpoint ${endpoint}, model ${config.embedding?.model ?? "unknown"}; credential: ${credential}${configFileSuffix}`,
259
+ });
260
+ }
261
+ // A targeted call starts from an already-published generation. Preserve its
262
+ // trust decision in O(1): successful writes for the changed IDs keep a
263
+ // healthy fast path healthy, but can never promote a generation already
264
+ // marked degraded. Global runs can afford to verify the entire derived set.
265
+ const vecFastPathWasReady = isVecFastPathReady(db);
266
+ const currentFingerprint = deriveSemanticProviderFingerprint(config.embedding);
267
+ const storedFingerprint = getMeta(db, "embeddingFingerprint");
268
+ let targetEntryIds = entryIds;
269
+ /** Set only on an actual rebuild, so the up-front "Re-embedding N entries" line names why. */
270
+ let rebuildReason;
271
+ // Resolved once and reused by both the canary (below) and the main pass's
272
+ // cap loop (further down) — the same cap must apply to both, or the canary
273
+ // compares a differently-capped text against the stored vector (#955).
274
+ const maxInputTokens = config.embedding?.maxInputTokens ?? DEFAULT_MAX_INPUT_TOKENS;
275
+ if (opts?.forceReembed) {
276
+ // `akm index --reembed`: an explicit operator override, skips the canary
277
+ // entirely. The new fingerprint (and identity, now stale/unknown until
278
+ // the next successful pass observes it) is written in the SAME
279
+ // transaction as the purge, before any embedding request — a restart
280
+ // then sees a matching fingerprint and only heals what is still missing
281
+ // instead of purging again from zero (#955/#956).
282
+ db.transaction(() => {
283
+ purgeEmbeddings(db, { dropVecTable: true });
284
+ // #955: an explicit forced rebuild must re-embed everything, not
285
+ // quietly satisfy some of it from stale salvage.
286
+ purgeEmbeddingSalvage(db);
287
+ deleteMeta(db, "embeddingDim");
288
+ setMeta(db, "embeddingFingerprint", currentFingerprint);
289
+ deleteMeta(db, "embeddingIdentity");
290
+ })();
291
+ targetEntryIds = undefined;
292
+ rebuildReason = "forced by --reembed";
293
+ }
294
+ else if (storedFingerprint && storedFingerprint !== currentFingerprint) {
295
+ const decision = await runEmbeddingCanary(db, config, signal, maxInputTokens);
296
+ if (decision.outcome === "unverifiable") {
297
+ // Destroying a good index because the server happens to be down right
298
+ // now is worse than leaving a rename unverified until the next run —
299
+ // keep the vectors AND the old fingerprint so the next `akm index`
300
+ // retries the canary instead of silently treating this as resolved.
301
+ warn(`[embed] ${decision.message}`);
302
+ onProgress({ phase: "embeddings", message: decision.message });
303
+ return { success: false, message: decision.message };
304
+ }
305
+ if (decision.outcome === "rebuild") {
306
+ db.transaction(() => {
307
+ purgeEmbeddings(db, { dropVecTable: true });
308
+ // #955: the stored vectors AND any leftover salvage both belong to
309
+ // a different model now — neither is reusable, so both go.
310
+ purgeEmbeddingSalvage(db);
311
+ deleteMeta(db, "embeddingDim");
312
+ setMeta(db, "embeddingFingerprint", currentFingerprint);
313
+ if (decision.identity)
314
+ setMeta(db, "embeddingIdentity", decision.identity);
315
+ else
316
+ deleteMeta(db, "embeddingIdentity");
317
+ })();
318
+ targetEntryIds = undefined;
319
+ rebuildReason = decision.reason;
320
+ }
321
+ else {
322
+ // Keep: adopt the new fingerprint (and identity, when observed)
323
+ // immediately rather than deferring to end-of-run — nothing was
324
+ // purged, so there is nothing an interruption could lose, and an
325
+ // immediate write means a crash right after this decision does not
326
+ // re-run the canary needlessly on the next attempt.
327
+ setMeta(db, "embeddingFingerprint", currentFingerprint);
328
+ // #955: the model did not actually change, only the fingerprint
329
+ // STRING did (e.g. a gateway rename) — any leftover salvage rows
330
+ // tagged with the OLD string are still valid vectors. Relabel them so
331
+ // the reuse step below (and any later pass) can still find them.
332
+ relabelEmbeddingSalvageFingerprint(db, storedFingerprint, currentFingerprint);
333
+ if (decision.identity)
334
+ setMeta(db, "embeddingIdentity", decision.identity);
335
+ if (decision.verified) {
336
+ const keptCount = getEmbeddingCount(db);
337
+ const detail = decision.viaIdentityMatch
338
+ ? "server-reported model unchanged"
339
+ : `stored vectors are compatible (median similarity ${decision.medianSimilarity?.toFixed(3)})`;
340
+ const message = `[embed] embedding model renamed (${storedFingerprint} → ${currentFingerprint}); ${detail}, keeping ${keptCount} embedding${keptCount === 1 ? "" : "s"}.`;
341
+ warn(message);
342
+ onProgress({ phase: "embeddings", message });
343
+ }
344
+ // Empty-sample case (decision.verified === false): nothing stored to
345
+ // lose or verify against — adopt the label silently, no purge line.
346
+ }
347
+ }
348
+ else {
349
+ // No rename to verify (either this is the very first
350
+ // pass ever for this db, or the fingerprint already matches the last
351
+ // successful one) — still record it NOW rather than deferring to a
352
+ // fully successful pass, mirroring the rebuild/keep branches above
353
+ // (#955/#956). Without this, an interrupted FIRST-EVER pass left
354
+ // `embeddingFingerprint` unset despite a per-batch commit below (#954)
355
+ // already having durably written real vectors — a later `akm index
356
+ // --full`'s salvage-before-discard step tags rows by this meta
357
+ // (`salvageEmbeddingsBeforeDiscard`) and treats an unset fingerprint as
358
+ // "nothing was ever verified", silently turning genuinely-embedded
359
+ // vectors into a full re-embed instead of a salvage-and-reuse.
360
+ setMeta(db, "embeddingFingerprint", currentFingerprint);
361
+ }
362
+ try {
363
+ throwIfAborted(signal);
364
+ if (entryIds === undefined && (!vecFastPathWasReady || !isVecFastPathComplete(db))) {
365
+ const storedDim = Number(getMeta(db, "embeddingDim"));
366
+ const expectedDim = Number.isInteger(storedDim) && storedDim > 0 ? storedDim : (config.embedding?.dimension ?? EMBEDDING_DIM);
367
+ const repair = repairVecFastPath(db, expectedDim);
368
+ if (repair.available &&
369
+ (repair.repaired > 0 || repair.removedOrphans > 0 || repair.rejected > 0 || repair.error !== undefined)) {
370
+ const detail = repair.error ? `; repair stopped: ${repair.error}` : "";
371
+ onProgress({
372
+ phase: "embeddings",
373
+ message: `[embed] Repaired ${repair.repaired} missing sqlite-vec row${repair.repaired === 1 ? "" : "s"}; removed ${repair.removedOrphans} orphan${repair.removedOrphans === 1 ? "" : "s"}; ${repair.rejected} rejected${detail}.`,
374
+ });
375
+ }
376
+ }
377
+ const allEntries = getAllEntriesForEmbedding(db, targetEntryIds);
378
+ let vecFailedCount = 0;
379
+ let vecUnavailableCount = 0;
380
+ // #955: before any provider call, hand back vectors salvaged from a
381
+ // full rebuild or a generation bump for entries whose search_text is
382
+ // byte-identical to what was salvaged under the SAME fingerprint — a
383
+ // fingerprint mismatch or a single-byte content change both correctly
384
+ // fall through to the provider below instead.
385
+ const { reusedCount, remaining: candidateEntries } = reuseSalvagedEmbeddings(db, allEntries, currentFingerprint, (entry, embedding) => {
386
+ const result = upsertEmbedding(db, entry.id, embedding);
387
+ if (result.vec === "failed")
388
+ vecFailedCount++;
389
+ if (result.vec === "unavailable")
390
+ vecUnavailableCount++;
391
+ return result.stored;
392
+ });
393
+ if (reusedCount > 0) {
394
+ onProgress({
395
+ phase: "embeddings",
396
+ message: `Reused ${reusedCount} embedding${reusedCount === 1 ? "" : "s"} from the previous generation; embedding ${candidateEntries.length} new.`,
397
+ });
398
+ }
399
+ if (candidateEntries.length === 0) {
400
+ onProgress({ phase: "embeddings", message: "Embeddings already up to date." });
401
+ setMeta(db, "embeddingFingerprint", currentFingerprint);
402
+ if (reusedCount > 0) {
403
+ const vecGenerationComplete = targetEntryIds === undefined ? isVecFastPathComplete(db) : vecFastPathWasReady;
404
+ setVecFastPathReady(db, vecFailedCount === 0 && vecUnavailableCount === 0 && vecGenerationComplete);
405
+ }
406
+ // A pass that completes (even one that did nothing but reuse) purges
407
+ // whatever is left — salvage is consumed by the NEXT pass, never kept
408
+ // around as a second cache.
409
+ purgeEmbeddingSalvage(db);
410
+ return reusedCount > 0 ? { success: true, vecInsertFailures: vecFailedCount } : { success: true };
411
+ }
412
+ // Cap each document's embedded text at
413
+ // embedding.maxInputTokens (default DEFAULT_MAX_INPUT_TOKENS, resolved
414
+ // once above so the canary uses the identical cap) instead of ever
415
+ // failing a whole batch over one oversized entry — truncation keeps the
416
+ // head of the text, unicode-safe. A document is skipped only when its
417
+ // head is empty (the impossible case: nothing left to embed), never
418
+ // merely for being long.
419
+ let truncatedCount = 0;
420
+ const texts = [];
421
+ const pendingEntries = [];
422
+ for (const entry of candidateEntries) {
423
+ const capped = capEmbeddingText(entry.searchText, maxInputTokens);
424
+ if (capped.text.length === 0)
425
+ continue;
426
+ if (capped.truncated)
427
+ truncatedCount++;
428
+ pendingEntries.push(entry);
429
+ texts.push(capped.text);
430
+ }
431
+ if (truncatedCount > 0) {
432
+ // Through onProgress ONLY, not warn() too — onProgress already reaches
433
+ // stderr at the default level in every output mode (#954), and the
434
+ // index CLI's progress handler writes it through info() (log-file
435
+ // aware), so calling warn() as well printed the identical sentence
436
+ // twice in text mode.
437
+ const message = `[embed] ${truncatedCount} entr${truncatedCount === 1 ? "y" : "ies"} truncated to the ${maxInputTokens}-token embedding cap (embedding.maxInputTokens); rerun with a higher cap to embed the full text.`;
438
+ onProgress({ phase: "embeddings", message });
439
+ }
440
+ if (rebuildReason) {
441
+ // See the truncation notice above: onProgress ONLY.
442
+ const message = `[embed] Re-embedding ${pendingEntries.length} entr${pendingEntries.length === 1 ? "y" : "ies"} because ${rebuildReason}`;
443
+ onProgress({ phase: "embeddings", message });
444
+ }
445
+ onProgress({
446
+ phase: "embeddings",
447
+ message: `Generating embeddings for ${pendingEntries.length} entr${pendingEntries.length === 1 ? "y" : "ies"}.`,
448
+ });
449
+ if (isVerbose()) {
450
+ // Mirror RemoteEmbedder's actual token-bounded batching (#874) so this
451
+ // log reflects the real request grouping rather than a fixed count of
452
+ // 100 that no longer matches what gets sent over the wire. Local runs
453
+ // don't batch by size at all (LocalEmbedder chunks by a fixed count
454
+ // for inference throughput only, never fails/skips), so there's
455
+ // nothing meaningful to report per-batch for them.
456
+ if (hasRemoteEndpoint(config.embedding ?? {})) {
457
+ // Mirrors RemoteEmbedder.embedBatch's own tokenBudget resolution
458
+ // (#956: contextLength no longer feeds this).
459
+ const tokenBudget = config.embedding?.maxTokens ?? DEFAULT_TOKEN_BUDGET;
460
+ const maxCount = config.embedding?.batchSize ?? DEFAULT_REMOTE_BATCH_SIZE;
461
+ const batches = buildTokenBoundedBatches(texts, tokenBudget, maxCount);
462
+ const batchNumberByIndex = new Map();
463
+ batches.forEach((batch, batchIdx) => {
464
+ for (const i of batch.indices)
465
+ batchNumberByIndex.set(i, batchIdx + 1);
466
+ });
467
+ for (const [i, entry] of pendingEntries.entries()) {
468
+ const chars = entry.searchText.length;
469
+ const tokens = estimateTokenCount(entry.searchText);
470
+ const batch = batches[batchNumberByIndex.get(i) - 1];
471
+ const label = batch?.oversized
472
+ ? "oversized (skipped)"
473
+ : `batch ${batchNumberByIndex.get(i)}/${batches.length}`;
474
+ warnVerbose(`[embed] ${entry.itemRef} (${chars} chars, est. ${tokens} tokens) → ${label}`);
475
+ }
476
+ }
477
+ else {
478
+ for (const entry of pendingEntries) {
479
+ warnVerbose(`[embed] ${entry.itemRef} (${entry.searchText.length} chars, est. ${estimateTokenCount(entry.searchText)} tokens)`);
480
+ }
481
+ }
482
+ }
483
+ let heartbeatTimer;
484
+ let storedCount = 0;
485
+ let skippedCount = 0;
486
+ let embedFailedCount = 0;
487
+ let storedTokens = 0;
488
+ try {
489
+ heartbeatTimer = setInterval(() => {
490
+ onProgress({
491
+ phase: "embeddings",
492
+ message: formatEmbeddingHeartbeat(storedCount, pendingEntries.length, embedFailedCount),
493
+ });
494
+ }, 15000);
495
+ // A failing sub-batch or an oversized document is SKIPPED by embedBatch,
496
+ // not thrown (#874) — collect what couldn't be embedded and why, so a
497
+ // few bad documents don't discard every other entry's embedding.
498
+ const skips = [];
499
+ const embedStart = Date.now();
500
+ // Circuit breaker (#954): stop
501
+ // dispatching further batches once either consecutive-failure streak
502
+ // below reaches CIRCUIT_BREAKER_THRESHOLD — a dead/hung provider used
503
+ // to grind through every remaining batch for hours, one 30s (now
504
+ // configurable, and now backed off/retried/split first — see
505
+ // RemoteEmbedder.embedBatch) timeout at a time, ending in one
506
+ // aggregate warning and `ok: true`. Counted per BATCH
507
+ // (`skip.batchStart`), not per document: a single failed 100-document
508
+ // batch must not look like 100 consecutive failures.
509
+ let consecutiveSingleDocFailures = 0;
510
+ let consecutiveNetworkErrorFailures = 0;
511
+ let circuitBreakerReason;
512
+ const onSkip = (skip) => {
513
+ skips.push(skip);
514
+ if (!skip.batchStart)
515
+ return undefined;
516
+ if (skip.reason === "context-window-exceeded") {
517
+ consecutiveSingleDocFailures = 0;
518
+ consecutiveNetworkErrorFailures = 0;
519
+ return undefined;
520
+ }
521
+ // "batch-request-failed": a timeout only counts once retries have
522
+ // already narrowed it down to a single document (embedBatch backs
523
+ // off, retries, and splits a multi-document timeout before ever
524
+ // reporting it here); a network error counts immediately at any
525
+ // size — it was never retried, so it is trusted right away.
526
+ consecutiveSingleDocFailures = skip.batchSize === 1 ? consecutiveSingleDocFailures + 1 : 0;
527
+ consecutiveNetworkErrorFailures =
528
+ skip.failureKind === "network-error" ? consecutiveNetworkErrorFailures + 1 : 0;
529
+ if (consecutiveSingleDocFailures >= CIRCUIT_BREAKER_THRESHOLD ||
530
+ consecutiveNetworkErrorFailures >= CIRCUIT_BREAKER_THRESHOLD) {
531
+ circuitBreakerReason = skip.message;
532
+ return false;
533
+ }
534
+ return undefined;
535
+ };
536
+ // Commit each provider batch in its own short transaction as it lands,
537
+ // rather than buffering the whole run in memory for one transaction at
538
+ // the very end (#954) — a competing-process lock error or any other
539
+ // interruption partway through now keeps whatever already committed
540
+ // instead of losing the entire pass.
541
+ // Tracks what this run actually observed, so a successful pass can
542
+ // record `embeddingIdentity` from real data rather than the config
543
+ // string alone (#955) — only the first non-empty batch's vector width
544
+ // is kept; every batch from one run shares the same provider/model.
545
+ let observedModel;
546
+ let observedVectorLen;
547
+ // Whether the remote provider's endpoint/model/token language is
548
+ // meaningful for this run — the per-batch diagnostic line below is
549
+ // remote-only, same gate the credential diagnostic (#953) above uses.
550
+ const reportPerBatchLine = hasRemoteEndpoint(config.embedding ?? {});
551
+ const onBatch = (indices, batchEmbeddings, model, outcome) => {
552
+ // #954 field-report follow-up: a "retrying" event carries nothing to
553
+ // commit — the request hasn't settled yet — only the notice that a
554
+ // back-off is about to be waited out, default-level so a run is
555
+ // never silently stalled indistinguishably from a hang.
556
+ if (outcome?.outcome === "retrying") {
557
+ if (reportPerBatchLine) {
558
+ onProgress({
559
+ phase: "embeddings",
560
+ message: `[embed] batch ${outcome.batchIndex}/${outcome.batchCount}: ${outcome.docCount} docs, ${outcome.requestTokens.toLocaleString()} tokens → retrying after ${(outcome.elapsedMs / 1000).toFixed(1)} s`,
561
+ });
562
+ }
563
+ return;
564
+ }
565
+ // #954: a "budget-lowered" event is the same kind of notice as
566
+ // "retrying" above — the run's first context-size rejection just
567
+ // shrank the request budget for everything not yet dispatched, but
568
+ // THIS rejected batch's own indices are still being split and
569
+ // retried by the embedder (their real stored/failed outcome lands in
570
+ // a later onBatch call). Nothing here has settled, so it must never
571
+ // touch storage, only report the notice — one line, at most once per
572
+ // run.
573
+ if (outcome?.outcome === "budget-lowered") {
574
+ if (reportPerBatchLine) {
575
+ onProgress({
576
+ phase: "embeddings",
577
+ message: `[embed] batch ${outcome.batchIndex}/${outcome.batchCount}: ${outcome.docCount} docs, ${outcome.requestTokens.toLocaleString()} tokens → ${outcome.reason}`,
578
+ });
579
+ }
580
+ return;
581
+ }
582
+ if (model)
583
+ observedModel = model;
584
+ // A batch that delivered at least one real embedding proves the
585
+ // provider is currently answering — reset both circuit-breaker
586
+ // streaks. (A wholly failed batch's `batchEmbeddings` are all
587
+ // `undefined`, per commitBatch's skip path, so this never
588
+ // re-triggers what onSkip just counted moments earlier.)
589
+ if (batchEmbeddings.some((embedding) => embedding !== undefined)) {
590
+ consecutiveSingleDocFailures = 0;
591
+ consecutiveNetworkErrorFailures = 0;
592
+ }
593
+ db.transaction(() => {
594
+ for (let k = 0; k < indices.length; k++) {
595
+ const index = indices[k];
596
+ const entry = pendingEntries[index];
597
+ if (!entry)
598
+ continue;
599
+ const embedding = batchEmbeddings[k];
600
+ if (!embedding) {
601
+ embedFailedCount++;
602
+ continue;
603
+ }
604
+ if (observedVectorLen === undefined)
605
+ observedVectorLen = embedding.length;
606
+ const result = upsertEmbedding(db, entry.id, embedding);
607
+ if (result.stored) {
608
+ storedCount++;
609
+ // #954: sum the estimate of the text actually sent —
610
+ // `texts[index]` is the capped string `embedBatch` was handed,
611
+ // parallel to `pendingEntries` by construction above (the
612
+ // `entry` guard covers both) — not `entry.searchText`, which is
613
+ // the pre-cap original and overstates throughput for every
614
+ // entry over the cap.
615
+ storedTokens += estimateTokenCount(texts[index]);
616
+ }
617
+ else {
618
+ skippedCount++;
619
+ }
620
+ if (result.vec === "failed")
621
+ vecFailedCount++;
622
+ if (result.vec === "unavailable")
623
+ vecUnavailableCount++;
624
+ }
625
+ })();
626
+ // Default level, one line per provider batch (#954, field-report
627
+ // follow-up): oversized documents never made a request
628
+ // (`reason === "oversized"`), so there is no batch outcome to
629
+ // report — they are covered by the run's final oversized-skip
630
+ // count and list instead.
631
+ if (outcome && outcome.reason !== "oversized" && reportPerBatchLine) {
632
+ const elapsedSeconds = (outcome.elapsedMs / 1000).toFixed(1);
633
+ const outcomeLabel = outcome.outcome === "stored"
634
+ ? `${outcome.docCount} stored (${elapsedSeconds} s)`
635
+ : `failed: ${outcome.reason}`;
636
+ onProgress({
637
+ phase: "embeddings",
638
+ message: `[embed] batch ${outcome.batchIndex}/${outcome.batchCount}: ${outcome.docCount} docs, ${outcome.requestTokens.toLocaleString()} tokens → ${outcomeLabel}`,
639
+ });
640
+ }
641
+ // Every committed batch, not just every 500 stored entries (#954)
642
+ // — the prior bucketing left a non-verbose run silent
643
+ // for the entire embedding phase on anything smaller than 500
644
+ // entries, indistinguishable from a hang.
645
+ onProgress({
646
+ phase: "embeddings",
647
+ message: `Embedded ${storedCount}/${pendingEntries.length} entries.`,
648
+ });
649
+ };
650
+ await embedBatch(texts, config.embedding, signal, onSkip, onBatch);
651
+ throwIfAborted(signal);
652
+ const elapsedSeconds = Math.max((Date.now() - embedStart) / 1000, 0.001);
653
+ if (skippedCount > 0) {
654
+ warn(`[embed] ${skippedCount} embedding${skippedCount === 1 ? "" : "s"} skipped (entry deleted between queue and write)`);
655
+ }
656
+ const vecGenerationComplete = targetEntryIds === undefined ? isVecFastPathComplete(db) : vecFastPathWasReady;
657
+ setVecFastPathReady(db, vecFailedCount === 0 && vecUnavailableCount === 0 && vecGenerationComplete);
658
+ if (vecFailedCount > 0) {
659
+ warn(`[embed] ${vecFailedCount} sqlite-vec fast-path insert${vecFailedCount === 1 ? "" : "s"} failed — ` +
660
+ "semantic search will use the slower JS-cosine fallback over stored embeddings. " +
661
+ "Rebuild with 'akm index --full' after resolving the vec table (often a vector-dimension mismatch).");
662
+ }
663
+ const entriesPerSec = storedCount / elapsedSeconds;
664
+ const tokensPerSec = storedTokens / elapsedSeconds;
665
+ const totalStored = storedCount + reusedCount;
666
+ // #954, field-report follow-up: the final line
667
+ // reports every outcome, not just what was stored — counts come from
668
+ // the same collected `skips` the circuit breaker already uses,
669
+ // categorized by `reason`/`failureKind`. "oversized skipped" =
670
+ // context-window-exceeded (never fit any request, at any size);
671
+ // "timed out" = a batch-request-failed skip whose last attempt timed
672
+ // out (retries/splits already exhausted before this counted); "failed"
673
+ // = every other batch-request-failed skip (a genuine, never-retried
674
+ // network/HTTP failure).
675
+ const oversizedSkips = skips.filter((skip) => skip.reason === "context-window-exceeded");
676
+ const timedOutSkips = skips.filter((skip) => skip.reason === "batch-request-failed" && skip.failureKind === "timeout");
677
+ const failedSkips = skips.filter((skip) => skip.reason === "batch-request-failed" && skip.failureKind !== "timeout");
678
+ const throughputLine = reusedCount > 0
679
+ ? // #955: report reused and newly-embedded counts separately — the
680
+ // rate figures below are provider throughput only (reuse is a
681
+ // plain DB write, not provider work) and would be misleadingly
682
+ // inflated if reused entries were folded into them.
683
+ `Stored ${totalStored} embedding${totalStored === 1 ? "" : "s"} (${reusedCount} reused, ${storedCount} newly embedded) in ${elapsedSeconds.toFixed(1)}s (${entriesPerSec.toFixed(1)} entries/s, ~${Math.round(tokensPerSec)} tokens/s)`
684
+ : `Stored ${storedCount} embedding${storedCount === 1 ? "" : "s"} in ${elapsedSeconds.toFixed(1)}s (${entriesPerSec.toFixed(1)} entries/s, ~${Math.round(tokensPerSec)} tokens/s)`;
685
+ onProgress({
686
+ phase: "embeddings",
687
+ message: `${throughputLine}; ${oversizedSkips.length} oversized skipped, ${timedOutSkips.length} timed out, ${failedSkips.length} failed.`,
688
+ });
689
+ // Bounded itemRef-level detail for every skip category, not just
690
+ // oversized — the aggregate counts above say HOW MANY documents timed
691
+ // out or failed, but give the operator no way to find out WHICH ones
692
+ // short of rerunning with --verbose and re-reading the whole log.
693
+ // Default level caps each list (there is nothing actionable about the
694
+ // 21st identical failure); --verbose prints every one, matching the
695
+ // per-document mapping lines' own verbosity gate above.
696
+ const printSkipList = (label, skipList) => {
697
+ if (skipList.length === 0)
698
+ return;
699
+ const limit = isVerbose() ? skipList.length : 20;
700
+ const listed = skipList
701
+ .slice(0, limit)
702
+ .map((skip) => ` - ${pendingEntries[skip.index]?.itemRef ?? skip.index}: ${skip.message}`)
703
+ .join("\n");
704
+ const more = skipList.length > limit ? `\n ...and ${skipList.length - limit} more` : "";
705
+ onProgress({ phase: "embeddings", message: `[embed] ${label} skipped:\n${listed}${more}` });
706
+ };
707
+ printSkipList("oversized documents", oversizedSkips);
708
+ printSkipList("timed-out documents", timedOutSkips);
709
+ printSkipList("failed documents", failedSkips);
710
+ setMeta(db, "embeddingFingerprint", currentFingerprint);
711
+ const observedIdentity = deriveObservedEmbeddingIdentity(config.embedding, observedModel, observedVectorLen);
712
+ if (observedIdentity)
713
+ setMeta(db, "embeddingIdentity", observedIdentity);
714
+ // Circuit breaker tripped (#954): committed batches are
715
+ // kept (nothing above discards them), but the pass is not a success —
716
+ // the provider looks dead, not just occasionally flaky.
717
+ if (circuitBreakerReason !== undefined) {
718
+ const message = `embedding provider failed ${CIRCUIT_BREAKER_THRESHOLD} consecutive batches ` +
719
+ `(last: ${circuitBreakerReason}); stopped after ${storedCount} embedding${storedCount === 1 ? "" : "s"} ` +
720
+ "were stored — rerun akm index when the endpoint is healthy";
721
+ warn(`[embed] ${message}`);
722
+ onProgress({ phase: "embeddings", message });
723
+ return { success: false, message, vecInsertFailures: vecFailedCount };
724
+ }
725
+ // Only a total failure (nothing at all embedded, despite having entries
726
+ // to embed) turns into a phase failure. Any partial success — the vast
727
+ // majority of a large bundle embedding fine around a handful of skips —
728
+ // must not discard what DID get stored (#874).
729
+ if (storedCount === 0 && embedFailedCount > 0) {
730
+ const firstMessage = skips[0]?.message ?? "All embeddings failed.";
731
+ // #873 removed the persisted semantic verdict, so there is no failure
732
+ // class to record — just report what happened on this run.
733
+ return {
734
+ success: false,
735
+ message: `All ${embedFailedCount} embedding batch(es) failed: ${firstMessage}`,
736
+ };
737
+ }
738
+ // A pass that completes without abort or circuit-break purges
739
+ // whatever salvage is left — consumed by this pass's reuse step
740
+ // above, or superseded by what it just embedded.
741
+ purgeEmbeddingSalvage(db);
742
+ return { success: true, vecInsertFailures: vecFailedCount };
743
+ }
744
+ finally {
745
+ if (heartbeatTimer)
746
+ clearInterval(heartbeatTimer);
747
+ }
748
+ }
749
+ catch (error) {
750
+ // Field follow-up to #956 (dev-team field review 2026-09-10): a
751
+ // contention-shaped error (another akm process writing index.db right
752
+ // now) used to escape this catch as a raw driver string ("database is
753
+ // locked"), reaching this user-facing message unclassified even though
754
+ // the acquisition-time path (`akmIndex`'s outer catch) already
755
+ // reclassifies the same shape into `TransientError("INDEX_DB_CONTENDED")`.
756
+ // Reuses that ONE shared classifier rather than a second one — see
757
+ // `index-db-contention.ts`. This catch stays non-fatal (a caller sees
758
+ // `success: false` and a message, never a thrown error): the run
759
+ // continues through the remaining index phases exactly as it did
760
+ // before, only the message is now classified when the error is
761
+ // contention-shaped.
762
+ const reclassified = reclassifyIndexDbContention(error);
763
+ const message = reclassified instanceof Error ? reclassified.message : String(reclassified);
764
+ warn("Embedding generation failed, continuing without:", message);
765
+ onProgress({ phase: "embeddings", message: `Embedding generation failed: ${message}` });
766
+ return {
767
+ success: false,
768
+ message: `Semantic search verification failed: ${message}`,
769
+ };
770
+ }
771
+ }
772
+ /**
773
+ * Update the `hasEmbeddings` DB fact after a targeted mutation, from the
774
+ * index's actual current embedding coverage — read fresh, not cached.
775
+ */
776
+ export function publishTargetedEmbeddingMeta(db, config) {
777
+ if (config.semanticSearchMode === "off") {
778
+ setMeta(db, "hasEmbeddings", "0");
779
+ return;
780
+ }
781
+ const entryCount = getEmbeddableEntryCount(db);
782
+ const embeddingCount = getEmbeddingCount(db);
783
+ const ready = entryCount > 0 && embeddingCount >= entryCount;
784
+ setMeta(db, "hasEmbeddings", ready ? "1" : "0");
785
+ }