akm-cli 0.9.16-alpha.1 → 0.9.16-alpha.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (144) hide show
  1. package/CHANGELOG.md +40 -132
  2. package/dist/assets/hints/cli-hints-full.md +13 -6
  3. package/dist/assets/tasks/core/index-refresh.yml +1 -1
  4. package/dist/assets/tasks/improve/akm-improve-catchup.yml +3 -6
  5. package/dist/cli/retired-commands.js +0 -4
  6. package/dist/cli/unknown-flags.js +3 -36
  7. package/dist/commands/env/env-binding.js +4 -4
  8. package/dist/commands/env/env-cli.js +3 -3
  9. package/dist/commands/improve/collapse-detector.js +2 -2
  10. package/dist/commands/improve/consolidate.js +4 -6
  11. package/dist/commands/improve/improve-cli.js +20 -15
  12. package/dist/commands/improve/reflect.js +23 -2
  13. package/dist/commands/lint/base-linter.js +9 -0
  14. package/dist/commands/lint/env-key-rules.js +2 -2
  15. package/dist/commands/proposal/propose.js +15 -1
  16. package/dist/commands/proposal/repository.js +3 -12
  17. package/dist/commands/proposal/validators/proposal-quality-validators.js +40 -3
  18. package/dist/commands/proposal/validators/proposal-validators.js +5 -4
  19. package/dist/commands/read/curate.js +44 -34
  20. package/dist/commands/read/search.js +35 -54
  21. package/dist/commands/read/show.js +21 -2
  22. package/dist/commands/registry-cli.js +5 -5
  23. package/dist/commands/sources/add-cli.js +59 -16
  24. package/dist/commands/sources/bundle-cli.js +35 -11
  25. package/dist/commands/sources/bundle-config-ops.js +30 -0
  26. package/dist/commands/sources/dangerous-env-audit.js +4 -4
  27. package/dist/commands/sources/info.js +8 -8
  28. package/dist/commands/sources/installed-stashes.js +55 -61
  29. package/dist/commands/sources/source-add.js +39 -38
  30. package/dist/commands/sources/source-manage.js +34 -12
  31. package/dist/commands/sources/stash-cli.js +111 -119
  32. package/dist/commands/sources/stash-skeleton.js +6 -3
  33. package/dist/commands/tasks/explain.js +4 -1
  34. package/dist/commands/tasks/tasks-cli.js +31 -9
  35. package/dist/commands/tasks/tasks.js +239 -194
  36. package/dist/commands/tasks/validate.js +20 -32
  37. package/dist/core/activation-policy.js +4 -4
  38. package/dist/core/adapter/adapters/akm-adapter.js +8 -35
  39. package/dist/core/adapter/adapters/akm-metadata.js +1 -11
  40. package/dist/core/adapter/execution-source.js +10 -29
  41. package/dist/core/asset/asset-placement.js +0 -35
  42. package/dist/core/config/config-schema.js +64 -8
  43. package/dist/core/config/config-sources.js +96 -2
  44. package/dist/core/config/config.js +190 -24
  45. package/dist/core/config/legacy-source-shape-shim.js +9 -0
  46. package/dist/core/config/schema/embedding.js +30 -7
  47. package/dist/core/config/schema/execution.js +23 -0
  48. package/dist/core/config/schema/experimental.js +1 -1
  49. package/dist/core/config/schema/scheduler.js +20 -0
  50. package/dist/core/config/schema/search.js +10 -12
  51. package/dist/core/config/schema/sources-bundles.js +32 -1
  52. package/dist/core/content-safety.js +52 -0
  53. package/dist/core/errors.js +2 -5
  54. package/dist/core/maintenance-barrier.js +11 -13
  55. package/dist/core/paths.js +11 -0
  56. package/dist/core/run-lock.js +2 -5
  57. package/dist/core/state/migrations.js +1 -26
  58. package/dist/core/state-db.js +27 -63
  59. package/dist/core/type-presentation.js +1 -1
  60. package/dist/core/write-source.js +13 -8
  61. package/dist/indexer/bundle-identity-guard.js +45 -8
  62. package/dist/indexer/ensure-index.js +0 -5
  63. package/dist/indexer/index-db-contention.js +56 -0
  64. package/dist/indexer/index-rebuild-lock.js +73 -0
  65. package/dist/indexer/index-written-assets.js +171 -133
  66. package/dist/indexer/indexer.js +1621 -458
  67. package/dist/indexer/lookup/adapter-concept-owner.js +5 -19
  68. package/dist/indexer/materialize-embeddings.js +785 -0
  69. package/dist/indexer/passes/dir-staleness.js +161 -0
  70. package/dist/indexer/passes/metadata.js +1 -18
  71. package/dist/indexer/scan/drain-dir.js +70 -27
  72. package/dist/indexer/search/db-search.js +89 -373
  73. package/dist/indexer/search/ranking-contributors.js +16 -21
  74. package/dist/indexer/search/ranking.js +57 -135
  75. package/dist/indexer/search/search-source.js +29 -11
  76. package/dist/integrations/agent/execution-lowering.js +3 -2
  77. package/dist/integrations/agent/execution-preparation.js +32 -1
  78. package/dist/integrations/agent/prompts.js +1 -1
  79. package/dist/integrations/agent/request-lowering.js +3 -2
  80. package/dist/llm/client.js +3 -11
  81. package/dist/llm/embedder.js +3 -10
  82. package/dist/llm/embedders/remote.js +104 -133
  83. package/dist/llm/feature-gate.js +2 -4
  84. package/dist/llm/rerank-client.js +3 -3
  85. package/dist/output/shapes/passthrough.js +2 -1
  86. package/dist/output/text/command-format.js +13 -19
  87. package/dist/output/text/helpers.js +1 -1
  88. package/dist/output/text/index.js +2 -5
  89. package/dist/registry/resolve.js +37 -10
  90. package/dist/scripts/akm-migrate-node.js +15197 -11351
  91. package/dist/scripts/akm-migrate.js +15514 -11668
  92. package/dist/setup/semantic-assets.js +2 -2
  93. package/dist/setup/setup.js +3 -3
  94. package/dist/setup/steps/connection.js +2 -3
  95. package/dist/setup/steps/tasks.js +29 -36
  96. package/dist/sources/providers/git-install.js +17 -11
  97. package/dist/sources/providers/git-provider.js +12 -5
  98. package/dist/sources/providers/git-stash.js +38 -16
  99. package/dist/sources/snapshot-fetchers/website-ingest.js +3 -3
  100. package/dist/storage/repositories/embedding-salvage-repository.js +184 -0
  101. package/dist/storage/repositories/index-connection.js +3 -1
  102. package/dist/storage/repositories/index-entries-repository.js +68 -77
  103. package/dist/storage/repositories/index-entry-schema.js +25 -16
  104. package/dist/storage/repositories/index-fts-repository.js +263 -29
  105. package/dist/storage/repositories/index-meta-repository.js +29 -0
  106. package/dist/storage/repositories/index-schema.js +122 -115
  107. package/dist/storage/repositories/index-utility-repository.js +1 -1
  108. package/dist/storage/repositories/index-vec-repository.js +435 -22
  109. package/dist/tasks/activation-config.js +90 -0
  110. package/dist/tasks/backends/cron.js +9 -0
  111. package/dist/tasks/backends/launchd.js +1 -0
  112. package/dist/tasks/backends/schtasks.js +2 -0
  113. package/dist/tasks/embedded.js +4 -5
  114. package/dist/tasks/scheduler-binding.js +2 -2
  115. package/dist/tasks/scheduler-sync-preview.js +8 -1
  116. package/dist/tasks/scheduler-sync.js +19 -10
  117. package/dist/tasks/source/parse-task-source.js +10 -113
  118. package/dist/tasks/source/project-v4.js +2 -2
  119. package/dist/tasks/source/task-source-v4.js +4 -12
  120. package/dist/tasks/source/task-to-v3.js +4 -12
  121. package/dist/tasks/source/task-to-v4.js +40 -7
  122. package/docs/migration/README.md +1 -0
  123. package/docs/migration/release-notes/0.9.15.md +36 -34
  124. package/docs/migration/release-notes/0.9.16.md +60 -98
  125. package/docs/migration/release-notes/README.md +0 -5
  126. package/docs/migration/v0.9.1-to-v0.9.2.md +6 -9
  127. package/docs/reference/cli.md +124 -122
  128. package/docs/reference/configuration.md +137 -133
  129. package/docs/reference/data-and-telemetry.md +1 -2
  130. package/docs/reference/tasks.md +34 -29
  131. package/package.json +1 -1
  132. package/schemas/akm-config.json +170 -6
  133. package/schemas/akm-task.json +1 -2
  134. package/dist/commands/sources/index-status.js +0 -99
  135. package/dist/core/hash.js +0 -18
  136. package/dist/indexer/drain.js +0 -306
  137. package/dist/indexer/embedding-identity.js +0 -20
  138. package/dist/indexer/enrich.js +0 -260
  139. package/dist/indexer/reconcile.js +0 -890
  140. package/dist/indexer/scan/parse-file.js +0 -66
  141. package/dist/indexer/units/unit.js +0 -159
  142. package/dist/llm/embedders/provider-limits.js +0 -288
  143. package/dist/storage/repositories/files-repository.js +0 -181
  144. package/dist/storage/repositories/units-repository.js +0 -510
@@ -1,306 +0,0 @@
1
- // This Source Code Form is subject to the terms of the Mozilla Public
2
- // License, v. 2.0. If a copy of the MPL was not distributed with this
3
- // file, You can obtain one at https://mozilla.org/MPL/2.0/.
4
- import { getConfigPath } from "../core/paths.js";
5
- import { isVerbose } from "../core/warn.js";
6
- import { embedBatch } from "../llm/embedder.js";
7
- import { probeProviderLimits } from "../llm/embedders/provider-limits.js";
8
- import { describeEmbeddingCredential, hasRemoteEndpoint, normalizeEmbeddingEndpoint, } from "../llm/embedders/remote.js";
9
- import { getMeta, setMeta } from "../storage/repositories/index-meta-repository.js";
10
- import { SQLITE_CHUNK_SIZE } from "../storage/repositories/index-sql.js";
11
- import { isVecAvailable } from "../storage/repositories/index-vec-repository.js";
12
- import { dropOtherIdentities, listMissingHashes, upsertUnitVectors } from "../storage/repositories/units-repository.js";
13
- import { deriveObservedEmbeddingIdentity } from "./embedding-identity.js";
14
- /**
15
- * Failure threshold, within the recent window below, that stops dispatching
16
- * further provider batches. Mirrors materialize-embeddings.ts's own
17
- * (unexported) `CIRCUIT_BREAKER_THRESHOLD`, #954 — reimplemented here at the
18
- * same value rather than imported, since that file is private and slated for
19
- * deletion by B5; the underlying stop-dispatch MECHANISM (`onSkip` returning
20
- * `false`) is still the real `RemoteEmbedder`'s, reused unmodified. Two
21
- * independent streaks share it: single-document failures (a multi-document
22
- * timeout is not yet evidence of a dead endpoint — `RemoteEmbedder` retries
23
- * and splits it smaller before ever reporting it this small), or network
24
- * errors at ANY size (never retried, trusted immediately). Storage-write
25
- * failures (E5b — `upsertUnitVectors`'s own per-row result) feed the SAME two
26
- * streaks: a sustained STORAGE failure (contention, permissions, a full
27
- * disk) must stop paying for provider requests just as surely as a
28
- * sustained PROVIDER failure, even while the provider itself keeps
29
- * succeeding.
30
- */
31
- const CIRCUIT_BREAKER_THRESHOLD = 3;
32
- /**
33
- * Recent-history window (in settled batch-starts) the two streaks above are
34
- * evaluated over, in place of a plain "reset to zero on any success" counter
35
- * (round-2 field finding): with concurrent dispatch (default 2, up to 16)
36
- * outcomes settle out of dispatch order, so a degraded endpoint failing MOST
37
- * requests never tripped the breaker as long as occasional successes
38
- * interleaved — reproduced with a 67% failure rate dispatching the entire
39
- * pending set. `CIRCUIT_BREAKER_THRESHOLD` failures within the last
40
- * `CIRCUIT_BREAKER_WINDOW` settled batch-starts of a streak's own kind (see
41
- * {@link pushBreakerOutcome}) trips it: a genuinely dead endpoint (no
42
- * successes at all) still trips in exactly `CIRCUIT_BREAKER_THRESHOLD`
43
- * batches, same as before; a single success now only AGES a failure out of
44
- * the window over time rather than erasing the whole run's evidence at once.
45
- */
46
- const CIRCUIT_BREAKER_WINDOW = CIRCUIT_BREAKER_THRESHOLD * 2;
47
- /** Record one settled batch-start's outcome into a breaker streak's window, capped at {@link CIRCUIT_BREAKER_WINDOW}. */
48
- function pushBreakerOutcome(window, isFailure) {
49
- window.push(isFailure);
50
- if (window.length > CIRCUIT_BREAKER_WINDOW)
51
- window.shift();
52
- }
53
- /** Failures currently recorded in a breaker streak's window. */
54
- function breakerFailureCount(window) {
55
- return window.reduce((n, isFailure) => n + (isFailure ? 1 : 0), 0);
56
- }
57
- /**
58
- * Prefix of the per-committed-batch progress line (`"${DRAIN_BATCH_PROGRESS_PREFIX}N: …"`,
59
- * emitted once per provider batch this call commits). Exported so a caller
60
- * juggling several `onProgress` sources (`stash-cli.ts`'s `akm index`) can
61
- * recognize — and, outside `--verbose`, suppress — this specific
62
- * high-frequency line by prefix rather than re-deriving its own copy of the
63
- * pattern (#954).
64
- */
65
- export const DRAIN_BATCH_PROGRESS_PREFIX = "[drain] batch ";
66
- function throwIfAborted(signal) {
67
- if (signal?.aborted) {
68
- throw signal.reason instanceof Error ? signal.reason : new Error("drain interrupted");
69
- }
70
- }
71
- /** Every distinct unit hash the index currently knows about, regardless of identity. */
72
- function selectAllUnitHashes(db) {
73
- return db.prepare("SELECT unit_hash FROM unit_texts ORDER BY unit_hash").all().map((row) => row.unit_hash);
74
- }
75
- /** `unit_hash -> text` for exactly `hashes`, chunked to respect SQLite's bound-parameter limit. */
76
- function fetchUnitTexts(db, hashes) {
77
- const texts = new Map();
78
- for (let offset = 0; offset < hashes.length; offset += SQLITE_CHUNK_SIZE) {
79
- const chunk = hashes.slice(offset, offset + SQLITE_CHUNK_SIZE);
80
- const placeholders = chunk.map(() => "?").join(",");
81
- const rows = db
82
- .prepare(`SELECT unit_hash, text FROM unit_texts WHERE unit_hash IN (${placeholders})`)
83
- .all(...chunk);
84
- for (const row of rows)
85
- texts.set(row.unit_hash, row.text);
86
- }
87
- return texts;
88
- }
89
- /**
90
- * Effective embedding config and request packing for this drain (index
91
- * redesign, B5 — replaces the retired `embedding.maxTokens`/`batchSize`/
92
- * `contextLength` config keys): `concurrency` (in-flight requests) defaults
93
- * to the provider's OWN observed slot count when `embedding.concurrency`
94
- * itself leaves it unset (`probeProviderLimits` already applies that same
95
- * override to `slots`). The request TOKEN WINDOW, chars-per-token ratio, and
96
- * Ollama `num_ctx` are no longer config fields at all — they are threaded
97
- * into `RemoteEmbedder.embedBatch` as `packing`, sourced straight from the
98
- * same probe: `windowTokens` for the per-request budget, `charsPerToken` for
99
- * the calibrated per-text token estimate, and `windowTokens` again for
100
- * Ollama's `num_ctx` when `source === "ollama"`. `windowIsKnown`
101
- * (`source !== "default"`) gates `RemoteEmbedder`'s same-run adaptive
102
- * shrink: a provider that reports nothing about its own context size still
103
- * gets that corrective, but a probed, authoritative window does not need it
104
- * second-guessed.
105
- */
106
- async function resolveEmbeddingPacking(config, signal) {
107
- const base = config.embedding ?? {};
108
- const limits = await probeProviderLimits(base, { signal });
109
- return {
110
- embeddingConfig: { ...base, concurrency: base.concurrency ?? limits.slots },
111
- packing: {
112
- tokenBudget: limits.windowTokens,
113
- charsPerToken: limits.charsPerToken,
114
- windowIsKnown: limits.source !== "default",
115
- ollamaNumCtx: limits.source === "ollama" ? limits.windowTokens : undefined,
116
- },
117
- };
118
- }
119
- /**
120
- * #953 field gap, ported from the deleted `materialize-embeddings.ts`
121
- * (`git show fc711fd6^:src/indexer/materialize-embeddings.ts`): a keyless
122
- * request against a remote embedding endpoint could not be reproduced in the
123
- * lab — every `RemoteEmbedder` path already resolves `secret://` through one
124
- * boundary, so a keyless request can only mean `embedding.apiKey` was absent
125
- * from the config THIS run loaded. The actionable outcome is a
126
- * self-diagnosing run, not a fix: one default-level line, emitted once
127
- * before the first provider request this call makes, naming the endpoint,
128
- * model, and credential SOURCE (never the value) so a field run can compare
129
- * it against what the gateway actually saw. A local (non-remote) endpoint,
130
- * or a call with nothing pending, has nothing to diagnose and stays silent.
131
- */
132
- function emitCredentialDiagnostic(config, onProgress) {
133
- if (!onProgress || !hasRemoteEndpoint(config.embedding ?? {}))
134
- return;
135
- const endpoint = normalizeEmbeddingEndpoint(config.embedding?.endpoint ?? "");
136
- const credential = describeEmbeddingCredential(config.embedding?.apiKey);
137
- const configFileSuffix = isVerbose() ? `; config: ${getConfigPath()}` : "";
138
- onProgress(`[embed] endpoint ${endpoint}, model ${config.embedding?.model ?? "unknown"}; credential: ${credential}${configFileSuffix}`);
139
- }
140
- function formatDoneLine(counts) {
141
- return (`[drain] done: ${counts.pending} pending, ${counts.embedded} embedded, ${counts.failed} failed, ` +
142
- `${counts.skipped} skipped (identity: ${counts.identity ?? "unknown"})`);
143
- }
144
- export async function drainEmbeddingQueue(db, config, opts = {}) {
145
- throwIfAborted(opts.signal);
146
- let identity = getMeta(db, "embeddingIdentity") ?? null;
147
- if (config.semanticSearchMode === "off") {
148
- return { pending: 0, embedded: 0, failed: 0, skipped: 0, identity };
149
- }
150
- const candidateHashes = opts.onlyHashes ? [...new Set(opts.onlyHashes)] : selectAllUnitHashes(db);
151
- const missingHashes = identity ? listMissingHashes(db, candidateHashes, identity) : candidateHashes;
152
- const pending = missingHashes.length;
153
- const emitDone = (counts) => {
154
- opts.onProgress?.(formatDoneLine(counts));
155
- return counts;
156
- };
157
- // upsertUnitVectors is a no-op without sqlite-vec (units-repository.ts), so
158
- // embedding the pending set here would just throw every vector away and
159
- // leave it "missing" again for the next call — pure wasted provider
160
- // traffic. `akmIndex`'s verification reports the missing extension as
161
- // blocked; this stays silent. `pending` is computed above so the done line and `akm
162
- // index status` stay truthful even though nothing was attempted.
163
- if (!isVecAvailable(db)) {
164
- return emitDone({ pending, embedded: 0, failed: 0, skipped: pending, identity });
165
- }
166
- if (pending === 0) {
167
- return emitDone({ pending: 0, embedded: 0, failed: 0, skipped: 0, identity });
168
- }
169
- const boundedHashes = opts.limit !== undefined ? missingHashes.slice(0, opts.limit) : missingHashes;
170
- const textByHash = fetchUnitTexts(db, boundedHashes);
171
- // A hash in `unit_texts` should always resolve to a row (it was just read
172
- // from that same table above), but a missing row is dropped rather than
173
- // sent to the provider as `undefined` text.
174
- const orderedHashes = boundedHashes.filter((hash) => textByHash.has(hash));
175
- const texts = orderedHashes.map((hash) => textByHash.get(hash));
176
- if (texts.length === 0) {
177
- return emitDone({ pending, embedded: 0, failed: 0, skipped: 0, identity });
178
- }
179
- emitCredentialDiagnostic(config, opts.onProgress);
180
- const { embeddingConfig, packing } = await resolveEmbeddingPacking(config, opts.signal);
181
- let embedded = 0;
182
- let failed = 0;
183
- let batchNumber = 0;
184
- // Two independent circuit-breaker streaks (single-document failures,
185
- // network errors at any size) — see CIRCUIT_BREAKER_WINDOW above.
186
- const singleDocFailureWindow = [];
187
- const networkErrorFailureWindow = [];
188
- // Whether this CALL has already decided the identity its first committed
189
- // row observed (E1) — adoption happens at most once per call; see the
190
- // module doc comment and the identity block in `onBatch` below.
191
- let identityDecidedThisCall = false;
192
- const onSkip = (skip) => {
193
- failed++;
194
- if (!skip.batchStart)
195
- return undefined;
196
- if (skip.reason === "context-window-exceeded") {
197
- // Proves the provider IS reachable; not evidence of a dead endpoint.
198
- singleDocFailureWindow.length = 0;
199
- networkErrorFailureWindow.length = 0;
200
- return undefined;
201
- }
202
- pushBreakerOutcome(singleDocFailureWindow, skip.batchSize === 1);
203
- pushBreakerOutcome(networkErrorFailureWindow, skip.failureKind === "network-error");
204
- if (breakerFailureCount(singleDocFailureWindow) >= CIRCUIT_BREAKER_THRESHOLD ||
205
- breakerFailureCount(networkErrorFailureWindow) >= CIRCUIT_BREAKER_THRESHOLD) {
206
- return false;
207
- }
208
- return undefined;
209
- };
210
- const onBatch = (indices, embeddings, model, outcome) => {
211
- // "retrying"/"budget-lowered" are in-flight notices for a batch that has
212
- // not settled yet (see EmbeddingBatchOutcome) — nothing to commit or
213
- // count, and not a distinct "batch" for the one-line-per-batch contract.
214
- if (outcome?.outcome === "retrying" || outcome?.outcome === "budget-lowered")
215
- return;
216
- const rows = [];
217
- for (let k = 0; k < indices.length; k++) {
218
- const embedding = embeddings[k];
219
- if (!embedding)
220
- continue;
221
- const learned = deriveObservedEmbeddingIdentity(config.embedding, model, embedding.length);
222
- if (!identityDecidedThisCall) {
223
- // This call's FIRST committed row decides the identity it adopts
224
- // (E1) — learned once, not re-derived per batch: a provider whose
225
- // responses alternate between models WITHIN one call (a
226
- // load-balanced gateway, a blue/green rollout behind one endpoint)
227
- // must not thrash the store between them (embed, delete, re-embed,
228
- // never converging). A genuine model change is still caught, just
229
- // not until the NEXT call's own first batch observes it and purges
230
- // whatever this call left behind.
231
- identityDecidedThisCall = true;
232
- if (learned && learned !== identity) {
233
- identity = learned;
234
- setMeta(db, "embeddingIdentity", identity);
235
- dropOtherIdentities(db, identity, embedding.length);
236
- }
237
- }
238
- const currentIdentity = identity;
239
- if (currentIdentity === null || learned !== currentIdentity) {
240
- // Either nothing has ever been learned, or a LATER batch this same
241
- // call reported an identity different from the one already adopted
242
- // — left missing rather than switched to; it becomes "missing"
243
- // again under whatever identity this call is using, and a later
244
- // call, whose own first batch observes it, picks it up. Counted in
245
- // `skipped` below (attempted minus embedded minus failed), not
246
- // `embedded`.
247
- continue;
248
- }
249
- const hash = orderedHashes[indices[k]];
250
- if (hash)
251
- rows.push({ hash, identity: currentIdentity, vector: embedding });
252
- }
253
- let storageBreakerTripped = false;
254
- if (rows.length > 0) {
255
- // upsertUnitVectors commits each row in its own transaction — this IS
256
- // "each provider batch commits durably" (a wrapping db.transaction()
257
- // here would only nest as an unobservable SAVEPOINT inside it, per the
258
- // ambient-transaction hazard materialize-embeddings.ts's own drift
259
- // guard documents), now made even finer-grained so one malformed
260
- // vector in a batch (e.g. a width mismatch) can't roll back the rest
261
- // of an otherwise-good response.
262
- const result = upsertUnitVectors(db, rows);
263
- embedded += result.inserted;
264
- failed += result.failed;
265
- if (result.failed > 0) {
266
- // E5b: a write failure is just as much evidence of a broken run as
267
- // a provider failure — a sustained STORAGE failure (contention,
268
- // permissions, a full disk) must not keep dispatching every
269
- // remaining batch to a perfectly healthy provider at full cost
270
- // while every write silently fails. One event per committed batch
271
- // (the "count batch starts, not documents" rule onSkip already
272
- // applies to provider failures), fed into the SAME two streaks.
273
- pushBreakerOutcome(singleDocFailureWindow, true);
274
- pushBreakerOutcome(networkErrorFailureWindow, true);
275
- if (breakerFailureCount(singleDocFailureWindow) >= CIRCUIT_BREAKER_THRESHOLD ||
276
- breakerFailureCount(networkErrorFailureWindow) >= CIRCUIT_BREAKER_THRESHOLD) {
277
- storageBreakerTripped = true;
278
- }
279
- }
280
- }
281
- if (embeddings.some((embedding) => embedding !== undefined)) {
282
- pushBreakerOutcome(singleDocFailureWindow, false);
283
- pushBreakerOutcome(networkErrorFailureWindow, false);
284
- }
285
- batchNumber++;
286
- if (opts.onProgress) {
287
- const docCount = outcome?.docCount ?? indices.length;
288
- const label = outcome && outcome.outcome !== "stored" ? `failed: ${outcome.reason ?? "unknown"}` : `${rows.length} stored`;
289
- opts.onProgress(`${DRAIN_BATCH_PROGRESS_PREFIX}${batchNumber}: ${docCount} docs → ${label}`);
290
- }
291
- if (storageBreakerTripped) {
292
- // onBatch has no `false`-return stop-dispatch contract the way onSkip
293
- // does (a storage failure can trip this even when the provider itself
294
- // keeps succeeding, so onSkip is never called at all) — this reuses
295
- // RemoteEmbedder's own documented mechanism instead: a throw from
296
- // onBatch stops the pool from dispatching any further provider
297
- // request, and is rethrown once every in-flight batch has settled.
298
- throw new Error(`Circuit breaker: ${CIRCUIT_BREAKER_THRESHOLD} storage write failures while embedding; stopping further provider requests this call.`);
299
- }
300
- };
301
- await embedBatch(texts, embeddingConfig, opts.signal, onSkip, onBatch, packing);
302
- throwIfAborted(opts.signal);
303
- const attempted = texts.length;
304
- const skipped = Math.max(0, attempted - embedded - failed);
305
- return emitDone({ pending, embedded, failed, skipped, identity });
306
- }
@@ -1,20 +0,0 @@
1
- // This Source Code Form is subject to the terms of the Mozilla Public
2
- // License, v. 2.0. If a copy of the MPL was not distributed with this
3
- // file, You can obtain one at https://mozilla.org/MPL/2.0/.
4
- import { DETERMINISTIC_EMBED_MODEL_ID, isDeterministicEmbedEnabled } from "../llm/embedders/deterministic.js";
5
- import { DEFAULT_LOCAL_MODEL } from "../llm/embedders/local.js";
6
- /**
7
- * Returns `undefined` when nothing was actually observed this call (no
8
- * vector to measure yet) — there is nothing to key an identity on.
9
- */
10
- export function deriveObservedEmbeddingIdentity(embedding, observedModel, observedVectorLen) {
11
- if (isDeterministicEmbedEnabled()) {
12
- return `deterministic:${DETERMINISTIC_EMBED_MODEL_ID}`;
13
- }
14
- if (observedVectorLen === undefined)
15
- return undefined;
16
- if (embedding?.endpoint) {
17
- return `remote:${observedModel ?? embedding.model ?? "unknown"}|${observedVectorLen}`;
18
- }
19
- return `local:${embedding?.localModel ?? DEFAULT_LOCAL_MODEL}|${observedVectorLen}`;
20
- }
@@ -1,260 +0,0 @@
1
- // This Source Code Form is subject to the terms of the Mozilla Public
2
- // License, v. 2.0. If a copy of the MPL was not distributed with this
3
- // file, You can obtain one at https://mozilla.org/MPL/2.0/.
4
- /**
5
- * LLM metadata-enrichment pass, restored on the reconcile path
6
- * (docs/plans/index-redesign-contract.md, B5e).
7
- *
8
- * Before the index redesign, `akm index` ran a config-driven metadata
9
- * enhancement pass over every "generated"-quality entry, keyed by a
10
- * `(assetRef, cacheVariant)` cache row whose `body_hash` column happened to
11
- * gate freshness. The reconcile rewrite (`reconcile.ts`, B1) dropped the
12
- * call site along with the rest of the old phase pipeline. This module
13
- * restores the feature on the NEW path, content-addressed throughout:
14
- * `reconcileRoots` collects one {@link MetadataEnrichmentCandidate} per file
15
- * it just upserted (added or changed) and hands the batch to
16
- * {@link enrichReconciledEntries} once, AFTER every per-file transaction in
17
- * this run has already committed — a provider call must never run inside
18
- * `applyChange`'s `BEGIN IMMEDIATE` transaction (docs/plans/index-redesign.md
19
- * rule 5: every index write stays a short, idempotent transaction; an LLM
20
- * call can take seconds to minutes and must not hold one open).
21
- *
22
- * **Content-addressed cache** — `llm_enrichment_cache` (still shared with
23
- * graph-extraction and memory-inference, which key it by absolute file path)
24
- * is used here with `asset_ref = body_hash = candidate.blobHash`: the cache
25
- * row IS the content address, so a cache hit means "this exact byte content
26
- * has already been enriched" regardless of which entry or how many entries
27
- * currently carry it, and a rename or an unrelated field edit elsewhere in
28
- * the same file never invalidates it. `withLlmCache` (`./db/llm-cache.ts`)
29
- * already implements exactly this hash-gated lookup/call/write shape, so this
30
- * module reuses it rather than duplicating the pattern a third time.
31
- *
32
- * **Fail-soft** — `enhanceMetadata`'s `EnhanceMetadataOutcome` distinguishes
33
- * `enriched` (real success — cache it) from `skipped` (feature gate closed)
34
- * and `failed` (provider/network error): `withLlmCache`'s "only cache a
35
- * defined result" contract means a `skipped`/`failed` outcome (mapped to
36
- * `undefined` below) writes no cache row and leaves the entry's `quality`
37
- * untouched, so a transient provider outage can never poison an entry into a
38
- * permanent enrichment skip.
39
- *
40
- * **`--full` re-applies without a new provider call** — `reconcileRoots`'s
41
- * `forceReparse` re-parses every file, so an unchanged file's fresh
42
- * `IndexDocument` is `quality: "generated"` again (enrichment only ever
43
- * updated the DB row, never the source file) and becomes an enrichment
44
- * candidate again on every `--full` run. Its `blobHash` is unchanged, so the
45
- * content-addressed cache lookup above hits and re-applies the SAME cached
46
- * fields with no new provider call — this falls out of content-addressing
47
- * for free and needs no `--full`-specific branch here.
48
- */
49
- import fs from "node:fs";
50
- import { concurrentMap } from "../core/concurrent.js";
51
- import { ConfigError } from "../core/errors.js";
52
- import { defaultConcurrencyForEndpoint } from "../core/loopback.js";
53
- import { withImmediateTransaction } from "../core/state-db.js";
54
- import { warn } from "../core/warn.js";
55
- import { resolveIndexPassExecution } from "../llm/index-passes.js";
56
- import { enhanceMetadata } from "../llm/metadata-enhance.js";
57
- import { insertNewUnitTexts } from "../storage/repositories/files-repository.js";
58
- import { upsertEntry } from "../storage/repositories/index-entries-repository.js";
59
- import { replaceEntryUnits } from "../storage/repositories/units-repository.js";
60
- import { withLlmCache } from "./db/llm-cache.js";
61
- import { getMarkdownFragmentContent, hasMarkdownFragmentContent, isEnrichmentComplete, setMarkdownFragmentContent, } from "./passes/metadata.js";
62
- import { buildSearchText } from "./search/search-fields.js";
63
- import { deriveUnits, toUnitSource } from "./units/unit.js";
64
- /**
65
- * Namespaces this pass's `llm_enrichment_cache` rows away from
66
- * graph-extraction's and memory-inference's own `cacheVariant` values, which
67
- * key the SAME shared table by absolute file path rather than content hash.
68
- */
69
- const METADATA_ENRICHMENT_CACHE_VARIANT = "metadata-enhance-v1";
70
- function emptyCounts() {
71
- return { attempted: 0, cacheHits: 0, enriched: 0, failed: 0, skipped: 0 };
72
- }
73
- /** Only "generated"-quality entries missing description/tags/searchHints are worth an LLM call — see `isEnrichmentComplete`. */
74
- function isEligibleForEnrichment(entry) {
75
- return entry.quality === "generated" && !isEnrichmentComplete(entry);
76
- }
77
- /**
78
- * Bounded-pool width for this pass — kept as a direct call to the shared
79
- * classifier (not `indexer.ts`'s `getDefaultLlmConcurrency` wrapper) to avoid
80
- * an indexer.ts → enrich.ts → indexer.ts import cycle, exactly like
81
- * `src/llm/embedders/remote.ts`'s `resolveEmbeddingConcurrency` — see that
82
- * function's neighboring comment. `tests/indexer/llm-concurrency-default.test.ts`
83
- * pins `getDefaultLlmConcurrency`'s behavior; this mirrors it exactly.
84
- */
85
- function resolveEnrichmentConcurrency(connection) {
86
- if (typeof connection?.concurrency === "number")
87
- return connection.concurrency;
88
- return defaultConcurrencyForEndpoint(connection?.endpoint);
89
- }
90
- /**
91
- * Run the metadata-enrichment pass over every eligible candidate
92
- * `reconcileRoots` collected this run, with a bounded concurrency pool
93
- * (`resolveEnrichmentConcurrency`). Only called when
94
- * `resolveIndexPassExecution("enrichment", config)` resolves a runner — an
95
- * unconfigured engine, or `index.enrichment.enabled: false`, is a no-op with
96
- * zero cache reads and zero provider calls. The separate `metadata_enhance`
97
- * feature gate (`index.metadataEnhance.enabled`, default `false`) is checked
98
- * per-call inside `enhanceMetadata` itself, so a closed gate still shows up
99
- * here as a cheap `skipped` outcome rather than being special-cased twice.
100
- *
101
- * A `ConfigError` (a required symbolic credential that resolved to nothing)
102
- * is not fail-soft like a provider error — `enhanceMetadata` lets it escape
103
- * `tryLlmFeature`'s normal fallback (`llm/structured-call.ts`'s
104
- * `callStructured`) precisely so a genuinely broken config surfaces loudly
105
- * instead of reading as an ordinary per-entry failure. `concurrentMap`
106
- * itself swallows a thrown callback into an `undefined` slot, so this
107
- * catches it per-candidate and rethrows the first occurrence once every
108
- * in-flight candidate has settled.
109
- */
110
- export async function enrichReconciledEntries(db, config, candidates, maxChars, opts) {
111
- const counts = emptyCounts();
112
- const eligible = candidates.filter((candidate) => isEligibleForEnrichment(candidate.entry));
113
- if (eligible.length === 0)
114
- return counts;
115
- const runner = resolveIndexPassExecution("enrichment", config).runner;
116
- if (!runner)
117
- return counts;
118
- const concurrency = resolveEnrichmentConcurrency(runner.connection);
119
- opts?.onProgress?.(`Metadata enrichment starting for ${eligible.length} entr${eligible.length === 1 ? "y" : "ies"} (concurrency ${concurrency}).`);
120
- let configFailure;
121
- await concurrentMap(eligible, async (candidate) => {
122
- if (opts?.signal?.aborted)
123
- return;
124
- counts.attempted++;
125
- try {
126
- await enrichOneCandidate(db, runner, config, candidate, maxChars, counts, opts?.signal);
127
- }
128
- catch (err) {
129
- if (err instanceof ConfigError) {
130
- configFailure ??= err;
131
- return;
132
- }
133
- throw err;
134
- }
135
- }, concurrency);
136
- if (configFailure)
137
- throw configFailure;
138
- opts?.onProgress?.(`Metadata enrichment finished: ${counts.enriched} enriched (${counts.cacheHits} from cache), ` +
139
- `${counts.failed} failed, ${counts.skipped} skipped.`);
140
- if (counts.failed > 0 && counts.enriched === 0 && counts.skipped === 0) {
141
- warn(`LLM metadata enrichment failed for all ${counts.failed} attempted entr${counts.failed === 1 ? "y" : "ies"} — ` +
142
- "index built without enrichment. Check the engine selected by index.enrichment.engine (or defaults.llmEngine).");
143
- }
144
- return counts;
145
- }
146
- async function enrichOneCandidate(db, runner, config, candidate, maxChars, counts, signal) {
147
- let sawOutcome;
148
- let cacheHit = false;
149
- const metadata = await withLlmCache(db, candidate.blobHash, "", false, async () => {
150
- let fileContent;
151
- try {
152
- fileContent = fs.readFileSync(candidate.filePath, "utf8");
153
- }
154
- catch {
155
- // Best-effort context for the prompt only — enhanceMetadata still
156
- // runs (with less context) when the file cannot be re-read.
157
- }
158
- const outcome = await enhanceMetadata(runner, candidate.entry, fileContent, signal, config);
159
- if (outcome.status !== "enriched") {
160
- sawOutcome = outcome.status;
161
- return undefined;
162
- }
163
- return outcome.metadata;
164
- }, (raw) => (raw !== null && typeof raw === "object" ? raw : undefined), candidate.blobHash, METADATA_ENRICHMENT_CACHE_VARIANT, { onCacheHit: () => (cacheHit = true) });
165
- if (metadata === undefined) {
166
- if (sawOutcome === "failed")
167
- counts.failed++;
168
- else
169
- counts.skipped++;
170
- return;
171
- }
172
- let applied;
173
- try {
174
- applied = applyEnrichmentToEntry(db, candidate, maxChars, metadata);
175
- }
176
- catch (err) {
177
- // A real write failure (not the stale-identity no-op below, which never
178
- // throws): `concurrentMap` would otherwise swallow this into a silent
179
- // undefined slot with `counts.enriched` never incremented but no record
180
- // of the failure either. Surface it the same way a provider failure is
181
- // already surfaced.
182
- counts.failed++;
183
- warn(`[index] Metadata enrichment write failed for ${candidate.filePath}: ` +
184
- (err instanceof Error ? err.message : String(err)));
185
- return;
186
- }
187
- if (!applied) {
188
- // The live `entries` row no longer matches the identity this candidate
189
- // was queued under (a concurrent rename or delete) — see
190
- // `applyEnrichmentToEntry`. The metadata is real and already cached
191
- // above, so this is reported as skipped rather than lost, and the next
192
- // ordinary run re-applies it from cache with no new provider call.
193
- counts.skipped++;
194
- return;
195
- }
196
- if (cacheHit)
197
- counts.cacheHits++;
198
- counts.enriched++;
199
- }
200
- /**
201
- * Merge enrichment fields onto the candidate's entry, re-derive its units,
202
- * and write both through the SAME canonical entry/FTS mutation and
203
- * `unit_texts`/`entry_units` maintenance `applyChange` (`reconcile.ts`) uses
204
- * — one short `BEGIN IMMEDIATE` transaction, no `content_hash` argument so
205
- * `upsertEntry`'s `COALESCE` preserves the scan-derived blob hash untouched.
206
- *
207
- * Fragment units are unaffected: only `description`/`tags`/`searchHints`
208
- * change, which feeds solely unit ordinal 0 (`structuredFieldsText`,
209
- * `units/unit.ts`); `replaceEntryUnits` is still a full delete-then-insert
210
- * for the entry, so `getMarkdownFragmentContent`/`setMarkdownFragmentContent`
211
- * re-tag the merged copy — otherwise `deriveUnits` would see no markdown
212
- * body at all and silently drop every fragment unit `applyChange` already
213
- * derived for this entry.
214
- *
215
- * **Stale-identity guard** — `candidate` was captured before the LLM round
216
- * trip above, which can take seconds to minutes. A concurrent reconcile can
217
- * rename (`repointEntry` updates the SAME `entries.id` in place with a new
218
- * `item_ref`/`content_hash`/`file_path`) or delete this row while that call
219
- * was in flight. Writing the captured values regardless would either (a)
220
- * `upsertEntry` under the stale `item_ref`, which no longer conflicts with
221
- * anything and INSERTs a ghost row pointing at an identity that no longer
222
- * exists, or (b) `replaceEntryUnits(candidate.entryId)` overwriting a live
223
- * renamed row's units with the old identity's hashes — or, if the row was
224
- * deleted outright, throw a foreign-key error. So this re-reads the live row
225
- * BY ID inside the same transaction and no-ops (returns `false`, counted as
226
- * skipped by the caller) unless its `content_hash` and `item_ref` still
227
- * match what this candidate was queued under; only then is `candidate.entryId`
228
- * — now confirmed live, never a captured id that may no longer exist — used
229
- * to write.
230
- */
231
- function applyEnrichmentToEntry(db, candidate, maxChars, metadata) {
232
- const merged = { ...candidate.entry, quality: "enriched" };
233
- if (metadata.description)
234
- merged.description = metadata.description;
235
- if (metadata.tags?.length)
236
- merged.tags = metadata.tags;
237
- if (metadata.searchHints?.length)
238
- merged.searchHints = metadata.searchHints;
239
- if (hasMarkdownFragmentContent(candidate.entry)) {
240
- setMarkdownFragmentContent(merged, getMarkdownFragmentContent(candidate.entry));
241
- }
242
- const searchText = buildSearchText(merged);
243
- return withImmediateTransaction(db, () => {
244
- const live = db
245
- .prepare("SELECT content_hash AS contentHash, item_ref AS itemRef FROM entries WHERE id = ?")
246
- .get(candidate.entryId);
247
- if (!live || live.contentHash !== candidate.blobHash || live.itemRef !== candidate.provenance.itemRef) {
248
- return false;
249
- }
250
- upsertEntry(db, candidate.filePath, merged, searchText, candidate.provenance);
251
- const units = deriveUnits(toUnitSource(candidate.entryId, merged), maxChars);
252
- insertNewUnitTexts(db, units.map((unit) => ({
253
- hash: unit.hash,
254
- kind: unit.fragmentId === null ? "card" : "fragment",
255
- text: unit.text,
256
- })));
257
- replaceEntryUnits(db, candidate.entryId, units.map((unit) => ({ ordinal: unit.ordinal, fragmentId: unit.fragmentId, hash: unit.hash })));
258
- return true;
259
- }, "index");
260
- }