akm-cli 0.9.17-alpha.7 → 0.9.17-alpha.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (116) hide show
  1. package/CHANGELOG.md +473 -0
  2. package/STABILITY.md +9 -8
  3. package/dist/akm +55 -22
  4. package/dist/akm-migrate +38 -19
  5. package/dist/assets/hints/cli-hints-full.md +6 -7
  6. package/dist/assets/improve-strategies/catchup.json +0 -3
  7. package/dist/assets/improve-strategies/consolidate.json +0 -1
  8. package/dist/assets/improve-strategies/default.json +1 -2
  9. package/dist/assets/improve-strategies/proactive-maintenance.json +1 -2
  10. package/dist/assets/improve-strategies/quick.json +1 -2
  11. package/dist/assets/improve-strategies/reflect-distill.json +1 -2
  12. package/dist/assets/improve-strategies/thorough.json +0 -3
  13. package/dist/assets/prompts/consolidate-pair.md +20 -0
  14. package/dist/assets/stash-skeleton/facts/conventions/backlinks.md +20 -20
  15. package/dist/assets/stash-skeleton/facts/conventions/domains.md +2 -2
  16. package/dist/assets/templates/html/health.html +3 -5
  17. package/dist/cli/retired-commands.js +1 -1
  18. package/dist/commands/health/archive-usage.js +98 -0
  19. package/dist/commands/health/data-dir-usage.js +25 -13
  20. package/dist/commands/health/html-report.js +1 -4
  21. package/dist/commands/health/improve-metrics.js +0 -25
  22. package/dist/commands/health/md-report.js +1 -6
  23. package/dist/commands/health/report-view-model.js +4 -14
  24. package/dist/commands/health/windows.js +0 -1
  25. package/dist/commands/health.js +13 -0
  26. package/dist/commands/improve/consolidate/continuity-check.js +137 -0
  27. package/dist/commands/improve/consolidate/pair-pass.js +791 -0
  28. package/dist/commands/improve/consolidate.js +38 -63
  29. package/dist/commands/improve/extract-prompt.js +1 -2
  30. package/dist/commands/improve/improve-cli.js +1 -1
  31. package/dist/commands/improve/improve-strategies.js +23 -5
  32. package/dist/commands/improve/improve.js +19 -30
  33. package/dist/commands/improve/ledger.js +3 -2
  34. package/dist/commands/improve/loop-stages.js +5 -84
  35. package/dist/commands/improve/memory/memory-belief.js +3 -1
  36. package/dist/commands/improve/memory/memory-improve.js +269 -11
  37. package/dist/commands/improve/planner.js +0 -5
  38. package/dist/commands/improve/preparation.js +20 -135
  39. package/dist/commands/improve/retrieval-scope.js +19 -4
  40. package/dist/commands/improve/salience.js +1 -14
  41. package/dist/commands/improve/stage.js +0 -1
  42. package/dist/commands/lint/base-linter.js +19 -11
  43. package/dist/commands/proposal/drain.js +8 -1
  44. package/dist/commands/proposal/proposal-cli.js +16 -2
  45. package/dist/commands/proposal/proposal-types.js +7 -0
  46. package/dist/commands/proposal/proposal.js +37 -6
  47. package/dist/commands/proposal/repository.js +613 -4
  48. package/dist/commands/proposal/validators/proposals.js +9 -0
  49. package/dist/commands/read/curate.js +40 -13
  50. package/dist/commands/read/knowledge.js +3 -2
  51. package/dist/commands/read/show.js +55 -16
  52. package/dist/commands/sources/info.js +3 -0
  53. package/dist/commands/sources/stash-cli.js +2 -2
  54. package/dist/core/adapter/adapters/akm-adapter.js +2 -0
  55. package/dist/core/adapter/adapters/akm-metadata.js +31 -0
  56. package/dist/core/bundle-rename.js +1 -7
  57. package/dist/core/config/config-schema.js +8 -1
  58. package/dist/core/config/config.js +23 -48
  59. package/dist/core/config/engine-semantics.js +0 -2
  60. package/dist/core/config/schema/improve-processes.js +17 -42
  61. package/dist/core/config/schema/index-config.js +5 -25
  62. package/dist/core/file-change.js +13 -5
  63. package/dist/core/improve-result.js +16 -5
  64. package/dist/core/improve-types.js +0 -1
  65. package/dist/core/loopback.js +7 -12
  66. package/dist/core/parse.js +13 -16
  67. package/dist/core/state/migrations.js +15 -0
  68. package/dist/core/time.js +0 -20
  69. package/dist/indexer/db/llm-cache.js +2 -2
  70. package/dist/indexer/ensure-index.js +2 -2
  71. package/dist/indexer/index-written-assets.js +2 -3
  72. package/dist/indexer/indexer.js +18 -418
  73. package/dist/indexer/links/declared-links.js +90 -0
  74. package/dist/indexer/passes/metadata.js +0 -19
  75. package/dist/indexer/scan/doc-to-entry.js +1 -0
  76. package/dist/indexer/walk/walker.js +3 -4
  77. package/dist/llm/client.js +8 -10
  78. package/dist/llm/embedders/remote.js +1 -2
  79. package/dist/llm/feature-gate.js +0 -5
  80. package/dist/output/shapes/helpers.js +23 -4
  81. package/dist/output/text/command-format.js +0 -8
  82. package/dist/output/text/proposal-format.js +47 -1
  83. package/dist/output/text/show-format.js +13 -17
  84. package/dist/scripts/akm-migrate-node.js +2754 -2836
  85. package/dist/scripts/akm-migrate.js +2754 -2836
  86. package/dist/setup/steps/connection.js +5 -6
  87. package/dist/setup/steps/platforms.js +2 -2
  88. package/dist/sources/providers/git-stash.js +55 -4
  89. package/dist/storage/repositories/improve-ledger-repository.js +48 -7
  90. package/dist/storage/repositories/index-entries-repository.js +16 -13
  91. package/dist/storage/repositories/index-entry-schema.js +22 -3
  92. package/dist/storage/repositories/index-links-repository.js +143 -0
  93. package/dist/storage/repositories/index-llm-cache-repository.js +7 -26
  94. package/dist/storage/repositories/index-schema.js +82 -104
  95. package/dist/storage/repositories/proposals-repository.js +61 -0
  96. package/dist/storage/repositories/salience-repository.js +1 -19
  97. package/dist/tasks/source/task-to-v4.js +462 -74
  98. package/docs/migration/release-notes/0.9.17.md +7 -5
  99. package/docs/reference/cli.md +33 -21
  100. package/docs/reference/configuration.md +21 -12
  101. package/docs/reference/data-and-telemetry.md +0 -1
  102. package/package.json +1 -1
  103. package/schemas/akm-config.json +0 -342
  104. package/dist/assets/improve-strategies/graph-refresh.json +0 -15
  105. package/dist/assets/prompts/contradiction-judge.md +0 -33
  106. package/dist/assets/prompts/graph-extract-system.md +0 -1
  107. package/dist/assets/prompts/graph-extract-user-prompt.md +0 -35
  108. package/dist/assets/prompts/metadata-enhance-system.md +0 -1
  109. package/dist/assets/tasks/improve/akm-graph-refresh-weekly.yml +0 -4
  110. package/dist/indexer/db/graph-db.js +0 -431
  111. package/dist/indexer/graph/graph-extraction.js +0 -807
  112. package/dist/indexer/graph/graph-related.js +0 -131
  113. package/dist/indexer/graph/graph-types.js +0 -4
  114. package/dist/llm/graph-extract.js +0 -903
  115. package/dist/llm/metadata-enhance.js +0 -95
  116. package/dist/tasks/source/task-to-v3.js +0 -453
@@ -1,807 +0,0 @@
1
- // This Source Code Form is subject to the terms of the Mozilla Public
2
- // License, v. 2.0. If a copy of the MPL was not distributed with this
3
- // file, You can obtain one at https://mozilla.org/MPL/2.0/.
4
- /**
5
- * Graph-extraction pass for `akm index` (#207).
6
- *
7
- * Walks the primary stash for `memory:` and `knowledge:` assets, asks the
8
- * configured LLM to extract entities and relations from each one, and
9
- * persists the result to stash-local SQLite graph tables keyed by stash root.
10
- * The artifact backs `akm show`'s `related` list and curate's support refs
11
- * (`src/indexer/graph/graph-related.ts`); it plays no part in search ranking.
12
- *
13
- * Disabling — three preconditions must ALL hold for the pass to run:
14
- * 1. An LLM profile must be configured (no provider = no extraction). When
15
- * absent, `resolveIndexPassExecution("graph", config).runner` is
16
- * `undefined` and the pass short-circuits.
17
- * 2. The selected strategy's `processes.graphExtraction.enabled !== false`
18
- * — the feature-gate layer (historically v1 spec §14, since superseded by
19
- * the 0.8.0 profile shape). Set to `false` to block the pass at the
20
- * feature-gate layer (no network call may ever issue).
21
- * 3. `index.graph.llm !== false` — the per-pass opt-out layer (#208).
22
- * Set to `false` to skip just this pass while leaving other passes
23
- * that share the same LLM profile enabled.
24
- * Toggling any one off does NOT delete the existing persisted graph — the
25
- * user keeps the related links they already have, they just stop
26
- * refreshing.
27
- *
28
- * Locked v1 contract:
29
- * - LLM access is exclusively via the frozen runner returned by
30
- * `resolveIndexPassExecution("graph", config)`.
31
- * - The graph rows are an indexer artifact, NOT a user-visible
32
- * asset. It does not have an asset ref, does not appear in search
33
- * hits, and is not addressable via `akm show`. The persisted artifact
34
- * lives in indexer-owned SQLite tables (`replaceStoredGraph` /
35
- * `loadStoredGraphSnapshot` in `../db/graph-db.ts`), NOT as a file on
36
- * disk (R-065 #3 — this comment previously described a retired
37
- * `fs.writeFile`-based storage layout) — `writeAssetToSource` is
38
- * reserved for asset writes (CLAUDE.md / spec §10 step 5).
39
- */
40
- import fs from "node:fs";
41
- import path from "node:path";
42
- import { stashDirFor } from "../../core/asset/asset-placement.js";
43
- import { parseFrontmatter } from "../../core/asset/frontmatter.js";
44
- import { concurrentMap } from "../../core/concurrent.js";
45
- import { getIndexPassConfig, resolveBatchSize } from "../../core/config/config.js";
46
- import { ConfigError } from "../../core/errors.js";
47
- import { warn, warnVerbose } from "../../core/warn.js";
48
- import { assertRunnerCredentials } from "../../integrations/agent/runner-dispatch.js";
49
- import { isProcessEnabled } from "../../llm/feature-gate.js";
50
- import * as graphExtract from "../../llm/graph-extract.js";
51
- import { resolveIndexPassExecution } from "../../llm/index-passes.js";
52
- import { computeBodyHash, getLlmCacheEntriesByRefs, upsertLlmCacheEntry, } from "../../storage/repositories/index-llm-cache-repository.js";
53
- import { loadStoredGraphMeta, loadStoredGraphSnapshot, replaceStoredGraph } from "../db/graph-db.js";
54
- import { walkMarkdownFiles } from "../walk/walker.js";
55
- /**
56
- * The frozen execution a graph call runs under: the invocation's own runner
57
- * when the caller passed one (it already passed its own gates), otherwise the
58
- * configured `graph` pass's. Undefined when the feature gate closes the pass.
59
- */
60
- function selectGraphExecution(holder, config) {
61
- if (Object.hasOwn(holder, "llmRunner")) {
62
- return {
63
- execution: Object.freeze({ runner: holder.llmRunner ?? undefined, notices: Object.freeze([]) }),
64
- featureConfig: { ...config, index: { ...config.index, graph: { ...config.index?.graph, enabled: true } } },
65
- };
66
- }
67
- if (!isProcessEnabled("index", "graph_extraction", config))
68
- return undefined;
69
- return { execution: resolveIndexPassExecution("graph", config), featureConfig: config };
70
- }
71
- const EMPTY_RESULT = {
72
- considered: 0,
73
- extracted: 0,
74
- totalEntities: 0,
75
- totalRelations: 0,
76
- written: false,
77
- quality: {
78
- consideredFiles: 0,
79
- extractedFiles: 0,
80
- entityCount: 0,
81
- relationCount: 0,
82
- extractionCoverage: 0,
83
- density: 0,
84
- },
85
- telemetry: {
86
- cacheHits: 0,
87
- cacheMisses: 0,
88
- truncationCount: 0,
89
- failureCount: 0,
90
- retryAttempts: 0,
91
- },
92
- warnings: [],
93
- };
94
- const DEFAULT_GRAPH_EXTRACTION_INCLUDE_TYPES = ["memory", "knowledge"];
95
- const SUPPORTED_GRAPH_EXTRACTION_INCLUDE_TYPES = new Set([
96
- "memory",
97
- "knowledge",
98
- "skill",
99
- "command",
100
- "agent",
101
- "workflow",
102
- "lesson",
103
- "task",
104
- ]);
105
- const GRAPH_CACHE_VARIANT_PREFIX = "graph-extraction";
106
- function normalizeConfidence(raw) {
107
- if (typeof raw !== "number" || !Number.isFinite(raw))
108
- return undefined;
109
- return Math.max(0, Math.min(1, raw));
110
- }
111
- export function getGraphExtractorId(config) {
112
- const fingerprint = computeBodyHash(JSON.stringify({
113
- promptVersion: graphExtract.GRAPH_EXTRACT_PROMPT_VERSION,
114
- model: config.model,
115
- batchSize: config.batchSize,
116
- includeTypes: config.includeTypes,
117
- maxChunkBodyChars: 1600,
118
- maxBatchBodyChars: 1600,
119
- })).slice(0, 16);
120
- return `${GRAPH_CACHE_VARIANT_PREFIX}:${graphExtract.GRAPH_EXTRACT_PROMPT_VERSION}:${config.model}:${fingerprint}`;
121
- }
122
- /**
123
- * GR-D16: one notice when this run's extractor differs from the one that last
124
- * wrote the graph. Cached extractions are keyed by extractor, so a config
125
- * change that alters it (model, batch size, included types, prompt version)
126
- * re-extracts every cached file; the notice says which change and how many.
127
- */
128
- function extractorChangeNotice(args) {
129
- const { previous, current, files, db } = args;
130
- if (!previous?.extractorId || previous.extractorId === current.extractorId)
131
- return undefined;
132
- const cachedUnder = (cacheVariant) => new Set(planEligibleGraphExtractions({ eligible: files, db, reEnrich: false, cacheVariant })
133
- .filter((plan) => plan.kind === "cache-hit")
134
- .map((plan) => plan.candidate.absPath));
135
- const stillCached = cachedUnder(current.extractorId);
136
- const reextracted = [...cachedUnder(previous.extractorId)].filter((file) => !stillCached.has(file)).length;
137
- const changes = [
138
- previous.model !== current.model ? `model ${previous.model} -> ${current.model}` : undefined,
139
- previous.batchSize !== current.batchSize ? `batch size ${previous.batchSize} -> ${current.batchSize}` : undefined,
140
- previous.promptVersion !== current.promptVersion
141
- ? `prompt ${previous.promptVersion} -> ${current.promptVersion}`
142
- : undefined,
143
- ].filter((change) => change !== undefined);
144
- return (`graph extraction: the extractor changed (${changes.join(", ") || "included asset types"}), ` +
145
- `so ${reextracted} file(s) with a cached extraction will be extracted again.`);
146
- }
147
- function buildLowQualityWarnings(quality, telemetry) {
148
- const warnings = [];
149
- if (quality.consideredFiles >= 5 && quality.extractionCoverage < 0.3) {
150
- warnings.push(`Low graph extraction coverage (${quality.extractedFiles}/${quality.consideredFiles}, ${quality.extractionCoverage}).`);
151
- }
152
- if (quality.entityCount >= 8 && quality.relationCount === 0) {
153
- warnings.push("Graph extraction produced many entities but no relations.");
154
- }
155
- if (telemetry.failureCount > 0) {
156
- warnings.push(`Graph extraction encountered ${telemetry.failureCount} failed file extraction(s).`);
157
- }
158
- return warnings;
159
- }
160
- /**
161
- * Failure-rate abort for the extraction run (R2), modelled on consolidate's
162
- * chunk-level guard (`ABORT_MIN_CHUNKS`/`ABORT_FAILURE_RATE` in
163
- * consolidate.ts, C-6/#392): rate-based over a minimum sample so a couple of
164
- * transient per-file failures cannot abort a run that would otherwise
165
- * recover, while a systemically dead provider stops burning through the
166
- * rest of the eligible set. The existing "one failure must not abort the
167
- * rest" behaviour for individual files is untouched — this only stops
168
- * further model calls once the failure rate itself is the signal.
169
- */
170
- const GRAPH_EXTRACTION_ABORT_MIN_ATTEMPTS = 4;
171
- const GRAPH_EXTRACTION_ABORT_FAILURE_RATE = 0.5;
172
- /** Records one attempted (non-cache-hit) model call and flips `aborted` once the failure-rate threshold is crossed. */
173
- function recordGraphExtractionAttempt(state, failed) {
174
- if (state.aborted)
175
- return;
176
- state.attempts += 1;
177
- if (failed)
178
- state.failures += 1;
179
- if (state.attempts < GRAPH_EXTRACTION_ABORT_MIN_ATTEMPTS)
180
- return;
181
- const failureRate = state.failures / state.attempts;
182
- if (failureRate < GRAPH_EXTRACTION_ABORT_FAILURE_RATE)
183
- return;
184
- state.aborted = true;
185
- state.message =
186
- `graph extraction aborted — failure rate ${(failureRate * 100).toFixed(0)}% over ${state.attempts} ` +
187
- `attempt(s) (>= ${GRAPH_EXTRACTION_ABORT_FAILURE_RATE * 100}% threshold). LLM may be unavailable.`;
188
- warn(state.message);
189
- }
190
- export function getGraphExtractionIncludeTypes(config) {
191
- const configured = getIndexPassConfig(config.index, "graph")?.graphExtractionIncludeTypes;
192
- if (!configured || configured.length === 0)
193
- return [...DEFAULT_GRAPH_EXTRACTION_INCLUDE_TYPES];
194
- const out = [];
195
- const seen = new Set();
196
- for (const rawType of configured) {
197
- const type = rawType.trim().toLowerCase();
198
- if (!type || seen.has(type))
199
- continue;
200
- if (!SUPPORTED_GRAPH_EXTRACTION_INCLUDE_TYPES.has(type))
201
- continue;
202
- seen.add(type);
203
- out.push(type);
204
- }
205
- return out.length > 0 ? out : [...DEFAULT_GRAPH_EXTRACTION_INCLUDE_TYPES];
206
- }
207
- function validateGraphCacheShape(raw) {
208
- if (!raw || typeof raw !== "object")
209
- return undefined;
210
- const obj = raw;
211
- if (!Array.isArray(obj.entities) || !obj.entities.every((e) => typeof e === "string"))
212
- return undefined;
213
- if (obj.relations !== undefined &&
214
- (!Array.isArray(obj.relations) ||
215
- !obj.relations.every((r) => {
216
- if (!r || typeof r !== "object")
217
- return false;
218
- const rel = r;
219
- if (typeof rel.from !== "string" || typeof rel.to !== "string")
220
- return false;
221
- if (rel.type !== undefined && typeof rel.type !== "string")
222
- return false;
223
- if (rel.confidence !== undefined && (typeof rel.confidence !== "number" || !Number.isFinite(rel.confidence))) {
224
- return false;
225
- }
226
- return true;
227
- }))) {
228
- return undefined;
229
- }
230
- return {
231
- entities: obj.entities,
232
- relations: Array.isArray(obj.relations) ? obj.relations : [],
233
- confidence: normalizeConfidence(obj.confidence),
234
- ...(typeof obj.status === "string" ? { status: obj.status } : {}),
235
- ...(typeof obj.reason === "string" ? { reason: obj.reason } : {}),
236
- };
237
- }
238
- /**
239
- * A `"failed"` extraction (provider error, invalid JSON, context overflow —
240
- * see {@link GraphExtractionStatus}) must never be reused as a cache hit or
241
- * re-persisted as one. R2: a dead provider upserted ~30,900 rows shaped
242
- * `{"entities":[],"relations":[],"status":"failed","reason":"llm_error"}`,
243
- * and both hit paths (the `llm_enrichment_cache` lookup and `reuseGraphNode`
244
- * over the previous graph) validated the shape without checking `status`, so
245
- * 92% of the persisted graph became a permanent hit that never retried. A
246
- * failed result becomes a miss naturally and is overwritten on the next
247
- * successful extraction; existing failed rows are left on disk untouched.
248
- */
249
- function isFailedExtractionStatus(status) {
250
- return status === "failed";
251
- }
252
- function loadGraphFile(stashRoot, db) {
253
- const graph = loadStoredGraphSnapshot(stashRoot, db);
254
- if (!graph)
255
- return { files: [] };
256
- const out = [];
257
- for (const node of graph.files) {
258
- const cacheShape = validateGraphCacheShape({ entities: node.entities, relations: node.relations });
259
- if (!cacheShape)
260
- continue;
261
- out.push({
262
- path: node.path,
263
- type: node.type,
264
- bodyHash: node.bodyHash,
265
- entities: cacheShape.entities,
266
- relations: cacheShape.relations,
267
- confidence: normalizeConfidence(node.confidence),
268
- ...(node.status ? { status: node.status } : {}),
269
- ...(node.reason ? { reason: node.reason } : {}),
270
- ...(node.extractionRunId ? { extractionRunId: node.extractionRunId } : {}),
271
- });
272
- }
273
- return {
274
- files: out,
275
- ...(graph.telemetry ? { telemetry: graph.telemetry } : {}),
276
- };
277
- }
278
- /**
279
- * The stored graph after a run: each refreshed node replaces the stored node
280
- * for its path, and every other stored node is kept as it was — files a
281
- * scoped run (`candidatePaths`, `topN`) did not select, and files an aborted
282
- * run never reached. With `keptPaths`, stored nodes outside it are dropped
283
- * (the file left the eligible set); without it, nothing is dropped.
284
- */
285
- function mergeGraphNodes(previousNodes, refreshedNodes, keptPaths) {
286
- const refreshedByPath = new Map(refreshedNodes.map((node) => [node.path, node]));
287
- const merged = [];
288
- for (const node of previousNodes) {
289
- const refreshed = refreshedByPath.get(node.path);
290
- if (refreshed) {
291
- merged.push(refreshed);
292
- refreshedByPath.delete(node.path);
293
- }
294
- else if (!keptPaths || keptPaths.has(node.path)) {
295
- merged.push(node);
296
- }
297
- }
298
- merged.push(...refreshedByPath.values());
299
- return merged;
300
- }
301
- /**
302
- * A file is a cache hit only through `llm_enrichment_cache`, whose variant is
303
- * the extractor id. A stored graph node is never reused here: the graph keeps
304
- * nodes that older extractors wrote (files a run did not reach), and reusing
305
- * one would record another extractor's output as this one's.
306
- */
307
- function planEligibleGraphExtractions(args) {
308
- const { eligible, db, reEnrich, cacheVariant } = args;
309
- const cacheEntries = reEnrich
310
- ? new Map()
311
- : getLlmCacheEntriesByRefs(db, eligible.map((candidate) => candidate.absPath), cacheVariant);
312
- return eligible.map((candidate) => {
313
- const bodyHash = computeBodyHash(candidate.body);
314
- if (reEnrich)
315
- return { kind: "model", candidate, bodyHash };
316
- const entry = cacheEntries.get(candidate.absPath);
317
- if (entry?.bodyHash === bodyHash) {
318
- try {
319
- const cached = validateGraphCacheShape(JSON.parse(entry.resultJson));
320
- if (cached && !isFailedExtractionStatus(cached.status)) {
321
- return { kind: "cache-hit", candidate, bodyHash, cached };
322
- }
323
- }
324
- catch {
325
- // A corrupt cache row is a miss.
326
- }
327
- }
328
- return { kind: "model", candidate, bodyHash };
329
- });
330
- }
331
- function extractionRecord(candidate, bodyHash, shape) {
332
- return {
333
- absPath: candidate.absPath,
334
- type: candidate.type,
335
- bodyHash,
336
- entities: shape.entities,
337
- relations: shape.relations,
338
- ...(shape.confidence !== undefined ? { confidence: shape.confidence } : {}),
339
- ...(shape.status ? { status: shape.status } : {}),
340
- ...(shape.reason ? { reason: shape.reason } : {}),
341
- };
342
- }
343
- /**
344
- * Run the planned extractions in chunks of `batchSize`: cache hits are taken
345
- * as-is, and each chunk's model plans go to the provider in one
346
- * `extractGraphFromBodies` call (a one-body call is the per-asset path).
347
- */
348
- async function extractGraphBatches(args) {
349
- const { plans, batchSize, signal, db, cacheVariant, telemetry, llmRunner, featureConfig, onFallback, abortState, batchState, runtimeTelemetry, onNotices, reportProgress, maxChunksPerAsset, } = args;
350
- const results = new Array(plans.length).fill(undefined);
351
- const chunkStarts = [];
352
- for (let start = 0; start < plans.length; start += batchSize)
353
- chunkStarts.push(start);
354
- let configFailure;
355
- await concurrentMap(chunkStarts, async (start) => {
356
- if (signal?.aborted)
357
- return;
358
- const chunk = plans.slice(start, start + batchSize);
359
- const reportChunkProgress = () => {
360
- for (const [j, plan] of chunk.entries())
361
- reportProgress(plan.candidate.absPath, results[start + j]);
362
- };
363
- const modelPlans = [];
364
- for (const [offset, plan] of chunk.entries()) {
365
- if (plan.kind === "model") {
366
- modelPlans.push({ plan, offset });
367
- continue;
368
- }
369
- telemetry.cacheHits += 1;
370
- results[start + offset] = extractionRecord(plan.candidate, plan.bodyHash, plan.cached);
371
- }
372
- if (modelPlans.length === 0 || abortState.aborted) {
373
- reportChunkProgress();
374
- return;
375
- }
376
- telemetry.cacheMisses += modelPlans.length;
377
- let batchExtractions;
378
- try {
379
- batchExtractions = await graphExtract.extractGraphFromBodies(llmRunner, modelPlans.map(({ plan }) => plan.candidate.body), signal, featureConfig, onFallback, {
380
- batchState,
381
- telemetry: runtimeTelemetry,
382
- onNotices,
383
- ...(maxChunksPerAsset != null ? { maxChunksPerAsset } : {}),
384
- });
385
- }
386
- catch (error) {
387
- if (error instanceof ConfigError) {
388
- configFailure ??= error;
389
- return;
390
- }
391
- throw error;
392
- }
393
- let dispatchHadResult = false;
394
- let dispatchAllFailed = true;
395
- for (const [i, { plan, offset }] of modelPlans.entries()) {
396
- const extraction = batchExtractions[i];
397
- if (!extraction)
398
- continue;
399
- const cacheShape = {
400
- entities: extraction.entities,
401
- relations: extraction.relations,
402
- ...(extraction.confidence !== undefined ? { confidence: extraction.confidence } : {}),
403
- ...(extraction.status ? { status: extraction.status } : {}),
404
- ...(extraction.reason ? { reason: extraction.reason } : {}),
405
- };
406
- dispatchHadResult = true;
407
- if (!isFailedExtractionStatus(cacheShape.status)) {
408
- dispatchAllFailed = false;
409
- upsertLlmCacheEntry(db, plan.candidate.absPath, plan.bodyHash, JSON.stringify(cacheShape), cacheVariant);
410
- }
411
- results[start + offset] = extractionRecord(plan.candidate, plan.bodyHash, cacheShape);
412
- }
413
- // One attempt per `extractGraphFromBodies` dispatch (this chunk's batch
414
- // call), not one per file it covers — mirrors consolidate.ts's
415
- // totalChunksProcessed++/totalChunksFailed, which count once per chunk
416
- // regardless of how many memories are in it. Counting per file let a
417
- // single batched provider_error satisfy GRAPH_EXTRACTION_ABORT_MIN_ATTEMPTS
418
- // after one HTTP failure whenever graphExtractionBatchSize >= 4.
419
- if (dispatchHadResult)
420
- recordGraphExtractionAttempt(abortState, dispatchAllFailed);
421
- reportChunkProgress();
422
- }, llmRunner.connection.concurrency ?? 1);
423
- return { results, ...(configFailure ? { configFailure } : {}) };
424
- }
425
- /**
426
- * Top-level entry point. Returns a no-op result when the pass is disabled.
427
- *
428
- * Three preconditions — ALL must hold for the pass to run:
429
- *
430
- * 1. **Provider configured** — an LLM profile must be selectable. Without a
431
- * configured provider, `resolveIndexPassExecution("graph", config).runner`
432
- * is `undefined` (the pass cannot run because there is no model to call).
433
- * 2. **Feature gate** — the selected strategy's `processes.graphExtraction.enabled`
434
- * (defaults to `true`). When `false`, no network call may issue regardless
435
- * of per-pass settings.
436
- * 3. **Per-pass gate** — `index.graph.llm` (defaults to `true`). When
437
- * `false`, the indexer simply skips this pass for the current run.
438
- *
439
- * If any of the three is missing or `false`, this function short-circuits
440
- * to an empty no-op result, leaving any existing persisted graph untouched.
441
- *
442
- * Eligible files are chunked by the resolved batch size
443
- * (`graphExtractionBatchSize`) and each chunk is one `extractGraphFromBodies`
444
- * call; a batch size of 1 is one call per asset.
445
- */
446
- export async function runGraphExtractionPass(ctx) {
447
- const { config, sources, signal, db, reEnrich, onProgress, options = {} } = ctx;
448
- // Gate 1 — the feature gate (selected strategy's
449
- // processes.graphExtraction.enabled, default enabled).
450
- const selection = selectGraphExecution(ctx, config);
451
- if (!selection)
452
- return { ...EMPTY_RESULT };
453
- const noticesByKey = new Map();
454
- const onNotices = (notices) => {
455
- for (const notice of notices)
456
- noticesByKey.set(JSON.stringify(notice), notice);
457
- };
458
- const emptyResult = () => ({
459
- ...EMPTY_RESULT,
460
- ...(noticesByKey.size > 0 ? { notices: Object.freeze([...noticesByKey.values()]) } : {}),
461
- });
462
- // Gate 2 — per-pass opt-out (#208). Retain the whole frozen resolution so
463
- // selection-time lowering notices cannot be separated from the runner.
464
- onNotices(selection.execution.notices);
465
- const llmRunner = selection.execution.runner;
466
- if (!llmRunner) {
467
- const reason = getIndexPassConfig(config.index, "graph")?.enabled === false
468
- ? "index.graph.enabled is false"
469
- : "no LLM engine is configured";
470
- warnVerbose(`graph extraction: skipped because ${reason}.`);
471
- return emptyResult();
472
- }
473
- const { featureConfig } = selection;
474
- // The pass only writes to the primary (working) stash. Read-only caches
475
- // (git, npm, website) are deliberately untouched — the graph artifact for
476
- // those sources would be clobbered by the next sync().
477
- const primary = sources[0];
478
- if (!primary) {
479
- warnVerbose("graph extraction: skipped because no primary stash source is available.");
480
- return emptyResult();
481
- }
482
- if (!db) {
483
- warn("graph extraction: no database handle available; skipping graph persistence.");
484
- return emptyResult();
485
- }
486
- const includeTypes = options.includeTypes ?? getGraphExtractionIncludeTypes(config);
487
- const previousGraph = loadGraphFile(primary.path, db);
488
- const batchSize = resolveBatchSize(options.batchSize ?? getIndexPassConfig(config.index, "graph")?.graphExtractionBatchSize, llmRunner.connection.contextLength);
489
- const extractorId = getGraphExtractorId({ model: llmRunner.connection.model, batchSize, includeTypes });
490
- const scan = collectEligibleFiles(primary.path, includeTypes);
491
- // The stored nodes this run keeps without touching them: every eligible file
492
- // (outside candidatePaths or topN, or never reached before an abort). Only a
493
- // node whose file left the eligible set — gone, emptied, inferred, or of a
494
- // type no longer included — is dropped, and an incomplete scan drops nothing.
495
- const keptPaths = scan.complete ? new Set(scan.files.map((file) => file.absPath)) : undefined;
496
- let eligible = scan.files.filter((candidate) => !options.candidatePaths || options.candidatePaths.has(candidate.absPath));
497
- // P2 (#624): when topN is set, rank the (already candidate-filtered)
498
- // eligible set by utility_scores DESC and keep only the top-N. Unset issues
499
- // no ranking query. Ranking composes WITH the candidatePaths filter:
500
- // scoped-then-ranked-then-sliced.
501
- if (options.topN != null && options.topN >= 0) {
502
- eligible = rankCandidatesByUtility(db, eligible).slice(0, options.topN);
503
- }
504
- const considered = eligible.length;
505
- const eligiblePlans = planEligibleGraphExtractions({ eligible, db, reEnrich, cacheVariant: extractorId });
506
- if (signal?.aborted)
507
- return emptyResult();
508
- // Validate exactly once iff classification found real model work. Cache
509
- // writes and graph replacement happen after this boundary, so a missing
510
- // credential cannot partially mutate a batch.
511
- if (eligiblePlans.some((plan) => plan.kind === "model"))
512
- assertRunnerCredentials(llmRunner);
513
- if (considered === 0) {
514
- const scoped = options.candidatePaths ? ` matching ${options.candidatePaths.size} candidate path(s)` : "";
515
- warnVerbose(`graph extraction: skipped because no eligible files${scoped} were found under ${primary.path}. ` +
516
- `includeTypes=${includeTypes.join(",")}`);
517
- return emptyResult();
518
- }
519
- let totalEntities = 0;
520
- let totalRelations = 0;
521
- let processed = 0;
522
- let extracted = 0;
523
- onProgress?.({ processed, total: considered, extracted, totalEntities, totalRelations });
524
- const reportProgress = (currentPath, result) => {
525
- processed += 1;
526
- if (result) {
527
- if (result.entities.length > 0)
528
- extracted += 1;
529
- totalEntities += result.entities.length;
530
- totalRelations += result.relations.length;
531
- }
532
- onProgress?.({
533
- processed,
534
- total: considered,
535
- extracted,
536
- totalEntities,
537
- totalRelations,
538
- currentPath,
539
- });
540
- };
541
- const extractionRunId = crypto.randomUUID();
542
- const telemetry = {
543
- extractorId,
544
- extractionRunId,
545
- model: llmRunner.connection.model,
546
- promptVersion: graphExtract.GRAPH_EXTRACT_PROMPT_VERSION,
547
- batchSize,
548
- cacheHits: 0,
549
- cacheMisses: 0,
550
- truncationCount: 0,
551
- failureCount: 0,
552
- htmlErrorCount: 0,
553
- retryAttempts: 0,
554
- nonArrayBatchFailures: 0,
555
- };
556
- const runtimeTelemetry = {
557
- truncationCount: 0,
558
- failureCount: 0,
559
- htmlErrorCount: 0,
560
- retryAttempts: 0,
561
- filteredGenericEntities: 0,
562
- filteredInvalidRelations: 0,
563
- filteredLowConfidenceRelations: 0,
564
- contextBatchRetries: 0,
565
- nonArrayBatchFailures: 0,
566
- };
567
- const abortState = { attempts: 0, failures: 0, aborted: false };
568
- const extractorNotice = extractorChangeNotice({
569
- previous: previousGraph.telemetry,
570
- current: {
571
- extractorId,
572
- model: llmRunner.connection.model,
573
- batchSize,
574
- promptVersion: graphExtract.GRAPH_EXTRACT_PROMPT_VERSION,
575
- },
576
- files: scan.files,
577
- db,
578
- });
579
- if (extractorNotice)
580
- warn(extractorNotice);
581
- warnVerbose(`graph extraction: starting for ${considered} eligible file(s) under ${primary.path}; ` +
582
- `includeTypes=${includeTypes.join(",")}, batchSize=${batchSize}, concurrency=${llmRunner.connection.concurrency ?? 1}, ` +
583
- `reEnrich=${reEnrich === true}, candidateScoped=${options.candidatePaths ? "true" : "false"}.`);
584
- const { results, configFailure } = await extractGraphBatches({
585
- plans: eligiblePlans,
586
- batchSize,
587
- signal,
588
- db,
589
- cacheVariant: extractorId,
590
- telemetry,
591
- llmRunner,
592
- featureConfig,
593
- onFallback: (evt) => warn(`[akm] LLM fallback for ${evt.feature}: ${evt.reason}`),
594
- batchState: { batchingDisabled: false, nonArrayBatchFailures: 0 },
595
- runtimeTelemetry,
596
- abortState,
597
- onNotices,
598
- reportProgress,
599
- ...(options.maxChunksPerAsset != null ? { maxChunksPerAsset: options.maxChunksPerAsset } : {}),
600
- });
601
- if (configFailure)
602
- throw configFailure;
603
- // A failed attempt says nothing about the file, so a stored node for it stays
604
- // as it was; only a file with no stored node records the failure.
605
- const storedPaths = new Set(previousGraph.files.map((node) => node.path));
606
- const nodes = results.flatMap((result) => !result || (isFailedExtractionStatus(result.status) && storedPaths.has(result.absPath))
607
- ? []
608
- : [toGraphNode(result, extractionRunId)]);
609
- telemetry.truncationCount = runtimeTelemetry.truncationCount ?? 0;
610
- telemetry.truncatedChunks = runtimeTelemetry.truncatedChunks ?? 0;
611
- telemetry.failureCount = runtimeTelemetry.failureCount ?? 0;
612
- telemetry.htmlErrorCount = runtimeTelemetry.htmlErrorCount ?? 0;
613
- telemetry.retryAttempts = runtimeTelemetry.retryAttempts ?? 0;
614
- telemetry.nonArrayBatchFailures = runtimeTelemetry.nonArrayBatchFailures ?? 0;
615
- telemetry.filteredGenericEntities = runtimeTelemetry.filteredGenericEntities ?? 0;
616
- telemetry.filteredInvalidRelations = runtimeTelemetry.filteredInvalidRelations ?? 0;
617
- telemetry.filteredLowConfidenceRelations = runtimeTelemetry.filteredLowConfidenceRelations ?? 0;
618
- telemetry.contextBatchRetries = runtimeTelemetry.contextBatchRetries ?? 0;
619
- telemetry.aborted = abortState.aborted;
620
- const graph = buildGraphFile(primary.path, mergeGraphNodes(previousGraph.files, nodes, keptPaths), telemetry);
621
- const written = writeGraphFile(db, graph);
622
- const quality = loadStoredGraphMeta(primary.path, db)?.quality ?? EMPTY_RESULT.quality;
623
- const warnings = buildLowQualityWarnings(quality, telemetry);
624
- if (extractorNotice)
625
- warnings.push(extractorNotice);
626
- if (abortState.message)
627
- warnings.push(abortState.message);
628
- for (const warning of warnings)
629
- warnVerbose(`graph extraction quality: ${warning}`);
630
- warnVerbose(`graph extraction: ${written ? "persisted" : "did not persist"} graph for ${primary.path}; ` +
631
- `considered=${considered}, extractedThisRun=${extracted}, storedFiles=${quality.consideredFiles}, ` +
632
- `entities=${quality.entityCount}, relations=${quality.relationCount}, coverage=${quality.extractionCoverage}.`);
633
- return {
634
- considered,
635
- extracted,
636
- totalEntities,
637
- totalRelations,
638
- written,
639
- quality,
640
- telemetry,
641
- warnings,
642
- ...(noticesByKey.size > 0 ? { notices: Object.freeze([...noticesByKey.values()]) } : {}),
643
- };
644
- }
645
- /**
646
- * The persisted node for one extraction outcome: entities trimmed and kept
647
- * once per {@link graphExtract.normalizeEntityKey} (the first form wins),
648
- * relations trimmed.
649
- */
650
- function toGraphNode(record, extractionRunId) {
651
- const confidence = normalizeConfidence(record.confidence);
652
- const entityKeys = new Set();
653
- const entities = record.entities
654
- .map((entity) => entity.trim())
655
- .filter((entity) => {
656
- const key = graphExtract.normalizeEntityKey(entity);
657
- if (!key || entityKeys.has(key))
658
- return false;
659
- entityKeys.add(key);
660
- return true;
661
- });
662
- return {
663
- path: record.absPath,
664
- type: record.type,
665
- bodyHash: record.bodyHash,
666
- entities,
667
- relations: record.relations
668
- .map((r) => ({
669
- from: r.from.trim(),
670
- to: r.to.trim(),
671
- ...(r.type ? { type: r.type.trim() } : {}),
672
- ...(normalizeConfidence(r.confidence) !== undefined ? { confidence: normalizeConfidence(r.confidence) } : {}),
673
- }))
674
- .filter((relation) => relation.from && relation.to),
675
- ...(confidence !== undefined ? { confidence } : {}),
676
- status: record.status ?? (record.entities.length > 0 ? "extracted" : "empty"),
677
- reason: record.reason ?? (record.entities.length > 0 ? "none" : "no_graph_content"),
678
- extractionRunId,
679
- };
680
- }
681
- /** The graph snapshot to store for `files`; its counts are derived from the stored rows on write. */
682
- function buildGraphFile(stashRoot, files, telemetry) {
683
- return {
684
- generatedAt: new Date().toISOString(),
685
- stashRoot,
686
- files,
687
- ...(telemetry ? { telemetry } : {}),
688
- };
689
- }
690
- // ── Eligible-file detection ─────────────────────────────────────────────────
691
- /**
692
- * Rank eligible graph-extraction candidates by their entry `utility_scores`,
693
- * highest first, for the incremental high-signal-first sweep (P2 of #624).
694
- *
695
- * The join is READ-ONLY (`entries.file_path = candidate.absPath`, then
696
- * `entries.id -> utility_scores.entry_id`) and does NOT re-couple the graph
697
- * rows to `entries`. It reads the GLOBAL `utility_scores` table (not the
698
- * per-scope `utility_scores_scoped`), so ranking is corpus-wide.
699
- *
700
- * Candidates with no matching `entries` row, or an entry with no
701
- * `utility_scores` row, get an effective utility of 0 (LEFT JOIN + COALESCE)
702
- * and sort LAST — they are deprioritized, never dropped, so a `topN >= total`
703
- * slice still includes them and they remain reachable on later runs.
704
- *
705
- * Ties (equal utility) break by `file_path` ASC for deterministic output.
706
- * Returns a NEW array; the input is not mutated. SQLite's ~999 bound-parameter
707
- * cap is respected by chunking the `IN (...)` lookup at 500.
708
- *
709
- * Exported for direct unit testing.
710
- */
711
- export function rankCandidatesByUtility(db, candidates) {
712
- if (!db || candidates.length === 0)
713
- return candidates;
714
- const utilityByPath = new Map();
715
- const CHUNK = 500;
716
- for (let start = 0; start < candidates.length; start += CHUNK) {
717
- const chunk = candidates.slice(start, start + CHUNK);
718
- const paths = chunk.map((c) => c.absPath);
719
- const placeholders = paths.map(() => "?").join(", ");
720
- const rows = db
721
- .prepare(`SELECT e.file_path AS file_path, COALESCE(MAX(u.utility), 0) AS utility
722
- FROM entries e
723
- LEFT JOIN utility_scores u ON u.entry_id = e.id
724
- WHERE e.file_path IN (${placeholders})
725
- GROUP BY e.file_path`)
726
- .all(...paths);
727
- for (const row of rows) {
728
- utilityByPath.set(row.file_path, row.utility ?? 0);
729
- }
730
- }
731
- return [...candidates].sort((a, b) => {
732
- const ua = utilityByPath.get(a.absPath) ?? 0;
733
- const ub = utilityByPath.get(b.absPath) ?? 0;
734
- if (ub !== ua)
735
- return ub - ua; // utility DESC
736
- return a.absPath < b.absPath ? -1 : a.absPath > b.absPath ? 1 : 0; // tie-break: path ASC
737
- });
738
- }
739
- /**
740
- * Scan the primary stash for `memory:` and `knowledge:` markdown files
741
- * suitable for graph extraction. The directory layout convention is the
742
- * same one the rest of the indexer uses: `<stashRoot>/<type>/...`.
743
- *
744
- * Inferred-child memories (frontmatter `inferred: true`) are skipped — they
745
- * are already derived summaries, with no additional internal graph structure worth
746
- * extracting.
747
- *
748
- * `complete` is false when a directory or a candidate file could not be read,
749
- * so the result may be missing eligible files.
750
- *
751
- * Exported for direct unit testing.
752
- */
753
- export function collectEligibleFiles(stashRoot, includeTypes = [...DEFAULT_GRAPH_EXTRACTION_INCLUDE_TYPES]) {
754
- const out = [];
755
- let complete = true;
756
- for (const rawType of includeTypes) {
757
- const type = rawType.trim().toLowerCase();
758
- if (!SUPPORTED_GRAPH_EXTRACTION_INCLUDE_TYPES.has(type))
759
- continue;
760
- const stashDir = stashDirFor(type);
761
- if (!stashDir)
762
- continue;
763
- const dir = path.join(stashRoot, stashDir);
764
- if (!fs.existsSync(dir))
765
- continue;
766
- const walked = walkMarkdownFiles(dir);
767
- if (!walked.complete) {
768
- complete = false;
769
- warn(`graph extraction: directory scan under ${dir} is incomplete — some files may be missing`);
770
- }
771
- for (const filePath of walked.files) {
772
- let raw;
773
- try {
774
- raw = fs.readFileSync(filePath, "utf8");
775
- }
776
- catch (err) {
777
- complete = false;
778
- warn(`graph extraction: failed to read candidate file ${filePath}: ${err instanceof Error ? err.message : String(err)}`);
779
- continue;
780
- }
781
- const parsed = parseFrontmatter(raw);
782
- // Skip inferred memory children — they are atomic and there's no
783
- // graph to extract from a single-fact body.
784
- if (type === "memory" && parsed.data.inferred === true)
785
- continue;
786
- const body = parsed.content.trim();
787
- if (!body)
788
- continue;
789
- out.push({ absPath: filePath, type, body });
790
- }
791
- }
792
- return { files: out, complete };
793
- }
794
- // ── Persistence ─────────────────────────────────────────────────────────────
795
- /**
796
- * Persist graph rows into the SQLite index DB.
797
- */
798
- function writeGraphFile(db, graph) {
799
- try {
800
- replaceStoredGraph(db, graph);
801
- return true;
802
- }
803
- catch (err) {
804
- warn(`graph extraction: failed to persist graph for ${graph.stashRoot}: ${err instanceof Error ? err.message : String(err)}`);
805
- return false;
806
- }
807
- }