akm-cli 0.9.17-alpha.4 → 0.9.17-alpha.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +167 -3
- package/dist/commands/improve/execution.js +5 -5
- package/dist/commands/improve/improve-strategies.js +3 -0
- package/dist/commands/improve/loop-stages.js +5 -4
- package/dist/commands/lint/base-linter.js +19 -5
- package/dist/commands/read/curate.js +31 -49
- package/dist/commands/read/show.js +2 -81
- package/dist/commands/sources/bundle-config-ops.js +3 -6
- package/dist/commands/sources/source-manage.js +9 -2
- package/dist/core/asset/asset-placement.js +4 -13
- package/dist/core/config/config.js +1 -1
- package/dist/core/config/schema/improve-processes.js +4 -3
- package/dist/core/config/schema/index-config.js +4 -23
- package/dist/indexer/db/graph-db.js +8 -81
- package/dist/indexer/graph/graph-extraction.js +74 -227
- package/dist/indexer/graph/graph-related.js +5 -4
- package/dist/indexer/search/db-search.js +10 -1
- package/dist/llm/feature-gate.js +0 -3
- package/dist/llm/graph-extract.js +81 -41
- package/dist/scripts/akm-migrate-node.js +4781 -4800
- package/dist/scripts/akm-migrate.js +4781 -4800
- package/dist/storage/repositories/index-entries-repository.js +5 -7
- package/dist/storage/repositories/index-schema.js +16 -34
- package/docs/reference/cli.md +9 -1
- package/docs/reference/configuration.md +12 -0
- package/package.json +1 -1
- package/schemas/akm-config.json +0 -6
|
@@ -26,7 +26,7 @@
|
|
|
26
26
|
import fs from "node:fs";
|
|
27
27
|
import path from "node:path";
|
|
28
28
|
import { toPosix } from "../common.js";
|
|
29
|
-
import {
|
|
29
|
+
import { SCRIPT_EXTENSIONS, WORKFLOW_EXTENSIONS } from "../recognition-util.js";
|
|
30
30
|
const workflowSpec = {
|
|
31
31
|
isRelevantFile: (fileName) => WORKFLOW_EXTENSIONS.includes(path.extname(fileName).toLowerCase()),
|
|
32
32
|
toCanonicalName: (typeRoot, filePath) => {
|
|
@@ -245,21 +245,12 @@ export function assetPathForName(assetType, typeRoot, name) {
|
|
|
245
245
|
* "default" alias is genuinely dual-owned: both `<dir>/.env` and
|
|
246
246
|
* `<dir>/default.env` derive the same canonical name (`toCanonicalName`
|
|
247
247
|
* above), so a physical-owner lookup must consider both without reading
|
|
248
|
-
* either file.
|
|
249
|
-
*
|
|
250
|
-
*
|
|
251
|
-
* the name (see `resolveParentRef`/`isDerivedMemory` in
|
|
252
|
-
* `commands/improve/memory/derived-ref.ts`, and the belief-edge identity
|
|
253
|
-
* channel's own `memory:<name>.derived` refs). The plain `.md` file wins when
|
|
254
|
-
* both exist, so it stays `primary` — first in the returned list — and every
|
|
255
|
-
* caller here already prefers the first candidate that exists on disk. Every
|
|
256
|
-
* other placement type has exactly one inverse spelling.
|
|
248
|
+
* either file. Every other placement type has exactly one inverse spelling —
|
|
249
|
+
* including `memory`: `<name>.derived.md` is a separate item that owns
|
|
250
|
+
* `<name>.derived`, never a second spelling of `<name>`.
|
|
257
251
|
*/
|
|
258
252
|
export function assetPathCandidatesForName(assetType, typeRoot, name) {
|
|
259
253
|
const primary = assetPathForName(assetType, typeRoot, name);
|
|
260
|
-
if (assetType === "memory" && !name.endsWith(DERIVED_SUFFIX)) {
|
|
261
|
-
return [primary, assetPathForName(assetType, typeRoot, `${name}${DERIVED_SUFFIX}`)];
|
|
262
|
-
}
|
|
263
254
|
if (assetType !== "env")
|
|
264
255
|
return [primary];
|
|
265
256
|
const base = name === "default" ? "" : name.endsWith("/default") ? name.slice(0, -"default".length) : undefined;
|
|
@@ -33,7 +33,7 @@ export { FEEDBACK_FAILURE_MODES } from "./config-schema.js";
|
|
|
33
33
|
* combined prompt size well under common 8K/16K context windows (each body is
|
|
34
34
|
* sliced to ~500 chars in the graph-extract prompt builder).
|
|
35
35
|
*/
|
|
36
|
-
|
|
36
|
+
const DEFAULT_GRAPH_EXTRACTION_BATCH_SIZE = 4;
|
|
37
37
|
/**
|
|
38
38
|
* Approximate character budget per asset body inside a batched
|
|
39
39
|
* graph-extraction prompt — used by {@link resolveBatchSize} to derive a
|
|
@@ -183,9 +183,10 @@ const MEMORY_INFERENCE_PROCESS_FIELDS = {
|
|
|
183
183
|
cls: clsField,
|
|
184
184
|
};
|
|
185
185
|
/**
|
|
186
|
-
* GraphExtraction process fields:
|
|
187
|
-
* batching.
|
|
188
|
-
*
|
|
186
|
+
* GraphExtraction process fields: one strategy's graph extraction scope and
|
|
187
|
+
* batching. `includeTypes` and `batchSize` override `index.graph`'s
|
|
188
|
+
* `graphExtractionIncludeTypes` and `graphExtractionBatchSize`; unset, the
|
|
189
|
+
* pass reads those.
|
|
189
190
|
*/
|
|
190
191
|
const GRAPH_EXTRACTION_PROCESS_FIELDS = {
|
|
191
192
|
// #624 P2: when set, rank eligible files by utility_scores DESC and process
|
|
@@ -32,22 +32,11 @@ const INDEX_PASS_RETIRED_KEYS = new Set([
|
|
|
32
32
|
"maxTokens",
|
|
33
33
|
"capabilities",
|
|
34
34
|
]);
|
|
35
|
-
const INDEX_PASS_KNOWN_KEYS = new Set([
|
|
36
|
-
"engine",
|
|
37
|
-
"model",
|
|
38
|
-
"timeoutMs",
|
|
39
|
-
"enabled",
|
|
40
|
-
"llm",
|
|
41
|
-
"graphExtractionBatchSize",
|
|
42
|
-
"graphExtractionIncludeTypes",
|
|
43
|
-
"lazyGraphExtraction",
|
|
44
|
-
]);
|
|
45
35
|
/**
|
|
46
|
-
* Per-pass `index.<pass>` entry.
|
|
47
|
-
*
|
|
48
|
-
*
|
|
49
|
-
*
|
|
50
|
-
* string` strings — keeps `akm` startup errors actionable.
|
|
36
|
+
* Per-pass `index.<pass>` entry. The preprocess names and drops the retired
|
|
37
|
+
* engine settings above with a targeted message. Any other unknown key is kept
|
|
38
|
+
* and named once by the config loader's schema walk, like an unknown key
|
|
39
|
+
* anywhere else in config.
|
|
51
40
|
*/
|
|
52
41
|
export const IndexPassConfigSchema = z.preprocess((raw, ctx) => {
|
|
53
42
|
if (typeof raw !== "object" || raw === null || Array.isArray(raw)) {
|
|
@@ -62,13 +51,6 @@ export const IndexPassConfigSchema = z.preprocess((raw, ctx) => {
|
|
|
62
51
|
cleaned ??= { ...obj };
|
|
63
52
|
delete cleaned[key];
|
|
64
53
|
}
|
|
65
|
-
else if (!INDEX_PASS_KNOWN_KEYS.has(key)) {
|
|
66
|
-
warnOnce(`index-pass:unknown:${dotted}`, `Unknown key \`${dotted}\` ignored. Per-pass entries support ` +
|
|
67
|
-
"`engine`, `model`, `timeoutMs`, `enabled`, `llm`, `graphExtractionBatchSize`, " +
|
|
68
|
-
"`graphExtractionIncludeTypes`, and `lazyGraphExtraction`.");
|
|
69
|
-
cleaned ??= { ...obj };
|
|
70
|
-
delete cleaned[key];
|
|
71
|
-
}
|
|
72
54
|
}
|
|
73
55
|
return cleaned ?? raw;
|
|
74
56
|
}, z
|
|
@@ -81,7 +63,6 @@ export const IndexPassConfigSchema = z.preprocess((raw, ctx) => {
|
|
|
81
63
|
graphExtractionBatchSize: positiveInt.optional(),
|
|
82
64
|
// Accept-any until Chunk 2 (WI-9.6c) — no longer enum-restricted.
|
|
83
65
|
graphExtractionIncludeTypes: z.array(z.string().min(1)).nonempty().optional(),
|
|
84
|
-
lazyGraphExtraction: z.boolean().optional(),
|
|
85
66
|
})
|
|
86
67
|
.passthrough());
|
|
87
68
|
const MetadataEnhanceSchema = z.object({ enabled: z.boolean().optional() }).passthrough();
|
|
@@ -4,7 +4,9 @@
|
|
|
4
4
|
import { rethrowIfDataDirUnreadable, rethrowIfTestIsolationError } from "../../core/errors.js";
|
|
5
5
|
import { isPathAbsent } from "../../core/path-access.js";
|
|
6
6
|
import { getDbPath } from "../../core/paths.js";
|
|
7
|
+
import { normalizeEntityKey } from "../../llm/graph-extract.js";
|
|
7
8
|
import { closeDatabase, openExistingDatabase } from "../../storage/repositories/index-connection.js";
|
|
9
|
+
import { GRAPH_SCHEMA_VERSION } from "../../storage/repositories/index-schema.js";
|
|
8
10
|
function withReadableGraphDb(db, fn) {
|
|
9
11
|
if (db)
|
|
10
12
|
return fn(db);
|
|
@@ -26,9 +28,6 @@ function withReadableGraphDb(db, fn) {
|
|
|
26
28
|
function uniqueSorted(values) {
|
|
27
29
|
return [...new Set(values)].sort((a, b) => a.localeCompare(b));
|
|
28
30
|
}
|
|
29
|
-
function normalizeEntity(value) {
|
|
30
|
-
return value.trim().toLowerCase();
|
|
31
|
-
}
|
|
32
31
|
/** Child rows joined to their file row, the same join the loaders read through. */
|
|
33
32
|
const STORED_ENTITIES = `graph_file_entities e
|
|
34
33
|
JOIN graph_files gf ON gf.stash_root = e.stash_root AND gf.file_path = e.file_path AND gf.body_hash = e.body_hash
|
|
@@ -113,9 +112,9 @@ function readStoredGraphQuality(db, stashRoot) {
|
|
|
113
112
|
* no entry_id resolution and no orphan-skip — a graph file no longer needs a
|
|
114
113
|
* matching entries row.
|
|
115
114
|
*
|
|
116
|
-
* graph_meta records the snapshot's
|
|
117
|
-
*
|
|
118
|
-
*
|
|
115
|
+
* graph_meta records the snapshot's time and run telemetry; its counts are
|
|
116
|
+
* derived from the rows as stored after the write, never from the caller's
|
|
117
|
+
* in-memory graph.
|
|
119
118
|
*/
|
|
120
119
|
export function replaceStoredGraph(db, graph) {
|
|
121
120
|
const upsertMeta = db.prepare(`INSERT INTO graph_meta (
|
|
@@ -215,10 +214,10 @@ export function replaceStoredGraph(db, graph) {
|
|
|
215
214
|
insertFile.run(graph.stashRoot, node.path, fileOrder, node.type, bodyHash, node.confidence ?? null, status, reason, runId);
|
|
216
215
|
}
|
|
217
216
|
for (const [entityOrder, entity] of node.entities.entries()) {
|
|
218
|
-
insertEntity.run(graph.stashRoot, node.path, bodyHash, entityOrder,
|
|
217
|
+
insertEntity.run(graph.stashRoot, node.path, bodyHash, entityOrder, normalizeEntityKey(entity), entity);
|
|
219
218
|
}
|
|
220
219
|
for (const [relationOrder, relation] of node.relations.entries()) {
|
|
221
|
-
insertRelation.run(graph.stashRoot, node.path, bodyHash, relationOrder,
|
|
220
|
+
insertRelation.run(graph.stashRoot, node.path, bodyHash, relationOrder, normalizeEntityKey(relation.from), relation.from, normalizeEntityKey(relation.to), relation.to, relation.type ?? null, relation.confidence ?? null);
|
|
222
221
|
}
|
|
223
222
|
}
|
|
224
223
|
// Delete files present in DB but absent from the new snapshot. Child
|
|
@@ -231,7 +230,7 @@ export function replaceStoredGraph(db, graph) {
|
|
|
231
230
|
}
|
|
232
231
|
}
|
|
233
232
|
const quality = readStoredGraphQuality(db, graph.stashRoot);
|
|
234
|
-
upsertMeta.run(graph.stashRoot,
|
|
233
|
+
upsertMeta.run(graph.stashRoot, GRAPH_SCHEMA_VERSION, graph.generatedAt, quality.consideredFiles, quality.extractedFiles, quality.entityCount, quality.relationCount, quality.extractionCoverage, quality.density, telemetry?.extractorId ?? null, telemetry?.extractionRunId ?? null, telemetry?.model ?? null, telemetry?.promptVersion ?? null, telemetry?.batchSize ?? null, telemetry?.cacheHits ?? 0, telemetry?.cacheMisses ?? 0, telemetry?.truncationCount ?? 0, telemetry?.failureCount ?? 0);
|
|
235
234
|
})();
|
|
236
235
|
}
|
|
237
236
|
export function deleteStoredGraph(db, stashPath) {
|
|
@@ -243,75 +242,6 @@ export function deleteStoredGraph(db, stashPath) {
|
|
|
243
242
|
db.prepare("DELETE FROM graph_meta WHERE stash_root = ?").run(stashPath);
|
|
244
243
|
})();
|
|
245
244
|
}
|
|
246
|
-
/**
|
|
247
|
-
* #624-P1 — does any graph data exist for a file_path under a stash root?
|
|
248
|
-
* Consumed by show/curate flows (P3) but defined here so the schema and its
|
|
249
|
-
* accessors land together.
|
|
250
|
-
*/
|
|
251
|
-
export function hasGraphData(db, stashRoot, filePath) {
|
|
252
|
-
try {
|
|
253
|
-
const row = db
|
|
254
|
-
.prepare("SELECT 1 AS present FROM graph_files WHERE stash_root = ? AND file_path = ? LIMIT 1")
|
|
255
|
-
.get(stashRoot, filePath);
|
|
256
|
-
return row !== undefined;
|
|
257
|
-
}
|
|
258
|
-
catch {
|
|
259
|
-
return false;
|
|
260
|
-
}
|
|
261
|
-
}
|
|
262
|
-
/**
|
|
263
|
-
* #624-P3 — enqueue a file for lazy graph extraction. Idempotent on the
|
|
264
|
-
* (stash_root, file_path) PK: a second enqueue refreshes body_hash + queued_at
|
|
265
|
-
* and keeps the HIGHER priority. Non-blocking, no LLM call — the queued row is
|
|
266
|
-
* drained later by the graph-extraction pass. Tolerant of a missing table /
|
|
267
|
-
* db error (best-effort), but never masks the bun-test isolation guard.
|
|
268
|
-
*/
|
|
269
|
-
export function enqueueGraphExtraction(db, stashRoot, filePath, bodyHash, priority = 0) {
|
|
270
|
-
try {
|
|
271
|
-
db.prepare(`INSERT INTO graph_extraction_queue (stash_root, file_path, body_hash, priority)
|
|
272
|
-
VALUES (?, ?, ?, ?)
|
|
273
|
-
ON CONFLICT(stash_root, file_path) DO UPDATE SET
|
|
274
|
-
body_hash = excluded.body_hash,
|
|
275
|
-
priority = MAX(graph_extraction_queue.priority, excluded.priority),
|
|
276
|
-
queued_at = datetime('now')`).run(stashRoot, filePath, bodyHash, priority);
|
|
277
|
-
}
|
|
278
|
-
catch (err) {
|
|
279
|
-
rethrowIfTestIsolationError(err);
|
|
280
|
-
}
|
|
281
|
-
}
|
|
282
|
-
/** Read queued graph work without claiming or deleting it. */
|
|
283
|
-
export function peekExtractionQueue(db, stashRoot, limit) {
|
|
284
|
-
try {
|
|
285
|
-
const rows = db
|
|
286
|
-
.prepare(`SELECT file_path, body_hash, priority
|
|
287
|
-
FROM graph_extraction_queue
|
|
288
|
-
WHERE stash_root = ?
|
|
289
|
-
ORDER BY priority DESC, queued_at ASC
|
|
290
|
-
LIMIT ?`)
|
|
291
|
-
.all(stashRoot, limit);
|
|
292
|
-
return rows.map((row) => ({ filePath: row.file_path, bodyHash: row.body_hash, priority: row.priority }));
|
|
293
|
-
}
|
|
294
|
-
catch (err) {
|
|
295
|
-
rethrowIfTestIsolationError(err);
|
|
296
|
-
return [];
|
|
297
|
-
}
|
|
298
|
-
}
|
|
299
|
-
/**
|
|
300
|
-
* Acknowledge the exact queued revision that was classified and handled.
|
|
301
|
-
* A concurrent re-enqueue with a new body hash therefore survives.
|
|
302
|
-
*/
|
|
303
|
-
export function acknowledgeExtractionQueueEntry(db, stashRoot, filePath, bodyHash) {
|
|
304
|
-
try {
|
|
305
|
-
const result = db
|
|
306
|
-
.prepare("DELETE FROM graph_extraction_queue WHERE stash_root = ? AND file_path = ? AND body_hash = ?")
|
|
307
|
-
.run(stashRoot, filePath, bodyHash);
|
|
308
|
-
return result.changes > 0;
|
|
309
|
-
}
|
|
310
|
-
catch (err) {
|
|
311
|
-
rethrowIfTestIsolationError(err);
|
|
312
|
-
return false;
|
|
313
|
-
}
|
|
314
|
-
}
|
|
315
245
|
/**
|
|
316
246
|
* Scoped loader — graph_files rows without entities/relations. Used for
|
|
317
247
|
* orphan detection and entity overview commands.
|
|
@@ -350,7 +280,6 @@ export function loadStoredGraphMeta(stashPath, db) {
|
|
|
350
280
|
const row = readDb
|
|
351
281
|
.prepare(`SELECT
|
|
352
282
|
stash_root,
|
|
353
|
-
schema_version,
|
|
354
283
|
generated_at,
|
|
355
284
|
considered_files,
|
|
356
285
|
extracted_files,
|
|
@@ -375,7 +304,6 @@ export function loadStoredGraphMeta(stashPath, db) {
|
|
|
375
304
|
return {
|
|
376
305
|
stashPath: row.stash_root,
|
|
377
306
|
graphPath: getDbPath(),
|
|
378
|
-
schemaVersion: row.schema_version,
|
|
379
307
|
generatedAt: row.generated_at,
|
|
380
308
|
quality: {
|
|
381
309
|
consideredFiles: row.considered_files,
|
|
@@ -484,7 +412,6 @@ export function loadStoredGraphSnapshot(stashPath, db) {
|
|
|
484
412
|
return {
|
|
485
413
|
stashPath: meta.stashPath,
|
|
486
414
|
graphPath: meta.graphPath,
|
|
487
|
-
schemaVersion: meta.schemaVersion,
|
|
488
415
|
generatedAt: meta.generatedAt,
|
|
489
416
|
...(meta.quality ? { quality: meta.quality } : {}),
|
|
490
417
|
...(meta.telemetry ? { telemetry: meta.telemetry } : {}),
|