akm-cli 0.9.17-alpha.4 → 0.9.17-alpha.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -26,7 +26,7 @@
26
26
  import fs from "node:fs";
27
27
  import path from "node:path";
28
28
  import { toPosix } from "../common.js";
29
- import { DERIVED_SUFFIX, SCRIPT_EXTENSIONS, WORKFLOW_EXTENSIONS } from "../recognition-util.js";
29
+ import { SCRIPT_EXTENSIONS, WORKFLOW_EXTENSIONS } from "../recognition-util.js";
30
30
  const workflowSpec = {
31
31
  isRelevantFile: (fileName) => WORKFLOW_EXTENSIONS.includes(path.extname(fileName).toLowerCase()),
32
32
  toCanonicalName: (typeRoot, filePath) => {
@@ -245,21 +245,12 @@ export function assetPathForName(assetType, typeRoot, name) {
245
245
  * "default" alias is genuinely dual-owned: both `<dir>/.env` and
246
246
  * `<dir>/default.env` derive the same canonical name (`toCanonicalName`
247
247
  * above), so a physical-owner lookup must consider both without reading
248
- * either file. `memory` has a second, analogous duality (#882): a ref to
249
- * `<name>` may own either `<name>.md` or the LLM-inferred `<name>.derived.md`
250
- * twin — `.derived` is a provenance marker on the SAME identity, not part of
251
- * the name (see `resolveParentRef`/`isDerivedMemory` in
252
- * `commands/improve/memory/derived-ref.ts`, and the belief-edge identity
253
- * channel's own `memory:<name>.derived` refs). The plain `.md` file wins when
254
- * both exist, so it stays `primary` — first in the returned list — and every
255
- * caller here already prefers the first candidate that exists on disk. Every
256
- * other placement type has exactly one inverse spelling.
248
+ * either file. Every other placement type has exactly one inverse spelling —
249
+ * including `memory`: `<name>.derived.md` is a separate item that owns
250
+ * `<name>.derived`, never a second spelling of `<name>`.
257
251
  */
258
252
  export function assetPathCandidatesForName(assetType, typeRoot, name) {
259
253
  const primary = assetPathForName(assetType, typeRoot, name);
260
- if (assetType === "memory" && !name.endsWith(DERIVED_SUFFIX)) {
261
- return [primary, assetPathForName(assetType, typeRoot, `${name}${DERIVED_SUFFIX}`)];
262
- }
263
254
  if (assetType !== "env")
264
255
  return [primary];
265
256
  const base = name === "default" ? "" : name.endsWith("/default") ? name.slice(0, -"default".length) : undefined;
@@ -33,7 +33,7 @@ export { FEEDBACK_FAILURE_MODES } from "./config-schema.js";
33
33
  * combined prompt size well under common 8K/16K context windows (each body is
34
34
  * sliced to ~500 chars in the graph-extract prompt builder).
35
35
  */
36
- export const DEFAULT_GRAPH_EXTRACTION_BATCH_SIZE = 4;
36
+ const DEFAULT_GRAPH_EXTRACTION_BATCH_SIZE = 4;
37
37
  /**
38
38
  * Approximate character budget per asset body inside a batched
39
39
  * graph-extraction prompt — used by {@link resolveBatchSize} to derive a
@@ -183,9 +183,10 @@ const MEMORY_INFERENCE_PROCESS_FIELDS = {
183
183
  cls: clsField,
184
184
  };
185
185
  /**
186
- * GraphExtraction process fields: improve-owned graph extraction scope and
187
- * batching. Passed to the invocation directly and never inherited from
188
- * standalone index.graph.
186
+ * GraphExtraction process fields: one strategy's graph extraction scope and
187
+ * batching. `includeTypes` and `batchSize` override `index.graph`'s
188
+ * `graphExtractionIncludeTypes` and `graphExtractionBatchSize`; unset, the
189
+ * pass reads those.
189
190
  */
190
191
  const GRAPH_EXTRACTION_PROCESS_FIELDS = {
191
192
  // #624 P2: when set, rank eligible files by utility_scores DESC and process
@@ -32,22 +32,11 @@ const INDEX_PASS_RETIRED_KEYS = new Set([
32
32
  "maxTokens",
33
33
  "capabilities",
34
34
  ]);
35
- const INDEX_PASS_KNOWN_KEYS = new Set([
36
- "engine",
37
- "model",
38
- "timeoutMs",
39
- "enabled",
40
- "llm",
41
- "graphExtractionBatchSize",
42
- "graphExtractionIncludeTypes",
43
- "lazyGraphExtraction",
44
- ]);
45
35
  /**
46
- * Per-pass `index.<pass>` entry. Uses preprocess + manual validation so we can
47
- * emit targeted error messages ("Retired or misplaced engine setting",
48
- * "Unknown key `index.<pass>.<key>`")
49
- * instead of Zod's generic `Unrecognized key` / `Expected boolean, received
50
- * string` strings — keeps `akm` startup errors actionable.
36
+ * Per-pass `index.<pass>` entry. The preprocess names and drops the retired
37
+ * engine settings above with a targeted message. Any other unknown key is kept
38
+ * and named once by the config loader's schema walk, like an unknown key
39
+ * anywhere else in config.
51
40
  */
52
41
  export const IndexPassConfigSchema = z.preprocess((raw, ctx) => {
53
42
  if (typeof raw !== "object" || raw === null || Array.isArray(raw)) {
@@ -62,13 +51,6 @@ export const IndexPassConfigSchema = z.preprocess((raw, ctx) => {
62
51
  cleaned ??= { ...obj };
63
52
  delete cleaned[key];
64
53
  }
65
- else if (!INDEX_PASS_KNOWN_KEYS.has(key)) {
66
- warnOnce(`index-pass:unknown:${dotted}`, `Unknown key \`${dotted}\` ignored. Per-pass entries support ` +
67
- "`engine`, `model`, `timeoutMs`, `enabled`, `llm`, `graphExtractionBatchSize`, " +
68
- "`graphExtractionIncludeTypes`, and `lazyGraphExtraction`.");
69
- cleaned ??= { ...obj };
70
- delete cleaned[key];
71
- }
72
54
  }
73
55
  return cleaned ?? raw;
74
56
  }, z
@@ -81,7 +63,6 @@ export const IndexPassConfigSchema = z.preprocess((raw, ctx) => {
81
63
  graphExtractionBatchSize: positiveInt.optional(),
82
64
  // Accept-any until Chunk 2 (WI-9.6c) — no longer enum-restricted.
83
65
  graphExtractionIncludeTypes: z.array(z.string().min(1)).nonempty().optional(),
84
- lazyGraphExtraction: z.boolean().optional(),
85
66
  })
86
67
  .passthrough());
87
68
  const MetadataEnhanceSchema = z.object({ enabled: z.boolean().optional() }).passthrough();
@@ -4,7 +4,9 @@
4
4
  import { rethrowIfDataDirUnreadable, rethrowIfTestIsolationError } from "../../core/errors.js";
5
5
  import { isPathAbsent } from "../../core/path-access.js";
6
6
  import { getDbPath } from "../../core/paths.js";
7
+ import { normalizeEntityKey } from "../../llm/graph-extract.js";
7
8
  import { closeDatabase, openExistingDatabase } from "../../storage/repositories/index-connection.js";
9
+ import { GRAPH_SCHEMA_VERSION } from "../../storage/repositories/index-schema.js";
8
10
  function withReadableGraphDb(db, fn) {
9
11
  if (db)
10
12
  return fn(db);
@@ -26,9 +28,6 @@ function withReadableGraphDb(db, fn) {
26
28
  function uniqueSorted(values) {
27
29
  return [...new Set(values)].sort((a, b) => a.localeCompare(b));
28
30
  }
29
- function normalizeEntity(value) {
30
- return value.trim().toLowerCase();
31
- }
32
31
  /** Child rows joined to their file row, the same join the loaders read through. */
33
32
  const STORED_ENTITIES = `graph_file_entities e
34
33
  JOIN graph_files gf ON gf.stash_root = e.stash_root AND gf.file_path = e.file_path AND gf.body_hash = e.body_hash
@@ -113,9 +112,9 @@ function readStoredGraphQuality(db, stashRoot) {
113
112
  * no entry_id resolution and no orphan-skip — a graph file no longer needs a
114
113
  * matching entries row.
115
114
  *
116
- * graph_meta records the snapshot's schema version, time and run telemetry;
117
- * its counts are derived from the rows as stored after the write, never from
118
- * the caller's in-memory graph.
115
+ * graph_meta records the snapshot's time and run telemetry; its counts are
116
+ * derived from the rows as stored after the write, never from the caller's
117
+ * in-memory graph.
119
118
  */
120
119
  export function replaceStoredGraph(db, graph) {
121
120
  const upsertMeta = db.prepare(`INSERT INTO graph_meta (
@@ -215,10 +214,10 @@ export function replaceStoredGraph(db, graph) {
215
214
  insertFile.run(graph.stashRoot, node.path, fileOrder, node.type, bodyHash, node.confidence ?? null, status, reason, runId);
216
215
  }
217
216
  for (const [entityOrder, entity] of node.entities.entries()) {
218
- insertEntity.run(graph.stashRoot, node.path, bodyHash, entityOrder, normalizeEntity(entity), entity);
217
+ insertEntity.run(graph.stashRoot, node.path, bodyHash, entityOrder, normalizeEntityKey(entity), entity);
219
218
  }
220
219
  for (const [relationOrder, relation] of node.relations.entries()) {
221
- insertRelation.run(graph.stashRoot, node.path, bodyHash, relationOrder, normalizeEntity(relation.from), relation.from, normalizeEntity(relation.to), relation.to, relation.type ?? null, relation.confidence ?? null);
220
+ insertRelation.run(graph.stashRoot, node.path, bodyHash, relationOrder, normalizeEntityKey(relation.from), relation.from, normalizeEntityKey(relation.to), relation.to, relation.type ?? null, relation.confidence ?? null);
222
221
  }
223
222
  }
224
223
  // Delete files present in DB but absent from the new snapshot. Child
@@ -231,7 +230,7 @@ export function replaceStoredGraph(db, graph) {
231
230
  }
232
231
  }
233
232
  const quality = readStoredGraphQuality(db, graph.stashRoot);
234
- upsertMeta.run(graph.stashRoot, graph.schemaVersion, graph.generatedAt, quality.consideredFiles, quality.extractedFiles, quality.entityCount, quality.relationCount, quality.extractionCoverage, quality.density, telemetry?.extractorId ?? null, telemetry?.extractionRunId ?? null, telemetry?.model ?? null, telemetry?.promptVersion ?? null, telemetry?.batchSize ?? null, telemetry?.cacheHits ?? 0, telemetry?.cacheMisses ?? 0, telemetry?.truncationCount ?? 0, telemetry?.failureCount ?? 0);
233
+ upsertMeta.run(graph.stashRoot, GRAPH_SCHEMA_VERSION, graph.generatedAt, quality.consideredFiles, quality.extractedFiles, quality.entityCount, quality.relationCount, quality.extractionCoverage, quality.density, telemetry?.extractorId ?? null, telemetry?.extractionRunId ?? null, telemetry?.model ?? null, telemetry?.promptVersion ?? null, telemetry?.batchSize ?? null, telemetry?.cacheHits ?? 0, telemetry?.cacheMisses ?? 0, telemetry?.truncationCount ?? 0, telemetry?.failureCount ?? 0);
235
234
  })();
236
235
  }
237
236
  export function deleteStoredGraph(db, stashPath) {
@@ -243,75 +242,6 @@ export function deleteStoredGraph(db, stashPath) {
243
242
  db.prepare("DELETE FROM graph_meta WHERE stash_root = ?").run(stashPath);
244
243
  })();
245
244
  }
246
- /**
247
- * #624-P1 — does any graph data exist for a file_path under a stash root?
248
- * Consumed by show/curate flows (P3) but defined here so the schema and its
249
- * accessors land together.
250
- */
251
- export function hasGraphData(db, stashRoot, filePath) {
252
- try {
253
- const row = db
254
- .prepare("SELECT 1 AS present FROM graph_files WHERE stash_root = ? AND file_path = ? LIMIT 1")
255
- .get(stashRoot, filePath);
256
- return row !== undefined;
257
- }
258
- catch {
259
- return false;
260
- }
261
- }
262
- /**
263
- * #624-P3 — enqueue a file for lazy graph extraction. Idempotent on the
264
- * (stash_root, file_path) PK: a second enqueue refreshes body_hash + queued_at
265
- * and keeps the HIGHER priority. Non-blocking, no LLM call — the queued row is
266
- * drained later by the graph-extraction pass. Tolerant of a missing table /
267
- * db error (best-effort), but never masks the bun-test isolation guard.
268
- */
269
- export function enqueueGraphExtraction(db, stashRoot, filePath, bodyHash, priority = 0) {
270
- try {
271
- db.prepare(`INSERT INTO graph_extraction_queue (stash_root, file_path, body_hash, priority)
272
- VALUES (?, ?, ?, ?)
273
- ON CONFLICT(stash_root, file_path) DO UPDATE SET
274
- body_hash = excluded.body_hash,
275
- priority = MAX(graph_extraction_queue.priority, excluded.priority),
276
- queued_at = datetime('now')`).run(stashRoot, filePath, bodyHash, priority);
277
- }
278
- catch (err) {
279
- rethrowIfTestIsolationError(err);
280
- }
281
- }
282
- /** Read queued graph work without claiming or deleting it. */
283
- export function peekExtractionQueue(db, stashRoot, limit) {
284
- try {
285
- const rows = db
286
- .prepare(`SELECT file_path, body_hash, priority
287
- FROM graph_extraction_queue
288
- WHERE stash_root = ?
289
- ORDER BY priority DESC, queued_at ASC
290
- LIMIT ?`)
291
- .all(stashRoot, limit);
292
- return rows.map((row) => ({ filePath: row.file_path, bodyHash: row.body_hash, priority: row.priority }));
293
- }
294
- catch (err) {
295
- rethrowIfTestIsolationError(err);
296
- return [];
297
- }
298
- }
299
- /**
300
- * Acknowledge the exact queued revision that was classified and handled.
301
- * A concurrent re-enqueue with a new body hash therefore survives.
302
- */
303
- export function acknowledgeExtractionQueueEntry(db, stashRoot, filePath, bodyHash) {
304
- try {
305
- const result = db
306
- .prepare("DELETE FROM graph_extraction_queue WHERE stash_root = ? AND file_path = ? AND body_hash = ?")
307
- .run(stashRoot, filePath, bodyHash);
308
- return result.changes > 0;
309
- }
310
- catch (err) {
311
- rethrowIfTestIsolationError(err);
312
- return false;
313
- }
314
- }
315
245
  /**
316
246
  * Scoped loader — graph_files rows without entities/relations. Used for
317
247
  * orphan detection and entity overview commands.
@@ -350,7 +280,6 @@ export function loadStoredGraphMeta(stashPath, db) {
350
280
  const row = readDb
351
281
  .prepare(`SELECT
352
282
  stash_root,
353
- schema_version,
354
283
  generated_at,
355
284
  considered_files,
356
285
  extracted_files,
@@ -375,7 +304,6 @@ export function loadStoredGraphMeta(stashPath, db) {
375
304
  return {
376
305
  stashPath: row.stash_root,
377
306
  graphPath: getDbPath(),
378
- schemaVersion: row.schema_version,
379
307
  generatedAt: row.generated_at,
380
308
  quality: {
381
309
  consideredFiles: row.considered_files,
@@ -484,7 +412,6 @@ export function loadStoredGraphSnapshot(stashPath, db) {
484
412
  return {
485
413
  stashPath: meta.stashPath,
486
414
  graphPath: meta.graphPath,
487
- schemaVersion: meta.schemaVersion,
488
415
  generatedAt: meta.generatedAt,
489
416
  ...(meta.quality ? { quality: meta.quality } : {}),
490
417
  ...(meta.telemetry ? { telemetry: meta.telemetry } : {}),