akm-cli 0.9.17-alpha.7 → 0.9.17-alpha.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -242,38 +242,6 @@ export function deleteStoredGraph(db, stashPath) {
242
242
  db.prepare("DELETE FROM graph_meta WHERE stash_root = ?").run(stashPath);
243
243
  })();
244
244
  }
245
- /**
246
- * Scoped loader — graph_files rows without entities/relations. Used for
247
- * orphan detection and entity overview commands.
248
- */
249
- export function loadGraphFilesOnly(stashPath, db) {
250
- try {
251
- return withReadableGraphDb(db, (readDb) => {
252
- const rows = readDb
253
- .prepare(`SELECT file_path, file_type, body_hash, confidence, status, reason
254
- FROM graph_files
255
- WHERE stash_root = ?
256
- ORDER BY file_order`)
257
- .all(stashPath);
258
- return rows.map((row) => ({
259
- path: row.file_path,
260
- type: row.file_type,
261
- bodyHash: row.body_hash,
262
- ...(typeof row.confidence === "number" ? { confidence: row.confidence } : {}),
263
- ...(row.status ? { status: row.status } : {}),
264
- ...(row.reason ? { reason: row.reason } : {}),
265
- }));
266
- });
267
- }
268
- catch (err) {
269
- // Never mask the bun-test isolation guard as "no stored graph files",
270
- // and never mask an index we are not allowed to read as one with no
271
- // graph in it (#791) — `GRAPH_DB_MISSING` above is the only "absent".
272
- rethrowIfTestIsolationError(err);
273
- rethrowIfDataDirUnreadable(err);
274
- return [];
275
- }
276
- }
277
245
  export function loadStoredGraphMeta(stashPath, db) {
278
246
  try {
279
247
  return withReadableGraphDb(db, (readDb) => {
@@ -7,7 +7,7 @@
7
7
  * Walks the primary stash for `memory:` and `knowledge:` assets, asks the
8
8
  * configured LLM to extract entities and relations from each one, and
9
9
  * persists the result to stash-local SQLite graph tables keyed by stash root.
10
- * The artifact backs `akm show`'s `related` list and curate's support refs
10
+ * The artifact backs `akm show`'s `related` list
11
11
  * (`src/indexer/graph/graph-related.ts`); it plays no part in search ranking.
12
12
  *
13
13
  * Disabling — three preconditions must ALL hold for the pass to run:
@@ -344,6 +344,8 @@ function extractionRecord(candidate, bodyHash, shape) {
344
344
  * Run the planned extractions in chunks of `batchSize`: cache hits are taken
345
345
  * as-is, and each chunk's model plans go to the provider in one
346
346
  * `extractGraphFromBodies` call (a one-body call is the per-asset path).
347
+ * Up to the runner's concurrency of chunks run at once, and each makes one
348
+ * call at a time, so that is also the most calls the run has in flight.
347
349
  */
348
350
  async function extractGraphBatches(args) {
349
351
  const { plans, batchSize, signal, db, cacheVariant, telemetry, llmRunner, featureConfig, onFallback, abortState, batchState, runtimeTelemetry, onNotices, reportProgress, maxChunksPerAsset, } = args;
@@ -0,0 +1,90 @@
1
+ // This Source Code Form is subject to the terms of the Mozilla Public
2
+ // License, v. 2.0. If a copy of the MPL was not distributed with this
3
+ // file, You can obtain one at https://mozilla.org/MPL/2.0/.
4
+ import { stashDirFor } from "../../core/asset/asset-placement.js";
5
+ import { isBundleSlug, parseBundleRef } from "../../core/asset/asset-ref.js";
6
+ /** Channel order is the order links are stored and shown in. */
7
+ const CHANNELS = [
8
+ ["xref", (doc) => doc.xrefs],
9
+ ["superseded_by", (doc) => doc.supersededBy],
10
+ ["contradicted_by", (doc) => doc.contradictedBy],
11
+ ["belief_peer", (doc) => doc.currentBeliefRefs],
12
+ ["derived_from", (doc) => doc.derivedFrom],
13
+ ["cites", (doc) => doc.sources],
14
+ ["links_to", (doc) => doc.links],
15
+ ["uses", (doc) => doc.uses],
16
+ ];
17
+ /** The links an indexed document declares, in channel then authored order; one per (kind, target). */
18
+ export function declaredLinks(doc, owner) {
19
+ const links = [];
20
+ const seen = new Set();
21
+ for (const [kind, read] of CHANNELS) {
22
+ for (const raw of stringValues(read(doc))) {
23
+ const target = linkTarget(raw, owner.conceptId);
24
+ if (!target)
25
+ continue;
26
+ const bundle = target.bundle ?? owner.bundleId;
27
+ if (bundle === owner.bundleId && target.conceptId === owner.conceptId)
28
+ continue;
29
+ const key = `${kind}\0${bundle}\0${target.conceptId}`;
30
+ if (seen.has(key))
31
+ continue;
32
+ seen.add(key);
33
+ links.push({ kind, raw, ...target });
34
+ }
35
+ }
36
+ return links;
37
+ }
38
+ function stringValues(value) {
39
+ if (typeof value === "string")
40
+ return [value];
41
+ return Array.isArray(value) ? value.filter((item) => typeof item === "string") : [];
42
+ }
43
+ /**
44
+ * The target a token names, or `undefined` when the token is not an asset ref
45
+ * (a session id, a URL, prose, a template placeholder). Retired spellings
46
+ * convert in memory: `<type>:<name>` (`memory:x`), `wiki:<wiki>/<path>` (0.8
47
+ * wikis, now knowledge under `wikis/`), a `.md` suffix, and a `#fragment`.
48
+ * Inside a wiki page, `raw/…` and `pages/…` are relative to the wiki root.
49
+ */
50
+ function linkTarget(raw, ownerConceptId) {
51
+ const token = raw.trim();
52
+ if (!token || /[\s<*]/.test(token) || token.includes("$(") || token.includes("${"))
53
+ return;
54
+ let bundle;
55
+ let body = token;
56
+ const boundary = token.indexOf("//");
57
+ if (boundary >= 0) {
58
+ bundle = token.slice(0, boundary);
59
+ body = token.slice(boundary + 2);
60
+ if (!isBundleSlug(bundle))
61
+ return;
62
+ }
63
+ const conceptId = legacyConceptId(body.split("#", 1)[0] ?? "", ownerConceptId);
64
+ if (conceptId === undefined || conceptId.length <= 1 || conceptId.includes("//"))
65
+ return;
66
+ try {
67
+ return { ...(bundle ? { bundle } : {}), conceptId: parseBundleRef(conceptId).conceptId };
68
+ }
69
+ catch {
70
+ return;
71
+ }
72
+ }
73
+ function legacyConceptId(body, ownerConceptId) {
74
+ const stripped = body.replace(/\.md$/i, "");
75
+ const colon = stripped.indexOf(":");
76
+ if (colon >= 0) {
77
+ const type = stripped.slice(0, colon);
78
+ const name = stripped.slice(colon + 1);
79
+ if (!name || name.includes(":"))
80
+ return;
81
+ if (type === "wiki")
82
+ return `knowledge/wikis/${name}`;
83
+ const stashDir = stashDirFor(type);
84
+ return stashDir === undefined ? undefined : `${stashDir}/${name}`;
85
+ }
86
+ const wikiRoot = /^((?:[^/]+\/)*?wikis\/[^/]+\/)/.exec(ownerConceptId)?.[1];
87
+ if (wikiRoot && /^(raw|pages)\//.test(stripped))
88
+ return `${wikiRoot}${stripped}`;
89
+ return stripped;
90
+ }
@@ -138,6 +138,7 @@ export function indexDocumentToStashEntry(doc) {
138
138
  entry.wikiRole = dj.wikiRole;
139
139
  assignStringList(entry, "sources", dj.sources);
140
140
  assignStringList(entry, "evidenceSources", dj.evidenceSources);
141
+ assignStringList(entry, "uses", dj.uses);
141
142
  // D2 (#730): OKF v0.2 provenance promoteProposal stamps onto AKM-native
142
143
  // writes, carried via DOCUMENT_JSON_CARRIED_FIELDS (akm-adapter.ts) —
143
144
  // unpacked back to a first-class member here exactly like `sources`/
@@ -185,30 +185,6 @@ function bumpTelemetry(telemetry, key, amount = 1) {
185
185
  return;
186
186
  telemetry[key] = (telemetry[key] ?? 0) + amount;
187
187
  }
188
- /**
189
- * `Promise.all(items.map(fn))` with at most `limit` calls in flight, so the
190
- * per-asset calls of one batch never outnumber the chunk pool's concurrency
191
- * (GR-D14). The first rejection rejects the map and stops further calls.
192
- */
193
- async function mapWithConcurrency(items, limit, fn) {
194
- const results = new Array(items.length);
195
- let next = 0;
196
- let failed = false;
197
- const worker = async () => {
198
- while (!failed && next < items.length) {
199
- const index = next++;
200
- try {
201
- results[index] = await fn(items[index]);
202
- }
203
- catch (error) {
204
- failed = true;
205
- throw error;
206
- }
207
- }
208
- };
209
- await Promise.all(Array.from({ length: Math.min(Math.max(1, limit), items.length) }, worker));
210
- return results;
211
- }
212
188
  /** Count what the parser dropped from one parsed extraction (single-asset or batch item). */
213
189
  function bumpFilterTelemetry(telemetry, extraction) {
214
190
  bumpTelemetry(telemetry, "filteredGenericEntities", extraction.filteredGenericEntities ?? 0);
@@ -260,6 +236,7 @@ function mergeGraphExtractions(extractions) {
260
236
  let filteredGenericEntities = 0;
261
237
  let filteredInvalidRelations = 0;
262
238
  let filteredLowConfidenceRelations = 0;
239
+ let anyChunkFailed = false;
263
240
  let firstFailureReason;
264
241
  for (const extraction of extractions) {
265
242
  truncationCount += extraction.truncationCount ?? 0;
@@ -267,8 +244,10 @@ function mergeGraphExtractions(extractions) {
267
244
  filteredGenericEntities += extraction.filteredGenericEntities ?? 0;
268
245
  filteredInvalidRelations += extraction.filteredInvalidRelations ?? 0;
269
246
  filteredLowConfidenceRelations += extraction.filteredLowConfidenceRelations ?? 0;
270
- if (extraction.status === "failed" && !firstFailureReason)
271
- firstFailureReason = extraction.reason;
247
+ if (extraction.status === "failed") {
248
+ anyChunkFailed = true;
249
+ firstFailureReason ??= extraction.reason;
250
+ }
272
251
  const nextConfidence = parseConfidence(extraction.confidence);
273
252
  if (nextConfidence !== undefined)
274
253
  confidence = confidence === undefined ? nextConfidence : Math.max(confidence, nextConfidence);
@@ -329,8 +308,12 @@ function mergeGraphExtractions(extractions) {
329
308
  const chunkCount = relationChunkCounts.get(key) ?? 1;
330
309
  relation.confidence = blendConsistency(relation.confidence, chunkCount);
331
310
  }
332
- const status = entities.length > 0 ? "extracted" : firstFailureReason ? "failed" : "empty";
333
- const reason = status === "extracted" ? "none" : (firstFailureReason ?? "no_graph_content");
311
+ // One failed chunk fails the body, whatever the others found: a failed
312
+ // extraction is never cached, so the next run extracts the body again
313
+ // (every chunk: partial failures are rare outside provider outages, which
314
+ // the run's failure-rate abort stops). The entities found are kept.
315
+ const status = anyChunkFailed ? "failed" : entities.length > 0 ? "extracted" : "empty";
316
+ const reason = status === "extracted" ? "none" : status === "failed" ? (firstFailureReason ?? "llm_error") : "no_graph_content";
334
317
  const mergedConfidence = confidence !== undefined ? blendConsistency(confidence, totalChunks) : totalChunks > 1 ? 1 : undefined;
335
318
  return {
336
319
  entities,
@@ -606,11 +589,21 @@ export async function extractGraphFromBodies(llmRunner, bodies, signal, akmConfi
606
589
  const result = await extractGraphFromBody(llmRunner, bodies[0] ?? "", signal, akmConfig, onFallback, options);
607
590
  return [result];
608
591
  }
609
- const concurrency = llmRunner.connection.concurrency ?? 1;
592
+ // A batch makes its per-asset calls one at a time: the caller's pool runs
593
+ // batches side by side and is the run's only concurrency, so the calls in
594
+ // flight never exceed it (not the pool's width squared).
610
595
  const extractOne = (body) => extractGraphFromBody(llmRunner, body, signal, akmConfig, onFallback, options);
596
+ const extractEach = async (indices, into) => {
597
+ for (const index of indices)
598
+ into[index] = await extractOne(bodies[index] ?? "");
599
+ };
611
600
  // With batching disabled every body takes the single-asset path, once.
612
- if (batchState?.batchingDisabled)
613
- return mapWithConcurrency(bodies, concurrency, extractOne);
601
+ if (batchState?.batchingDisabled) {
602
+ const results = [];
603
+ for (const body of bodies)
604
+ results.push(await extractOne(body));
605
+ return results;
606
+ }
614
607
  // Filter out bodies that are empty so we don't waste tokens, but keep
615
608
  // index correspondence by tracking which indices were non-empty.
616
609
  const results = bodies.map(empty);
@@ -629,9 +622,7 @@ export async function extractGraphFromBodies(llmRunner, bodies, signal, akmConfi
629
622
  }
630
623
  }
631
624
  }
632
- await mapWithConcurrency(oversizedIndices, concurrency, async (index) => {
633
- results[index] = await extractOne(bodies[index] ?? "");
634
- });
625
+ await extractEach(oversizedIndices, results);
635
626
  if (nonEmptyBodies.length === 0)
636
627
  return results;
637
628
  const systemPrompt = buildBatchSystemPrompt();
@@ -780,9 +771,7 @@ export async function extractGraphFromBodies(llmRunner, bodies, signal, akmConfi
780
771
  warn(`graph extraction (batch): response had ${batchResult.length} items for ${nonEmptyBodies.length} assets; ` +
781
772
  `falling back to individual calls for ${fallbackIndices.length} missing asset(s).`);
782
773
  }
783
- await mapWithConcurrency(fallbackIndices, concurrency, async (origIdx) => {
784
- results[origIdx] = await extractOne(bodies[origIdx] ?? "");
785
- });
774
+ await extractEach(fallbackIndices, results);
786
775
  }
787
776
  else if (batchContextError) {
788
777
  warn(`graph extraction (batch): skipped ${nonEmptyBodies.length} asset(s) due to context size error; ` +
@@ -358,6 +358,7 @@ export function shapeShowOutput(result, detail, shape = "human") {
358
358
  "steps",
359
359
  "keys",
360
360
  "related",
361
+ "links",
361
362
  ...FRAGMENT_PROVENANCE_FIELDS,
362
363
  ...FRAGMENT_CONTEXT_FIELDS,
363
364
  ]);
@@ -382,6 +383,7 @@ export function shapeShowOutput(result, detail, shape = "human") {
382
383
  "origin",
383
384
  "keys",
384
385
  "related",
386
+ "links",
385
387
  ...FRAGMENT_PROVENANCE_FIELDS,
386
388
  ...FRAGMENT_CONTEXT_FIELDS,
387
389
  ]);
@@ -411,6 +413,7 @@ export function shapeShowOutput(result, detail, shape = "human") {
411
413
  "activeRun",
412
414
  "keys",
413
415
  "related",
416
+ "links",
414
417
  ...FRAGMENT_PROVENANCE_FIELDS,
415
418
  ...FRAGMENT_CONTEXT_FIELDS,
416
419
  // ref, path, and editable are always projected — at every --detail level,
@@ -67,6 +67,22 @@ export function formatShowPlain(r, detail) {
67
67
  lines.push(` relationCount: ${String(hit.relationCount ?? 0)}`);
68
68
  }
69
69
  }
70
+ const links = typeof r.links === "object" && r.links !== null ? r.links : undefined;
71
+ if (links) {
72
+ lines.push("");
73
+ lines.push("links:");
74
+ for (const part of ["outgoing", "incoming", "unresolved"]) {
75
+ const groups = links[part];
76
+ if (typeof groups !== "object" || groups === null)
77
+ continue;
78
+ lines.push(` ${part}:`);
79
+ for (const [kind, group] of Object.entries(groups)) {
80
+ const refs = Array.isArray(group.refs) ? group.refs.map(String) : [];
81
+ const more = (group.total ?? refs.length) > refs.length ? ` (+${(group.total ?? 0) - refs.length} more)` : "";
82
+ lines.push(` ${kind}: ${refs.join(", ")}${more}`);
83
+ }
84
+ }
85
+ }
70
86
  const payloads = [r.content, r.template, r.prompt].filter((value) => value != null).map(String);
71
87
  if (Array.isArray(r.steps) && r.steps.length > 0) {
72
88
  if (lines.length > 0)