akm-cli 0.9.17-alpha.7 → 0.9.17-alpha.8
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +139 -0
- package/dist/akm +55 -22
- package/dist/akm-migrate +38 -19
- package/dist/assets/stash-skeleton/facts/conventions/backlinks.md +5 -3
- package/dist/commands/improve/loop-stages.js +4 -3
- package/dist/commands/read/curate.js +40 -13
- package/dist/commands/read/show.js +55 -2
- package/dist/commands/sources/info.js +3 -0
- package/dist/core/adapter/adapters/akm-adapter.js +2 -0
- package/dist/core/adapter/adapters/akm-metadata.js +31 -0
- package/dist/indexer/db/graph-db.js +0 -32
- package/dist/indexer/graph/graph-extraction.js +3 -1
- package/dist/indexer/links/declared-links.js +90 -0
- package/dist/indexer/scan/doc-to-entry.js +1 -0
- package/dist/llm/graph-extract.js +26 -37
- package/dist/output/shapes/helpers.js +3 -0
- package/dist/output/text/show-format.js +16 -0
- package/dist/scripts/akm-migrate-node.js +351 -400
- package/dist/scripts/akm-migrate.js +351 -400
- package/dist/storage/repositories/index-entries-repository.js +12 -6
- package/dist/storage/repositories/index-entry-schema.js +18 -1
- package/dist/storage/repositories/index-links-repository.js +143 -0
- package/dist/storage/repositories/index-schema.js +27 -0
- package/dist/tasks/source/task-to-v4.js +462 -74
- package/docs/migration/release-notes/0.9.17.md +7 -5
- package/docs/reference/cli.md +22 -7
- package/package.json +1 -1
- package/dist/tasks/source/task-to-v3.js +0 -453
|
@@ -242,38 +242,6 @@ export function deleteStoredGraph(db, stashPath) {
|
|
|
242
242
|
db.prepare("DELETE FROM graph_meta WHERE stash_root = ?").run(stashPath);
|
|
243
243
|
})();
|
|
244
244
|
}
|
|
245
|
-
/**
|
|
246
|
-
* Scoped loader — graph_files rows without entities/relations. Used for
|
|
247
|
-
* orphan detection and entity overview commands.
|
|
248
|
-
*/
|
|
249
|
-
export function loadGraphFilesOnly(stashPath, db) {
|
|
250
|
-
try {
|
|
251
|
-
return withReadableGraphDb(db, (readDb) => {
|
|
252
|
-
const rows = readDb
|
|
253
|
-
.prepare(`SELECT file_path, file_type, body_hash, confidence, status, reason
|
|
254
|
-
FROM graph_files
|
|
255
|
-
WHERE stash_root = ?
|
|
256
|
-
ORDER BY file_order`)
|
|
257
|
-
.all(stashPath);
|
|
258
|
-
return rows.map((row) => ({
|
|
259
|
-
path: row.file_path,
|
|
260
|
-
type: row.file_type,
|
|
261
|
-
bodyHash: row.body_hash,
|
|
262
|
-
...(typeof row.confidence === "number" ? { confidence: row.confidence } : {}),
|
|
263
|
-
...(row.status ? { status: row.status } : {}),
|
|
264
|
-
...(row.reason ? { reason: row.reason } : {}),
|
|
265
|
-
}));
|
|
266
|
-
});
|
|
267
|
-
}
|
|
268
|
-
catch (err) {
|
|
269
|
-
// Never mask the bun-test isolation guard as "no stored graph files",
|
|
270
|
-
// and never mask an index we are not allowed to read as one with no
|
|
271
|
-
// graph in it (#791) — `GRAPH_DB_MISSING` above is the only "absent".
|
|
272
|
-
rethrowIfTestIsolationError(err);
|
|
273
|
-
rethrowIfDataDirUnreadable(err);
|
|
274
|
-
return [];
|
|
275
|
-
}
|
|
276
|
-
}
|
|
277
245
|
export function loadStoredGraphMeta(stashPath, db) {
|
|
278
246
|
try {
|
|
279
247
|
return withReadableGraphDb(db, (readDb) => {
|
|
@@ -7,7 +7,7 @@
|
|
|
7
7
|
* Walks the primary stash for `memory:` and `knowledge:` assets, asks the
|
|
8
8
|
* configured LLM to extract entities and relations from each one, and
|
|
9
9
|
* persists the result to stash-local SQLite graph tables keyed by stash root.
|
|
10
|
-
* The artifact backs `akm show`'s `related` list
|
|
10
|
+
* The artifact backs `akm show`'s `related` list
|
|
11
11
|
* (`src/indexer/graph/graph-related.ts`); it plays no part in search ranking.
|
|
12
12
|
*
|
|
13
13
|
* Disabling — three preconditions must ALL hold for the pass to run:
|
|
@@ -344,6 +344,8 @@ function extractionRecord(candidate, bodyHash, shape) {
|
|
|
344
344
|
* Run the planned extractions in chunks of `batchSize`: cache hits are taken
|
|
345
345
|
* as-is, and each chunk's model plans go to the provider in one
|
|
346
346
|
* `extractGraphFromBodies` call (a one-body call is the per-asset path).
|
|
347
|
+
* Up to the runner's concurrency of chunks run at once, and each makes one
|
|
348
|
+
* call at a time, so that is also the most calls the run has in flight.
|
|
347
349
|
*/
|
|
348
350
|
async function extractGraphBatches(args) {
|
|
349
351
|
const { plans, batchSize, signal, db, cacheVariant, telemetry, llmRunner, featureConfig, onFallback, abortState, batchState, runtimeTelemetry, onNotices, reportProgress, maxChunksPerAsset, } = args;
|
|
@@ -0,0 +1,90 @@
|
|
|
1
|
+
// This Source Code Form is subject to the terms of the Mozilla Public
|
|
2
|
+
// License, v. 2.0. If a copy of the MPL was not distributed with this
|
|
3
|
+
// file, You can obtain one at https://mozilla.org/MPL/2.0/.
|
|
4
|
+
import { stashDirFor } from "../../core/asset/asset-placement.js";
|
|
5
|
+
import { isBundleSlug, parseBundleRef } from "../../core/asset/asset-ref.js";
|
|
6
|
+
/** Channel order is the order links are stored and shown in. */
|
|
7
|
+
const CHANNELS = [
|
|
8
|
+
["xref", (doc) => doc.xrefs],
|
|
9
|
+
["superseded_by", (doc) => doc.supersededBy],
|
|
10
|
+
["contradicted_by", (doc) => doc.contradictedBy],
|
|
11
|
+
["belief_peer", (doc) => doc.currentBeliefRefs],
|
|
12
|
+
["derived_from", (doc) => doc.derivedFrom],
|
|
13
|
+
["cites", (doc) => doc.sources],
|
|
14
|
+
["links_to", (doc) => doc.links],
|
|
15
|
+
["uses", (doc) => doc.uses],
|
|
16
|
+
];
|
|
17
|
+
/** The links an indexed document declares, in channel then authored order; one per (kind, target). */
|
|
18
|
+
export function declaredLinks(doc, owner) {
|
|
19
|
+
const links = [];
|
|
20
|
+
const seen = new Set();
|
|
21
|
+
for (const [kind, read] of CHANNELS) {
|
|
22
|
+
for (const raw of stringValues(read(doc))) {
|
|
23
|
+
const target = linkTarget(raw, owner.conceptId);
|
|
24
|
+
if (!target)
|
|
25
|
+
continue;
|
|
26
|
+
const bundle = target.bundle ?? owner.bundleId;
|
|
27
|
+
if (bundle === owner.bundleId && target.conceptId === owner.conceptId)
|
|
28
|
+
continue;
|
|
29
|
+
const key = `${kind}\0${bundle}\0${target.conceptId}`;
|
|
30
|
+
if (seen.has(key))
|
|
31
|
+
continue;
|
|
32
|
+
seen.add(key);
|
|
33
|
+
links.push({ kind, raw, ...target });
|
|
34
|
+
}
|
|
35
|
+
}
|
|
36
|
+
return links;
|
|
37
|
+
}
|
|
38
|
+
function stringValues(value) {
|
|
39
|
+
if (typeof value === "string")
|
|
40
|
+
return [value];
|
|
41
|
+
return Array.isArray(value) ? value.filter((item) => typeof item === "string") : [];
|
|
42
|
+
}
|
|
43
|
+
/**
|
|
44
|
+
* The target a token names, or `undefined` when the token is not an asset ref
|
|
45
|
+
* (a session id, a URL, prose, a template placeholder). Retired spellings
|
|
46
|
+
* convert in memory: `<type>:<name>` (`memory:x`), `wiki:<wiki>/<path>` (0.8
|
|
47
|
+
* wikis, now knowledge under `wikis/`), a `.md` suffix, and a `#fragment`.
|
|
48
|
+
* Inside a wiki page, `raw/…` and `pages/…` are relative to the wiki root.
|
|
49
|
+
*/
|
|
50
|
+
function linkTarget(raw, ownerConceptId) {
|
|
51
|
+
const token = raw.trim();
|
|
52
|
+
if (!token || /[\s<*]/.test(token) || token.includes("$(") || token.includes("${"))
|
|
53
|
+
return;
|
|
54
|
+
let bundle;
|
|
55
|
+
let body = token;
|
|
56
|
+
const boundary = token.indexOf("//");
|
|
57
|
+
if (boundary >= 0) {
|
|
58
|
+
bundle = token.slice(0, boundary);
|
|
59
|
+
body = token.slice(boundary + 2);
|
|
60
|
+
if (!isBundleSlug(bundle))
|
|
61
|
+
return;
|
|
62
|
+
}
|
|
63
|
+
const conceptId = legacyConceptId(body.split("#", 1)[0] ?? "", ownerConceptId);
|
|
64
|
+
if (conceptId === undefined || conceptId.length <= 1 || conceptId.includes("//"))
|
|
65
|
+
return;
|
|
66
|
+
try {
|
|
67
|
+
return { ...(bundle ? { bundle } : {}), conceptId: parseBundleRef(conceptId).conceptId };
|
|
68
|
+
}
|
|
69
|
+
catch {
|
|
70
|
+
return;
|
|
71
|
+
}
|
|
72
|
+
}
|
|
73
|
+
function legacyConceptId(body, ownerConceptId) {
|
|
74
|
+
const stripped = body.replace(/\.md$/i, "");
|
|
75
|
+
const colon = stripped.indexOf(":");
|
|
76
|
+
if (colon >= 0) {
|
|
77
|
+
const type = stripped.slice(0, colon);
|
|
78
|
+
const name = stripped.slice(colon + 1);
|
|
79
|
+
if (!name || name.includes(":"))
|
|
80
|
+
return;
|
|
81
|
+
if (type === "wiki")
|
|
82
|
+
return `knowledge/wikis/${name}`;
|
|
83
|
+
const stashDir = stashDirFor(type);
|
|
84
|
+
return stashDir === undefined ? undefined : `${stashDir}/${name}`;
|
|
85
|
+
}
|
|
86
|
+
const wikiRoot = /^((?:[^/]+\/)*?wikis\/[^/]+\/)/.exec(ownerConceptId)?.[1];
|
|
87
|
+
if (wikiRoot && /^(raw|pages)\//.test(stripped))
|
|
88
|
+
return `${wikiRoot}${stripped}`;
|
|
89
|
+
return stripped;
|
|
90
|
+
}
|
|
@@ -138,6 +138,7 @@ export function indexDocumentToStashEntry(doc) {
|
|
|
138
138
|
entry.wikiRole = dj.wikiRole;
|
|
139
139
|
assignStringList(entry, "sources", dj.sources);
|
|
140
140
|
assignStringList(entry, "evidenceSources", dj.evidenceSources);
|
|
141
|
+
assignStringList(entry, "uses", dj.uses);
|
|
141
142
|
// D2 (#730): OKF v0.2 provenance promoteProposal stamps onto AKM-native
|
|
142
143
|
// writes, carried via DOCUMENT_JSON_CARRIED_FIELDS (akm-adapter.ts) —
|
|
143
144
|
// unpacked back to a first-class member here exactly like `sources`/
|
|
@@ -185,30 +185,6 @@ function bumpTelemetry(telemetry, key, amount = 1) {
|
|
|
185
185
|
return;
|
|
186
186
|
telemetry[key] = (telemetry[key] ?? 0) + amount;
|
|
187
187
|
}
|
|
188
|
-
/**
|
|
189
|
-
* `Promise.all(items.map(fn))` with at most `limit` calls in flight, so the
|
|
190
|
-
* per-asset calls of one batch never outnumber the chunk pool's concurrency
|
|
191
|
-
* (GR-D14). The first rejection rejects the map and stops further calls.
|
|
192
|
-
*/
|
|
193
|
-
async function mapWithConcurrency(items, limit, fn) {
|
|
194
|
-
const results = new Array(items.length);
|
|
195
|
-
let next = 0;
|
|
196
|
-
let failed = false;
|
|
197
|
-
const worker = async () => {
|
|
198
|
-
while (!failed && next < items.length) {
|
|
199
|
-
const index = next++;
|
|
200
|
-
try {
|
|
201
|
-
results[index] = await fn(items[index]);
|
|
202
|
-
}
|
|
203
|
-
catch (error) {
|
|
204
|
-
failed = true;
|
|
205
|
-
throw error;
|
|
206
|
-
}
|
|
207
|
-
}
|
|
208
|
-
};
|
|
209
|
-
await Promise.all(Array.from({ length: Math.min(Math.max(1, limit), items.length) }, worker));
|
|
210
|
-
return results;
|
|
211
|
-
}
|
|
212
188
|
/** Count what the parser dropped from one parsed extraction (single-asset or batch item). */
|
|
213
189
|
function bumpFilterTelemetry(telemetry, extraction) {
|
|
214
190
|
bumpTelemetry(telemetry, "filteredGenericEntities", extraction.filteredGenericEntities ?? 0);
|
|
@@ -260,6 +236,7 @@ function mergeGraphExtractions(extractions) {
|
|
|
260
236
|
let filteredGenericEntities = 0;
|
|
261
237
|
let filteredInvalidRelations = 0;
|
|
262
238
|
let filteredLowConfidenceRelations = 0;
|
|
239
|
+
let anyChunkFailed = false;
|
|
263
240
|
let firstFailureReason;
|
|
264
241
|
for (const extraction of extractions) {
|
|
265
242
|
truncationCount += extraction.truncationCount ?? 0;
|
|
@@ -267,8 +244,10 @@ function mergeGraphExtractions(extractions) {
|
|
|
267
244
|
filteredGenericEntities += extraction.filteredGenericEntities ?? 0;
|
|
268
245
|
filteredInvalidRelations += extraction.filteredInvalidRelations ?? 0;
|
|
269
246
|
filteredLowConfidenceRelations += extraction.filteredLowConfidenceRelations ?? 0;
|
|
270
|
-
if (extraction.status === "failed"
|
|
271
|
-
|
|
247
|
+
if (extraction.status === "failed") {
|
|
248
|
+
anyChunkFailed = true;
|
|
249
|
+
firstFailureReason ??= extraction.reason;
|
|
250
|
+
}
|
|
272
251
|
const nextConfidence = parseConfidence(extraction.confidence);
|
|
273
252
|
if (nextConfidence !== undefined)
|
|
274
253
|
confidence = confidence === undefined ? nextConfidence : Math.max(confidence, nextConfidence);
|
|
@@ -329,8 +308,12 @@ function mergeGraphExtractions(extractions) {
|
|
|
329
308
|
const chunkCount = relationChunkCounts.get(key) ?? 1;
|
|
330
309
|
relation.confidence = blendConsistency(relation.confidence, chunkCount);
|
|
331
310
|
}
|
|
332
|
-
|
|
333
|
-
|
|
311
|
+
// One failed chunk fails the body, whatever the others found: a failed
|
|
312
|
+
// extraction is never cached, so the next run extracts the body again
|
|
313
|
+
// (every chunk: partial failures are rare outside provider outages, which
|
|
314
|
+
// the run's failure-rate abort stops). The entities found are kept.
|
|
315
|
+
const status = anyChunkFailed ? "failed" : entities.length > 0 ? "extracted" : "empty";
|
|
316
|
+
const reason = status === "extracted" ? "none" : status === "failed" ? (firstFailureReason ?? "llm_error") : "no_graph_content";
|
|
334
317
|
const mergedConfidence = confidence !== undefined ? blendConsistency(confidence, totalChunks) : totalChunks > 1 ? 1 : undefined;
|
|
335
318
|
return {
|
|
336
319
|
entities,
|
|
@@ -606,11 +589,21 @@ export async function extractGraphFromBodies(llmRunner, bodies, signal, akmConfi
|
|
|
606
589
|
const result = await extractGraphFromBody(llmRunner, bodies[0] ?? "", signal, akmConfig, onFallback, options);
|
|
607
590
|
return [result];
|
|
608
591
|
}
|
|
609
|
-
|
|
592
|
+
// A batch makes its per-asset calls one at a time: the caller's pool runs
|
|
593
|
+
// batches side by side and is the run's only concurrency, so the calls in
|
|
594
|
+
// flight never exceed it (not the pool's width squared).
|
|
610
595
|
const extractOne = (body) => extractGraphFromBody(llmRunner, body, signal, akmConfig, onFallback, options);
|
|
596
|
+
const extractEach = async (indices, into) => {
|
|
597
|
+
for (const index of indices)
|
|
598
|
+
into[index] = await extractOne(bodies[index] ?? "");
|
|
599
|
+
};
|
|
611
600
|
// With batching disabled every body takes the single-asset path, once.
|
|
612
|
-
if (batchState?.batchingDisabled)
|
|
613
|
-
|
|
601
|
+
if (batchState?.batchingDisabled) {
|
|
602
|
+
const results = [];
|
|
603
|
+
for (const body of bodies)
|
|
604
|
+
results.push(await extractOne(body));
|
|
605
|
+
return results;
|
|
606
|
+
}
|
|
614
607
|
// Filter out bodies that are empty so we don't waste tokens, but keep
|
|
615
608
|
// index correspondence by tracking which indices were non-empty.
|
|
616
609
|
const results = bodies.map(empty);
|
|
@@ -629,9 +622,7 @@ export async function extractGraphFromBodies(llmRunner, bodies, signal, akmConfi
|
|
|
629
622
|
}
|
|
630
623
|
}
|
|
631
624
|
}
|
|
632
|
-
await
|
|
633
|
-
results[index] = await extractOne(bodies[index] ?? "");
|
|
634
|
-
});
|
|
625
|
+
await extractEach(oversizedIndices, results);
|
|
635
626
|
if (nonEmptyBodies.length === 0)
|
|
636
627
|
return results;
|
|
637
628
|
const systemPrompt = buildBatchSystemPrompt();
|
|
@@ -780,9 +771,7 @@ export async function extractGraphFromBodies(llmRunner, bodies, signal, akmConfi
|
|
|
780
771
|
warn(`graph extraction (batch): response had ${batchResult.length} items for ${nonEmptyBodies.length} assets; ` +
|
|
781
772
|
`falling back to individual calls for ${fallbackIndices.length} missing asset(s).`);
|
|
782
773
|
}
|
|
783
|
-
await
|
|
784
|
-
results[origIdx] = await extractOne(bodies[origIdx] ?? "");
|
|
785
|
-
});
|
|
774
|
+
await extractEach(fallbackIndices, results);
|
|
786
775
|
}
|
|
787
776
|
else if (batchContextError) {
|
|
788
777
|
warn(`graph extraction (batch): skipped ${nonEmptyBodies.length} asset(s) due to context size error; ` +
|
|
@@ -358,6 +358,7 @@ export function shapeShowOutput(result, detail, shape = "human") {
|
|
|
358
358
|
"steps",
|
|
359
359
|
"keys",
|
|
360
360
|
"related",
|
|
361
|
+
"links",
|
|
361
362
|
...FRAGMENT_PROVENANCE_FIELDS,
|
|
362
363
|
...FRAGMENT_CONTEXT_FIELDS,
|
|
363
364
|
]);
|
|
@@ -382,6 +383,7 @@ export function shapeShowOutput(result, detail, shape = "human") {
|
|
|
382
383
|
"origin",
|
|
383
384
|
"keys",
|
|
384
385
|
"related",
|
|
386
|
+
"links",
|
|
385
387
|
...FRAGMENT_PROVENANCE_FIELDS,
|
|
386
388
|
...FRAGMENT_CONTEXT_FIELDS,
|
|
387
389
|
]);
|
|
@@ -411,6 +413,7 @@ export function shapeShowOutput(result, detail, shape = "human") {
|
|
|
411
413
|
"activeRun",
|
|
412
414
|
"keys",
|
|
413
415
|
"related",
|
|
416
|
+
"links",
|
|
414
417
|
...FRAGMENT_PROVENANCE_FIELDS,
|
|
415
418
|
...FRAGMENT_CONTEXT_FIELDS,
|
|
416
419
|
// ref, path, and editable are always projected — at every --detail level,
|
|
@@ -67,6 +67,22 @@ export function formatShowPlain(r, detail) {
|
|
|
67
67
|
lines.push(` relationCount: ${String(hit.relationCount ?? 0)}`);
|
|
68
68
|
}
|
|
69
69
|
}
|
|
70
|
+
const links = typeof r.links === "object" && r.links !== null ? r.links : undefined;
|
|
71
|
+
if (links) {
|
|
72
|
+
lines.push("");
|
|
73
|
+
lines.push("links:");
|
|
74
|
+
for (const part of ["outgoing", "incoming", "unresolved"]) {
|
|
75
|
+
const groups = links[part];
|
|
76
|
+
if (typeof groups !== "object" || groups === null)
|
|
77
|
+
continue;
|
|
78
|
+
lines.push(` ${part}:`);
|
|
79
|
+
for (const [kind, group] of Object.entries(groups)) {
|
|
80
|
+
const refs = Array.isArray(group.refs) ? group.refs.map(String) : [];
|
|
81
|
+
const more = (group.total ?? refs.length) > refs.length ? ` (+${(group.total ?? 0) - refs.length} more)` : "";
|
|
82
|
+
lines.push(` ${kind}: ${refs.join(", ")}${more}`);
|
|
83
|
+
}
|
|
84
|
+
}
|
|
85
|
+
}
|
|
70
86
|
const payloads = [r.content, r.template, r.prompt].filter((value) => value != null).map(String);
|
|
71
87
|
if (Array.isArray(r.steps) && r.steps.length > 0) {
|
|
72
88
|
if (lines.length > 0)
|