akm-cli 0.9.17-alpha.6 → 0.9.17-alpha.8
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +285 -0
- package/STABILITY.md +2 -2
- package/dist/akm +62 -29
- package/dist/akm-migrate +38 -19
- package/dist/assets/prompts/retrieval-relevance-judge.md +6 -0
- package/dist/assets/stash-skeleton/facts/conventions/backlinks.md +5 -3
- package/dist/commands/improve/consolidate.js +11 -0
- package/dist/commands/improve/improve-cli.js +27 -7
- package/dist/commands/improve/ledger.js +7 -3
- package/dist/commands/improve/loop-stages.js +4 -3
- package/dist/commands/improve/preparation.js +40 -10
- package/dist/commands/improve/reflect.js +46 -22
- package/dist/commands/improve/retrieval-gate.js +127 -0
- package/dist/commands/improve/retrieval-scope.js +77 -0
- package/dist/commands/read/curate.js +41 -30
- package/dist/commands/read/show.js +55 -2
- package/dist/commands/sources/info.js +3 -0
- package/dist/commands/tasks/tasks-cli.js +10 -12
- package/dist/commands/tasks/tasks.js +57 -56
- package/dist/commands/tasks/validate.js +27 -46
- package/dist/core/adapter/adapters/akm-adapter.js +2 -0
- package/dist/core/adapter/adapters/akm-metadata.js +31 -0
- package/dist/core/adapter/adapters/akm-task-adapter.js +29 -8
- package/dist/core/improve-result.js +4 -1
- package/dist/core/non-task-input.js +20 -0
- package/dist/core/paths.js +0 -4
- package/dist/indexer/db/graph-db.js +0 -32
- package/dist/indexer/graph/graph-extraction.js +3 -1
- package/dist/indexer/indexer.js +1 -3
- package/dist/indexer/links/declared-links.js +90 -0
- package/dist/indexer/scan/doc-to-entry.js +1 -0
- package/dist/indexer/usage/usage-events.js +34 -0
- package/dist/llm/graph-extract.js +26 -37
- package/dist/output/shapes/helpers.js +3 -0
- package/dist/output/text/show-format.js +16 -0
- package/dist/scripts/akm-migrate-node.js +6377 -6437
- package/dist/scripts/akm-migrate.js +6860 -6920
- package/dist/storage/repositories/index-entries-repository.js +12 -6
- package/dist/storage/repositories/index-entry-schema.js +18 -1
- package/dist/storage/repositories/index-links-repository.js +143 -0
- package/dist/storage/repositories/index-schema.js +27 -0
- package/dist/storage/repositories/proposals-repository.js +4 -0
- package/dist/tasks/backends/cron.js +80 -43
- package/dist/tasks/backends/launchd.js +28 -15
- package/dist/tasks/backends/schtasks.js +25 -10
- package/dist/tasks/run/load-task.js +1 -1
- package/dist/tasks/scheduler-binding.js +4 -2
- package/dist/tasks/scheduler-invocation.js +127 -235
- package/dist/tasks/scheduler-sync.js +13 -8
- package/dist/tasks/source/parse-task-source.js +22 -126
- package/dist/tasks/source/task-to-v4.js +463 -87
- package/docs/migration/release-notes/0.9.17.md +7 -5
- package/docs/migration/v0.9.1-to-v0.9.2.md +7 -3
- package/docs/reference/cli.md +26 -10
- package/docs/reference/tasks.md +58 -40
- package/package.json +1 -1
- package/dist/tasks/source/task-to-v3.js +0 -507
|
@@ -306,6 +306,8 @@ function validateImprovePlan(value, dryRun, plannedRefNames) {
|
|
|
306
306
|
fail("plan.limits.totalCeiling must equal plan.limits.effective + plan.limits.additiveReplayAllowance");
|
|
307
307
|
}
|
|
308
308
|
const gateNames = new Set(["profile", "cleanup", "validation", "signal", "disk", "limit"]);
|
|
309
|
+
// Plans stored before 0.9.17-alpha.7 (#986) have no retrieval gate.
|
|
310
|
+
const optionalGateNames = new Set(["retrieval"]);
|
|
309
311
|
if (!Array.isArray(value.gates))
|
|
310
312
|
fail("plan.gates must be an array");
|
|
311
313
|
const gateRemovedByName = new Map();
|
|
@@ -313,8 +315,9 @@ function validateImprovePlan(value, dryRun, plannedRefNames) {
|
|
|
313
315
|
if (!isRecord(gate))
|
|
314
316
|
fail("plan.gates entries must be objects");
|
|
315
317
|
requireExactFields(gate, new Set(["name", "removed", "reason"]));
|
|
316
|
-
if (typeof gate.name !== "string" || !gateNames.has(gate.name))
|
|
318
|
+
if (typeof gate.name !== "string" || !(gateNames.has(gate.name) || optionalGateNames.has(gate.name))) {
|
|
317
319
|
fail("plan.gates.name is invalid");
|
|
320
|
+
}
|
|
318
321
|
requireCount(gate, "removed", "plan.gates entry");
|
|
319
322
|
if (typeof gate.reason !== "string")
|
|
320
323
|
fail("plan.gates.reason must be a string");
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
// This Source Code Form is subject to the terms of the Mozilla Public
|
|
2
|
+
// License, v. 2.0. If a copy of the MPL was not distributed with this
|
|
3
|
+
// file, You can obtain one at https://mozilla.org/MPL/2.0/.
|
|
4
|
+
/** The line of `src/assets/stash-skeleton/README.md` that reaches curate verbatim as a query. */
|
|
5
|
+
const STASH_README_LINE = "This is an **AKM stash** — a structured knowledge repository that stores reusable";
|
|
6
|
+
/**
|
|
7
|
+
* What the (trimmed) curate input is when it is not a task, else undefined.
|
|
8
|
+
* Harness and tool envelopes (`<task-notification>…`, `<system-reminder>…`,
|
|
9
|
+
* `<cross-session-message …>…`) start with a tag and close one, and the stash
|
|
10
|
+
* README line arrives verbatim; on the retrieval suite neither shape occurs in
|
|
11
|
+
* a real query. Length is not a signal: prompts over 2,000 characters found
|
|
12
|
+
* relevant assets at about the rate of shorter long prompts.
|
|
13
|
+
*/
|
|
14
|
+
export function nonTaskInput(query) {
|
|
15
|
+
if (query.startsWith("<") && query.includes("</"))
|
|
16
|
+
return "a harness or tool envelope";
|
|
17
|
+
if (query === STASH_README_LINE)
|
|
18
|
+
return "the akm stash README boilerplate";
|
|
19
|
+
return undefined;
|
|
20
|
+
}
|
package/dist/core/paths.js
CHANGED
|
@@ -249,10 +249,6 @@ export function getIndexRebuildLockPath() {
|
|
|
249
249
|
export function getStateDbPathInDataDir() {
|
|
250
250
|
return path.join(getDataDir(), "state.db");
|
|
251
251
|
}
|
|
252
|
-
/** Content-addressed scheduler runtime descriptors. */
|
|
253
|
-
export function getTaskContextDir(env = process.env) {
|
|
254
|
-
return path.join(getDataDir(env), "tasks", "context");
|
|
255
|
-
}
|
|
256
252
|
/** Path to the akm.lock file in $DATA. */
|
|
257
253
|
export function getLockfilePath() {
|
|
258
254
|
return path.join(getDataDir(), "akm.lock");
|
|
@@ -242,38 +242,6 @@ export function deleteStoredGraph(db, stashPath) {
|
|
|
242
242
|
db.prepare("DELETE FROM graph_meta WHERE stash_root = ?").run(stashPath);
|
|
243
243
|
})();
|
|
244
244
|
}
|
|
245
|
-
/**
|
|
246
|
-
* Scoped loader — graph_files rows without entities/relations. Used for
|
|
247
|
-
* orphan detection and entity overview commands.
|
|
248
|
-
*/
|
|
249
|
-
export function loadGraphFilesOnly(stashPath, db) {
|
|
250
|
-
try {
|
|
251
|
-
return withReadableGraphDb(db, (readDb) => {
|
|
252
|
-
const rows = readDb
|
|
253
|
-
.prepare(`SELECT file_path, file_type, body_hash, confidence, status, reason
|
|
254
|
-
FROM graph_files
|
|
255
|
-
WHERE stash_root = ?
|
|
256
|
-
ORDER BY file_order`)
|
|
257
|
-
.all(stashPath);
|
|
258
|
-
return rows.map((row) => ({
|
|
259
|
-
path: row.file_path,
|
|
260
|
-
type: row.file_type,
|
|
261
|
-
bodyHash: row.body_hash,
|
|
262
|
-
...(typeof row.confidence === "number" ? { confidence: row.confidence } : {}),
|
|
263
|
-
...(row.status ? { status: row.status } : {}),
|
|
264
|
-
...(row.reason ? { reason: row.reason } : {}),
|
|
265
|
-
}));
|
|
266
|
-
});
|
|
267
|
-
}
|
|
268
|
-
catch (err) {
|
|
269
|
-
// Never mask the bun-test isolation guard as "no stored graph files",
|
|
270
|
-
// and never mask an index we are not allowed to read as one with no
|
|
271
|
-
// graph in it (#791) — `GRAPH_DB_MISSING` above is the only "absent".
|
|
272
|
-
rethrowIfTestIsolationError(err);
|
|
273
|
-
rethrowIfDataDirUnreadable(err);
|
|
274
|
-
return [];
|
|
275
|
-
}
|
|
276
|
-
}
|
|
277
245
|
export function loadStoredGraphMeta(stashPath, db) {
|
|
278
246
|
try {
|
|
279
247
|
return withReadableGraphDb(db, (readDb) => {
|
|
@@ -7,7 +7,7 @@
|
|
|
7
7
|
* Walks the primary stash for `memory:` and `knowledge:` assets, asks the
|
|
8
8
|
* configured LLM to extract entities and relations from each one, and
|
|
9
9
|
* persists the result to stash-local SQLite graph tables keyed by stash root.
|
|
10
|
-
* The artifact backs `akm show`'s `related` list
|
|
10
|
+
* The artifact backs `akm show`'s `related` list
|
|
11
11
|
* (`src/indexer/graph/graph-related.ts`); it plays no part in search ranking.
|
|
12
12
|
*
|
|
13
13
|
* Disabling — three preconditions must ALL hold for the pass to run:
|
|
@@ -344,6 +344,8 @@ function extractionRecord(candidate, bodyHash, shape) {
|
|
|
344
344
|
* Run the planned extractions in chunks of `batchSize`: cache hits are taken
|
|
345
345
|
* as-is, and each chunk's model plans go to the provider in one
|
|
346
346
|
* `extractGraphFromBodies` call (a one-body call is the per-asset path).
|
|
347
|
+
* Up to the runner's concurrency of chunks run at once, and each makes one
|
|
348
|
+
* call at a time, so that is also the most calls the run has in flight.
|
|
347
349
|
*/
|
|
348
350
|
async function extractGraphBatches(args) {
|
|
349
351
|
const { plans, batchSize, signal, db, cacheVariant, telemetry, llmRunner, featureConfig, onFallback, abortState, batchState, runtimeTelemetry, onNotices, reportProgress, maxChunksPerAsset, } = args;
|
package/dist/indexer/indexer.js
CHANGED
|
@@ -35,7 +35,7 @@ import { generateEmbeddingsForDb } from "./materialize-embeddings.js";
|
|
|
35
35
|
import { canUseIncrementalSkip, computeDirFingerprint, getCachedDirState, getDirIndexState, inferZeroRowReason, } from "./passes/dir-staleness.js";
|
|
36
36
|
import { isEnrichmentComplete, isWorkflowSkipWarning, withFileSize, } from "./passes/metadata.js";
|
|
37
37
|
import { drainDirDocuments } from "./scan/drain-dir.js";
|
|
38
|
-
import { purgeOldUsageEvents } from "./usage/usage-events.js";
|
|
38
|
+
import { purgeOldUsageEvents, USAGE_EVENT_RETENTION_DAYS } from "./usage/usage-events.js";
|
|
39
39
|
import { walkStashFlatWithStatus } from "./walk/walker.js";
|
|
40
40
|
function collectLoweringNotices(target, notices) {
|
|
41
41
|
const keys = new Set(target.map((notice) => JSON.stringify(notice)));
|
|
@@ -1624,8 +1624,6 @@ export async function lookup(ref) {
|
|
|
1624
1624
|
return lookupBundleRef({ bundle: ref.origin, conceptId: conceptIdFromTypeName(ref.type, ref.name) });
|
|
1625
1625
|
}
|
|
1626
1626
|
// ── Utility score recomputation ──────────────────────────────────────────────
|
|
1627
|
-
/** Retention window for usage events: events older than this are purged. */
|
|
1628
|
-
const USAGE_EVENT_RETENTION_DAYS = 90;
|
|
1629
1627
|
/**
|
|
1630
1628
|
* Recompute utility scores for all entries based on usage_events data.
|
|
1631
1629
|
*
|
|
@@ -0,0 +1,90 @@
|
|
|
1
|
+
// This Source Code Form is subject to the terms of the Mozilla Public
|
|
2
|
+
// License, v. 2.0. If a copy of the MPL was not distributed with this
|
|
3
|
+
// file, You can obtain one at https://mozilla.org/MPL/2.0/.
|
|
4
|
+
import { stashDirFor } from "../../core/asset/asset-placement.js";
|
|
5
|
+
import { isBundleSlug, parseBundleRef } from "../../core/asset/asset-ref.js";
|
|
6
|
+
/** Channel order is the order links are stored and shown in. */
|
|
7
|
+
const CHANNELS = [
|
|
8
|
+
["xref", (doc) => doc.xrefs],
|
|
9
|
+
["superseded_by", (doc) => doc.supersededBy],
|
|
10
|
+
["contradicted_by", (doc) => doc.contradictedBy],
|
|
11
|
+
["belief_peer", (doc) => doc.currentBeliefRefs],
|
|
12
|
+
["derived_from", (doc) => doc.derivedFrom],
|
|
13
|
+
["cites", (doc) => doc.sources],
|
|
14
|
+
["links_to", (doc) => doc.links],
|
|
15
|
+
["uses", (doc) => doc.uses],
|
|
16
|
+
];
|
|
17
|
+
/** The links an indexed document declares, in channel then authored order; one per (kind, target). */
|
|
18
|
+
export function declaredLinks(doc, owner) {
|
|
19
|
+
const links = [];
|
|
20
|
+
const seen = new Set();
|
|
21
|
+
for (const [kind, read] of CHANNELS) {
|
|
22
|
+
for (const raw of stringValues(read(doc))) {
|
|
23
|
+
const target = linkTarget(raw, owner.conceptId);
|
|
24
|
+
if (!target)
|
|
25
|
+
continue;
|
|
26
|
+
const bundle = target.bundle ?? owner.bundleId;
|
|
27
|
+
if (bundle === owner.bundleId && target.conceptId === owner.conceptId)
|
|
28
|
+
continue;
|
|
29
|
+
const key = `${kind}\0${bundle}\0${target.conceptId}`;
|
|
30
|
+
if (seen.has(key))
|
|
31
|
+
continue;
|
|
32
|
+
seen.add(key);
|
|
33
|
+
links.push({ kind, raw, ...target });
|
|
34
|
+
}
|
|
35
|
+
}
|
|
36
|
+
return links;
|
|
37
|
+
}
|
|
38
|
+
function stringValues(value) {
|
|
39
|
+
if (typeof value === "string")
|
|
40
|
+
return [value];
|
|
41
|
+
return Array.isArray(value) ? value.filter((item) => typeof item === "string") : [];
|
|
42
|
+
}
|
|
43
|
+
/**
|
|
44
|
+
* The target a token names, or `undefined` when the token is not an asset ref
|
|
45
|
+
* (a session id, a URL, prose, a template placeholder). Retired spellings
|
|
46
|
+
* convert in memory: `<type>:<name>` (`memory:x`), `wiki:<wiki>/<path>` (0.8
|
|
47
|
+
* wikis, now knowledge under `wikis/`), a `.md` suffix, and a `#fragment`.
|
|
48
|
+
* Inside a wiki page, `raw/…` and `pages/…` are relative to the wiki root.
|
|
49
|
+
*/
|
|
50
|
+
function linkTarget(raw, ownerConceptId) {
|
|
51
|
+
const token = raw.trim();
|
|
52
|
+
if (!token || /[\s<*]/.test(token) || token.includes("$(") || token.includes("${"))
|
|
53
|
+
return;
|
|
54
|
+
let bundle;
|
|
55
|
+
let body = token;
|
|
56
|
+
const boundary = token.indexOf("//");
|
|
57
|
+
if (boundary >= 0) {
|
|
58
|
+
bundle = token.slice(0, boundary);
|
|
59
|
+
body = token.slice(boundary + 2);
|
|
60
|
+
if (!isBundleSlug(bundle))
|
|
61
|
+
return;
|
|
62
|
+
}
|
|
63
|
+
const conceptId = legacyConceptId(body.split("#", 1)[0] ?? "", ownerConceptId);
|
|
64
|
+
if (conceptId === undefined || conceptId.length <= 1 || conceptId.includes("//"))
|
|
65
|
+
return;
|
|
66
|
+
try {
|
|
67
|
+
return { ...(bundle ? { bundle } : {}), conceptId: parseBundleRef(conceptId).conceptId };
|
|
68
|
+
}
|
|
69
|
+
catch {
|
|
70
|
+
return;
|
|
71
|
+
}
|
|
72
|
+
}
|
|
73
|
+
function legacyConceptId(body, ownerConceptId) {
|
|
74
|
+
const stripped = body.replace(/\.md$/i, "");
|
|
75
|
+
const colon = stripped.indexOf(":");
|
|
76
|
+
if (colon >= 0) {
|
|
77
|
+
const type = stripped.slice(0, colon);
|
|
78
|
+
const name = stripped.slice(colon + 1);
|
|
79
|
+
if (!name || name.includes(":"))
|
|
80
|
+
return;
|
|
81
|
+
if (type === "wiki")
|
|
82
|
+
return `knowledge/wikis/${name}`;
|
|
83
|
+
const stashDir = stashDirFor(type);
|
|
84
|
+
return stashDir === undefined ? undefined : `${stashDir}/${name}`;
|
|
85
|
+
}
|
|
86
|
+
const wikiRoot = /^((?:[^/]+\/)*?wikis\/[^/]+\/)/.exec(ownerConceptId)?.[1];
|
|
87
|
+
if (wikiRoot && /^(raw|pages)\//.test(stripped))
|
|
88
|
+
return `${wikiRoot}${stripped}`;
|
|
89
|
+
return stripped;
|
|
90
|
+
}
|
|
@@ -138,6 +138,7 @@ export function indexDocumentToStashEntry(doc) {
|
|
|
138
138
|
entry.wikiRole = dj.wikiRole;
|
|
139
139
|
assignStringList(entry, "sources", dj.sources);
|
|
140
140
|
assignStringList(entry, "evidenceSources", dj.evidenceSources);
|
|
141
|
+
assignStringList(entry, "uses", dj.uses);
|
|
141
142
|
// D2 (#730): OKF v0.2 provenance promoteProposal stamps onto AKM-native
|
|
142
143
|
// writes, carried via DOCUMENT_JSON_CARRIED_FIELDS (akm-adapter.ts) —
|
|
143
144
|
// unpacked back to a first-class member here exactly like `sources`/
|
|
@@ -91,6 +91,40 @@ export function countUsageEventsByType(db, eventType) {
|
|
|
91
91
|
return db.prepare("SELECT COUNT(*) AS cnt FROM usage_events WHERE event_type = ?").get(eventType)
|
|
92
92
|
.cnt;
|
|
93
93
|
}
|
|
94
|
+
/**
|
|
95
|
+
* Durable refs a user-attributed `search`, `curate` or `show` returned, or a
|
|
96
|
+
* user `feedback` named, at or after `sinceIso`. Machine traffic (`improve`,
|
|
97
|
+
* `task`, `audit`, `unknown`) is not demand, as in {@link countFeedbackSignals}.
|
|
98
|
+
*/
|
|
99
|
+
export function listUsedEntryRefs(db, sinceIso) {
|
|
100
|
+
const rows = db
|
|
101
|
+
.prepare(`SELECT DISTINCT entry_ref FROM usage_events
|
|
102
|
+
WHERE event_type IN ('search', 'curate', 'show', 'feedback')
|
|
103
|
+
AND source = 'user'
|
|
104
|
+
AND entry_ref IS NOT NULL
|
|
105
|
+
AND julianday(created_at) >= julianday(?)`)
|
|
106
|
+
.all(sinceIso);
|
|
107
|
+
return rows.map((row) => row.entry_ref);
|
|
108
|
+
}
|
|
109
|
+
/**
|
|
110
|
+
* The distinct query texts of user `search` and `curate` events that returned
|
|
111
|
+
* `conceptId` (any bundle prefix), most recent first.
|
|
112
|
+
*/
|
|
113
|
+
export function listRetrievalQueries(db, conceptId) {
|
|
114
|
+
const rows = db
|
|
115
|
+
.prepare(`SELECT query, MAX(created_at) AS last_at FROM usage_events
|
|
116
|
+
WHERE event_type IN ('search', 'curate')
|
|
117
|
+
AND source = 'user'
|
|
118
|
+
AND query IS NOT NULL AND trim(query) != ''
|
|
119
|
+
AND instr(entry_ref, '//') > 0
|
|
120
|
+
AND substr(entry_ref, instr(entry_ref, '//') + 2) = ?
|
|
121
|
+
GROUP BY query
|
|
122
|
+
ORDER BY last_at DESC`)
|
|
123
|
+
.all(conceptId);
|
|
124
|
+
return rows.map((row) => row.query);
|
|
125
|
+
}
|
|
126
|
+
/** Usage events older than this many days are purged on every `akm index`. */
|
|
127
|
+
export const USAGE_EVENT_RETENTION_DAYS = 90;
|
|
94
128
|
/**
|
|
95
129
|
* Delete usage events older than the given number of days.
|
|
96
130
|
*/
|
|
@@ -185,30 +185,6 @@ function bumpTelemetry(telemetry, key, amount = 1) {
|
|
|
185
185
|
return;
|
|
186
186
|
telemetry[key] = (telemetry[key] ?? 0) + amount;
|
|
187
187
|
}
|
|
188
|
-
/**
|
|
189
|
-
* `Promise.all(items.map(fn))` with at most `limit` calls in flight, so the
|
|
190
|
-
* per-asset calls of one batch never outnumber the chunk pool's concurrency
|
|
191
|
-
* (GR-D14). The first rejection rejects the map and stops further calls.
|
|
192
|
-
*/
|
|
193
|
-
async function mapWithConcurrency(items, limit, fn) {
|
|
194
|
-
const results = new Array(items.length);
|
|
195
|
-
let next = 0;
|
|
196
|
-
let failed = false;
|
|
197
|
-
const worker = async () => {
|
|
198
|
-
while (!failed && next < items.length) {
|
|
199
|
-
const index = next++;
|
|
200
|
-
try {
|
|
201
|
-
results[index] = await fn(items[index]);
|
|
202
|
-
}
|
|
203
|
-
catch (error) {
|
|
204
|
-
failed = true;
|
|
205
|
-
throw error;
|
|
206
|
-
}
|
|
207
|
-
}
|
|
208
|
-
};
|
|
209
|
-
await Promise.all(Array.from({ length: Math.min(Math.max(1, limit), items.length) }, worker));
|
|
210
|
-
return results;
|
|
211
|
-
}
|
|
212
188
|
/** Count what the parser dropped from one parsed extraction (single-asset or batch item). */
|
|
213
189
|
function bumpFilterTelemetry(telemetry, extraction) {
|
|
214
190
|
bumpTelemetry(telemetry, "filteredGenericEntities", extraction.filteredGenericEntities ?? 0);
|
|
@@ -260,6 +236,7 @@ function mergeGraphExtractions(extractions) {
|
|
|
260
236
|
let filteredGenericEntities = 0;
|
|
261
237
|
let filteredInvalidRelations = 0;
|
|
262
238
|
let filteredLowConfidenceRelations = 0;
|
|
239
|
+
let anyChunkFailed = false;
|
|
263
240
|
let firstFailureReason;
|
|
264
241
|
for (const extraction of extractions) {
|
|
265
242
|
truncationCount += extraction.truncationCount ?? 0;
|
|
@@ -267,8 +244,10 @@ function mergeGraphExtractions(extractions) {
|
|
|
267
244
|
filteredGenericEntities += extraction.filteredGenericEntities ?? 0;
|
|
268
245
|
filteredInvalidRelations += extraction.filteredInvalidRelations ?? 0;
|
|
269
246
|
filteredLowConfidenceRelations += extraction.filteredLowConfidenceRelations ?? 0;
|
|
270
|
-
if (extraction.status === "failed"
|
|
271
|
-
|
|
247
|
+
if (extraction.status === "failed") {
|
|
248
|
+
anyChunkFailed = true;
|
|
249
|
+
firstFailureReason ??= extraction.reason;
|
|
250
|
+
}
|
|
272
251
|
const nextConfidence = parseConfidence(extraction.confidence);
|
|
273
252
|
if (nextConfidence !== undefined)
|
|
274
253
|
confidence = confidence === undefined ? nextConfidence : Math.max(confidence, nextConfidence);
|
|
@@ -329,8 +308,12 @@ function mergeGraphExtractions(extractions) {
|
|
|
329
308
|
const chunkCount = relationChunkCounts.get(key) ?? 1;
|
|
330
309
|
relation.confidence = blendConsistency(relation.confidence, chunkCount);
|
|
331
310
|
}
|
|
332
|
-
|
|
333
|
-
|
|
311
|
+
// One failed chunk fails the body, whatever the others found: a failed
|
|
312
|
+
// extraction is never cached, so the next run extracts the body again
|
|
313
|
+
// (every chunk: partial failures are rare outside provider outages, which
|
|
314
|
+
// the run's failure-rate abort stops). The entities found are kept.
|
|
315
|
+
const status = anyChunkFailed ? "failed" : entities.length > 0 ? "extracted" : "empty";
|
|
316
|
+
const reason = status === "extracted" ? "none" : status === "failed" ? (firstFailureReason ?? "llm_error") : "no_graph_content";
|
|
334
317
|
const mergedConfidence = confidence !== undefined ? blendConsistency(confidence, totalChunks) : totalChunks > 1 ? 1 : undefined;
|
|
335
318
|
return {
|
|
336
319
|
entities,
|
|
@@ -606,11 +589,21 @@ export async function extractGraphFromBodies(llmRunner, bodies, signal, akmConfi
|
|
|
606
589
|
const result = await extractGraphFromBody(llmRunner, bodies[0] ?? "", signal, akmConfig, onFallback, options);
|
|
607
590
|
return [result];
|
|
608
591
|
}
|
|
609
|
-
|
|
592
|
+
// A batch makes its per-asset calls one at a time: the caller's pool runs
|
|
593
|
+
// batches side by side and is the run's only concurrency, so the calls in
|
|
594
|
+
// flight never exceed it (not the pool's width squared).
|
|
610
595
|
const extractOne = (body) => extractGraphFromBody(llmRunner, body, signal, akmConfig, onFallback, options);
|
|
596
|
+
const extractEach = async (indices, into) => {
|
|
597
|
+
for (const index of indices)
|
|
598
|
+
into[index] = await extractOne(bodies[index] ?? "");
|
|
599
|
+
};
|
|
611
600
|
// With batching disabled every body takes the single-asset path, once.
|
|
612
|
-
if (batchState?.batchingDisabled)
|
|
613
|
-
|
|
601
|
+
if (batchState?.batchingDisabled) {
|
|
602
|
+
const results = [];
|
|
603
|
+
for (const body of bodies)
|
|
604
|
+
results.push(await extractOne(body));
|
|
605
|
+
return results;
|
|
606
|
+
}
|
|
614
607
|
// Filter out bodies that are empty so we don't waste tokens, but keep
|
|
615
608
|
// index correspondence by tracking which indices were non-empty.
|
|
616
609
|
const results = bodies.map(empty);
|
|
@@ -629,9 +622,7 @@ export async function extractGraphFromBodies(llmRunner, bodies, signal, akmConfi
|
|
|
629
622
|
}
|
|
630
623
|
}
|
|
631
624
|
}
|
|
632
|
-
await
|
|
633
|
-
results[index] = await extractOne(bodies[index] ?? "");
|
|
634
|
-
});
|
|
625
|
+
await extractEach(oversizedIndices, results);
|
|
635
626
|
if (nonEmptyBodies.length === 0)
|
|
636
627
|
return results;
|
|
637
628
|
const systemPrompt = buildBatchSystemPrompt();
|
|
@@ -780,9 +771,7 @@ export async function extractGraphFromBodies(llmRunner, bodies, signal, akmConfi
|
|
|
780
771
|
warn(`graph extraction (batch): response had ${batchResult.length} items for ${nonEmptyBodies.length} assets; ` +
|
|
781
772
|
`falling back to individual calls for ${fallbackIndices.length} missing asset(s).`);
|
|
782
773
|
}
|
|
783
|
-
await
|
|
784
|
-
results[origIdx] = await extractOne(bodies[origIdx] ?? "");
|
|
785
|
-
});
|
|
774
|
+
await extractEach(fallbackIndices, results);
|
|
786
775
|
}
|
|
787
776
|
else if (batchContextError) {
|
|
788
777
|
warn(`graph extraction (batch): skipped ${nonEmptyBodies.length} asset(s) due to context size error; ` +
|
|
@@ -358,6 +358,7 @@ export function shapeShowOutput(result, detail, shape = "human") {
|
|
|
358
358
|
"steps",
|
|
359
359
|
"keys",
|
|
360
360
|
"related",
|
|
361
|
+
"links",
|
|
361
362
|
...FRAGMENT_PROVENANCE_FIELDS,
|
|
362
363
|
...FRAGMENT_CONTEXT_FIELDS,
|
|
363
364
|
]);
|
|
@@ -382,6 +383,7 @@ export function shapeShowOutput(result, detail, shape = "human") {
|
|
|
382
383
|
"origin",
|
|
383
384
|
"keys",
|
|
384
385
|
"related",
|
|
386
|
+
"links",
|
|
385
387
|
...FRAGMENT_PROVENANCE_FIELDS,
|
|
386
388
|
...FRAGMENT_CONTEXT_FIELDS,
|
|
387
389
|
]);
|
|
@@ -411,6 +413,7 @@ export function shapeShowOutput(result, detail, shape = "human") {
|
|
|
411
413
|
"activeRun",
|
|
412
414
|
"keys",
|
|
413
415
|
"related",
|
|
416
|
+
"links",
|
|
414
417
|
...FRAGMENT_PROVENANCE_FIELDS,
|
|
415
418
|
...FRAGMENT_CONTEXT_FIELDS,
|
|
416
419
|
// ref, path, and editable are always projected — at every --detail level,
|
|
@@ -67,6 +67,22 @@ export function formatShowPlain(r, detail) {
|
|
|
67
67
|
lines.push(` relationCount: ${String(hit.relationCount ?? 0)}`);
|
|
68
68
|
}
|
|
69
69
|
}
|
|
70
|
+
const links = typeof r.links === "object" && r.links !== null ? r.links : undefined;
|
|
71
|
+
if (links) {
|
|
72
|
+
lines.push("");
|
|
73
|
+
lines.push("links:");
|
|
74
|
+
for (const part of ["outgoing", "incoming", "unresolved"]) {
|
|
75
|
+
const groups = links[part];
|
|
76
|
+
if (typeof groups !== "object" || groups === null)
|
|
77
|
+
continue;
|
|
78
|
+
lines.push(` ${part}:`);
|
|
79
|
+
for (const [kind, group] of Object.entries(groups)) {
|
|
80
|
+
const refs = Array.isArray(group.refs) ? group.refs.map(String) : [];
|
|
81
|
+
const more = (group.total ?? refs.length) > refs.length ? ` (+${(group.total ?? 0) - refs.length} more)` : "";
|
|
82
|
+
lines.push(` ${kind}: ${refs.join(", ")}${more}`);
|
|
83
|
+
}
|
|
84
|
+
}
|
|
85
|
+
}
|
|
70
86
|
const payloads = [r.content, r.template, r.prompt].filter((value) => value != null).map(String);
|
|
71
87
|
if (Array.isArray(r.steps) && r.steps.length > 0) {
|
|
72
88
|
if (lines.length > 0)
|