akm-cli 0.9.17-alpha.7 → 0.9.17-alpha.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +473 -0
- package/STABILITY.md +9 -8
- package/dist/akm +55 -22
- package/dist/akm-migrate +38 -19
- package/dist/assets/hints/cli-hints-full.md +6 -7
- package/dist/assets/improve-strategies/catchup.json +0 -3
- package/dist/assets/improve-strategies/consolidate.json +0 -1
- package/dist/assets/improve-strategies/default.json +1 -2
- package/dist/assets/improve-strategies/proactive-maintenance.json +1 -2
- package/dist/assets/improve-strategies/quick.json +1 -2
- package/dist/assets/improve-strategies/reflect-distill.json +1 -2
- package/dist/assets/improve-strategies/thorough.json +0 -3
- package/dist/assets/prompts/consolidate-pair.md +20 -0
- package/dist/assets/stash-skeleton/facts/conventions/backlinks.md +20 -20
- package/dist/assets/stash-skeleton/facts/conventions/domains.md +2 -2
- package/dist/assets/templates/html/health.html +3 -5
- package/dist/cli/retired-commands.js +1 -1
- package/dist/commands/health/archive-usage.js +98 -0
- package/dist/commands/health/data-dir-usage.js +25 -13
- package/dist/commands/health/html-report.js +1 -4
- package/dist/commands/health/improve-metrics.js +0 -25
- package/dist/commands/health/md-report.js +1 -6
- package/dist/commands/health/report-view-model.js +4 -14
- package/dist/commands/health/windows.js +0 -1
- package/dist/commands/health.js +13 -0
- package/dist/commands/improve/consolidate/continuity-check.js +137 -0
- package/dist/commands/improve/consolidate/pair-pass.js +791 -0
- package/dist/commands/improve/consolidate.js +38 -63
- package/dist/commands/improve/extract-prompt.js +1 -2
- package/dist/commands/improve/improve-cli.js +1 -1
- package/dist/commands/improve/improve-strategies.js +23 -5
- package/dist/commands/improve/improve.js +19 -30
- package/dist/commands/improve/ledger.js +3 -2
- package/dist/commands/improve/loop-stages.js +5 -84
- package/dist/commands/improve/memory/memory-belief.js +3 -1
- package/dist/commands/improve/memory/memory-improve.js +269 -11
- package/dist/commands/improve/planner.js +0 -5
- package/dist/commands/improve/preparation.js +20 -135
- package/dist/commands/improve/retrieval-scope.js +19 -4
- package/dist/commands/improve/salience.js +1 -14
- package/dist/commands/improve/stage.js +0 -1
- package/dist/commands/lint/base-linter.js +19 -11
- package/dist/commands/proposal/drain.js +8 -1
- package/dist/commands/proposal/proposal-cli.js +16 -2
- package/dist/commands/proposal/proposal-types.js +7 -0
- package/dist/commands/proposal/proposal.js +37 -6
- package/dist/commands/proposal/repository.js +613 -4
- package/dist/commands/proposal/validators/proposals.js +9 -0
- package/dist/commands/read/curate.js +40 -13
- package/dist/commands/read/knowledge.js +3 -2
- package/dist/commands/read/show.js +55 -16
- package/dist/commands/sources/info.js +3 -0
- package/dist/commands/sources/stash-cli.js +2 -2
- package/dist/core/adapter/adapters/akm-adapter.js +2 -0
- package/dist/core/adapter/adapters/akm-metadata.js +31 -0
- package/dist/core/bundle-rename.js +1 -7
- package/dist/core/config/config-schema.js +8 -1
- package/dist/core/config/config.js +23 -48
- package/dist/core/config/engine-semantics.js +0 -2
- package/dist/core/config/schema/improve-processes.js +17 -42
- package/dist/core/config/schema/index-config.js +5 -25
- package/dist/core/file-change.js +13 -5
- package/dist/core/improve-result.js +16 -5
- package/dist/core/improve-types.js +0 -1
- package/dist/core/loopback.js +7 -12
- package/dist/core/parse.js +13 -16
- package/dist/core/state/migrations.js +15 -0
- package/dist/core/time.js +0 -20
- package/dist/indexer/db/llm-cache.js +2 -2
- package/dist/indexer/ensure-index.js +2 -2
- package/dist/indexer/index-written-assets.js +2 -3
- package/dist/indexer/indexer.js +18 -418
- package/dist/indexer/links/declared-links.js +90 -0
- package/dist/indexer/passes/metadata.js +0 -19
- package/dist/indexer/scan/doc-to-entry.js +1 -0
- package/dist/indexer/walk/walker.js +3 -4
- package/dist/llm/client.js +8 -10
- package/dist/llm/embedders/remote.js +1 -2
- package/dist/llm/feature-gate.js +0 -5
- package/dist/output/shapes/helpers.js +23 -4
- package/dist/output/text/command-format.js +0 -8
- package/dist/output/text/proposal-format.js +47 -1
- package/dist/output/text/show-format.js +13 -17
- package/dist/scripts/akm-migrate-node.js +2754 -2836
- package/dist/scripts/akm-migrate.js +2754 -2836
- package/dist/setup/steps/connection.js +5 -6
- package/dist/setup/steps/platforms.js +2 -2
- package/dist/sources/providers/git-stash.js +55 -4
- package/dist/storage/repositories/improve-ledger-repository.js +48 -7
- package/dist/storage/repositories/index-entries-repository.js +16 -13
- package/dist/storage/repositories/index-entry-schema.js +22 -3
- package/dist/storage/repositories/index-links-repository.js +143 -0
- package/dist/storage/repositories/index-llm-cache-repository.js +7 -26
- package/dist/storage/repositories/index-schema.js +82 -104
- package/dist/storage/repositories/proposals-repository.js +61 -0
- package/dist/storage/repositories/salience-repository.js +1 -19
- package/dist/tasks/source/task-to-v4.js +462 -74
- package/docs/migration/release-notes/0.9.17.md +7 -5
- package/docs/reference/cli.md +33 -21
- package/docs/reference/configuration.md +21 -12
- package/docs/reference/data-and-telemetry.md +0 -1
- package/package.json +1 -1
- package/schemas/akm-config.json +0 -342
- package/dist/assets/improve-strategies/graph-refresh.json +0 -15
- package/dist/assets/prompts/contradiction-judge.md +0 -33
- package/dist/assets/prompts/graph-extract-system.md +0 -1
- package/dist/assets/prompts/graph-extract-user-prompt.md +0 -35
- package/dist/assets/prompts/metadata-enhance-system.md +0 -1
- package/dist/assets/tasks/improve/akm-graph-refresh-weekly.yml +0 -4
- package/dist/indexer/db/graph-db.js +0 -431
- package/dist/indexer/graph/graph-extraction.js +0 -807
- package/dist/indexer/graph/graph-related.js +0 -131
- package/dist/indexer/graph/graph-types.js +0 -4
- package/dist/llm/graph-extract.js +0 -903
- package/dist/llm/metadata-enhance.js +0 -95
- package/dist/tasks/source/task-to-v3.js +0 -453
|
@@ -1,903 +0,0 @@
|
|
|
1
|
-
// This Source Code Form is subject to the terms of the Mozilla Public
|
|
2
|
-
// License, v. 2.0. If a copy of the MPL was not distributed with this
|
|
3
|
-
// file, You can obtain one at https://mozilla.org/MPL/2.0/.
|
|
4
|
-
/**
|
|
5
|
-
* LLM helper for the `akm index` graph-extraction pass (#207).
|
|
6
|
-
*
|
|
7
|
-
* Given a single asset body (typically a `memory:` or `knowledge:` file),
|
|
8
|
-
* asks the configured LLM to surface the entities mentioned in it and the
|
|
9
|
-
* relations between them. The pass itself
|
|
10
|
-
* (`src/indexer/graph/graph-extraction.ts`) is responsible for deciding which
|
|
11
|
-
* files to extract, persisting the resulting nodes/edges to the index DB,
|
|
12
|
-
* and feeding the graph data into the FTS5+boosts
|
|
13
|
-
* search pipeline as a single boost component.
|
|
14
|
-
*
|
|
15
|
-
* This module is intentionally tiny and stateless so tests can stub it via
|
|
16
|
-
* `mock.module("../src/llm/graph-extract", ...)` without hitting a network.
|
|
17
|
-
*
|
|
18
|
-
* The symbolic LLM runner comes from the current index-pass execution
|
|
19
|
-
* resolution and is passed straight through.
|
|
20
|
-
*/
|
|
21
|
-
import systemPromptTemplate from "../assets/prompts/graph-extract-system.md" with { type: "text" };
|
|
22
|
-
import userPromptTemplate from "../assets/prompts/graph-extract-user-prompt.md" with { type: "text" };
|
|
23
|
-
import { splitMarkdownFragmentStats } from "../core/asset/markdown-fragments.js";
|
|
24
|
-
import { toErrorMessage } from "../core/common.js";
|
|
25
|
-
import { ConfigError } from "../core/errors.js";
|
|
26
|
-
import { parseEmbeddedJsonResponse } from "../core/parse.js";
|
|
27
|
-
import { warn, warnVerbose } from "../core/warn.js";
|
|
28
|
-
import { isContextSizeError, isTransportFailure, LlmCallError } from "./client.js";
|
|
29
|
-
import { tryLlmFeature } from "./feature-gate.js";
|
|
30
|
-
import { callStructured } from "./structured-call.js";
|
|
31
|
-
/**
|
|
32
|
-
* Separator token used between assets in a batch prompt.
|
|
33
|
-
* Chosen to be visually clear and unlikely to appear verbatim in asset bodies.
|
|
34
|
-
*/
|
|
35
|
-
const BATCH_ASSET_SEPARATOR = "=== ASSET";
|
|
36
|
-
/**
|
|
37
|
-
* Part of the extractor id that keys cached extractions; the prompt text is
|
|
38
|
-
* not. Bump it with any change to either prompt, and expect every cached file
|
|
39
|
-
* to be extracted again. Pending for the next bump (GR-D12): the batch prompt
|
|
40
|
-
* still asks for "file/dir names", which the single-asset prompt rules out.
|
|
41
|
-
*/
|
|
42
|
-
export const GRAPH_EXTRACT_PROMPT_VERSION = "v3";
|
|
43
|
-
/** Asset bodies longer than this are chunked instead of truncated. */
|
|
44
|
-
const MAX_CHUNK_BODY_CHARS = 1600;
|
|
45
|
-
/** Bodies longer than this are excluded from multi-asset batch prompts. */
|
|
46
|
-
const MAX_BATCH_BODY_CHARS = 1600;
|
|
47
|
-
const MIN_RELATION_CONFIDENCE = 0.5;
|
|
48
|
-
const NON_ARRAY_BATCH_DISABLE_THRESHOLD = 2;
|
|
49
|
-
/** Hard cap on entities returned per asset — guards against runaway LLM output. */
|
|
50
|
-
const MAX_ENTITIES_PER_ASSET = 32;
|
|
51
|
-
/** Hard cap on relations returned per asset. */
|
|
52
|
-
const MAX_RELATIONS_PER_ASSET = 32;
|
|
53
|
-
/**
|
|
54
|
-
* Default cap on chunks processed per asset (R12b + R20) — overridable via
|
|
55
|
-
* `processes.graphExtraction.maxChunksPerAsset`. Without a cap, one long file
|
|
56
|
-
* chunked at MAX_CHUNK_BODY_CHARS could spend dozens of calls on a single
|
|
57
|
-
* asset (one file spent 21 of 27 run calls this way) before its output was
|
|
58
|
-
* sliced to MAX_ENTITIES_PER_ASSET/MAX_RELATIONS_PER_ASSET anyway.
|
|
59
|
-
*/
|
|
60
|
-
const DEFAULT_MAX_CHUNKS_PER_ASSET = 8;
|
|
61
|
-
const SYSTEM_PROMPT = systemPromptTemplate;
|
|
62
|
-
const USER_PROMPT_PREFIX = userPromptTemplate
|
|
63
|
-
.replace("{{MAX_ENTITIES}}", String(MAX_ENTITIES_PER_ASSET))
|
|
64
|
-
.replace("{{MAX_RELATIONS}}", String(MAX_RELATIONS_PER_ASSET));
|
|
65
|
-
/**
|
|
66
|
-
* Strict JSON Schema for one asset's extraction payload (R12b, compacted for
|
|
67
|
-
* R12). Sent via `responseSchema` to providers that opt into structured
|
|
68
|
-
* output (`runner.connection.supportsJsonSchema` — same lift as
|
|
69
|
-
* memory-infer.ts's `DERIVED_MEMORY_JSON_SCHEMA`); the client silently drops
|
|
70
|
-
* it otherwise. `maxItems` mirrors MAX_ENTITIES_PER_ASSET/MAX_RELATIONS_PER_ASSET
|
|
71
|
-
* so a compliant provider cannot pay for output beyond what
|
|
72
|
-
* parseGraphExtraction keeps. Relations are compact `[from, type, to]`
|
|
73
|
-
* triples (`type` may be `""`) rather than `{"from","to","type"}` objects —
|
|
74
|
-
* the object-keyed form cost 10+ tokens per relation for no signal, and
|
|
75
|
-
* output tokens cost far more than prompt tokens. There is deliberately no
|
|
76
|
-
* relation-level `confidence` in the schema (the prompt never asks for one);
|
|
77
|
-
* `parseGraphExtraction` still reads it from a legacy object-shaped relation
|
|
78
|
-
* for backward compatibility. `confidence` stays at the extraction level —
|
|
79
|
-
* parseGraphExtraction reads it into the merged confidence.
|
|
80
|
-
* `additionalProperties: false` forbids anything else. Reused as the `items`
|
|
81
|
-
* schema of a batch call's array response (see
|
|
82
|
-
* {@link buildBatchResponseSchema}) so a single-asset and a batched call
|
|
83
|
-
* bound entities/relations identically.
|
|
84
|
-
*/
|
|
85
|
-
const GRAPH_EXTRACTION_ITEM_SCHEMA = {
|
|
86
|
-
type: "object",
|
|
87
|
-
properties: {
|
|
88
|
-
entities: { type: "array", items: { type: "string" }, maxItems: MAX_ENTITIES_PER_ASSET },
|
|
89
|
-
relations: {
|
|
90
|
-
type: "array",
|
|
91
|
-
maxItems: MAX_RELATIONS_PER_ASSET,
|
|
92
|
-
items: {
|
|
93
|
-
type: "array",
|
|
94
|
-
items: { type: "string" },
|
|
95
|
-
minItems: 3,
|
|
96
|
-
maxItems: 3,
|
|
97
|
-
},
|
|
98
|
-
},
|
|
99
|
-
confidence: { type: "number" },
|
|
100
|
-
},
|
|
101
|
-
required: ["entities", "relations"],
|
|
102
|
-
additionalProperties: false,
|
|
103
|
-
};
|
|
104
|
-
/** Schema for {@link extractGraphFromBody}'s single-asset `responseSchema`. */
|
|
105
|
-
const GRAPH_EXTRACTION_JSON_SCHEMA = GRAPH_EXTRACTION_ITEM_SCHEMA;
|
|
106
|
-
/**
|
|
107
|
-
* Schema for {@link extractGraphFromBodies}' batch `responseSchema` — an
|
|
108
|
-
* array of exactly `count` {@link GRAPH_EXTRACTION_ITEM_SCHEMA} elements, one
|
|
109
|
-
* per asset in the batch, matching the batch prompt's contract.
|
|
110
|
-
*/
|
|
111
|
-
function buildBatchResponseSchema(count) {
|
|
112
|
-
return {
|
|
113
|
-
type: "array",
|
|
114
|
-
minItems: count,
|
|
115
|
-
maxItems: count,
|
|
116
|
-
items: GRAPH_EXTRACTION_ITEM_SCHEMA,
|
|
117
|
-
};
|
|
118
|
-
}
|
|
119
|
-
const GENERIC_ENTITIES = new Set([
|
|
120
|
-
"agent",
|
|
121
|
-
"application",
|
|
122
|
-
"assistant",
|
|
123
|
-
"code",
|
|
124
|
-
"content",
|
|
125
|
-
"data",
|
|
126
|
-
"developer",
|
|
127
|
-
"document",
|
|
128
|
-
"file",
|
|
129
|
-
"knowledge",
|
|
130
|
-
"memory",
|
|
131
|
-
"note",
|
|
132
|
-
"notes",
|
|
133
|
-
"project",
|
|
134
|
-
"service",
|
|
135
|
-
"system",
|
|
136
|
-
"task",
|
|
137
|
-
"team",
|
|
138
|
-
"text",
|
|
139
|
-
"thing",
|
|
140
|
-
"user",
|
|
141
|
-
]);
|
|
142
|
-
const GENERIC_RELATION_TYPES = new Set(["has", "is", "mentions", "references", "related to"]);
|
|
143
|
-
function parseConfidence(raw) {
|
|
144
|
-
if (typeof raw !== "number" || !Number.isFinite(raw))
|
|
145
|
-
return undefined;
|
|
146
|
-
return Math.max(0, Math.min(1, raw));
|
|
147
|
-
}
|
|
148
|
-
function normalizeEntityName(raw) {
|
|
149
|
-
return raw
|
|
150
|
-
.trim()
|
|
151
|
-
.replace(/^[`"']+|[`"']+$/g, "")
|
|
152
|
-
.replace(/\s+/g, " ")
|
|
153
|
-
.replace(/[;,!?]+$/g, "")
|
|
154
|
-
.trim();
|
|
155
|
-
}
|
|
156
|
-
function normalizeRelationType(raw) {
|
|
157
|
-
const normalized = raw
|
|
158
|
-
.trim()
|
|
159
|
-
.toLowerCase()
|
|
160
|
-
.replace(/^[`"']+|[`"']+$/g, "")
|
|
161
|
-
.replace(/\s+/g, " ")
|
|
162
|
-
.replace(/[.;,!?]+$/g, "")
|
|
163
|
-
.trim();
|
|
164
|
-
if (!normalized)
|
|
165
|
-
return undefined;
|
|
166
|
-
if (normalized === "use" || normalized === "utilizes")
|
|
167
|
-
return "uses";
|
|
168
|
-
if (normalized === "depend on" || normalized === "depends")
|
|
169
|
-
return "depends on";
|
|
170
|
-
if (normalized === "integrates" || normalized === "integration with")
|
|
171
|
-
return "integrates with";
|
|
172
|
-
return normalized;
|
|
173
|
-
}
|
|
174
|
-
/**
|
|
175
|
-
* The key under which two entity names are the same entity: the display
|
|
176
|
-
* clean-up above, case-folded. The one normalization for graph entities:
|
|
177
|
-
* extraction deduplicates on it, the pass deduplicates on it before writing,
|
|
178
|
-
* and it is the stored `entity_norm` that `related` joins on (GR-D17).
|
|
179
|
-
*/
|
|
180
|
-
export function normalizeEntityKey(raw) {
|
|
181
|
-
return normalizeEntityName(raw).toLowerCase();
|
|
182
|
-
}
|
|
183
|
-
function bumpTelemetry(telemetry, key, amount = 1) {
|
|
184
|
-
if (!telemetry)
|
|
185
|
-
return;
|
|
186
|
-
telemetry[key] = (telemetry[key] ?? 0) + amount;
|
|
187
|
-
}
|
|
188
|
-
/**
|
|
189
|
-
* `Promise.all(items.map(fn))` with at most `limit` calls in flight, so the
|
|
190
|
-
* per-asset calls of one batch never outnumber the chunk pool's concurrency
|
|
191
|
-
* (GR-D14). The first rejection rejects the map and stops further calls.
|
|
192
|
-
*/
|
|
193
|
-
async function mapWithConcurrency(items, limit, fn) {
|
|
194
|
-
const results = new Array(items.length);
|
|
195
|
-
let next = 0;
|
|
196
|
-
let failed = false;
|
|
197
|
-
const worker = async () => {
|
|
198
|
-
while (!failed && next < items.length) {
|
|
199
|
-
const index = next++;
|
|
200
|
-
try {
|
|
201
|
-
results[index] = await fn(items[index]);
|
|
202
|
-
}
|
|
203
|
-
catch (error) {
|
|
204
|
-
failed = true;
|
|
205
|
-
throw error;
|
|
206
|
-
}
|
|
207
|
-
}
|
|
208
|
-
};
|
|
209
|
-
await Promise.all(Array.from({ length: Math.min(Math.max(1, limit), items.length) }, worker));
|
|
210
|
-
return results;
|
|
211
|
-
}
|
|
212
|
-
/** Count what the parser dropped from one parsed extraction (single-asset or batch item). */
|
|
213
|
-
function bumpFilterTelemetry(telemetry, extraction) {
|
|
214
|
-
bumpTelemetry(telemetry, "filteredGenericEntities", extraction.filteredGenericEntities ?? 0);
|
|
215
|
-
bumpTelemetry(telemetry, "filteredInvalidRelations", extraction.filteredInvalidRelations ?? 0);
|
|
216
|
-
bumpTelemetry(telemetry, "filteredLowConfidenceRelations", extraction.filteredLowConfidenceRelations ?? 0);
|
|
217
|
-
}
|
|
218
|
-
function normalizeBatchState(state) {
|
|
219
|
-
if (!state)
|
|
220
|
-
return undefined;
|
|
221
|
-
state.batchingDisabled = state.batchingDisabled === true;
|
|
222
|
-
state.nonArrayBatchFailures = Math.max(0, state.nonArrayBatchFailures ?? 0);
|
|
223
|
-
return state;
|
|
224
|
-
}
|
|
225
|
-
function splitBodyIntoChunks(body, maxChars = MAX_CHUNK_BODY_CHARS) {
|
|
226
|
-
const split = splitMarkdownFragmentStats(body, maxChars);
|
|
227
|
-
// Graph extraction keeps its historical cost shape: adjacent safe fragments
|
|
228
|
-
// share one LLM call whenever they fit. The fragment splitter is still the
|
|
229
|
-
// sole boundary authority; this is only prompt packing, never a second
|
|
230
|
-
// parser/chunker. Hard splits stay isolated and telemetry remains the core
|
|
231
|
-
// split count rather than counting ordinary heading boundaries.
|
|
232
|
-
const chunks = [];
|
|
233
|
-
let current = "";
|
|
234
|
-
for (const fragment of split.fragments) {
|
|
235
|
-
const candidate = current ? `${current}\n\n${fragment.text}` : fragment.text;
|
|
236
|
-
if (candidate.length <= maxChars) {
|
|
237
|
-
current = candidate;
|
|
238
|
-
}
|
|
239
|
-
else {
|
|
240
|
-
if (current)
|
|
241
|
-
chunks.push(current);
|
|
242
|
-
current = fragment.text;
|
|
243
|
-
}
|
|
244
|
-
}
|
|
245
|
-
if (current)
|
|
246
|
-
chunks.push(current);
|
|
247
|
-
return { chunks, truncationCount: split.hardSplitCount };
|
|
248
|
-
}
|
|
249
|
-
/** Consistency weight for blending chunk-agreement with LLM confidence. */
|
|
250
|
-
const CONSISTENCY_WEIGHT = 0.4;
|
|
251
|
-
function mergeGraphExtractions(extractions) {
|
|
252
|
-
const totalChunks = extractions.length;
|
|
253
|
-
const entityCanonical = new Map();
|
|
254
|
-
const entityChunkCounts = new Map();
|
|
255
|
-
const relationByKey = new Map();
|
|
256
|
-
const relationChunkCounts = new Map();
|
|
257
|
-
let confidence;
|
|
258
|
-
let truncationCount = 0;
|
|
259
|
-
let truncatedChunks = 0;
|
|
260
|
-
let filteredGenericEntities = 0;
|
|
261
|
-
let filteredInvalidRelations = 0;
|
|
262
|
-
let filteredLowConfidenceRelations = 0;
|
|
263
|
-
let firstFailureReason;
|
|
264
|
-
for (const extraction of extractions) {
|
|
265
|
-
truncationCount += extraction.truncationCount ?? 0;
|
|
266
|
-
truncatedChunks += extraction.truncatedChunks ?? 0;
|
|
267
|
-
filteredGenericEntities += extraction.filteredGenericEntities ?? 0;
|
|
268
|
-
filteredInvalidRelations += extraction.filteredInvalidRelations ?? 0;
|
|
269
|
-
filteredLowConfidenceRelations += extraction.filteredLowConfidenceRelations ?? 0;
|
|
270
|
-
if (extraction.status === "failed" && !firstFailureReason)
|
|
271
|
-
firstFailureReason = extraction.reason;
|
|
272
|
-
const nextConfidence = parseConfidence(extraction.confidence);
|
|
273
|
-
if (nextConfidence !== undefined)
|
|
274
|
-
confidence = confidence === undefined ? nextConfidence : Math.max(confidence, nextConfidence);
|
|
275
|
-
for (const entity of extraction.entities) {
|
|
276
|
-
const key = normalizeEntityKey(entity);
|
|
277
|
-
if (!key)
|
|
278
|
-
continue;
|
|
279
|
-
if (!entityCanonical.has(key))
|
|
280
|
-
entityCanonical.set(key, entity);
|
|
281
|
-
entityChunkCounts.set(key, (entityChunkCounts.get(key) ?? 0) + 1);
|
|
282
|
-
}
|
|
283
|
-
}
|
|
284
|
-
for (const extraction of extractions) {
|
|
285
|
-
for (const relation of extraction.relations) {
|
|
286
|
-
const fromKey = normalizeEntityKey(relation.from);
|
|
287
|
-
const toKey = normalizeEntityKey(relation.to);
|
|
288
|
-
const type = normalizeRelationType(relation.type ?? "");
|
|
289
|
-
if (!fromKey || !toKey || !type)
|
|
290
|
-
continue;
|
|
291
|
-
const from = entityCanonical.get(fromKey);
|
|
292
|
-
const to = entityCanonical.get(toKey);
|
|
293
|
-
if (!from || !to)
|
|
294
|
-
continue;
|
|
295
|
-
const key = `${fromKey}\u0000${toKey}\u0000${type}`;
|
|
296
|
-
if (!relationByKey.has(key)) {
|
|
297
|
-
relationByKey.set(key, {
|
|
298
|
-
from,
|
|
299
|
-
to,
|
|
300
|
-
type,
|
|
301
|
-
});
|
|
302
|
-
relationChunkCounts.set(key, 0);
|
|
303
|
-
}
|
|
304
|
-
relationChunkCounts.set(key, (relationChunkCounts.get(key) ?? 0) + 1);
|
|
305
|
-
const nextConfidence = parseConfidence(relation.confidence);
|
|
306
|
-
const existing = relationByKey.get(key);
|
|
307
|
-
if (existing && nextConfidence !== undefined) {
|
|
308
|
-
const current = parseConfidence(existing.confidence) ?? 0;
|
|
309
|
-
if (nextConfidence > current)
|
|
310
|
-
existing.confidence = nextConfidence;
|
|
311
|
-
}
|
|
312
|
-
}
|
|
313
|
-
}
|
|
314
|
-
function blendConsistency(llmConfidence, chunkCount) {
|
|
315
|
-
const consistency = totalChunks > 1 ? chunkCount / totalChunks : 1;
|
|
316
|
-
if (llmConfidence === undefined)
|
|
317
|
-
return consistency;
|
|
318
|
-
return (1 - CONSISTENCY_WEIGHT) * llmConfidence + CONSISTENCY_WEIGHT * consistency;
|
|
319
|
-
}
|
|
320
|
-
const entities = [...entityCanonical.values()].slice(0, MAX_ENTITIES_PER_ASSET);
|
|
321
|
-
const relations = [...relationByKey.values()].slice(0, MAX_RELATIONS_PER_ASSET);
|
|
322
|
-
for (const relation of relations) {
|
|
323
|
-
const fromKey = normalizeEntityKey(relation.from);
|
|
324
|
-
const toKey = normalizeEntityKey(relation.to);
|
|
325
|
-
const type = normalizeRelationType(relation.type ?? "");
|
|
326
|
-
if (!fromKey || !toKey || !type)
|
|
327
|
-
continue;
|
|
328
|
-
const key = `${fromKey}\u0000${toKey}\u0000${type}`;
|
|
329
|
-
const chunkCount = relationChunkCounts.get(key) ?? 1;
|
|
330
|
-
relation.confidence = blendConsistency(relation.confidence, chunkCount);
|
|
331
|
-
}
|
|
332
|
-
const status = entities.length > 0 ? "extracted" : firstFailureReason ? "failed" : "empty";
|
|
333
|
-
const reason = status === "extracted" ? "none" : (firstFailureReason ?? "no_graph_content");
|
|
334
|
-
const mergedConfidence = confidence !== undefined ? blendConsistency(confidence, totalChunks) : totalChunks > 1 ? 1 : undefined;
|
|
335
|
-
return {
|
|
336
|
-
entities,
|
|
337
|
-
relations,
|
|
338
|
-
...(mergedConfidence !== undefined ? { confidence: mergedConfidence } : {}),
|
|
339
|
-
status,
|
|
340
|
-
reason,
|
|
341
|
-
chunkCount: extractions.length,
|
|
342
|
-
truncationCount,
|
|
343
|
-
truncatedChunks,
|
|
344
|
-
filteredGenericEntities,
|
|
345
|
-
filteredInvalidRelations,
|
|
346
|
-
filteredLowConfidenceRelations,
|
|
347
|
-
};
|
|
348
|
-
}
|
|
349
|
-
function parseGraphExtraction(raw) {
|
|
350
|
-
const empty = (reason = "no_graph_content") => ({
|
|
351
|
-
entities: [],
|
|
352
|
-
relations: [],
|
|
353
|
-
status: reason === "llm_error" || reason === "invalid_json" || reason === "context_limit" ? "failed" : "empty",
|
|
354
|
-
reason,
|
|
355
|
-
});
|
|
356
|
-
if (typeof raw !== "object" || raw === null || Array.isArray(raw))
|
|
357
|
-
return empty();
|
|
358
|
-
const item = raw;
|
|
359
|
-
const extractionConfidence = parseConfidence(item.confidence);
|
|
360
|
-
const entityCanonical = new Map();
|
|
361
|
-
let filteredGenericEntities = 0;
|
|
362
|
-
if (Array.isArray(item.entities)) {
|
|
363
|
-
for (const value of item.entities) {
|
|
364
|
-
if (typeof value !== "string")
|
|
365
|
-
continue;
|
|
366
|
-
const normalized = normalizeEntityName(value);
|
|
367
|
-
if (!normalized)
|
|
368
|
-
continue;
|
|
369
|
-
const normalizedKey = normalized.toLowerCase();
|
|
370
|
-
// Drop generic/empty entities AND raw file/dir paths (anything with a
|
|
371
|
-
// path separator) — the prompt no longer asks for them and isJunkEntity
|
|
372
|
-
// discards them downstream, so emitting them is pure waste/junk (#632).
|
|
373
|
-
if (!/[a-z0-9]/i.test(normalized) ||
|
|
374
|
-
GENERIC_ENTITIES.has(normalizedKey) ||
|
|
375
|
-
normalized.includes("/") ||
|
|
376
|
-
normalized.includes("\\")) {
|
|
377
|
-
filteredGenericEntities += 1;
|
|
378
|
-
continue;
|
|
379
|
-
}
|
|
380
|
-
const key = normalized.toLowerCase();
|
|
381
|
-
if (!entityCanonical.has(key))
|
|
382
|
-
entityCanonical.set(key, normalized);
|
|
383
|
-
if (entityCanonical.size >= MAX_ENTITIES_PER_ASSET)
|
|
384
|
-
break;
|
|
385
|
-
}
|
|
386
|
-
}
|
|
387
|
-
const entities = Array.from(entityCanonical.values());
|
|
388
|
-
const relations = [];
|
|
389
|
-
let filteredInvalidRelations = 0;
|
|
390
|
-
let filteredLowConfidenceRelations = 0;
|
|
391
|
-
if (Array.isArray(item.relations)) {
|
|
392
|
-
for (const relation of item.relations) {
|
|
393
|
-
// Compact triple form `[from, type, to]` (R12) — `type` may be "".
|
|
394
|
-
// Legacy `{"from","to","type","confidence"}` object form still parses
|
|
395
|
-
// for backward compatibility (older cached prompts, other callers).
|
|
396
|
-
let fromRaw;
|
|
397
|
-
let toRaw;
|
|
398
|
-
let typeRaw;
|
|
399
|
-
let confidenceRaw;
|
|
400
|
-
if (Array.isArray(relation)) {
|
|
401
|
-
if (relation.length !== 3 ||
|
|
402
|
-
typeof relation[0] !== "string" ||
|
|
403
|
-
typeof relation[1] !== "string" ||
|
|
404
|
-
typeof relation[2] !== "string") {
|
|
405
|
-
filteredInvalidRelations += 1;
|
|
406
|
-
continue;
|
|
407
|
-
}
|
|
408
|
-
fromRaw = normalizeEntityName(relation[0]);
|
|
409
|
-
typeRaw = relation[1];
|
|
410
|
-
toRaw = normalizeEntityName(relation[2]);
|
|
411
|
-
}
|
|
412
|
-
else if (typeof relation === "object" && relation !== null) {
|
|
413
|
-
const rel = relation;
|
|
414
|
-
fromRaw = typeof rel.from === "string" ? normalizeEntityName(rel.from) : "";
|
|
415
|
-
toRaw = typeof rel.to === "string" ? normalizeEntityName(rel.to) : "";
|
|
416
|
-
typeRaw = typeof rel.type === "string" ? rel.type : undefined;
|
|
417
|
-
confidenceRaw = rel.confidence;
|
|
418
|
-
}
|
|
419
|
-
else {
|
|
420
|
-
filteredInvalidRelations += 1;
|
|
421
|
-
continue;
|
|
422
|
-
}
|
|
423
|
-
if (!fromRaw || !toRaw) {
|
|
424
|
-
filteredInvalidRelations += 1;
|
|
425
|
-
continue;
|
|
426
|
-
}
|
|
427
|
-
const from = entityCanonical.get(fromRaw.toLowerCase());
|
|
428
|
-
const to = entityCanonical.get(toRaw.toLowerCase());
|
|
429
|
-
if (!from || !to || from.toLowerCase() === to.toLowerCase()) {
|
|
430
|
-
filteredInvalidRelations += 1;
|
|
431
|
-
continue;
|
|
432
|
-
}
|
|
433
|
-
const type = typeRaw !== undefined ? normalizeRelationType(typeRaw) : undefined;
|
|
434
|
-
if (type !== undefined && GENERIC_RELATION_TYPES.has(type)) {
|
|
435
|
-
filteredInvalidRelations += 1;
|
|
436
|
-
continue;
|
|
437
|
-
}
|
|
438
|
-
const confidence = parseConfidence(confidenceRaw);
|
|
439
|
-
if (confidence !== undefined && confidence < MIN_RELATION_CONFIDENCE) {
|
|
440
|
-
filteredLowConfidenceRelations += 1;
|
|
441
|
-
continue;
|
|
442
|
-
}
|
|
443
|
-
relations.push({
|
|
444
|
-
from,
|
|
445
|
-
to,
|
|
446
|
-
...(type ? { type } : {}),
|
|
447
|
-
...(confidence !== undefined ? { confidence } : {}),
|
|
448
|
-
});
|
|
449
|
-
if (relations.length >= MAX_RELATIONS_PER_ASSET)
|
|
450
|
-
break;
|
|
451
|
-
}
|
|
452
|
-
}
|
|
453
|
-
const confidence = extractionConfidence;
|
|
454
|
-
const status = entities.length > 0 ? "extracted" : "empty";
|
|
455
|
-
const reason = entities.length > 0 ? "none" : filteredGenericEntities > 0 ? "generic_entities_only" : "no_graph_content";
|
|
456
|
-
return {
|
|
457
|
-
entities,
|
|
458
|
-
relations,
|
|
459
|
-
status,
|
|
460
|
-
reason,
|
|
461
|
-
filteredGenericEntities,
|
|
462
|
-
filteredInvalidRelations,
|
|
463
|
-
filteredLowConfidenceRelations,
|
|
464
|
-
...(confidence !== undefined ? { confidence } : {}),
|
|
465
|
-
};
|
|
466
|
-
}
|
|
467
|
-
/**
|
|
468
|
-
* Build the system prompt for a batched graph-extraction call.
|
|
469
|
-
*
|
|
470
|
-
* The prompt instructs the model to return a JSON array of exactly `count`
|
|
471
|
-
* objects, one per asset, in input order. Index alignment is the critical
|
|
472
|
-
* invariant — if the model drops an asset it still must emit an empty
|
|
473
|
-
* placeholder `{"entities":[],"relations":[]}` at that position.
|
|
474
|
-
*
|
|
475
|
-
* Worked example (3 assets, abbreviated):
|
|
476
|
-
*
|
|
477
|
-
* Input user message:
|
|
478
|
-
* Extract entities and relations from the N=3 assets below.
|
|
479
|
-
* ...rules...
|
|
480
|
-
* === ASSET 1 ===
|
|
481
|
-
* ServiceA integrates with ServiceB.
|
|
482
|
-
* === ASSET 2 ===
|
|
483
|
-
* Terraform provisions the Prod cluster.
|
|
484
|
-
* === ASSET 3 ===
|
|
485
|
-
* No extractable graph content here.
|
|
486
|
-
*
|
|
487
|
-
* Expected model output (valid JSON array, no prose):
|
|
488
|
-
* [
|
|
489
|
-
* {"entities":["ServiceA","ServiceB"],"relations":[["ServiceA","integrates with","ServiceB"]]},
|
|
490
|
-
* {"entities":["Terraform","Prod cluster"],"relations":[["Terraform","provisions","Prod cluster"]]},
|
|
491
|
-
* {"entities":[],"relations":[]}
|
|
492
|
-
* ]
|
|
493
|
-
*
|
|
494
|
-
* If the model returns fewer than 3 items (partial failure), the caller
|
|
495
|
-
* (`extractGraphFromBodies`) falls back to individual calls for missing indices.
|
|
496
|
-
*/
|
|
497
|
-
function buildBatchSystemPrompt() {
|
|
498
|
-
return ("You extract knowledge graphs from developer notes. " +
|
|
499
|
-
"Return ONLY a valid JSON array — no prose, no markdown fences, no preamble. " +
|
|
500
|
-
"Each element of the array corresponds to one input asset, in order. " +
|
|
501
|
-
"The array length MUST equal the number of assets provided. " +
|
|
502
|
-
'Use {"entities":[],"relations":[]} for assets with no extractable graph content.');
|
|
503
|
-
}
|
|
504
|
-
/**
|
|
505
|
-
* Hardened system prompt for the single batch retry (#635). Used only after a
|
|
506
|
-
* first response failed array salvage — leans harder on "raw array only" so a
|
|
507
|
-
* model that wrapped the array in prose/fences corrects itself before we pay
|
|
508
|
-
* the per-asset fallback.
|
|
509
|
-
*/
|
|
510
|
-
function buildBatchRetrySystemPrompt() {
|
|
511
|
-
return (`${buildBatchSystemPrompt()} ` +
|
|
512
|
-
"Your previous response could NOT be parsed as a JSON array. " +
|
|
513
|
-
"Respond with ONLY the raw JSON array — start with '[' and end with ']'. " +
|
|
514
|
-
"No prose, no explanation, no markdown code fences, no preamble.");
|
|
515
|
-
}
|
|
516
|
-
function buildBatchUserPrompt(bodies) {
|
|
517
|
-
const count = bodies.length;
|
|
518
|
-
const assetBlocks = bodies.map((body, i) => `${BATCH_ASSET_SEPARATOR} ${i + 1} ===\n${body.trim()}`).join("\n\n");
|
|
519
|
-
return (`Extract entities and relations from the N=${count} assets below.\n\n` +
|
|
520
|
-
`Rules:\n` +
|
|
521
|
-
`- Output ONLY a JSON array of exactly ${count} objects, one per asset, preserving input order.\n` +
|
|
522
|
-
`- Each object: {"entities": ["Entity One", ...], "relations": [["A", "uses", "B"], ...]}\n` +
|
|
523
|
-
`- Entities are short, canonical noun phrases (project names, services, tools, people, file/dir names, technical concepts).\n` +
|
|
524
|
-
`- Each relation is a 3-element array: [from, type, to]. Relations connect two entities that both appear in that asset's entities array.\n` +
|
|
525
|
-
`- "type" is a short verb phrase (e.g. "uses", "depends on", "owns"). Use "" when unsure.\n` +
|
|
526
|
-
`- Drop pleasantries, meta-commentary, and timestamps.\n` +
|
|
527
|
-
`- Limit to at most ${MAX_ENTITIES_PER_ASSET} entities and ${MAX_RELATIONS_PER_ASSET} relations per asset.\n` +
|
|
528
|
-
`- Use {"entities":[],"relations":[]} for assets with no extractable graph content.\n` +
|
|
529
|
-
`- The array MUST have exactly ${count} elements — one placeholder per asset even if empty.\n\n` +
|
|
530
|
-
assetBlocks);
|
|
531
|
-
}
|
|
532
|
-
function formatContextHint(llmRunner) {
|
|
533
|
-
return llmRunner.connection.contextLength ? `, configured contextLength=${llmRunner.connection.contextLength}` : "";
|
|
534
|
-
}
|
|
535
|
-
/** Dispatch one raw graph prompt through the common resolved-request adapter. */
|
|
536
|
-
async function callGraphLlm(runner, messages, request, onNotices) {
|
|
537
|
-
return callStructured({
|
|
538
|
-
feature: "graph_extraction",
|
|
539
|
-
runner,
|
|
540
|
-
messages,
|
|
541
|
-
request,
|
|
542
|
-
onNotices,
|
|
543
|
-
parse: (raw) => raw ?? "",
|
|
544
|
-
onError: (_cls, error) => {
|
|
545
|
-
throw error;
|
|
546
|
-
},
|
|
547
|
-
fallback: "",
|
|
548
|
-
});
|
|
549
|
-
}
|
|
550
|
-
/**
|
|
551
|
-
* Parse and validate a single item from the batch response array.
|
|
552
|
-
* Mirrors the validation logic in `extractGraphFromBody`.
|
|
553
|
-
*/
|
|
554
|
-
function parseBatchItem(raw) {
|
|
555
|
-
return parseGraphExtraction(raw);
|
|
556
|
-
}
|
|
557
|
-
function applySuccessfulBatchResults(results, batchResult, nonEmptyBodies, nonEmptyIndices, batchState, telemetry) {
|
|
558
|
-
if (batchState)
|
|
559
|
-
batchState.nonArrayBatchFailures = 0;
|
|
560
|
-
if (batchResult.length > nonEmptyBodies.length) {
|
|
561
|
-
warn(`graph extraction (batch): response had ${batchResult.length} items for ${nonEmptyBodies.length} assets; ` +
|
|
562
|
-
`ignoring ${batchResult.length - nonEmptyBodies.length} extra item(s).`);
|
|
563
|
-
}
|
|
564
|
-
for (let j = 0; j < nonEmptyBodies.length; j++) {
|
|
565
|
-
const originalIndex = nonEmptyIndices[j];
|
|
566
|
-
if (originalIndex === undefined)
|
|
567
|
-
continue;
|
|
568
|
-
if (j >= batchResult.length)
|
|
569
|
-
continue;
|
|
570
|
-
const extraction = parseBatchItem(batchResult[j]);
|
|
571
|
-
bumpFilterTelemetry(telemetry, extraction);
|
|
572
|
-
results[originalIndex] = extraction;
|
|
573
|
-
}
|
|
574
|
-
}
|
|
575
|
-
/**
|
|
576
|
-
* Extract entities and relations from multiple asset bodies in a single LLM
|
|
577
|
-
* call (batched graph extraction).
|
|
578
|
-
*
|
|
579
|
-
* Sends all `bodies` as a single prompt with `=== ASSET N ===` separators
|
|
580
|
-
* and expects a JSON array where element `i` corresponds to `bodies[i]`.
|
|
581
|
-
*
|
|
582
|
-
* **Partial-failure handling**: if the model returns fewer elements than
|
|
583
|
-
* `bodies.length`, missing indices are filled by falling back to individual
|
|
584
|
-
* `extractGraphFromBody` calls — ensuring every input always has a result.
|
|
585
|
-
*
|
|
586
|
-
* Returns an array of the same length as `bodies` (never shorter). A body
|
|
587
|
-
* whose extraction failed gets no entities and `status: "failed"`.
|
|
588
|
-
*
|
|
589
|
-
* Routes through `tryLlmFeature("graph_extraction", ...)` so the feature gate
|
|
590
|
-
* and onFallback hook are honoured uniformly.
|
|
591
|
-
*
|
|
592
|
-
* @param llmRunner - Symbolic LLM runner selected through shared execution lowering.
|
|
593
|
-
* @param bodies - Asset body strings to process in one batch.
|
|
594
|
-
* @param signal - Optional AbortSignal for cancellation.
|
|
595
|
-
* @param akmConfig - Full AKM config (for feature-gate checks).
|
|
596
|
-
* @param onFallback - Optional fallback event sink.
|
|
597
|
-
*/
|
|
598
|
-
export async function extractGraphFromBodies(llmRunner, bodies, signal, akmConfig, onFallback, options = {}) {
|
|
599
|
-
const empty = () => ({ entities: [], relations: [] });
|
|
600
|
-
const batchState = normalizeBatchState(options.batchState);
|
|
601
|
-
// Degenerate case: no bodies → empty array (not an error).
|
|
602
|
-
if (bodies.length === 0)
|
|
603
|
-
return [];
|
|
604
|
-
// Single body: delegate to the single-asset path for identical behaviour.
|
|
605
|
-
if (bodies.length === 1) {
|
|
606
|
-
const result = await extractGraphFromBody(llmRunner, bodies[0] ?? "", signal, akmConfig, onFallback, options);
|
|
607
|
-
return [result];
|
|
608
|
-
}
|
|
609
|
-
const concurrency = llmRunner.connection.concurrency ?? 1;
|
|
610
|
-
const extractOne = (body) => extractGraphFromBody(llmRunner, body, signal, akmConfig, onFallback, options);
|
|
611
|
-
// With batching disabled every body takes the single-asset path, once.
|
|
612
|
-
if (batchState?.batchingDisabled)
|
|
613
|
-
return mapWithConcurrency(bodies, concurrency, extractOne);
|
|
614
|
-
// Filter out bodies that are empty so we don't waste tokens, but keep
|
|
615
|
-
// index correspondence by tracking which indices were non-empty.
|
|
616
|
-
const results = bodies.map(empty);
|
|
617
|
-
const nonEmptyIndices = [];
|
|
618
|
-
const nonEmptyBodies = [];
|
|
619
|
-
const oversizedIndices = [];
|
|
620
|
-
for (let i = 0; i < bodies.length; i++) {
|
|
621
|
-
const trimmed = (bodies[i] ?? "").trim();
|
|
622
|
-
if (trimmed) {
|
|
623
|
-
if (trimmed.length > MAX_BATCH_BODY_CHARS) {
|
|
624
|
-
oversizedIndices.push(i);
|
|
625
|
-
}
|
|
626
|
-
else {
|
|
627
|
-
nonEmptyIndices.push(i);
|
|
628
|
-
nonEmptyBodies.push(trimmed);
|
|
629
|
-
}
|
|
630
|
-
}
|
|
631
|
-
}
|
|
632
|
-
await mapWithConcurrency(oversizedIndices, concurrency, async (index) => {
|
|
633
|
-
results[index] = await extractOne(bodies[index] ?? "");
|
|
634
|
-
});
|
|
635
|
-
if (nonEmptyBodies.length === 0)
|
|
636
|
-
return results;
|
|
637
|
-
const systemPrompt = buildBatchSystemPrompt();
|
|
638
|
-
const userPrompt = buildBatchUserPrompt(nonEmptyBodies);
|
|
639
|
-
// Same responseSchema lift as extractGraphFromBody (R12b), scoped to this
|
|
640
|
-
// batch's asset count so a compliant provider bounds every element's
|
|
641
|
-
// entities/relations by the same maxItems as the single-asset path.
|
|
642
|
-
const batchResponseSchema = buildBatchResponseSchema(nonEmptyBodies.length);
|
|
643
|
-
let batchContextError = false;
|
|
644
|
-
// R2: a dead/erroring provider must not be hammered with a per-asset
|
|
645
|
-
// fallback retry for every body in the batch — that is what turned one
|
|
646
|
-
// outage into 15,453 additional retry attempts. `isTransportFailure`
|
|
647
|
-
// (shared with client.ts's `isRetryable` — see its definition) covers
|
|
648
|
-
// `provider_error`, `network_error`, and `provider_html_error`: the
|
|
649
|
-
// provider itself is failing (a dead endpoint more often raises
|
|
650
|
-
// `network_error` or `provider_html_error` than a plain 5xx), not that this
|
|
651
|
-
// particular response was malformed; skip the fallback and record every
|
|
652
|
-
// asset as failed instead. A batch that got no answer at all (the gate's
|
|
653
|
-
// timeout, or a closed gate) is handled the same way.
|
|
654
|
-
let batchProviderError = false;
|
|
655
|
-
const batchOutcome = await tryLlmFeature("graph_extraction", akmConfig, async () => {
|
|
656
|
-
try {
|
|
657
|
-
const raw = await callGraphLlm(llmRunner, [
|
|
658
|
-
{ role: "system", content: systemPrompt },
|
|
659
|
-
{ role: "user", content: userPrompt },
|
|
660
|
-
], {
|
|
661
|
-
temperature: 0.1,
|
|
662
|
-
timeoutMs: llmRunner.timeoutMs,
|
|
663
|
-
signal,
|
|
664
|
-
responseSchema: batchResponseSchema,
|
|
665
|
-
onRetryAttempt: () => bumpTelemetry(options.telemetry, "retryAttempts"),
|
|
666
|
-
}, options.onNotices);
|
|
667
|
-
if (!raw)
|
|
668
|
-
return { kind: "value", value: null };
|
|
669
|
-
// Array-preferring salvage (#635): the batch contract is a top-level
|
|
670
|
-
// JSON array. A leading/example `{…}` object in the response must not
|
|
671
|
-
// mask a valid `[…]` array as a false "non-array" failure.
|
|
672
|
-
let parsed = parseEmbeddedJsonResponse(raw, { expect: "array" });
|
|
673
|
-
if (!Array.isArray(parsed)) {
|
|
674
|
-
// One stricter-reprompt retry before paying the per-asset fallback
|
|
675
|
-
// (#635). Many genuine non-array responses recover when the model is
|
|
676
|
-
// told explicitly to emit only the raw array.
|
|
677
|
-
bumpTelemetry(options.telemetry, "retryAttempts");
|
|
678
|
-
const retryRaw = await callGraphLlm(llmRunner, [
|
|
679
|
-
{ role: "system", content: buildBatchRetrySystemPrompt() },
|
|
680
|
-
{ role: "user", content: userPrompt },
|
|
681
|
-
], {
|
|
682
|
-
temperature: 0,
|
|
683
|
-
timeoutMs: llmRunner.timeoutMs,
|
|
684
|
-
signal,
|
|
685
|
-
responseSchema: batchResponseSchema,
|
|
686
|
-
}, options.onNotices);
|
|
687
|
-
parsed = retryRaw ? parseEmbeddedJsonResponse(retryRaw, { expect: "array" }) : undefined;
|
|
688
|
-
}
|
|
689
|
-
if (!Array.isArray(parsed)) {
|
|
690
|
-
bumpTelemetry(options.telemetry, "nonArrayBatchFailures");
|
|
691
|
-
if (batchState) {
|
|
692
|
-
batchState.nonArrayBatchFailures += 1;
|
|
693
|
-
if (batchState.nonArrayBatchFailures >= NON_ARRAY_BATCH_DISABLE_THRESHOLD) {
|
|
694
|
-
batchState.batchingDisabled = true;
|
|
695
|
-
}
|
|
696
|
-
}
|
|
697
|
-
warn(`graph extraction (batch): LLM response was not a JSON array for ${nonEmptyBodies.length} asset(s) ` +
|
|
698
|
-
`even after a stricter retry; will fall back per-asset. ` +
|
|
699
|
-
`promptChars=${userPrompt.length}${formatContextHint(llmRunner)}`);
|
|
700
|
-
return { kind: "value", value: null };
|
|
701
|
-
}
|
|
702
|
-
return { kind: "value", value: parsed };
|
|
703
|
-
}
|
|
704
|
-
catch (err) {
|
|
705
|
-
if (err instanceof ConfigError)
|
|
706
|
-
return { kind: "config-error", error: err };
|
|
707
|
-
const errMsg = toErrorMessage(err);
|
|
708
|
-
if (isContextSizeError(errMsg)) {
|
|
709
|
-
batchContextError = true;
|
|
710
|
-
bumpTelemetry(options.telemetry, "contextBatchRetries");
|
|
711
|
-
warn(`graph extraction (batch): context size exceeded for ${nonEmptyBodies.length} asset(s); ` +
|
|
712
|
-
`skipping batch. promptChars=${userPrompt.length}${formatContextHint(llmRunner)}`);
|
|
713
|
-
}
|
|
714
|
-
else if (err instanceof LlmCallError && isTransportFailure(err)) {
|
|
715
|
-
batchProviderError = true;
|
|
716
|
-
bumpTelemetry(options.telemetry, "failureCount", nonEmptyBodies.length);
|
|
717
|
-
warn(`graph extraction (batch): provider error (${err.code}) for ${nonEmptyBodies.length} asset(s); ` +
|
|
718
|
-
`skipping per-asset fallback retries. promptChars=${userPrompt.length}${formatContextHint(llmRunner)}: ${errMsg}`);
|
|
719
|
-
}
|
|
720
|
-
else {
|
|
721
|
-
warn(`graph extraction (batch) failed for ${nonEmptyBodies.length} asset(s); ` +
|
|
722
|
-
`promptChars=${userPrompt.length}${formatContextHint(llmRunner)}: ${errMsg}`);
|
|
723
|
-
}
|
|
724
|
-
return { kind: "value", value: null };
|
|
725
|
-
}
|
|
726
|
-
}, { kind: "no-answer" }, {
|
|
727
|
-
timeoutMs: llmRunner.timeoutMs,
|
|
728
|
-
onFallback,
|
|
729
|
-
});
|
|
730
|
-
if (batchOutcome.kind === "config-error")
|
|
731
|
-
throw batchOutcome.error;
|
|
732
|
-
if (batchOutcome.kind === "no-answer") {
|
|
733
|
-
batchProviderError = true;
|
|
734
|
-
bumpTelemetry(options.telemetry, "failureCount", nonEmptyBodies.length);
|
|
735
|
-
}
|
|
736
|
-
const batchResult = batchOutcome.kind === "value" ? batchOutcome.value : null;
|
|
737
|
-
// Map successful batch results back to their original indices.
|
|
738
|
-
if (batchResult !== null) {
|
|
739
|
-
applySuccessfulBatchResults(results, batchResult, nonEmptyBodies, nonEmptyIndices, batchState, options.telemetry);
|
|
740
|
-
}
|
|
741
|
-
else if (batchProviderError) {
|
|
742
|
-
// No per-asset fallback against a failing provider — record every asset
|
|
743
|
-
// in this batch as a genuine failure so it is neither silently empty nor
|
|
744
|
-
// retried again below.
|
|
745
|
-
for (const origIdx of nonEmptyIndices)
|
|
746
|
-
results[origIdx] = { entities: [], relations: [], status: "failed", reason: "llm_error" };
|
|
747
|
-
}
|
|
748
|
-
if (batchContextError && nonEmptyBodies.length > 1) {
|
|
749
|
-
const splitAt = Math.ceil(nonEmptyBodies.length / 2);
|
|
750
|
-
const left = await extractGraphFromBodies(llmRunner, nonEmptyBodies.slice(0, splitAt), signal, akmConfig, onFallback, options);
|
|
751
|
-
const right = await extractGraphFromBodies(llmRunner, nonEmptyBodies.slice(splitAt), signal, akmConfig, onFallback, options);
|
|
752
|
-
const combined = [...left, ...right];
|
|
753
|
-
for (let j = 0; j < nonEmptyIndices.length; j++) {
|
|
754
|
-
const origIdx = nonEmptyIndices[j];
|
|
755
|
-
if (origIdx === undefined)
|
|
756
|
-
continue;
|
|
757
|
-
results[origIdx] = combined[j] ?? empty();
|
|
758
|
-
}
|
|
759
|
-
return results;
|
|
760
|
-
}
|
|
761
|
-
// Partial-failure fallback: any non-empty body whose result is still the
|
|
762
|
-
// empty placeholder (either because batchResult was null or the array was
|
|
763
|
-
// shorter than expected) gets an individual retry — unless the batch failed
|
|
764
|
-
// due to context size, in which case individual calls would also fail.
|
|
765
|
-
const fallbackIndices = nonEmptyIndices.filter((_origIdx, j) => {
|
|
766
|
-
if (batchContextError)
|
|
767
|
-
return false; // skip individual retries on context error
|
|
768
|
-
if (batchProviderError)
|
|
769
|
-
return false; // skip individual retries against a failing provider
|
|
770
|
-
// Result is still empty → needs a fallback call.
|
|
771
|
-
if (batchResult === null)
|
|
772
|
-
return true;
|
|
773
|
-
// batchResult was shorter than the number of non-empty bodies.
|
|
774
|
-
return j >= batchResult.length;
|
|
775
|
-
});
|
|
776
|
-
if (fallbackIndices.length > 0) {
|
|
777
|
-
if (batchResult !== null) {
|
|
778
|
-
// Only warn on partial failure (not when the whole batch failed, which
|
|
779
|
-
// already emitted a warn above).
|
|
780
|
-
warn(`graph extraction (batch): response had ${batchResult.length} items for ${nonEmptyBodies.length} assets; ` +
|
|
781
|
-
`falling back to individual calls for ${fallbackIndices.length} missing asset(s).`);
|
|
782
|
-
}
|
|
783
|
-
await mapWithConcurrency(fallbackIndices, concurrency, async (origIdx) => {
|
|
784
|
-
results[origIdx] = await extractOne(bodies[origIdx] ?? "");
|
|
785
|
-
});
|
|
786
|
-
}
|
|
787
|
-
else if (batchContextError) {
|
|
788
|
-
warn(`graph extraction (batch): skipped ${nonEmptyBodies.length} asset(s) due to context size error; ` +
|
|
789
|
-
`consider increasing llm.contextLength or reducing index.graph.graphExtractionBatchSize to 1.`);
|
|
790
|
-
}
|
|
791
|
-
return results;
|
|
792
|
-
}
|
|
793
|
-
/**
|
|
794
|
-
* Extract entities and relations from a single asset body via the configured LLM.
|
|
795
|
-
*
|
|
796
|
-
* Any failure (timeout, invalid JSON, empty response, provider error) returns
|
|
797
|
-
* no entities with `status: "failed"`, which is never cached, so the next run
|
|
798
|
-
* retries it. Errors are logged via `warn()` but never thrown — a failed
|
|
799
|
-
* extraction for one asset must not abort the rest of the index pass.
|
|
800
|
-
*
|
|
801
|
-
* Routes through `tryLlmFeature("graph_extraction", ...)` so the feature gate
|
|
802
|
-
* and onFallback hook are honoured uniformly (Fix C5).
|
|
803
|
-
*/
|
|
804
|
-
export async function extractGraphFromBody(llmRunner, body, signal, akmConfig, onFallback, options = {}) {
|
|
805
|
-
const empty = (reason, status) => ({
|
|
806
|
-
entities: [],
|
|
807
|
-
relations: [],
|
|
808
|
-
...(status ? { status } : {}),
|
|
809
|
-
...(reason ? { reason } : {}),
|
|
810
|
-
});
|
|
811
|
-
const trimmedBody = body.trim();
|
|
812
|
-
if (!trimmedBody)
|
|
813
|
-
return empty();
|
|
814
|
-
const chunked = splitBodyIntoChunks(trimmedBody, MAX_CHUNK_BODY_CHARS);
|
|
815
|
-
if (chunked.truncationCount > 0) {
|
|
816
|
-
bumpTelemetry(options.telemetry, "truncationCount", chunked.truncationCount);
|
|
817
|
-
warnVerbose(`graph extraction: split a long asset into ${chunked.chunks.length} chunk(s) with ${chunked.truncationCount} hard split(s).`);
|
|
818
|
-
}
|
|
819
|
-
// R12b + R20: bound per-asset cost by capping how many chunks of a long
|
|
820
|
-
// asset are ever sent to the LLM. Excess chunks are dropped, never
|
|
821
|
-
// processed — the coverage loss is recorded as truncatedChunks rather than
|
|
822
|
-
// silently absorbed.
|
|
823
|
-
const maxChunksPerAsset = options.maxChunksPerAsset ?? DEFAULT_MAX_CHUNKS_PER_ASSET;
|
|
824
|
-
const cappedChunks = chunked.chunks.length > maxChunksPerAsset ? chunked.chunks.slice(0, maxChunksPerAsset) : chunked.chunks;
|
|
825
|
-
const truncatedChunkCount = chunked.chunks.length - cappedChunks.length;
|
|
826
|
-
if (truncatedChunkCount > 0) {
|
|
827
|
-
bumpTelemetry(options.telemetry, "truncatedChunks", truncatedChunkCount);
|
|
828
|
-
warnVerbose(`graph extraction: capped a long asset to ${cappedChunks.length} of ${chunked.chunks.length} chunk(s) ` +
|
|
829
|
-
`(maxChunksPerAsset=${maxChunksPerAsset}); ${truncatedChunkCount} chunk(s) not processed.`);
|
|
830
|
-
}
|
|
831
|
-
if (cappedChunks.length > 1) {
|
|
832
|
-
const chunkResults = [];
|
|
833
|
-
for (const chunk of cappedChunks) {
|
|
834
|
-
chunkResults.push(await extractGraphFromBody(llmRunner, chunk, signal, akmConfig, onFallback, options));
|
|
835
|
-
}
|
|
836
|
-
const merged = mergeGraphExtractions(chunkResults);
|
|
837
|
-
merged.truncationCount = (merged.truncationCount ?? 0) + chunked.truncationCount;
|
|
838
|
-
merged.truncatedChunks = (merged.truncatedChunks ?? 0) + truncatedChunkCount;
|
|
839
|
-
return merged;
|
|
840
|
-
}
|
|
841
|
-
// When capped down to exactly one chunk from a body that originally split
|
|
842
|
-
// into more, that single surviving chunk (not the full trimmedBody) is what
|
|
843
|
-
// must be sent — otherwise the cap would have no effect on prompt size.
|
|
844
|
-
const bodyForCall = truncatedChunkCount > 0 ? (cappedChunks[0] ?? trimmedBody) : trimmedBody;
|
|
845
|
-
const userPrompt = `${USER_PROMPT_PREFIX}${bodyForCall}`;
|
|
846
|
-
const result = await callStructured({
|
|
847
|
-
feature: "graph_extraction",
|
|
848
|
-
akmConfig,
|
|
849
|
-
runner: llmRunner,
|
|
850
|
-
messages: [
|
|
851
|
-
{ role: "system", content: SYSTEM_PROMPT },
|
|
852
|
-
{ role: "user", content: userPrompt },
|
|
853
|
-
],
|
|
854
|
-
request: {
|
|
855
|
-
temperature: 0.1,
|
|
856
|
-
timeoutMs: llmRunner.timeoutMs,
|
|
857
|
-
signal,
|
|
858
|
-
responseSchema: GRAPH_EXTRACTION_JSON_SCHEMA,
|
|
859
|
-
onRetryAttempt: () => bumpTelemetry(options.telemetry, "retryAttempts"),
|
|
860
|
-
},
|
|
861
|
-
onNotices: options.onNotices,
|
|
862
|
-
parse: (raw) => {
|
|
863
|
-
// An empty response is not JSON either; "nothing to extract" is `{"entities": []}`.
|
|
864
|
-
const parsed = raw ? parseEmbeddedJsonResponse(raw) : undefined;
|
|
865
|
-
if (!parsed) {
|
|
866
|
-
warn("graph extraction: invalid JSON response from LLM; skipping asset.");
|
|
867
|
-
bumpTelemetry(options.telemetry, "failureCount");
|
|
868
|
-
return empty("invalid_json", "failed");
|
|
869
|
-
}
|
|
870
|
-
const extraction = parseGraphExtraction(parsed);
|
|
871
|
-
bumpFilterTelemetry(options.telemetry, extraction);
|
|
872
|
-
if (extraction.status === "failed")
|
|
873
|
-
bumpTelemetry(options.telemetry, "failureCount");
|
|
874
|
-
return extraction;
|
|
875
|
-
},
|
|
876
|
-
onError: (cls, err) => {
|
|
877
|
-
const errMsg = toErrorMessage(err);
|
|
878
|
-
if (cls === "context_limit") {
|
|
879
|
-
bumpTelemetry(options.telemetry, "failureCount");
|
|
880
|
-
warn(`graph extraction: context size exceeded for asset; promptChars=${userPrompt.length}${formatContextHint(llmRunner)}. ` +
|
|
881
|
-
`Consider increasing llm.contextLength in config.json.`);
|
|
882
|
-
return empty("context_limit", "failed");
|
|
883
|
-
}
|
|
884
|
-
else if (cls === "html") {
|
|
885
|
-
bumpTelemetry(options.telemetry, "htmlErrorCount");
|
|
886
|
-
warn(`graph extraction: provider returned HTML instead of JSON for asset; promptChars=${userPrompt.length}${formatContextHint(llmRunner)}: ${errMsg}`);
|
|
887
|
-
return empty("llm_error", "failed");
|
|
888
|
-
}
|
|
889
|
-
else {
|
|
890
|
-
bumpTelemetry(options.telemetry, "failureCount");
|
|
891
|
-
warn(`graph extraction failed for asset; promptChars=${userPrompt.length}${formatContextHint(llmRunner)}: ${errMsg}`);
|
|
892
|
-
return empty("llm_error", "failed");
|
|
893
|
-
}
|
|
894
|
-
},
|
|
895
|
-
// No answer from the model (the gate's timeout, or a closed gate) is a
|
|
896
|
-
// failure the next run retries, never a cacheable "no entities".
|
|
897
|
-
fallback: empty("llm_error", "failed"),
|
|
898
|
-
onFallback,
|
|
899
|
-
});
|
|
900
|
-
if (truncatedChunkCount > 0)
|
|
901
|
-
result.truncatedChunks = (result.truncatedChunks ?? 0) + truncatedChunkCount;
|
|
902
|
-
return result;
|
|
903
|
-
}
|