akm-cli 0.9.16 → 0.9.17-alpha.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. package/CHANGELOG.md +478 -0
  2. package/dist/assets/prompts/consolidate-system.md +4 -11
  3. package/dist/assets/prompts/graph-extract-user-prompt.md +5 -5
  4. package/dist/commands/health/accept-rate.js +6 -0
  5. package/dist/commands/health/checks.js +54 -0
  6. package/dist/commands/health/improve-metrics.js +1 -5
  7. package/dist/commands/health/report-view-model.js +0 -1
  8. package/dist/commands/health.js +10 -0
  9. package/dist/commands/improve/consolidate/chunking.js +19 -35
  10. package/dist/commands/improve/consolidate/merge.js +6 -9
  11. package/dist/commands/improve/consolidate.js +104 -91
  12. package/dist/commands/improve/distill/promote-memory.js +40 -2
  13. package/dist/commands/improve/distill/quality-gate.js +186 -23
  14. package/dist/commands/improve/distill.js +42 -8
  15. package/dist/commands/improve/eligibility.js +13 -3
  16. package/dist/commands/improve/improve-cli.js +32 -9
  17. package/dist/commands/improve/improve-strategies.js +23 -1
  18. package/dist/commands/improve/improve.js +121 -84
  19. package/dist/commands/improve/loop-stages.js +241 -108
  20. package/dist/commands/improve/preparation.js +50 -17
  21. package/dist/commands/improve/reflect.js +16 -5
  22. package/dist/commands/improve/shared.js +0 -10
  23. package/dist/commands/proposal/drain.js +79 -10
  24. package/dist/commands/proposal/proposal-types.js +21 -0
  25. package/dist/commands/proposal/repository.js +108 -29
  26. package/dist/core/asset/frontmatter.js +106 -1
  27. package/dist/core/config/schema/improve-processes.js +29 -2
  28. package/dist/core/improve-result.js +9 -0
  29. package/dist/core/paths.js +7 -0
  30. package/dist/indexer/ensure-index.js +52 -7
  31. package/dist/indexer/graph/graph-extraction.js +82 -8
  32. package/dist/indexer/passes/memory-inference.js +16 -1
  33. package/dist/llm/client.js +16 -2
  34. package/dist/llm/graph-extract.js +162 -18
  35. package/dist/scripts/akm-migrate-node.js +20 -4
  36. package/dist/scripts/akm-migrate.js +20 -4
  37. package/dist/storage/repositories/index-entries-repository.js +43 -0
  38. package/dist/storage/repositories/proposals-repository.js +4 -1
  39. package/dist/storage/state-db-integrity.js +123 -0
  40. package/dist/workflows/program/schema.js +1 -0
  41. package/docs/reference/cli.md +4 -3
  42. package/docs/reference/data-and-telemetry.md +1 -0
  43. package/package.json +1 -1
  44. package/schemas/akm-config.json +44 -0
  45. package/schemas/akm-workflow.json +1 -0
  46. package/dist/commands/improve/eval-cases.js +0 -52
@@ -128,7 +128,7 @@ function normalizeConfidence(raw) {
128
128
  return undefined;
129
129
  return Math.max(0, Math.min(1, raw));
130
130
  }
131
- function getGraphExtractorId(config) {
131
+ export function getGraphExtractorId(config) {
132
132
  const fingerprint = computeBodyHash(JSON.stringify({
133
133
  promptVersion: graphExtract.GRAPH_EXTRACT_PROMPT_VERSION,
134
134
  model: config.model,
@@ -152,6 +152,36 @@ function buildLowQualityWarnings(quality, telemetry) {
152
152
  }
153
153
  return warnings;
154
154
  }
155
+ /**
156
+ * Failure-rate abort for the extraction run (R2), modelled on consolidate's
157
+ * chunk-level guard (`ABORT_MIN_CHUNKS`/`ABORT_FAILURE_RATE` in
158
+ * consolidate.ts, C-6/#392): rate-based over a minimum sample so a couple of
159
+ * transient per-file failures cannot abort a run that would otherwise
160
+ * recover, while a systemically dead provider stops burning through the
161
+ * rest of the eligible set. The existing "one failure must not abort the
162
+ * rest" behaviour for individual files is untouched — this only stops
163
+ * further model calls once the failure rate itself is the signal.
164
+ */
165
+ const GRAPH_EXTRACTION_ABORT_MIN_ATTEMPTS = 4;
166
+ const GRAPH_EXTRACTION_ABORT_FAILURE_RATE = 0.5;
167
+ /** Records one attempted (non-cache-hit) model call and flips `aborted` once the failure-rate threshold is crossed. */
168
+ function recordGraphExtractionAttempt(state, failed) {
169
+ if (state.aborted)
170
+ return;
171
+ state.attempts += 1;
172
+ if (failed)
173
+ state.failures += 1;
174
+ if (state.attempts < GRAPH_EXTRACTION_ABORT_MIN_ATTEMPTS)
175
+ return;
176
+ const failureRate = state.failures / state.attempts;
177
+ if (failureRate < GRAPH_EXTRACTION_ABORT_FAILURE_RATE)
178
+ return;
179
+ state.aborted = true;
180
+ state.message =
181
+ `graph extraction aborted — failure rate ${(failureRate * 100).toFixed(0)}% over ${state.attempts} ` +
182
+ `attempt(s) (>= ${GRAPH_EXTRACTION_ABORT_FAILURE_RATE * 100}% threshold). LLM may be unavailable.`;
183
+ warn(state.message);
184
+ }
155
185
  export function getGraphExtractionIncludeTypes(config) {
156
186
  const configured = getIndexPassConfig(config.index, "graph")?.graphExtractionIncludeTypes;
157
187
  if (!configured || configured.length === 0)
@@ -200,6 +230,20 @@ function validateGraphCacheShape(raw) {
200
230
  ...(typeof obj.reason === "string" ? { reason: obj.reason } : {}),
201
231
  };
202
232
  }
233
+ /**
234
+ * A `"failed"` extraction (provider error, invalid JSON, context overflow —
235
+ * see {@link GraphExtractionStatus}) must never be reused as a cache hit or
236
+ * re-persisted as one. R2: a dead provider upserted ~30,900 rows shaped
237
+ * `{"entities":[],"relations":[],"status":"failed","reason":"llm_error"}`,
238
+ * and both hit paths (the `llm_enrichment_cache` lookup and `reuseGraphNode`
239
+ * over the previous graph) validated the shape without checking `status`, so
240
+ * 92% of the persisted graph became a permanent hit that never retried. A
241
+ * failed result becomes a miss naturally and is overwritten on the next
242
+ * successful extraction; existing failed rows are left on disk untouched.
243
+ */
244
+ function isFailedExtractionStatus(status) {
245
+ return status === "failed";
246
+ }
203
247
  function loadGraphFile(stashRoot, db) {
204
248
  if (!db)
205
249
  return { files: [] };
@@ -252,6 +296,8 @@ function reuseGraphNode(previousNodes, candidate, bodyHash) {
252
296
  return undefined;
253
297
  if (node.bodyHash !== bodyHash)
254
298
  return undefined;
299
+ if (isFailedExtractionStatus(node.status))
300
+ return undefined;
255
301
  const validated = validateGraphCacheShape({ entities: node.entities, relations: node.relations });
256
302
  if (!validated)
257
303
  return undefined;
@@ -276,8 +322,9 @@ function planEligibleGraphExtractions(args) {
276
322
  if (entry?.bodyHash === bodyHash) {
277
323
  try {
278
324
  const cached = validateGraphCacheShape(JSON.parse(entry.resultJson));
279
- if (cached)
325
+ if (cached && !isFailedExtractionStatus(cached.status)) {
280
326
  return { kind: "cache-hit", candidate, bodyHash, cached, persistCache: false };
327
+ }
281
328
  }
282
329
  catch {
283
330
  // Corrupt cache rows are immutable model plans for this pass.
@@ -305,7 +352,7 @@ function graphRecordFromCachePlan(plan) {
305
352
  };
306
353
  }
307
354
  async function extractGraphBatches(args) {
308
- const { plans, batchSize, signal, db, cacheVariant, telemetry, llmRunner, lease, featureConfig, onFallback, batchState, runtimeTelemetry, onNotices, reportProgress, } = args;
355
+ const { plans, batchSize, signal, db, cacheVariant, telemetry, llmRunner, lease, featureConfig, onFallback, abortState, batchState, runtimeTelemetry, onNotices, reportProgress, maxChunksPerAsset, } = args;
309
356
  const results = new Array(plans.length).fill(undefined);
310
357
  const chunkStarts = [];
311
358
  for (let start = 0; start < plans.length; start += batchSize)
@@ -328,12 +375,12 @@ async function extractGraphBatches(args) {
328
375
  continue;
329
376
  telemetry.cacheHits += 1;
330
377
  results[start + index] = graphRecordFromCachePlan(plan);
331
- if (db && plan.persistCache) {
378
+ if (db && plan.persistCache && !isFailedExtractionStatus(plan.cached.status)) {
332
379
  upsertLlmCacheEntry(db, plan.candidate.absPath, plan.bodyHash, JSON.stringify(plan.cached), cacheVariant);
333
380
  }
334
381
  }
335
382
  const modelPlans = chunk.filter((plan) => plan.kind === "model");
336
- if (modelPlans.length === 0) {
383
+ if (modelPlans.length === 0 || abortState.aborted) {
337
384
  reportChunkProgress();
338
385
  return;
339
386
  }
@@ -345,6 +392,7 @@ async function extractGraphBatches(args) {
345
392
  telemetry: runtimeTelemetry,
346
393
  onNotices,
347
394
  ...(lease ? { lease } : {}),
395
+ ...(maxChunksPerAsset != null ? { maxChunksPerAsset } : {}),
348
396
  });
349
397
  }
350
398
  catch (error) {
@@ -355,6 +403,8 @@ async function extractGraphBatches(args) {
355
403
  throw error;
356
404
  }
357
405
  let llmIndex = 0;
406
+ let dispatchHadResult = false;
407
+ let dispatchAllFailed = true;
358
408
  for (let index = 0; index < chunk.length; index++) {
359
409
  const plan = chunk[index];
360
410
  if (!plan || plan.kind !== "model")
@@ -369,7 +419,10 @@ async function extractGraphBatches(args) {
369
419
  ...(extraction.status ? { status: extraction.status } : {}),
370
420
  ...(extraction.reason ? { reason: extraction.reason } : {}),
371
421
  };
372
- if (db) {
422
+ dispatchHadResult = true;
423
+ if (!isFailedExtractionStatus(cacheShape.status))
424
+ dispatchAllFailed = false;
425
+ if (db && !isFailedExtractionStatus(cacheShape.status)) {
373
426
  upsertLlmCacheEntry(db, plan.candidate.absPath, plan.bodyHash, JSON.stringify(cacheShape), cacheVariant);
374
427
  }
375
428
  results[start + index] = {
@@ -379,6 +432,14 @@ async function extractGraphBatches(args) {
379
432
  ...cacheShape,
380
433
  };
381
434
  }
435
+ // One attempt per `extractGraphFromBodies` dispatch (this chunk's batch
436
+ // call), not one per file it covers — mirrors consolidate.ts's
437
+ // totalChunksProcessed++/totalChunksFailed, which count once per chunk
438
+ // regardless of how many memories are in it. Counting per file let a
439
+ // single batched provider_error satisfy GRAPH_EXTRACTION_ABORT_MIN_ATTEMPTS
440
+ // after one HTTP failure whenever graphExtractionBatchSize >= 4.
441
+ if (dispatchHadResult)
442
+ recordGraphExtractionAttempt(abortState, dispatchAllFailed);
382
443
  reportChunkProgress();
383
444
  }, llmRunner.connection.concurrency ?? 1);
384
445
  return { results, ...(configFailure ? { configFailure } : {}) };
@@ -670,6 +731,7 @@ export async function runGraphExtractionPass(ctx) {
670
731
  batchingDisabled: false,
671
732
  nonArrayBatchFailures: 0,
672
733
  };
734
+ const abortState = { attempts: 0, failures: 0, aborted: false };
673
735
  warnVerbose(`graph extraction: starting for ${considered} eligible file(s) under ${primary.path}; ` +
674
736
  `includeTypes=${includeTypes.join(",")}, batchSize=${batchSize}, concurrency=${llmRunner.connection.concurrency ?? 1}, ` +
675
737
  `reEnrich=${reEnrich === true}, candidateScoped=${options.candidatePaths ? "true" : "false"}.`);
@@ -690,11 +752,15 @@ export async function runGraphExtractionPass(ctx) {
690
752
  if (plan.kind === "cache-hit") {
691
753
  telemetry.cacheHits += 1;
692
754
  cached = plan.cached;
693
- if (db && plan.persistCache) {
755
+ if (db && plan.persistCache && !isFailedExtractionStatus(cached.status)) {
694
756
  upsertLlmCacheEntry(db, candidate.absPath, bodyHash, JSON.stringify(cached), cacheVariant);
695
757
  }
696
758
  }
697
759
  else {
760
+ if (abortState.aborted) {
761
+ reportProgress(candidate.absPath, undefined);
762
+ return undefined;
763
+ }
698
764
  telemetry.cacheMisses += 1;
699
765
  let extraction;
700
766
  try {
@@ -703,6 +769,7 @@ export async function runGraphExtractionPass(ctx) {
703
769
  telemetry: runtimeTelemetry,
704
770
  onNotices,
705
771
  ...(dispatchLease ? { lease: dispatchLease } : {}),
772
+ ...(options.maxChunksPerAsset != null ? { maxChunksPerAsset: options.maxChunksPerAsset } : {}),
706
773
  });
707
774
  }
708
775
  catch (err) {
@@ -719,7 +786,8 @@ export async function runGraphExtractionPass(ctx) {
719
786
  ...(extraction.status ? { status: extraction.status } : {}),
720
787
  ...(extraction.reason ? { reason: extraction.reason } : {}),
721
788
  };
722
- if (db) {
789
+ recordGraphExtractionAttempt(abortState, isFailedExtractionStatus(cached.status));
790
+ if (db && !isFailedExtractionStatus(cached.status)) {
723
791
  upsertLlmCacheEntry(db, candidate.absPath, bodyHash, JSON.stringify(cached), cacheVariant);
724
792
  }
725
793
  }
@@ -754,8 +822,10 @@ export async function runGraphExtractionPass(ctx) {
754
822
  onFallback,
755
823
  batchState,
756
824
  runtimeTelemetry,
825
+ abortState,
757
826
  onNotices,
758
827
  reportProgress,
828
+ ...(options.maxChunksPerAsset != null ? { maxChunksPerAsset: options.maxChunksPerAsset } : {}),
759
829
  });
760
830
  extractionResults = batch.results;
761
831
  configFailure ??= batch.configFailure;
@@ -797,14 +867,18 @@ export async function runGraphExtractionPass(ctx) {
797
867
  const assetRefs = mergedNodes.map((node) => node.path);
798
868
  const deduped = deduplicateGraph(mergedNodes.map((node) => ({ entities: node.entities, relations: node.relations })), assetRefs);
799
869
  telemetry.truncationCount = runtimeTelemetry.truncationCount ?? 0;
870
+ telemetry.truncatedChunks = runtimeTelemetry.truncatedChunks ?? 0;
800
871
  telemetry.failureCount = runtimeTelemetry.failureCount ?? 0;
801
872
  telemetry.htmlErrorCount = runtimeTelemetry.htmlErrorCount ?? 0;
802
873
  telemetry.retryAttempts = runtimeTelemetry.retryAttempts ?? 0;
803
874
  telemetry.nonArrayBatchFailures = runtimeTelemetry.nonArrayBatchFailures ?? 0;
875
+ telemetry.aborted = abortState.aborted;
804
876
  const qualityConsidered = mergedNodes.length;
805
877
  const qualityExtracted = mergedNodes.filter((node) => node.status === "extracted" && node.entities.length > 0).length;
806
878
  const quality = computeGraphQualityTelemetry(qualityConsidered, qualityExtracted, deduped.entities.length, deduped.relations.length);
807
879
  const warnings = buildLowQualityWarnings(quality, telemetry);
880
+ if (abortState.message)
881
+ warnings.push(abortState.message);
808
882
  for (const warning of warnings)
809
883
  warnVerbose(`graph extraction quality: ${warning}`);
810
884
  const graph = {
@@ -46,7 +46,7 @@ import { todayIso } from "../../core/common.js";
46
46
  import { concurrentMap } from "../../core/concurrent.js";
47
47
  import { ConfigError } from "../../core/errors.js";
48
48
  import { warn } from "../../core/warn.js";
49
- import { recordWrittenPath } from "../../core/write-provenance.js";
49
+ import { beginWriteProvenance, recordWrittenPath } from "../../core/write-provenance.js";
50
50
  import { writeAssetToSource } from "../../core/write-source.js";
51
51
  import { disposeLoweredExecutionDispatchLease, } from "../../integrations/agent/execution-lowering.js";
52
52
  import { isProcessEnabled } from "../../llm/feature-gate.js";
@@ -188,6 +188,18 @@ async function inferPendingMemoryRecord(plan, ctx) {
188
188
  * short-circuits to a no-op result.
189
189
  */
190
190
  export async function runMemoryInferencePass(ctx) {
191
+ // R78 (tier1-0917): owns the write-provenance journal end-to-end so it closes on every
192
+ // exit path, including a throw out of the body below — an unclosed journal
193
+ // would keep reporting every later write in this process as this call's own.
194
+ const provenance = beginWriteProvenance();
195
+ try {
196
+ return await runMemoryInferencePassBody(ctx, provenance);
197
+ }
198
+ finally {
199
+ provenance.end();
200
+ }
201
+ }
202
+ async function runMemoryInferencePassBody(ctx, provenance) {
191
203
  const { config, sources, signal, db, reEnrich, onProgress, options = {} } = ctx;
192
204
  const invocationOwnsRunner = Object.hasOwn(ctx, "llmRunner");
193
205
  const compressMemoryToDerivedMemory = options.compressMemoryToDerivedMemory ?? memoryInfer.compressMemoryToDerivedMemory;
@@ -202,6 +214,7 @@ export async function runMemoryInferencePass(ctx) {
202
214
  skippedAborted: 0,
203
215
  unaccounted: 0,
204
216
  htmlErrorCount: 0,
217
+ writtenPaths: [],
205
218
  };
206
219
  // Mutable sink threaded into compressMemoryToDerivedMemory so the per-call
207
220
  // HTML-error categorization (which is otherwise swallowed inside the feature
@@ -215,6 +228,8 @@ export async function runMemoryInferencePass(ctx) {
215
228
  const completeResult = () => {
216
229
  if (noticesByKey.size > 0)
217
230
  result.notices = Object.freeze([...noticesByKey.values()]);
231
+ // Non-destructive read — the wrapper's `finally` owns closing the journal.
232
+ result.writtenPaths = provenance.writtenPaths();
218
233
  return result;
219
234
  };
220
235
  // Gate 1 — feature gate via isProcessEnabled, which reads the 0.8.0 path
@@ -117,12 +117,24 @@ export function isContextSizeError(message) {
117
117
  /exceeded|over.*limit|too.*long/.test(lower);
118
118
  return evidence;
119
119
  }
120
+ /**
121
+ * Codes describing a failure to reach or get a usable response from the
122
+ * provider transport itself, as opposed to a malformed-but-received response
123
+ * (`parse_error`) or a request-shape rejection (`rate_limited`). Shared by
124
+ * {@link isRetryable} (which additionally requires evidence the failure is
125
+ * transient) and the batch graph-extraction storm guard in `graph-extract.ts`
126
+ * (which treats any of these as "the provider is down, stop retrying
127
+ * per-asset") so the two classifications cannot drift apart.
128
+ */
129
+ export function isTransportFailure(err) {
130
+ return err.code === "provider_error" || err.code === "network_error" || err.code === "provider_html_error";
131
+ }
120
132
  /**
121
133
  * Decide whether a first-attempt {@link LlmCallError} is eligible for a single
122
134
  * retry. Retryable: HTTP 5xx (`provider_error` with statusCode >= 500) and
123
135
  * `network_error` whose message looks like a transient connection drop.
124
- * NOT retryable: 4xx, `rate_limited` (429), `timeout`, `parse_error`, and
125
- * context-overflow-classified errors.
136
+ * NOT retryable: 4xx, `rate_limited` (429), `timeout`, `parse_error`,
137
+ * `provider_html_error`, and context-overflow-classified errors.
126
138
  *
127
139
  * The connection-drop heuristic covers the substrings emitted across runtimes
128
140
  * for a mid-flight socket close:
@@ -141,6 +153,8 @@ export function isContextSizeError(message) {
141
153
  function isRetryable(err) {
142
154
  if (isContextSizeError(err.message))
143
155
  return false;
156
+ if (!isTransportFailure(err))
157
+ return false;
144
158
  if (err.code === "provider_error") {
145
159
  return typeof err.statusCode === "number" && err.statusCode >= 500;
146
160
  }
@@ -25,7 +25,7 @@ import { toErrorMessage } from "../core/common.js";
25
25
  import { ConfigError } from "../core/errors.js";
26
26
  import { parseEmbeddedJsonResponse } from "../core/parse.js";
27
27
  import { warn, warnVerbose } from "../core/warn.js";
28
- import { isContextSizeError } from "./client.js";
28
+ import { isContextSizeError, isTransportFailure, LlmCallError } from "./client.js";
29
29
  import { tryLlmFeature } from "./feature-gate.js";
30
30
  import { callStructured } from "./structured-call.js";
31
31
  /**
@@ -33,7 +33,7 @@ import { callStructured } from "./structured-call.js";
33
33
  * Chosen to be visually clear and unlikely to appear verbatim in asset bodies.
34
34
  */
35
35
  const BATCH_ASSET_SEPARATOR = "=== ASSET";
36
- export const GRAPH_EXTRACT_PROMPT_VERSION = "v2";
36
+ export const GRAPH_EXTRACT_PROMPT_VERSION = "v3";
37
37
  /** Asset bodies longer than this are chunked instead of truncated. */
38
38
  const MAX_CHUNK_BODY_CHARS = 1600;
39
39
  /** Bodies longer than this are excluded from multi-asset batch prompts. */
@@ -44,10 +44,72 @@ const NON_ARRAY_BATCH_DISABLE_THRESHOLD = 2;
44
44
  const MAX_ENTITIES_PER_ASSET = 32;
45
45
  /** Hard cap on relations returned per asset. */
46
46
  const MAX_RELATIONS_PER_ASSET = 32;
47
+ /**
48
+ * Default cap on chunks processed per asset (R12b + R20) — overridable via
49
+ * `processes.graphExtraction.maxChunksPerAsset`. Without a cap, one long file
50
+ * chunked at MAX_CHUNK_BODY_CHARS could spend dozens of calls on a single
51
+ * asset (one file spent 21 of 27 run calls this way) before its output was
52
+ * sliced to MAX_ENTITIES_PER_ASSET/MAX_RELATIONS_PER_ASSET anyway.
53
+ */
54
+ const DEFAULT_MAX_CHUNKS_PER_ASSET = 8;
47
55
  const SYSTEM_PROMPT = systemPromptTemplate;
48
56
  const USER_PROMPT_PREFIX = userPromptTemplate
49
57
  .replace("{{MAX_ENTITIES}}", String(MAX_ENTITIES_PER_ASSET))
50
58
  .replace("{{MAX_RELATIONS}}", String(MAX_RELATIONS_PER_ASSET));
59
+ /**
60
+ * Strict JSON Schema for one asset's extraction payload (R12b, compacted for
61
+ * R12). Sent via `responseSchema` to providers that opt into structured
62
+ * output (`runner.connection.supportsJsonSchema` — same lift as
63
+ * memory-infer.ts's `DERIVED_MEMORY_JSON_SCHEMA`); the client silently drops
64
+ * it otherwise. `maxItems` mirrors MAX_ENTITIES_PER_ASSET/MAX_RELATIONS_PER_ASSET
65
+ * so a compliant provider cannot pay for output beyond what
66
+ * parseGraphExtraction keeps. Relations are compact `[from, type, to]`
67
+ * triples (`type` may be `""`) rather than `{"from","to","type"}` objects —
68
+ * the object-keyed form cost 10+ tokens per relation for no signal, and
69
+ * output tokens cost far more than prompt tokens. There is deliberately no
70
+ * relation-level `confidence` in the schema (the prompt never asks for one);
71
+ * `parseGraphExtraction` still reads it from a legacy object-shaped relation
72
+ * for backward compatibility. `confidence` stays at the extraction level —
73
+ * parseGraphExtraction reads it into the merged confidence.
74
+ * `additionalProperties: false` forbids anything else. Reused as the `items`
75
+ * schema of a batch call's array response (see
76
+ * {@link buildBatchResponseSchema}) so a single-asset and a batched call
77
+ * bound entities/relations identically.
78
+ */
79
+ const GRAPH_EXTRACTION_ITEM_SCHEMA = {
80
+ type: "object",
81
+ properties: {
82
+ entities: { type: "array", items: { type: "string" }, maxItems: MAX_ENTITIES_PER_ASSET },
83
+ relations: {
84
+ type: "array",
85
+ maxItems: MAX_RELATIONS_PER_ASSET,
86
+ items: {
87
+ type: "array",
88
+ items: { type: "string" },
89
+ minItems: 3,
90
+ maxItems: 3,
91
+ },
92
+ },
93
+ confidence: { type: "number" },
94
+ },
95
+ required: ["entities", "relations"],
96
+ additionalProperties: false,
97
+ };
98
+ /** Schema for {@link extractGraphFromBody}'s single-asset `responseSchema`. */
99
+ const GRAPH_EXTRACTION_JSON_SCHEMA = GRAPH_EXTRACTION_ITEM_SCHEMA;
100
+ /**
101
+ * Schema for {@link extractGraphFromBodies}' batch `responseSchema` — an
102
+ * array of exactly `count` {@link GRAPH_EXTRACTION_ITEM_SCHEMA} elements, one
103
+ * per asset in the batch, matching the batch prompt's contract.
104
+ */
105
+ function buildBatchResponseSchema(count) {
106
+ return {
107
+ type: "array",
108
+ minItems: count,
109
+ maxItems: count,
110
+ items: GRAPH_EXTRACTION_ITEM_SCHEMA,
111
+ };
112
+ }
51
113
  const GENERIC_ENTITIES = new Set([
52
114
  "agent",
53
115
  "application",
@@ -152,12 +214,14 @@ function mergeGraphExtractions(extractions) {
152
214
  const relationChunkCounts = new Map();
153
215
  let confidence;
154
216
  let truncationCount = 0;
217
+ let truncatedChunks = 0;
155
218
  let filteredGenericEntities = 0;
156
219
  let filteredInvalidRelations = 0;
157
220
  let filteredLowConfidenceRelations = 0;
158
221
  let firstFailureReason;
159
222
  for (const extraction of extractions) {
160
223
  truncationCount += extraction.truncationCount ?? 0;
224
+ truncatedChunks += extraction.truncatedChunks ?? 0;
161
225
  filteredGenericEntities += extraction.filteredGenericEntities ?? 0;
162
226
  filteredInvalidRelations += extraction.filteredInvalidRelations ?? 0;
163
227
  filteredLowConfidenceRelations += extraction.filteredLowConfidenceRelations ?? 0;
@@ -234,6 +298,7 @@ function mergeGraphExtractions(extractions) {
234
298
  reason,
235
299
  chunkCount: extractions.length,
236
300
  truncationCount,
301
+ truncatedChunks,
237
302
  filteredGenericEntities,
238
303
  filteredInvalidRelations,
239
304
  filteredLowConfidenceRelations,
@@ -283,13 +348,36 @@ function parseGraphExtraction(raw) {
283
348
  let filteredLowConfidenceRelations = 0;
284
349
  if (Array.isArray(item.relations)) {
285
350
  for (const relation of item.relations) {
286
- if (typeof relation !== "object" || relation === null || Array.isArray(relation)) {
351
+ // Compact triple form `[from, type, to]` (R12) — `type` may be "".
352
+ // Legacy `{"from","to","type","confidence"}` object form still parses
353
+ // for backward compatibility (older cached prompts, other callers).
354
+ let fromRaw;
355
+ let toRaw;
356
+ let typeRaw;
357
+ let confidenceRaw;
358
+ if (Array.isArray(relation)) {
359
+ if (relation.length !== 3 ||
360
+ typeof relation[0] !== "string" ||
361
+ typeof relation[1] !== "string" ||
362
+ typeof relation[2] !== "string") {
363
+ filteredInvalidRelations += 1;
364
+ continue;
365
+ }
366
+ fromRaw = normalizeEntityName(relation[0]);
367
+ typeRaw = relation[1];
368
+ toRaw = normalizeEntityName(relation[2]);
369
+ }
370
+ else if (typeof relation === "object" && relation !== null) {
371
+ const rel = relation;
372
+ fromRaw = typeof rel.from === "string" ? normalizeEntityName(rel.from) : "";
373
+ toRaw = typeof rel.to === "string" ? normalizeEntityName(rel.to) : "";
374
+ typeRaw = typeof rel.type === "string" ? rel.type : undefined;
375
+ confidenceRaw = rel.confidence;
376
+ }
377
+ else {
287
378
  filteredInvalidRelations += 1;
288
379
  continue;
289
380
  }
290
- const rel = relation;
291
- const fromRaw = typeof rel.from === "string" ? normalizeEntityName(rel.from) : "";
292
- const toRaw = typeof rel.to === "string" ? normalizeEntityName(rel.to) : "";
293
381
  if (!fromRaw || !toRaw) {
294
382
  filteredInvalidRelations += 1;
295
383
  continue;
@@ -300,12 +388,12 @@ function parseGraphExtraction(raw) {
300
388
  filteredInvalidRelations += 1;
301
389
  continue;
302
390
  }
303
- const type = typeof rel.type === "string" ? normalizeRelationType(rel.type) : undefined;
391
+ const type = typeRaw !== undefined ? normalizeRelationType(typeRaw) : undefined;
304
392
  if (type !== undefined && GENERIC_RELATION_TYPES.has(type)) {
305
393
  filteredInvalidRelations += 1;
306
394
  continue;
307
395
  }
308
- const confidence = parseConfidence(rel.confidence);
396
+ const confidence = parseConfidence(confidenceRaw);
309
397
  if (confidence !== undefined && confidence < MIN_RELATION_CONFIDENCE) {
310
398
  filteredLowConfidenceRelations += 1;
311
399
  continue;
@@ -356,8 +444,8 @@ function parseGraphExtraction(raw) {
356
444
  *
357
445
  * Expected model output (valid JSON array, no prose):
358
446
  * [
359
- * {"entities":["ServiceA","ServiceB"],"relations":[{"from":"ServiceA","to":"ServiceB","type":"integrates with"}]},
360
- * {"entities":["Terraform","Prod cluster"],"relations":[{"from":"Terraform","to":"Prod cluster","type":"provisions"}]},
447
+ * {"entities":["ServiceA","ServiceB"],"relations":[["ServiceA","integrates with","ServiceB"]]},
448
+ * {"entities":["Terraform","Prod cluster"],"relations":[["Terraform","provisions","Prod cluster"]]},
361
449
  * {"entities":[],"relations":[]}
362
450
  * ]
363
451
  *
@@ -389,10 +477,10 @@ function buildBatchUserPrompt(bodies) {
389
477
  return (`Extract entities and relations from the N=${count} assets below.\n\n` +
390
478
  `Rules:\n` +
391
479
  `- Output ONLY a JSON array of exactly ${count} objects, one per asset, preserving input order.\n` +
392
- `- Each object: {"entities": ["Entity One", ...], "relations": [{"from": "A", "to": "B", "type": "uses"}, ...]}\n` +
480
+ `- Each object: {"entities": ["Entity One", ...], "relations": [["A", "uses", "B"], ...]}\n` +
393
481
  `- Entities are short, canonical noun phrases (project names, services, tools, people, file/dir names, technical concepts).\n` +
394
- `- Relations connect two entities that both appear in that asset's entities array.\n` +
395
- `- "type" is a short verb phrase (e.g. "uses", "depends on", "owns"). Optional; omit when unsure.\n` +
482
+ `- Each relation is a 3-element array: [from, type, to]. Relations connect two entities that both appear in that asset's entities array.\n` +
483
+ `- "type" is a short verb phrase (e.g. "uses", "depends on", "owns"). Use "" when unsure.\n` +
396
484
  `- Drop pleasantries, meta-commentary, and timestamps.\n` +
397
485
  `- Limit to at most ${MAX_ENTITIES_PER_ASSET} entities and ${MAX_RELATIONS_PER_ASSET} relations per asset.\n` +
398
486
  `- Use {"entities":[],"relations":[]} for assets with no extractable graph content.\n` +
@@ -504,12 +592,26 @@ export async function extractGraphFromBodies(llmRunner, bodies, signal, akmConfi
504
592
  }
505
593
  const systemPrompt = buildBatchSystemPrompt();
506
594
  const userPrompt = buildBatchUserPrompt(nonEmptyBodies);
595
+ // Same responseSchema lift as extractGraphFromBody (R12b), scoped to this
596
+ // batch's asset count so a compliant provider bounds every element's
597
+ // entities/relations by the same maxItems as the single-asset path.
598
+ const batchResponseSchema = buildBatchResponseSchema(nonEmptyBodies.length);
507
599
  const truncatedBodies = nonEmptyBodies.filter((body) => body.length > MAX_BATCH_BODY_CHARS).length;
508
600
  if (truncatedBodies > 0) {
509
601
  warnVerbose(`graph extraction (batch): ${truncatedBodies}/${nonEmptyBodies.length} asset body/bodies exceed the batch body threshold of ${MAX_BATCH_BODY_CHARS} chars.`);
510
602
  }
511
603
  let batchContextError = false;
512
604
  let nonArrayResponse = false;
605
+ // R2: a dead/erroring provider must not be hammered with a per-asset
606
+ // fallback retry for every body in the batch — that is what turned one
607
+ // outage into 15,453 additional retry attempts. `isTransportFailure`
608
+ // (shared with client.ts's `isRetryable` — see its definition) covers
609
+ // `provider_error`, `network_error`, and `provider_html_error`: the
610
+ // provider itself is failing (a dead endpoint more often raises
611
+ // `network_error` or `provider_html_error` than a plain 5xx), not that this
612
+ // particular response was malformed; skip the fallback and record every
613
+ // asset as failed instead.
614
+ let batchProviderError = false;
513
615
  const batchOutcome = await tryLlmFeature("graph_extraction", akmConfig, async () => {
514
616
  try {
515
617
  const raw = await callGraphLlm(llmRunner, [
@@ -519,6 +621,7 @@ export async function extractGraphFromBodies(llmRunner, bodies, signal, akmConfi
519
621
  temperature: 0.1,
520
622
  timeoutMs: llmRunner.timeoutMs,
521
623
  signal,
624
+ responseSchema: batchResponseSchema,
522
625
  onRetryAttempt: () => bumpTelemetry(options.telemetry, "retryAttempts"),
523
626
  }, options.lease, options.onNotices);
524
627
  if (!raw)
@@ -535,7 +638,12 @@ export async function extractGraphFromBodies(llmRunner, bodies, signal, akmConfi
535
638
  const retryRaw = await callGraphLlm(llmRunner, [
536
639
  { role: "system", content: buildBatchRetrySystemPrompt() },
537
640
  { role: "user", content: userPrompt },
538
- ], { temperature: 0, timeoutMs: llmRunner.timeoutMs, signal }, options.lease, options.onNotices);
641
+ ], {
642
+ temperature: 0,
643
+ timeoutMs: llmRunner.timeoutMs,
644
+ signal,
645
+ responseSchema: batchResponseSchema,
646
+ }, options.lease, options.onNotices);
539
647
  parsed = retryRaw ? parseEmbeddedJsonResponse(retryRaw, { expect: "array" }) : undefined;
540
648
  }
541
649
  if (!Array.isArray(parsed)) {
@@ -564,6 +672,12 @@ export async function extractGraphFromBodies(llmRunner, bodies, signal, akmConfi
564
672
  warn(`graph extraction (batch): context size exceeded for ${nonEmptyBodies.length} asset(s); ` +
565
673
  `skipping batch. promptChars=${userPrompt.length}${formatContextHint(llmRunner)}`);
566
674
  }
675
+ else if (err instanceof LlmCallError && isTransportFailure(err)) {
676
+ batchProviderError = true;
677
+ bumpTelemetry(options.telemetry, "failureCount", nonEmptyBodies.length);
678
+ warn(`graph extraction (batch): provider error (${err.code}) for ${nonEmptyBodies.length} asset(s); ` +
679
+ `skipping per-asset fallback retries. promptChars=${userPrompt.length}${formatContextHint(llmRunner)}: ${errMsg}`);
680
+ }
567
681
  else {
568
682
  warn(`graph extraction (batch) failed for ${nonEmptyBodies.length} asset(s); ` +
569
683
  `promptChars=${userPrompt.length}${formatContextHint(llmRunner)}: ${errMsg}`);
@@ -581,6 +695,13 @@ export async function extractGraphFromBodies(llmRunner, bodies, signal, akmConfi
581
695
  if (batchResult !== null) {
582
696
  applySuccessfulBatchResults(results, batchResult, nonEmptyBodies, nonEmptyIndices, batchState);
583
697
  }
698
+ else if (batchProviderError) {
699
+ // No per-asset fallback against a failing provider — record every asset
700
+ // in this batch as a genuine failure so it is neither silently empty nor
701
+ // retried again below.
702
+ for (const origIdx of nonEmptyIndices)
703
+ results[origIdx] = { entities: [], relations: [], status: "failed", reason: "llm_error" };
704
+ }
584
705
  if (batchContextError && nonEmptyBodies.length > 1) {
585
706
  const splitAt = Math.ceil(nonEmptyBodies.length / 2);
586
707
  const left = await extractGraphFromBodies(llmRunner, nonEmptyBodies.slice(0, splitAt), signal, akmConfig, onFallback, options);
@@ -601,6 +722,8 @@ export async function extractGraphFromBodies(llmRunner, bodies, signal, akmConfi
601
722
  const fallbackIndices = nonEmptyIndices.filter((_origIdx, j) => {
602
723
  if (batchContextError)
603
724
  return false; // skip individual retries on context error
725
+ if (batchProviderError)
726
+ return false; // skip individual retries against a failing provider
604
727
  // Result is still empty → needs a fallback call.
605
728
  if (batchResult === null)
606
729
  return true;
@@ -653,17 +776,34 @@ export async function extractGraphFromBody(llmRunner, body, signal, akmConfig, o
653
776
  bumpTelemetry(options.telemetry, "truncationCount", chunked.truncationCount);
654
777
  warnVerbose(`graph extraction: split a long asset into ${chunked.chunks.length} chunk(s) with ${chunked.truncationCount} hard split(s).`);
655
778
  }
656
- if (chunked.chunks.length > 1) {
779
+ // R12b + R20: bound per-asset cost by capping how many chunks of a long
780
+ // asset are ever sent to the LLM. Excess chunks are dropped, never
781
+ // processed — the coverage loss is recorded as truncatedChunks rather than
782
+ // silently absorbed.
783
+ const maxChunksPerAsset = options.maxChunksPerAsset ?? DEFAULT_MAX_CHUNKS_PER_ASSET;
784
+ const cappedChunks = chunked.chunks.length > maxChunksPerAsset ? chunked.chunks.slice(0, maxChunksPerAsset) : chunked.chunks;
785
+ const truncatedChunkCount = chunked.chunks.length - cappedChunks.length;
786
+ if (truncatedChunkCount > 0) {
787
+ bumpTelemetry(options.telemetry, "truncatedChunks", truncatedChunkCount);
788
+ warnVerbose(`graph extraction: capped a long asset to ${cappedChunks.length} of ${chunked.chunks.length} chunk(s) ` +
789
+ `(maxChunksPerAsset=${maxChunksPerAsset}); ${truncatedChunkCount} chunk(s) not processed.`);
790
+ }
791
+ if (cappedChunks.length > 1) {
657
792
  const chunkResults = [];
658
- for (const chunk of chunked.chunks) {
793
+ for (const chunk of cappedChunks) {
659
794
  chunkResults.push(await extractGraphFromBody(llmRunner, chunk, signal, akmConfig, onFallback, options));
660
795
  }
661
796
  const merged = mergeGraphExtractions(chunkResults);
662
797
  merged.truncationCount = (merged.truncationCount ?? 0) + chunked.truncationCount;
798
+ merged.truncatedChunks = (merged.truncatedChunks ?? 0) + truncatedChunkCount;
663
799
  return merged;
664
800
  }
665
- const userPrompt = `${USER_PROMPT_PREFIX}${trimmedBody}`;
666
- return callStructured({
801
+ // When capped down to exactly one chunk from a body that originally split
802
+ // into more, that single surviving chunk (not the full trimmedBody) is what
803
+ // must be sent — otherwise the cap would have no effect on prompt size.
804
+ const bodyForCall = truncatedChunkCount > 0 ? (cappedChunks[0] ?? trimmedBody) : trimmedBody;
805
+ const userPrompt = `${USER_PROMPT_PREFIX}${bodyForCall}`;
806
+ const result = await callStructured({
667
807
  feature: "graph_extraction",
668
808
  akmConfig,
669
809
  runner: llmRunner,
@@ -676,6 +816,7 @@ export async function extractGraphFromBody(llmRunner, body, signal, akmConfig, o
676
816
  temperature: 0.1,
677
817
  timeoutMs: llmRunner.timeoutMs,
678
818
  signal,
819
+ responseSchema: GRAPH_EXTRACTION_JSON_SCHEMA,
679
820
  onRetryAttempt: () => bumpTelemetry(options.telemetry, "retryAttempts"),
680
821
  },
681
822
  onNotices: options.onNotices,
@@ -718,6 +859,9 @@ export async function extractGraphFromBody(llmRunner, body, signal, akmConfig, o
718
859
  fallback: empty(),
719
860
  onFallback,
720
861
  });
862
+ if (truncatedChunkCount > 0)
863
+ result.truncatedChunks = (result.truncatedChunks ?? 0) + truncatedChunkCount;
864
+ return result;
721
865
  }
722
866
  // deduplicateGraph lives in src/indexer/graph/graph-dedup.ts (pure utility, no
723
867
  // LLM calls) — import it from there directly. The re-export that used to live