akm-cli 0.9.17-alpha.4 → 0.9.17-alpha.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -33,6 +33,12 @@ import { callStructured } from "./structured-call.js";
33
33
  * Chosen to be visually clear and unlikely to appear verbatim in asset bodies.
34
34
  */
35
35
  const BATCH_ASSET_SEPARATOR = "=== ASSET";
36
+ /**
37
+ * Part of the extractor id that keys cached extractions; the prompt text is
38
+ * not. Bump it with any change to either prompt, and expect every cached file
39
+ * to be extracted again. Pending for the next bump (GR-D12): the batch prompt
40
+ * still asks for "file/dir names", which the single-asset prompt rules out.
41
+ */
36
42
  export const GRAPH_EXTRACT_PROMPT_VERSION = "v3";
37
43
  /** Asset bodies longer than this are chunked instead of truncated. */
38
44
  const MAX_CHUNK_BODY_CHARS = 1600;
@@ -165,7 +171,13 @@ function normalizeRelationType(raw) {
165
171
  return "integrates with";
166
172
  return normalized;
167
173
  }
168
- function normalizeEntityKey(raw) {
174
+ /**
175
+ * The key under which two entity names are the same entity: the display
176
+ * clean-up above, case-folded. The one normalization for graph entities:
177
+ * extraction deduplicates on it, the pass deduplicates on it before writing,
178
+ * and it is the stored `entity_norm` that `related` joins on (GR-D17).
179
+ */
180
+ export function normalizeEntityKey(raw) {
169
181
  return normalizeEntityName(raw).toLowerCase();
170
182
  }
171
183
  function bumpTelemetry(telemetry, key, amount = 1) {
@@ -173,6 +185,36 @@ function bumpTelemetry(telemetry, key, amount = 1) {
173
185
  return;
174
186
  telemetry[key] = (telemetry[key] ?? 0) + amount;
175
187
  }
188
+ /**
189
+ * `Promise.all(items.map(fn))` with at most `limit` calls in flight, so the
190
+ * per-asset calls of one batch never outnumber the chunk pool's concurrency
191
+ * (GR-D14). The first rejection rejects the map and stops further calls.
192
+ */
193
+ async function mapWithConcurrency(items, limit, fn) {
194
+ const results = new Array(items.length);
195
+ let next = 0;
196
+ let failed = false;
197
+ const worker = async () => {
198
+ while (!failed && next < items.length) {
199
+ const index = next++;
200
+ try {
201
+ results[index] = await fn(items[index]);
202
+ }
203
+ catch (error) {
204
+ failed = true;
205
+ throw error;
206
+ }
207
+ }
208
+ };
209
+ await Promise.all(Array.from({ length: Math.min(Math.max(1, limit), items.length) }, worker));
210
+ return results;
211
+ }
212
+ /** Count what the parser dropped from one parsed extraction (single-asset or batch item). */
213
+ function bumpFilterTelemetry(telemetry, extraction) {
214
+ bumpTelemetry(telemetry, "filteredGenericEntities", extraction.filteredGenericEntities ?? 0);
215
+ bumpTelemetry(telemetry, "filteredInvalidRelations", extraction.filteredInvalidRelations ?? 0);
216
+ bumpTelemetry(telemetry, "filteredLowConfidenceRelations", extraction.filteredLowConfidenceRelations ?? 0);
217
+ }
176
218
  function normalizeBatchState(state) {
177
219
  if (!state)
178
220
  return undefined;
@@ -512,7 +554,7 @@ async function callGraphLlm(runner, messages, request, onNotices) {
512
554
  function parseBatchItem(raw) {
513
555
  return parseGraphExtraction(raw);
514
556
  }
515
- function applySuccessfulBatchResults(results, batchResult, nonEmptyBodies, nonEmptyIndices, batchState) {
557
+ function applySuccessfulBatchResults(results, batchResult, nonEmptyBodies, nonEmptyIndices, batchState, telemetry) {
516
558
  if (batchState)
517
559
  batchState.nonArrayBatchFailures = 0;
518
560
  if (batchResult.length > nonEmptyBodies.length) {
@@ -523,8 +565,11 @@ function applySuccessfulBatchResults(results, batchResult, nonEmptyBodies, nonEm
523
565
  const originalIndex = nonEmptyIndices[j];
524
566
  if (originalIndex === undefined)
525
567
  continue;
526
- if (j < batchResult.length)
527
- results[originalIndex] = parseBatchItem(batchResult[j]);
568
+ if (j >= batchResult.length)
569
+ continue;
570
+ const extraction = parseBatchItem(batchResult[j]);
571
+ bumpFilterTelemetry(telemetry, extraction);
572
+ results[originalIndex] = extraction;
528
573
  }
529
574
  }
530
575
  /**
@@ -538,8 +583,8 @@ function applySuccessfulBatchResults(results, batchResult, nonEmptyBodies, nonEm
538
583
  * `bodies.length`, missing indices are filled by falling back to individual
539
584
  * `extractGraphFromBody` calls — ensuring every input always has a result.
540
585
  *
541
- * Returns an array of the same length as `bodies` (never shorter).
542
- * Individual elements default to `{entities:[], relations:[]}` on failure.
586
+ * Returns an array of the same length as `bodies` (never shorter). A body
587
+ * whose extraction failed gets no entities and `status: "failed"`.
543
588
  *
544
589
  * Routes through `tryLlmFeature("graph_extraction", ...)` so the feature gate
545
590
  * and onFallback hook are honoured uniformly.
@@ -561,6 +606,11 @@ export async function extractGraphFromBodies(llmRunner, bodies, signal, akmConfi
561
606
  const result = await extractGraphFromBody(llmRunner, bodies[0] ?? "", signal, akmConfig, onFallback, options);
562
607
  return [result];
563
608
  }
609
+ const concurrency = llmRunner.connection.concurrency ?? 1;
610
+ const extractOne = (body) => extractGraphFromBody(llmRunner, body, signal, akmConfig, onFallback, options);
611
+ // With batching disabled every body takes the single-asset path, once.
612
+ if (batchState?.batchingDisabled)
613
+ return mapWithConcurrency(bodies, concurrency, extractOne);
564
614
  // Filter out bodies that are empty so we don't waste tokens, but keep
565
615
  // index correspondence by tracking which indices were non-empty.
566
616
  const results = bodies.map(empty);
@@ -579,28 +629,18 @@ export async function extractGraphFromBodies(llmRunner, bodies, signal, akmConfi
579
629
  }
580
630
  }
581
631
  }
582
- if (oversizedIndices.length > 0) {
583
- await Promise.all(oversizedIndices.map(async (index) => {
584
- results[index] = await extractGraphFromBody(llmRunner, bodies[index] ?? "", signal, akmConfig, onFallback, options);
585
- }));
586
- }
632
+ await mapWithConcurrency(oversizedIndices, concurrency, async (index) => {
633
+ results[index] = await extractOne(bodies[index] ?? "");
634
+ });
587
635
  if (nonEmptyBodies.length === 0)
588
636
  return results;
589
- if (batchState?.batchingDisabled) {
590
- return Promise.all(bodies.map((body) => extractGraphFromBody(llmRunner, body, signal, akmConfig, onFallback, options)));
591
- }
592
637
  const systemPrompt = buildBatchSystemPrompt();
593
638
  const userPrompt = buildBatchUserPrompt(nonEmptyBodies);
594
639
  // Same responseSchema lift as extractGraphFromBody (R12b), scoped to this
595
640
  // batch's asset count so a compliant provider bounds every element's
596
641
  // entities/relations by the same maxItems as the single-asset path.
597
642
  const batchResponseSchema = buildBatchResponseSchema(nonEmptyBodies.length);
598
- const truncatedBodies = nonEmptyBodies.filter((body) => body.length > MAX_BATCH_BODY_CHARS).length;
599
- if (truncatedBodies > 0) {
600
- warnVerbose(`graph extraction (batch): ${truncatedBodies}/${nonEmptyBodies.length} asset body/bodies exceed the batch body threshold of ${MAX_BATCH_BODY_CHARS} chars.`);
601
- }
602
643
  let batchContextError = false;
603
- let nonArrayResponse = false;
604
644
  // R2: a dead/erroring provider must not be hammered with a per-asset
605
645
  // fallback retry for every body in the batch — that is what turned one
606
646
  // outage into 15,453 additional retry attempts. `isTransportFailure`
@@ -609,7 +649,8 @@ export async function extractGraphFromBodies(llmRunner, bodies, signal, akmConfi
609
649
  // provider itself is failing (a dead endpoint more often raises
610
650
  // `network_error` or `provider_html_error` than a plain 5xx), not that this
611
651
  // particular response was malformed; skip the fallback and record every
612
- // asset as failed instead.
652
+ // asset as failed instead. A batch that got no answer at all (the gate's
653
+ // timeout, or a closed gate) is handled the same way.
613
654
  let batchProviderError = false;
614
655
  const batchOutcome = await tryLlmFeature("graph_extraction", akmConfig, async () => {
615
656
  try {
@@ -646,7 +687,6 @@ export async function extractGraphFromBodies(llmRunner, bodies, signal, akmConfi
646
687
  parsed = retryRaw ? parseEmbeddedJsonResponse(retryRaw, { expect: "array" }) : undefined;
647
688
  }
648
689
  if (!Array.isArray(parsed)) {
649
- nonArrayResponse = true;
650
690
  bumpTelemetry(options.telemetry, "nonArrayBatchFailures");
651
691
  if (batchState) {
652
692
  batchState.nonArrayBatchFailures += 1;
@@ -683,16 +723,20 @@ export async function extractGraphFromBodies(llmRunner, bodies, signal, akmConfi
683
723
  }
684
724
  return { kind: "value", value: null };
685
725
  }
686
- }, { kind: "value", value: null }, {
726
+ }, { kind: "no-answer" }, {
687
727
  timeoutMs: llmRunner.timeoutMs,
688
728
  onFallback,
689
729
  });
690
730
  if (batchOutcome.kind === "config-error")
691
731
  throw batchOutcome.error;
692
- const batchResult = batchOutcome.value;
732
+ if (batchOutcome.kind === "no-answer") {
733
+ batchProviderError = true;
734
+ bumpTelemetry(options.telemetry, "failureCount", nonEmptyBodies.length);
735
+ }
736
+ const batchResult = batchOutcome.kind === "value" ? batchOutcome.value : null;
693
737
  // Map successful batch results back to their original indices.
694
738
  if (batchResult !== null) {
695
- applySuccessfulBatchResults(results, batchResult, nonEmptyBodies, nonEmptyIndices, batchState);
739
+ applySuccessfulBatchResults(results, batchResult, nonEmptyBodies, nonEmptyIndices, batchState, options.telemetry);
696
740
  }
697
741
  else if (batchProviderError) {
698
742
  // No per-asset fallback against a failing provider — record every asset
@@ -736,26 +780,23 @@ export async function extractGraphFromBodies(llmRunner, bodies, signal, akmConfi
736
780
  warn(`graph extraction (batch): response had ${batchResult.length} items for ${nonEmptyBodies.length} assets; ` +
737
781
  `falling back to individual calls for ${fallbackIndices.length} missing asset(s).`);
738
782
  }
739
- await Promise.all(fallbackIndices.map(async (origIdx) => {
740
- const body = bodies[origIdx] ?? "";
741
- results[origIdx] = await extractGraphFromBody(llmRunner, body, signal, akmConfig, onFallback, options);
742
- }));
783
+ await mapWithConcurrency(fallbackIndices, concurrency, async (origIdx) => {
784
+ results[origIdx] = await extractOne(bodies[origIdx] ?? "");
785
+ });
743
786
  }
744
787
  else if (batchContextError) {
745
788
  warn(`graph extraction (batch): skipped ${nonEmptyBodies.length} asset(s) due to context size error; ` +
746
789
  `consider increasing llm.contextLength or reducing index.graph.graphExtractionBatchSize to 1.`);
747
790
  }
748
- else if (nonArrayResponse && batchState?.batchingDisabled) {
749
- warn("graph extraction (batch): disabling batching for the rest of this run after repeated non-array responses.");
750
- }
751
791
  return results;
752
792
  }
753
793
  /**
754
794
  * Extract entities and relations from a single asset body via the configured LLM.
755
795
  *
756
- * Returns `{entities: [], relations: []}` on any failure (timeout, invalid
757
- * JSON, empty response). Errors are logged via `warn()` but never thrown — a
758
- * failed extraction for one asset must not abort the rest of the index pass.
796
+ * Any failure (timeout, invalid JSON, empty response, provider error) returns
797
+ * no entities with `status: "failed"`, which is never cached, so the next run
798
+ * retries it. Errors are logged via `warn()` but never thrown — a failed
799
+ * extraction for one asset must not abort the rest of the index pass.
759
800
  *
760
801
  * Routes through `tryLlmFeature("graph_extraction", ...)` so the feature gate
761
802
  * and onFallback hook are honoured uniformly (Fix C5).
@@ -819,18 +860,15 @@ export async function extractGraphFromBody(llmRunner, body, signal, akmConfig, o
819
860
  },
820
861
  onNotices: options.onNotices,
821
862
  parse: (raw) => {
822
- if (!raw)
823
- return empty();
824
- const parsed = parseEmbeddedJsonResponse(raw);
863
+ // An empty response is not JSON either; "nothing to extract" is `{"entities": []}`.
864
+ const parsed = raw ? parseEmbeddedJsonResponse(raw) : undefined;
825
865
  if (!parsed) {
826
866
  warn("graph extraction: invalid JSON response from LLM; skipping asset.");
827
867
  bumpTelemetry(options.telemetry, "failureCount");
828
868
  return empty("invalid_json", "failed");
829
869
  }
830
870
  const extraction = parseGraphExtraction(parsed);
831
- bumpTelemetry(options.telemetry, "filteredGenericEntities", extraction.filteredGenericEntities ?? 0);
832
- bumpTelemetry(options.telemetry, "filteredInvalidRelations", extraction.filteredInvalidRelations ?? 0);
833
- bumpTelemetry(options.telemetry, "filteredLowConfidenceRelations", extraction.filteredLowConfidenceRelations ?? 0);
871
+ bumpFilterTelemetry(options.telemetry, extraction);
834
872
  if (extraction.status === "failed")
835
873
  bumpTelemetry(options.telemetry, "failureCount");
836
874
  return extraction;
@@ -854,7 +892,9 @@ export async function extractGraphFromBody(llmRunner, body, signal, akmConfig, o
854
892
  return empty("llm_error", "failed");
855
893
  }
856
894
  },
857
- fallback: empty(),
895
+ // No answer from the model (the gate's timeout, or a closed gate) is a
896
+ // failure the next run retries, never a cacheable "no entities".
897
+ fallback: empty("llm_error", "failed"),
858
898
  onFallback,
859
899
  });
860
900
  if (truncatedChunkCount > 0)